This commit is contained in:
@@ -0,0 +1,22 @@
|
||||
Copyright (C) 1993-2013 Yukihiro Matsumoto. All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions
|
||||
are met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in the
|
||||
documentation and/or other materials provided with the distribution.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
|
||||
ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
|
||||
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
|
||||
OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
|
||||
HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
|
||||
LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
|
||||
OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
|
||||
SUCH DAMAGE.
|
||||
@@ -0,0 +1,843 @@
|
||||
# News
|
||||
|
||||
## 3.4.4 - 2025-09-10 {#version-3-4-4}
|
||||
|
||||
### Improvement
|
||||
|
||||
* Accept `REXML::Document.new("")` for backward compatibility
|
||||
* GH-296
|
||||
* GH-295
|
||||
* Patch by NAITOH Jun
|
||||
* Reported by Joe Rafaniello
|
||||
|
||||
### Thanks
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* Joe Rafaniello
|
||||
|
||||
## 3.4.3 - 2025-09-07 {#version-3-4-3}
|
||||
|
||||
### Improvement
|
||||
|
||||
* Reject no root element XML as an invalid XML
|
||||
* GH-289
|
||||
* GH-291
|
||||
* Patch by NAITOH Jun
|
||||
* Reported by Sutou Kouhei
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed an issue with `IOSource#read_until` when reaching the end of a file
|
||||
* GH-287
|
||||
* GH-288
|
||||
* Patch by NAITOH Jun
|
||||
* Reported by Jason Thomas
|
||||
|
||||
### Thanks
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* Sutou Kouhei
|
||||
|
||||
* Jason Thomas
|
||||
|
||||
## 3.4.2 - 2025-08-26 {#version-3-4-2}
|
||||
|
||||
### Improvement
|
||||
|
||||
* Improved performance.
|
||||
* GH-244
|
||||
* GH-245
|
||||
* GH-246
|
||||
* GH-249
|
||||
* GH-256
|
||||
* Patch by NAITOH Jun
|
||||
|
||||
* Raise appropriate exception when failing to match start tag in DOCTYPE
|
||||
* GH-247
|
||||
* Patch by NAITOH Jun
|
||||
|
||||
* Deprecate accepting array as an element in XPath.match, first and each
|
||||
* GH-252
|
||||
* Patch by tomoya ishida
|
||||
|
||||
* Don't call needless encoding_updated
|
||||
* GH-259
|
||||
* Patch by Sutou Kouhei
|
||||
|
||||
* Reuse XPath::match
|
||||
* GH-263
|
||||
* Patch by pboling
|
||||
|
||||
* Cache redundant calls for doctype
|
||||
* GH-264
|
||||
* Patch by pboling
|
||||
|
||||
* Use Safe Navigation (&.) from Ruby 2.3
|
||||
* GH-265
|
||||
* Patch by pboling
|
||||
|
||||
* Remove redundant return statements
|
||||
* GH-266
|
||||
* Patch by pboling
|
||||
|
||||
* Added XML declaration check & Source#skip_spaces method
|
||||
* GH-282
|
||||
* Patch by NAITOH Jun
|
||||
* Reported by Sofi Aberegg
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fix docs typo
|
||||
* GH-248
|
||||
* Patch by James Coleman
|
||||
|
||||
* Fix reverse sort in xpath_parser
|
||||
* GH-251
|
||||
* Patch by tomoya ishida
|
||||
|
||||
* Fix duplicate responses in XPath following, following-sibling, preceding, preceding-sibling
|
||||
* GH-255
|
||||
* Patch by NAITOH Jun
|
||||
|
||||
* Fix wrong Encoding resolution
|
||||
* GH-258
|
||||
* Patch by Sutou Kouhei
|
||||
|
||||
* Handle nil when parsing fragment
|
||||
* GH-267
|
||||
* GH-268
|
||||
* Patch by pboling
|
||||
|
||||
* [Documentation] Use # to reference instance methods
|
||||
* GH-269
|
||||
* GH-270
|
||||
* Patch by pboling
|
||||
|
||||
* Fix & Deprecate REXML::Text#text_indent
|
||||
* GH-273
|
||||
* GH-275
|
||||
* Patch by pboling
|
||||
|
||||
* remove bundler from dev deps
|
||||
* GH-276
|
||||
* GH-277
|
||||
* Patch by pboling
|
||||
|
||||
* remove ostruct from dev deps
|
||||
* GH-280
|
||||
* GH-281
|
||||
* Patch by pboling
|
||||
|
||||
### Thanks
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* tomoya ishida
|
||||
|
||||
* James Coleman
|
||||
|
||||
* pboling
|
||||
|
||||
* Sutou Kouhei
|
||||
|
||||
* Sofi Aberegg
|
||||
|
||||
## 3.4.1 - 2025-02-16 {#version-3-4-1}
|
||||
|
||||
### Improvement
|
||||
|
||||
* Improved performance.
|
||||
* GH-226
|
||||
* GH-227
|
||||
* GH-237
|
||||
* Patch by NAITOH Jun
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fix serialization of ATTLIST is incorrect
|
||||
* GH-233
|
||||
* GH-234
|
||||
* Patch by OlofKalufs
|
||||
* Reported by OlofKalufs
|
||||
|
||||
### Thanks
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* OlofKalufs
|
||||
|
||||
## 3.4.0 - 2024-12-15 {#version-3-4-0}
|
||||
|
||||
### Improvement
|
||||
|
||||
* Improved performance.
|
||||
* GH-216
|
||||
* Patch by NAITOH Jun
|
||||
|
||||
* JRuby: Improved parse performance.
|
||||
* GH-219
|
||||
* Patch by João Duarte
|
||||
|
||||
* Added support for reusing pull parser.
|
||||
* GH-214
|
||||
* GH-220
|
||||
* Patch by Dmitry Pogrebnoy
|
||||
|
||||
* Improved error handling when source is `IO`.
|
||||
* GH-221
|
||||
* Patch by NAITOH Jun
|
||||
|
||||
### Thanks
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* João Duarte
|
||||
|
||||
* Dmitry Pogrebnoy
|
||||
|
||||
## 3.3.9 - 2024-10-24 {#version-3-3-9}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Improved performance.
|
||||
* GH-210
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a parse bug for text only invalid XML.
|
||||
* GH-215
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Fixed a parse bug that `�x...;` is accepted as a character
|
||||
reference.
|
||||
|
||||
### Thanks
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
## 3.3.8 - 2024-09-29 {#version-3-3-8}
|
||||
|
||||
### Improvements
|
||||
|
||||
* SAX2: Improve parse performance.
|
||||
* GH-207
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that unexpected attribute namespace conflict error for
|
||||
the predefined "xml" namespace is reported.
|
||||
* GH-208
|
||||
* Patch by KITAITI Makoto
|
||||
|
||||
### Thanks
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* KITAITI Makoto
|
||||
|
||||
## 3.3.7 - 2024-09-04 {#version-3-3-7}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Added local entity expansion limit methods
|
||||
* GH-192
|
||||
* GH-202
|
||||
* Reported by takuya kodama.
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Removed explicit strscan dependency
|
||||
* GH-204
|
||||
* Patch by Bo Anderson.
|
||||
|
||||
### Thanks
|
||||
|
||||
* takuya kodama
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* Bo Anderson
|
||||
|
||||
## 3.3.6 - 2024-08-22 {#version-3-3-6}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Removed duplicated entity expansions for performance.
|
||||
* GH-194
|
||||
* Patch by Viktor Ivarsson.
|
||||
|
||||
* Improved namespace conflicted attribute check performance. It was
|
||||
too slow for deep elements.
|
||||
* Reported by l33thaxor.
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that default entity expansions are counted for
|
||||
security check. Default entity expansions should not be counted
|
||||
because they don't have a security risk.
|
||||
* GH-198
|
||||
* GH-199
|
||||
* Patch Viktor Ivarsson
|
||||
|
||||
* Fixed a parser bug that parameter entity references in internal
|
||||
subsets are expanded. It's not allowed in the XML specification.
|
||||
* GH-191
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Fixed a stream parser bug that user-defined entity references in
|
||||
text aren't expanded.
|
||||
* GH-200
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
### Thanks
|
||||
|
||||
* Viktor Ivarsson
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* l33thaxor
|
||||
|
||||
## 3.3.5 - 2024-08-12 {#version-3-3-5}
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that `REXML::Security.entity_expansion_text_limit`
|
||||
check has wrong text size calculation in SAX and pull parsers.
|
||||
* GH-193
|
||||
* GH-195
|
||||
* Reported by Viktor Ivarsson.
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
### Thanks
|
||||
|
||||
* Viktor Ivarsson
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
## 3.3.4 - 2024-08-01 {#version-3-3-4}
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that `REXML::Security` isn't defined when
|
||||
`REXML::Parsers::StreamParser` is used and
|
||||
`rexml/parsers/streamparser` is only required.
|
||||
* GH-189
|
||||
* Patch by takuya kodama.
|
||||
|
||||
### Thanks
|
||||
|
||||
* takuya kodama
|
||||
|
||||
## 3.3.3 - 2024-08-01 {#version-3-3-3}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Added support for detecting invalid XML that has unsupported
|
||||
content before root element
|
||||
* GH-184
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Added support for `REXML::Security.entity_expansion_limit=` and
|
||||
`REXML::Security.entity_expansion_text_limit=` in SAX2 and pull
|
||||
parsers
|
||||
* GH-187
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Added more tests for invalid XMLs.
|
||||
* GH-183
|
||||
* Patch by Watson.
|
||||
|
||||
* Added more performance tests.
|
||||
* Patch by Watson.
|
||||
|
||||
* Improved parse performance.
|
||||
* GH-186
|
||||
* Patch by tomoya ishida.
|
||||
|
||||
### Thanks
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* Watson
|
||||
|
||||
* tomoya ishida
|
||||
|
||||
## 3.3.2 - 2024-07-16 {#version-3-3-2}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Improved parse performance.
|
||||
* GH-160
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Improved parse performance.
|
||||
* GH-169
|
||||
* GH-170
|
||||
* GH-171
|
||||
* GH-172
|
||||
* GH-173
|
||||
* GH-174
|
||||
* GH-175
|
||||
* GH-176
|
||||
* GH-177
|
||||
* Patch by Watson.
|
||||
|
||||
* Added support for raising a parse exception when an XML has extra
|
||||
content after the root element.
|
||||
* GH-161
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Added support for raising a parse exception when an XML
|
||||
declaration exists in wrong position.
|
||||
* GH-162
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Removed needless a space after XML declaration in pretty print mode.
|
||||
* GH-164
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Stopped to emit `:text` event after the root element.
|
||||
* GH-167
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that SAX2 parser doesn't expand predefined entities for
|
||||
`characters` callback.
|
||||
* GH-168
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
### Thanks
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* Watson
|
||||
|
||||
## 3.3.1 - 2024-06-25 {#version-3-3-1}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Added support for detecting malformed top-level comments.
|
||||
* GH-145
|
||||
* Patch by Hiroya Fujinami.
|
||||
|
||||
* Improved `REXML::Element#attribute` performance.
|
||||
* GH-146
|
||||
* Patch by Hiroya Fujinami.
|
||||
|
||||
* Added support for detecting malformed `<!-->` comments.
|
||||
* GH-147
|
||||
* Patch by Hiroya Fujinami.
|
||||
|
||||
* Added support for detecting unclosed `DOCTYPE`.
|
||||
* GH-152
|
||||
* Patch by Hiroya Fujinami.
|
||||
|
||||
* Added `changlog_uri` metadata to gemspec.
|
||||
* GH-156
|
||||
* Patch by fynsta.
|
||||
|
||||
* Improved parse performance.
|
||||
* GH-157
|
||||
* GH-158
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that large XML can't be parsed.
|
||||
* GH-154
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Fixed a bug that private constants are visible.
|
||||
* GH-155
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
### Thanks
|
||||
|
||||
* Hiroya Fujinami
|
||||
|
||||
* NAITOH Jun
|
||||
|
||||
* fynsta
|
||||
|
||||
## 3.3.0 - 2024-06-11 {#version-3-3-0}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Added support for strscan 0.7.0 installed with Ruby 2.6.
|
||||
* GH-142
|
||||
* Reported by Fernando Trigoso.
|
||||
|
||||
### Thanks
|
||||
|
||||
* Fernando Trigoso
|
||||
|
||||
## 3.2.9 - 2024-06-09 {#version-3-2-9}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Added support for old strscan.
|
||||
* GH-132
|
||||
* Reported by Adam.
|
||||
|
||||
* Improved attribute value parse performance.
|
||||
* GH-135
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Improved `REXML::Node#each_recursive` performance.
|
||||
* GH-134
|
||||
* GH-139
|
||||
* Patch by Hiroya Fujinami.
|
||||
|
||||
* Improved text parse performance.
|
||||
* Reported by mprogrammer.
|
||||
|
||||
### Thanks
|
||||
|
||||
* Adam
|
||||
* NAITOH Jun
|
||||
* Hiroya Fujinami
|
||||
* mprogrammer
|
||||
|
||||
## 3.2.8 - 2024-05-16 {#version-3-2-8}
|
||||
|
||||
### Fixes
|
||||
|
||||
* Suppressed a warning
|
||||
|
||||
## 3.2.7 - 2024-05-16 {#version-3-2-7}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Improve parse performance by using `StringScanner`.
|
||||
|
||||
* GH-106
|
||||
* GH-107
|
||||
* GH-108
|
||||
* GH-109
|
||||
* GH-112
|
||||
* GH-113
|
||||
* GH-114
|
||||
* GH-115
|
||||
* GH-116
|
||||
* GH-117
|
||||
* GH-118
|
||||
* GH-119
|
||||
* GH-121
|
||||
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Improved parse performance when an attribute has many `>`s.
|
||||
|
||||
* GH-126
|
||||
|
||||
### Fixes
|
||||
|
||||
* XPath: Fixed a bug of `normalize_space(array)`.
|
||||
|
||||
* GH-110
|
||||
* GH-111
|
||||
|
||||
* Patch by flatisland.
|
||||
|
||||
* XPath: Fixed a bug that wrong position is used with nested path.
|
||||
|
||||
* GH-110
|
||||
* GH-122
|
||||
|
||||
* Reported by jcavalieri.
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
* Fixed a bug that an exception message can't be generated for
|
||||
invalid encoding XML.
|
||||
|
||||
* GH-29
|
||||
* GH-123
|
||||
|
||||
* Reported by DuKewu.
|
||||
* Patch by NAITOH Jun.
|
||||
|
||||
### Thanks
|
||||
|
||||
* NAITOH Jun
|
||||
* flatisland
|
||||
* jcavalieri
|
||||
* DuKewu
|
||||
|
||||
## 3.2.6 - 2023-07-27 {#version-3-2-6}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Required Ruby 2.5 or later explicitly.
|
||||
[GH-69][gh-69]
|
||||
[Patch by Ivo Anjo]
|
||||
|
||||
* Added documentation for maintenance cycle.
|
||||
[GH-71][gh-71]
|
||||
[Patch by Ivo Anjo]
|
||||
|
||||
* Added tutorial.
|
||||
[GH-77][gh-77]
|
||||
[GH-78][gh-78]
|
||||
[Patch by Burdette Lamar]
|
||||
|
||||
* Improved performance and memory usage.
|
||||
[GH-94][gh-94]
|
||||
[Patch by fatkodima]
|
||||
|
||||
* `REXML::Parsers::XPathParser#abbreviate`: Added support for
|
||||
function arguments.
|
||||
[GH-95][gh-95]
|
||||
[Reported by pulver]
|
||||
|
||||
* `REXML::Parsers::XPathParser#abbreviate`: Added support for string
|
||||
literal that contains double-quote.
|
||||
[GH-96][gh-96]
|
||||
[Patch by pulver]
|
||||
|
||||
* `REXML::Parsers::XPathParser#abbreviate`: Added missing `/` to
|
||||
`:descendant_or_self/:self/:parent`.
|
||||
[GH-97][gh-97]
|
||||
[Reported by pulver]
|
||||
|
||||
* `REXML::Parsers::XPathParser#abbreviate`: Added support for more patterns.
|
||||
[GH-97][gh-97]
|
||||
[Reported by pulver]
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a typo in NEWS.
|
||||
[GH-72][gh-72]
|
||||
[Patch by Spencer Goodman]
|
||||
|
||||
* Fixed a typo in NEWS.
|
||||
[GH-75][gh-75]
|
||||
[Patch by Andrew Bromwich]
|
||||
|
||||
* Fixed documents.
|
||||
[GH-87][gh-87]
|
||||
[Patch by Alexander Ilyin]
|
||||
|
||||
* Fixed a bug that `Attriute` convert `'` and `'` even when
|
||||
`attribute_quote: :quote` is used.
|
||||
[GH-92][gh-92]
|
||||
[Reported by Edouard Brière]
|
||||
|
||||
* Fixed links in tutorial.
|
||||
[GH-99][gh-99]
|
||||
[Patch by gemmaro]
|
||||
|
||||
|
||||
### Thanks
|
||||
|
||||
* Ivo Anjo
|
||||
|
||||
* Spencer Goodman
|
||||
|
||||
* Andrew Bromwich
|
||||
|
||||
* Burdette Lamar
|
||||
|
||||
* Alexander Ilyin
|
||||
|
||||
* Edouard Brière
|
||||
|
||||
* fatkodima
|
||||
|
||||
* pulver
|
||||
|
||||
* gemmaro
|
||||
|
||||
[gh-69]:https://github.com/ruby/rexml/issues/69
|
||||
[gh-71]:https://github.com/ruby/rexml/issues/71
|
||||
[gh-72]:https://github.com/ruby/rexml/issues/72
|
||||
[gh-75]:https://github.com/ruby/rexml/issues/75
|
||||
[gh-77]:https://github.com/ruby/rexml/issues/77
|
||||
[gh-87]:https://github.com/ruby/rexml/issues/87
|
||||
[gh-92]:https://github.com/ruby/rexml/issues/92
|
||||
[gh-94]:https://github.com/ruby/rexml/issues/94
|
||||
[gh-95]:https://github.com/ruby/rexml/issues/95
|
||||
[gh-96]:https://github.com/ruby/rexml/issues/96
|
||||
[gh-97]:https://github.com/ruby/rexml/issues/97
|
||||
[gh-98]:https://github.com/ruby/rexml/issues/98
|
||||
[gh-99]:https://github.com/ruby/rexml/issues/99
|
||||
|
||||
## 3.2.5 - 2021-04-05 {#version-3-2-5}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Add more validations to XPath parser.
|
||||
|
||||
* `require "rexml/document"` by default.
|
||||
[GitHub#36][Patch by Koichi ITO]
|
||||
|
||||
* Don't add `#dclone` method to core classes globally.
|
||||
[GitHub#37][Patch by Akira Matsuda]
|
||||
|
||||
* Add more documentations.
|
||||
[Patch by Burdette Lamar]
|
||||
|
||||
* Added `REXML::Elements#parent`.
|
||||
[GitHub#52][Patch by Burdette Lamar]
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that `REXML::DocType#clone` doesn't copy external ID
|
||||
information.
|
||||
|
||||
* Fixed round-trip vulnerability bugs.
|
||||
See also: https://www.ruby-lang.org/en/news/2021/04/05/xml-round-trip-vulnerability-in-rexml-cve-2021-28965/
|
||||
[HackerOne#1104077][CVE-2021-28965][Reported by Juho Nurminen]
|
||||
|
||||
### Thanks
|
||||
|
||||
* Koichi ITO
|
||||
|
||||
* Akira Matsuda
|
||||
|
||||
* Burdette Lamar
|
||||
|
||||
* Juho Nurminen
|
||||
|
||||
## 3.2.4 - 2020-01-31 {#version-3-2-4}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Don't use `taint` with Ruby 2.7 or later.
|
||||
[GitHub#21][Patch by Jeremy Evans]
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a `elsif` typo.
|
||||
[GitHub#22][Patch by Nobuyoshi Nakada]
|
||||
|
||||
### Thanks
|
||||
|
||||
* Jeremy Evans
|
||||
|
||||
* Nobuyoshi Nakada
|
||||
|
||||
## 3.2.3 - 2019-10-12 {#version-3-2-3}
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that `REXML::XMLDecl#close` doesn't copy `@writethis`.
|
||||
[GitHub#20][Patch by hirura]
|
||||
|
||||
### Thanks
|
||||
|
||||
* hirura
|
||||
|
||||
## 3.2.2 - 2019-06-03 {#version-3-2-2}
|
||||
|
||||
### Fixes
|
||||
|
||||
* xpath: Fixed a bug for equality and relational expressions.
|
||||
[GitHub#17][Reported by Mirko Budszuhn]
|
||||
|
||||
* xpath: Fixed `boolean()` implementation.
|
||||
|
||||
* xpath: Fixed `local_name()` with nonexistent node.
|
||||
|
||||
* xpath: Fixed `number()` implementation with node set.
|
||||
[GitHub#18][Reported by Mirko Budszuhn]
|
||||
|
||||
### Thanks
|
||||
|
||||
* Mirko Budszuhn
|
||||
|
||||
## 3.2.1 - 2019-05-04 {#version-3-2-1}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Improved error message.
|
||||
[GitHub#12][Patch by FUJI Goro]
|
||||
|
||||
* Improved error message.
|
||||
[GitHub#16][Patch by ujihisa]
|
||||
|
||||
* Improved documentation markup.
|
||||
[GitHub#14][Patch by Alyssa Ross]
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that `nil` variable value raises an unexpected exception.
|
||||
[GitHub#13][Patch by Alyssa Ross]
|
||||
|
||||
### Thanks
|
||||
|
||||
* FUJI Goro
|
||||
|
||||
* Alyssa Ross
|
||||
|
||||
* ujihisa
|
||||
|
||||
## 3.2.0 - 2019-01-01 {#version-3-2-0}
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that no namespace attribute isn't matched with prefix.
|
||||
|
||||
[ruby-list:50731][Reported by Yasuhiro KIMURA]
|
||||
|
||||
* Fixed a bug that the default namespace is applied to attribute names.
|
||||
|
||||
NOTE: It's a backward incompatible change. If your program has any
|
||||
problem with this change, please report it. We may revert this fix.
|
||||
|
||||
* `REXML::Attribute#prefix` returns `""` for no namespace attribute.
|
||||
|
||||
* `REXML::Attribute#namespace` returns `""` for no namespace attribute.
|
||||
|
||||
### Thanks
|
||||
|
||||
* Yasuhiro KIMURA
|
||||
|
||||
## 3.1.9 - 2018-12-20 {#version-3-1-9}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Improved backward compatibility.
|
||||
|
||||
Restored `REXML::Parsers::BaseParser::UNQME_STR` because it's used
|
||||
by kramdown.
|
||||
|
||||
## 3.1.8 - 2018-12-20 {#version-3-1-8}
|
||||
|
||||
### Improvements
|
||||
|
||||
* Added support for customizing quote character in prologue.
|
||||
[GitHub#8][Bug #9367][Reported by Takashi Oguma]
|
||||
|
||||
* You can use `"` as quote character by specifying `:quote` to
|
||||
`REXML::Document#context[:prologue_quote]`.
|
||||
|
||||
* You can use `'` as quote character by specifying `:apostrophe`
|
||||
to `REXML::Document#context[:prologue_quote]`.
|
||||
|
||||
* Added processing instruction target check. The target must not nil.
|
||||
[GitHub#7][Reported by Ariel Zelivansky]
|
||||
|
||||
* Added name check for element and attribute.
|
||||
[GitHub#7][Reported by Ariel Zelivansky]
|
||||
|
||||
* Stopped to use `Exception`.
|
||||
[GitHub#9][Patch by Jean Boussier]
|
||||
|
||||
### Fixes
|
||||
|
||||
* Fixed a bug that `REXML::Text#clone` escapes value twice.
|
||||
[ruby-dev:50626][Bug #15058][Reported by Ryosuke Nanba]
|
||||
|
||||
### Thanks
|
||||
|
||||
* Takashi Oguma
|
||||
|
||||
* Ariel Zelivansky
|
||||
|
||||
* Jean Boussier
|
||||
|
||||
* Ryosuke Nanba
|
||||
@@ -0,0 +1,57 @@
|
||||
# REXML
|
||||
|
||||
REXML was inspired by the Electric XML library for Java, which features an easy-to-use API, small size, and speed. Hopefully, REXML, designed with the same philosophy, has these same features. I've tried to keep the API as intuitive as possible, and have followed the Ruby methodology for method naming and code flow, rather than mirroring the Java API.
|
||||
|
||||
REXML supports both tree and stream document parsing. Stream parsing is faster (about 1.5 times as fast). However, with stream parsing, you don't get access to features such as XPath.
|
||||
|
||||
## API
|
||||
|
||||
See the [API documentation](https://ruby.github.io/rexml/).
|
||||
|
||||
## Usage
|
||||
|
||||
We'll start with parsing an XML document
|
||||
|
||||
```ruby
|
||||
require "rexml/document"
|
||||
file = File.new( "mydoc.xml" )
|
||||
doc = REXML::Document.new file
|
||||
```
|
||||
|
||||
Line 3 creates a new document and parses the supplied file. You can also do the following
|
||||
|
||||
```ruby
|
||||
require "rexml/document"
|
||||
include REXML # so that we don't have to prefix everything with REXML::...
|
||||
string = <<EOF
|
||||
<mydoc>
|
||||
<someelement attribute="nanoo">Text, text, text</someelement>
|
||||
</mydoc>
|
||||
EOF
|
||||
doc = Document.new string
|
||||
```
|
||||
|
||||
So parsing a string is just as easy as parsing a file.
|
||||
|
||||
## Support
|
||||
|
||||
REXML support follows the same maintenance cycle as Ruby releases, as shown on <https://www.ruby-lang.org/en/downloads/branches/>.
|
||||
|
||||
If you are running on an end-of-life Ruby, do not expect modern REXML releases to be compatible with it; in fact, it's recommended that you DO NOT use this gem, and instead use the REXML version that came bundled with your end-of-life Ruby version.
|
||||
|
||||
The `required_ruby_version` on the gemspec is kept updated on a [best-effort basis](https://github.com/ruby/rexml/pull/70) by the community.
|
||||
Up to version 3.2.5, this information was not set. That version [is known broken with at least Ruby < 2.3](https://github.com/ruby/rexml/issues/69).
|
||||
|
||||
## Development
|
||||
|
||||
After checking out the repo, run `rake test` to run the tests.
|
||||
|
||||
To install this gem onto your local machine, run `bundle exec rake install`. To release a new version, update the version number in `version.rb`, and then run `bundle exec rake release`, which will create a git tag for the version, push git commits and tags, and push the `.gem` file to [rubygems.org](https://rubygems.org).
|
||||
|
||||
## Contributing
|
||||
|
||||
Bug reports and pull requests are welcome on GitHub at https://github.com/ruby/rexml.
|
||||
|
||||
## License
|
||||
|
||||
The gem is available as open source under the terms of the [BSD-2-Clause](LICENSE.txt).
|
||||
@@ -0,0 +1,143 @@
|
||||
== Element Context
|
||||
|
||||
Notes:
|
||||
- All code on this page presupposes that the following has been executed:
|
||||
|
||||
require 'rexml/document'
|
||||
|
||||
- For convenience, examples on this page use +REXML::Document.new+, not +REXML::Element.new+.
|
||||
This is completely valid, because REXML::Document is a subclass of REXML::Element.
|
||||
|
||||
The context for an element is a hash of processing directives
|
||||
that influence the way \XML is read, stored, and written.
|
||||
The context entries are:
|
||||
|
||||
- +:respect_whitespace+: controls treatment of whitespace.
|
||||
- +:compress_whitespace+: determines whether whitespace is compressed.
|
||||
- +:ignore_whitespace_nodes+: determines whether whitespace-only nodes are to be ignored.
|
||||
- +:raw+: controls treatment of special characters and entities.
|
||||
|
||||
The default context for a new element is <tt>{}</tt>.
|
||||
You can set the context at element-creation time:
|
||||
|
||||
d = REXML::Document.new('', {compress_whitespace: :all, raw: :all})
|
||||
d.context # => {:compress_whitespace=>:all, :raw=>:all}
|
||||
|
||||
You can reset the entire context by assigning a new hash:
|
||||
|
||||
d.context = {ignore_whitespace_nodes: :all}
|
||||
d.context # => {:ignore_whitespace_nodes=>:all}
|
||||
|
||||
Or you can create or modify an individual entry:
|
||||
|
||||
d.context[:raw] = :all
|
||||
d.context # => {:ignore_whitespace_nodes=>:all, :raw=>:all}
|
||||
|
||||
=== +:respect_whitespace+
|
||||
|
||||
Affects: +REXML::Element.new+, +REXML::Element.text=+.
|
||||
|
||||
By default, all parsed whitespace is respected (that is, stored whitespace not compressed):
|
||||
|
||||
xml_string = '<root><foo>a b</foo> <bar>c d</bar> <baz>e f</baz></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.to_s # => "<root><foo>a b</foo> <bar>c d</bar> <baz>e f</baz></root>"
|
||||
|
||||
Use +:respect_whitespace+ with an array of element names
|
||||
to specify the elements that _are_ to have their whitespace respected;
|
||||
other elements' whitespace, and whitespace between elements, will be compressed.
|
||||
|
||||
In this example: +foo+ and +baz+ will have their whitespace respected;
|
||||
+bar+ and the space between elements will have their whitespace compressed:
|
||||
|
||||
d = REXML::Document.new(xml_string, {respect_whitespace: ['foo', 'baz']})
|
||||
d.to_s # => "<root><foo>a b</foo> <bar>c d</bar> <baz>e f</baz></root>"
|
||||
bar = d.root[2] # => <bar> ... </>
|
||||
bar.text = 'X Y'
|
||||
d.to_s # => "<root><foo>a b</foo> <bar>X Y</bar> <baz>e f</baz></root>"
|
||||
|
||||
=== +:compress_whitespace+
|
||||
|
||||
Affects: +REXML::Element.new+, +REXML::Element.text=+.
|
||||
|
||||
Use <tt>compress_whitespace: :all</tt>
|
||||
to compress whitespace both within and between elements:
|
||||
|
||||
xml_string = '<root><foo>a b</foo> <bar>c d</bar> <baz>e f</baz></root>'
|
||||
d = REXML::Document.new(xml_string, {compress_whitespace: :all})
|
||||
d.to_s # => "<root><foo>a b</foo> <bar>c d</bar> <baz>e f</baz></root>"
|
||||
|
||||
Use +:compress_whitespace+ with an array of element names
|
||||
to compress whitespace in those elements,
|
||||
but not in other elements nor between elements.
|
||||
|
||||
In this example, +foo+ and +baz+ will have their whitespace compressed;
|
||||
+bar+ and the space between elements will not:
|
||||
|
||||
d = REXML::Document.new(xml_string, {compress_whitespace: ['foo', 'baz']})
|
||||
d.to_s # => "<root><foo>a b</foo> <bar>c d</bar> <baz>e f</baz></root>"
|
||||
foo = d.root[0] # => <foo> ... </>
|
||||
foo.text= 'X Y'
|
||||
d.to_s # => "<root><foo>X Y</foo> <bar>c d</bar> <baz>e f</baz></root>"
|
||||
|
||||
=== +:ignore_whitespace_nodes+
|
||||
|
||||
Affects: +REXML::Element.new+.
|
||||
|
||||
Use <tt>ignore_whitespace_nodes: :all</tt> to omit all whitespace-only elements.
|
||||
|
||||
In this example, +bar+ has a text node, while nodes +foo+ and +baz+ do not:
|
||||
|
||||
xml_string = '<root><foo> </foo><bar> BAR </bar><baz> </baz></root>'
|
||||
d = REXML::Document.new(xml_string, {ignore_whitespace_nodes: :all})
|
||||
d.to_s # => "<root><foo> FOO </foo><bar/><baz> BAZ </baz></root>"
|
||||
root = d.root # => <root> ... </>
|
||||
foo = root[0] # => <foo/>
|
||||
bar = root[1] # => <bar> ... </>
|
||||
baz = root[2] # => <baz/>
|
||||
foo.first.class # => NilClass
|
||||
bar.first.class # => REXML::Text
|
||||
baz.first.class # => NilClass
|
||||
|
||||
Use +:ignore_whitespace_nodes+ with an array of element names
|
||||
to specify the elements that are to have whitespace nodes ignored.
|
||||
|
||||
In this example, +bar+ and +baz+ have text nodes, while node +foo+ does not.
|
||||
|
||||
xml_string = '<root><foo> </foo><bar> BAR </bar><baz> </baz></root>'
|
||||
d = REXML::Document.new(xml_string, {ignore_whitespace_nodes: ['foo']})
|
||||
d.to_s # => "<root><foo/><bar> BAR </bar><baz> </baz></root>"
|
||||
root = d.root # => <root> ... </>
|
||||
foo = root[0] # => <foo/>
|
||||
bar = root[1] # => <bar> ... </>
|
||||
baz = root[2] # => <baz> ... </>
|
||||
foo.first.class # => NilClass
|
||||
bar.first.class # => REXML::Text
|
||||
baz.first.class # => REXML::Text
|
||||
|
||||
=== +:raw+
|
||||
|
||||
Affects: +Element.text=+, +Element.add_text+, +Text.to_s+.
|
||||
|
||||
Parsing of +a+ elements is not affected by +raw+:
|
||||
|
||||
xml_string = '<root><a>0 < 1</a><b>1 > 0</b></root>'
|
||||
d = REXML::Document.new(xml_string, {:raw => ['a']})
|
||||
d.root.to_s # => "<root><a>0 < 1</a><b>1 > 0</b></root>"
|
||||
a, b = *d.root.elements
|
||||
a.to_s # => "<a>0 < 1</a>"
|
||||
b.to_s # => "<b>1 > 0</b>"
|
||||
|
||||
But Element#text= is affected:
|
||||
|
||||
a.text = '0 < 1'
|
||||
b.text = '1 > 0'
|
||||
a.to_s # => "<a>0 < 1</a>"
|
||||
b.to_s # => "<b>1 &gt; 0</b>"
|
||||
|
||||
As is Element.add_text:
|
||||
|
||||
a.add_text(' so 1 > 0')
|
||||
b.add_text(' so 0 < 1')
|
||||
a.to_s # => "<a>0 < 1 so 1 > 0</a>"
|
||||
b.to_s # => "<b>1 &gt; 0 so 0 &lt; 1</b>"
|
||||
@@ -0,0 +1,87 @@
|
||||
== Class Child
|
||||
|
||||
Class Child includes module Node;
|
||||
see {Tasks for Node}[node_rdoc.html].
|
||||
|
||||
:include: ../tocs/child_toc.rdoc
|
||||
|
||||
=== Relationships
|
||||
|
||||
==== Task: Set the Parent
|
||||
|
||||
Use method {Child#parent=}[../../../../REXML/Parent.html#method-i-parent-3D]
|
||||
to set the parent:
|
||||
|
||||
e0 = REXML::Element.new('foo')
|
||||
e1 = REXML::Element.new('bar')
|
||||
e1.parent # => nil
|
||||
e1.parent = e0
|
||||
e1.parent # => <foo/>
|
||||
|
||||
==== Task: Insert Previous Sibling
|
||||
|
||||
Use method {Child#previous_sibling=}[../../../../REXML/Parent.html#method-i-previous_sibling-3D]
|
||||
to insert a previous sibling:
|
||||
|
||||
xml_string = '<root><a/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.to_a # => [<a/>, <c/>]
|
||||
c = d.root[1] # => <c/>
|
||||
b = REXML::Element.new('b')
|
||||
c.previous_sibling = b
|
||||
d.root.to_a # => [<a/>, <b/>, <c/>]
|
||||
|
||||
==== Task: Insert Next Sibling
|
||||
|
||||
Use method {Child#next_sibling=}[../../../../REXML/Parent.html#method-i-next-sibling-3D]
|
||||
to insert a previous sibling:
|
||||
|
||||
xml_string = '<root><a/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.to_a # => [<a/>, <c/>]
|
||||
a = d.root[0] # => <a/>
|
||||
b = REXML::Element.new('b')
|
||||
a.next_sibling = b
|
||||
d.root.to_a # => [<a/>, <b/>, <c/>]
|
||||
|
||||
=== Removal or Replacement
|
||||
|
||||
==== Task: Remove Child from Parent
|
||||
|
||||
Use method {Child#remove}[../../../../REXML/Parent.html#method-i-remove]
|
||||
to remove a child from its parent; returns the removed child:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.to_a # => [<a/>, <b/>, <c/>]
|
||||
b = d.root[1] # => <b/>
|
||||
b.remove # => <b/>
|
||||
d.root.to_a # => [<a/>, <c/>]
|
||||
|
||||
==== Task: Replace Child
|
||||
|
||||
Use method {Child#replace_with}[../../../../REXML/Parent.html#method-i-replace]
|
||||
to replace a child;
|
||||
returns the replaced child:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.to_a # => [<a/>, <b/>, <c/>]
|
||||
b = d.root[1] # => <b/>
|
||||
d = REXML::Element.new('d')
|
||||
b.replace_with(d) # => <b/>
|
||||
d.root.to_a # => [<a/>, <d/>, <c/>]
|
||||
|
||||
=== Document
|
||||
|
||||
==== Task: Get the Document
|
||||
|
||||
Use method {Child#document}[../../../../REXML/Parent.html#method-i-document]
|
||||
to get the document for the child:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.to_a # => [<a/>, <b/>, <c/>]
|
||||
b = d.root[1] # => <b/>
|
||||
b.document == d # => true
|
||||
REXML::Child.new.document # => nil
|
||||
@@ -0,0 +1,276 @@
|
||||
== Class Document
|
||||
|
||||
Class Document has methods from its superclasses and included modules;
|
||||
see:
|
||||
|
||||
- {Tasks for Element}[element_rdoc.html].
|
||||
- {Tasks for Parent}[parent_rdoc.html].
|
||||
- {Tasks for Child}[child_rdoc.html].
|
||||
- {Tasks for Node}[node_rdoc.html].
|
||||
- {Module Enumerable}[https://docs.ruby-lang.org/en/master/Enumerable.html].
|
||||
|
||||
:include: ../tocs/document_toc.rdoc
|
||||
|
||||
=== New Document
|
||||
|
||||
==== Task: Create an Empty Document
|
||||
|
||||
Use method {Document::new}[../../../../REXML/Document.html#method-c-new]
|
||||
to create an empty document.
|
||||
|
||||
d = REXML::Document.new
|
||||
|
||||
==== Task: Parse a \String into a New Document
|
||||
|
||||
Use method {Document::new}[../../../../REXML/Document.html#method-c-new]
|
||||
to parse an XML string into a new document:
|
||||
|
||||
xml_string = '<root><a/>text<b/>more<c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root # => <root> ... </>
|
||||
|
||||
==== Task: Parse an \IO Stream into a New Document
|
||||
|
||||
Use method {Document::new}[../../../../REXML/Document.html#method-c-new]
|
||||
to parse an XML \IO stream into a new document:
|
||||
|
||||
xml_string = '<root><a/>text<b/>more<c/></root>'
|
||||
File.write('t.xml', xml_string)
|
||||
d = File.open('t.xml', 'r') do |file|
|
||||
REXML::Document.new(file)
|
||||
end
|
||||
d.root # => <root> ... </>
|
||||
|
||||
==== Task: Create a Document from an Existing Document
|
||||
|
||||
Use method {Document::new}[../../../../REXML/Document.html#method-c-new]
|
||||
to create a document from an existing document.
|
||||
The context and attributes are copied to the new document,
|
||||
but not the children:
|
||||
|
||||
xml_string = '<root><a/>text<b/>more<c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.children # => [<root> ... </>]
|
||||
d.context = {raw: :all, compress_whitespace: :all}
|
||||
d.add_attributes({'bar' => 0, 'baz' => 1})
|
||||
d1 = REXML::Document.new(d)
|
||||
d1.context # => {:raw=>:all, :compress_whitespace=>:all}
|
||||
d1.attributes # => {"bar"=>bar='0', "baz"=>baz='1'}
|
||||
d1.children # => []
|
||||
|
||||
==== Task: Clone a Document
|
||||
|
||||
Use method {Document#clone}[../../../../REXML/Document.html#method-i-clone]
|
||||
to clone a document.
|
||||
The context and attributes are copied to the new document,
|
||||
but not the children:
|
||||
|
||||
xml_string = '<root><a/>text<b/>more<c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.children # => [<root> ... </>]
|
||||
d.context = {raw: :all, compress_whitespace: :all}
|
||||
d.add_attributes({'bar' => 0, 'baz' => 1})
|
||||
d1 = d.clone # => < bar='0' baz='1'/>
|
||||
d1.context # => {:raw=>:all, :compress_whitespace=>:all}
|
||||
d1.attributes # => {"bar"=>bar='0', "baz"=>baz='1'}
|
||||
d1.children # => []
|
||||
|
||||
=== Document Type
|
||||
|
||||
==== Task: Get the Document Type
|
||||
|
||||
Use method {Document#doctype}[../../../../REXML/Document.html#method-i-doctype]
|
||||
to get the document type:
|
||||
|
||||
d = REXML::Document.new('<!DOCTYPE document SYSTEM "subjects.dtd">')
|
||||
d.doctype.class # => REXML::DocType
|
||||
d = REXML::Document.new('')
|
||||
d.doctype.class # => nil
|
||||
|
||||
==== Task: Set the Document Type
|
||||
|
||||
Use method {document#add}[../../../../REXML/Document.html#method-i-add]
|
||||
to add or replace the document type:
|
||||
|
||||
d = REXML::Document.new('')
|
||||
d.doctype.class # => nil
|
||||
d.add(REXML::DocType.new('foo'))
|
||||
d.doctype.class # => REXML::DocType
|
||||
|
||||
=== XML Declaration
|
||||
|
||||
==== Task: Get the XML Declaration
|
||||
|
||||
Use method {document#xml_decl}[../../../../REXML/Document.html#method-i-xml_decl]
|
||||
to get the XML declaration:
|
||||
|
||||
d = REXML::Document.new('<!DOCTYPE document SYSTEM "subjects.dtd">')
|
||||
d.xml_decl.class # => REXML::XMLDecl
|
||||
d.xml_decl # => <?xml ... ?>
|
||||
d = REXML::Document.new('')
|
||||
d.xml_decl.class # => REXML::XMLDecl
|
||||
d.xml_decl # => <?xml ... ?>
|
||||
|
||||
==== Task: Set the XML Declaration
|
||||
|
||||
Use method {document#add}[../../../../REXML/Document.html#method-i-add]
|
||||
to replace the XML declaration:
|
||||
|
||||
d = REXML::Document.new('<!DOCTYPE document SYSTEM "subjects.dtd">')
|
||||
d.add(REXML::XMLDecl.new)
|
||||
|
||||
=== Children
|
||||
|
||||
==== Task: Add an Element Child
|
||||
|
||||
Use method
|
||||
{document#add_element}[../../../../REXML/Document.html#method-i-add_element]
|
||||
to add an element to the document:
|
||||
|
||||
d = REXML::Document.new('')
|
||||
d.add_element(REXML::Element.new('root'))
|
||||
d.children # => [<root/>]
|
||||
|
||||
==== Task: Add a Non-Element Child
|
||||
|
||||
Use method
|
||||
{document#add}[../../../../REXML/Document.html#method-i-add]
|
||||
to add a non-element to the document:
|
||||
|
||||
xml_string = '<root><a/>text<b/>more<c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.add(REXML::Text.new('foo'))
|
||||
d.children # => [<root> ... </>, "foo"]
|
||||
|
||||
=== Writing
|
||||
|
||||
==== Task: Write to $stdout
|
||||
|
||||
Use method
|
||||
{document#write}[../../../../REXML/Document.html#method-i-write]
|
||||
to write the document to <tt>$stdout</tt>:
|
||||
|
||||
xml_string = '<root><a/>text<b/>more<c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.write
|
||||
|
||||
Output:
|
||||
|
||||
<root><a/>text<b/>more<c/></root>
|
||||
|
||||
==== Task: Write to IO Stream
|
||||
|
||||
Use method
|
||||
{document#write}[../../../../REXML/Document.html#method-i-write]
|
||||
to write the document to <tt>$stdout</tt>:
|
||||
|
||||
xml_string = '<root><a/>text<b/>more<c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
File.open('t.xml', 'w') do |file|
|
||||
d.write(file)
|
||||
end
|
||||
p File.read('t.xml')
|
||||
|
||||
Output:
|
||||
|
||||
"<root><a/>text<b/>more<c/></root>"
|
||||
|
||||
==== Task: Write with No Indentation
|
||||
|
||||
Use method
|
||||
{document#write}[../../../../REXML/Document.html#method-i-write]
|
||||
to write the document with no indentation:
|
||||
|
||||
xml_string = '<root><a><b><c></c></b></a></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.write({indent: 0})
|
||||
|
||||
Output:
|
||||
|
||||
<root>
|
||||
<a>
|
||||
<b>
|
||||
<c/>
|
||||
</b>
|
||||
</a>
|
||||
</root>
|
||||
|
||||
==== Task: Write with Specified Indentation
|
||||
|
||||
Use method
|
||||
{document#write}[../../../../REXML/Document.html#method-i-write]
|
||||
to write the document with a specified indentation:
|
||||
|
||||
xml_string = '<root><a><b><c></c></b></a></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.write({indent: 2})
|
||||
|
||||
Output:
|
||||
|
||||
<root>
|
||||
<a>
|
||||
<b>
|
||||
<c/>
|
||||
</b>
|
||||
</a>
|
||||
</root>
|
||||
|
||||
=== Querying
|
||||
|
||||
==== Task: Get the Document
|
||||
|
||||
Use method
|
||||
{document#document}[../../../../REXML/Document.html#method-i-document]
|
||||
to get the document (+self+); overrides <tt>Element#document</tt>:
|
||||
|
||||
xml_string = '<root><a><b><c></c></b></a></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.document == d # => true
|
||||
|
||||
==== Task: Get the Encoding
|
||||
|
||||
Use method
|
||||
{document#document}[../../../../REXML/Document.html#method-i-document]
|
||||
to get the document (+self+); overrides <tt>Element#document</tt>:
|
||||
|
||||
xml_string = '<root><a><b><c></c></b></a></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.encoding # => "UTF-8"
|
||||
|
||||
==== Task: Get the Node Type
|
||||
|
||||
Use method
|
||||
{document#node_type}[../../../../REXML/Document.html#method-i-node_type]
|
||||
to get the node type (+:document+); overrides <tt>Element#node_type</tt>:
|
||||
|
||||
xml_string = '<root><a><b><c></c></b></a></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.node_type # => :document
|
||||
|
||||
==== Task: Get the Root Element
|
||||
|
||||
Use method
|
||||
{document#root}[../../../../REXML/Document.html#method-i-root]
|
||||
to get the root element:
|
||||
|
||||
xml_string = '<root><a><b><c></c></b></a></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root # => <root> ... </>
|
||||
|
||||
==== Task: Determine Whether Stand-Alone
|
||||
|
||||
Use method
|
||||
{document#stand_alone?}[../../../../REXML/Document.html#method-i-stand_alone-3F]
|
||||
to get the stand-alone value:
|
||||
|
||||
d = REXML::Document.new('<?xml standalone="yes"?>')
|
||||
d.stand_alone? # => "yes"
|
||||
|
||||
==== Task: Get the Version
|
||||
|
||||
Use method
|
||||
{document#version}[../../../../REXML/Document.html#method-i-version]
|
||||
to get the version:
|
||||
|
||||
d = REXML::Document.new('<?xml version="2.0" encoding="UTF-8"?>')
|
||||
d.version # => "2.0"
|
||||
@@ -0,0 +1,602 @@
|
||||
== Class Element
|
||||
|
||||
Class Element has methods from its superclasses and included modules;
|
||||
see:
|
||||
|
||||
- {Tasks for Parent}[parent_rdoc.html].
|
||||
- {Tasks for Child}[child_rdoc.html].
|
||||
- {Tasks for Node}[node_rdoc.html].
|
||||
- {Module Enumerable}[https://docs.ruby-lang.org/en/master/Enumerable.html].
|
||||
|
||||
:include: ../tocs/element_toc.rdoc
|
||||
|
||||
=== New Element
|
||||
|
||||
==== Task: Create a Default Element
|
||||
|
||||
Use method
|
||||
{Element::new}[../../../../REXML/Element.html#method-c-new]
|
||||
with no arguments to create a default element:
|
||||
|
||||
e = REXML::Element.new
|
||||
e.name # => "UNDEFINED"
|
||||
e.parent # => nil
|
||||
e.context # => nil
|
||||
|
||||
==== Task: Create a Named Element
|
||||
|
||||
Use method
|
||||
{Element::new}[../../../../REXML/Element.html#method-c-new]
|
||||
with a string name argument
|
||||
to create a named element:
|
||||
|
||||
e = REXML::Element.new('foo')
|
||||
e.name # => "foo"
|
||||
e.parent # => nil
|
||||
e.context # => nil
|
||||
|
||||
==== Task: Create an Element with Name and Parent
|
||||
|
||||
Use method
|
||||
{Element::new}[../../../../REXML/Element.html#method-c-new]
|
||||
with name and parent arguments
|
||||
to create an element with name and parent:
|
||||
|
||||
p = REXML::Parent.new
|
||||
e = REXML::Element.new('foo', p)
|
||||
e.name # => "foo"
|
||||
e.parent # => #<REXML::Parent @parent=nil, @children=[<foo/>]>
|
||||
e.context # => nil
|
||||
|
||||
==== Task: Create an Element with Name, Parent, and Context
|
||||
|
||||
Use method
|
||||
{Element::new}[../../../../REXML/Element.html#method-c-new]
|
||||
with name, parent, and context arguments
|
||||
to create an element with name, parent, and context:
|
||||
|
||||
p = REXML::Parent.new
|
||||
e = REXML::Element.new('foo', p, {compress_whitespace: :all})
|
||||
e.name # => "foo"
|
||||
e.parent # => #<REXML::Parent @parent=nil, @children=[<foo/>]>
|
||||
e.context # => {:compress_whitespace=>:all}
|
||||
|
||||
==== Task: Create a Shallow Clone
|
||||
|
||||
Use method
|
||||
{Element#clone}[../../../../REXML/Element.html#method-i-clone]
|
||||
to create a shallow clone of an element,
|
||||
copying only the name, attributes, and context:
|
||||
|
||||
e0 = REXML::Element.new('foo', nil, {compress_whitespace: :all})
|
||||
e0.add_attribute(REXML::Attribute.new('bar', 'baz'))
|
||||
e0.context = {compress_whitespace: :all}
|
||||
e1 = e0.clone # => <foo bar='baz'/>
|
||||
e1.name # => "foo"
|
||||
e1.context # => {:compress_whitespace=>:all}
|
||||
|
||||
=== Attributes
|
||||
|
||||
==== Task: Create and Add an Attribute
|
||||
|
||||
Use method
|
||||
{Element#add_attribute}[../../../../REXML/Element.html#method-i-add_attribute]
|
||||
to create and add an attribute:
|
||||
|
||||
e = REXML::Element.new
|
||||
e.add_attribute('attr', 'value') # => "value"
|
||||
e['attr'] # => "value"
|
||||
e.add_attribute('attr', 'VALUE') # => "VALUE"
|
||||
e['attr'] # => "VALUE"
|
||||
|
||||
==== Task: Add an Existing Attribute
|
||||
|
||||
Use method
|
||||
{Element#add_attribute}[../../../../REXML/Element.html#method-i-add_attribute]
|
||||
to add an existing attribute:
|
||||
|
||||
e = REXML::Element.new
|
||||
a = REXML::Attribute.new('attr', 'value')
|
||||
e.add_attribute(a)
|
||||
e['attr'] # => "value"
|
||||
a = REXML::Attribute.new('attr', 'VALUE')
|
||||
e.add_attribute(a)
|
||||
e['attr'] # => "VALUE"
|
||||
|
||||
==== Task: Add Multiple Attributes from a Hash
|
||||
|
||||
Use method
|
||||
{Element#add_attributes}[../../../../REXML/Element.html#method-i-add_attributes]
|
||||
to add multiple attributes from a hash:
|
||||
|
||||
e = REXML::Element.new
|
||||
h = {'foo' => 0, 'bar' => 1}
|
||||
e.add_attributes(h)
|
||||
e['foo'] # => "0"
|
||||
e['bar'] # => "1"
|
||||
|
||||
==== Task: Add Multiple Attributes from an Array
|
||||
|
||||
Use method
|
||||
{Element#add_attributes}[../../../../REXML/Element.html#method-i-add_attributes]
|
||||
to add multiple attributes from an array:
|
||||
|
||||
e = REXML::Element.new
|
||||
a = [['foo', 0], ['bar', 1]]
|
||||
e.add_attributes(a)
|
||||
e['foo'] # => "0"
|
||||
e['bar'] # => "1"
|
||||
|
||||
==== Task: Retrieve the Value for an Attribute Name
|
||||
|
||||
Use method
|
||||
{Element#[]}[../../../../REXML/Element.html#method-i-5B-5D]
|
||||
to retrieve the value for an attribute name:
|
||||
|
||||
e = REXML::Element.new
|
||||
e.add_attribute('attr', 'value') # => "value"
|
||||
e['attr'] # => "value"
|
||||
|
||||
==== Task: Retrieve the Attribute Value for a Name and Namespace
|
||||
|
||||
Use method
|
||||
{Element#attribute}[../../../../REXML/Element.html#method-i-attribute]
|
||||
to retrieve the value for an attribute name:
|
||||
|
||||
xml_string = "<root xmlns:a='a' a:x='a:x' x='x'/>"
|
||||
d = REXML::Document.new(xml_string)
|
||||
e = d.root
|
||||
e.attribute("x") # => x='x'
|
||||
e.attribute("x", "a") # => a:x='a:x'
|
||||
|
||||
==== Task: Delete an Attribute
|
||||
|
||||
Use method
|
||||
{Element#delete_attribute}[../../../../REXML/Element.html#method-i-delete_attribute]
|
||||
to remove an attribute:
|
||||
|
||||
e = REXML::Element.new('foo')
|
||||
e.add_attribute('bar', 'baz')
|
||||
e.delete_attribute('bar')
|
||||
e.delete_attribute('bar')
|
||||
e['bar'] # => nil
|
||||
|
||||
==== Task: Determine Whether the Element Has Attributes
|
||||
|
||||
Use method
|
||||
{Element#has_attributes?}[../../../../REXML/Element.html#method-i-has_attributes-3F]
|
||||
to determine whether the element has attributes:
|
||||
|
||||
e = REXML::Element.new('foo')
|
||||
e.has_attributes? # => false
|
||||
e.add_attribute('bar', 'baz')
|
||||
e.has_attributes? # => true
|
||||
|
||||
=== Children
|
||||
|
||||
<em>Element Children</em>
|
||||
|
||||
==== Task: Create and Add an Element
|
||||
|
||||
Use method
|
||||
{Element#add_element}[../../../../REXML/Element.html#method-i-add_element]
|
||||
to create a new element and add it to this element:
|
||||
|
||||
e0 = REXML::Element.new('foo')
|
||||
e0.add_element('bar')
|
||||
e0.children # => [<bar/>]
|
||||
|
||||
==== Task: Add an Existing Element
|
||||
|
||||
Use method
|
||||
{Element#add_element}[../../../../REXML/Element.html#method-i-add_element]
|
||||
to add an element to this element:
|
||||
|
||||
e0 = REXML::Element.new('foo')
|
||||
e1 = REXML::Element.new('bar')
|
||||
e0.add_element(e1)
|
||||
e0.children # => [<bar/>]
|
||||
|
||||
==== Task: Create and Add an Element with Attributes
|
||||
|
||||
Use method
|
||||
{Element#add_element}[../../../../REXML/Element.html#method-i-add_element]
|
||||
to create a new element with attributes, and add it to this element:
|
||||
|
||||
e0 = REXML::Element.new('foo')
|
||||
e0.add_element('bar', {'name' => 'value'})
|
||||
e0.children # => [<bar name='value'/>]
|
||||
|
||||
==== Task: Add an Existing Element with Added Attributes
|
||||
|
||||
Use method
|
||||
{Element#add_element}[../../../../REXML/Element.html#method-i-add_element]
|
||||
to add an element to this element:
|
||||
|
||||
e0 = REXML::Element.new('foo')
|
||||
e1 = REXML::Element.new('bar')
|
||||
e0.add_element(e1, {'name' => 'value'})
|
||||
e0.children # => [<bar name='value'/>]
|
||||
|
||||
==== Task: Delete a Specified Element
|
||||
|
||||
Use method
|
||||
{Element#delete_element}[../../../../REXML/Element.html#method-i-delete_element]
|
||||
to remove a specified element from this element:
|
||||
|
||||
e0 = REXML::Element.new('foo')
|
||||
e1 = REXML::Element.new('bar')
|
||||
e0.add_element(e1)
|
||||
e0.children # => [<bar/>]
|
||||
e0.delete_element(e1)
|
||||
e0.children # => []
|
||||
|
||||
==== Task: Delete an Element by Index
|
||||
|
||||
Use method
|
||||
{Element#delete_element}[../../../../REXML/Element.html#method-i-delete_element]
|
||||
to remove an element from this element by index:
|
||||
|
||||
e0 = REXML::Element.new('foo')
|
||||
e1 = REXML::Element.new('bar')
|
||||
e0.add_element(e1)
|
||||
e0.children # => [<bar/>]
|
||||
e0.delete_element(1)
|
||||
e0.children # => []
|
||||
|
||||
==== Task: Delete an Element by XPath
|
||||
|
||||
Use method
|
||||
{Element#delete_element}[../../../../REXML/Element.html#method-i-delete_element]
|
||||
to remove an element from this element by XPath:
|
||||
|
||||
e0 = REXML::Element.new('foo')
|
||||
e1 = REXML::Element.new('bar')
|
||||
e0.add_element(e1)
|
||||
e0.children # => [<bar/>]
|
||||
e0.delete_element('//bar/')
|
||||
e0.children # => []
|
||||
|
||||
==== Task: Determine Whether Element Children
|
||||
|
||||
Use method
|
||||
{Element#has_elements?}[../../../../REXML/Element.html#method-i-has_elements-3F]
|
||||
to determine whether the element has element children:
|
||||
|
||||
e0 = REXML::Element.new('foo')
|
||||
e0.has_elements? # => false
|
||||
e0.add_element(REXML::Element.new('bar'))
|
||||
e0.has_elements? # => true
|
||||
|
||||
==== Task: Get Element Descendants by XPath
|
||||
|
||||
Use method
|
||||
{Element#get_elements}[../../../../REXML/Element.html#method-i-get_elements]
|
||||
to fetch all element descendant children by XPath:
|
||||
|
||||
xml_string = <<-EOT
|
||||
<root>
|
||||
<a level='1'>
|
||||
<a level='2'/>
|
||||
</a>
|
||||
</root>
|
||||
EOT
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.get_elements('//a') # => [<a level='1'> ... </>, <a level='2'/>]
|
||||
|
||||
==== Task: Get Next Element Sibling
|
||||
|
||||
Use method
|
||||
{Element#next_element}[../../../../REXML/Element.html#method-i-next_element]
|
||||
to retrieve the next element sibling:
|
||||
|
||||
d = REXML::Document.new '<a><b/>text<c/></a>'
|
||||
d.root.elements['b'].next_element #-> <c/>
|
||||
d.root.elements['c'].next_element #-> nil
|
||||
|
||||
==== Task: Get Previous Element Sibling
|
||||
|
||||
Use method
|
||||
{Element#previous_element}[../../../../REXML/Element.html#method-i-previous_element]
|
||||
to retrieve the previous element sibling:
|
||||
|
||||
d = REXML::Document.new '<a><b/>text<c/></a>'
|
||||
d.root.elements['c'].previous_element #-> <b/>
|
||||
d.root.elements['b'].previous_element #-> nil
|
||||
|
||||
<em>Text Children</em>
|
||||
|
||||
==== Task: Add a Text Node
|
||||
|
||||
Use method
|
||||
{Element#add_text}[../../../../REXML/Element.html#method-i-add_text]
|
||||
to add a text node to the element:
|
||||
|
||||
d = REXML::Document.new('<a>foo<b/>bar</a>')
|
||||
e = d.root
|
||||
e.add_text(REXML::Text.new('baz'))
|
||||
e.to_a # => ["foo", <b/>, "bar", "baz"]
|
||||
e.add_text(REXML::Text.new('baz'))
|
||||
e.to_a # => ["foo", <b/>, "bar", "baz", "baz"]
|
||||
|
||||
==== Task: Replace the First Text Node
|
||||
|
||||
Use method
|
||||
{Element#text=}[../../../../REXML/Element.html#method-i-text-3D]
|
||||
to replace the first text node in the element:
|
||||
|
||||
d = REXML::Document.new('<root><a/>text<b/>more<c/></root>')
|
||||
e = d.root
|
||||
e.to_a # => [<a/>, "text", <b/>, "more", <c/>]
|
||||
e.text = 'oops'
|
||||
e.to_a # => [<a/>, "oops", <b/>, "more", <c/>]
|
||||
|
||||
==== Task: Remove the First Text Node
|
||||
|
||||
Use method
|
||||
{Element#text=}[../../../../REXML/Element.html#method-i-text-3D]
|
||||
to remove the first text node in the element:
|
||||
|
||||
d = REXML::Document.new('<root><a/>text<b/>more<c/></root>')
|
||||
e = d.root
|
||||
e.to_a # => [<a/>, "text", <b/>, "more", <c/>]
|
||||
e.text = nil
|
||||
e.to_a # => [<a/>, <b/>, "more", <c/>]
|
||||
|
||||
==== Task: Retrieve the First Text Node
|
||||
|
||||
Use method
|
||||
{Element#get_text}[../../../../REXML/Element.html#method-i-get_text]
|
||||
to retrieve the first text node in the element:
|
||||
|
||||
d = REXML::Document.new('<root><a/>text<b/>more<c/></root>')
|
||||
e = d.root
|
||||
e.to_a # => [<a/>, "text", <b/>, "more", <c/>]
|
||||
e.get_text # => "text"
|
||||
|
||||
==== Task: Retrieve a Specific Text Node
|
||||
|
||||
Use method
|
||||
{Element#get_text}[../../../../REXML/Element.html#method-i-get_text]
|
||||
to retrieve the first text node in a specified element:
|
||||
|
||||
d = REXML::Document.new "<root>some text <b>this is bold!</b> more text</root>"
|
||||
e = d.root
|
||||
e.get_text('//root') # => "some text "
|
||||
e.get_text('//b') # => "this is bold!"
|
||||
|
||||
==== Task: Determine Whether the Element has Text Nodes
|
||||
|
||||
Use method
|
||||
{Element#has_text?}[../../../../REXML/Element.html#method-i-has_text-3F]
|
||||
to determine whether the element has text:
|
||||
|
||||
e = REXML::Element.new('foo')
|
||||
e.has_text? # => false
|
||||
e.add_text('bar')
|
||||
e.has_text? # => true
|
||||
|
||||
<em>Other Children</em>
|
||||
|
||||
==== Task: Get the Child at a Given Index
|
||||
|
||||
Use method
|
||||
{Element#[]}[../../../../REXML/Element.html#method-i-5B-5D]
|
||||
to retrieve the child at a given index:
|
||||
|
||||
d = REXML::Document.new '><root><a/>text<b/>more<c/></root>'
|
||||
e = d.root
|
||||
e[0] # => <a/>
|
||||
e[1] # => "text"
|
||||
e[2] # => <b/>
|
||||
|
||||
==== Task: Get All CDATA Children
|
||||
|
||||
Use method
|
||||
{Element#cdatas}[../../../../REXML/Element.html#method-i-cdatas]
|
||||
to retrieve all CDATA children:
|
||||
|
||||
xml_string = <<-EOT
|
||||
<root>
|
||||
<![CDATA[foo]]>
|
||||
<![CDATA[bar]]>
|
||||
</root>
|
||||
EOT
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.cdatas # => ["foo", "bar"]
|
||||
|
||||
==== Task: Get All Comment Children
|
||||
|
||||
Use method
|
||||
{Element#comments}[../../../../REXML/Element.html#method-i-comments]
|
||||
to retrieve all comment children:
|
||||
|
||||
xml_string = <<-EOT
|
||||
<root>
|
||||
<!--foo-->
|
||||
<!--bar-->
|
||||
</root>
|
||||
EOT
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.comments.map {|comment| comment.to_s } # => ["foo", "bar"]
|
||||
|
||||
==== Task: Get All Processing Instruction Children
|
||||
|
||||
Use method
|
||||
{Element#instructions}[../../../../REXML/Element.html#method-i-instructions]
|
||||
to retrieve all processing instruction children:
|
||||
|
||||
xml_string = <<-EOT
|
||||
<root>
|
||||
<?target0 foo?>
|
||||
<?target1 bar?>
|
||||
</root>
|
||||
EOT
|
||||
d = REXML::Document.new(xml_string)
|
||||
instructions = d.root.instructions.map {|instruction| instruction.to_s }
|
||||
instructions # => ["<?target0 foo?>", "<?target1 bar?>"]
|
||||
|
||||
==== Task: Get All Text Children
|
||||
|
||||
Use method
|
||||
{Element#texts}[../../../../REXML/Element.html#method-i-texts]
|
||||
to retrieve all text children:
|
||||
|
||||
xml_string = '<root><a/>text<b/>more<c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.texts # => ["text", "more"]
|
||||
|
||||
=== Namespaces
|
||||
|
||||
==== Task: Add a Namespace
|
||||
|
||||
Use method
|
||||
{Element#add_namespace}[../../../../REXML/Element.html#method-i-add_namespace]
|
||||
to add a namespace to the element:
|
||||
|
||||
e = REXML::Element.new('foo')
|
||||
e.add_namespace('bar')
|
||||
e.namespaces # => {"xmlns"=>"bar"}
|
||||
|
||||
==== Task: Delete the Default Namespace
|
||||
|
||||
Use method
|
||||
{Element#delete_namespace}[../../../../REXML/Element.html#method-i-delete_namespace]
|
||||
to remove the default namespace from the element:
|
||||
|
||||
d = REXML::Document.new "<a xmlns:foo='bar' xmlns='twiddle'/>"
|
||||
d.to_s # => "<a xmlns:foo='bar' xmlns='twiddle'/>"
|
||||
d.root.delete_namespace # => <a xmlns:foo='bar'/>
|
||||
d.to_s # => "<a xmlns:foo='bar'/>"
|
||||
|
||||
==== Task: Delete a Specific Namespace
|
||||
|
||||
Use method
|
||||
{Element#delete_namespace}[../../../../REXML/Element.html#method-i-delete_namespace]
|
||||
to remove a specific namespace from the element:
|
||||
|
||||
d = REXML::Document.new "<a xmlns:foo='bar' xmlns='twiddle'/>"
|
||||
d.to_s # => "<a xmlns:foo='bar' xmlns='twiddle'/>"
|
||||
d.root.delete_namespace # => <a xmlns:foo='bar'/>
|
||||
d.to_s # => "<a xmlns:foo='bar'/>"
|
||||
d.root.delete_namespace('foo')
|
||||
d.to_s # => "<a/>"
|
||||
|
||||
==== Task: Get a Namespace URI
|
||||
|
||||
Use method
|
||||
{Element#namespace}[../../../../REXML/Element.html#method-i-namespace]
|
||||
to retrieve a specific namespace URI for the element:
|
||||
|
||||
xml_string = <<-EOT
|
||||
<root>
|
||||
<a xmlns='1' xmlns:y='2'>
|
||||
<b/>
|
||||
<c xmlns:z='3'/>
|
||||
</a>
|
||||
</root>
|
||||
EOT
|
||||
d = REXML::Document.new(xml_string)
|
||||
b = d.elements['//b']
|
||||
b.namespace # => "1"
|
||||
b.namespace('y') # => "2"
|
||||
|
||||
==== Task: Retrieve Namespaces
|
||||
|
||||
Use method
|
||||
{Element#namespaces}[../../../../REXML/Element.html#method-i-namespaces]
|
||||
to retrieve all namespaces for the element:
|
||||
|
||||
xml_string = '<a xmlns="foo" xmlns:x="bar" xmlns:y="twee" z="glorp"/>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.attributes.namespaces # => {"xmlns"=>"foo", "x"=>"bar", "y"=>"twee"}
|
||||
|
||||
==== Task: Retrieve Namespace Prefixes
|
||||
|
||||
Use method
|
||||
{Element#prefixes}[../../../../REXML/Element.html#method-i-prefixes]
|
||||
to retrieve all prefixes (namespace names) for the element:
|
||||
|
||||
xml_string = <<-EOT
|
||||
<root>
|
||||
<a xmlns:x='1' xmlns:y='2'>
|
||||
<b/>
|
||||
<c xmlns:z='3'/>
|
||||
</a>
|
||||
</root>
|
||||
EOT
|
||||
d = REXML::Document.new(xml_string, {compress_whitespace: :all})
|
||||
d.elements['//a'].prefixes # => ["x", "y"]
|
||||
d.elements['//b'].prefixes # => ["x", "y"]
|
||||
d.elements['//c'].prefixes # => ["x", "y", "z"]
|
||||
|
||||
=== Iteration
|
||||
|
||||
==== Task: Iterate Over Elements
|
||||
|
||||
Use method
|
||||
{Element#each_element}[../../../../REXML/Element.html#method-i-each_element]
|
||||
to iterate over element children:
|
||||
|
||||
d = REXML::Document.new '<a><b>b</b><c>b</c><d>d</d><e/></a>'
|
||||
d.root.each_element {|e| p e }
|
||||
|
||||
Output:
|
||||
|
||||
<b> ... </>
|
||||
<c> ... </>
|
||||
<d> ... </>
|
||||
<e/>
|
||||
|
||||
==== Task: Iterate Over Elements Having a Specified Attribute
|
||||
|
||||
Use method
|
||||
{Element#each_element_with_attribute}[../../../../REXML/Element.html#method-i-each_element_with_attribute]
|
||||
to iterate over element children that have a specified attribute:
|
||||
|
||||
d = REXML::Document.new '<a><b id="1"/><c id="2"/><d id="1"/><e/></a>'
|
||||
a = d.root
|
||||
a.each_element_with_attribute('id') {|e| p e }
|
||||
|
||||
Output:
|
||||
|
||||
<b id='1'/>
|
||||
<c id='2'/>
|
||||
<d id='1'/>
|
||||
|
||||
==== Task: Iterate Over Elements Having a Specified Attribute and Value
|
||||
|
||||
Use method
|
||||
{Element#each_element_with_attribute}[../../../../REXML/Element.html#method-i-each_element_with_attribute]
|
||||
to iterate over element children that have a specified attribute and value:
|
||||
|
||||
d = REXML::Document.new '<a><b id="1"/><c id="2"/><d id="1"/><e/></a>'
|
||||
a = d.root
|
||||
a.each_element_with_attribute('id', '1') {|e| p e }
|
||||
|
||||
Output:
|
||||
|
||||
<b id='1'/>
|
||||
<d id='1'/>
|
||||
|
||||
==== Task: Iterate Over Elements Having Specified Text
|
||||
|
||||
Use method
|
||||
{Element#each_element_with_text}[../../../../REXML/Element.html#method-i-each_element_with_text]
|
||||
to iterate over element children that have specified text:
|
||||
|
||||
|
||||
=== Context
|
||||
|
||||
#whitespace
|
||||
#ignore_whitespace_nodes
|
||||
#raw
|
||||
|
||||
=== Other Getters
|
||||
|
||||
#document
|
||||
#root
|
||||
#root_node
|
||||
#node_type
|
||||
#xpath
|
||||
#inspect
|
||||
@@ -0,0 +1,97 @@
|
||||
== Module Node
|
||||
|
||||
:include: ../tocs/node_toc.rdoc
|
||||
|
||||
=== Siblings
|
||||
|
||||
==== Task: Find Previous Sibling
|
||||
|
||||
Use method
|
||||
{Node.previous_sibling_node}[../../../../REXML/Node.html#method-i-previous_sibling]
|
||||
to retrieve the previous sibling:
|
||||
|
||||
d = REXML::Document.new('<root><a/><b/><c/></root>')
|
||||
b = d.root[1] # => <b/>
|
||||
b.previous_sibling_node # => <a/>
|
||||
|
||||
==== Task: Find Next Sibling
|
||||
|
||||
Use method
|
||||
{Node.next_sibling_node}[../../../../REXML/Node.html#method-i-next_sibling]
|
||||
to retrieve the next sibling:
|
||||
|
||||
d = REXML::Document.new('<root><a/><b/><c/></root>')
|
||||
b = d.root[1] # => <b/>
|
||||
b.next_sibling_node # => <c/>
|
||||
|
||||
=== Position
|
||||
|
||||
==== Task: Find Own Index Among Siblings
|
||||
|
||||
Use method
|
||||
{Node.index_in_parent}[../../../../REXML/Node.html#method-i-index_in_parent]
|
||||
to retrieve the 1-based index of this node among its siblings:
|
||||
|
||||
d = REXML::Document.new('<root><a/><b/><c/></root>')
|
||||
b = d.root[1] # => <b/>
|
||||
b.index_in_parent # => 2
|
||||
|
||||
=== Recursive Traversal
|
||||
|
||||
==== Task: Traverse Each Recursively
|
||||
|
||||
Use method
|
||||
{Node.each_recursive}[../../../../REXML/Node.html#method-i-each_recursive]
|
||||
to traverse a tree of nodes recursively:
|
||||
|
||||
xml_string = '<root><a><b><c></c></b><b><c></c></b></a></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.each_recursive {|node| p node }
|
||||
|
||||
Output:
|
||||
|
||||
<a> ... </>
|
||||
<b> ... </>
|
||||
<c/>
|
||||
<b> ... </>
|
||||
<c/>
|
||||
|
||||
=== Recursive Search
|
||||
|
||||
==== Task: Traverse Each Recursively
|
||||
|
||||
Use method
|
||||
{Node.find_first_recursive}[../../../../REXML/Node.html#method-i-find_first_recursive]
|
||||
to search a tree of nodes recursively:
|
||||
|
||||
xml_string = '<root><a><b><c></c></b><b><c></c></b></a></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.find_first_recursive {|node| node.name == 'c' } # => <c/>
|
||||
|
||||
=== Representation
|
||||
|
||||
==== Task: Represent a String
|
||||
|
||||
Use method {Node.to_s}[../../../../REXML/Node.html#method-i-to_s]
|
||||
to represent the node as a string:
|
||||
|
||||
xml_string = '<root><a><b><c></c></b><b><c></c></b></a></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.to_s # => "<root><a><b><c/></b><b><c/></b></a></root>"
|
||||
|
||||
=== Parent?
|
||||
|
||||
==== Task: Determine Whether the Node is a Parent
|
||||
|
||||
Use method {Node.parent?}[../../../../REXML/Node.html#method-i-parent-3F]
|
||||
to determine whether the node is a parent;
|
||||
class Text derives from Node:
|
||||
|
||||
d = REXML::Document.new('<root><a/>text<b/>more<c/></root>')
|
||||
t = d.root[1] # => "text"
|
||||
t.parent? # => false
|
||||
|
||||
Class Parent also derives from Node, but overrides this method:
|
||||
|
||||
p = REXML::Parent.new
|
||||
p.parent? # => true
|
||||
@@ -0,0 +1,267 @@
|
||||
== Class Parent
|
||||
|
||||
Class Parent has methods from its superclasses and included modules;
|
||||
see:
|
||||
|
||||
- {Tasks for Child}[child_rdoc.html].
|
||||
- {Tasks for Node}[node_rdoc.html].
|
||||
- {Module Enumerable}[https://docs.ruby-lang.org/en/master/Enumerable.html].
|
||||
|
||||
:include: ../tocs/parent_toc.rdoc
|
||||
|
||||
=== Queries
|
||||
|
||||
==== Task: Get the Count of Children
|
||||
|
||||
Use method {Parent#size}[../../../../REXML/Parent.html#method-i-size]
|
||||
(or its alias +length+) to get the count of the parent's children:
|
||||
|
||||
p = REXML::Parent.new
|
||||
p.size # => 0
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.size # => 3
|
||||
|
||||
==== Task: Get the Child at a Given Index
|
||||
|
||||
Use method {Parent#[]}[../../../../REXML/Parent.html#method-i-5B-5D]
|
||||
to get the child at a given index:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root[1] # => <b/>
|
||||
d.root[-1] # => <c/>
|
||||
d.root[50] # => nil
|
||||
|
||||
==== Task: Get the Index of a Given Child
|
||||
|
||||
Use method {Parent#index}[../../../../REXML/Parent.html#method-i-index]
|
||||
to get the index (0-based offset) of a child:
|
||||
|
||||
d = REXML::Document.new('<root></root>')
|
||||
root = d.root
|
||||
e0 = REXML::Element.new('foo')
|
||||
e1 = REXML::Element.new('bar')
|
||||
root.add(e0) # => <foo/>
|
||||
root.add(e1) # => <bar/>
|
||||
root.add(e0) # => <foo/>
|
||||
root.add(e1) # => <bar/>
|
||||
root.index(e0) # => 0
|
||||
root.index(e1) # => 1
|
||||
|
||||
==== Task: Get the Children
|
||||
|
||||
Use method {Parent#children}[../../../../REXML/Parent.html#method-i-children]
|
||||
(or its alias +to_a+) to get the parent's children:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.children # => [<a/>, <b/>, <c/>]
|
||||
|
||||
==== Task: Determine Whether the Node is a Parent
|
||||
|
||||
Use method {Parent#parent?}[../../../../REXML/Parent.html#method-i-parent-3F]
|
||||
to determine whether the node is a parent;
|
||||
class Text derives from Node:
|
||||
|
||||
d = REXML::Document.new('<root><a/>text<b/>more<c/></root>')
|
||||
t = d.root[1] # => "text"
|
||||
t.parent? # => false
|
||||
|
||||
Class Parent also derives from Node, but overrides this method:
|
||||
|
||||
p = REXML::Parent.new
|
||||
p.parent? # => true
|
||||
|
||||
=== Additions
|
||||
|
||||
==== Task: Add a Child at the Beginning
|
||||
|
||||
Use method {Parent#unshift}[../../../../REXML/Parent.html#method-i-unshift]
|
||||
to add a child as at the beginning of the children:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.children # => [<a/>, <b/>, <c/>]
|
||||
d.root.unshift REXML::Element.new('d')
|
||||
d.root.children # => [<d/>, <a/>, <b/>, <c/>]
|
||||
|
||||
==== Task: Add a Child at the End
|
||||
|
||||
Use method {Parent#<<}[../../../../REXML/Parent.html#method-i-3C-3C]
|
||||
(or an alias +push+ or +add+) to add a child as at the end of the children:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.children # => [<a/>, <b/>, <c/>]
|
||||
d.root << REXML::Element.new('d')
|
||||
d.root.children # => [<a/>, <b/>, <c/>, <d/>]
|
||||
|
||||
==== Task: Replace a Child with Another Child
|
||||
|
||||
Use method {Parent#replace}[../../../../REXML/Parent.html#method-i-replace]
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.children # => [<a/>, <b/>, <c/>]
|
||||
b = d.root[1] # => <b/>
|
||||
d.replace_child(b, REXML::Element.new('d'))
|
||||
d.root.children # => [<a/>, <c/>]
|
||||
|
||||
==== Task: Replace Multiple Children with Another Child
|
||||
|
||||
Use method {Parent#[]=}[../../../../REXML/Parent.html#method-i-parent-5B-5D-3D]
|
||||
to replace multiple consecutive children with another child:
|
||||
|
||||
xml_string = '<root><a/><b/><c/><d/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.children # => [<a/>, <b/>, <c/>, <d/>]
|
||||
d.root[1, 2] = REXML::Element.new('x')
|
||||
d.root.children # => [<a/>, <x/>, <d/>]
|
||||
d.root[1, 5] = REXML::Element.new('x')
|
||||
d.root.children # => [<a/>, <x/>] # BUG?
|
||||
|
||||
==== Task: Insert Child Before a Given Child
|
||||
|
||||
Use method {Parent#insert_before}[../../../../REXML/Parent.html#method-i-insert_before]
|
||||
to insert a child immediately before a given child:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.children # => [<a/>, <b/>, <c/>]
|
||||
b = d.root[1] # => <b/>
|
||||
x = REXML::Element.new('x')
|
||||
d.root.insert_before(b, x)
|
||||
d.root.children # => [<a/>, <x/>, <b/>, <c/>]
|
||||
|
||||
==== Task: Insert Child After a Given Child
|
||||
|
||||
Use method {Parent#insert_after}[../../../../REXML/Parent.html#method-i-insert_after]
|
||||
to insert a child immediately after a given child:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.children # => [<a/>, <b/>, <c/>]
|
||||
b = d.root[1] # => <b/>
|
||||
x = REXML::Element.new('x')
|
||||
d.root.insert_after(b, x)
|
||||
d.root.children # => [<a/>, <b/>, <x/>, <c/>]
|
||||
|
||||
=== Deletions
|
||||
|
||||
==== Task: Remove a Given Child
|
||||
|
||||
Use method {Parent#delete}[../../../../REXML/Parent.html#method-i-delete]
|
||||
to remove all occurrences of a given child:
|
||||
|
||||
d = REXML::Document.new('<root></root>')
|
||||
a = REXML::Element.new('a')
|
||||
b = REXML::Element.new('b')
|
||||
d.root.add(a)
|
||||
d.root.add(b)
|
||||
d.root.add(a)
|
||||
d.root.add(b)
|
||||
d.root.children # => [<a/>, <b/>, <a/>, <b/>]
|
||||
d.root.delete(b)
|
||||
d.root.children # => [<a/>, <a/>]
|
||||
|
||||
==== Task: Remove the Child at a Specified Offset
|
||||
|
||||
Use method {Parent#delete_at}[../../../../REXML/Parent.html#method-i-delete_at]
|
||||
to remove the child at a specified offset:
|
||||
|
||||
d = REXML::Document.new('<root></root>')
|
||||
a = REXML::Element.new('a')
|
||||
b = REXML::Element.new('b')
|
||||
d.root.add(a)
|
||||
d.root.add(b)
|
||||
d.root.add(a)
|
||||
d.root.add(b)
|
||||
d.root.children # => [<a/>, <b/>, <a/>, <b/>]
|
||||
d.root.delete_at(2)
|
||||
d.root.children # => [<a/>, <b/>, <b/>]
|
||||
|
||||
==== Task: Remove Children That Meet Specified Criteria
|
||||
|
||||
Use method {Parent#delete_if}[../../../../REXML/Parent.html#method-i-delete_if]
|
||||
to remove children that meet criteria specified in the given block:
|
||||
|
||||
d = REXML::Document.new('<root></root>')
|
||||
d.root.add(REXML::Element.new('x'))
|
||||
d.root.add(REXML::Element.new('xx'))
|
||||
d.root.add(REXML::Element.new('xxx'))
|
||||
d.root.add(REXML::Element.new('xxxx'))
|
||||
d.root.children # => [<x/>, <xx/>, <xxx/>, <xxxx/>]
|
||||
d.root.delete_if {|child| child.name.size.odd? }
|
||||
d.root.children # => [<xx/>, <xxxx/>]
|
||||
|
||||
=== Iterations
|
||||
|
||||
==== Task: Iterate Over Children
|
||||
|
||||
Use method {Parent#each_child}[../../../../REXML/Parent.html#method-i-each_child]
|
||||
(or its alias +each+) to iterate over all children:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.children # => [<a/>, <b/>, <c/>]
|
||||
d.root.each_child {|child| p child }
|
||||
|
||||
Output:
|
||||
|
||||
<a/>
|
||||
<b/>
|
||||
<c/>
|
||||
|
||||
==== Task: Iterate Over Child Indexes
|
||||
|
||||
Use method {Parent#each_index}[../../../../REXML/Parent.html#method-i-each_index]
|
||||
to iterate over all child indexes:
|
||||
|
||||
xml_string = '<root><a/><b/><c/></root>'
|
||||
d = REXML::Document.new(xml_string)
|
||||
d.root.children # => [<a/>, <b/>, <c/>]
|
||||
d.root.each_index {|child| p child }
|
||||
|
||||
Output:
|
||||
|
||||
0
|
||||
1
|
||||
2
|
||||
|
||||
=== Clones
|
||||
|
||||
==== Task: Clone Deeply
|
||||
|
||||
Use method {Parent#deep_clone}[../../../../REXML/Parent.html#method-i-deep_clone]
|
||||
to clone deeply; that is, to clone every nested node that is a Parent object:
|
||||
|
||||
xml_string = <<-EOT
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<bookstore>
|
||||
<book category="cooking">
|
||||
<title lang="en">Everyday Italian</title>
|
||||
<author>Giada De Laurentiis</author>
|
||||
<year>2005</year>
|
||||
<price>30.00</price>
|
||||
</book>
|
||||
<book category="children">
|
||||
<title lang="en">Harry Potter</title>
|
||||
<author>J K. Rowling</author>
|
||||
<year>2005</year>
|
||||
<price>29.99</price>
|
||||
</book>
|
||||
<book category="web">
|
||||
<title lang="en">Learning XML</title>
|
||||
<author>Erik T. Ray</author>
|
||||
<year>2003</year>
|
||||
<price>39.95</price>
|
||||
</book>
|
||||
</bookstore>
|
||||
EOT
|
||||
d = REXML::Document.new(xml_string)
|
||||
root = d.root
|
||||
shallow = root.clone
|
||||
deep = root.deep_clone
|
||||
shallow.to_s.size # => 12
|
||||
deep.to_s.size # => 590
|
||||
@@ -0,0 +1,12 @@
|
||||
Tasks on this page:
|
||||
|
||||
- {Relationships}[#label-Relationships]
|
||||
- {Task: Set the Parent}[#label-Task-3A+Set+the+Parent]
|
||||
- {Task: Insert Previous Sibling}[#label-Task-3A+Insert+Previous+Sibling]
|
||||
- {Task: Insert Next Sibling}[#label-Task-3A+Insert+Next+Sibling]
|
||||
- {Removal or Replacement}[#label-Removal+or+Replacement]
|
||||
- {Task: Remove Child from Parent}[#label-Task-3A+Remove+Child+from+Parent]
|
||||
- {Task: Replace Child}[#label-Task-3A+Replace+Child]
|
||||
- {Document}[#label-Document]
|
||||
- {Task: Get the Document}[#label-Task-3A+Get+the+Document]
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
Tasks on this page:
|
||||
|
||||
- {New Document}[#label-New+Document]
|
||||
- {Task: Create an Empty Document}[#label-Task-3A+Create+an+Empty+Document]
|
||||
- {Task: Parse a String into a New Document}[#label-Task-3A+Parse+a+String+into+a+New+Document]
|
||||
- {Task: Parse an IO Stream into a New Document}[#label-Task-3A+Parse+an+IO+Stream+into+a+New+Document]
|
||||
- {Task: Create a Document from an Existing Document}[#label-Task-3A+Create+a+Document+from+an+Existing+Document]
|
||||
- {Task: Clone a Document}[#label-Task-3A+Clone+a+Document]
|
||||
- {Document Type}[#label-Document+Type]
|
||||
- {Task: Get the Document Type}[#label-Task-3A+Get+the+Document+Type]
|
||||
- {Task: Set the Document Type}[#label-Task-3A+Set+the+Document+Type]
|
||||
- {XML Declaration}[#label-XML+Declaration]
|
||||
- {Task: Get the XML Declaration}[#label-Task-3A+Get+the+XML+Declaration]
|
||||
- {Task: Set the XML Declaration}[#label-Task-3A+Set+the+XML+Declaration]
|
||||
- {Children}[#label-Children]
|
||||
- {Task: Add an Element Child}[#label-Task-3A+Add+an+Element+Child]
|
||||
- {Task: Add a Non-Element Child}[#label-Task-3A+Add+a+Non-Element+Child]
|
||||
- {Writing}[#label-Writing]
|
||||
- {Task: Write to $stdout}[#label-Task-3A+Write+to+-24stdout]
|
||||
- {Task: Write to IO Stream}[#label-Task-3A+Write+to+IO+Stream]
|
||||
- {Task: Write with No Indentation}[#label-Task-3A+Write+with+No+Indentation]
|
||||
- {Task: Write with Specified Indentation}[#label-Task-3A+Write+with+Specified+Indentation]
|
||||
- {Querying}[#label-Querying]
|
||||
- {Task: Get the Document}[#label-Task-3A+Get+the+Document]
|
||||
- {Task: Get the Encoding}[#label-Task-3A+Get+the+Encoding]
|
||||
- {Task: Get the Node Type}[#label-Task-3A+Get+the+Node+Type]
|
||||
- {Task: Get the Root Element}[#label-Task-3A+Get+the+Root+Element]
|
||||
- {Task: Determine Whether Stand-Alone}[#label-Task-3A+Determine+Whether+Stand-Alone]
|
||||
- {Task: Get the Version}[#label-Task-3A+Get+the+Version]
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
Tasks on this page:
|
||||
|
||||
- {New Element}[#label-New+Element]
|
||||
- {Task: Create a Default Element}[#label-Task-3A+Create+a+Default+Element]
|
||||
- {Task: Create a Named Element}[#label-Task-3A+Create+a+Named+Element]
|
||||
- {Task: Create an Element with Name and Parent}[#label-Task-3A+Create+an+Element+with+Name+and+Parent]
|
||||
- {Task: Create an Element with Name, Parent, and Context}[#label-Task-3A+Create+an+Element+with+Name-2C+Parent-2C+and+Context]
|
||||
- {Task: Create a Shallow Clone}[#label-Task-3A+Create+a+Shallow+Clone]
|
||||
- {Attributes}[#label-Attributes]
|
||||
- {Task: Create and Add an Attribute}[#label-Task-3A+Create+and+Add+an+Attribute]
|
||||
- {Task: Add an Existing Attribute}[#label-Task-3A+Add+an+Existing+Attribute]
|
||||
- {Task: Add Multiple Attributes from a Hash}[#label-Task-3A+Add+Multiple+Attributes+from+a+Hash]
|
||||
- {Task: Add Multiple Attributes from an Array}[#label-Task-3A+Add+Multiple+Attributes+from+an+Array]
|
||||
- {Task: Retrieve the Value for an Attribute Name}[#label-Task-3A+Retrieve+the+Value+for+an+Attribute+Name]
|
||||
- {Task: Retrieve the Attribute Value for a Name and Namespace}[#label-Task-3A+Retrieve+the+Attribute+Value+for+a+Name+and+Namespace]
|
||||
- {Task: Delete an Attribute}[#label-Task-3A+Delete+an+Attribute]
|
||||
- {Task: Determine Whether the Element Has Attributes}[#label-Task-3A+Determine+Whether+the+Element+Has+Attributes]
|
||||
- {Children}[#label-Children]
|
||||
- {Task: Create and Add an Element}[#label-Task-3A+Create+and+Add+an+Element]
|
||||
- {Task: Add an Existing Element}[#label-Task-3A+Add+an+Existing+Element]
|
||||
- {Task: Create and Add an Element with Attributes}[#label-Task-3A+Create+and+Add+an+Element+with+Attributes]
|
||||
- {Task: Add an Existing Element with Added Attributes}[#label-Task-3A+Add+an+Existing+Element+with+Added+Attributes]
|
||||
- {Task: Delete a Specified Element}[#label-Task-3A+Delete+a+Specified+Element]
|
||||
- {Task: Delete an Element by Index}[#label-Task-3A+Delete+an+Element+by+Index]
|
||||
- {Task: Delete an Element by XPath}[#label-Task-3A+Delete+an+Element+by+XPath]
|
||||
- {Task: Determine Whether Element Children}[#label-Task-3A+Determine+Whether+Element+Children]
|
||||
- {Task: Get Element Descendants by XPath}[#label-Task-3A+Get+Element+Descendants+by+XPath]
|
||||
- {Task: Get Next Element Sibling}[#label-Task-3A+Get+Next+Element+Sibling]
|
||||
- {Task: Get Previous Element Sibling}[#label-Task-3A+Get+Previous+Element+Sibling]
|
||||
- {Task: Add a Text Node}[#label-Task-3A+Add+a+Text+Node]
|
||||
- {Task: Replace the First Text Node}[#label-Task-3A+Replace+the+First+Text+Node]
|
||||
- {Task: Remove the First Text Node}[#label-Task-3A+Remove+the+First+Text+Node]
|
||||
- {Task: Retrieve the First Text Node}[#label-Task-3A+Retrieve+the+First+Text+Node]
|
||||
- {Task: Retrieve a Specific Text Node}[#label-Task-3A+Retrieve+a+Specific+Text+Node]
|
||||
- {Task: Determine Whether the Element has Text Nodes}[#label-Task-3A+Determine+Whether+the+Element+has+Text+Nodes]
|
||||
- {Task: Get the Child at a Given Index}[#label-Task-3A+Get+the+Child+at+a+Given+Index]
|
||||
- {Task: Get All CDATA Children}[#label-Task-3A+Get+All+CDATA+Children]
|
||||
- {Task: Get All Comment Children}[#label-Task-3A+Get+All+Comment+Children]
|
||||
- {Task: Get All Processing Instruction Children}[#label-Task-3A+Get+All+Processing+Instruction+Children]
|
||||
- {Task: Get All Text Children}[#label-Task-3A+Get+All+Text+Children]
|
||||
- {Namespaces}[#label-Namespaces]
|
||||
- {Task: Add a Namespace}[#label-Task-3A+Add+a+Namespace]
|
||||
- {Task: Delete the Default Namespace}[#label-Task-3A+Delete+the+Default+Namespace]
|
||||
- {Task: Delete a Specific Namespace}[#label-Task-3A+Delete+a+Specific+Namespace]
|
||||
- {Task: Get a Namespace URI}[#label-Task-3A+Get+a+Namespace+URI]
|
||||
- {Task: Retrieve Namespaces}[#label-Task-3A+Retrieve+Namespaces]
|
||||
- {Task: Retrieve Namespace Prefixes}[#label-Task-3A+Retrieve+Namespace+Prefixes]
|
||||
- {Iteration}[#label-Iteration]
|
||||
- {Task: Iterate Over Elements}[#label-Task-3A+Iterate+Over+Elements]
|
||||
- {Task: Iterate Over Elements Having a Specified Attribute}[#label-Task-3A+Iterate+Over+Elements+Having+a+Specified+Attribute]
|
||||
- {Task: Iterate Over Elements Having a Specified Attribute and Value}[#label-Task-3A+Iterate+Over+Elements+Having+a+Specified+Attribute+and+Value]
|
||||
- {Task: Iterate Over Elements Having Specified Text}[#label-Task-3A+Iterate+Over+Elements+Having+Specified+Text]
|
||||
- {Context}[#label-Context]
|
||||
- {Other Getters}[#label-Other+Getters]
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
== Tasks
|
||||
|
||||
=== {Child}[../../tasks/rdoc/child_rdoc.html]
|
||||
- {Relationships}[../../tasks/rdoc/child_rdoc.html#label-Relationships]
|
||||
- {Task: Set the Parent}[../../tasks/rdoc/child_rdoc.html#label-Task-3A+Set+the+Parent]
|
||||
- {Task: Insert Previous Sibling}[../../tasks/rdoc/child_rdoc.html#label-Task-3A+Insert+Previous+Sibling]
|
||||
- {Task: Insert Next Sibling}[../../tasks/rdoc/child_rdoc.html#label-Task-3A+Insert+Next+Sibling]
|
||||
- {Removal or Replacement}[../../tasks/rdoc/child_rdoc.html#label-Removal+or+Replacement]
|
||||
- {Task: Remove Child from Parent}[../../tasks/rdoc/child_rdoc.html#label-Task-3A+Remove+Child+from+Parent]
|
||||
- {Task: Replace Child}[../../tasks/rdoc/child_rdoc.html#label-Task-3A+Replace+Child]
|
||||
- {Document}[../../tasks/rdoc/child_rdoc.html#label-Document]
|
||||
- {Task: Get the Document}[../../tasks/rdoc/child_rdoc.html#label-Task-3A+Get+the+Document]
|
||||
|
||||
=== {Document}[../../tasks/rdoc/document_rdoc.html]
|
||||
- {New Document}[../../tasks/rdoc/document_rdoc.html#label-New+Document]
|
||||
- {Task: Create an Empty Document}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Create+an+Empty+Document]
|
||||
- {Task: Parse a String into a New Document}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Parse+a+String+into+a+New+Document]
|
||||
- {Task: Parse an IO Stream into a New Document}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Parse+an+IO+Stream+into+a+New+Document]
|
||||
- {Task: Create a Document from an Existing Document}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Create+a+Document+from+an+Existing+Document]
|
||||
- {Task: Clone a Document}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Clone+a+Document]
|
||||
- {Document Type}[../../tasks/rdoc/document_rdoc.html#label-Document+Type]
|
||||
- {Task: Get the Document Type}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Get+the+Document+Type]
|
||||
- {Task: Set the Document Type}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Set+the+Document+Type]
|
||||
- {XML Declaration}[../../tasks/rdoc/document_rdoc.html#label-XML+Declaration]
|
||||
- {Task: Get the XML Declaration}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Get+the+XML+Declaration]
|
||||
- {Task: Set the XML Declaration}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Set+the+XML+Declaration]
|
||||
- {Children}[../../tasks/rdoc/document_rdoc.html#label-Children]
|
||||
- {Task: Add an Element Child}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Add+an+Element+Child]
|
||||
- {Task: Add a Non-Element Child}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Add+a+Non-Element+Child]
|
||||
- {Writing}[../../tasks/rdoc/document_rdoc.html#label-Writing]
|
||||
- {Task: Write to $stdout}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Write+to+-24stdout]
|
||||
- {Task: Write to IO Stream}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Write+to+IO+Stream]
|
||||
- {Task: Write with No Indentation}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Write+with+No+Indentation]
|
||||
- {Task: Write with Specified Indentation}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Write+with+Specified+Indentation]
|
||||
- {Querying}[../../tasks/rdoc/document_rdoc.html#label-Querying]
|
||||
- {Task: Get the Document}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Get+the+Document]
|
||||
- {Task: Get the Encoding}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Get+the+Encoding]
|
||||
- {Task: Get the Node Type}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Get+the+Node+Type]
|
||||
- {Task: Get the Root Element}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Get+the+Root+Element]
|
||||
- {Task: Determine Whether Stand-Alone}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Determine+Whether+Stand-Alone]
|
||||
- {Task: Get the Version}[../../tasks/rdoc/document_rdoc.html#label-Task-3A+Get+the+Version]
|
||||
|
||||
=== {Element}[../../tasks/rdoc/element_rdoc.html]
|
||||
- {New Element}[../../tasks/rdoc/element_rdoc.html#label-New+Element]
|
||||
- {Task: Create a Default Element}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Create+a+Default+Element]
|
||||
- {Task: Create a Named Element}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Create+a+Named+Element]
|
||||
- {Task: Create an Element with Name and Parent}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Create+an+Element+with+Name+and+Parent]
|
||||
- {Task: Create an Element with Name, Parent, and Context}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Create+an+Element+with+Name-2C+Parent-2C+and+Context]
|
||||
- {Task: Create a Shallow Clone}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Create+a+Shallow+Clone]
|
||||
- {Attributes}[../../tasks/rdoc/element_rdoc.html#label-Attributes]
|
||||
- {Task: Create and Add an Attribute}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Create+and+Add+an+Attribute]
|
||||
- {Task: Add an Existing Attribute}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Add+an+Existing+Attribute]
|
||||
- {Task: Add Multiple Attributes from a Hash}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Add+Multiple+Attributes+from+a+Hash]
|
||||
- {Task: Add Multiple Attributes from an Array}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Add+Multiple+Attributes+from+an+Array]
|
||||
- {Task: Retrieve the Value for an Attribute Name}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Retrieve+the+Value+for+an+Attribute+Name]
|
||||
- {Task: Retrieve the Attribute Value for a Name and Namespace}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Retrieve+the+Attribute+Value+for+a+Name+and+Namespace]
|
||||
- {Task: Delete an Attribute}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Delete+an+Attribute]
|
||||
- {Task: Determine Whether the Element Has Attributes}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Determine+Whether+the+Element+Has+Attributes]
|
||||
- {Children}[../../tasks/rdoc/element_rdoc.html#label-Children]
|
||||
- {Task: Create and Add an Element}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Create+and+Add+an+Element]
|
||||
- {Task: Add an Existing Element}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Add+an+Existing+Element]
|
||||
- {Task: Create and Add an Element with Attributes}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Create+and+Add+an+Element+with+Attributes]
|
||||
- {Task: Add an Existing Element with Added Attributes}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Add+an+Existing+Element+with+Added+Attributes]
|
||||
- {Task: Delete a Specified Element}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Delete+a+Specified+Element]
|
||||
- {Task: Delete an Element by Index}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Delete+an+Element+by+Index]
|
||||
- {Task: Delete an Element by XPath}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Delete+an+Element+by+XPath]
|
||||
- {Task: Determine Whether Element Children}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Determine+Whether+Element+Children]
|
||||
- {Task: Get Element Descendants by XPath}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Get+Element+Descendants+by+XPath]
|
||||
- {Task: Get Next Element Sibling}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Get+Next+Element+Sibling]
|
||||
- {Task: Get Previous Element Sibling}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Get+Previous+Element+Sibling]
|
||||
- {Task: Add a Text Node}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Add+a+Text+Node]
|
||||
- {Task: Replace the First Text Node}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Replace+the+First+Text+Node]
|
||||
- {Task: Remove the First Text Node}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Remove+the+First+Text+Node]
|
||||
- {Task: Retrieve the First Text Node}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Retrieve+the+First+Text+Node]
|
||||
- {Task: Retrieve a Specific Text Node}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Retrieve+a+Specific+Text+Node]
|
||||
- {Task: Determine Whether the Element has Text Nodes}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Determine+Whether+the+Element+has+Text+Nodes]
|
||||
- {Task: Get the Child at a Given Index}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Get+the+Child+at+a+Given+Index]
|
||||
- {Task: Get All CDATA Children}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Get+All+CDATA+Children]
|
||||
- {Task: Get All Comment Children}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Get+All+Comment+Children]
|
||||
- {Task: Get All Processing Instruction Children}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Get+All+Processing+Instruction+Children]
|
||||
- {Task: Get All Text Children}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Get+All+Text+Children]
|
||||
- {Namespaces}[../../tasks/rdoc/element_rdoc.html#label-Namespaces]
|
||||
- {Task: Add a Namespace}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Add+a+Namespace]
|
||||
- {Task: Delete the Default Namespace}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Delete+the+Default+Namespace]
|
||||
- {Task: Delete a Specific Namespace}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Delete+a+Specific+Namespace]
|
||||
- {Task: Get a Namespace URI}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Get+a+Namespace+URI]
|
||||
- {Task: Retrieve Namespaces}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Retrieve+Namespaces]
|
||||
- {Task: Retrieve Namespace Prefixes}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Retrieve+Namespace+Prefixes]
|
||||
- {Iteration}[../../tasks/rdoc/element_rdoc.html#label-Iteration]
|
||||
- {Task: Iterate Over Elements}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Iterate+Over+Elements]
|
||||
- {Task: Iterate Over Elements Having a Specified Attribute}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Iterate+Over+Elements+Having+a+Specified+Attribute]
|
||||
- {Task: Iterate Over Elements Having a Specified Attribute and Value}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Iterate+Over+Elements+Having+a+Specified+Attribute+and+Value]
|
||||
- {Task: Iterate Over Elements Having Specified Text}[../../tasks/rdoc/element_rdoc.html#label-Task-3A+Iterate+Over+Elements+Having+Specified+Text]
|
||||
- {Context}[../../tasks/rdoc/element_rdoc.html#label-Context]
|
||||
- {Other Getters}[../../tasks/rdoc/element_rdoc.html#label-Other+Getters]
|
||||
|
||||
=== {Node}[../../tasks/rdoc/node_rdoc.html]
|
||||
- {Siblings}[../../tasks/rdoc/node_rdoc.html#label-Siblings]
|
||||
- {Task: Find Previous Sibling}[../../tasks/rdoc/node_rdoc.html#label-Task-3A+Find+Previous+Sibling]
|
||||
- {Task: Find Next Sibling}[../../tasks/rdoc/node_rdoc.html#label-Task-3A+Find+Next+Sibling]
|
||||
- {Position}[../../tasks/rdoc/node_rdoc.html#label-Position]
|
||||
- {Task: Find Own Index Among Siblings}[../../tasks/rdoc/node_rdoc.html#label-Task-3A+Find+Own+Index+Among+Siblings]
|
||||
- {Recursive Traversal}[../../tasks/rdoc/node_rdoc.html#label-Recursive+Traversal]
|
||||
- {Task: Traverse Each Recursively}[../../tasks/rdoc/node_rdoc.html#label-Task-3A+Traverse+Each+Recursively]
|
||||
- {Recursive Search}[../../tasks/rdoc/node_rdoc.html#label-Recursive+Search]
|
||||
- {Task: Traverse Each Recursively}[../../tasks/rdoc/node_rdoc.html#label-Task-3A+Traverse+Each+Recursively]
|
||||
- {Representation}[../../tasks/rdoc/node_rdoc.html#label-Representation]
|
||||
- {Task: Represent a String}[../../tasks/rdoc/node_rdoc.html#label-Task-3A+Represent+a+String]
|
||||
- {Parent?}[../../tasks/rdoc/node_rdoc.html#label-Parent-3F]
|
||||
- {Task: Determine Whether the Node is a Parent}[../../tasks/rdoc/node_rdoc.html#label-Task-3A+Determine+Whether+the+Node+is+a+Parent]
|
||||
|
||||
=== {Parent}[../../tasks/rdoc/parent_rdoc.html]
|
||||
- {Queries}[../../tasks/rdoc/parent_rdoc.html#label-Queries]
|
||||
- {Task: Get the Count of Children}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Get+the+Count+of+Children]
|
||||
- {Task: Get the Child at a Given Index}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Get+the+Child+at+a+Given+Index]
|
||||
- {Task: Get the Index of a Given Child}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Get+the+Index+of+a+Given+Child]
|
||||
- {Task: Get the Children}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Get+the+Children]
|
||||
- {Task: Determine Whether the Node is a Parent}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Determine+Whether+the+Node+is+a+Parent]
|
||||
- {Additions}[../../tasks/rdoc/parent_rdoc.html#label-Additions]
|
||||
- {Task: Add a Child at the Beginning}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Add+a+Child+at+the+Beginning]
|
||||
- {Task: Add a Child at the End}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Add+a+Child+at+the+End]
|
||||
- {Task: Replace a Child with Another Child}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Replace+a+Child+with+Another+Child]
|
||||
- {Task: Replace Multiple Children with Another Child}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Replace+Multiple+Children+with+Another+Child]
|
||||
- {Task: Insert Child Before a Given Child}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Insert+Child+Before+a+Given+Child]
|
||||
- {Task: Insert Child After a Given Child}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Insert+Child+After+a+Given+Child]
|
||||
- {Deletions}[../../tasks/rdoc/parent_rdoc.html#label-Deletions]
|
||||
- {Task: Remove a Given Child}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Remove+a+Given+Child]
|
||||
- {Task: Remove the Child at a Specified Offset}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Remove+the+Child+at+a+Specified+Offset]
|
||||
- {Task: Remove Children That Meet Specified Criteria}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Remove+Children+That+Meet+Specified+Criteria]
|
||||
- {Iterations}[../../tasks/rdoc/parent_rdoc.html#label-Iterations]
|
||||
- {Task: Iterate Over Children}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Iterate+Over+Children]
|
||||
- {Task: Iterate Over Child Indexes}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Iterate+Over+Child+Indexes]
|
||||
- {Clones}[../../tasks/rdoc/parent_rdoc.html#label-Clones]
|
||||
- {Task: Clone Deeply}[../../tasks/rdoc/parent_rdoc.html#label-Task-3A+Clone+Deeply]
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
Tasks on this page:
|
||||
|
||||
- {Siblings}[#label-Siblings]
|
||||
- {Task: Find Previous Sibling}[#label-Task-3A+Find+Previous+Sibling]
|
||||
- {Task: Find Next Sibling}[#label-Task-3A+Find+Next+Sibling]
|
||||
- {Position}[#label-Position]
|
||||
- {Task: Find Own Index Among Siblings}[#label-Task-3A+Find+Own+Index+Among+Siblings]
|
||||
- {Recursive Traversal}[#label-Recursive+Traversal]
|
||||
- {Task: Traverse Each Recursively}[#label-Task-3A+Traverse+Each+Recursively]
|
||||
- {Recursive Search}[#label-Recursive+Search]
|
||||
- {Task: Traverse Each Recursively}[#label-Task-3A+Traverse+Each+Recursively]
|
||||
- {Representation}[#label-Representation]
|
||||
- {Task: Represent a String}[#label-Task-3A+Represent+a+String]
|
||||
- {Parent?}[#label-Parent-3F]
|
||||
- {Task: Determine Whether the Node is a Parent}[#label-Task-3A+Determine+Whether+the+Node+is+a+Parent]
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
Tasks on this page:
|
||||
|
||||
- {Queries}[#label-Queries]
|
||||
- {Task: Get the Count of Children}[#label-Task-3A+Get+the+Count+of+Children]
|
||||
- {Task: Get the Child at a Given Index}[#label-Task-3A+Get+the+Child+at+a+Given+Index]
|
||||
- {Task: Get the Index of a Given Child}[#label-Task-3A+Get+the+Index+of+a+Given+Child]
|
||||
- {Task: Get the Children}[#label-Task-3A+Get+the+Children]
|
||||
- {Task: Determine Whether the Node is a Parent}[#label-Task-3A+Determine+Whether+the+Node+is+a+Parent]
|
||||
- {Additions}[#label-Additions]
|
||||
- {Task: Add a Child at the Beginning}[#label-Task-3A+Add+a+Child+at+the+Beginning]
|
||||
- {Task: Add a Child at the End}[#label-Task-3A+Add+a+Child+at+the+End]
|
||||
- {Task: Replace a Child with Another Child}[#label-Task-3A+Replace+a+Child+with+Another+Child]
|
||||
- {Task: Replace Multiple Children with Another Child}[#label-Task-3A+Replace+Multiple+Children+with+Another+Child]
|
||||
- {Task: Insert Child Before a Given Child}[#label-Task-3A+Insert+Child+Before+a+Given+Child]
|
||||
- {Task: Insert Child After a Given Child}[#label-Task-3A+Insert+Child+After+a+Given+Child]
|
||||
- {Deletions}[#label-Deletions]
|
||||
- {Task: Remove a Given Child}[#label-Task-3A+Remove+a+Given+Child]
|
||||
- {Task: Remove the Child at a Specified Offset}[#label-Task-3A+Remove+the+Child+at+a+Specified+Offset]
|
||||
- {Task: Remove Children That Meet Specified Criteria}[#label-Task-3A+Remove+Children+That+Meet+Specified+Criteria]
|
||||
- {Iterations}[#label-Iterations]
|
||||
- {Task: Iterate Over Children}[#label-Task-3A+Iterate+Over+Children]
|
||||
- {Task: Iterate Over Child Indexes}[#label-Task-3A+Iterate+Over+Child+Indexes]
|
||||
- {Clones}[#label-Clones]
|
||||
- {Task: Clone Deeply}[#label-Task-3A+Clone+Deeply]
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,3 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
require_relative "rexml/document"
|
||||
@@ -0,0 +1,63 @@
|
||||
# frozen_string_literal: false
|
||||
#vim:ts=2 sw=2 noexpandtab:
|
||||
require_relative 'child'
|
||||
require_relative 'source'
|
||||
|
||||
module REXML
|
||||
# This class needs:
|
||||
# * Documentation
|
||||
# * Work! Not all types of attlists are intelligently parsed, so we just
|
||||
# spew back out what we get in. This works, but it would be better if
|
||||
# we formatted the output ourselves.
|
||||
#
|
||||
# AttlistDecls provide *just* enough support to allow namespace
|
||||
# declarations. If you need some sort of generalized support, or have an
|
||||
# interesting idea about how to map the hideous, terrible design of DTD
|
||||
# AttlistDecls onto an intuitive Ruby interface, let me know. I'm desperate
|
||||
# for anything to make DTDs more palateable.
|
||||
class AttlistDecl < Child
|
||||
include Enumerable
|
||||
|
||||
# What is this? Got me.
|
||||
attr_reader :element_name
|
||||
|
||||
# Create an AttlistDecl, pulling the information from a Source. Notice
|
||||
# that this isn't very convenient; to create an AttlistDecl, you basically
|
||||
# have to format it yourself, and then have the initializer parse it.
|
||||
# Sorry, but for the foreseeable future, DTD support in REXML is pretty
|
||||
# weak on convenience. Have I mentioned how much I hate DTDs?
|
||||
def initialize(source)
|
||||
super()
|
||||
if (source.kind_of? Array)
|
||||
@element_name, @pairs, @contents = *source
|
||||
end
|
||||
end
|
||||
|
||||
# Access the attlist attribute/value pairs.
|
||||
# value = attlist_decl[ attribute_name ]
|
||||
def [](key)
|
||||
@pairs[key]
|
||||
end
|
||||
|
||||
# Whether an attlist declaration includes the given attribute definition
|
||||
# if attlist_decl.include? "xmlns:foobar"
|
||||
def include?(key)
|
||||
@pairs.keys.include? key
|
||||
end
|
||||
|
||||
# Iterate over the key/value pairs:
|
||||
# attlist_decl.each { |attribute_name, attribute_value| ... }
|
||||
def each(&block)
|
||||
@pairs.each(&block)
|
||||
end
|
||||
|
||||
# Write out exactly what we got in.
|
||||
def write out, indent=-1
|
||||
out << @contents
|
||||
end
|
||||
|
||||
def node_type
|
||||
:attlistdecl
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,210 @@
|
||||
# frozen_string_literal: true
|
||||
require_relative "namespace"
|
||||
require_relative 'text'
|
||||
|
||||
module REXML
|
||||
# Defines an Element Attribute; IE, a attribute=value pair, as in:
|
||||
# <element attribute="value"/>. Attributes can be in their own
|
||||
# namespaces. General users of REXML will not interact with the
|
||||
# Attribute class much.
|
||||
class Attribute
|
||||
include Node
|
||||
include Namespace
|
||||
|
||||
# The element to which this attribute belongs
|
||||
attr_reader :element
|
||||
PATTERN = /\s*(#{NAME_STR})\s*=\s*(["'])(.*?)\2/um
|
||||
|
||||
NEEDS_A_SECOND_CHECK = /(<|&((#{Entity::NAME});|(#0*((?:\d+)|(?:x[a-fA-F0-9]+)));)?)/um
|
||||
|
||||
# Constructor.
|
||||
# FIXME: The parser doesn't catch illegal characters in attributes
|
||||
#
|
||||
# first::
|
||||
# Either: an Attribute, which this new attribute will become a
|
||||
# clone of; or a String, which is the name of this attribute
|
||||
# second::
|
||||
# If +first+ is an Attribute, then this may be an Element, or nil.
|
||||
# If nil, then the Element parent of this attribute is the parent
|
||||
# of the +first+ Attribute. If the first argument is a String,
|
||||
# then this must also be a String, and is the content of the attribute.
|
||||
# If this is the content, it must be fully normalized (contain no
|
||||
# illegal characters).
|
||||
# parent::
|
||||
# Ignored unless +first+ is a String; otherwise, may be the Element
|
||||
# parent of this attribute, or nil.
|
||||
#
|
||||
#
|
||||
# Attribute.new( attribute_to_clone )
|
||||
# Attribute.new( attribute_to_clone, parent_element )
|
||||
# Attribute.new( "attr", "attr_value" )
|
||||
# Attribute.new( "attr", "attr_value", parent_element )
|
||||
def initialize( first, second=nil, parent=nil )
|
||||
@normalized = @unnormalized = @element = nil
|
||||
if first.kind_of? Attribute
|
||||
self.name = first.expanded_name
|
||||
@unnormalized = first.value
|
||||
if second.kind_of? Element
|
||||
@element = second
|
||||
else
|
||||
@element = first.element
|
||||
end
|
||||
elsif first.kind_of? String
|
||||
@element = parent
|
||||
self.name = first
|
||||
@normalized = second.to_s
|
||||
else
|
||||
raise "illegal argument #{first.class.name} to Attribute constructor"
|
||||
end
|
||||
end
|
||||
|
||||
# Returns the namespace of the attribute.
|
||||
#
|
||||
# e = Element.new( "elns:myelement" )
|
||||
# e.add_attribute( "nsa:a", "aval" )
|
||||
# e.add_attribute( "b", "bval" )
|
||||
# e.attributes.get_attribute( "a" ).prefix # -> "nsa"
|
||||
# e.attributes.get_attribute( "b" ).prefix # -> ""
|
||||
# a = Attribute.new( "x", "y" )
|
||||
# a.prefix # -> ""
|
||||
def prefix
|
||||
super
|
||||
end
|
||||
|
||||
# Returns the namespace URL, if defined, or nil otherwise
|
||||
#
|
||||
# e = Element.new("el")
|
||||
# e.add_namespace("ns", "http://url")
|
||||
# e.add_attribute("ns:a", "b")
|
||||
# e.add_attribute("nsx:a", "c")
|
||||
# e.attribute("ns:a").namespace # => "http://url"
|
||||
# e.attribute("nsx:a").namespace # => nil
|
||||
#
|
||||
# This method always returns "" for no namespace attribute. Because
|
||||
# the default namespace doesn't apply to attribute names.
|
||||
#
|
||||
# From https://www.w3.org/TR/xml-names/#uniqAttrs
|
||||
#
|
||||
# > the default namespace does not apply to attribute names
|
||||
#
|
||||
# e = REXML::Element.new("el")
|
||||
# e.add_namespace("", "http://example.com/")
|
||||
# e.namespace # => "http://example.com/"
|
||||
# e.add_attribute("a", "b")
|
||||
# e.attribute("a").namespace # => ""
|
||||
def namespace arg=nil
|
||||
arg = prefix if arg.nil?
|
||||
if arg == ""
|
||||
""
|
||||
else
|
||||
@element.namespace(arg)
|
||||
end
|
||||
end
|
||||
|
||||
# Returns true if other is an Attribute and has the same name and value,
|
||||
# false otherwise.
|
||||
def ==( other )
|
||||
other.kind_of?(Attribute) and other.name==name and other.value==value
|
||||
end
|
||||
|
||||
# Creates (and returns) a hash from both the name and value
|
||||
def hash
|
||||
name.hash + value.hash
|
||||
end
|
||||
|
||||
# Returns this attribute out as XML source, expanding the name
|
||||
#
|
||||
# a = Attribute.new( "x", "y" )
|
||||
# a.to_string # -> "x='y'"
|
||||
# b = Attribute.new( "ns:x", "y" )
|
||||
# b.to_string # -> "ns:x='y'"
|
||||
def to_string
|
||||
value = to_s
|
||||
if @element and @element.context and @element.context[:attribute_quote] == :quote
|
||||
value = value.gsub('"', '"') if value.include?('"')
|
||||
%Q^#@expanded_name="#{value}"^
|
||||
else
|
||||
value = value.gsub("'", ''') if value.include?("'")
|
||||
"#@expanded_name='#{value}'"
|
||||
end
|
||||
end
|
||||
|
||||
def doctype
|
||||
@element&.document&.doctype
|
||||
end
|
||||
|
||||
# Returns the attribute value, with entities replaced
|
||||
def to_s
|
||||
return @normalized if @normalized
|
||||
|
||||
@normalized = Text::normalize( @unnormalized, doctype )
|
||||
@normalized
|
||||
end
|
||||
|
||||
# Returns the UNNORMALIZED value of this attribute. That is, entities
|
||||
# have been expanded to their values
|
||||
def value
|
||||
return @unnormalized if @unnormalized
|
||||
|
||||
@unnormalized = Text::unnormalize(@normalized, doctype,
|
||||
entity_expansion_text_limit: @element&.document&.entity_expansion_text_limit)
|
||||
end
|
||||
|
||||
# The normalized value of this attribute. That is, the attribute with
|
||||
# entities intact.
|
||||
def normalized=(new_normalized)
|
||||
@normalized = new_normalized
|
||||
@unnormalized = nil
|
||||
end
|
||||
|
||||
# Returns a copy of this attribute
|
||||
def clone
|
||||
Attribute.new self
|
||||
end
|
||||
|
||||
# Sets the element of which this object is an attribute. Normally, this
|
||||
# is not directly called.
|
||||
#
|
||||
# Returns this attribute
|
||||
def element=( element )
|
||||
@element = element
|
||||
|
||||
if @normalized
|
||||
Text.check( @normalized, NEEDS_A_SECOND_CHECK )
|
||||
end
|
||||
|
||||
self
|
||||
end
|
||||
|
||||
# Removes this Attribute from the tree, and returns true if successful
|
||||
#
|
||||
# This method is usually not called directly.
|
||||
def remove
|
||||
@element.attributes.delete self.name unless @element.nil?
|
||||
end
|
||||
|
||||
# Writes this attribute (EG, puts 'key="value"' to the output)
|
||||
def write( output, indent=-1 )
|
||||
output << to_string
|
||||
end
|
||||
|
||||
def node_type
|
||||
:attribute
|
||||
end
|
||||
|
||||
def inspect
|
||||
rv = +""
|
||||
write( rv )
|
||||
rv
|
||||
end
|
||||
|
||||
def xpath
|
||||
@element.xpath + "/@#{self.expanded_name}"
|
||||
end
|
||||
|
||||
def document
|
||||
@element&.document
|
||||
end
|
||||
end
|
||||
end
|
||||
#vim:ts=2 sw=2 noexpandtab:
|
||||
@@ -0,0 +1,68 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "text"
|
||||
|
||||
module REXML
|
||||
class CData < Text
|
||||
START = '<![CDATA['
|
||||
STOP = ']]>'
|
||||
ILLEGAL = /(\]\]>)/
|
||||
|
||||
# Constructor. CData is data between <![CDATA[ ... ]]>
|
||||
#
|
||||
# _Examples_
|
||||
# CData.new( source )
|
||||
# CData.new( "Here is some CDATA" )
|
||||
# CData.new( "Some unprocessed data", respect_whitespace_TF, parent_element )
|
||||
def initialize( first, whitespace=true, parent=nil )
|
||||
super( first, whitespace, parent, false, true, ILLEGAL )
|
||||
end
|
||||
|
||||
# Make a copy of this object
|
||||
#
|
||||
# _Examples_
|
||||
# c = CData.new( "Some text" )
|
||||
# d = c.clone
|
||||
# d.to_s # -> "Some text"
|
||||
def clone
|
||||
CData.new self
|
||||
end
|
||||
|
||||
# Returns the content of this CData object
|
||||
#
|
||||
# _Examples_
|
||||
# c = CData.new( "Some text" )
|
||||
# c.to_s # -> "Some text"
|
||||
def to_s
|
||||
@string
|
||||
end
|
||||
|
||||
def value
|
||||
@string
|
||||
end
|
||||
|
||||
# == DEPRECATED
|
||||
# See the rexml/formatters package
|
||||
#
|
||||
# Generates XML output of this object
|
||||
#
|
||||
# output::
|
||||
# Where to write the string. Defaults to $stdout
|
||||
# indent::
|
||||
# The amount to indent this node by
|
||||
# transitive::
|
||||
# Ignored
|
||||
# ie_hack::
|
||||
# Ignored
|
||||
#
|
||||
# _Examples_
|
||||
# c = CData.new( " Some text " )
|
||||
# c.write( $stdout ) #-> <![CDATA[ Some text ]]>
|
||||
def write( output=$stdout, indent=-1, transitive=false, ie_hack=false )
|
||||
Kernel.warn( "#{self.class.name}#write is deprecated", uplevel: 1)
|
||||
indent( output, indent )
|
||||
output << START
|
||||
output << @string
|
||||
output << STOP
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,96 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "node"
|
||||
|
||||
module REXML
|
||||
##
|
||||
# A Child object is something contained by a parent, and this class
|
||||
# contains methods to support that. Most user code will not use this
|
||||
# class directly.
|
||||
class Child
|
||||
include Node
|
||||
attr_reader :parent # The Parent of this object
|
||||
|
||||
# Constructor. Any inheritors of this class should call super to make
|
||||
# sure this method is called.
|
||||
# parent::
|
||||
# if supplied, the parent of this child will be set to the
|
||||
# supplied value, and self will be added to the parent
|
||||
def initialize( parent = nil )
|
||||
@parent = nil
|
||||
# Declare @parent, but don't define it. The next line sets the
|
||||
# parent.
|
||||
parent.add( self ) if parent
|
||||
end
|
||||
|
||||
# Replaces this object with another object. Basically, calls
|
||||
# Parent.replace_child
|
||||
#
|
||||
# Returns:: self
|
||||
def replace_with( child )
|
||||
@parent.replace_child( self, child )
|
||||
self
|
||||
end
|
||||
|
||||
# Removes this child from the parent.
|
||||
#
|
||||
# Returns:: self
|
||||
def remove
|
||||
unless @parent.nil?
|
||||
@parent.delete self
|
||||
end
|
||||
self
|
||||
end
|
||||
|
||||
# Sets the parent of this child to the supplied argument.
|
||||
#
|
||||
# other::
|
||||
# Must be a Parent object. If this object is the same object as the
|
||||
# existing parent of this child, no action is taken. Otherwise, this
|
||||
# child is removed from the current parent (if one exists), and is added
|
||||
# to the new parent.
|
||||
# Returns:: The parent added
|
||||
def parent=( other )
|
||||
return @parent if @parent == other
|
||||
@parent.delete self if defined? @parent and @parent
|
||||
@parent = other
|
||||
end
|
||||
|
||||
alias :next_sibling :next_sibling_node
|
||||
alias :previous_sibling :previous_sibling_node
|
||||
|
||||
# Sets the next sibling of this child. This can be used to insert a child
|
||||
# after some other child.
|
||||
# a = Element.new("a")
|
||||
# b = a.add_element("b")
|
||||
# c = Element.new("c")
|
||||
# b.next_sibling = c
|
||||
# # => <a><b/><c/></a>
|
||||
def next_sibling=( other )
|
||||
parent.insert_after self, other
|
||||
end
|
||||
|
||||
# Sets the previous sibling of this child. This can be used to insert a
|
||||
# child before some other child.
|
||||
# a = Element.new("a")
|
||||
# b = a.add_element("b")
|
||||
# c = Element.new("c")
|
||||
# b.previous_sibling = c
|
||||
# # => <a><b/><c/></a>
|
||||
def previous_sibling=(other)
|
||||
parent.insert_before self, other
|
||||
end
|
||||
|
||||
# Returns:: the document this child belongs to, or nil if this child
|
||||
# belongs to no document
|
||||
def document
|
||||
parent&.document
|
||||
end
|
||||
|
||||
# This doesn't yet handle encodings
|
||||
def bytes
|
||||
document&.encoding
|
||||
|
||||
to_s
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,80 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "child"
|
||||
|
||||
module REXML
|
||||
##
|
||||
# Represents an XML comment; that is, text between \<!-- ... -->
|
||||
class Comment < Child
|
||||
include Comparable
|
||||
START = "<!--"
|
||||
STOP = "-->"
|
||||
|
||||
# The content text
|
||||
|
||||
attr_accessor :string
|
||||
|
||||
##
|
||||
# Constructor. The first argument can be one of three types:
|
||||
# @param first If String, the contents of this comment are set to the
|
||||
# argument. If Comment, the argument is duplicated. If
|
||||
# Source, the argument is scanned for a comment.
|
||||
# @param second If the first argument is a Source, this argument
|
||||
# should be nil, not supplied, or a Parent to be set as the parent
|
||||
# of this object
|
||||
def initialize( first, second = nil )
|
||||
super(second)
|
||||
if first.kind_of? String
|
||||
@string = first
|
||||
elsif first.kind_of? Comment
|
||||
@string = first.string
|
||||
end
|
||||
end
|
||||
|
||||
def clone
|
||||
Comment.new self
|
||||
end
|
||||
|
||||
# == DEPRECATED
|
||||
# See REXML::Formatters
|
||||
#
|
||||
# output::
|
||||
# Where to write the string
|
||||
# indent::
|
||||
# An integer. If -1, no indenting will be used; otherwise, the
|
||||
# indentation will be this number of spaces, and children will be
|
||||
# indented an additional amount.
|
||||
# transitive::
|
||||
# Ignored by this class. The contents of comments are never modified.
|
||||
# ie_hack::
|
||||
# Needed for conformity to the child API, but not used by this class.
|
||||
def write( output, indent=-1, transitive=false, ie_hack=false )
|
||||
Kernel.warn("#{self.class.name}#write is deprecated. See REXML::Formatters", uplevel: 1)
|
||||
indent( output, indent )
|
||||
output << START
|
||||
output << @string
|
||||
output << STOP
|
||||
end
|
||||
|
||||
alias :to_s :string
|
||||
|
||||
##
|
||||
# Compares this Comment to another; the contents of the comment are used
|
||||
# in the comparison.
|
||||
def <=>(other)
|
||||
other.to_s <=> @string
|
||||
end
|
||||
|
||||
##
|
||||
# Compares this Comment to another; the contents of the comment are used
|
||||
# in the comparison.
|
||||
def ==( other )
|
||||
other.kind_of? Comment and
|
||||
(other <=> self) == 0
|
||||
end
|
||||
|
||||
def node_type
|
||||
:comment
|
||||
end
|
||||
end
|
||||
end
|
||||
#vim:ts=2 sw=2 noexpandtab:
|
||||
@@ -0,0 +1,306 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "parent"
|
||||
require_relative "parseexception"
|
||||
require_relative "namespace"
|
||||
require_relative 'entity'
|
||||
require_relative 'attlistdecl'
|
||||
require_relative 'xmltokens'
|
||||
|
||||
module REXML
|
||||
class ReferenceWriter
|
||||
def initialize(id_type,
|
||||
public_id_literal,
|
||||
system_literal,
|
||||
context=nil)
|
||||
@id_type = id_type
|
||||
@public_id_literal = public_id_literal
|
||||
@system_literal = system_literal
|
||||
if context and context[:prologue_quote] == :apostrophe
|
||||
@default_quote = "'"
|
||||
else
|
||||
@default_quote = "\""
|
||||
end
|
||||
end
|
||||
|
||||
def write(output)
|
||||
output << " #{@id_type}"
|
||||
if @public_id_literal
|
||||
if @public_id_literal.include?("'")
|
||||
quote = "\""
|
||||
else
|
||||
quote = @default_quote
|
||||
end
|
||||
output << " #{quote}#{@public_id_literal}#{quote}"
|
||||
end
|
||||
if @system_literal
|
||||
if @system_literal.include?("'")
|
||||
quote = "\""
|
||||
elsif @system_literal.include?("\"")
|
||||
quote = "'"
|
||||
else
|
||||
quote = @default_quote
|
||||
end
|
||||
output << " #{quote}#{@system_literal}#{quote}"
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# Represents an XML DOCTYPE declaration; that is, the contents of <!DOCTYPE
|
||||
# ... >. DOCTYPES can be used to declare the DTD of a document, as well as
|
||||
# being used to declare entities used in the document.
|
||||
class DocType < Parent
|
||||
include XMLTokens
|
||||
START = "<!DOCTYPE"
|
||||
STOP = ">"
|
||||
SYSTEM = "SYSTEM"
|
||||
PUBLIC = "PUBLIC"
|
||||
DEFAULT_ENTITIES = {
|
||||
'gt'=>EntityConst::GT,
|
||||
'lt'=>EntityConst::LT,
|
||||
'quot'=>EntityConst::QUOT,
|
||||
"apos"=>EntityConst::APOS
|
||||
}
|
||||
|
||||
# name is the name of the doctype
|
||||
# external_id is the referenced DTD, if given
|
||||
attr_reader :name, :external_id, :entities, :namespaces
|
||||
|
||||
# Constructor
|
||||
#
|
||||
# dt = DocType.new( 'foo', '-//I/Hate/External/IDs' )
|
||||
# # <!DOCTYPE foo '-//I/Hate/External/IDs'>
|
||||
# dt = DocType.new( doctype_to_clone )
|
||||
# # Incomplete. Shallow clone of doctype
|
||||
#
|
||||
# +Note+ that the constructor:
|
||||
#
|
||||
# Doctype.new( Source.new( "<!DOCTYPE foo 'bar'>" ) )
|
||||
#
|
||||
# is _deprecated_. Do not use it. It will probably disappear.
|
||||
def initialize( first, parent=nil )
|
||||
@entities = DEFAULT_ENTITIES
|
||||
@long_name = @uri = nil
|
||||
if first.kind_of? String
|
||||
super()
|
||||
@name = first
|
||||
@external_id = parent
|
||||
elsif first.kind_of? DocType
|
||||
super( parent )
|
||||
@name = first.name
|
||||
@external_id = first.external_id
|
||||
@long_name = first.instance_variable_get(:@long_name)
|
||||
@uri = first.instance_variable_get(:@uri)
|
||||
elsif first.kind_of? Array
|
||||
super( parent )
|
||||
@name = first[0]
|
||||
@external_id = first[1]
|
||||
@long_name = first[2]
|
||||
@uri = first[3]
|
||||
elsif first.kind_of? Source
|
||||
super( parent )
|
||||
parser = Parsers::BaseParser.new( first )
|
||||
event = parser.pull
|
||||
if event[0] == :start_doctype
|
||||
@name, @external_id, @long_name, @uri, = event[1..-1]
|
||||
end
|
||||
else
|
||||
super()
|
||||
end
|
||||
end
|
||||
|
||||
def node_type
|
||||
:doctype
|
||||
end
|
||||
|
||||
def attributes_of element
|
||||
rv = []
|
||||
each do |child|
|
||||
child.each do |key,val|
|
||||
rv << Attribute.new(key,val)
|
||||
end if child.kind_of? AttlistDecl and child.element_name == element
|
||||
end
|
||||
rv
|
||||
end
|
||||
|
||||
def attribute_of element, attribute
|
||||
att_decl = find do |child|
|
||||
child.kind_of? AttlistDecl and
|
||||
child.element_name == element and
|
||||
child.include? attribute
|
||||
end
|
||||
return nil unless att_decl
|
||||
att_decl[attribute]
|
||||
end
|
||||
|
||||
def clone
|
||||
DocType.new self
|
||||
end
|
||||
|
||||
# output::
|
||||
# Where to write the string
|
||||
# indent::
|
||||
# An integer. If -1, no indentation will be used; otherwise, the
|
||||
# indentation will be this number of spaces, and children will be
|
||||
# indented an additional amount.
|
||||
# transitive::
|
||||
# Ignored
|
||||
# ie_hack::
|
||||
# Ignored
|
||||
def write( output, indent=0, transitive=false, ie_hack=false )
|
||||
f = REXML::Formatters::Default.new
|
||||
indent( output, indent )
|
||||
output << START
|
||||
output << ' '
|
||||
output << @name
|
||||
if @external_id
|
||||
reference_writer = ReferenceWriter.new(@external_id,
|
||||
@long_name,
|
||||
@uri,
|
||||
context)
|
||||
reference_writer.write(output)
|
||||
end
|
||||
unless @children.empty?
|
||||
output << ' ['
|
||||
@children.each { |child|
|
||||
output << "\n"
|
||||
f.write( child, output )
|
||||
}
|
||||
output << "\n]"
|
||||
end
|
||||
output << STOP
|
||||
end
|
||||
|
||||
def context
|
||||
@parent&.context
|
||||
end
|
||||
|
||||
def entity( name )
|
||||
@entities[name]&.unnormalized
|
||||
end
|
||||
|
||||
def add child
|
||||
super(child)
|
||||
@entities = DEFAULT_ENTITIES.clone if @entities == DEFAULT_ENTITIES
|
||||
@entities[ child.name ] = child if child.kind_of? Entity
|
||||
end
|
||||
|
||||
# This method retrieves the public identifier identifying the document's
|
||||
# DTD.
|
||||
#
|
||||
# Method contributed by Henrik Martensson
|
||||
def public
|
||||
case @external_id
|
||||
when "SYSTEM"
|
||||
nil
|
||||
when "PUBLIC"
|
||||
@long_name
|
||||
end
|
||||
end
|
||||
|
||||
# This method retrieves the system identifier identifying the document's DTD
|
||||
#
|
||||
# Method contributed by Henrik Martensson
|
||||
def system
|
||||
case @external_id
|
||||
when "SYSTEM"
|
||||
@long_name
|
||||
when "PUBLIC"
|
||||
@uri.kind_of?(String) ? @uri : nil
|
||||
end
|
||||
end
|
||||
|
||||
# This method returns a list of notations that have been declared in the
|
||||
# _internal_ DTD subset. Notations in the external DTD subset are not
|
||||
# listed.
|
||||
#
|
||||
# Method contributed by Henrik Martensson
|
||||
def notations
|
||||
children().select {|node| node.kind_of?(REXML::NotationDecl)}
|
||||
end
|
||||
|
||||
# Retrieves a named notation. Only notations declared in the internal
|
||||
# DTD subset can be retrieved.
|
||||
#
|
||||
# Method contributed by Henrik Martensson
|
||||
def notation(name)
|
||||
notations.find { |notation_decl|
|
||||
notation_decl.name == name
|
||||
}
|
||||
end
|
||||
end
|
||||
|
||||
# We don't really handle any of these since we're not a validating
|
||||
# parser, so we can be pretty dumb about them. All we need to be able
|
||||
# to do is spew them back out on a write()
|
||||
|
||||
# This is an abstract class. You never use this directly; it serves as a
|
||||
# parent class for the specific declarations.
|
||||
class Declaration < Child
|
||||
def initialize src
|
||||
super()
|
||||
@string = src
|
||||
end
|
||||
|
||||
def to_s
|
||||
@string+'>'
|
||||
end
|
||||
|
||||
# == DEPRECATED
|
||||
# See REXML::Formatters
|
||||
#
|
||||
def write( output, indent )
|
||||
output << to_s
|
||||
end
|
||||
end
|
||||
|
||||
public
|
||||
class ElementDecl < Declaration
|
||||
def initialize( src )
|
||||
super
|
||||
end
|
||||
end
|
||||
|
||||
class ExternalEntity < Child
|
||||
def initialize( src )
|
||||
super()
|
||||
@entity = src
|
||||
end
|
||||
def to_s
|
||||
@entity
|
||||
end
|
||||
def write( output, indent )
|
||||
output << @entity
|
||||
end
|
||||
end
|
||||
|
||||
class NotationDecl < Child
|
||||
attr_accessor :public, :system
|
||||
def initialize name, middle, pub, sys
|
||||
super(nil)
|
||||
@name = name
|
||||
@middle = middle
|
||||
@public = pub
|
||||
@system = sys
|
||||
end
|
||||
|
||||
def to_s
|
||||
context = parent&.context
|
||||
notation = "<!NOTATION #{@name}"
|
||||
reference_writer = ReferenceWriter.new(@middle, @public, @system, context)
|
||||
reference_writer.write(notation)
|
||||
notation << ">"
|
||||
notation
|
||||
end
|
||||
|
||||
def write( output, indent=-1 )
|
||||
output << to_s
|
||||
end
|
||||
|
||||
# This method retrieves the name of the notation.
|
||||
#
|
||||
# Method contributed by Henrik Martensson
|
||||
def name
|
||||
@name
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,471 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "security"
|
||||
require_relative "element"
|
||||
require_relative "xmldecl"
|
||||
require_relative "source"
|
||||
require_relative "comment"
|
||||
require_relative "doctype"
|
||||
require_relative "instruction"
|
||||
require_relative "rexml"
|
||||
require_relative "parseexception"
|
||||
require_relative "output"
|
||||
require_relative "parsers/baseparser"
|
||||
require_relative "parsers/streamparser"
|
||||
require_relative "parsers/treeparser"
|
||||
|
||||
module REXML
|
||||
# Represents an XML document.
|
||||
#
|
||||
# A document may have:
|
||||
#
|
||||
# - A single child that may be accessed via method #root.
|
||||
# - An XML declaration.
|
||||
# - A document type.
|
||||
# - Processing instructions.
|
||||
#
|
||||
# == In a Hurry?
|
||||
#
|
||||
# If you're somewhat familiar with XML
|
||||
# and have a particular task in mind,
|
||||
# you may want to see the
|
||||
# {tasks pages}[../doc/rexml/tasks/tocs/master_toc_rdoc.html],
|
||||
# and in particular, the
|
||||
# {tasks page for documents}[../doc/rexml/tasks/tocs/document_toc_rdoc.html].
|
||||
#
|
||||
class Document < Element
|
||||
# A convenient default XML declaration. Use:
|
||||
#
|
||||
# mydoc << XMLDecl.default
|
||||
#
|
||||
DECLARATION = XMLDecl.default
|
||||
|
||||
# :call-seq:
|
||||
# new(string = nil, context = {}) -> new_document
|
||||
# new(io_stream = nil, context = {}) -> new_document
|
||||
# new(document = nil, context = {}) -> new_document
|
||||
#
|
||||
# Returns a new \REXML::Document object.
|
||||
#
|
||||
# When no arguments are given,
|
||||
# returns an empty document:
|
||||
#
|
||||
# d = REXML::Document.new
|
||||
# d.to_s # => ""
|
||||
#
|
||||
# When argument +string+ is given, it must be a string
|
||||
# containing a valid XML document:
|
||||
#
|
||||
# xml_string = '<root><foo>Foo</foo><bar>Bar</bar></root>'
|
||||
# d = REXML::Document.new(xml_string)
|
||||
# d.to_s # => "<root><foo>Foo</foo><bar>Bar</bar></root>"
|
||||
#
|
||||
# When argument +io_stream+ is given, it must be an \IO object
|
||||
# that is opened for reading, and when read must return a valid XML document:
|
||||
#
|
||||
# File.write('t.xml', xml_string)
|
||||
# d = File.open('t.xml', 'r') do |io|
|
||||
# REXML::Document.new(io)
|
||||
# end
|
||||
# d.to_s # => "<root><foo>Foo</foo><bar>Bar</bar></root>"
|
||||
#
|
||||
# When argument +document+ is given, it must be an existing
|
||||
# document object, whose context and attributes (but not children)
|
||||
# are cloned into the new document:
|
||||
#
|
||||
# d = REXML::Document.new(xml_string)
|
||||
# d.children # => [<root> ... </>]
|
||||
# d.context = {raw: :all, compress_whitespace: :all}
|
||||
# d.add_attributes({'bar' => 0, 'baz' => 1})
|
||||
# d1 = REXML::Document.new(d)
|
||||
# d1.children # => []
|
||||
# d1.context # => {:raw=>:all, :compress_whitespace=>:all}
|
||||
# d1.attributes # => {"bar"=>bar='0', "baz"=>baz='1'}
|
||||
#
|
||||
# When argument +context+ is given, it must be a hash
|
||||
# containing context entries for the document;
|
||||
# see {Element Context}[../doc/rexml/context_rdoc.html]:
|
||||
#
|
||||
# context = {raw: :all, compress_whitespace: :all}
|
||||
# d = REXML::Document.new(xml_string, context)
|
||||
# d.context # => {:raw=>:all, :compress_whitespace=>:all}
|
||||
#
|
||||
def initialize( source = nil, context = {} )
|
||||
@entity_expansion_count = 0
|
||||
@entity_expansion_limit = Security.entity_expansion_limit
|
||||
@entity_expansion_text_limit = Security.entity_expansion_text_limit
|
||||
super()
|
||||
@context = context
|
||||
# `source = ""` is an invalid usage because no root element XML is an invalid XML.
|
||||
# But we accept `""` for backward compatibility.
|
||||
return if source.nil? or source == ""
|
||||
if source.kind_of? Document
|
||||
@context = source.context
|
||||
super source
|
||||
else
|
||||
build( source )
|
||||
end
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# node_type -> :document
|
||||
#
|
||||
# Returns the symbol +:document+.
|
||||
#
|
||||
def node_type
|
||||
:document
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# clone -> new_document
|
||||
#
|
||||
# Returns the new document resulting from executing
|
||||
# <tt>Document.new(self)</tt>. See Document.new.
|
||||
#
|
||||
def clone
|
||||
Document.new self
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# expanded_name -> empty_string
|
||||
#
|
||||
# Returns an empty string.
|
||||
#
|
||||
def expanded_name
|
||||
''
|
||||
#d = doc_type
|
||||
#d ? d.name : "UNDEFINED"
|
||||
end
|
||||
alias :name :expanded_name
|
||||
|
||||
# :call-seq:
|
||||
# add(xml_decl) -> self
|
||||
# add(doc_type) -> self
|
||||
# add(object) -> self
|
||||
#
|
||||
# Adds an object to the document; returns +self+.
|
||||
#
|
||||
# When argument +xml_decl+ is given,
|
||||
# it must be an REXML::XMLDecl object,
|
||||
# which becomes the XML declaration for the document,
|
||||
# replacing the previous XML declaration if any:
|
||||
#
|
||||
# d = REXML::Document.new
|
||||
# d.xml_decl.to_s # => ""
|
||||
# d.add(REXML::XMLDecl.new('2.0'))
|
||||
# d.xml_decl.to_s # => "<?xml version='2.0'?>"
|
||||
#
|
||||
# When argument +doc_type+ is given,
|
||||
# it must be an REXML::DocType object,
|
||||
# which becomes the document type for the document,
|
||||
# replacing the previous document type, if any:
|
||||
#
|
||||
# d = REXML::Document.new
|
||||
# d.doctype.to_s # => ""
|
||||
# d.add(REXML::DocType.new('foo'))
|
||||
# d.doctype.to_s # => "<!DOCTYPE foo>"
|
||||
#
|
||||
# When argument +object+ (not an REXML::XMLDecl or REXML::DocType object)
|
||||
# is given it is added as the last child:
|
||||
#
|
||||
# d = REXML::Document.new
|
||||
# d.add(REXML::Element.new('foo'))
|
||||
# d.to_s # => "<foo/>"
|
||||
#
|
||||
def add( child )
|
||||
if child.kind_of? XMLDecl
|
||||
if @children[0].kind_of? XMLDecl
|
||||
@children[0] = child
|
||||
else
|
||||
@children.unshift child
|
||||
end
|
||||
child.parent = self
|
||||
elsif child.kind_of? DocType
|
||||
# Find first Element or DocType node and insert the decl right
|
||||
# before it. If there is no such node, just insert the child at the
|
||||
# end. If there is a child and it is an DocType, then replace it.
|
||||
insert_before_index = @children.find_index { |x|
|
||||
x.kind_of?(Element) || x.kind_of?(DocType)
|
||||
}
|
||||
if insert_before_index # Not null = not end of list
|
||||
if @children[ insert_before_index ].kind_of? DocType
|
||||
@children[ insert_before_index ] = child
|
||||
else
|
||||
@children[ insert_before_index-1, 0 ] = child
|
||||
end
|
||||
else # Insert at end of list
|
||||
@children << child
|
||||
end
|
||||
child.parent = self
|
||||
else
|
||||
rv = super
|
||||
raise "attempted adding second root element to document" if @elements.size > 1
|
||||
rv
|
||||
end
|
||||
end
|
||||
alias :<< :add
|
||||
|
||||
# :call-seq:
|
||||
# add_element(name_or_element = nil, attributes = nil) -> new_element
|
||||
#
|
||||
# Adds an element to the document by calling REXML::Element.add_element:
|
||||
#
|
||||
# REXML::Element.add_element(name_or_element, attributes)
|
||||
def add_element(arg=nil, arg2=nil)
|
||||
rv = super
|
||||
raise "attempted adding second root element to document" if @elements.size > 1
|
||||
rv
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# root -> root_element or nil
|
||||
#
|
||||
# Returns the root element of the document, if it exists, otherwise +nil+:
|
||||
#
|
||||
# d = REXML::Document.new('<root></root>')
|
||||
# d.root # => <root/>
|
||||
# d = REXML::Document.new('')
|
||||
# d.root # => nil
|
||||
#
|
||||
def root
|
||||
elements[1]
|
||||
#self
|
||||
#@children.find { |item| item.kind_of? Element }
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# doctype -> doc_type or nil
|
||||
#
|
||||
# Returns the DocType object for the document, if it exists, otherwise +nil+:
|
||||
#
|
||||
# d = REXML::Document.new('<!DOCTYPE document SYSTEM "subjects.dtd">')
|
||||
# d.doctype.class # => REXML::DocType
|
||||
# d = REXML::Document.new('')
|
||||
# d.doctype.class # => nil
|
||||
#
|
||||
def doctype
|
||||
@children.find { |item| item.kind_of? DocType }
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# xml_decl -> xml_decl
|
||||
#
|
||||
# Returns the XMLDecl object for the document, if it exists,
|
||||
# otherwise the default XMLDecl object:
|
||||
#
|
||||
# d = REXML::Document.new('<?xml version="1.0" encoding="UTF-8"?>')
|
||||
# d.xml_decl.class # => REXML::XMLDecl
|
||||
# d.xml_decl.to_s # => "<?xml version='1.0' encoding='UTF-8'?>"
|
||||
# d = REXML::Document.new('')
|
||||
# d.xml_decl.class # => REXML::XMLDecl
|
||||
# d.xml_decl.to_s # => ""
|
||||
#
|
||||
def xml_decl
|
||||
rv = @children[0]
|
||||
return rv if rv.kind_of? XMLDecl
|
||||
@children.unshift(XMLDecl.default)[0]
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# version -> version_string
|
||||
#
|
||||
# Returns the XMLDecl version of this document as a string,
|
||||
# if it has been set, otherwise the default version:
|
||||
#
|
||||
# d = REXML::Document.new('<?xml version="2.0" encoding="UTF-8"?>')
|
||||
# d.version # => "2.0"
|
||||
# d = REXML::Document.new('')
|
||||
# d.version # => "1.0"
|
||||
#
|
||||
def version
|
||||
xml_decl().version
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# encoding -> encoding_string
|
||||
#
|
||||
# Returns the XMLDecl encoding of the document,
|
||||
# if it has been set, otherwise the default encoding:
|
||||
#
|
||||
# d = REXML::Document.new('<?xml version="1.0" encoding="UTF-16"?>')
|
||||
# d.encoding # => "UTF-16"
|
||||
# d = REXML::Document.new('')
|
||||
# d.encoding # => "UTF-8"
|
||||
#
|
||||
def encoding
|
||||
xml_decl().encoding
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# stand_alone?
|
||||
#
|
||||
# Returns the XMLDecl standalone value of the document as a string,
|
||||
# if it has been set, otherwise the default standalone value:
|
||||
#
|
||||
# d = REXML::Document.new('<?xml standalone="yes"?>')
|
||||
# d.stand_alone? # => "yes"
|
||||
# d = REXML::Document.new('')
|
||||
# d.stand_alone? # => nil
|
||||
#
|
||||
def stand_alone?
|
||||
xml_decl().stand_alone?
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# doc.write(output=$stdout, indent=-1, transitive=false, ie_hack=false, encoding=nil)
|
||||
# doc.write(options={:output => $stdout, :indent => -1, :transitive => false, :ie_hack => false, :encoding => nil})
|
||||
#
|
||||
# Write the XML tree out, optionally with indent. This writes out the
|
||||
# entire XML document, including XML declarations, doctype declarations,
|
||||
# and processing instructions (if any are given).
|
||||
#
|
||||
# A controversial point is whether Document should always write the XML
|
||||
# declaration (<?xml version='1.0'?>) whether or not one is given by the
|
||||
# user (or source document). REXML does not write one if one was not
|
||||
# specified, because it adds unnecessary bandwidth to applications such
|
||||
# as XML-RPC.
|
||||
#
|
||||
# Accept Nth argument style and options Hash style as argument.
|
||||
# The recommended style is options Hash style for one or more
|
||||
# arguments case.
|
||||
#
|
||||
# _Examples_
|
||||
# Document.new("<a><b/></a>").write
|
||||
#
|
||||
# output = ""
|
||||
# Document.new("<a><b/></a>").write(output)
|
||||
#
|
||||
# output = ""
|
||||
# Document.new("<a><b/></a>").write(:output => output, :indent => 2)
|
||||
#
|
||||
# See also the classes in the rexml/formatters package for the proper way
|
||||
# to change the default formatting of XML output.
|
||||
#
|
||||
# _Examples_
|
||||
#
|
||||
# output = ""
|
||||
# tr = Transitive.new
|
||||
# tr.write(Document.new("<a><b/></a>"), output)
|
||||
#
|
||||
# output::
|
||||
# output an object which supports '<< string'; this is where the
|
||||
# document will be written.
|
||||
# indent::
|
||||
# An integer. If -1, no indenting will be used; otherwise, the
|
||||
# indentation will be twice this number of spaces, and children will be
|
||||
# indented an additional amount. For a value of 3, every item will be
|
||||
# indented 3 more levels, or 6 more spaces (2 * 3). Defaults to -1
|
||||
# transitive::
|
||||
# If transitive is true and indent is >= 0, then the output will be
|
||||
# pretty-printed in such a way that the added whitespace does not affect
|
||||
# the absolute *value* of the document -- that is, it leaves the value
|
||||
# and number of Text nodes in the document unchanged.
|
||||
# ie_hack::
|
||||
# This hack inserts a space before the /> on empty tags to address
|
||||
# a limitation of Internet Explorer. Defaults to false
|
||||
# encoding::
|
||||
# Encoding name as String. Change output encoding to specified encoding
|
||||
# instead of encoding in XML declaration.
|
||||
# Defaults to nil. It means encoding in XML declaration is used.
|
||||
def write(*arguments)
|
||||
if arguments.size == 1 and arguments[0].class == Hash
|
||||
options = arguments[0]
|
||||
|
||||
output = options[:output]
|
||||
indent = options[:indent]
|
||||
transitive = options[:transitive]
|
||||
ie_hack = options[:ie_hack]
|
||||
encoding = options[:encoding]
|
||||
else
|
||||
output, indent, transitive, ie_hack, encoding, = *arguments
|
||||
end
|
||||
|
||||
output ||= $stdout
|
||||
indent ||= -1
|
||||
transitive = false if transitive.nil?
|
||||
ie_hack = false if ie_hack.nil?
|
||||
encoding ||= xml_decl.encoding
|
||||
|
||||
if encoding != 'UTF-8' && !output.kind_of?(Output)
|
||||
output = Output.new( output, encoding )
|
||||
end
|
||||
formatter = if indent > -1
|
||||
if transitive
|
||||
require_relative "formatters/transitive"
|
||||
REXML::Formatters::Transitive.new( indent, ie_hack )
|
||||
else
|
||||
REXML::Formatters::Pretty.new( indent, ie_hack )
|
||||
end
|
||||
else
|
||||
REXML::Formatters::Default.new( ie_hack )
|
||||
end
|
||||
formatter.write( self, output )
|
||||
end
|
||||
|
||||
|
||||
def Document::parse_stream( source, listener )
|
||||
Parsers::StreamParser.new( source, listener ).parse
|
||||
end
|
||||
|
||||
# Set the entity expansion limit. By default the limit is set to 10000.
|
||||
#
|
||||
# Deprecated. Use REXML::Security.entity_expansion_limit= instead.
|
||||
def Document::entity_expansion_limit=( val )
|
||||
Security.entity_expansion_limit = val
|
||||
end
|
||||
|
||||
# Get the entity expansion limit. By default the limit is set to 10000.
|
||||
#
|
||||
# Deprecated. Use REXML::Security.entity_expansion_limit= instead.
|
||||
def Document::entity_expansion_limit
|
||||
Security.entity_expansion_limit
|
||||
end
|
||||
|
||||
# Set the entity expansion limit. By default the limit is set to 10240.
|
||||
#
|
||||
# Deprecated. Use REXML::Security.entity_expansion_text_limit= instead.
|
||||
def Document::entity_expansion_text_limit=( val )
|
||||
Security.entity_expansion_text_limit = val
|
||||
end
|
||||
|
||||
# Get the entity expansion limit. By default the limit is set to 10240.
|
||||
#
|
||||
# Deprecated. Use REXML::Security.entity_expansion_text_limit instead.
|
||||
def Document::entity_expansion_text_limit
|
||||
Security.entity_expansion_text_limit
|
||||
end
|
||||
|
||||
attr_reader :entity_expansion_count
|
||||
attr_writer :entity_expansion_limit
|
||||
attr_accessor :entity_expansion_text_limit
|
||||
|
||||
def record_entity_expansion
|
||||
@entity_expansion_count += 1
|
||||
if @entity_expansion_count > @entity_expansion_limit
|
||||
raise "number of entity expansions exceeded, processing aborted."
|
||||
end
|
||||
end
|
||||
|
||||
def document
|
||||
self
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
attr_accessor :namespaces_cache
|
||||
|
||||
# New document level cache is created and available in this block.
|
||||
# This API is thread unsafe. Users can't change this document in this block.
|
||||
def enable_cache
|
||||
@namespaces_cache = {}
|
||||
begin
|
||||
yield
|
||||
ensure
|
||||
@namespaces_cache = nil
|
||||
end
|
||||
end
|
||||
|
||||
def build( source )
|
||||
Parsers::TreeParser.new( source, self ).parse
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,11 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "../child"
|
||||
module REXML
|
||||
module DTD
|
||||
class AttlistDecl < Child
|
||||
START = "<!ATTLIST"
|
||||
START_RE = /^\s*#{START}/um
|
||||
PATTERN_RE = /\s*(#{START}.*?>)/um
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,47 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "elementdecl"
|
||||
require_relative "entitydecl"
|
||||
require_relative "../comment"
|
||||
require_relative "notationdecl"
|
||||
require_relative "attlistdecl"
|
||||
require_relative "../parent"
|
||||
|
||||
module REXML
|
||||
module DTD
|
||||
class Parser
|
||||
def Parser.parse( input )
|
||||
case input
|
||||
when String
|
||||
parse_helper input
|
||||
when File
|
||||
parse_helper input.read
|
||||
end
|
||||
end
|
||||
|
||||
# Takes a String and parses it out
|
||||
def Parser.parse_helper( input )
|
||||
contents = Parent.new
|
||||
while input.size > 0
|
||||
case input
|
||||
when ElementDecl.PATTERN_RE
|
||||
match = $&
|
||||
contents << ElementDecl.new( match )
|
||||
when AttlistDecl.PATTERN_RE
|
||||
matchdata = $~
|
||||
contents << AttlistDecl.new( matchdata )
|
||||
when EntityDecl.PATTERN_RE
|
||||
matchdata = $~
|
||||
contents << EntityDecl.new( matchdata )
|
||||
when Comment.PATTERN_RE
|
||||
matchdata = $~
|
||||
contents << Comment.new( matchdata )
|
||||
when NotationDecl.PATTERN_RE
|
||||
matchdata = $~
|
||||
contents << NotationDecl.new( matchdata )
|
||||
end
|
||||
end
|
||||
contents
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,18 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "../child"
|
||||
module REXML
|
||||
module DTD
|
||||
class ElementDecl < Child
|
||||
START = "<!ELEMENT"
|
||||
START_RE = /^\s*#{START}/um
|
||||
# PATTERN_RE = /^\s*(#{START}.*?)>/um
|
||||
PATTERN_RE = /^\s*#{START}\s+((?:[:\w][-\.\w]*:)?[-!\*\.\w]*)(.*?)>/
|
||||
#\s*((((["']).*?\5)|[^\/'">]*)*?)(\/)?>/um, true)
|
||||
|
||||
def initialize match
|
||||
@name = match[1]
|
||||
@rest = match[2]
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,57 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "../child"
|
||||
module REXML
|
||||
module DTD
|
||||
class EntityDecl < Child
|
||||
START = "<!ENTITY"
|
||||
START_RE = /^\s*#{START}/um
|
||||
PUBLIC = /^\s*#{START}\s+(?:%\s+)?(\w+)\s+PUBLIC\s+((["']).*?\3)\s+((["']).*?\5)\s*>/um
|
||||
SYSTEM = /^\s*#{START}\s+(?:%\s+)?(\w+)\s+SYSTEM\s+((["']).*?\3)(?:\s+NDATA\s+\w+)?\s*>/um
|
||||
PLAIN = /^\s*#{START}\s+(\w+)\s+((["']).*?\3)\s*>/um
|
||||
PERCENT = /^\s*#{START}\s+%\s+(\w+)\s+((["']).*?\3)\s*>/um
|
||||
# <!ENTITY name SYSTEM "...">
|
||||
# <!ENTITY name "...">
|
||||
def initialize src
|
||||
super()
|
||||
md = nil
|
||||
if src.match( PUBLIC )
|
||||
md = src.match( PUBLIC, true )
|
||||
@middle = "PUBLIC"
|
||||
@content = "#{md[2]} #{md[4]}"
|
||||
elsif src.match( SYSTEM )
|
||||
md = src.match( SYSTEM, true )
|
||||
@middle = "SYSTEM"
|
||||
@content = md[2]
|
||||
elsif src.match( PLAIN )
|
||||
md = src.match( PLAIN, true )
|
||||
@middle = ""
|
||||
@content = md[2]
|
||||
elsif src.match( PERCENT )
|
||||
md = src.match( PERCENT, true )
|
||||
@middle = ""
|
||||
@content = md[2]
|
||||
end
|
||||
raise ParseException.new("failed Entity match", src) if md.nil?
|
||||
@name = md[1]
|
||||
end
|
||||
|
||||
def to_s
|
||||
rv = "<!ENTITY #@name "
|
||||
rv << "#@middle " if @middle.size > 0
|
||||
rv << @content
|
||||
rv
|
||||
end
|
||||
|
||||
def write( output, indent )
|
||||
indent( output, indent )
|
||||
output << to_s
|
||||
end
|
||||
|
||||
def EntityDecl.parse_source source, listener
|
||||
md = source.match( PATTERN_RE, true )
|
||||
thing = md[0].squeeze(" \t\n\r")
|
||||
listener.send inspect.downcase, thing
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,40 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "../child"
|
||||
module REXML
|
||||
module DTD
|
||||
class NotationDecl < Child
|
||||
START = "<!NOTATION"
|
||||
START_RE = /^\s*#{START}/um
|
||||
PUBLIC = /^\s*#{START}\s+(\w[\w-]*)\s+(PUBLIC)\s+((["']).*?\4)\s*>/um
|
||||
SYSTEM = /^\s*#{START}\s+(\w[\w-]*)\s+(SYSTEM)\s+((["']).*?\4)\s*>/um
|
||||
def initialize src
|
||||
super()
|
||||
if src.match( PUBLIC )
|
||||
md = src.match( PUBLIC, true )
|
||||
elsif src.match( SYSTEM )
|
||||
md = src.match( SYSTEM, true )
|
||||
else
|
||||
raise ParseException.new( "error parsing notation: no matching pattern", src )
|
||||
end
|
||||
@name = md[1]
|
||||
@middle = md[2]
|
||||
@rest = md[3]
|
||||
end
|
||||
|
||||
def to_s
|
||||
"<!NOTATION #@name #@middle #@rest>"
|
||||
end
|
||||
|
||||
def write( output, indent )
|
||||
indent( output, indent )
|
||||
output << to_s
|
||||
end
|
||||
|
||||
def NotationDecl.parse_source source, listener
|
||||
md = source.match( PATTERN_RE, true )
|
||||
thing = md[0].squeeze(" \t\n\r")
|
||||
listener.send inspect.downcase, thing
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,48 @@
|
||||
# coding: US-ASCII
|
||||
# frozen_string_literal: false
|
||||
module REXML
|
||||
module Encoding
|
||||
# ID ---> Encoding name
|
||||
attr_reader :encoding
|
||||
def encoding=(encoding)
|
||||
encoding = encoding.name if encoding.is_a?(::Encoding)
|
||||
if encoding.is_a?(String)
|
||||
original_encoding = encoding
|
||||
encoding = find_encoding(encoding)
|
||||
unless encoding
|
||||
raise ArgumentError, "Bad encoding name #{original_encoding}"
|
||||
end
|
||||
end
|
||||
encoding = encoding.upcase if encoding
|
||||
return false if defined?(@encoding) and encoding == @encoding
|
||||
@encoding = encoding || "UTF-8"
|
||||
true
|
||||
end
|
||||
|
||||
def encode(string)
|
||||
string.encode(@encoding)
|
||||
end
|
||||
|
||||
def decode(string)
|
||||
string.encode(::Encoding::UTF_8, @encoding)
|
||||
end
|
||||
|
||||
private
|
||||
def find_encoding(name)
|
||||
case name
|
||||
when /\Ashift-jis\z/i
|
||||
return "SHIFT_JIS"
|
||||
when /\ACP-(\d+)\z/
|
||||
name = "CP#{$1}"
|
||||
when /\AUTF-8\z/i
|
||||
return name
|
||||
end
|
||||
begin
|
||||
::Encoding::Converter.search_convpath(name, 'UTF-8')
|
||||
rescue ::Encoding::ConverterNotFoundError
|
||||
return nil
|
||||
end
|
||||
name
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,142 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative 'child'
|
||||
require_relative 'source'
|
||||
require_relative 'xmltokens'
|
||||
|
||||
module REXML
|
||||
class Entity < Child
|
||||
include XMLTokens
|
||||
PUBIDCHAR = "\x20\x0D\x0Aa-zA-Z0-9\\-()+,./:=?;!*@$_%#"
|
||||
SYSTEMLITERAL = %Q{((?:"[^"]*")|(?:'[^']*'))}
|
||||
PUBIDLITERAL = %Q{("[#{PUBIDCHAR}']*"|'[#{PUBIDCHAR}]*')}
|
||||
EXTERNALID = "(?:(?:(SYSTEM)\\s+#{SYSTEMLITERAL})|(?:(PUBLIC)\\s+#{PUBIDLITERAL}\\s+#{SYSTEMLITERAL}))"
|
||||
NDATADECL = "\\s+NDATA\\s+#{NAME}"
|
||||
PEREFERENCE = "%#{NAME};"
|
||||
PEREFERENCE_RE = /#{PEREFERENCE}/um
|
||||
ENTITYVALUE = %Q{((?:"(?:[^%&"]|#{PEREFERENCE}|#{REFERENCE})*")|(?:'([^%&']|#{PEREFERENCE}|#{REFERENCE})*'))}
|
||||
PEDEF = "(?:#{ENTITYVALUE}|#{EXTERNALID})"
|
||||
ENTITYDEF = "(?:#{ENTITYVALUE}|(?:#{EXTERNALID}(#{NDATADECL})?))"
|
||||
PEDECL = "<!ENTITY\\s+(%)\\s+#{NAME}\\s+#{PEDEF}\\s*>"
|
||||
GEDECL = "<!ENTITY\\s+#{NAME}\\s+#{ENTITYDEF}\\s*>"
|
||||
ENTITYDECL = /\s*(?:#{GEDECL})|(?:#{PEDECL})/um
|
||||
|
||||
attr_reader :name, :external, :ref, :ndata, :pubid, :value
|
||||
|
||||
# Create a new entity. Simple entities can be constructed by passing a
|
||||
# name, value to the constructor; this creates a generic, plain entity
|
||||
# reference. For anything more complicated, you have to pass a Source to
|
||||
# the constructor with the entity definition, or use the accessor methods.
|
||||
# +WARNING+: There is no validation of entity state except when the entity
|
||||
# is read from a stream. If you start poking around with the accessors,
|
||||
# you can easily create a non-conformant Entity.
|
||||
#
|
||||
# e = Entity.new( 'amp', '&' )
|
||||
def initialize stream, value=nil, parent=nil, reference=false
|
||||
super(parent)
|
||||
@ndata = @pubid = @value = @external = nil
|
||||
if stream.kind_of? Array
|
||||
@name = stream[1]
|
||||
if stream[-1] == '%'
|
||||
@reference = true
|
||||
stream.pop
|
||||
else
|
||||
@reference = false
|
||||
end
|
||||
if stream[2] =~ /SYSTEM|PUBLIC/
|
||||
@external = stream[2]
|
||||
if @external == 'SYSTEM'
|
||||
@ref = stream[3]
|
||||
@ndata = stream[4] if stream.size == 5
|
||||
else
|
||||
@pubid = stream[3]
|
||||
@ref = stream[4]
|
||||
end
|
||||
else
|
||||
@value = stream[2]
|
||||
end
|
||||
else
|
||||
@reference = reference
|
||||
@external = nil
|
||||
@name = stream
|
||||
@value = value
|
||||
end
|
||||
end
|
||||
|
||||
# Evaluates whether the given string matches an entity definition,
|
||||
# returning true if so, and false otherwise.
|
||||
def Entity::matches? string
|
||||
(ENTITYDECL =~ string) == 0
|
||||
end
|
||||
|
||||
# Evaluates to the unnormalized value of this entity; that is, replacing
|
||||
# &ent; entities.
|
||||
def unnormalized
|
||||
document&.record_entity_expansion
|
||||
|
||||
return nil if @value.nil?
|
||||
|
||||
@unnormalized = Text::unnormalize(@value, parent,
|
||||
entity_expansion_text_limit: document&.entity_expansion_text_limit)
|
||||
end
|
||||
|
||||
#once :unnormalized
|
||||
|
||||
# Returns the value of this entity unprocessed -- raw. This is the
|
||||
# normalized value; that is, with all %ent; and &ent; entities intact
|
||||
def normalized
|
||||
@value
|
||||
end
|
||||
|
||||
# Write out a fully formed, correct entity definition (assuming the Entity
|
||||
# object itself is valid.)
|
||||
#
|
||||
# out::
|
||||
# An object implementing <TT><<</TT> to which the entity will be
|
||||
# output
|
||||
# indent::
|
||||
# *DEPRECATED* and ignored
|
||||
def write out, indent=-1
|
||||
out << '<!ENTITY '
|
||||
out << '% ' if @reference
|
||||
out << @name
|
||||
out << ' '
|
||||
if @external
|
||||
out << @external << ' '
|
||||
if @pubid
|
||||
q = @pubid.include?('"')?"'":'"'
|
||||
out << q << @pubid << q << ' '
|
||||
end
|
||||
q = @ref.include?('"')?"'":'"'
|
||||
out << q << @ref << q
|
||||
out << ' NDATA ' << @ndata if @ndata
|
||||
else
|
||||
q = @value.include?('"')?"'":'"'
|
||||
out << q << @value << q
|
||||
end
|
||||
out << '>'
|
||||
end
|
||||
|
||||
# Returns this entity as a string. See write().
|
||||
def to_s
|
||||
rv = ''
|
||||
write rv
|
||||
rv
|
||||
end
|
||||
end
|
||||
|
||||
# This is a set of entity constants -- the ones defined in the XML
|
||||
# specification. These are +gt+, +lt+, +amp+, +quot+ and +apos+.
|
||||
# CAUTION: these entities does not have parent and document
|
||||
module EntityConst
|
||||
# +>+
|
||||
GT = Entity.new( 'gt', '>' )
|
||||
# +<+
|
||||
LT = Entity.new( 'lt', '<' )
|
||||
# +&+
|
||||
AMP = Entity.new( 'amp', '&' )
|
||||
# +"+
|
||||
QUOT = Entity.new( 'quot', '"' )
|
||||
# +'+
|
||||
APOS = Entity.new( 'apos', "'" )
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,116 @@
|
||||
# frozen_string_literal: false
|
||||
|
||||
module REXML
|
||||
module Formatters
|
||||
class Default
|
||||
# Prints out the XML document with no formatting -- except if ie_hack is
|
||||
# set.
|
||||
#
|
||||
# ie_hack::
|
||||
# If set to true, then inserts whitespace before the close of an empty
|
||||
# tag, so that IE's bad XML parser doesn't choke.
|
||||
def initialize( ie_hack=false )
|
||||
@ie_hack = ie_hack
|
||||
end
|
||||
|
||||
# Writes the node to some output.
|
||||
#
|
||||
# node::
|
||||
# The node to write
|
||||
# output::
|
||||
# A class implementing <TT><<</TT>. Pass in an Output object to
|
||||
# change the output encoding.
|
||||
def write( node, output )
|
||||
case node
|
||||
|
||||
when Document
|
||||
if node.xml_decl.encoding != 'UTF-8' && !output.kind_of?(Output)
|
||||
output = Output.new( output, node.xml_decl.encoding )
|
||||
end
|
||||
write_document( node, output )
|
||||
|
||||
when Element
|
||||
write_element( node, output )
|
||||
|
||||
when Declaration, ElementDecl, NotationDecl, ExternalEntity, Entity,
|
||||
Attribute, AttlistDecl
|
||||
node.write( output,-1 )
|
||||
|
||||
when Instruction
|
||||
write_instruction( node, output )
|
||||
|
||||
when DocType, XMLDecl
|
||||
node.write( output )
|
||||
|
||||
when Comment
|
||||
write_comment( node, output )
|
||||
|
||||
when CData
|
||||
write_cdata( node, output )
|
||||
|
||||
when Text
|
||||
write_text( node, output )
|
||||
|
||||
else
|
||||
raise Exception.new("XML FORMATTING ERROR")
|
||||
|
||||
end
|
||||
end
|
||||
|
||||
protected
|
||||
def write_document( node, output )
|
||||
node.children.each { |child| write( child, output ) }
|
||||
end
|
||||
|
||||
def write_element( node, output )
|
||||
output << "<#{node.expanded_name}"
|
||||
|
||||
node.attributes.to_a.map { |a|
|
||||
Hash === a ? a.values : a
|
||||
}.flatten.sort_by {|attr| attr.name}.each do |attr|
|
||||
output << " "
|
||||
attr.write( output )
|
||||
end unless node.attributes.empty?
|
||||
|
||||
if node.children.empty?
|
||||
output << " " if @ie_hack
|
||||
output << "/"
|
||||
else
|
||||
output << ">"
|
||||
node.children.each { |child|
|
||||
write( child, output )
|
||||
}
|
||||
output << "</#{node.expanded_name}"
|
||||
end
|
||||
output << ">"
|
||||
end
|
||||
|
||||
def write_text( node, output )
|
||||
output << node.to_s()
|
||||
end
|
||||
|
||||
def write_comment( node, output )
|
||||
output << Comment::START
|
||||
output << node.to_s
|
||||
output << Comment::STOP
|
||||
end
|
||||
|
||||
def write_cdata( node, output )
|
||||
output << CData::START
|
||||
output << node.to_s
|
||||
output << CData::STOP
|
||||
end
|
||||
|
||||
def write_instruction( node, output )
|
||||
output << Instruction::START
|
||||
output << node.target
|
||||
content = node.content
|
||||
if content
|
||||
output << ' '
|
||||
output << content
|
||||
end
|
||||
output << Instruction::STOP
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,142 @@
|
||||
# frozen_string_literal: true
|
||||
require_relative 'default'
|
||||
|
||||
module REXML
|
||||
module Formatters
|
||||
# Pretty-prints an XML document. This destroys whitespace in text nodes
|
||||
# and will insert carriage returns and indentations.
|
||||
#
|
||||
# TODO: Add an option to print attributes on new lines
|
||||
class Pretty < Default
|
||||
|
||||
# If compact is set to true, then the formatter will attempt to use as
|
||||
# little space as possible
|
||||
attr_accessor :compact
|
||||
# The width of a page. Used for formatting text
|
||||
attr_accessor :width
|
||||
|
||||
# Create a new pretty printer.
|
||||
#
|
||||
# output::
|
||||
# An object implementing '<<(String)', to which the output will be written.
|
||||
# indentation::
|
||||
# An integer greater than 0. The indentation of each level will be
|
||||
# this number of spaces. If this is < 1, the behavior of this object
|
||||
# is undefined. Defaults to 2.
|
||||
# ie_hack::
|
||||
# If true, the printer will insert whitespace before closing empty
|
||||
# tags, thereby allowing Internet Explorer's XML parser to
|
||||
# function. Defaults to false.
|
||||
def initialize( indentation=2, ie_hack=false )
|
||||
@indentation = indentation
|
||||
@level = 0
|
||||
@ie_hack = ie_hack
|
||||
@width = 80
|
||||
@compact = false
|
||||
end
|
||||
|
||||
protected
|
||||
def write_element(node, output)
|
||||
output << ' '*@level
|
||||
output << "<#{node.expanded_name}"
|
||||
|
||||
node.attributes.each_attribute do |attr|
|
||||
output << " "
|
||||
attr.write( output )
|
||||
end unless node.attributes.empty?
|
||||
|
||||
if node.children.empty?
|
||||
if @ie_hack
|
||||
output << " "
|
||||
end
|
||||
output << "/"
|
||||
else
|
||||
output << ">"
|
||||
# If compact and all children are text, and if the formatted output
|
||||
# is less than the specified width, then try to print everything on
|
||||
# one line
|
||||
skip = false
|
||||
if compact
|
||||
if node.children.inject(true) {|s,c| s & c.kind_of?(Text)}
|
||||
string = +""
|
||||
old_level = @level
|
||||
@level = 0
|
||||
node.children.each { |child| write( child, string ) }
|
||||
@level = old_level
|
||||
if string.length < @width
|
||||
output << string
|
||||
skip = true
|
||||
end
|
||||
end
|
||||
end
|
||||
unless skip
|
||||
output << "\n"
|
||||
@level += @indentation
|
||||
node.children.each { |child|
|
||||
next if child.kind_of?(Text) and child.to_s.strip.length == 0
|
||||
write( child, output )
|
||||
output << "\n"
|
||||
}
|
||||
@level -= @indentation
|
||||
output << ' '*@level
|
||||
end
|
||||
output << "</#{node.expanded_name}"
|
||||
end
|
||||
output << ">"
|
||||
end
|
||||
|
||||
def write_text( node, output )
|
||||
s = node.to_s()
|
||||
s.gsub!(/\s/,' ')
|
||||
s.squeeze!(" ")
|
||||
s = wrap(s, @width - @level)
|
||||
s = indent_text(s, @level, " ", true)
|
||||
output << (' '*@level + s)
|
||||
end
|
||||
|
||||
def write_comment( node, output)
|
||||
output << ' ' * @level
|
||||
super
|
||||
end
|
||||
|
||||
def write_cdata( node, output)
|
||||
output << ' ' * @level
|
||||
super
|
||||
end
|
||||
|
||||
def write_document( node, output )
|
||||
# Ok, this is a bit odd. All XML documents have an XML declaration,
|
||||
# but it may not write itself if the user didn't specifically add it,
|
||||
# either through the API or in the input document. If it doesn't write
|
||||
# itself, then we don't need a carriage return... which makes this
|
||||
# logic more complex.
|
||||
node.children.each { |child|
|
||||
next if child.instance_of?(Text)
|
||||
unless child == node.children[0] or child.instance_of?(Text) or
|
||||
(child == node.children[1] and !node.children[0].writethis)
|
||||
output << "\n"
|
||||
end
|
||||
write( child, output )
|
||||
}
|
||||
end
|
||||
|
||||
private
|
||||
def indent_text(string, level=1, style="\t", indentfirstline=true)
|
||||
return string if level < 0
|
||||
string.gsub(/\n/, "\n#{style*level}")
|
||||
end
|
||||
|
||||
def wrap(string, width)
|
||||
parts = []
|
||||
while string.length > width and place = string.rindex(' ', width)
|
||||
parts << string[0...place]
|
||||
string = string[place+1..-1]
|
||||
end
|
||||
parts << string
|
||||
parts.join("\n")
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative 'pretty'
|
||||
|
||||
module REXML
|
||||
module Formatters
|
||||
# The Transitive formatter writes an XML document that parses to an
|
||||
# identical document as the source document. This means that no extra
|
||||
# whitespace nodes are inserted, and whitespace within text nodes is
|
||||
# preserved. Within these constraints, the document is pretty-printed,
|
||||
# with whitespace inserted into the metadata to introduce formatting.
|
||||
#
|
||||
# Note that this is only useful if the original XML is not already
|
||||
# formatted. Since this formatter does not alter whitespace nodes, the
|
||||
# results of formatting already formatted XML will be odd.
|
||||
class Transitive < Default
|
||||
def initialize( indentation=2, ie_hack=false )
|
||||
@indentation = indentation
|
||||
@level = 0
|
||||
@ie_hack = ie_hack
|
||||
end
|
||||
|
||||
protected
|
||||
def write_element( node, output )
|
||||
output << "<#{node.expanded_name}"
|
||||
|
||||
node.attributes.each_attribute do |attr|
|
||||
output << " "
|
||||
attr.write( output )
|
||||
end unless node.attributes.empty?
|
||||
|
||||
output << "\n"
|
||||
output << ' '*@level
|
||||
if node.children.empty?
|
||||
output << " " if @ie_hack
|
||||
output << "/"
|
||||
else
|
||||
output << ">"
|
||||
# If compact and all children are text, and if the formatted output
|
||||
# is less than the specified width, then try to print everything on
|
||||
# one line
|
||||
@level += @indentation
|
||||
node.children.each { |child|
|
||||
write( child, output )
|
||||
}
|
||||
@level -= @indentation
|
||||
output << "</#{node.expanded_name}"
|
||||
output << "\n"
|
||||
output << ' '*@level
|
||||
end
|
||||
output << ">"
|
||||
end
|
||||
|
||||
def write_text( node, output )
|
||||
output << node.to_s()
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,446 @@
|
||||
# frozen_string_literal: false
|
||||
module REXML
|
||||
# If you add a method, keep in mind two things:
|
||||
# (1) the first argument will always be a list of nodes from which to
|
||||
# filter. In the case of context methods (such as position), the function
|
||||
# should return an array with a value for each child in the array.
|
||||
# (2) all method calls from XML will have "-" replaced with "_".
|
||||
# Therefore, in XML, "local-name()" is identical (and actually becomes)
|
||||
# "local_name()"
|
||||
module Functions
|
||||
@@available_functions = {}
|
||||
@@context = nil
|
||||
@@namespace_context = {}
|
||||
@@variables = {}
|
||||
|
||||
INTERNAL_METHODS = [
|
||||
:namespace_context,
|
||||
:namespace_context=,
|
||||
:variables,
|
||||
:variables=,
|
||||
:context=,
|
||||
:get_namespace,
|
||||
:send,
|
||||
]
|
||||
class << self
|
||||
def singleton_method_added(name)
|
||||
unless INTERNAL_METHODS.include?(name)
|
||||
@@available_functions[name] = true
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
def Functions::namespace_context=(x) ; @@namespace_context=x ; end
|
||||
def Functions::variables=(x) ; @@variables=x ; end
|
||||
def Functions::namespace_context ; @@namespace_context ; end
|
||||
def Functions::variables ; @@variables ; end
|
||||
|
||||
def Functions::context=(value); @@context = value; end
|
||||
|
||||
def Functions::text( )
|
||||
if @@context[:node].node_type == :element
|
||||
@@context[:node].find_all{|n| n.node_type == :text}.collect{|n| n.value}
|
||||
elsif @@context[:node].node_type == :text
|
||||
@@context[:node].value
|
||||
else
|
||||
false
|
||||
end
|
||||
end
|
||||
|
||||
# Returns the last node of the given list of nodes.
|
||||
def Functions::last( )
|
||||
@@context[:size]
|
||||
end
|
||||
|
||||
def Functions::position( )
|
||||
@@context[:index]
|
||||
end
|
||||
|
||||
# Returns the size of the given list of nodes.
|
||||
def Functions::count( node_set )
|
||||
node_set.size
|
||||
end
|
||||
|
||||
# Since REXML is non-validating, this method is not implemented as it
|
||||
# requires a DTD
|
||||
def Functions::id( object )
|
||||
end
|
||||
|
||||
def Functions::local_name(node_set=nil)
|
||||
get_namespace(node_set) do |node|
|
||||
return node.local_name
|
||||
end
|
||||
""
|
||||
end
|
||||
|
||||
def Functions::namespace_uri( node_set=nil )
|
||||
get_namespace( node_set ) {|node| node.namespace}
|
||||
end
|
||||
|
||||
def Functions::name( node_set=nil )
|
||||
get_namespace( node_set ) do |node|
|
||||
node.expanded_name
|
||||
end
|
||||
end
|
||||
|
||||
# Helper method.
|
||||
def Functions::get_namespace( node_set = nil )
|
||||
if node_set == nil
|
||||
yield @@context[:node] if @@context[:node].respond_to?(:namespace)
|
||||
else
|
||||
if node_set.respond_to? :each
|
||||
result = []
|
||||
node_set.each do |node|
|
||||
result << yield(node) if node.respond_to?(:namespace)
|
||||
end
|
||||
result
|
||||
elsif node_set.respond_to? :namespace
|
||||
yield node_set
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# A node-set is converted to a string by returning the string-value of the
|
||||
# node in the node-set that is first in document order. If the node-set is
|
||||
# empty, an empty string is returned.
|
||||
#
|
||||
# A number is converted to a string as follows
|
||||
#
|
||||
# NaN is converted to the string NaN
|
||||
#
|
||||
# positive zero is converted to the string 0
|
||||
#
|
||||
# negative zero is converted to the string 0
|
||||
#
|
||||
# positive infinity is converted to the string Infinity
|
||||
#
|
||||
# negative infinity is converted to the string -Infinity
|
||||
#
|
||||
# if the number is an integer, the number is represented in decimal form
|
||||
# as a Number with no decimal point and no leading zeros, preceded by a
|
||||
# minus sign (-) if the number is negative
|
||||
#
|
||||
# otherwise, the number is represented in decimal form as a Number
|
||||
# including a decimal point with at least one digit before the decimal
|
||||
# point and at least one digit after the decimal point, preceded by a
|
||||
# minus sign (-) if the number is negative; there must be no leading zeros
|
||||
# before the decimal point apart possibly from the one required digit
|
||||
# immediately before the decimal point; beyond the one required digit
|
||||
# after the decimal point there must be as many, but only as many, more
|
||||
# digits as are needed to uniquely distinguish the number from all other
|
||||
# IEEE 754 numeric values.
|
||||
#
|
||||
# The boolean false value is converted to the string false. The boolean
|
||||
# true value is converted to the string true.
|
||||
#
|
||||
# An object of a type other than the four basic types is converted to a
|
||||
# string in a way that is dependent on that type.
|
||||
def Functions::string( object=@@context[:node] )
|
||||
if object.respond_to?(:node_type)
|
||||
case object.node_type
|
||||
when :attribute
|
||||
object.value
|
||||
when :element
|
||||
string_value(object)
|
||||
when :document
|
||||
string_value(object.root)
|
||||
when :processing_instruction
|
||||
object.content
|
||||
else
|
||||
object.to_s
|
||||
end
|
||||
else
|
||||
case object
|
||||
when Array
|
||||
string(object[0])
|
||||
when Float
|
||||
if object.nan?
|
||||
"NaN"
|
||||
else
|
||||
integer = object.to_i
|
||||
if object == integer
|
||||
"%d" % integer
|
||||
else
|
||||
object.to_s
|
||||
end
|
||||
end
|
||||
else
|
||||
object.to_s
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# A node-set is converted to a string by
|
||||
# returning the concatenation of the string-value
|
||||
# of each of the children of the node in the
|
||||
# node-set that is first in document order.
|
||||
# If the node-set is empty, an empty string is returned.
|
||||
def Functions::string_value( o )
|
||||
rv = ""
|
||||
o.children.each { |e|
|
||||
if e.node_type == :text
|
||||
rv << e.to_s
|
||||
elsif e.node_type == :element
|
||||
rv << string_value( e )
|
||||
end
|
||||
}
|
||||
rv
|
||||
end
|
||||
|
||||
def Functions::concat( *objects )
|
||||
concatenated = ""
|
||||
objects.each do |object|
|
||||
concatenated << string(object)
|
||||
end
|
||||
concatenated
|
||||
end
|
||||
|
||||
# Fixed by Mike Stok
|
||||
def Functions::starts_with( string, test )
|
||||
string(string).index(string(test)) == 0
|
||||
end
|
||||
|
||||
# Fixed by Mike Stok
|
||||
def Functions::contains( string, test )
|
||||
string(string).include?(string(test))
|
||||
end
|
||||
|
||||
# Kouhei fixed this
|
||||
def Functions::substring_before( string, test )
|
||||
ruby_string = string(string)
|
||||
ruby_index = ruby_string.index(string(test))
|
||||
if ruby_index.nil?
|
||||
""
|
||||
else
|
||||
ruby_string[ 0...ruby_index ]
|
||||
end
|
||||
end
|
||||
|
||||
# Kouhei fixed this too
|
||||
def Functions::substring_after( string, test )
|
||||
ruby_string = string(string)
|
||||
return $1 if ruby_string =~ /#{test}(.*)/
|
||||
""
|
||||
end
|
||||
|
||||
# Take equal portions of Mike Stok and Sean Russell; mix
|
||||
# vigorously, and pour into a tall, chilled glass. Serves 10,000.
|
||||
def Functions::substring( string, start, length=nil )
|
||||
ruby_string = string(string)
|
||||
ruby_length = if length.nil?
|
||||
ruby_string.length.to_f
|
||||
else
|
||||
number(length)
|
||||
end
|
||||
ruby_start = number(start)
|
||||
|
||||
# Handle the special cases
|
||||
return '' if (
|
||||
ruby_length.nan? or
|
||||
ruby_start.nan? or
|
||||
ruby_start.infinite?
|
||||
)
|
||||
|
||||
infinite_length = ruby_length.infinite? == 1
|
||||
ruby_length = ruby_string.length if infinite_length
|
||||
|
||||
# Now, get the bounds. The XPath bounds are 1..length; the ruby bounds
|
||||
# are 0..length. Therefore, we have to offset the bounds by one.
|
||||
ruby_start = round(ruby_start) - 1
|
||||
ruby_length = round(ruby_length)
|
||||
|
||||
if ruby_start < 0
|
||||
ruby_length += ruby_start unless infinite_length
|
||||
ruby_start = 0
|
||||
end
|
||||
return '' if ruby_length <= 0
|
||||
ruby_string[ruby_start,ruby_length]
|
||||
end
|
||||
|
||||
# UNTESTED
|
||||
def Functions::string_length( string )
|
||||
string(string).length
|
||||
end
|
||||
|
||||
def Functions::normalize_space( string=nil )
|
||||
string = string(@@context[:node]) if string.nil?
|
||||
if string.kind_of? Array
|
||||
string.collect{|x| x.to_s.strip.gsub(/\s+/um, ' ') if x}
|
||||
else
|
||||
string.to_s.strip.gsub(/\s+/um, ' ')
|
||||
end
|
||||
end
|
||||
|
||||
# This is entirely Mike Stok's beast
|
||||
def Functions::translate( string, tr1, tr2 )
|
||||
from = string(tr1)
|
||||
to = string(tr2)
|
||||
|
||||
# the map is our translation table.
|
||||
#
|
||||
# if a character occurs more than once in the
|
||||
# from string then we ignore the second &
|
||||
# subsequent mappings
|
||||
#
|
||||
# if a character maps to nil then we delete it
|
||||
# in the output. This happens if the from
|
||||
# string is longer than the to string
|
||||
#
|
||||
# there's nothing about - or ^ being special in
|
||||
# http://www.w3.org/TR/xpath#function-translate
|
||||
# so we don't build ranges or negated classes
|
||||
|
||||
map = Hash.new
|
||||
0.upto(from.length - 1) { |pos|
|
||||
from_char = from[pos]
|
||||
unless map.has_key? from_char
|
||||
map[from_char] =
|
||||
if pos < to.length
|
||||
to[pos]
|
||||
else
|
||||
nil
|
||||
end
|
||||
end
|
||||
}
|
||||
|
||||
if ''.respond_to? :chars
|
||||
string(string).chars.collect { |c|
|
||||
if map.has_key? c then map[c] else c end
|
||||
}.compact.join
|
||||
else
|
||||
string(string).unpack('U*').collect { |c|
|
||||
if map.has_key? c then map[c] else c end
|
||||
}.compact.pack('U*')
|
||||
end
|
||||
end
|
||||
|
||||
def Functions::boolean(object=@@context[:node])
|
||||
case object
|
||||
when true, false
|
||||
object
|
||||
when Float
|
||||
return false if object.zero?
|
||||
return false if object.nan?
|
||||
true
|
||||
when Numeric
|
||||
not object.zero?
|
||||
when String
|
||||
not object.empty?
|
||||
when Array
|
||||
not object.empty?
|
||||
else
|
||||
object ? true : false
|
||||
end
|
||||
end
|
||||
|
||||
# UNTESTED
|
||||
def Functions::not( object )
|
||||
not boolean( object )
|
||||
end
|
||||
|
||||
# UNTESTED
|
||||
def Functions::true( )
|
||||
true
|
||||
end
|
||||
|
||||
# UNTESTED
|
||||
def Functions::false( )
|
||||
false
|
||||
end
|
||||
|
||||
# UNTESTED
|
||||
def Functions::lang( language )
|
||||
lang = false
|
||||
node = @@context[:node]
|
||||
attr = nil
|
||||
until node.nil?
|
||||
if node.node_type == :element
|
||||
attr = node.attributes["xml:lang"]
|
||||
unless attr.nil?
|
||||
lang = compare_language(string(language), attr)
|
||||
break
|
||||
else
|
||||
end
|
||||
end
|
||||
node = node.parent
|
||||
end
|
||||
lang
|
||||
end
|
||||
|
||||
def Functions::compare_language lang1, lang2
|
||||
lang2.downcase.index(lang1.downcase) == 0
|
||||
end
|
||||
|
||||
# a string that consists of optional whitespace followed by an optional
|
||||
# minus sign followed by a Number followed by whitespace is converted to
|
||||
# the IEEE 754 number that is nearest (according to the IEEE 754
|
||||
# round-to-nearest rule) to the mathematical value represented by the
|
||||
# string; any other string is converted to NaN
|
||||
#
|
||||
# boolean true is converted to 1; boolean false is converted to 0
|
||||
#
|
||||
# a node-set is first converted to a string as if by a call to the string
|
||||
# function and then converted in the same way as a string argument
|
||||
#
|
||||
# an object of a type other than the four basic types is converted to a
|
||||
# number in a way that is dependent on that type
|
||||
def Functions::number(object=@@context[:node])
|
||||
case object
|
||||
when true
|
||||
Float(1)
|
||||
when false
|
||||
Float(0)
|
||||
when Array
|
||||
number(string(object))
|
||||
when Numeric
|
||||
object.to_f
|
||||
else
|
||||
str = string(object)
|
||||
case str.strip
|
||||
when /\A\s*(-?(?:\d+(?:\.\d*)?|\.\d+))\s*\z/
|
||||
$1.to_f
|
||||
else
|
||||
Float::NAN
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
def Functions::sum( nodes )
|
||||
nodes = [nodes] unless nodes.kind_of? Array
|
||||
nodes.inject(0) { |r,n| r + number(string(n)) }
|
||||
end
|
||||
|
||||
def Functions::floor( number )
|
||||
number(number).floor
|
||||
end
|
||||
|
||||
def Functions::ceiling( number )
|
||||
number(number).ceil
|
||||
end
|
||||
|
||||
def Functions::round( number )
|
||||
number = number(number)
|
||||
begin
|
||||
neg = number.negative?
|
||||
number = number.abs.round
|
||||
neg ? -number : number
|
||||
rescue FloatDomainError
|
||||
number
|
||||
end
|
||||
end
|
||||
|
||||
def Functions::processing_instruction( node )
|
||||
node.node_type == :processing_instruction
|
||||
end
|
||||
|
||||
def Functions::send(name, *args)
|
||||
if @@available_functions[name.to_sym]
|
||||
super
|
||||
else
|
||||
# TODO: Maybe, this is not XPath spec behavior.
|
||||
# This behavior must be reconsidered.
|
||||
XPath.match(@@context[:node], name.to_s)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,79 @@
|
||||
# frozen_string_literal: false
|
||||
|
||||
require_relative "child"
|
||||
require_relative "source"
|
||||
|
||||
module REXML
|
||||
# Represents an XML Instruction; IE, <? ... ?>
|
||||
# TODO: Add parent arg (3rd arg) to constructor
|
||||
class Instruction < Child
|
||||
START = "<?"
|
||||
STOP = "?>"
|
||||
|
||||
# target is the "name" of the Instruction; IE, the "tag" in <?tag ...?>
|
||||
# content is everything else.
|
||||
attr_accessor :target, :content
|
||||
|
||||
# Constructs a new Instruction
|
||||
# @param target can be one of a number of things. If String, then
|
||||
# the target of this instruction is set to this. If an Instruction,
|
||||
# then the Instruction is shallowly cloned (target and content are
|
||||
# copied).
|
||||
# @param content Must be either a String, or a Parent. Can only
|
||||
# be a Parent if the target argument is a Source. Otherwise, this
|
||||
# String is set as the content of this instruction.
|
||||
def initialize(target, content=nil)
|
||||
case target
|
||||
when String
|
||||
super()
|
||||
@target = target
|
||||
@content = content
|
||||
when Instruction
|
||||
super(content)
|
||||
@target = target.target
|
||||
@content = target.content
|
||||
else
|
||||
message =
|
||||
"processing instruction target must be String or REXML::Instruction: "
|
||||
message << "<#{target.inspect}>"
|
||||
raise ArgumentError, message
|
||||
end
|
||||
@content.strip! if @content
|
||||
end
|
||||
|
||||
def clone
|
||||
Instruction.new self
|
||||
end
|
||||
|
||||
# == DEPRECATED
|
||||
# See the rexml/formatters package
|
||||
#
|
||||
def write writer, indent=-1, transitive=false, ie_hack=false
|
||||
Kernel.warn( "#{self.class.name}#write is deprecated", uplevel: 1)
|
||||
indent(writer, indent)
|
||||
writer << START
|
||||
writer << @target
|
||||
if @content
|
||||
writer << ' '
|
||||
writer << @content
|
||||
end
|
||||
writer << STOP
|
||||
end
|
||||
|
||||
# @return true if other is an Instruction, and the content and target
|
||||
# of the other matches the target and content of this object.
|
||||
def ==( other )
|
||||
other.kind_of? Instruction and
|
||||
other.target == @target and
|
||||
other.content == @content
|
||||
end
|
||||
|
||||
def node_type
|
||||
:processing_instruction
|
||||
end
|
||||
|
||||
def inspect
|
||||
"<?p-i #{target} ...?>"
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,188 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative '../xmltokens'
|
||||
|
||||
module REXML
|
||||
module Light
|
||||
# Represents a tagged XML element. Elements are characterized by
|
||||
# having children, attributes, and names, and can themselves be
|
||||
# children.
|
||||
class Node
|
||||
NAMESPLIT = /^(?:(#{XMLTokens::NCNAME_STR}):)?(#{XMLTokens::NCNAME_STR})/u
|
||||
PARENTS = [ :element, :document, :doctype ]
|
||||
# Create a new element.
|
||||
def initialize node=nil
|
||||
@node = node
|
||||
if node.kind_of? String
|
||||
node = [ :text, node ]
|
||||
elsif node.nil?
|
||||
node = [ :document, nil, nil ]
|
||||
elsif node[0] == :start_element
|
||||
node[0] = :element
|
||||
elsif node[0] == :start_doctype
|
||||
node[0] = :doctype
|
||||
elsif node[0] == :start_document
|
||||
node[0] = :document
|
||||
end
|
||||
end
|
||||
|
||||
def size
|
||||
if PARENTS.include? @node[0]
|
||||
@node[-1].size
|
||||
else
|
||||
0
|
||||
end
|
||||
end
|
||||
|
||||
def each
|
||||
size.times { |x| yield( at(x+4) ) }
|
||||
end
|
||||
|
||||
def name
|
||||
at(2)
|
||||
end
|
||||
|
||||
def name=( name_str, ns=nil )
|
||||
pfx = ''
|
||||
pfx = "#{prefix(ns)}:" if ns
|
||||
_old_put(2, "#{pfx}#{name_str}")
|
||||
end
|
||||
|
||||
def parent=( node )
|
||||
_old_put(1,node)
|
||||
end
|
||||
|
||||
def local_name
|
||||
namesplit
|
||||
@name
|
||||
end
|
||||
|
||||
def local_name=( name_str )
|
||||
_old_put( 1, "#@prefix:#{name_str}" )
|
||||
end
|
||||
|
||||
def prefix( namespace=nil )
|
||||
prefix_of( self, namespace )
|
||||
end
|
||||
|
||||
def namespace( prefix=prefix() )
|
||||
namespace_of( self, prefix )
|
||||
end
|
||||
|
||||
def namespace=( namespace )
|
||||
@prefix = prefix( namespace )
|
||||
pfx = ''
|
||||
pfx = "#@prefix:" if @prefix.size > 0
|
||||
_old_put(1, "#{pfx}#@name")
|
||||
end
|
||||
|
||||
def []( reference, ns=nil )
|
||||
if reference.kind_of? String
|
||||
pfx = ''
|
||||
pfx = "#{prefix(ns)}:" if ns
|
||||
at(3)["#{pfx}#{reference}"]
|
||||
elsif reference.kind_of? Range
|
||||
_old_get( Range.new(4+reference.begin, reference.end, reference.exclude_end?) )
|
||||
else
|
||||
_old_get( 4+reference )
|
||||
end
|
||||
end
|
||||
|
||||
def =~( path )
|
||||
XPath.match( self, path )
|
||||
end
|
||||
|
||||
# Doesn't handle namespaces yet
|
||||
def []=( reference, ns, value=nil )
|
||||
if reference.kind_of? String
|
||||
value = ns unless value
|
||||
at( 3 )[reference] = value
|
||||
elsif reference.kind_of? Range
|
||||
_old_put( Range.new(3+reference.begin, reference.end, reference.exclude_end?), ns )
|
||||
else
|
||||
if value
|
||||
_old_put( 4+reference, ns, value )
|
||||
else
|
||||
_old_put( 4+reference, ns )
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# Append a child to this element, optionally under a provided namespace.
|
||||
# The namespace argument is ignored if the element argument is an Element
|
||||
# object. Otherwise, the element argument is a string, the namespace (if
|
||||
# provided) is the namespace the element is created in.
|
||||
def << element
|
||||
if node_type() == :text
|
||||
at(-1) << element
|
||||
else
|
||||
newnode = Node.new( element )
|
||||
newnode.parent = self
|
||||
self.push( newnode )
|
||||
end
|
||||
at(-1)
|
||||
end
|
||||
|
||||
def node_type
|
||||
_old_get(0)
|
||||
end
|
||||
|
||||
def text=( foo )
|
||||
replace = at(4).kind_of?(String)? 1 : 0
|
||||
self._old_put(4,replace, normalizefoo)
|
||||
end
|
||||
|
||||
def root
|
||||
context = self
|
||||
context = context.at(1) while context.at(1)
|
||||
end
|
||||
|
||||
def has_name?( name, namespace = '' )
|
||||
at(3) == name and namespace() == namespace
|
||||
end
|
||||
|
||||
def children
|
||||
self
|
||||
end
|
||||
|
||||
def parent
|
||||
at(1)
|
||||
end
|
||||
|
||||
def to_s
|
||||
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def namesplit
|
||||
return if @name.defined?
|
||||
at(2) =~ NAMESPLIT
|
||||
@prefix = '' || $1
|
||||
@name = $2
|
||||
end
|
||||
|
||||
def namespace_of( node, prefix=nil )
|
||||
if not prefix
|
||||
name = at(2)
|
||||
name =~ NAMESPLIT
|
||||
prefix = $1
|
||||
end
|
||||
to_find = 'xmlns'
|
||||
to_find = "xmlns:#{prefix}" if not prefix.nil?
|
||||
ns = at(3)[ to_find ]
|
||||
ns ? ns : namespace_of( @node[0], prefix )
|
||||
end
|
||||
|
||||
def prefix_of( node, namespace=nil )
|
||||
if not namespace
|
||||
name = node.name
|
||||
name =~ NAMESPLIT
|
||||
$1
|
||||
else
|
||||
ns = at(3).find { |k,v| v == namespace }
|
||||
ns ? ns : prefix_of( node.parent, namespace )
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,63 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
require_relative 'xmltokens'
|
||||
|
||||
module REXML
|
||||
# Adds named attributes to an object.
|
||||
module Namespace
|
||||
# The name of the object, valid if set
|
||||
attr_reader :name, :expanded_name
|
||||
# The expanded name of the object, valid if name is set
|
||||
attr_accessor :prefix
|
||||
include XMLTokens
|
||||
NAME_WITHOUT_NAMESPACE = /\A#{NCNAME_STR}\z/
|
||||
NAMESPLIT = /^(?:(#{NCNAME_STR}):)?(#{NCNAME_STR})/u
|
||||
|
||||
# Sets the name and the expanded name
|
||||
def name=( name )
|
||||
@expanded_name = name
|
||||
if name.match?(NAME_WITHOUT_NAMESPACE)
|
||||
@prefix = ""
|
||||
@namespace = ""
|
||||
@name = name
|
||||
elsif name =~ NAMESPLIT
|
||||
if $1
|
||||
@prefix = $1
|
||||
else
|
||||
@prefix = ""
|
||||
@namespace = ""
|
||||
end
|
||||
@name = $2
|
||||
elsif name == ""
|
||||
@prefix = nil
|
||||
@namespace = nil
|
||||
@name = nil
|
||||
else
|
||||
message = "name must be \#{PREFIX}:\#{LOCAL_NAME} or \#{LOCAL_NAME}: "
|
||||
message += "<#{name.inspect}>"
|
||||
raise ArgumentError, message
|
||||
end
|
||||
end
|
||||
|
||||
# Compares names optionally WITH namespaces
|
||||
def has_name?( other, ns=nil )
|
||||
if ns
|
||||
namespace() == ns and name() == other
|
||||
elsif other.include? ":"
|
||||
fully_expanded_name == other
|
||||
else
|
||||
name == other
|
||||
end
|
||||
end
|
||||
|
||||
alias :local_name :name
|
||||
|
||||
# Fully expand the name, even if the prefix wasn't specified in the
|
||||
# source file.
|
||||
def fully_expanded_name
|
||||
ns = prefix
|
||||
return "#{ns}:#@name" if ns.size > 0
|
||||
@name
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,80 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "parseexception"
|
||||
require_relative "formatters/pretty"
|
||||
require_relative "formatters/default"
|
||||
|
||||
module REXML
|
||||
# Represents a node in the tree. Nodes are never encountered except as
|
||||
# superclasses of other objects. Nodes have siblings.
|
||||
module Node
|
||||
# @return the next sibling (nil if unset)
|
||||
def next_sibling_node
|
||||
return nil if @parent.nil?
|
||||
@parent[ @parent.index(self) + 1 ]
|
||||
end
|
||||
|
||||
# @return the previous sibling (nil if unset)
|
||||
def previous_sibling_node
|
||||
return nil if @parent.nil?
|
||||
ind = @parent.index(self)
|
||||
return nil if ind == 0
|
||||
@parent[ ind - 1 ]
|
||||
end
|
||||
|
||||
# indent::
|
||||
# *DEPRECATED* This parameter is now ignored. See the formatters in the
|
||||
# REXML::Formatters package for changing the output style.
|
||||
def to_s indent=nil
|
||||
unless indent.nil?
|
||||
Kernel.warn( "#{self.class.name}#to_s(indent) parameter is deprecated", uplevel: 1)
|
||||
f = REXML::Formatters::Pretty.new( indent )
|
||||
f.write( self, rv = "" )
|
||||
else
|
||||
f = REXML::Formatters::Default.new
|
||||
f.write( self, rv = "" )
|
||||
end
|
||||
return rv
|
||||
end
|
||||
|
||||
def indent to, ind
|
||||
if @parent and @parent.context and not @parent.context[:indentstyle].nil? then
|
||||
indentstyle = @parent.context[:indentstyle]
|
||||
else
|
||||
indentstyle = ' '
|
||||
end
|
||||
to << indentstyle*ind unless ind<1
|
||||
end
|
||||
|
||||
def parent?
|
||||
false;
|
||||
end
|
||||
|
||||
|
||||
# Visit all subnodes of +self+ recursively
|
||||
def each_recursive(&block) # :yields: node
|
||||
stack = []
|
||||
each { |child| stack.unshift child if child.node_type == :element }
|
||||
until stack.empty?
|
||||
child = stack.pop
|
||||
yield child
|
||||
n = stack.size
|
||||
child.each { |grandchild| stack.insert n, grandchild if grandchild.node_type == :element }
|
||||
end
|
||||
end
|
||||
|
||||
# Find (and return) first subnode (recursively) for which the block
|
||||
# evaluates to true. Returns +nil+ if none was found.
|
||||
def find_first_recursive(&block) # :yields: node
|
||||
each_recursive {|node|
|
||||
return node if block.call(node)
|
||||
}
|
||||
nil
|
||||
end
|
||||
|
||||
# Returns the position that +self+ holds in its parent's array, indexed
|
||||
# from 1.
|
||||
def index_in_parent
|
||||
parent.index(self)+1
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,30 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative 'encoding'
|
||||
|
||||
module REXML
|
||||
class Output
|
||||
include Encoding
|
||||
|
||||
attr_reader :encoding
|
||||
|
||||
def initialize real_IO, encd="iso-8859-1"
|
||||
@output = real_IO
|
||||
self.encoding = encd
|
||||
|
||||
@to_utf = encoding != 'UTF-8'
|
||||
|
||||
if encoding == "UTF-16"
|
||||
@output << "\ufeff".encode("UTF-16BE")
|
||||
self.encoding = "UTF-16BE"
|
||||
end
|
||||
end
|
||||
|
||||
def <<( content )
|
||||
@output << (@to_utf ? self.encode(content) : content)
|
||||
end
|
||||
|
||||
def to_s
|
||||
"Output[#{encoding}]"
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,166 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "child"
|
||||
|
||||
module REXML
|
||||
# A parent has children, and has methods for accessing them. The Parent
|
||||
# class is never encountered except as the superclass for some other
|
||||
# object.
|
||||
class Parent < Child
|
||||
include Enumerable
|
||||
|
||||
# Constructor
|
||||
# @param parent if supplied, will be set as the parent of this object
|
||||
def initialize parent=nil
|
||||
super(parent)
|
||||
@children = []
|
||||
end
|
||||
|
||||
def add( object )
|
||||
object.parent = self
|
||||
@children << object
|
||||
object
|
||||
end
|
||||
|
||||
alias :push :add
|
||||
alias :<< :push
|
||||
|
||||
def unshift( object )
|
||||
object.parent = self
|
||||
@children.unshift object
|
||||
end
|
||||
|
||||
def delete( object )
|
||||
found = false
|
||||
@children.delete_if {|c| c.equal?(object) and found = true }
|
||||
object.parent = nil if found
|
||||
found ? object : nil
|
||||
end
|
||||
|
||||
def each(&block)
|
||||
@children.each(&block)
|
||||
end
|
||||
|
||||
def delete_if( &block )
|
||||
@children.delete_if(&block)
|
||||
end
|
||||
|
||||
def delete_at( index )
|
||||
@children.delete_at index
|
||||
end
|
||||
|
||||
def each_index( &block )
|
||||
@children.each_index(&block)
|
||||
end
|
||||
|
||||
# Fetches a child at a given index
|
||||
# @param index the Integer index of the child to fetch
|
||||
def []( index )
|
||||
@children[index]
|
||||
end
|
||||
|
||||
alias :each_child :each
|
||||
|
||||
|
||||
|
||||
# Set an index entry. See Array.[]=
|
||||
# @param index the index of the element to set
|
||||
# @param opt either the object to set, or an Integer length
|
||||
# @param child if opt is an Integer, this is the child to set
|
||||
# @return the parent (self)
|
||||
def []=( *args )
|
||||
args[-1].parent = self
|
||||
@children[*args[0..-2]] = args[-1]
|
||||
end
|
||||
|
||||
# Inserts an child before another child
|
||||
# @param child1 this is either an xpath or an Element. If an Element,
|
||||
# child2 will be inserted before child1 in the child list of the parent.
|
||||
# If an xpath, child2 will be inserted before the first child to match
|
||||
# the xpath.
|
||||
# @param child2 the child to insert
|
||||
# @return the parent (self)
|
||||
def insert_before( child1, child2 )
|
||||
if child1.kind_of? String
|
||||
child1 = XPath.first( self, child1 )
|
||||
child1.parent.insert_before child1, child2
|
||||
else
|
||||
ind = index(child1)
|
||||
child2.parent.delete(child2) if child2.parent
|
||||
@children[ind,0] = child2
|
||||
child2.parent = self
|
||||
end
|
||||
self
|
||||
end
|
||||
|
||||
# Inserts an child after another child
|
||||
# @param child1 this is either an xpath or an Element. If an Element,
|
||||
# child2 will be inserted after child1 in the child list of the parent.
|
||||
# If an xpath, child2 will be inserted after the first child to match
|
||||
# the xpath.
|
||||
# @param child2 the child to insert
|
||||
# @return the parent (self)
|
||||
def insert_after( child1, child2 )
|
||||
if child1.kind_of? String
|
||||
child1 = XPath.first( self, child1 )
|
||||
child1.parent.insert_after child1, child2
|
||||
else
|
||||
ind = index(child1)+1
|
||||
child2.parent.delete(child2) if child2.parent
|
||||
@children[ind,0] = child2
|
||||
child2.parent = self
|
||||
end
|
||||
self
|
||||
end
|
||||
|
||||
def to_a
|
||||
@children.dup
|
||||
end
|
||||
|
||||
# Fetches the index of a given child
|
||||
# @param child the child to get the index of
|
||||
# @return the index of the child, or nil if the object is not a child
|
||||
# of this parent.
|
||||
def index( child )
|
||||
count = -1
|
||||
@children.find { |i| count += 1 ; i.hash == child.hash }
|
||||
count
|
||||
end
|
||||
|
||||
# @return the number of children of this parent
|
||||
def size
|
||||
@children.size
|
||||
end
|
||||
|
||||
alias :length :size
|
||||
|
||||
# Replaces one child with another, making sure the nodelist is correct
|
||||
# @param to_replace the child to replace (must be a Child)
|
||||
# @param replacement the child to insert into the nodelist (must be a
|
||||
# Child)
|
||||
def replace_child( to_replace, replacement )
|
||||
@children.map! {|c| c.equal?( to_replace ) ? replacement : c }
|
||||
to_replace.parent = nil
|
||||
replacement.parent = self
|
||||
end
|
||||
|
||||
# Deeply clones this object. This creates a complete duplicate of this
|
||||
# Parent, including all descendants.
|
||||
def deep_clone
|
||||
cl = clone()
|
||||
each do |child|
|
||||
if child.kind_of? Parent
|
||||
cl << child.deep_clone
|
||||
else
|
||||
cl << child.clone
|
||||
end
|
||||
end
|
||||
cl
|
||||
end
|
||||
|
||||
alias :children :to_a
|
||||
|
||||
def parent?
|
||||
true
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,53 @@
|
||||
# frozen_string_literal: false
|
||||
module REXML
|
||||
class ParseException < RuntimeError
|
||||
attr_accessor :source, :parser, :continued_exception
|
||||
|
||||
def initialize( message, source=nil, parser=nil, exception=nil )
|
||||
super(message)
|
||||
@source = source
|
||||
@parser = parser
|
||||
@continued_exception = exception
|
||||
end
|
||||
|
||||
def to_s
|
||||
# Quote the original exception, if there was one
|
||||
if @continued_exception
|
||||
err = @continued_exception.inspect
|
||||
err << "\n"
|
||||
err << @continued_exception.backtrace.join("\n")
|
||||
err << "\n...\n"
|
||||
else
|
||||
err = ""
|
||||
end
|
||||
|
||||
# Get the stack trace and error message
|
||||
err << super
|
||||
|
||||
# Add contextual information
|
||||
if @source
|
||||
err << "\nLine: #{line}\n"
|
||||
err << "Position: #{position}\n"
|
||||
err << "Last 80 unconsumed characters:\n"
|
||||
err.force_encoding("ASCII-8BIT")
|
||||
err << @source.buffer[0..80].force_encoding("ASCII-8BIT").gsub(/\n/, ' ')
|
||||
end
|
||||
|
||||
err
|
||||
end
|
||||
|
||||
def position
|
||||
@source.current_line[0] if @source and defined? @source.current_line and
|
||||
@source.current_line
|
||||
end
|
||||
|
||||
def line
|
||||
@source.current_line[2] if @source and defined? @source.current_line and
|
||||
@source.current_line
|
||||
end
|
||||
|
||||
def context
|
||||
@source.current_line
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,949 @@
|
||||
# frozen_string_literal: true
|
||||
require_relative '../parseexception'
|
||||
require_relative '../undefinednamespaceexception'
|
||||
require_relative '../security'
|
||||
require_relative '../source'
|
||||
require 'set'
|
||||
require "strscan"
|
||||
|
||||
module REXML
|
||||
module Parsers
|
||||
unless [].respond_to?(:tally)
|
||||
module EnumerableTally
|
||||
refine Enumerable do
|
||||
def tally
|
||||
counts = {}
|
||||
each do |item|
|
||||
counts[item] ||= 0
|
||||
counts[item] += 1
|
||||
end
|
||||
counts
|
||||
end
|
||||
end
|
||||
end
|
||||
using EnumerableTally
|
||||
end
|
||||
|
||||
if StringScanner::Version < "3.0.8"
|
||||
module StringScannerCaptures
|
||||
refine StringScanner do
|
||||
def captures
|
||||
values_at(*(1...size))
|
||||
end
|
||||
end
|
||||
end
|
||||
using StringScannerCaptures
|
||||
end
|
||||
|
||||
# = Using the Pull Parser
|
||||
# <em>This API is experimental, and subject to change.</em>
|
||||
# parser = PullParser.new( "<a>text<b att='val'/>txet</a>" )
|
||||
# while parser.has_next?
|
||||
# res = parser.next
|
||||
# puts res[1]['att'] if res.start_tag? and res[0] == 'b'
|
||||
# end
|
||||
# See the PullEvent class for information on the content of the results.
|
||||
# The data is identical to the arguments passed for the various events to
|
||||
# the StreamListener API.
|
||||
#
|
||||
# Notice that:
|
||||
# parser = PullParser.new( "<a>BAD DOCUMENT" )
|
||||
# while parser.has_next?
|
||||
# res = parser.next
|
||||
# raise res[1] if res.error?
|
||||
# end
|
||||
#
|
||||
# Nat Price gave me some good ideas for the API.
|
||||
class BaseParser
|
||||
LETTER = '[:alpha:]'
|
||||
DIGIT = '[:digit:]'
|
||||
|
||||
COMBININGCHAR = '' # TODO
|
||||
EXTENDER = '' # TODO
|
||||
|
||||
NCNAME_STR= "[#{LETTER}_][-[:alnum:]._#{COMBININGCHAR}#{EXTENDER}]*"
|
||||
QNAME_STR= "(?:(#{NCNAME_STR}):)?(#{NCNAME_STR})"
|
||||
QNAME = /(#{QNAME_STR})/
|
||||
|
||||
# Just for backward compatibility. For example, kramdown uses this.
|
||||
# It's not used in REXML.
|
||||
UNAME_STR= "(?:#{NCNAME_STR}:)?#{NCNAME_STR}"
|
||||
|
||||
NAMECHAR = '[\-\w\.:]'
|
||||
NAME = "([\\w:]#{NAMECHAR}*)"
|
||||
NMTOKEN = "(?:#{NAMECHAR})+"
|
||||
NMTOKENS = "#{NMTOKEN}(\\s+#{NMTOKEN})*"
|
||||
REFERENCE = "&(?:#{NAME};|#\\d+;|#x[0-9a-fA-F]+;)"
|
||||
REFERENCE_RE = /#{REFERENCE}/
|
||||
|
||||
DOCTYPE_START = /\A\s*<!DOCTYPE\s/um
|
||||
DOCTYPE_END = /\A\s*\]\s*>/um
|
||||
ATTRIBUTE_PATTERN = /\s*(#{QNAME_STR})\s*=\s*(["'])(.*?)\4/um
|
||||
COMMENT_START = /\A<!--/u
|
||||
COMMENT_PATTERN = /<!--(.*?)-->/um
|
||||
CDATA_START = /\A<!\[CDATA\[/u
|
||||
CDATA_END = /\A\s*\]\s*>/um
|
||||
CDATA_PATTERN = /<!\[CDATA\[(.*?)\]\]>/um
|
||||
XMLDECL_START = /\A<\?xml\s/u;
|
||||
XMLDECL_PATTERN = /<\?xml\s+(.*?)\?>/um
|
||||
INSTRUCTION_START = /\A<\?/u
|
||||
INSTRUCTION_PATTERN = /<\?#{NAME}(\s+.*?)?\?>/um
|
||||
TAG_MATCH = /\A<((?>#{QNAME_STR}))/um
|
||||
CLOSE_MATCH = /\A\s*<\/(#{QNAME_STR})\s*>/um
|
||||
|
||||
VERSION = /\bversion\s*=\s*["'](.*?)['"]/um
|
||||
ENCODING = /\bencoding\s*=\s*["'](.*?)['"]/um
|
||||
STANDALONE = /\bstandalone\s*=\s*["'](.*?)['"]/um
|
||||
|
||||
ENTITY_START = /\A\s*<!ENTITY/
|
||||
ELEMENTDECL_START = /\A\s*<!ELEMENT/um
|
||||
ELEMENTDECL_PATTERN = /\A\s*(<!ELEMENT.*?)>/um
|
||||
SYSTEMENTITY = /\A\s*(%.*?;)\s*$/um
|
||||
ENUMERATION = "\\(\\s*#{NMTOKEN}(?:\\s*\\|\\s*#{NMTOKEN})*\\s*\\)"
|
||||
NOTATIONTYPE = "NOTATION\\s+\\(\\s*#{NAME}(?:\\s*\\|\\s*#{NAME})*\\s*\\)"
|
||||
ENUMERATEDTYPE = "(?:(?:#{NOTATIONTYPE})|(?:#{ENUMERATION}))"
|
||||
ATTTYPE = "(CDATA|ID|IDREF|IDREFS|ENTITY|ENTITIES|NMTOKEN|NMTOKENS|#{ENUMERATEDTYPE})"
|
||||
ATTVALUE = "(?:\"((?:[^<&\"]|#{REFERENCE})*)\")|(?:'((?:[^<&']|#{REFERENCE})*)')"
|
||||
DEFAULTDECL = "(#REQUIRED|#IMPLIED|(?:(#FIXED\\s+)?#{ATTVALUE}))"
|
||||
ATTDEF = "\\s+#{NAME}\\s+#{ATTTYPE}\\s+#{DEFAULTDECL}"
|
||||
ATTDEF_RE = /#{ATTDEF}/
|
||||
ATTLISTDECL_START = /\A\s*<!ATTLIST/um
|
||||
ATTLISTDECL_PATTERN = /\A\s*<!ATTLIST\s+#{NAME}(?:#{ATTDEF})*\s*>/um
|
||||
|
||||
TEXT_PATTERN = /\A([^<]*)/um
|
||||
|
||||
# Entity constants
|
||||
PUBIDCHAR = "\x20\x0D\x0Aa-zA-Z0-9\\-()+,./:=?;!*@$_%#"
|
||||
SYSTEMLITERAL = %Q{((?:"[^"]*")|(?:'[^']*'))}
|
||||
PUBIDLITERAL = %Q{("[#{PUBIDCHAR}']*"|'[#{PUBIDCHAR}]*')}
|
||||
EXTERNALID = "(?:(?:(SYSTEM)\\s+#{SYSTEMLITERAL})|(?:(PUBLIC)\\s+#{PUBIDLITERAL}\\s+#{SYSTEMLITERAL}))"
|
||||
NDATADECL = "\\s+NDATA\\s+#{NAME}"
|
||||
PEREFERENCE = "%#{NAME};"
|
||||
ENTITYVALUE = %Q{((?:"(?:[^%&"]|#{PEREFERENCE}|#{REFERENCE})*")|(?:'([^%&']|#{PEREFERENCE}|#{REFERENCE})*'))}
|
||||
PEDEF = "(?:#{ENTITYVALUE}|#{EXTERNALID})"
|
||||
ENTITYDEF = "(?:#{ENTITYVALUE}|(?:#{EXTERNALID}(#{NDATADECL})?))"
|
||||
PEDECL = "<!ENTITY\\s+(%)\\s+#{NAME}\\s+#{PEDEF}\\s*>"
|
||||
GEDECL = "<!ENTITY\\s+#{NAME}\\s+#{ENTITYDEF}\\s*>"
|
||||
ENTITYDECL = /\s*(?:#{GEDECL})|\s*(?:#{PEDECL})/um
|
||||
|
||||
NOTATIONDECL_START = /\A\s*<!NOTATION/um
|
||||
EXTERNAL_ID_PUBLIC = /\A\s*PUBLIC\s+#{PUBIDLITERAL}\s+#{SYSTEMLITERAL}\s*/um
|
||||
EXTERNAL_ID_SYSTEM = /\A\s*SYSTEM\s+#{SYSTEMLITERAL}\s*/um
|
||||
PUBLIC_ID = /\A\s*PUBLIC\s+#{PUBIDLITERAL}\s*/um
|
||||
|
||||
EREFERENCE = /&(?!#{NAME};)/
|
||||
|
||||
DEFAULT_ENTITIES = {
|
||||
'gt' => [/>/, '>', '>', />/],
|
||||
'lt' => [/</, '<', '<', /</],
|
||||
'quot' => [/"/, '"', '"', /"/],
|
||||
"apos" => [/'/, "'", "'", /'/]
|
||||
}
|
||||
|
||||
module Private
|
||||
PEREFERENCE_PATTERN = /#{PEREFERENCE}/um
|
||||
TAG_PATTERN = /((?>#{QNAME_STR}))\s*/um
|
||||
CLOSE_PATTERN = /(#{QNAME_STR})\s*>/um
|
||||
EQUAL_PATTERN = /\s*=\s*/um
|
||||
ATTLISTDECL_END = /\s+#{NAME}(?:#{ATTDEF})*\s*>/um
|
||||
NAME_PATTERN = /#{NAME}/um
|
||||
GEDECL_PATTERN = "\\s+#{NAME}\\s+#{ENTITYDEF}\\s*>"
|
||||
PEDECL_PATTERN = "\\s+(%)\\s+#{NAME}\\s+#{PEDEF}\\s*>"
|
||||
ENTITYDECL_PATTERN = /(?:#{GEDECL_PATTERN})|(?:#{PEDECL_PATTERN})/um
|
||||
CARRIAGE_RETURN_NEWLINE_PATTERN = /\r\n?/
|
||||
CHARACTER_REFERENCES = /&#((?:\d+)|(?:x[a-fA-F0-9]+));/
|
||||
DEFAULT_ENTITIES_PATTERNS = {}
|
||||
default_entities = ['gt', 'lt', 'quot', 'apos', 'amp']
|
||||
default_entities.each do |term|
|
||||
DEFAULT_ENTITIES_PATTERNS[term] = /&#{term};/
|
||||
end
|
||||
XML_PREFIXED_NAMESPACE = "http://www.w3.org/XML/1998/namespace"
|
||||
end
|
||||
private_constant :Private
|
||||
|
||||
def initialize( source )
|
||||
self.stream = source
|
||||
@listeners = []
|
||||
@prefixes = Set.new
|
||||
@entity_expansion_count = 0
|
||||
@entity_expansion_limit = Security.entity_expansion_limit
|
||||
@entity_expansion_text_limit = Security.entity_expansion_text_limit
|
||||
@source.ensure_buffer
|
||||
@version = nil
|
||||
end
|
||||
|
||||
def add_listener( listener )
|
||||
@listeners << listener
|
||||
end
|
||||
|
||||
attr_reader :source
|
||||
attr_reader :entity_expansion_count
|
||||
attr_writer :entity_expansion_limit
|
||||
attr_writer :entity_expansion_text_limit
|
||||
|
||||
def stream=( source )
|
||||
@source = SourceFactory.create_from( source )
|
||||
reset
|
||||
end
|
||||
|
||||
def reset
|
||||
@closed = nil
|
||||
@have_root = false
|
||||
@document_status = nil
|
||||
@tags = []
|
||||
@stack = []
|
||||
@entities = []
|
||||
@namespaces = {"xml" => Private::XML_PREFIXED_NAMESPACE}
|
||||
@namespaces_restore_stack = []
|
||||
end
|
||||
|
||||
def position
|
||||
if @source.respond_to? :position
|
||||
@source.position
|
||||
else
|
||||
# FIXME
|
||||
0
|
||||
end
|
||||
end
|
||||
|
||||
# Returns true if there are no more events
|
||||
def empty?
|
||||
(@source.empty? and @stack.empty?)
|
||||
end
|
||||
|
||||
# Returns true if there are more events. Synonymous with !empty?
|
||||
def has_next?
|
||||
!(@source.empty? and @stack.empty?)
|
||||
end
|
||||
|
||||
# Push an event back on the head of the stream. This method
|
||||
# has (theoretically) infinite depth.
|
||||
def unshift token
|
||||
@stack.unshift(token)
|
||||
end
|
||||
|
||||
# Peek at the +depth+ event in the stack. The first element on the stack
|
||||
# is at depth 0. If +depth+ is -1, will parse to the end of the input
|
||||
# stream and return the last event, which is always :end_document.
|
||||
# Be aware that this causes the stream to be parsed up to the +depth+
|
||||
# event, so you can effectively pre-parse the entire document (pull the
|
||||
# entire thing into memory) using this method.
|
||||
def peek depth=0
|
||||
raise %Q[Illegal argument "#{depth}"] if depth < -1
|
||||
temp = []
|
||||
if depth == -1
|
||||
temp.push(pull()) until empty?
|
||||
else
|
||||
while @stack.size+temp.size < depth+1
|
||||
temp.push(pull())
|
||||
end
|
||||
end
|
||||
@stack += temp if temp.size > 0
|
||||
@stack[depth]
|
||||
end
|
||||
|
||||
# Returns the next event. This is a +PullEvent+ object.
|
||||
def pull
|
||||
@source.drop_parsed_content
|
||||
|
||||
pull_event.tap do |event|
|
||||
@listeners.each do |listener|
|
||||
listener.receive event
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
def pull_event
|
||||
if @closed
|
||||
x, @closed = @closed, nil
|
||||
return [ :end_element, x ]
|
||||
end
|
||||
if empty?
|
||||
if @document_status == :in_doctype
|
||||
raise ParseException.new("Malformed DOCTYPE: unclosed", @source)
|
||||
end
|
||||
unless @tags.empty?
|
||||
path = "/" + @tags.join("/")
|
||||
raise ParseException.new("Missing end tag for '#{path}'", @source)
|
||||
end
|
||||
|
||||
unless @document_status == :in_element
|
||||
raise ParseException.new("Malformed XML: No root element", @source)
|
||||
end
|
||||
|
||||
return [ :end_document ]
|
||||
end
|
||||
return @stack.shift if @stack.size > 0
|
||||
#STDERR.puts @source.encoding
|
||||
#STDERR.puts "BUFFER = #{@source.buffer.inspect}"
|
||||
|
||||
@source.ensure_buffer
|
||||
if @document_status == nil
|
||||
start_position = @source.position
|
||||
if @source.match?("<?", true)
|
||||
return process_instruction
|
||||
elsif @source.match?("<!", true)
|
||||
if @source.match?("--", true)
|
||||
return [ :comment, process_comment ]
|
||||
elsif @source.match?("DOCTYPE", true)
|
||||
base_error_message = "Malformed DOCTYPE"
|
||||
unless @source.skip_spaces
|
||||
if @source.match?(">")
|
||||
message = "#{base_error_message}: name is missing"
|
||||
else
|
||||
message = "#{base_error_message}: invalid name"
|
||||
end
|
||||
@source.position = start_position
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
name = parse_name(base_error_message)
|
||||
@source.skip_spaces
|
||||
if @source.match?("[", true)
|
||||
id = [nil, nil, nil]
|
||||
@document_status = :in_doctype
|
||||
elsif @source.match?(">", true)
|
||||
id = [nil, nil, nil]
|
||||
@document_status = :after_doctype
|
||||
@source.ensure_buffer
|
||||
else
|
||||
id = parse_id(base_error_message,
|
||||
accept_external_id: true,
|
||||
accept_public_id: false)
|
||||
if id[0] == "SYSTEM"
|
||||
# For backward compatibility
|
||||
id[1], id[2] = id[2], nil
|
||||
end
|
||||
@source.skip_spaces
|
||||
if @source.match?("[", true)
|
||||
@document_status = :in_doctype
|
||||
elsif @source.match?(">", true)
|
||||
@document_status = :after_doctype
|
||||
@source.ensure_buffer
|
||||
else
|
||||
message = "#{base_error_message}: garbage after external ID"
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
end
|
||||
args = [:start_doctype, name, *id]
|
||||
if @document_status == :after_doctype
|
||||
@source.skip_spaces
|
||||
@stack << [ :end_doctype ]
|
||||
end
|
||||
return args
|
||||
else
|
||||
message = "Invalid XML"
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
end
|
||||
end
|
||||
if @document_status == :in_doctype
|
||||
@source.skip_spaces
|
||||
start_position = @source.position
|
||||
if @source.match?("<!", true)
|
||||
if @source.match?("ELEMENT", true)
|
||||
md = @source.match(/(.*?)>/um, true)
|
||||
raise REXML::ParseException.new( "Bad ELEMENT declaration!", @source ) if md.nil?
|
||||
return [ :elementdecl, "<!ELEMENT" + md[1] ]
|
||||
elsif @source.match?("ENTITY", true)
|
||||
match_data = @source.match(Private::ENTITYDECL_PATTERN, true)
|
||||
unless match_data
|
||||
raise REXML::ParseException.new("Malformed entity declaration", @source)
|
||||
end
|
||||
match = [:entitydecl, *match_data.captures.compact]
|
||||
ref = false
|
||||
if match[1] == '%'
|
||||
ref = true
|
||||
match.delete_at 1
|
||||
end
|
||||
# Now we have to sort out what kind of entity reference this is
|
||||
if match[2] == 'SYSTEM'
|
||||
# External reference
|
||||
match[3] = match[3][1..-2] # PUBID
|
||||
match.delete_at(4) if match.size > 4 # Chop out NDATA decl
|
||||
# match is [ :entity, name, SYSTEM, pubid(, ndata)? ]
|
||||
elsif match[2] == 'PUBLIC'
|
||||
# External reference
|
||||
match[3] = match[3][1..-2] # PUBID
|
||||
match[4] = match[4][1..-2] # HREF
|
||||
match.delete_at(5) if match.size > 5 # Chop out NDATA decl
|
||||
# match is [ :entity, name, PUBLIC, pubid, href(, ndata)? ]
|
||||
elsif Private::PEREFERENCE_PATTERN.match?(match[2])
|
||||
raise REXML::ParseException.new("Parameter entity references forbidden in internal subset: #{match[2]}", @source)
|
||||
else
|
||||
match[2] = match[2][1..-2]
|
||||
match.pop if match.size == 4
|
||||
# match is [ :entity, name, value ]
|
||||
end
|
||||
match << '%' if ref
|
||||
return match
|
||||
elsif @source.match?("ATTLIST", true)
|
||||
md = @source.match(Private::ATTLISTDECL_END, true)
|
||||
raise REXML::ParseException.new( "Bad ATTLIST declaration!", @source ) if md.nil?
|
||||
element = md[1]
|
||||
contents = "<!ATTLIST" + md[0]
|
||||
|
||||
pairs = {}
|
||||
values = md[0].strip.scan( ATTDEF_RE )
|
||||
values.each do |attdef|
|
||||
unless attdef[3] == "#IMPLIED"
|
||||
attdef.compact!
|
||||
val = attdef[3]
|
||||
val = attdef[4] if val == "#FIXED "
|
||||
pairs[attdef[0]] = val
|
||||
if attdef[0] =~ /^xmlns:(.*)/
|
||||
@namespaces[$1] = val
|
||||
end
|
||||
end
|
||||
end
|
||||
return [ :attlistdecl, element, pairs, contents ]
|
||||
elsif @source.match?("NOTATION", true)
|
||||
base_error_message = "Malformed notation declaration"
|
||||
unless @source.skip_spaces
|
||||
if @source.match?(">")
|
||||
message = "#{base_error_message}: name is missing"
|
||||
else
|
||||
message = "#{base_error_message}: invalid name"
|
||||
end
|
||||
@source.position = start_position
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
name = parse_name(base_error_message)
|
||||
id = parse_id(base_error_message,
|
||||
accept_external_id: true,
|
||||
accept_public_id: true)
|
||||
@source.skip_spaces
|
||||
unless @source.match?(">", true)
|
||||
message = "#{base_error_message}: garbage before end >"
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
return [:notationdecl, name, *id]
|
||||
elsif @source.match?("--", true)
|
||||
return [ :comment, process_comment ]
|
||||
else
|
||||
raise REXML::ParseException.new("Malformed node: Started with '<!' but not a comment nor ELEMENT,ENTITY,ATTLIST,NOTATION", @source)
|
||||
end
|
||||
elsif match = @source.match(/(%.*?;)\s*/um, true)
|
||||
return [ :externalentity, match[1] ]
|
||||
elsif @source.match?(/\]\s*>/um, true)
|
||||
@document_status = :after_doctype
|
||||
return [ :end_doctype ]
|
||||
else
|
||||
raise ParseException.new("Malformed DOCTYPE: invalid declaration", @source)
|
||||
end
|
||||
end
|
||||
if @document_status == :after_doctype
|
||||
@source.skip_spaces
|
||||
end
|
||||
begin
|
||||
start_position = @source.position
|
||||
if @source.match?("<", true)
|
||||
# :text's read_until may remain only "<" in buffer. In the
|
||||
# case, buffer is empty here. So we need to fill buffer
|
||||
# here explicitly.
|
||||
@source.ensure_buffer
|
||||
if @source.match?("/", true)
|
||||
@namespaces_restore_stack.pop
|
||||
last_tag = @tags.pop
|
||||
md = @source.match(Private::CLOSE_PATTERN, true)
|
||||
if md and !last_tag
|
||||
message = "Unexpected top-level end tag (got '#{md[1]}')"
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
if md.nil? or last_tag != md[1]
|
||||
message = "Missing end tag for '#{last_tag}'"
|
||||
message += " (got '#{md[1]}')" if md
|
||||
@source.position = start_position if md.nil?
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
return [ :end_element, last_tag ]
|
||||
elsif @source.match?("!", true)
|
||||
#STDERR.puts "SOURCE BUFFER = #{source.buffer}, #{source.buffer.size}"
|
||||
if @source.match?("--", true)
|
||||
return [ :comment, process_comment ]
|
||||
elsif @source.match?("[CDATA[", true)
|
||||
text = @source.read_until("]]>")
|
||||
if text.chomp!("]]>")
|
||||
return [ :cdata, text ]
|
||||
else
|
||||
raise REXML::ParseException.new("Malformed CDATA: Missing end ']]>'", @source)
|
||||
end
|
||||
else
|
||||
raise REXML::ParseException.new("Malformed node: Started with '<!' but not a comment nor CDATA", @source)
|
||||
end
|
||||
elsif @source.match?("?", true)
|
||||
return process_instruction
|
||||
else
|
||||
# Get the next tag
|
||||
md = @source.match(Private::TAG_PATTERN, true)
|
||||
unless md
|
||||
@source.position = start_position
|
||||
raise REXML::ParseException.new("malformed XML: missing tag start", @source)
|
||||
end
|
||||
tag = md[1]
|
||||
@document_status = :in_element
|
||||
@prefixes.clear
|
||||
@prefixes << md[2] if md[2]
|
||||
push_namespaces_restore
|
||||
attributes, closed = parse_attributes(@prefixes)
|
||||
# Verify that all of the prefixes have been defined
|
||||
for prefix in @prefixes
|
||||
unless @namespaces.key?(prefix)
|
||||
raise UndefinedNamespaceException.new(prefix,@source,self)
|
||||
end
|
||||
end
|
||||
|
||||
if closed
|
||||
@closed = tag
|
||||
pop_namespaces_restore
|
||||
else
|
||||
if @tags.empty? and @have_root
|
||||
raise ParseException.new("Malformed XML: Extra tag at the end of the document (got '<#{tag}')", @source)
|
||||
end
|
||||
@tags.push( tag )
|
||||
end
|
||||
@have_root = true
|
||||
return [ :start_element, tag, attributes ]
|
||||
end
|
||||
else
|
||||
text = @source.read_until("<")
|
||||
if text.chomp!("<")
|
||||
@source.position -= "<".bytesize
|
||||
end
|
||||
if @tags.empty?
|
||||
unless /\A\s*\z/.match?(text)
|
||||
if @have_root
|
||||
raise ParseException.new("Malformed XML: Extra content at the end of the document (got '#{text}')", @source)
|
||||
else
|
||||
raise ParseException.new("Malformed XML: Content at the start of the document (got '#{text}')", @source)
|
||||
end
|
||||
end
|
||||
return pull_event if @have_root
|
||||
end
|
||||
return [ :text, text ]
|
||||
end
|
||||
rescue REXML::UndefinedNamespaceException
|
||||
raise
|
||||
rescue REXML::ParseException
|
||||
raise
|
||||
rescue => error
|
||||
raise REXML::ParseException.new( "Exception parsing",
|
||||
@source, self, (error ? error : $!) )
|
||||
end
|
||||
# NOTE: The end of the method never runs, because it is unreachable.
|
||||
# All branches of code above have explicit unconditional return or raise statements.
|
||||
end
|
||||
private :pull_event
|
||||
|
||||
def entity( reference, entities )
|
||||
return unless entities
|
||||
|
||||
value = entities[ reference ]
|
||||
return if value.nil?
|
||||
|
||||
record_entity_expansion
|
||||
unnormalize( value, entities )
|
||||
end
|
||||
|
||||
# Escapes all possible entities
|
||||
def normalize( input, entities=nil, entity_filter=nil )
|
||||
copy = input.clone
|
||||
# Doing it like this rather than in a loop improves the speed
|
||||
copy.gsub!( EREFERENCE, '&' )
|
||||
entities.each do |key, value|
|
||||
copy.gsub!( value, "&#{key};" ) unless entity_filter and
|
||||
entity_filter.include?(entity)
|
||||
end if entities
|
||||
copy.gsub!( EREFERENCE, '&' )
|
||||
DEFAULT_ENTITIES.each do |key, value|
|
||||
copy.gsub!( value[3], value[1] )
|
||||
end
|
||||
copy
|
||||
end
|
||||
|
||||
# Unescapes all possible entities
|
||||
def unnormalize( string, entities=nil, filter=nil )
|
||||
if string.include?("\r")
|
||||
rv = string.gsub( Private::CARRIAGE_RETURN_NEWLINE_PATTERN, "\n" )
|
||||
else
|
||||
rv = string.dup
|
||||
end
|
||||
matches = rv.scan( REFERENCE_RE )
|
||||
return rv if matches.size == 0
|
||||
rv.gsub!( Private::CHARACTER_REFERENCES ) {
|
||||
m=$1
|
||||
if m.start_with?("x")
|
||||
code_point = Integer(m[1..-1], 16)
|
||||
else
|
||||
code_point = Integer(m, 10)
|
||||
end
|
||||
[code_point].pack('U*')
|
||||
}
|
||||
matches.collect!{|x|x[0]}.compact!
|
||||
if filter
|
||||
matches.reject! do |entity_reference|
|
||||
filter.include?(entity_reference)
|
||||
end
|
||||
end
|
||||
if matches.size > 0
|
||||
matches.tally.each do |entity_reference, n|
|
||||
entity_expansion_count_before = @entity_expansion_count
|
||||
entity_value = entity( entity_reference, entities )
|
||||
if entity_value
|
||||
if n > 1
|
||||
entity_expansion_count_delta =
|
||||
@entity_expansion_count - entity_expansion_count_before
|
||||
record_entity_expansion(entity_expansion_count_delta * (n - 1))
|
||||
end
|
||||
re = Private::DEFAULT_ENTITIES_PATTERNS[entity_reference] || /&#{entity_reference};/
|
||||
rv.gsub!( re, entity_value )
|
||||
if rv.bytesize > @entity_expansion_text_limit
|
||||
raise "entity expansion has grown too large"
|
||||
end
|
||||
else
|
||||
er = DEFAULT_ENTITIES[entity_reference]
|
||||
rv.gsub!( er[0], er[2] ) if er
|
||||
end
|
||||
end
|
||||
rv.gsub!( Private::DEFAULT_ENTITIES_PATTERNS['amp'], '&' )
|
||||
end
|
||||
rv
|
||||
end
|
||||
|
||||
private
|
||||
def add_namespace(prefix, uri)
|
||||
@namespaces_restore_stack.last[prefix] = @namespaces[prefix]
|
||||
if uri.nil?
|
||||
@namespaces.delete(prefix)
|
||||
else
|
||||
@namespaces[prefix] = uri
|
||||
end
|
||||
end
|
||||
|
||||
def push_namespaces_restore
|
||||
namespaces_restore = {}
|
||||
@namespaces_restore_stack.push(namespaces_restore)
|
||||
namespaces_restore
|
||||
end
|
||||
|
||||
def pop_namespaces_restore
|
||||
namespaces_restore = @namespaces_restore_stack.pop
|
||||
namespaces_restore.each do |prefix, uri|
|
||||
if uri.nil?
|
||||
@namespaces.delete(prefix)
|
||||
else
|
||||
@namespaces[prefix] = uri
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
def record_entity_expansion(delta=1)
|
||||
@entity_expansion_count += delta
|
||||
if @entity_expansion_count > @entity_expansion_limit
|
||||
raise "number of entity expansions exceeded, processing aborted."
|
||||
end
|
||||
end
|
||||
|
||||
def need_source_encoding_update?(xml_declaration_encoding)
|
||||
return false if xml_declaration_encoding.nil?
|
||||
return false if /\AUTF-16\z/i =~ xml_declaration_encoding
|
||||
true
|
||||
end
|
||||
|
||||
def normalize_xml_declaration_encoding(xml_declaration_encoding)
|
||||
/\AUTF-16(?:BE|LE)\z/i.match?(xml_declaration_encoding) ? "UTF-16" : nil
|
||||
end
|
||||
|
||||
def parse_name(base_error_message)
|
||||
md = @source.match(Private::NAME_PATTERN, true)
|
||||
unless md
|
||||
if @source.match?(/\S/um)
|
||||
message = "#{base_error_message}: invalid name"
|
||||
else
|
||||
message = "#{base_error_message}: name is missing"
|
||||
end
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
md[0]
|
||||
end
|
||||
|
||||
def parse_id(base_error_message,
|
||||
accept_external_id:,
|
||||
accept_public_id:)
|
||||
if accept_external_id and (md = @source.match(EXTERNAL_ID_PUBLIC, true))
|
||||
pubid = system = nil
|
||||
pubid_literal = md[1]
|
||||
pubid = pubid_literal[1..-2] if pubid_literal # Remove quote
|
||||
system_literal = md[2]
|
||||
system = system_literal[1..-2] if system_literal # Remove quote
|
||||
["PUBLIC", pubid, system]
|
||||
elsif accept_public_id and (md = @source.match(PUBLIC_ID, true))
|
||||
pubid = system = nil
|
||||
pubid_literal = md[1]
|
||||
pubid = pubid_literal[1..-2] if pubid_literal # Remove quote
|
||||
["PUBLIC", pubid, nil]
|
||||
elsif accept_external_id and (md = @source.match(EXTERNAL_ID_SYSTEM, true))
|
||||
system = nil
|
||||
system_literal = md[1]
|
||||
system = system_literal[1..-2] if system_literal # Remove quote
|
||||
["SYSTEM", nil, system]
|
||||
else
|
||||
details = parse_id_invalid_details(accept_external_id: accept_external_id,
|
||||
accept_public_id: accept_public_id)
|
||||
message = "#{base_error_message}: #{details}"
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
end
|
||||
|
||||
def parse_id_invalid_details(accept_external_id:,
|
||||
accept_public_id:)
|
||||
public = /\A\s*PUBLIC/um
|
||||
system = /\A\s*SYSTEM/um
|
||||
if (accept_external_id or accept_public_id) and @source.match?(/#{public}/um)
|
||||
if @source.match?(/#{public}(?:\s+[^'"]|\s*[\[>])/um)
|
||||
return "public ID literal is missing"
|
||||
end
|
||||
unless @source.match?(/#{public}\s+#{PUBIDLITERAL}/um)
|
||||
return "invalid public ID literal"
|
||||
end
|
||||
if accept_public_id
|
||||
if @source.match?(/#{public}\s+#{PUBIDLITERAL}\s+[^'"]/um)
|
||||
return "system ID literal is missing"
|
||||
end
|
||||
unless @source.match?(/#{public}\s+#{PUBIDLITERAL}\s+#{SYSTEMLITERAL}/um)
|
||||
return "invalid system literal"
|
||||
end
|
||||
"garbage after system literal"
|
||||
else
|
||||
"garbage after public ID literal"
|
||||
end
|
||||
elsif accept_external_id and @source.match?(/#{system}/um)
|
||||
if @source.match?(/#{system}(?:\s+[^'"]|\s*[\[>])/um)
|
||||
return "system literal is missing"
|
||||
end
|
||||
unless @source.match?(/#{system}\s+#{SYSTEMLITERAL}/um)
|
||||
return "invalid system literal"
|
||||
end
|
||||
"garbage after system literal"
|
||||
else
|
||||
unless @source.match?(/\A\s*(?:PUBLIC|SYSTEM)\s/um)
|
||||
return "invalid ID type"
|
||||
end
|
||||
"ID type is missing"
|
||||
end
|
||||
end
|
||||
|
||||
def process_comment
|
||||
text = @source.read_until("-->")
|
||||
unless text.chomp!("-->")
|
||||
raise REXML::ParseException.new("Unclosed comment: Missing end '-->'", @source)
|
||||
end
|
||||
|
||||
if text.include? "--" or text.end_with?("-")
|
||||
raise REXML::ParseException.new("Malformed comment", @source)
|
||||
end
|
||||
text
|
||||
end
|
||||
|
||||
def process_instruction
|
||||
name = parse_name("Malformed XML: Invalid processing instruction node")
|
||||
if name == "xml"
|
||||
xml_declaration
|
||||
else # PITarget
|
||||
if @source.skip_spaces # e.g. <?name content?>
|
||||
start_position = @source.position
|
||||
content = @source.read_until("?>")
|
||||
unless content.chomp!("?>")
|
||||
@source.position = start_position
|
||||
raise ParseException.new("Malformed XML: Unclosed processing instruction: <#{name}>", @source)
|
||||
end
|
||||
else # e.g. <?name?>
|
||||
content = nil
|
||||
unless @source.match?("?>", true)
|
||||
raise ParseException.new("Malformed XML: Unclosed processing instruction: <#{name}>", @source)
|
||||
end
|
||||
end
|
||||
[:processing_instruction, name, content]
|
||||
end
|
||||
end
|
||||
|
||||
def xml_declaration
|
||||
unless @version.nil?
|
||||
raise ParseException.new("Malformed XML: XML declaration is duplicated", @source)
|
||||
end
|
||||
if @document_status
|
||||
raise ParseException.new("Malformed XML: XML declaration is not at the start", @source)
|
||||
end
|
||||
unless @source.skip_spaces
|
||||
raise ParseException.new("Malformed XML: XML declaration misses spaces before version", @source)
|
||||
end
|
||||
unless @source.match?("version", true)
|
||||
raise ParseException.new("Malformed XML: XML declaration misses version", @source)
|
||||
end
|
||||
@version = parse_attribute_value_with_equal("xml")
|
||||
unless @source.skip_spaces
|
||||
unless @source.match?("?>", true)
|
||||
raise ParseException.new("Malformed XML: Unclosed XML declaration", @source)
|
||||
end
|
||||
encoding = normalize_xml_declaration_encoding(@source.encoding)
|
||||
return [ :xmldecl, @version, encoding, nil ] # e.g. <?xml version="1.0"?>
|
||||
end
|
||||
|
||||
if @source.match?("encoding", true)
|
||||
encoding = parse_attribute_value_with_equal("xml")
|
||||
unless @source.skip_spaces
|
||||
unless @source.match?("?>", true)
|
||||
raise ParseException.new("Malformed XML: Unclosed XML declaration", @source)
|
||||
end
|
||||
if need_source_encoding_update?(encoding)
|
||||
@source.encoding = encoding
|
||||
end
|
||||
encoding ||= normalize_xml_declaration_encoding(@source.encoding)
|
||||
return [ :xmldecl, @version, encoding, nil ] # e.g. <?xml version="1.1" encoding="UTF-8"?>
|
||||
end
|
||||
end
|
||||
|
||||
if @source.match?("standalone", true)
|
||||
standalone = parse_attribute_value_with_equal("xml")
|
||||
case standalone
|
||||
when "yes", "no"
|
||||
else
|
||||
raise ParseException.new("Malformed XML: XML declaration standalone is not yes or no : <#{standalone}>", @source)
|
||||
end
|
||||
end
|
||||
@source.skip_spaces
|
||||
unless @source.match?("?>", true)
|
||||
raise ParseException.new("Malformed XML: Unclosed XML declaration", @source)
|
||||
end
|
||||
|
||||
if need_source_encoding_update?(encoding)
|
||||
@source.encoding = encoding
|
||||
end
|
||||
encoding ||= normalize_xml_declaration_encoding(@source.encoding)
|
||||
|
||||
# e.g. <?xml version="1.0" ?>
|
||||
# <?xml version="1.1" encoding="UTF-8" ?>
|
||||
# <?xml version="1.1" standalone="yes"?>
|
||||
# <?xml version="1.1" encoding="UTF-8" standalone="yes" ?>
|
||||
[ :xmldecl, @version, encoding, standalone ]
|
||||
end
|
||||
|
||||
if StringScanner::Version < "3.1.1"
|
||||
def scan_quote
|
||||
@source.match(/(['"])/, true)&.[](1)
|
||||
end
|
||||
else
|
||||
def scan_quote
|
||||
case @source.peek_byte
|
||||
when 34 # '"'.ord
|
||||
@source.scan_byte
|
||||
'"'
|
||||
when 39 # "'".ord
|
||||
@source.scan_byte
|
||||
"'"
|
||||
else
|
||||
nil
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
def parse_attribute_value_with_equal(name)
|
||||
unless @source.match?(Private::EQUAL_PATTERN, true)
|
||||
message = "Missing attribute equal: <#{name}>"
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
unless quote = scan_quote
|
||||
message = "Missing attribute value start quote: <#{name}>"
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
start_position = @source.position
|
||||
value = @source.read_until(quote)
|
||||
unless value.chomp!(quote)
|
||||
@source.position = start_position
|
||||
message = "Missing attribute value end quote: <#{name}>: <#{quote}>"
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
value
|
||||
end
|
||||
|
||||
def parse_attributes(prefixes)
|
||||
attributes = {}
|
||||
expanded_names = {}
|
||||
closed = false
|
||||
while true
|
||||
if @source.match?(">", true)
|
||||
return attributes, closed
|
||||
elsif @source.match?("/>", true)
|
||||
closed = true
|
||||
return attributes, closed
|
||||
elsif match = @source.match(QNAME, true)
|
||||
name = match[1]
|
||||
prefix = match[2]
|
||||
local_part = match[3]
|
||||
value = parse_attribute_value_with_equal(name)
|
||||
@source.skip_spaces
|
||||
if prefix == "xmlns"
|
||||
if local_part == "xml"
|
||||
if value != Private::XML_PREFIXED_NAMESPACE
|
||||
msg = "The 'xml' prefix must not be bound to any other namespace "+
|
||||
"(http://www.w3.org/TR/REC-xml-names/#ns-decl)"
|
||||
raise REXML::ParseException.new( msg, @source, self )
|
||||
end
|
||||
elsif local_part == "xmlns"
|
||||
msg = "The 'xmlns' prefix must not be declared "+
|
||||
"(http://www.w3.org/TR/REC-xml-names/#ns-decl)"
|
||||
raise REXML::ParseException.new( msg, @source, self)
|
||||
end
|
||||
add_namespace(local_part, value)
|
||||
elsif prefix
|
||||
prefixes << prefix unless prefix == "xml"
|
||||
end
|
||||
|
||||
if attributes[name]
|
||||
msg = "Duplicate attribute #{name.inspect}"
|
||||
raise REXML::ParseException.new(msg, @source, self)
|
||||
end
|
||||
|
||||
unless prefix == "xmlns"
|
||||
uri = @namespaces[prefix]
|
||||
expanded_name = [uri, local_part]
|
||||
existing_prefix = expanded_names[expanded_name]
|
||||
if existing_prefix
|
||||
message = "Namespace conflict in adding attribute " +
|
||||
"\"#{local_part}\": " +
|
||||
"Prefix \"#{existing_prefix}\" = \"#{uri}\" and " +
|
||||
"prefix \"#{prefix}\" = \"#{uri}\""
|
||||
raise REXML::ParseException.new(message, @source, self)
|
||||
end
|
||||
expanded_names[expanded_name] = prefix
|
||||
end
|
||||
|
||||
attributes[name] = value
|
||||
else
|
||||
message = "Invalid attribute name: <#{@source.buffer.split(%r{[/>\s]}).first}>"
|
||||
raise REXML::ParseException.new(message, @source)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
=begin
|
||||
case event[0]
|
||||
when :start_element
|
||||
when :text
|
||||
when :end_element
|
||||
when :processing_instruction
|
||||
when :cdata
|
||||
when :comment
|
||||
when :xmldecl
|
||||
when :start_doctype
|
||||
when :end_doctype
|
||||
when :externalentity
|
||||
when :elementdecl
|
||||
when :entity
|
||||
when :attlistdecl
|
||||
when :notationdecl
|
||||
when :end_doctype
|
||||
end
|
||||
=end
|
||||
@@ -0,0 +1,59 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative 'streamparser'
|
||||
require_relative 'baseparser'
|
||||
require_relative '../light/node'
|
||||
|
||||
module REXML
|
||||
module Parsers
|
||||
class LightParser
|
||||
def initialize stream
|
||||
@stream = stream
|
||||
@parser = REXML::Parsers::BaseParser.new( stream )
|
||||
end
|
||||
|
||||
def add_listener( listener )
|
||||
@parser.add_listener( listener )
|
||||
end
|
||||
|
||||
def rewind
|
||||
@stream.rewind
|
||||
@parser.stream = @stream
|
||||
end
|
||||
|
||||
def parse
|
||||
root = context = [ :document ]
|
||||
while true
|
||||
event = @parser.pull
|
||||
case event[0]
|
||||
when :end_document
|
||||
break
|
||||
when :start_element, :start_doctype
|
||||
new_node = event
|
||||
context << new_node
|
||||
new_node[1,0] = [context]
|
||||
context = new_node
|
||||
when :end_element, :end_doctype
|
||||
context = context[1]
|
||||
else
|
||||
new_node = event
|
||||
context << new_node
|
||||
new_node[1,0] = [context]
|
||||
end
|
||||
end
|
||||
root
|
||||
end
|
||||
end
|
||||
|
||||
# An element is an array. The array contains:
|
||||
# 0 The parent element
|
||||
# 1 The tag name
|
||||
# 2 A hash of attributes
|
||||
# 3..-1 The child elements
|
||||
# An element is an array of size > 3
|
||||
# Text is a String
|
||||
# PIs are [ :processing_instruction, target, data ]
|
||||
# Comments are [ :comment, data ]
|
||||
# DocTypes are DocType structs
|
||||
# The root is an array with XMLDecls, Text, DocType, Array, Text
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,213 @@
|
||||
# frozen_string_literal: false
|
||||
require 'forwardable'
|
||||
|
||||
require_relative '../parseexception'
|
||||
require_relative 'baseparser'
|
||||
require_relative '../xmltokens'
|
||||
|
||||
module REXML
|
||||
module Parsers
|
||||
# = Using the Pull Parser
|
||||
# <em>This API is experimental, and subject to change.</em>
|
||||
# parser = PullParser.new( "<a>text<b att='val'/>txet</a>" )
|
||||
# while parser.has_next?
|
||||
# res = parser.next
|
||||
# puts res[1]['att'] if res.start_tag? and res[0] == 'b'
|
||||
# end
|
||||
# See the PullEvent class for information on the content of the results.
|
||||
# The data is identical to the arguments passed for the various events to
|
||||
# the StreamListener API.
|
||||
#
|
||||
# Notice that:
|
||||
# parser = PullParser.new( "<a>BAD DOCUMENT" )
|
||||
# while parser.has_next?
|
||||
# res = parser.next
|
||||
# raise res[1] if res.error?
|
||||
# end
|
||||
#
|
||||
# Nat Price gave me some good ideas for the API.
|
||||
class PullParser
|
||||
include XMLTokens
|
||||
extend Forwardable
|
||||
|
||||
def_delegators( :@parser, :has_next? )
|
||||
def_delegators( :@parser, :entity )
|
||||
def_delegators( :@parser, :empty? )
|
||||
def_delegators( :@parser, :source )
|
||||
|
||||
def initialize stream
|
||||
@entities = {}
|
||||
@listeners = nil
|
||||
@parser = BaseParser.new( stream )
|
||||
@my_stack = []
|
||||
end
|
||||
|
||||
def add_listener( listener )
|
||||
@listeners = [] unless @listeners
|
||||
@listeners << listener
|
||||
end
|
||||
|
||||
def entity_expansion_count
|
||||
@parser.entity_expansion_count
|
||||
end
|
||||
|
||||
def entity_expansion_limit=( limit )
|
||||
@parser.entity_expansion_limit = limit
|
||||
end
|
||||
|
||||
def entity_expansion_text_limit=( limit )
|
||||
@parser.entity_expansion_text_limit = limit
|
||||
end
|
||||
|
||||
def each
|
||||
while has_next?
|
||||
yield self.pull
|
||||
end
|
||||
end
|
||||
|
||||
def peek depth=0
|
||||
if @my_stack.length <= depth
|
||||
(depth - @my_stack.length + 1).times {
|
||||
e = PullEvent.new(@parser.pull)
|
||||
@my_stack.push(e)
|
||||
}
|
||||
end
|
||||
@my_stack[depth]
|
||||
end
|
||||
|
||||
def pull
|
||||
return @my_stack.shift if @my_stack.length > 0
|
||||
|
||||
event = @parser.pull
|
||||
case event[0]
|
||||
when :entitydecl
|
||||
@entities[ event[1] ] =
|
||||
event[2] unless event[2] =~ /PUBLIC|SYSTEM/
|
||||
when :text
|
||||
unnormalized = @parser.unnormalize( event[1], @entities )
|
||||
event << unnormalized
|
||||
end
|
||||
PullEvent.new( event )
|
||||
end
|
||||
|
||||
def unshift token
|
||||
@my_stack.unshift token
|
||||
end
|
||||
|
||||
def reset
|
||||
@parser.reset
|
||||
end
|
||||
end
|
||||
|
||||
# A parsing event. The contents of the event are accessed as an +Array?,
|
||||
# and the type is given either by the ...? methods, or by accessing the
|
||||
# +type+ accessor. The contents of this object vary from event to event,
|
||||
# but are identical to the arguments passed to +StreamListener+s for each
|
||||
# event.
|
||||
class PullEvent
|
||||
# The type of this event. Will be one of :tag_start, :tag_end, :text,
|
||||
# :processing_instruction, :comment, :doctype, :attlistdecl, :entitydecl,
|
||||
# :notationdecl, :entity, :cdata, :xmldecl, or :error.
|
||||
def initialize(arg)
|
||||
@contents = arg
|
||||
end
|
||||
|
||||
def []( start, endd=nil)
|
||||
if start.kind_of? Range
|
||||
@contents.slice( start.begin+1 .. start.end )
|
||||
elsif start.kind_of? Numeric
|
||||
if endd.nil?
|
||||
@contents.slice( start+1 )
|
||||
else
|
||||
@contents.slice( start+1, endd )
|
||||
end
|
||||
else
|
||||
raise "Illegal argument #{start.inspect} (#{start.class})"
|
||||
end
|
||||
end
|
||||
|
||||
def event_type
|
||||
@contents[0]
|
||||
end
|
||||
|
||||
# Content: [ String tag_name, Hash attributes ]
|
||||
def start_element?
|
||||
@contents[0] == :start_element
|
||||
end
|
||||
|
||||
# Content: [ String tag_name ]
|
||||
def end_element?
|
||||
@contents[0] == :end_element
|
||||
end
|
||||
|
||||
# Content: [ String raw_text, String unnormalized_text ]
|
||||
def text?
|
||||
@contents[0] == :text
|
||||
end
|
||||
|
||||
# Content: [ String text ]
|
||||
def instruction?
|
||||
@contents[0] == :processing_instruction
|
||||
end
|
||||
|
||||
# Content: [ String text ]
|
||||
def comment?
|
||||
@contents[0] == :comment
|
||||
end
|
||||
|
||||
# Content: [ String name, String pub_sys, String long_name, String uri ]
|
||||
def doctype?
|
||||
@contents[0] == :start_doctype
|
||||
end
|
||||
|
||||
# Content: [ String text ]
|
||||
def attlistdecl?
|
||||
@contents[0] == :attlistdecl
|
||||
end
|
||||
|
||||
# Content: [ String text ]
|
||||
def elementdecl?
|
||||
@contents[0] == :elementdecl
|
||||
end
|
||||
|
||||
# Due to the wonders of DTDs, an entity declaration can be just about
|
||||
# anything. There's no way to normalize it; you'll have to interpret the
|
||||
# content yourself. However, the following is true:
|
||||
#
|
||||
# * If the entity declaration is an internal entity:
|
||||
# [ String name, String value ]
|
||||
# Content: [ String text ]
|
||||
def entitydecl?
|
||||
@contents[0] == :entitydecl
|
||||
end
|
||||
|
||||
# Content: [ String text ]
|
||||
def notationdecl?
|
||||
@contents[0] == :notationdecl
|
||||
end
|
||||
|
||||
# Content: [ String text ]
|
||||
def entity?
|
||||
@contents[0] == :entity
|
||||
end
|
||||
|
||||
# Content: [ String text ]
|
||||
def cdata?
|
||||
@contents[0] == :cdata
|
||||
end
|
||||
|
||||
# Content: [ String version, String encoding, String standalone ]
|
||||
def xmldecl?
|
||||
@contents[0] == :xmldecl
|
||||
end
|
||||
|
||||
def error?
|
||||
@contents[0] == :error
|
||||
end
|
||||
|
||||
def inspect
|
||||
@contents[0].to_s + ": " + @contents[1..-1].inspect
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,270 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative 'baseparser'
|
||||
require_relative '../parseexception'
|
||||
require_relative '../namespace'
|
||||
require_relative '../text'
|
||||
|
||||
module REXML
|
||||
module Parsers
|
||||
# SAX2Parser
|
||||
class SAX2Parser
|
||||
def initialize source
|
||||
@parser = BaseParser.new(source)
|
||||
@listeners = []
|
||||
@procs = []
|
||||
@namespace_stack = []
|
||||
@has_listeners = false
|
||||
@tag_stack = []
|
||||
@entities = {}
|
||||
end
|
||||
|
||||
def source
|
||||
@parser.source
|
||||
end
|
||||
|
||||
def entity_expansion_count
|
||||
@parser.entity_expansion_count
|
||||
end
|
||||
|
||||
def entity_expansion_limit=( limit )
|
||||
@parser.entity_expansion_limit = limit
|
||||
end
|
||||
|
||||
def entity_expansion_text_limit=( limit )
|
||||
@parser.entity_expansion_text_limit = limit
|
||||
end
|
||||
|
||||
def add_listener( listener )
|
||||
@parser.add_listener( listener )
|
||||
end
|
||||
|
||||
# Listen arguments:
|
||||
#
|
||||
# Symbol, Array, Block
|
||||
# Listen to Symbol events on Array elements
|
||||
# Symbol, Block
|
||||
# Listen to Symbol events
|
||||
# Array, Listener
|
||||
# Listen to all events on Array elements
|
||||
# Array, Block
|
||||
# Listen to :start_element events on Array elements
|
||||
# Listener
|
||||
# Listen to All events
|
||||
#
|
||||
# Symbol can be one of: :start_element, :end_element,
|
||||
# :start_prefix_mapping, :end_prefix_mapping, :characters,
|
||||
# :processing_instruction, :doctype, :attlistdecl, :elementdecl,
|
||||
# :entitydecl, :notationdecl, :cdata, :xmldecl, :comment
|
||||
#
|
||||
# There is an additional symbol that can be listened for: :progress.
|
||||
# This will be called for every event generated, passing in the current
|
||||
# stream position.
|
||||
#
|
||||
# Array contains regular expressions or strings which will be matched
|
||||
# against fully qualified element names.
|
||||
#
|
||||
# Listener must implement the methods in SAX2Listener
|
||||
#
|
||||
# Block will be passed the same arguments as a SAX2Listener method would
|
||||
# be, where the method name is the same as the matched Symbol.
|
||||
# See the SAX2Listener for more information.
|
||||
def listen( *args, &blok )
|
||||
if args[0].kind_of? Symbol
|
||||
if args.size == 2
|
||||
args[1].each { |match| @procs << [args[0], match, blok] }
|
||||
else
|
||||
add( [args[0], nil, blok] )
|
||||
end
|
||||
elsif args[0].kind_of? Array
|
||||
if args.size == 2
|
||||
args[0].each { |match| add( [nil, match, args[1]] ) }
|
||||
else
|
||||
args[0].each { |match| add( [ :start_element, match, blok ] ) }
|
||||
end
|
||||
else
|
||||
add([nil, nil, args[0]])
|
||||
end
|
||||
end
|
||||
|
||||
def deafen( listener=nil, &blok )
|
||||
if listener
|
||||
@listeners.delete_if {|item| item[-1] == listener }
|
||||
@has_listeners = false if @listeners.size == 0
|
||||
else
|
||||
@procs.delete_if {|item| item[-1] == blok }
|
||||
end
|
||||
end
|
||||
|
||||
def parse
|
||||
@procs.each { |sym,match,block| block.call if sym == :start_document }
|
||||
@listeners.each { |sym,match,block|
|
||||
block.start_document if sym == :start_document or sym.nil?
|
||||
}
|
||||
context = []
|
||||
while true
|
||||
event = @parser.pull
|
||||
case event[0]
|
||||
when :end_document
|
||||
handle( :end_document )
|
||||
break
|
||||
when :start_doctype
|
||||
handle( :doctype, *event[1..-1])
|
||||
when :end_doctype
|
||||
context = context[1]
|
||||
when :start_element
|
||||
@tag_stack.push(event[1])
|
||||
# find the observers for namespaces
|
||||
procs = get_procs( :start_prefix_mapping, event[1] )
|
||||
listeners = get_listeners( :start_prefix_mapping, event[1] )
|
||||
if procs or listeners
|
||||
# break out the namespace declarations
|
||||
# The attributes live in event[2]
|
||||
event[2].each {|n, v| event[2][n] = @parser.normalize(v)}
|
||||
nsdecl = event[2].find_all { |n, value| n =~ /^xmlns(:|$)/ }
|
||||
nsdecl.collect! { |n, value| [ n[6..-1], value ] }
|
||||
@namespace_stack.push({})
|
||||
nsdecl.each do |n,v|
|
||||
@namespace_stack[-1][n] = v
|
||||
# notify observers of namespaces
|
||||
procs.each { |ob| ob.call( n, v ) } if procs
|
||||
listeners.each { |ob| ob.start_prefix_mapping(n, v) } if listeners
|
||||
end
|
||||
end
|
||||
event[1] =~ Namespace::NAMESPLIT
|
||||
prefix = $1
|
||||
local = $2
|
||||
uri = get_namespace(prefix)
|
||||
# find the observers for start_element
|
||||
procs = get_procs( :start_element, event[1] )
|
||||
listeners = get_listeners( :start_element, event[1] )
|
||||
# notify observers
|
||||
procs.each { |ob| ob.call( uri, local, event[1], event[2] ) } if procs
|
||||
listeners.each { |ob|
|
||||
ob.start_element( uri, local, event[1], event[2] )
|
||||
} if listeners
|
||||
when :end_element
|
||||
@tag_stack.pop
|
||||
event[1] =~ Namespace::NAMESPLIT
|
||||
prefix = $1
|
||||
local = $2
|
||||
uri = get_namespace(prefix)
|
||||
# find the observers for start_element
|
||||
procs = get_procs( :end_element, event[1] )
|
||||
listeners = get_listeners( :end_element, event[1] )
|
||||
# notify observers
|
||||
procs.each { |ob| ob.call( uri, local, event[1] ) } if procs
|
||||
listeners.each { |ob|
|
||||
ob.end_element( uri, local, event[1] )
|
||||
} if listeners
|
||||
|
||||
namespace_mapping = @namespace_stack.pop
|
||||
# find the observers for namespaces
|
||||
procs = get_procs( :end_prefix_mapping, event[1] )
|
||||
listeners = get_listeners( :end_prefix_mapping, event[1] )
|
||||
if procs or listeners
|
||||
namespace_mapping.each do |ns_prefix, ns_uri|
|
||||
# notify observers of namespaces
|
||||
procs.each { |ob| ob.call( ns_prefix ) } if procs
|
||||
listeners.each { |ob| ob.end_prefix_mapping(ns_prefix) } if listeners
|
||||
end
|
||||
end
|
||||
when :text
|
||||
unnormalized = @parser.unnormalize( event[1], @entities )
|
||||
handle( :characters, unnormalized )
|
||||
when :entitydecl
|
||||
handle_entitydecl( event )
|
||||
when :processing_instruction, :comment, :attlistdecl,
|
||||
:elementdecl, :cdata, :notationdecl, :xmldecl
|
||||
handle( *event )
|
||||
end
|
||||
handle( :progress, @parser.position )
|
||||
end
|
||||
end
|
||||
|
||||
private
|
||||
def handle( symbol, *arguments )
|
||||
tag = @tag_stack[-1]
|
||||
procs = get_procs( symbol, tag )
|
||||
listeners = get_listeners( symbol, tag )
|
||||
# notify observers
|
||||
procs.each { |ob| ob.call( *arguments ) } if procs
|
||||
listeners.each { |l|
|
||||
l.send( symbol.to_s, *arguments )
|
||||
} if listeners
|
||||
end
|
||||
|
||||
def handle_entitydecl( event )
|
||||
@entities[ event[1] ] = event[2] if event.size == 3
|
||||
parameter_reference_p = false
|
||||
case event[2]
|
||||
when "SYSTEM"
|
||||
if event.size == 5
|
||||
if event.last == "%"
|
||||
parameter_reference_p = true
|
||||
else
|
||||
event[4, 0] = "NDATA"
|
||||
end
|
||||
end
|
||||
when "PUBLIC"
|
||||
if event.size == 6
|
||||
if event.last == "%"
|
||||
parameter_reference_p = true
|
||||
else
|
||||
event[5, 0] = "NDATA"
|
||||
end
|
||||
end
|
||||
else
|
||||
parameter_reference_p = (event.size == 4)
|
||||
end
|
||||
event[1, 0] = event.pop if parameter_reference_p
|
||||
handle( event[0], event[1..-1] )
|
||||
end
|
||||
|
||||
# The following methods are duplicates, but it is faster than using
|
||||
# a helper
|
||||
def get_procs( symbol, name )
|
||||
return nil if @procs.size == 0
|
||||
@procs.find_all do |sym, match, block|
|
||||
(
|
||||
(sym.nil? or symbol == sym) and
|
||||
((name.nil? and match.nil?) or match.nil? or (
|
||||
(name == match) or
|
||||
(match.kind_of? Regexp and name =~ match)
|
||||
)
|
||||
)
|
||||
)
|
||||
end.collect{|x| x[-1]}
|
||||
end
|
||||
def get_listeners( symbol, name )
|
||||
return nil if @listeners.size == 0
|
||||
@listeners.find_all do |sym, match, block|
|
||||
(
|
||||
(sym.nil? or symbol == sym) and
|
||||
((name.nil? and match.nil?) or match.nil? or (
|
||||
(name == match) or
|
||||
(match.kind_of? Regexp and name =~ match)
|
||||
)
|
||||
)
|
||||
)
|
||||
end.collect{|x| x[-1]}
|
||||
end
|
||||
|
||||
def add( pair )
|
||||
if pair[-1].respond_to? :call
|
||||
@procs << pair unless @procs.include? pair
|
||||
else
|
||||
@listeners << pair unless @listeners.include? pair
|
||||
@has_listeners = true
|
||||
end
|
||||
end
|
||||
|
||||
def get_namespace( prefix )
|
||||
return nil if @namespace_stack.empty?
|
||||
|
||||
uris = (@namespace_stack.find_all { |ns| not ns[prefix].nil? }) ||
|
||||
(@namespace_stack.find { |ns| not ns[nil].nil? })
|
||||
uris[-1][prefix] unless uris.nil? or 0 == uris.size
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,67 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "baseparser"
|
||||
|
||||
module REXML
|
||||
module Parsers
|
||||
class StreamParser
|
||||
def initialize source, listener
|
||||
@listener = listener
|
||||
@parser = BaseParser.new( source )
|
||||
@entities = {}
|
||||
end
|
||||
|
||||
def add_listener( listener )
|
||||
@parser.add_listener( listener )
|
||||
end
|
||||
|
||||
def entity_expansion_count
|
||||
@parser.entity_expansion_count
|
||||
end
|
||||
|
||||
def entity_expansion_limit=( limit )
|
||||
@parser.entity_expansion_limit = limit
|
||||
end
|
||||
|
||||
def entity_expansion_text_limit=( limit )
|
||||
@parser.entity_expansion_text_limit = limit
|
||||
end
|
||||
|
||||
def parse
|
||||
# entity string
|
||||
while true
|
||||
event = @parser.pull
|
||||
case event[0]
|
||||
when :end_document
|
||||
return
|
||||
when :start_element
|
||||
attrs = event[2].each do |n, v|
|
||||
event[2][n] = @parser.unnormalize( v )
|
||||
end
|
||||
@listener.tag_start( event[1], attrs )
|
||||
when :end_element
|
||||
@listener.tag_end( event[1] )
|
||||
when :text
|
||||
unnormalized = @parser.unnormalize( event[1], @entities )
|
||||
@listener.text( unnormalized )
|
||||
when :processing_instruction
|
||||
@listener.instruction( *event[1,2] )
|
||||
when :start_doctype
|
||||
@listener.doctype( *event[1..-1] )
|
||||
when :end_doctype
|
||||
# FIXME: remove this condition for milestone:3.2
|
||||
@listener.doctype_end if @listener.respond_to? :doctype_end
|
||||
when :comment, :attlistdecl, :cdata, :xmldecl, :elementdecl
|
||||
@listener.send( event[0].to_s, *event[1..-1] )
|
||||
when :entitydecl, :notationdecl
|
||||
@entities[ event[1] ] = event[2] if event.size == 3
|
||||
@listener.send( event[0].to_s, event[1..-1] )
|
||||
when :externalentity
|
||||
entity_reference = event[1]
|
||||
content = entity_reference.gsub(/\A%|;\z/, "")
|
||||
@listener.entity(content)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,89 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative '../validation/validationexception'
|
||||
require_relative '../undefinednamespaceexception'
|
||||
|
||||
module REXML
|
||||
module Parsers
|
||||
class TreeParser
|
||||
def initialize( source, build_context = Document.new )
|
||||
@build_context = build_context
|
||||
@parser = Parsers::BaseParser.new( source )
|
||||
end
|
||||
|
||||
def add_listener( listener )
|
||||
@parser.add_listener( listener )
|
||||
end
|
||||
|
||||
def parse
|
||||
entities = nil
|
||||
begin
|
||||
while true
|
||||
event = @parser.pull
|
||||
#STDERR.puts "TREEPARSER GOT #{event.inspect}"
|
||||
case event[0]
|
||||
when :end_document
|
||||
return
|
||||
when :start_element
|
||||
el = @build_context = @build_context.add_element( event[1] )
|
||||
event[2].each do |key, value|
|
||||
el.attributes[key]=Attribute.new(key,value,self)
|
||||
end
|
||||
when :end_element
|
||||
@build_context = @build_context.parent
|
||||
when :text
|
||||
if @build_context[-1].instance_of? Text
|
||||
@build_context[-1] << event[1]
|
||||
else
|
||||
@build_context.add(
|
||||
Text.new(event[1], @build_context.whitespace, nil, true)
|
||||
) unless (
|
||||
@build_context.ignore_whitespace_nodes and
|
||||
event[1].strip.size==0
|
||||
)
|
||||
end
|
||||
when :comment
|
||||
c = Comment.new( event[1] )
|
||||
@build_context.add( c )
|
||||
when :cdata
|
||||
c = CData.new( event[1] )
|
||||
@build_context.add( c )
|
||||
when :processing_instruction
|
||||
@build_context.add( Instruction.new( event[1], event[2] ) )
|
||||
when :end_doctype
|
||||
entities.each { |k,v| entities[k] = @build_context.entities[k].value }
|
||||
@build_context = @build_context.parent
|
||||
when :start_doctype
|
||||
doctype = DocType.new( event[1..-1], @build_context )
|
||||
@build_context = doctype
|
||||
entities = {}
|
||||
when :attlistdecl
|
||||
n = AttlistDecl.new( event[1..-1] )
|
||||
@build_context.add( n )
|
||||
when :externalentity
|
||||
n = ExternalEntity.new( event[1] )
|
||||
@build_context.add( n )
|
||||
when :elementdecl
|
||||
n = ElementDecl.new( event[1] )
|
||||
@build_context.add(n)
|
||||
when :entitydecl
|
||||
entities[ event[1] ] = event[2] unless event[2] =~ /PUBLIC|SYSTEM/
|
||||
@build_context.add(Entity.new(event))
|
||||
when :notationdecl
|
||||
n = NotationDecl.new( *event[1..-1] )
|
||||
@build_context.add( n )
|
||||
when :xmldecl
|
||||
x = XMLDecl.new( event[1], event[2], event[3] )
|
||||
@build_context.add( x )
|
||||
end
|
||||
end
|
||||
rescue REXML::Validation::ValidationException
|
||||
raise
|
||||
rescue REXML::ParseException
|
||||
raise
|
||||
rescue
|
||||
raise ParseException.new( $!.message, @parser.source, @parser, $! )
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,57 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative 'streamparser'
|
||||
require_relative 'baseparser'
|
||||
|
||||
module REXML
|
||||
module Parsers
|
||||
class UltraLightParser
|
||||
def initialize stream
|
||||
@stream = stream
|
||||
@parser = REXML::Parsers::BaseParser.new( stream )
|
||||
end
|
||||
|
||||
def add_listener( listener )
|
||||
@parser.add_listener( listener )
|
||||
end
|
||||
|
||||
def rewind
|
||||
@stream.rewind
|
||||
@parser.stream = @stream
|
||||
end
|
||||
|
||||
def parse
|
||||
root = context = []
|
||||
while true
|
||||
event = @parser.pull
|
||||
case event[0]
|
||||
when :end_document
|
||||
break
|
||||
when :end_doctype
|
||||
context = context[1]
|
||||
when :start_element, :start_doctype
|
||||
context << event
|
||||
event[1,0] = [context]
|
||||
context = event
|
||||
when :end_element
|
||||
context = context[1]
|
||||
else
|
||||
context << event
|
||||
end
|
||||
end
|
||||
root
|
||||
end
|
||||
end
|
||||
|
||||
# An element is an array. The array contains:
|
||||
# 0 The parent element
|
||||
# 1 The tag name
|
||||
# 2 A hash of attributes
|
||||
# 3..-1 The child elements
|
||||
# An element is an array of size > 3
|
||||
# Text is a String
|
||||
# PIs are [ :processing_instruction, target, data ]
|
||||
# Comments are [ :comment, data ]
|
||||
# DocTypes are DocType structs
|
||||
# The root is an array with XMLDecls, Text, DocType, Array, Text
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,739 @@
|
||||
# frozen_string_literal: false
|
||||
|
||||
require_relative '../namespace'
|
||||
require_relative '../xmltokens'
|
||||
|
||||
module REXML
|
||||
module Parsers
|
||||
# You don't want to use this class. Really. Use XPath, which is a wrapper
|
||||
# for this class. Believe me. You don't want to poke around in here.
|
||||
# There is strange, dark magic at work in this code. Beware. Go back! Go
|
||||
# back while you still can!
|
||||
class XPathParser
|
||||
include XMLTokens
|
||||
LITERAL = /^'([^']*)'|^"([^"]*)"/u
|
||||
|
||||
def namespaces=( namespaces )
|
||||
Functions::namespace_context = namespaces
|
||||
@namespaces = namespaces
|
||||
end
|
||||
|
||||
def parse path
|
||||
path = path.dup
|
||||
path.gsub!(/([\(\[])\s+/, '\1') # Strip ignorable spaces
|
||||
path.gsub!( /\s+([\]\)])/, '\1')
|
||||
parsed = []
|
||||
rest = OrExpr(path, parsed)
|
||||
if rest
|
||||
unless rest.strip.empty?
|
||||
raise ParseException.new("Garbage component exists at the end: " +
|
||||
"<#{rest}>: <#{path}>")
|
||||
end
|
||||
end
|
||||
parsed
|
||||
end
|
||||
|
||||
def predicate path
|
||||
parsed = []
|
||||
Predicate( "[#{path}]", parsed )
|
||||
parsed
|
||||
end
|
||||
|
||||
def abbreviate(path_or_parsed)
|
||||
if path_or_parsed.kind_of?(String)
|
||||
parsed = parse(path_or_parsed)
|
||||
else
|
||||
parsed = path_or_parsed
|
||||
end
|
||||
components = []
|
||||
component = nil
|
||||
while parsed.size > 0
|
||||
op = parsed.shift
|
||||
case op
|
||||
when :node
|
||||
component << "node()"
|
||||
when :attribute
|
||||
component = "@"
|
||||
components << component
|
||||
when :child
|
||||
component = ""
|
||||
components << component
|
||||
when :descendant_or_self
|
||||
next_op = parsed[0]
|
||||
if next_op == :node
|
||||
parsed.shift
|
||||
component = ""
|
||||
components << component
|
||||
else
|
||||
component = "descendant-or-self::"
|
||||
components << component
|
||||
end
|
||||
when :self
|
||||
next_op = parsed[0]
|
||||
if next_op == :node
|
||||
parsed.shift
|
||||
components << "."
|
||||
else
|
||||
component = "self::"
|
||||
components << component
|
||||
end
|
||||
when :parent
|
||||
next_op = parsed[0]
|
||||
if next_op == :node
|
||||
parsed.shift
|
||||
components << ".."
|
||||
else
|
||||
component = "parent::"
|
||||
components << component
|
||||
end
|
||||
when :any
|
||||
component << "*"
|
||||
when :text
|
||||
component << "text()"
|
||||
when :following, :following_sibling,
|
||||
:ancestor, :ancestor_or_self, :descendant,
|
||||
:namespace, :preceding, :preceding_sibling
|
||||
component = op.to_s.tr("_", "-") << "::"
|
||||
components << component
|
||||
when :qname
|
||||
prefix = parsed.shift
|
||||
name = parsed.shift
|
||||
component << prefix+":" if prefix.size > 0
|
||||
component << name
|
||||
when :predicate
|
||||
component << '['
|
||||
component << predicate_to_path(parsed.shift) {|x| abbreviate(x)}
|
||||
component << ']'
|
||||
when :document
|
||||
components << ""
|
||||
when :function
|
||||
component << parsed.shift
|
||||
component << "( "
|
||||
component << predicate_to_path(parsed.shift[0]) {|x| abbreviate(x)}
|
||||
component << " )"
|
||||
when :literal
|
||||
component << quote_literal(parsed.shift)
|
||||
else
|
||||
component << "UNKNOWN("
|
||||
component << op.inspect
|
||||
component << ")"
|
||||
end
|
||||
end
|
||||
case components
|
||||
when [""]
|
||||
"/"
|
||||
when ["", ""]
|
||||
"//"
|
||||
else
|
||||
components.join("/")
|
||||
end
|
||||
end
|
||||
|
||||
def expand(path_or_parsed)
|
||||
if path_or_parsed.kind_of?(String)
|
||||
parsed = parse(path_or_parsed)
|
||||
else
|
||||
parsed = path_or_parsed
|
||||
end
|
||||
path = ""
|
||||
document = false
|
||||
while parsed.size > 0
|
||||
op = parsed.shift
|
||||
case op
|
||||
when :node
|
||||
path << "node()"
|
||||
when :attribute, :child, :following, :following_sibling,
|
||||
:ancestor, :ancestor_or_self, :descendant, :descendant_or_self,
|
||||
:namespace, :preceding, :preceding_sibling, :self, :parent
|
||||
path << "/" unless path.size == 0
|
||||
path << op.to_s.tr("_", "-")
|
||||
path << "::"
|
||||
when :any
|
||||
path << "*"
|
||||
when :qname
|
||||
prefix = parsed.shift
|
||||
name = parsed.shift
|
||||
path << prefix+":" if prefix.size > 0
|
||||
path << name
|
||||
when :predicate
|
||||
path << '['
|
||||
path << predicate_to_path( parsed.shift ) { |x| expand(x) }
|
||||
path << ']'
|
||||
when :document
|
||||
document = true
|
||||
else
|
||||
path << "UNKNOWN("
|
||||
path << op.inspect
|
||||
path << ")"
|
||||
end
|
||||
end
|
||||
path = "/"+path if document
|
||||
path
|
||||
end
|
||||
|
||||
def predicate_to_path(parsed, &block)
|
||||
path = ""
|
||||
case parsed[0]
|
||||
when :and, :or, :mult, :plus, :minus, :neq, :eq, :lt, :gt, :lteq, :gteq, :div, :mod, :union
|
||||
op = parsed.shift
|
||||
case op
|
||||
when :eq
|
||||
op = "="
|
||||
when :lt
|
||||
op = "<"
|
||||
when :gt
|
||||
op = ">"
|
||||
when :lteq
|
||||
op = "<="
|
||||
when :gteq
|
||||
op = ">="
|
||||
when :neq
|
||||
op = "!="
|
||||
when :union
|
||||
op = "|"
|
||||
end
|
||||
left = predicate_to_path( parsed.shift, &block )
|
||||
right = predicate_to_path( parsed.shift, &block )
|
||||
path << left
|
||||
path << " "
|
||||
path << op.to_s
|
||||
path << " "
|
||||
path << right
|
||||
when :function
|
||||
parsed.shift
|
||||
name = parsed.shift
|
||||
path << name
|
||||
path << "("
|
||||
parsed.shift.each_with_index do |argument, i|
|
||||
path << ", " if i > 0
|
||||
path << predicate_to_path(argument, &block)
|
||||
end
|
||||
path << ")"
|
||||
when :literal
|
||||
parsed.shift
|
||||
path << quote_literal(parsed.shift)
|
||||
else
|
||||
path << yield( parsed )
|
||||
end
|
||||
path.squeeze(" ")
|
||||
end
|
||||
# For backward compatibility
|
||||
alias_method :preciate_to_string, :predicate_to_path
|
||||
|
||||
private
|
||||
def quote_literal( literal )
|
||||
case literal
|
||||
when String
|
||||
# XPath 1.0 does not support escape characters.
|
||||
# Assumes literal does not contain both single and double quotes.
|
||||
if literal.include?("'")
|
||||
"\"#{literal}\""
|
||||
else
|
||||
"'#{literal}'"
|
||||
end
|
||||
else
|
||||
literal.inspect
|
||||
end
|
||||
end
|
||||
|
||||
#LocationPath
|
||||
# | RelativeLocationPath
|
||||
# | '/' RelativeLocationPath?
|
||||
# | '//' RelativeLocationPath
|
||||
def LocationPath path, parsed
|
||||
path = path.lstrip
|
||||
if path[0] == ?/
|
||||
parsed << :document
|
||||
if path[1] == ?/
|
||||
parsed << :descendant_or_self
|
||||
parsed << :node
|
||||
path = path[2..-1]
|
||||
else
|
||||
path = path[1..-1]
|
||||
end
|
||||
end
|
||||
RelativeLocationPath( path, parsed ) if path.size > 0
|
||||
end
|
||||
|
||||
#RelativeLocationPath
|
||||
# | Step
|
||||
# | (AXIS_NAME '::' | '@' | '') AxisSpecifier
|
||||
# NodeTest
|
||||
# Predicate
|
||||
# | '.' | '..' AbbreviatedStep
|
||||
# | RelativeLocationPath '/' Step
|
||||
# | RelativeLocationPath '//' Step
|
||||
AXIS = /^(ancestor|ancestor-or-self|attribute|child|descendant|descendant-or-self|following|following-sibling|namespace|parent|preceding|preceding-sibling|self)::/
|
||||
def RelativeLocationPath path, parsed
|
||||
loop do
|
||||
original_path = path
|
||||
path = path.lstrip
|
||||
|
||||
return original_path if path.empty?
|
||||
|
||||
# (axis or @ or <child::>) nodetest predicate >
|
||||
# OR > / Step
|
||||
# (. or ..) >
|
||||
if path[0] == ?.
|
||||
if path[1] == ?.
|
||||
parsed << :parent
|
||||
parsed << :node
|
||||
path = path[2..-1]
|
||||
else
|
||||
parsed << :self
|
||||
parsed << :node
|
||||
path = path[1..-1]
|
||||
end
|
||||
else
|
||||
path_before_axis_specifier = path
|
||||
parsed_not_abberviated = []
|
||||
if path[0] == ?@
|
||||
parsed_not_abberviated << :attribute
|
||||
path = path[1..-1]
|
||||
# Goto Nodetest
|
||||
elsif path =~ AXIS
|
||||
parsed_not_abberviated << $1.tr('-','_').intern
|
||||
path = $'
|
||||
# Goto Nodetest
|
||||
else
|
||||
parsed_not_abberviated << :child
|
||||
end
|
||||
|
||||
path_before_node_test = path
|
||||
path = NodeTest(path, parsed_not_abberviated)
|
||||
if path == path_before_node_test
|
||||
return path_before_axis_specifier
|
||||
end
|
||||
path = Predicate(path, parsed_not_abberviated)
|
||||
|
||||
parsed.concat(parsed_not_abberviated)
|
||||
end
|
||||
|
||||
original_path = path
|
||||
path = path.lstrip
|
||||
return original_path if path.empty?
|
||||
|
||||
return original_path if path[0] != ?/
|
||||
|
||||
if path[1] == ?/
|
||||
parsed << :descendant_or_self
|
||||
parsed << :node
|
||||
path = path[2..-1]
|
||||
else
|
||||
path = path[1..-1]
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# Returns a 1-1 map of the nodeset
|
||||
# The contents of the resulting array are either:
|
||||
# true/false, if a positive match
|
||||
# String, if a name match
|
||||
#NodeTest
|
||||
# | ('*' | NCNAME ':' '*' | QNAME) NameTest
|
||||
# | '*' ':' NCNAME NameTest since XPath 2.0
|
||||
# | NODE_TYPE '(' ')' NodeType
|
||||
# | PI '(' LITERAL ')' PI
|
||||
# | '[' expr ']' Predicate
|
||||
PREFIX_WILDCARD = /^\*:(#{NCNAME_STR})/u
|
||||
LOCAL_NAME_WILDCARD = /^(#{NCNAME_STR}):\*/u
|
||||
QNAME = Namespace::NAMESPLIT
|
||||
NODE_TYPE = /^(comment|text|node)\(\s*\)/m
|
||||
PI = /^processing-instruction\(/
|
||||
def NodeTest path, parsed
|
||||
original_path = path
|
||||
path = path.lstrip
|
||||
case path
|
||||
when PREFIX_WILDCARD
|
||||
prefix = nil
|
||||
name = $1
|
||||
path = $'
|
||||
parsed << :qname
|
||||
parsed << prefix
|
||||
parsed << name
|
||||
when /^\*/
|
||||
path = $'
|
||||
parsed << :any
|
||||
when NODE_TYPE
|
||||
type = $1
|
||||
path = $'
|
||||
parsed << type.tr('-', '_').intern
|
||||
when PI
|
||||
path = $'
|
||||
literal = nil
|
||||
if path =~ /^\s*\)/
|
||||
path = $'
|
||||
else
|
||||
path =~ LITERAL
|
||||
literal = $1
|
||||
path = $'
|
||||
raise ParseException.new("Missing ')' after processing instruction") if path[0] != ?)
|
||||
path = path[1..-1]
|
||||
end
|
||||
parsed << :processing_instruction
|
||||
parsed << (literal || '')
|
||||
when LOCAL_NAME_WILDCARD
|
||||
prefix = $1
|
||||
path = $'
|
||||
parsed << :namespace
|
||||
parsed << prefix
|
||||
when QNAME
|
||||
prefix = $1
|
||||
name = $2
|
||||
path = $'
|
||||
prefix = "" unless prefix
|
||||
parsed << :qname
|
||||
parsed << prefix
|
||||
parsed << name
|
||||
else
|
||||
path = original_path
|
||||
end
|
||||
path
|
||||
end
|
||||
|
||||
# Filters the supplied nodeset on the predicate(s)
|
||||
def Predicate path, parsed
|
||||
original_path = path
|
||||
path = path.lstrip
|
||||
return original_path unless path[0] == ?[
|
||||
predicates = []
|
||||
while path[0] == ?[
|
||||
path, expr = get_group(path)
|
||||
predicates << expr[1..-2] if expr
|
||||
end
|
||||
predicates.each{ |pred|
|
||||
preds = []
|
||||
parsed << :predicate
|
||||
parsed << preds
|
||||
OrExpr(pred, preds)
|
||||
}
|
||||
path
|
||||
end
|
||||
|
||||
# The following return arrays of true/false, a 1-1 mapping of the
|
||||
# supplied nodeset, except for axe(), which returns a filtered
|
||||
# nodeset
|
||||
|
||||
#| OrExpr S 'or' S AndExpr
|
||||
#| AndExpr
|
||||
def OrExpr path, parsed
|
||||
n = []
|
||||
rest = AndExpr( path, n )
|
||||
if rest != path
|
||||
while rest =~ /^\s*( or )/
|
||||
n = [ :or, n, [] ]
|
||||
rest = AndExpr( $', n[-1] )
|
||||
end
|
||||
end
|
||||
if parsed.size == 0 and n.size != 0
|
||||
parsed.replace(n)
|
||||
elsif n.size > 0
|
||||
parsed << n
|
||||
end
|
||||
rest
|
||||
end
|
||||
|
||||
#| AndExpr S 'and' S EqualityExpr
|
||||
#| EqualityExpr
|
||||
def AndExpr path, parsed
|
||||
n = []
|
||||
rest = EqualityExpr( path, n )
|
||||
if rest != path
|
||||
while rest =~ /^\s*( and )/
|
||||
n = [ :and, n, [] ]
|
||||
rest = EqualityExpr( $', n[-1] )
|
||||
end
|
||||
end
|
||||
if parsed.size == 0 and n.size != 0
|
||||
parsed.replace(n)
|
||||
elsif n.size > 0
|
||||
parsed << n
|
||||
end
|
||||
rest
|
||||
end
|
||||
|
||||
#| EqualityExpr ('=' | '!=') RelationalExpr
|
||||
#| RelationalExpr
|
||||
def EqualityExpr path, parsed
|
||||
n = []
|
||||
rest = RelationalExpr( path, n )
|
||||
if rest != path
|
||||
while rest =~ /^\s*(!?=)\s*/
|
||||
if $1[0] == ?!
|
||||
n = [ :neq, n, [] ]
|
||||
else
|
||||
n = [ :eq, n, [] ]
|
||||
end
|
||||
rest = RelationalExpr( $', n[-1] )
|
||||
end
|
||||
end
|
||||
if parsed.size == 0 and n.size != 0
|
||||
parsed.replace(n)
|
||||
elsif n.size > 0
|
||||
parsed << n
|
||||
end
|
||||
rest
|
||||
end
|
||||
|
||||
#| RelationalExpr ('<' | '>' | '<=' | '>=') AdditiveExpr
|
||||
#| AdditiveExpr
|
||||
def RelationalExpr path, parsed
|
||||
n = []
|
||||
rest = AdditiveExpr( path, n )
|
||||
if rest != path
|
||||
while rest =~ /^\s*([<>]=?)\s*/
|
||||
if $1[0] == ?<
|
||||
sym = "lt"
|
||||
else
|
||||
sym = "gt"
|
||||
end
|
||||
sym << "eq" if $1[-1] == ?=
|
||||
n = [ sym.intern, n, [] ]
|
||||
rest = AdditiveExpr( $', n[-1] )
|
||||
end
|
||||
end
|
||||
if parsed.size == 0 and n.size != 0
|
||||
parsed.replace(n)
|
||||
elsif n.size > 0
|
||||
parsed << n
|
||||
end
|
||||
rest
|
||||
end
|
||||
|
||||
#| AdditiveExpr ('+' | '-') MultiplicativeExpr
|
||||
#| MultiplicativeExpr
|
||||
def AdditiveExpr path, parsed
|
||||
n = []
|
||||
rest = MultiplicativeExpr( path, n )
|
||||
if rest != path
|
||||
while rest =~ /^\s*(\+|-)\s*/
|
||||
if $1[0] == ?+
|
||||
n = [ :plus, n, [] ]
|
||||
else
|
||||
n = [ :minus, n, [] ]
|
||||
end
|
||||
rest = MultiplicativeExpr( $', n[-1] )
|
||||
end
|
||||
end
|
||||
if parsed.size == 0 and n.size != 0
|
||||
parsed.replace(n)
|
||||
elsif n.size > 0
|
||||
parsed << n
|
||||
end
|
||||
rest
|
||||
end
|
||||
|
||||
#| MultiplicativeExpr ('*' | S ('div' | 'mod') S) UnaryExpr
|
||||
#| UnaryExpr
|
||||
def MultiplicativeExpr path, parsed
|
||||
n = []
|
||||
rest = UnaryExpr( path, n )
|
||||
if rest != path
|
||||
while rest =~ /^\s*(\*| div | mod )\s*/
|
||||
if $1[0] == ?*
|
||||
n = [ :mult, n, [] ]
|
||||
elsif $1.include?( "div" )
|
||||
n = [ :div, n, [] ]
|
||||
else
|
||||
n = [ :mod, n, [] ]
|
||||
end
|
||||
rest = UnaryExpr( $', n[-1] )
|
||||
end
|
||||
end
|
||||
if parsed.size == 0 and n.size != 0
|
||||
parsed.replace(n)
|
||||
elsif n.size > 0
|
||||
parsed << n
|
||||
end
|
||||
rest
|
||||
end
|
||||
|
||||
#| '-' UnaryExpr
|
||||
#| UnionExpr
|
||||
def UnaryExpr path, parsed
|
||||
path =~ /^(\-*)/
|
||||
path = $'
|
||||
if $1 and (($1.size % 2) != 0)
|
||||
mult = -1
|
||||
else
|
||||
mult = 1
|
||||
end
|
||||
parsed << :neg if mult < 0
|
||||
|
||||
n = []
|
||||
path = UnionExpr( path, n )
|
||||
parsed.concat( n )
|
||||
path
|
||||
end
|
||||
|
||||
#| UnionExpr '|' PathExpr
|
||||
#| PathExpr
|
||||
def UnionExpr path, parsed
|
||||
n = []
|
||||
rest = PathExpr( path, n )
|
||||
if rest != path
|
||||
while rest =~ /^\s*(\|)\s*/
|
||||
n = [ :union, n, [] ]
|
||||
rest = PathExpr( $', n[-1] )
|
||||
end
|
||||
end
|
||||
if parsed.size == 0 and n.size != 0
|
||||
parsed.replace( n )
|
||||
elsif n.size > 0
|
||||
parsed << n
|
||||
end
|
||||
rest
|
||||
end
|
||||
|
||||
#| LocationPath
|
||||
#| FilterExpr ('/' | '//') RelativeLocationPath
|
||||
def PathExpr path, parsed
|
||||
path = path.lstrip
|
||||
n = []
|
||||
rest = FilterExpr( path, n )
|
||||
if rest != path
|
||||
if rest and rest[0] == ?/
|
||||
rest = RelativeLocationPath(rest, n)
|
||||
parsed.concat(n)
|
||||
return rest
|
||||
end
|
||||
end
|
||||
rest = LocationPath(rest, n) if rest =~ /\A[\/\.\@\[\w*]/
|
||||
parsed.concat(n)
|
||||
rest
|
||||
end
|
||||
|
||||
#| FilterExpr Predicate
|
||||
#| PrimaryExpr
|
||||
def FilterExpr path, parsed
|
||||
n = []
|
||||
path_before_primary_expr = path
|
||||
path = PrimaryExpr(path, n)
|
||||
return path_before_primary_expr if path == path_before_primary_expr
|
||||
path = Predicate(path, n)
|
||||
parsed.concat(n)
|
||||
path
|
||||
end
|
||||
|
||||
#| VARIABLE_REFERENCE
|
||||
#| '(' expr ')'
|
||||
#| LITERAL
|
||||
#| NUMBER
|
||||
#| FunctionCall
|
||||
VARIABLE_REFERENCE = /^\$(#{NAME_STR})/u
|
||||
NUMBER = /^(\d*\.?\d+)/
|
||||
NT = /^comment|text|processing-instruction|node$/
|
||||
def PrimaryExpr path, parsed
|
||||
case path
|
||||
when VARIABLE_REFERENCE
|
||||
varname = $1
|
||||
path = $'
|
||||
parsed << :variable
|
||||
parsed << varname
|
||||
#arry << @variables[ varname ]
|
||||
when /^(\w[-\w]*)(?:\()/
|
||||
fname = $1
|
||||
tmp = $'
|
||||
return path if fname =~ NT
|
||||
path = tmp
|
||||
parsed << :function
|
||||
parsed << fname
|
||||
path = FunctionCall(path, parsed)
|
||||
when NUMBER
|
||||
varname = $1.nil? ? $2 : $1
|
||||
path = $'
|
||||
parsed << :literal
|
||||
parsed << (varname.include?('.') ? varname.to_f : varname.to_i)
|
||||
when LITERAL
|
||||
varname = $1.nil? ? $2 : $1
|
||||
path = $'
|
||||
parsed << :literal
|
||||
parsed << varname
|
||||
when /^\(/ #/
|
||||
path, contents = get_group(path)
|
||||
contents = contents[1..-2]
|
||||
n = []
|
||||
OrExpr( contents, n )
|
||||
parsed.concat(n)
|
||||
end
|
||||
path
|
||||
end
|
||||
|
||||
#| FUNCTION_NAME '(' ( expr ( ',' expr )* )? ')'
|
||||
def FunctionCall rest, parsed
|
||||
path, arguments = parse_args(rest)
|
||||
argset = []
|
||||
for argument in arguments
|
||||
args = []
|
||||
OrExpr( argument, args )
|
||||
argset << args
|
||||
end
|
||||
parsed << argset
|
||||
path
|
||||
end
|
||||
|
||||
# get_group( '[foo]bar' ) -> ['bar', '[foo]']
|
||||
def get_group string
|
||||
ind = 0
|
||||
depth = 0
|
||||
st = string[0,1]
|
||||
en = (st == "(" ? ")" : "]")
|
||||
begin
|
||||
case string[ind,1]
|
||||
when st
|
||||
depth += 1
|
||||
when en
|
||||
depth -= 1
|
||||
end
|
||||
ind += 1
|
||||
end while depth > 0 and ind < string.length
|
||||
return nil unless depth==0
|
||||
[string[ind..-1], string[0..ind-1]]
|
||||
end
|
||||
|
||||
def parse_args( string )
|
||||
arguments = []
|
||||
ind = 0
|
||||
inquot = false
|
||||
inapos = false
|
||||
depth = 1
|
||||
begin
|
||||
case string[ind]
|
||||
when ?"
|
||||
inquot = !inquot unless inapos
|
||||
when ?'
|
||||
inapos = !inapos unless inquot
|
||||
else
|
||||
unless inquot or inapos
|
||||
case string[ind]
|
||||
when ?(
|
||||
depth += 1
|
||||
if depth == 1
|
||||
string = string[1..-1]
|
||||
ind -= 1
|
||||
end
|
||||
when ?)
|
||||
depth -= 1
|
||||
if depth == 0
|
||||
s = string[0,ind].strip
|
||||
arguments << s unless s == ""
|
||||
string = string[ind+1..-1]
|
||||
end
|
||||
when ?,
|
||||
if depth == 1
|
||||
s = string[0,ind].strip
|
||||
arguments << s unless s == ""
|
||||
string = string[ind+1..-1]
|
||||
ind = -1
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
ind += 1
|
||||
end while depth > 0 and ind < string.length
|
||||
return nil unless depth==0
|
||||
[string,arguments]
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,267 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative 'functions'
|
||||
require_relative 'xmltokens'
|
||||
|
||||
module REXML
|
||||
class QuickPath
|
||||
include Functions
|
||||
include XMLTokens
|
||||
|
||||
# A base Hash object to be used when initializing a
|
||||
# default empty namespaces set.
|
||||
EMPTY_HASH = {}
|
||||
|
||||
def QuickPath::first element, path, namespaces=EMPTY_HASH
|
||||
match(element, path, namespaces)[0]
|
||||
end
|
||||
|
||||
def QuickPath::each element, path, namespaces=EMPTY_HASH, &block
|
||||
path = "*" unless path
|
||||
match(element, path, namespaces).each( &block )
|
||||
end
|
||||
|
||||
def QuickPath::match element, path, namespaces=EMPTY_HASH
|
||||
raise "nil is not a valid xpath" unless path
|
||||
results = nil
|
||||
Functions::namespace_context = namespaces
|
||||
case path
|
||||
when /^\/([^\/]|$)/u
|
||||
# match on root
|
||||
path = path[1..-1]
|
||||
return [element.root.parent] if path == ''
|
||||
results = filter([element.root], path)
|
||||
when /^[-\w]*::/u
|
||||
results = filter([element], path)
|
||||
when /^\*/u
|
||||
results = filter(element.to_a, path)
|
||||
when /^[\[!\w:]/u
|
||||
# match on child
|
||||
children = element.to_a
|
||||
results = filter(children, path)
|
||||
else
|
||||
results = filter([element], path)
|
||||
end
|
||||
results
|
||||
end
|
||||
|
||||
# Given an array of nodes it filters the array based on the path. The
|
||||
# result is that when this method returns, the array will contain elements
|
||||
# which match the path
|
||||
def QuickPath::filter elements, path
|
||||
return elements if path.nil? or path == '' or elements.size == 0
|
||||
case path
|
||||
when /^\/\//u # Descendant
|
||||
axe( elements, "descendant-or-self", $' )
|
||||
when /^\/?\b(\w[-\w]*)\b::/u # Axe
|
||||
axe( elements, $1, $' )
|
||||
when /^\/(?=\b([:!\w][-\.\w]*:)?[-!\*\.\w]*\b([^:(]|$)|\*)/u # Child
|
||||
rest = $'
|
||||
results = []
|
||||
elements.each do |element|
|
||||
results |= filter( element.to_a, rest )
|
||||
end
|
||||
results
|
||||
when /^\/?(\w[-\w]*)\(/u # / Function
|
||||
function( elements, $1, $' )
|
||||
when Namespace::NAMESPLIT # Element name
|
||||
name = $2
|
||||
ns = $1
|
||||
rest = $'
|
||||
elements.delete_if do |element|
|
||||
!(element.kind_of? Element and
|
||||
(element.expanded_name == name or
|
||||
(element.name == name and
|
||||
element.namespace == Functions.namespace_context[ns])))
|
||||
end
|
||||
filter( elements, rest )
|
||||
when /^\/\[/u
|
||||
matches = []
|
||||
elements.each do |element|
|
||||
matches |= predicate( element.to_a, path[1..-1] ) if element.kind_of? Element
|
||||
end
|
||||
matches
|
||||
when /^\[/u # Predicate
|
||||
predicate( elements, path )
|
||||
when /^\/?\.\.\./u # Ancestor
|
||||
axe( elements, "ancestor", $' )
|
||||
when /^\/?\.\./u # Parent
|
||||
filter( elements.collect{|e|e.parent}, $' )
|
||||
when /^\/?\./u # Self
|
||||
filter( elements, $' )
|
||||
when /^\*/u # Any
|
||||
results = []
|
||||
elements.each do |element|
|
||||
results |= filter( [element], $' ) if element.kind_of? Element
|
||||
#if element.kind_of? Element
|
||||
# children = element.to_a
|
||||
# children.delete_if { |child| !child.kind_of?(Element) }
|
||||
# results |= filter( children, $' )
|
||||
#end
|
||||
end
|
||||
results
|
||||
else
|
||||
[]
|
||||
end
|
||||
end
|
||||
|
||||
def QuickPath::axe( elements, axe_name, rest )
|
||||
matches = []
|
||||
matches = filter( elements.dup, rest ) if axe_name =~ /-or-self$/u
|
||||
case axe_name
|
||||
when /^descendant/u
|
||||
elements.each do |element|
|
||||
matches |= filter( element.to_a, "descendant-or-self::#{rest}" ) if element.kind_of? Element
|
||||
end
|
||||
when /^ancestor/u
|
||||
elements.each do |element|
|
||||
while element.parent
|
||||
matches << element.parent
|
||||
element = element.parent
|
||||
end
|
||||
end
|
||||
matches = filter( matches, rest )
|
||||
when "self"
|
||||
matches = filter( elements, rest )
|
||||
when "child"
|
||||
elements.each do |element|
|
||||
matches |= filter( element.to_a, rest ) if element.kind_of? Element
|
||||
end
|
||||
when "attribute"
|
||||
elements.each do |element|
|
||||
matches << element.attributes[ rest ] if element.kind_of? Element
|
||||
end
|
||||
when "parent"
|
||||
matches = filter(elements.collect{|element| element.parent}.uniq, rest)
|
||||
when "following-sibling"
|
||||
matches = filter(elements.collect{|element| element.next_sibling}.uniq,
|
||||
rest)
|
||||
when "previous-sibling"
|
||||
matches = filter(elements.collect{|element|
|
||||
element.previous_sibling}.uniq, rest )
|
||||
end
|
||||
matches.uniq
|
||||
end
|
||||
|
||||
OPERAND_ = '((?=(?:(?!and|or).)*[^\s<>=])[^\s<>=]+)'
|
||||
# A predicate filters a node-set with respect to an axis to produce a
|
||||
# new node-set. For each node in the node-set to be filtered, the
|
||||
# PredicateExpr is evaluated with that node as the context node, with
|
||||
# the number of nodes in the node-set as the context size, and with the
|
||||
# proximity position of the node in the node-set with respect to the
|
||||
# axis as the context position; if PredicateExpr evaluates to true for
|
||||
# that node, the node is included in the new node-set; otherwise, it is
|
||||
# not included.
|
||||
#
|
||||
# A PredicateExpr is evaluated by evaluating the Expr and converting
|
||||
# the result to a boolean. If the result is a number, the result will
|
||||
# be converted to true if the number is equal to the context position
|
||||
# and will be converted to false otherwise; if the result is not a
|
||||
# number, then the result will be converted as if by a call to the
|
||||
# boolean function. Thus a location path para[3] is equivalent to
|
||||
# para[position()=3].
|
||||
def QuickPath::predicate( elements, path )
|
||||
ind = 1
|
||||
bcount = 1
|
||||
while bcount > 0
|
||||
bcount += 1 if path[ind] == ?[
|
||||
bcount -= 1 if path[ind] == ?]
|
||||
ind += 1
|
||||
end
|
||||
ind -= 1
|
||||
predicate = path[1..ind-1]
|
||||
rest = path[ind+1..-1]
|
||||
|
||||
# have to change 'a [=<>] b [=<>] c' into 'a [=<>] b and b [=<>] c'
|
||||
#
|
||||
predicate.gsub!(
|
||||
/#{OPERAND_}\s*([<>=])\s*#{OPERAND_}\s*([<>=])\s*#{OPERAND_}/u,
|
||||
'\1 \2 \3 and \3 \4 \5' )
|
||||
# Let's do some Ruby trickery to avoid some work:
|
||||
predicate.gsub!( /&/u, "&&" )
|
||||
predicate.gsub!( /=/u, "==" )
|
||||
predicate.gsub!( /@(\w[-\w.]*)/u, 'attribute("\1")' )
|
||||
predicate.gsub!( /\bmod\b/u, "%" )
|
||||
predicate.gsub!( /\b(\w[-\w.]*\()/u ) {
|
||||
fname = $1
|
||||
fname.gsub( /-/u, "_" )
|
||||
}
|
||||
|
||||
Functions.pair = [ 0, elements.size ]
|
||||
results = []
|
||||
elements.each do |element|
|
||||
Functions.pair[0] += 1
|
||||
Functions.node = element
|
||||
res = eval( predicate )
|
||||
case res
|
||||
when true
|
||||
results << element
|
||||
when Integer
|
||||
results << element if Functions.pair[0] == res
|
||||
when String
|
||||
results << element
|
||||
end
|
||||
end
|
||||
filter( results, rest )
|
||||
end
|
||||
|
||||
def QuickPath::attribute( name )
|
||||
Functions.node.attributes[name] if Functions.node.kind_of? Element
|
||||
end
|
||||
|
||||
def QuickPath::name()
|
||||
Functions.node.name if Functions.node.kind_of? Element
|
||||
end
|
||||
|
||||
def QuickPath::method_missing( id, *args )
|
||||
begin
|
||||
Functions.send( id.id2name, *args )
|
||||
rescue Exception
|
||||
raise "METHOD: #{id.id2name}(#{args.join ', '})\n#{$!.message}"
|
||||
end
|
||||
end
|
||||
|
||||
def QuickPath::function( elements, fname, rest )
|
||||
args = parse_args( elements, rest )
|
||||
Functions.pair = [0, elements.size]
|
||||
results = []
|
||||
elements.each do |element|
|
||||
Functions.pair[0] += 1
|
||||
Functions.node = element
|
||||
res = Functions.send( fname, *args )
|
||||
case res
|
||||
when true
|
||||
results << element
|
||||
when Integer
|
||||
results << element if Functions.pair[0] == res
|
||||
end
|
||||
end
|
||||
results
|
||||
end
|
||||
|
||||
def QuickPath::parse_args( element, string )
|
||||
# /.*?(?:\)|,)/
|
||||
arguments = []
|
||||
buffer = ""
|
||||
while string and string != ""
|
||||
c = string[0]
|
||||
string.sub!(/^./u, "")
|
||||
case c
|
||||
when ?,
|
||||
# if depth = 1, then we start a new argument
|
||||
arguments << evaluate( buffer )
|
||||
#arguments << evaluate( string[0..count] )
|
||||
when ?(
|
||||
# start a new method call
|
||||
function( element, buffer, string )
|
||||
buffer = ""
|
||||
when ?)
|
||||
# close the method call and return arguments
|
||||
return arguments
|
||||
else
|
||||
buffer << c
|
||||
end
|
||||
end
|
||||
""
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,39 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# frozen_string_literal: false
|
||||
#
|
||||
# \Module \REXML provides classes and methods for parsing,
|
||||
# editing, and generating XML.
|
||||
#
|
||||
# == Implementation
|
||||
#
|
||||
# \REXML:
|
||||
# - Is pure Ruby.
|
||||
# - Provides tree, stream, SAX2, pull, and lightweight APIs.
|
||||
# - Conforms to {XML version 1.0}[https://www.w3.org/TR/REC-xml/].
|
||||
# - Fully implements {XPath version 1.0}[http://www.w3c.org/tr/xpath].
|
||||
# - Is {non-validating}[https://www.w3.org/TR/xml/].
|
||||
# - Passes 100% of the non-validating {Oasis tests}[http://www.oasis-open.org/committees/xml-conformance/xml-test-suite.shtml].
|
||||
#
|
||||
# == In a Hurry?
|
||||
#
|
||||
# If you're somewhat familiar with XML
|
||||
# and have a particular task in mind,
|
||||
# you may want to see {the tasks pages}[doc/rexml/tasks/tocs/master_toc_rdoc.html].
|
||||
#
|
||||
# == API
|
||||
#
|
||||
# Among the most important classes for using \REXML are:
|
||||
# - REXML::Document.
|
||||
# - REXML::Element.
|
||||
#
|
||||
# There's also an {REXML tutorial}[doc/rexml/tutorial_rdoc.html].
|
||||
#
|
||||
module REXML
|
||||
COPYRIGHT = "Copyright © 2001-2008 Sean Russell <ser@germane-software.com>"
|
||||
DATE = "2008/019"
|
||||
VERSION = "3.4.4"
|
||||
REVISION = ""
|
||||
|
||||
Copyright = COPYRIGHT
|
||||
Version = VERSION
|
||||
end
|
||||
@@ -0,0 +1,98 @@
|
||||
# frozen_string_literal: false
|
||||
module REXML
|
||||
# A template for stream parser listeners.
|
||||
# Note that the declarations (attlistdecl, elementdecl, etc) are trivially
|
||||
# processed; REXML doesn't yet handle doctype entity declarations, so you
|
||||
# have to parse them out yourself.
|
||||
# === Missing methods from SAX2
|
||||
# ignorable_whitespace
|
||||
# === Methods extending SAX2
|
||||
# +WARNING+
|
||||
# These methods are certainly going to change, until DTDs are fully
|
||||
# supported. Be aware of this.
|
||||
# start_document
|
||||
# end_document
|
||||
# doctype
|
||||
# elementdecl
|
||||
# attlistdecl
|
||||
# entitydecl
|
||||
# notationdecl
|
||||
# cdata
|
||||
# xmldecl
|
||||
# comment
|
||||
module SAX2Listener
|
||||
def start_document
|
||||
end
|
||||
def end_document
|
||||
end
|
||||
def start_prefix_mapping prefix, uri
|
||||
end
|
||||
def end_prefix_mapping prefix
|
||||
end
|
||||
def start_element uri, localname, qname, attributes
|
||||
end
|
||||
def end_element uri, localname, qname
|
||||
end
|
||||
def characters text
|
||||
end
|
||||
def processing_instruction target, data
|
||||
end
|
||||
# Handles a doctype declaration. Any attributes of the doctype which are
|
||||
# not supplied will be nil. # EG, <!DOCTYPE me PUBLIC "foo" "bar">
|
||||
# @p name the name of the doctype; EG, "me"
|
||||
# @p pub_sys "PUBLIC", "SYSTEM", or nil. EG, "PUBLIC"
|
||||
# @p long_name the supplied long name, or nil. EG, "foo"
|
||||
# @p uri the uri of the doctype, or nil. EG, "bar"
|
||||
def doctype name, pub_sys, long_name, uri
|
||||
end
|
||||
# If a doctype includes an ATTLIST declaration, it will cause this
|
||||
# method to be called. The content is the declaration itself, unparsed.
|
||||
# EG, <!ATTLIST el attr CDATA #REQUIRED> will come to this method as "el
|
||||
# attr CDATA #REQUIRED". This is the same for all of the .*decl
|
||||
# methods.
|
||||
def attlistdecl(element, pairs, contents)
|
||||
end
|
||||
# <!ELEMENT ...>
|
||||
def elementdecl content
|
||||
end
|
||||
# <!ENTITY ...>
|
||||
# The argument passed to this method is an array of the entity
|
||||
# declaration. It can be in a number of formats, but in general it
|
||||
# returns (example, result):
|
||||
# <!ENTITY % YN '"Yes"'>
|
||||
# ["%", "YN", "\"Yes\""]
|
||||
# <!ENTITY % YN 'Yes'>
|
||||
# ["%", "YN", "Yes"]
|
||||
# <!ENTITY WhatHeSaid "He said %YN;">
|
||||
# ["WhatHeSaid", "He said %YN;"]
|
||||
# <!ENTITY open-hatch SYSTEM "http://www.textuality.com/boilerplate/OpenHatch.xml">
|
||||
# ["open-hatch", "SYSTEM", "http://www.textuality.com/boilerplate/OpenHatch.xml"]
|
||||
# <!ENTITY open-hatch PUBLIC "-//Textuality//TEXT Standard open-hatch boilerplate//EN" "http://www.textuality.com/boilerplate/OpenHatch.xml">
|
||||
# ["open-hatch", "PUBLIC", "-//Textuality//TEXT Standard open-hatch boilerplate//EN", "http://www.textuality.com/boilerplate/OpenHatch.xml"]
|
||||
# <!ENTITY hatch-pic SYSTEM "../grafix/OpenHatch.gif" NDATA gif>
|
||||
# ["hatch-pic", "SYSTEM", "../grafix/OpenHatch.gif", "NDATA", "gif"]
|
||||
def entitydecl declaration
|
||||
end
|
||||
# <!NOTATION ...>
|
||||
def notationdecl name, public_or_system, public_id, system_id
|
||||
end
|
||||
# Called when <![CDATA[ ... ]]> is encountered in a document.
|
||||
# @p content "..."
|
||||
def cdata content
|
||||
end
|
||||
# Called when an XML PI is encountered in the document.
|
||||
# EG: <?xml version="1.0" encoding="utf"?>
|
||||
# @p version the version attribute value. EG, "1.0"
|
||||
# @p encoding the encoding attribute value, or nil. EG, "utf"
|
||||
# @p standalone the standalone attribute value, or nil. EG, nil
|
||||
# @p spaced the declaration is followed by a line break
|
||||
def xmldecl version, encoding, standalone
|
||||
end
|
||||
# Called when a comment is encountered.
|
||||
# @p comment The content of the comment
|
||||
def comment comment
|
||||
end
|
||||
def progress position
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,28 @@
|
||||
# frozen_string_literal: false
|
||||
module REXML
|
||||
module Security
|
||||
@@entity_expansion_limit = 10_000
|
||||
|
||||
# Set the entity expansion limit. By default the limit is set to 10000.
|
||||
def self.entity_expansion_limit=( val )
|
||||
@@entity_expansion_limit = val
|
||||
end
|
||||
|
||||
# Get the entity expansion limit. By default the limit is set to 10000.
|
||||
def self.entity_expansion_limit
|
||||
@@entity_expansion_limit
|
||||
end
|
||||
|
||||
@@entity_expansion_text_limit = 10_240
|
||||
|
||||
# Set the entity expansion limit. By default the limit is set to 10240.
|
||||
def self.entity_expansion_text_limit=( val )
|
||||
@@entity_expansion_text_limit = val
|
||||
end
|
||||
|
||||
# Get the entity expansion limit. By default the limit is set to 10240.
|
||||
def self.entity_expansion_text_limit
|
||||
@@entity_expansion_text_limit
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,388 @@
|
||||
# coding: US-ASCII
|
||||
# frozen_string_literal: false
|
||||
|
||||
require "stringio"
|
||||
require "strscan"
|
||||
|
||||
require_relative 'encoding'
|
||||
|
||||
module REXML
|
||||
if StringScanner::Version < "1.0.0"
|
||||
module StringScannerCheckScanString
|
||||
refine StringScanner do
|
||||
def check(pattern)
|
||||
pattern = /#{Regexp.escape(pattern)}/ if pattern.is_a?(String)
|
||||
super(pattern)
|
||||
end
|
||||
|
||||
def scan(pattern)
|
||||
pattern = /#{Regexp.escape(pattern)}/ if pattern.is_a?(String)
|
||||
super(pattern)
|
||||
end
|
||||
|
||||
def match?(pattern)
|
||||
pattern = /#{Regexp.escape(pattern)}/ if pattern.is_a?(String)
|
||||
super(pattern)
|
||||
end
|
||||
|
||||
def skip(pattern)
|
||||
pattern = /#{Regexp.escape(pattern)}/ if pattern.is_a?(String)
|
||||
super(pattern)
|
||||
end
|
||||
end
|
||||
end
|
||||
using StringScannerCheckScanString
|
||||
end
|
||||
|
||||
# Generates Source-s. USE THIS CLASS.
|
||||
class SourceFactory
|
||||
# Generates a Source object
|
||||
# @param arg Either a String, or an IO
|
||||
# @return a Source, or nil if a bad argument was given
|
||||
def SourceFactory::create_from(arg)
|
||||
if arg.respond_to? :read and
|
||||
arg.respond_to? :readline and
|
||||
arg.respond_to? :nil? and
|
||||
arg.respond_to? :eof?
|
||||
IOSource.new(arg)
|
||||
elsif arg.respond_to? :to_str
|
||||
IOSource.new(StringIO.new(arg))
|
||||
elsif arg.kind_of? Source
|
||||
arg
|
||||
else
|
||||
raise "#{arg.class} is not a valid input stream. It must walk \n"+
|
||||
"like either a String, an IO, or a Source."
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# A Source can be searched for patterns, and wraps buffers and other
|
||||
# objects and provides consumption of text
|
||||
class Source
|
||||
include Encoding
|
||||
# The line number of the last consumed text
|
||||
attr_reader :line
|
||||
attr_reader :encoding
|
||||
|
||||
module Private
|
||||
SPACES_PATTERN = /\s+/um
|
||||
SCANNER_RESET_SIZE = 100000
|
||||
PRE_DEFINED_TERM_PATTERNS = {}
|
||||
pre_defined_terms = ["'", '"', "<", "]]>", "?>"]
|
||||
if StringScanner::Version < "3.1.1"
|
||||
pre_defined_terms.each do |term|
|
||||
PRE_DEFINED_TERM_PATTERNS[term] = /#{Regexp.escape(term)}/
|
||||
end
|
||||
else
|
||||
pre_defined_terms.each do |term|
|
||||
PRE_DEFINED_TERM_PATTERNS[term] = term
|
||||
end
|
||||
end
|
||||
end
|
||||
private_constant :Private
|
||||
|
||||
# Constructor
|
||||
# @param arg must be a String, and should be a valid XML document
|
||||
# @param encoding if non-null, sets the encoding of the source to this
|
||||
# value, overriding all encoding detection
|
||||
def initialize(arg, encoding=nil)
|
||||
@orig = arg
|
||||
@scanner = StringScanner.new(@orig)
|
||||
if encoding
|
||||
self.encoding = encoding
|
||||
else
|
||||
detect_encoding
|
||||
end
|
||||
@line = 0
|
||||
@encoded_terms = {}
|
||||
end
|
||||
|
||||
# The current buffer (what we're going to read next)
|
||||
def buffer
|
||||
@scanner.rest
|
||||
end
|
||||
|
||||
def drop_parsed_content
|
||||
if @scanner.pos > Private::SCANNER_RESET_SIZE
|
||||
@scanner.string = @scanner.rest
|
||||
end
|
||||
end
|
||||
|
||||
def buffer_encoding=(encoding)
|
||||
@scanner.string.force_encoding(encoding)
|
||||
end
|
||||
|
||||
# Inherited from Encoding
|
||||
# Overridden to support optimized en/decoding
|
||||
def encoding=(enc)
|
||||
return unless super
|
||||
encoding_updated
|
||||
end
|
||||
|
||||
def read(term = nil)
|
||||
end
|
||||
|
||||
def read_until(term)
|
||||
pattern = Private::PRE_DEFINED_TERM_PATTERNS[term] || /#{Regexp.escape(term)}/
|
||||
data = @scanner.scan_until(pattern)
|
||||
unless data
|
||||
data = @scanner.rest
|
||||
@scanner.pos = @scanner.string.bytesize
|
||||
end
|
||||
data
|
||||
end
|
||||
|
||||
def ensure_buffer
|
||||
end
|
||||
|
||||
def match(pattern, cons=false)
|
||||
if cons
|
||||
@scanner.scan(pattern).nil? ? nil : @scanner
|
||||
else
|
||||
@scanner.check(pattern).nil? ? nil : @scanner
|
||||
end
|
||||
end
|
||||
|
||||
def match?(pattern, cons=false)
|
||||
if cons
|
||||
!@scanner.skip(pattern).nil?
|
||||
else
|
||||
!@scanner.match?(pattern).nil?
|
||||
end
|
||||
end
|
||||
|
||||
def skip_spaces
|
||||
@scanner.skip(Private::SPACES_PATTERN) ? true : false
|
||||
end
|
||||
|
||||
def position
|
||||
@scanner.pos
|
||||
end
|
||||
|
||||
def position=(pos)
|
||||
@scanner.pos = pos
|
||||
end
|
||||
|
||||
def peek_byte
|
||||
@scanner.peek_byte
|
||||
end
|
||||
|
||||
def scan_byte
|
||||
@scanner.scan_byte
|
||||
end
|
||||
|
||||
# @return true if the Source is exhausted
|
||||
def empty?
|
||||
@scanner.eos?
|
||||
end
|
||||
|
||||
# @return the current line in the source
|
||||
def current_line
|
||||
lines = @orig.split
|
||||
res = lines.grep @scanner.rest[0..30]
|
||||
res = res[-1] if res.kind_of? Array
|
||||
lines.index( res ) if res
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def detect_encoding
|
||||
scanner_encoding = @scanner.rest.encoding
|
||||
detected_encoding = "UTF-8"
|
||||
begin
|
||||
@scanner.string.force_encoding("ASCII-8BIT")
|
||||
if @scanner.scan(/\xfe\xff/n)
|
||||
detected_encoding = "UTF-16BE"
|
||||
elsif @scanner.scan(/\xff\xfe/n)
|
||||
detected_encoding = "UTF-16LE"
|
||||
elsif @scanner.scan(/\xef\xbb\xbf/n)
|
||||
detected_encoding = "UTF-8"
|
||||
end
|
||||
ensure
|
||||
@scanner.string.force_encoding(scanner_encoding)
|
||||
end
|
||||
self.encoding = detected_encoding
|
||||
end
|
||||
|
||||
def encoding_updated
|
||||
if @encoding != 'UTF-8'
|
||||
@scanner.string = decode(@scanner.rest)
|
||||
@to_utf = true
|
||||
else
|
||||
@to_utf = false
|
||||
@scanner.string.force_encoding(::Encoding::UTF_8)
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# A Source that wraps an IO. See the Source class for method
|
||||
# documentation
|
||||
class IOSource < Source
|
||||
#attr_reader :block_size
|
||||
|
||||
# block_size has been deprecated
|
||||
def initialize(arg, block_size=500, encoding=nil)
|
||||
@er_source = @source = arg
|
||||
@to_utf = false
|
||||
@pending_buffer = nil
|
||||
|
||||
if encoding
|
||||
super("", encoding)
|
||||
else
|
||||
super(@source.read(3) || "")
|
||||
end
|
||||
|
||||
if !@to_utf and
|
||||
@orig.respond_to?(:force_encoding) and
|
||||
@source.respond_to?(:external_encoding) and
|
||||
@source.external_encoding != ::Encoding::UTF_8
|
||||
@force_utf8 = true
|
||||
else
|
||||
@force_utf8 = false
|
||||
end
|
||||
end
|
||||
|
||||
def read(term = nil, min_bytes = 1)
|
||||
term = encode(term) if term
|
||||
begin
|
||||
str = readline(term)
|
||||
@scanner << str
|
||||
read_bytes = str.bytesize
|
||||
begin
|
||||
while read_bytes < min_bytes
|
||||
str = readline(term)
|
||||
@scanner << str
|
||||
read_bytes += str.bytesize
|
||||
end
|
||||
rescue IOError
|
||||
end
|
||||
true
|
||||
rescue Exception, NameError
|
||||
@source = nil
|
||||
false
|
||||
end
|
||||
end
|
||||
|
||||
def read_until(term)
|
||||
pattern = Private::PRE_DEFINED_TERM_PATTERNS[term] || /#{Regexp.escape(term)}/
|
||||
term = @encoded_terms[term] ||= encode(term)
|
||||
until str = @scanner.scan_until(pattern)
|
||||
break if @source.nil?
|
||||
break if @source.eof?
|
||||
@scanner << readline(term)
|
||||
end
|
||||
if str
|
||||
read if @scanner.eos? and @source and !@source.eof?
|
||||
str
|
||||
else
|
||||
rest = @scanner.rest
|
||||
@scanner.pos = @scanner.string.bytesize
|
||||
rest
|
||||
end
|
||||
end
|
||||
|
||||
def ensure_buffer
|
||||
read if @scanner.eos? && @source
|
||||
end
|
||||
|
||||
def match( pattern, cons=false )
|
||||
# To avoid performance issue, we need to increase bytes to read per scan
|
||||
min_bytes = 1
|
||||
while true
|
||||
if cons
|
||||
md = @scanner.scan(pattern)
|
||||
else
|
||||
md = @scanner.check(pattern)
|
||||
end
|
||||
break if md
|
||||
return nil if pattern.is_a?(String)
|
||||
return nil if @source.nil?
|
||||
return nil unless read(nil, min_bytes)
|
||||
min_bytes *= 2
|
||||
end
|
||||
|
||||
md.nil? ? nil : @scanner
|
||||
end
|
||||
|
||||
def match?( pattern, cons=false )
|
||||
# To avoid performance issue, we need to increase bytes to read per scan
|
||||
min_bytes = 1
|
||||
while true
|
||||
if cons
|
||||
n_matched_bytes = @scanner.skip(pattern)
|
||||
else
|
||||
n_matched_bytes = @scanner.match?(pattern)
|
||||
end
|
||||
return true if n_matched_bytes
|
||||
return false if pattern.is_a?(String)
|
||||
return false if @source.nil?
|
||||
return false unless read(nil, min_bytes)
|
||||
min_bytes *= 2
|
||||
end
|
||||
end
|
||||
|
||||
def empty?
|
||||
super and ( @source.nil? || @source.eof? )
|
||||
end
|
||||
|
||||
# @return the current line in the source
|
||||
def current_line
|
||||
begin
|
||||
pos = @er_source.pos # The byte position in the source
|
||||
lineno = @er_source.lineno # The XML < position in the source
|
||||
@er_source.rewind
|
||||
line = 0 # The \r\n position in the source
|
||||
begin
|
||||
while @er_source.pos < pos
|
||||
@er_source.readline
|
||||
line += 1
|
||||
end
|
||||
rescue
|
||||
end
|
||||
@er_source.seek(pos)
|
||||
rescue IOError, SystemCallError
|
||||
pos = -1
|
||||
line = -1
|
||||
end
|
||||
[pos, lineno, line]
|
||||
end
|
||||
|
||||
private
|
||||
def readline(term = nil)
|
||||
if @pending_buffer
|
||||
begin
|
||||
str = @source.readline(term || @line_break)
|
||||
rescue IOError
|
||||
end
|
||||
if str.nil?
|
||||
str = @pending_buffer
|
||||
else
|
||||
str = @pending_buffer + str
|
||||
end
|
||||
@pending_buffer = nil
|
||||
else
|
||||
str = @source.readline(term || @line_break)
|
||||
end
|
||||
return nil if str.nil?
|
||||
|
||||
if @to_utf
|
||||
decode(str)
|
||||
else
|
||||
str.force_encoding(::Encoding::UTF_8) if @force_utf8
|
||||
str
|
||||
end
|
||||
end
|
||||
|
||||
def encoding_updated
|
||||
case @encoding
|
||||
when "UTF-16BE", "UTF-16LE"
|
||||
@source.binmode
|
||||
@source.set_encoding(@encoding, @encoding)
|
||||
end
|
||||
@line_break = encode(">")
|
||||
@pending_buffer, @scanner.string = @scanner.rest, ""
|
||||
@pending_buffer.force_encoding(@encoding)
|
||||
super
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,93 @@
|
||||
# frozen_string_literal: false
|
||||
module REXML
|
||||
# A template for stream parser listeners.
|
||||
# Note that the declarations (attlistdecl, elementdecl, etc) are trivially
|
||||
# processed; REXML doesn't yet handle doctype entity declarations, so you
|
||||
# have to parse them out yourself.
|
||||
module StreamListener
|
||||
# Called when a tag is encountered.
|
||||
# @p name the tag name
|
||||
# @p attrs an array of arrays of attribute/value pairs, suitable for
|
||||
# use with assoc or rassoc. IE, <tag attr1="value1" attr2="value2">
|
||||
# will result in
|
||||
# tag_start( "tag", # [["attr1","value1"],["attr2","value2"]])
|
||||
def tag_start name, attrs
|
||||
end
|
||||
# Called when the end tag is reached. In the case of <tag/>, tag_end
|
||||
# will be called immediately after tag_start
|
||||
# @p the name of the tag
|
||||
def tag_end name
|
||||
end
|
||||
# Called when text is encountered in the document
|
||||
# @p text the text content.
|
||||
def text text
|
||||
end
|
||||
# Called when an instruction is encountered. EG: <?xsl sheet='foo'?>
|
||||
# @p name the instruction name; in the example, "xsl"
|
||||
# @p instruction the rest of the instruction. In the example,
|
||||
# "sheet='foo'"
|
||||
def instruction name, instruction
|
||||
end
|
||||
# Called when a comment is encountered.
|
||||
# @p comment The content of the comment
|
||||
def comment comment
|
||||
end
|
||||
# Handles a doctype declaration. Any attributes of the doctype which are
|
||||
# not supplied will be nil. # EG, <!DOCTYPE me PUBLIC "foo" "bar">
|
||||
# @p name the name of the doctype; EG, "me"
|
||||
# @p pub_sys "PUBLIC", "SYSTEM", or nil. EG, "PUBLIC"
|
||||
# @p long_name the supplied long name, or nil. EG, "foo"
|
||||
# @p uri the uri of the doctype, or nil. EG, "bar"
|
||||
def doctype name, pub_sys, long_name, uri
|
||||
end
|
||||
# Called when the doctype is done
|
||||
def doctype_end
|
||||
end
|
||||
# If a doctype includes an ATTLIST declaration, it will cause this
|
||||
# method to be called. The content is the declaration itself, unparsed.
|
||||
# EG, <!ATTLIST el attr CDATA #REQUIRED> will come to this method as "el
|
||||
# attr CDATA #REQUIRED". This is the same for all of the .*decl
|
||||
# methods.
|
||||
def attlistdecl element_name, attributes, raw_content
|
||||
end
|
||||
# <!ELEMENT ...>
|
||||
def elementdecl content
|
||||
end
|
||||
# <!ENTITY ...>
|
||||
# The argument passed to this method is an array of the entity
|
||||
# declaration. It can be in a number of formats, but in general it
|
||||
# returns (example, result):
|
||||
# <!ENTITY % YN '"Yes"'>
|
||||
# ["YN", "\"Yes\"", "%"]
|
||||
# <!ENTITY % YN 'Yes'>
|
||||
# ["YN", "Yes", "%"]
|
||||
# <!ENTITY WhatHeSaid "He said %YN;">
|
||||
# ["WhatHeSaid", "He said %YN;"]
|
||||
# <!ENTITY open-hatch SYSTEM "http://www.textuality.com/boilerplate/OpenHatch.xml">
|
||||
# ["open-hatch", "SYSTEM", "http://www.textuality.com/boilerplate/OpenHatch.xml"]
|
||||
# <!ENTITY open-hatch PUBLIC "-//Textuality//TEXT Standard open-hatch boilerplate//EN" "http://www.textuality.com/boilerplate/OpenHatch.xml">
|
||||
# ["open-hatch", "PUBLIC", "-//Textuality//TEXT Standard open-hatch boilerplate//EN", "http://www.textuality.com/boilerplate/OpenHatch.xml"]
|
||||
# <!ENTITY hatch-pic SYSTEM "../grafix/OpenHatch.gif" NDATA gif>
|
||||
# ["hatch-pic", "SYSTEM", "../grafix/OpenHatch.gif", "gif"]
|
||||
def entitydecl content
|
||||
end
|
||||
# <!NOTATION ...>
|
||||
def notationdecl content
|
||||
end
|
||||
# Called when %foo; is encountered in a doctype declaration.
|
||||
# @p content "foo"
|
||||
def entity content
|
||||
end
|
||||
# Called when <![CDATA[ ... ]]> is encountered in a document.
|
||||
# @p content "..."
|
||||
def cdata content
|
||||
end
|
||||
# Called when an XML PI is encountered in the document.
|
||||
# EG: <?xml version="1.0" encoding="utf"?>
|
||||
# @p version the version attribute value. EG, "1.0"
|
||||
# @p encoding the encoding attribute value, or nil. EG, "utf"
|
||||
# @p standalone the standalone attribute value, or nil. EG, nil
|
||||
def xmldecl version, encoding, standalone
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,420 @@
|
||||
# frozen_string_literal: true
|
||||
require_relative 'security'
|
||||
require_relative 'entity'
|
||||
require_relative 'doctype'
|
||||
require_relative 'child'
|
||||
require_relative 'doctype'
|
||||
require_relative 'parseexception'
|
||||
|
||||
module REXML
|
||||
# Represents text nodes in an XML document
|
||||
class Text < Child
|
||||
include Comparable
|
||||
# The order in which the substitutions occur
|
||||
SPECIALS = [ /&(?!#?[\w-]+;)/u, /</u, />/u, /"/u, /'/u, /\r/u ]
|
||||
SUBSTITUTES = ['&', '<', '>', '"', ''', ' ']
|
||||
# Characters which are substituted in written strings
|
||||
SLAICEPS = [ '<', '>', '"', "'", '&' ]
|
||||
SETUTITSBUS = [ /</u, />/u, /"/u, /'/u, /&/u ]
|
||||
|
||||
# If +raw+ is true, then REXML leaves the value alone
|
||||
attr_accessor :raw
|
||||
|
||||
NEEDS_A_SECOND_CHECK = /(<|&((#{Entity::NAME});|(#0*((?:\d+)|(?:x[a-fA-F0-9]+)));)?)/um
|
||||
NUMERICENTITY = /�*((?:\d+)|(?:x[a-fA-F0-9]+));/
|
||||
VALID_CHAR = [
|
||||
0x9, 0xA, 0xD,
|
||||
(0x20..0xD7FF),
|
||||
(0xE000..0xFFFD),
|
||||
(0x10000..0x10FFFF)
|
||||
]
|
||||
|
||||
VALID_XML_CHARS = Regexp.new('^['+
|
||||
VALID_CHAR.map { |item|
|
||||
case item
|
||||
when Integer
|
||||
[item].pack('U').force_encoding('utf-8')
|
||||
when Range
|
||||
[item.first, '-'.ord, item.last].pack('UUU').force_encoding('utf-8')
|
||||
end
|
||||
}.join +
|
||||
']*$')
|
||||
|
||||
# Constructor
|
||||
# +arg+ if a String, the content is set to the String. If a Text,
|
||||
# the object is shallowly cloned.
|
||||
#
|
||||
# +respect_whitespace+ (boolean, false) if true, whitespace is
|
||||
# respected
|
||||
#
|
||||
# +parent+ (nil) if this is a Parent object, the parent
|
||||
# will be set to this.
|
||||
#
|
||||
# +raw+ (nil) This argument can be given three values.
|
||||
# If true, then the value of used to construct this object is expected to
|
||||
# contain no unescaped XML markup, and REXML will not change the text. If
|
||||
# this value is false, the string may contain any characters, and REXML will
|
||||
# escape any and all defined entities whose values are contained in the
|
||||
# text. If this value is nil (the default), then the raw value of the
|
||||
# parent will be used as the raw value for this node. If there is no raw
|
||||
# value for the parent, and no value is supplied, the default is false.
|
||||
# Use this field if you have entities defined for some text, and you don't
|
||||
# want REXML to escape that text in output.
|
||||
# Text.new( "<&", false, nil, false ) #-> "<&"
|
||||
# Text.new( "<&", false, nil, false ) #-> "&lt;&amp;"
|
||||
# Text.new( "<&", false, nil, true ) #-> Parse exception
|
||||
# Text.new( "<&", false, nil, true ) #-> "<&"
|
||||
# # Assume that the entity "s" is defined to be "sean"
|
||||
# # and that the entity "r" is defined to be "russell"
|
||||
# Text.new( "sean russell" ) #-> "&s; &r;"
|
||||
# Text.new( "sean russell", false, nil, true ) #-> "sean russell"
|
||||
#
|
||||
# +entity_filter+ (nil) This can be an array of entities to match in the
|
||||
# supplied text. This argument is only useful if +raw+ is set to false.
|
||||
# Text.new( "sean russell", false, nil, false, ["s"] ) #-> "&s; russell"
|
||||
# Text.new( "sean russell", false, nil, true, ["s"] ) #-> "sean russell"
|
||||
# In the last example, the +entity_filter+ argument is ignored.
|
||||
#
|
||||
# +illegal+ INTERNAL USE ONLY
|
||||
def initialize(arg, respect_whitespace=false, parent=nil, raw=nil,
|
||||
entity_filter=nil, illegal=NEEDS_A_SECOND_CHECK )
|
||||
|
||||
@raw = false
|
||||
@parent = nil
|
||||
@entity_filter = nil
|
||||
|
||||
if parent
|
||||
super( parent )
|
||||
@raw = parent.raw
|
||||
end
|
||||
|
||||
if arg.kind_of? String
|
||||
@string = arg.dup
|
||||
elsif arg.kind_of? Text
|
||||
@string = arg.instance_variable_get(:@string).dup
|
||||
@raw = arg.raw
|
||||
@entity_filter = arg.instance_variable_get(:@entity_filter)
|
||||
else
|
||||
raise "Illegal argument of type #{arg.type} for Text constructor (#{arg})"
|
||||
end
|
||||
|
||||
@string.squeeze!(" \n\t") unless respect_whitespace
|
||||
@string.gsub!(/\r\n?/, "\n")
|
||||
@raw = raw unless raw.nil?
|
||||
@entity_filter = entity_filter if entity_filter
|
||||
clear_cache
|
||||
|
||||
Text.check(@string, illegal) if @raw
|
||||
end
|
||||
|
||||
def parent= parent
|
||||
super(parent)
|
||||
Text.check(@string, NEEDS_A_SECOND_CHECK) if @raw and @parent
|
||||
end
|
||||
|
||||
# check for illegal characters
|
||||
def Text.check string, pattern, doctype = nil
|
||||
|
||||
# illegal anywhere
|
||||
if !string.match?(VALID_XML_CHARS)
|
||||
string.chars.each do |c|
|
||||
case c.ord
|
||||
when *VALID_CHAR
|
||||
else
|
||||
raise "Illegal character #{c.inspect} in raw string #{string.inspect}"
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
pos = 0
|
||||
while (index = string.index(/<|&/, pos))
|
||||
if string[index] == "<"
|
||||
raise "Illegal character \"#{string[index]}\" in raw string #{string.inspect}"
|
||||
end
|
||||
|
||||
unless (end_index = string.index(/[^\s];/, index + 1))
|
||||
raise "Illegal character \"#{string[index]}\" in raw string #{string.inspect}"
|
||||
end
|
||||
|
||||
value = string[(index + 1)..end_index]
|
||||
if /\s/.match?(value)
|
||||
raise "Illegal character \"#{string[index]}\" in raw string #{string.inspect}"
|
||||
end
|
||||
|
||||
if value[0] == "#"
|
||||
character_reference = value[1..-1]
|
||||
|
||||
unless (/\A(\d+|x[0-9a-fA-F]+)\z/.match?(character_reference))
|
||||
if character_reference[0] == "x" || character_reference[-1] == "x"
|
||||
raise "Illegal character \"#{string[index]}\" in raw string #{string.inspect}"
|
||||
else
|
||||
raise "Illegal character #{string.inspect} in raw string #{string.inspect}"
|
||||
end
|
||||
end
|
||||
|
||||
case (character_reference[0] == "x" ? character_reference[1..-1].to_i(16) : character_reference[0..-1].to_i)
|
||||
when *VALID_CHAR
|
||||
else
|
||||
raise "Illegal character #{string.inspect} in raw string #{string.inspect}"
|
||||
end
|
||||
elsif !(/\A#{Entity::NAME}\z/um.match?(value))
|
||||
raise "Illegal character \"#{string[index]}\" in raw string #{string.inspect}"
|
||||
end
|
||||
|
||||
pos = end_index + 1
|
||||
end
|
||||
|
||||
string
|
||||
end
|
||||
|
||||
def node_type
|
||||
:text
|
||||
end
|
||||
|
||||
def empty?
|
||||
@string.size==0
|
||||
end
|
||||
|
||||
|
||||
def clone
|
||||
Text.new(self, true)
|
||||
end
|
||||
|
||||
|
||||
# Appends text to this text node. The text is appended in the +raw+ mode
|
||||
# of this text node.
|
||||
#
|
||||
# +returns+ the text itself to enable method chain like
|
||||
# 'text << "XXX" << "YYY"'.
|
||||
def <<( to_append )
|
||||
@string << to_append.gsub( /\r\n?/, "\n" )
|
||||
clear_cache
|
||||
self
|
||||
end
|
||||
|
||||
|
||||
# +other+ a String or a Text
|
||||
# +returns+ the result of (to_s <=> arg.to_s)
|
||||
def <=>( other )
|
||||
to_s() <=> other.to_s
|
||||
end
|
||||
|
||||
def doctype
|
||||
@parent&.document&.doctype
|
||||
end
|
||||
|
||||
REFERENCE = /#{Entity::REFERENCE}/
|
||||
# Returns the string value of this text node. This string is always
|
||||
# escaped, meaning that it is a valid XML text node string, and all
|
||||
# entities that can be escaped, have been inserted. This method respects
|
||||
# the entity filter set in the constructor.
|
||||
#
|
||||
# # Assume that the entity "s" is defined to be "sean", and that the
|
||||
# # entity "r" is defined to be "russell"
|
||||
# t = Text.new( "< & sean russell", false, nil, false, ['s'] )
|
||||
# t.to_s #-> "< & &s; russell"
|
||||
# t = Text.new( "< & &s; russell", false, nil, false )
|
||||
# t.to_s #-> "< & &s; russell"
|
||||
# u = Text.new( "sean russell", false, nil, true )
|
||||
# u.to_s #-> "sean russell"
|
||||
def to_s
|
||||
return @string if @raw
|
||||
@normalized ||= Text::normalize( @string, doctype, @entity_filter )
|
||||
end
|
||||
|
||||
def inspect
|
||||
@string.inspect
|
||||
end
|
||||
|
||||
# Returns the string value of this text. This is the text without
|
||||
# entities, as it might be used programmatically, or printed to the
|
||||
# console. This ignores the 'raw' attribute setting, and any
|
||||
# entity_filter.
|
||||
#
|
||||
# # Assume that the entity "s" is defined to be "sean", and that the
|
||||
# # entity "r" is defined to be "russell"
|
||||
# t = Text.new( "< & sean russell", false, nil, false, ['s'] )
|
||||
# t.value #-> "< & sean russell"
|
||||
# t = Text.new( "< & &s; russell", false, nil, false )
|
||||
# t.value #-> "< & sean russell"
|
||||
# u = Text.new( "sean russell", false, nil, true )
|
||||
# u.value #-> "sean russell"
|
||||
def value
|
||||
@unnormalized ||= Text::unnormalize(@string, doctype,
|
||||
entity_expansion_text_limit: document&.entity_expansion_text_limit)
|
||||
end
|
||||
|
||||
# Sets the contents of this text node. This expects the text to be
|
||||
# unnormalized. It returns self.
|
||||
#
|
||||
# e = Element.new( "a" )
|
||||
# e.add_text( "foo" ) # <a>foo</a>
|
||||
# e[0].value = "bar" # <a>bar</a>
|
||||
# e[0].value = "<a>" # <a><a></a>
|
||||
def value=( val )
|
||||
@string = val.gsub( /\r\n?/, "\n" )
|
||||
clear_cache
|
||||
@raw = false
|
||||
end
|
||||
|
||||
def wrap(string, width, addnewline=false)
|
||||
# Recursively wrap string at width.
|
||||
return string if string.length <= width
|
||||
place = string.rindex(' ', width) # Position in string with last ' ' before cutoff
|
||||
if addnewline
|
||||
"\n" + string[0,place] + "\n" + wrap(string[place+1..-1], width)
|
||||
else
|
||||
string[0,place] + "\n" + wrap(string[place+1..-1], width)
|
||||
end
|
||||
end
|
||||
|
||||
def indent_text(string, level=1, style="\t", indentfirstline=true)
|
||||
Kernel.warn("#{self.class.name}#indent_text is deprecated. See REXML::Formatters", uplevel: 1)
|
||||
return string if level < 0
|
||||
|
||||
new_string = +''
|
||||
string.each_line { |line|
|
||||
indent_string = style * level
|
||||
new_line = (indent_string + line).sub(/[\s]+$/,'')
|
||||
new_string << new_line
|
||||
}
|
||||
new_string.strip! unless indentfirstline
|
||||
new_string
|
||||
end
|
||||
|
||||
# == DEPRECATED
|
||||
# See REXML::Formatters
|
||||
#
|
||||
def write( writer, indent=-1, transitive=false, ie_hack=false )
|
||||
Kernel.warn("#{self.class.name}#write is deprecated. See REXML::Formatters", uplevel: 1)
|
||||
formatter = if indent > -1
|
||||
REXML::Formatters::Pretty.new( indent )
|
||||
else
|
||||
REXML::Formatters::Default.new
|
||||
end
|
||||
formatter.write( self, writer )
|
||||
end
|
||||
|
||||
# FIXME
|
||||
# This probably won't work properly
|
||||
def xpath
|
||||
@parent.xpath + "/text()"
|
||||
end
|
||||
|
||||
# Writes out text, substituting special characters beforehand.
|
||||
# +out+ A String, IO, or any other object supporting <<( String )
|
||||
# +input+ the text to substitute and the write out
|
||||
#
|
||||
# z=utf8.unpack("U*")
|
||||
# ascOut=""
|
||||
# z.each{|r|
|
||||
# if r < 0x100
|
||||
# ascOut.concat(r.chr)
|
||||
# else
|
||||
# ascOut.concat(sprintf("&#x%x;", r))
|
||||
# end
|
||||
# }
|
||||
# puts ascOut
|
||||
def write_with_substitution out, input
|
||||
copy = input.clone
|
||||
# Doing it like this rather than in a loop improves the speed
|
||||
copy.gsub!( SPECIALS[0], SUBSTITUTES[0] )
|
||||
copy.gsub!( SPECIALS[1], SUBSTITUTES[1] )
|
||||
copy.gsub!( SPECIALS[2], SUBSTITUTES[2] )
|
||||
copy.gsub!( SPECIALS[3], SUBSTITUTES[3] )
|
||||
copy.gsub!( SPECIALS[4], SUBSTITUTES[4] )
|
||||
copy.gsub!( SPECIALS[5], SUBSTITUTES[5] )
|
||||
out << copy
|
||||
end
|
||||
|
||||
private
|
||||
def clear_cache
|
||||
@normalized = nil
|
||||
@unnormalized = nil
|
||||
end
|
||||
|
||||
# Reads text, substituting entities
|
||||
def Text::read_with_substitution( input, illegal=nil )
|
||||
copy = input.clone
|
||||
|
||||
if copy =~ illegal
|
||||
raise ParseException.new( "malformed text: Illegal character #$& in \"#{copy}\"" )
|
||||
end if illegal
|
||||
|
||||
copy.gsub!( /\r\n?/, "\n" )
|
||||
if copy.include? ?&
|
||||
copy.gsub!( SETUTITSBUS[0], SLAICEPS[0] )
|
||||
copy.gsub!( SETUTITSBUS[1], SLAICEPS[1] )
|
||||
copy.gsub!( SETUTITSBUS[2], SLAICEPS[2] )
|
||||
copy.gsub!( SETUTITSBUS[3], SLAICEPS[3] )
|
||||
copy.gsub!( SETUTITSBUS[4], SLAICEPS[4] )
|
||||
copy.gsub!( /�*((?:\d+)|(?:x[a-f0-9]+));/ ) {
|
||||
m=$1
|
||||
#m='0' if m==''
|
||||
m = "0#{m}" if m[0] == ?x
|
||||
[Integer(m)].pack('U*')
|
||||
}
|
||||
end
|
||||
copy
|
||||
end
|
||||
|
||||
EREFERENCE = /&(?!#{Entity::NAME};)/
|
||||
# Escapes all possible entities
|
||||
def Text::normalize( input, doctype=nil, entity_filter=nil )
|
||||
copy = input.to_s
|
||||
# Doing it like this rather than in a loop improves the speed
|
||||
#copy = copy.gsub( EREFERENCE, '&' )
|
||||
copy = copy.gsub( "&", "&" ) if copy.include?("&")
|
||||
if doctype
|
||||
# Replace all ampersands that aren't part of an entity
|
||||
doctype.entities.each_value do |entity|
|
||||
copy = copy.gsub( entity.value,
|
||||
"&#{entity.name};" ) if entity.value and
|
||||
not( entity_filter and entity_filter.include?(entity.name) )
|
||||
end
|
||||
else
|
||||
# Replace all ampersands that aren't part of an entity
|
||||
DocType::DEFAULT_ENTITIES.each_value do |entity|
|
||||
if copy.include?(entity.value)
|
||||
copy = copy.gsub(entity.value, "&#{entity.name};" )
|
||||
end
|
||||
end
|
||||
end
|
||||
copy
|
||||
end
|
||||
|
||||
# Unescapes all possible entities
|
||||
def Text::unnormalize( string, doctype=nil, filter=nil, illegal=nil, entity_expansion_text_limit: nil )
|
||||
entity_expansion_text_limit ||= Security.entity_expansion_text_limit
|
||||
sum = 0
|
||||
string.gsub( /\r\n?/, "\n" ).gsub( REFERENCE ) {
|
||||
s = Text.expand($&, doctype, filter)
|
||||
if sum + s.bytesize > entity_expansion_text_limit
|
||||
raise "entity expansion has grown too large"
|
||||
else
|
||||
sum += s.bytesize
|
||||
end
|
||||
s
|
||||
}
|
||||
end
|
||||
|
||||
def Text.expand(ref, doctype, filter)
|
||||
if ref[1] == ?#
|
||||
if ref[2] == ?x
|
||||
[ref[3...-1].to_i(16)].pack('U*')
|
||||
else
|
||||
[ref[2...-1].to_i].pack('U*')
|
||||
end
|
||||
elsif ref == '&'
|
||||
'&'
|
||||
elsif filter and filter.include?( ref[1...-1] )
|
||||
ref
|
||||
elsif doctype
|
||||
doctype.entity( ref[1...-1] ) or ref
|
||||
else
|
||||
entity_value = DocType::DEFAULT_ENTITIES[ ref[1...-1] ]
|
||||
entity_value ? entity_value.value : ref
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,9 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative 'parseexception'
|
||||
module REXML
|
||||
class UndefinedNamespaceException < ParseException
|
||||
def initialize( prefix, source, parser )
|
||||
super( "Undefined prefix #{prefix} found" )
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,540 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative "validation"
|
||||
require_relative "../parsers/baseparser"
|
||||
|
||||
module REXML
|
||||
module Validation
|
||||
# Implemented:
|
||||
# * empty
|
||||
# * element
|
||||
# * attribute
|
||||
# * text
|
||||
# * optional
|
||||
# * choice
|
||||
# * oneOrMore
|
||||
# * zeroOrMore
|
||||
# * group
|
||||
# * value
|
||||
# * interleave
|
||||
# * mixed
|
||||
# * ref
|
||||
# * grammar
|
||||
# * start
|
||||
# * define
|
||||
#
|
||||
# Not implemented:
|
||||
# * data
|
||||
# * param
|
||||
# * include
|
||||
# * externalRef
|
||||
# * notAllowed
|
||||
# * anyName
|
||||
# * nsName
|
||||
# * except
|
||||
# * name
|
||||
class RelaxNG
|
||||
include Validator
|
||||
|
||||
INFINITY = 1.0 / 0.0
|
||||
EMPTY = Event.new( nil )
|
||||
TEXT = [:start_element, "text"]
|
||||
attr_accessor :current
|
||||
attr_accessor :count
|
||||
attr_reader :references
|
||||
|
||||
# FIXME: Namespaces
|
||||
def initialize source
|
||||
parser = REXML::Parsers::BaseParser.new( source )
|
||||
|
||||
@count = 0
|
||||
@references = {}
|
||||
@root = @current = Sequence.new(self)
|
||||
@root.previous = true
|
||||
states = [ @current ]
|
||||
begin
|
||||
event = parser.pull
|
||||
case event[0]
|
||||
when :start_element
|
||||
case event[1]
|
||||
when "empty"
|
||||
when "element", "attribute", "text", "value"
|
||||
states[-1] << event
|
||||
when "optional"
|
||||
states << Optional.new( self )
|
||||
states[-2] << states[-1]
|
||||
when "choice"
|
||||
states << Choice.new( self )
|
||||
states[-2] << states[-1]
|
||||
when "oneOrMore"
|
||||
states << OneOrMore.new( self )
|
||||
states[-2] << states[-1]
|
||||
when "zeroOrMore"
|
||||
states << ZeroOrMore.new( self )
|
||||
states[-2] << states[-1]
|
||||
when "group"
|
||||
states << Sequence.new( self )
|
||||
states[-2] << states[-1]
|
||||
when "interleave"
|
||||
states << Interleave.new( self )
|
||||
states[-2] << states[-1]
|
||||
when "mixed"
|
||||
states << Interleave.new( self )
|
||||
states[-2] << states[-1]
|
||||
states[-1] << TEXT
|
||||
when "define"
|
||||
states << [ event[2]["name"] ]
|
||||
when "ref"
|
||||
states[-1] << Ref.new( event[2]["name"] )
|
||||
when "anyName"
|
||||
states << AnyName.new( self )
|
||||
states[-2] << states[-1]
|
||||
when "nsName"
|
||||
when "except"
|
||||
when "name"
|
||||
when "data"
|
||||
when "param"
|
||||
when "include"
|
||||
when "grammar"
|
||||
when "start"
|
||||
when "externalRef"
|
||||
when "notAllowed"
|
||||
end
|
||||
when :end_element
|
||||
case event[1]
|
||||
when "element", "attribute"
|
||||
states[-1] << event
|
||||
when "zeroOrMore", "oneOrMore", "choice", "optional",
|
||||
"interleave", "group", "mixed"
|
||||
states.pop
|
||||
when "define"
|
||||
ref = states.pop
|
||||
@references[ ref.shift ] = ref
|
||||
#when "empty"
|
||||
end
|
||||
when :end_document
|
||||
states[-1] << event
|
||||
when :text
|
||||
states[-1] << event
|
||||
end
|
||||
end while event[0] != :end_document
|
||||
end
|
||||
|
||||
def receive event
|
||||
validate( event )
|
||||
end
|
||||
end
|
||||
|
||||
class State
|
||||
def initialize( context )
|
||||
@previous = []
|
||||
@events = []
|
||||
@current = 0
|
||||
@count = context.count += 1
|
||||
@references = context.references
|
||||
@value = false
|
||||
end
|
||||
|
||||
def reset
|
||||
return if @current == 0
|
||||
@current = 0
|
||||
@events.each {|s| s.reset if s.kind_of? State }
|
||||
end
|
||||
|
||||
def previous=( previous )
|
||||
@previous << previous
|
||||
end
|
||||
|
||||
def next( event )
|
||||
#print "In next with #{event.inspect}. "
|
||||
#p @previous
|
||||
return @previous.pop.next( event ) if @events[@current].nil?
|
||||
expand_ref_in( @events, @current ) if @events[@current].class == Ref
|
||||
if ( @events[@current].kind_of? State )
|
||||
@current += 1
|
||||
@events[@current-1].previous = self
|
||||
return @events[@current-1].next( event )
|
||||
end
|
||||
if ( @events[@current].matches?(event) )
|
||||
@current += 1
|
||||
if @events[@current].nil?
|
||||
@previous.pop
|
||||
elsif @events[@current].kind_of? State
|
||||
@current += 1
|
||||
@events[@current-1].previous = self
|
||||
@events[@current-1]
|
||||
else
|
||||
self
|
||||
end
|
||||
else
|
||||
nil
|
||||
end
|
||||
end
|
||||
|
||||
def to_s
|
||||
# Abbreviated:
|
||||
self.class.name =~ /(?:::)(\w)\w+$/
|
||||
# Full:
|
||||
#self.class.name =~ /(?:::)(\w+)$/
|
||||
"#$1.#@count"
|
||||
end
|
||||
|
||||
def inspect
|
||||
"< #{to_s} #{@events.collect{|e|
|
||||
pre = e == @events[@current] ? '#' : ''
|
||||
pre + e.inspect unless self == e
|
||||
}.join(', ')} >"
|
||||
end
|
||||
|
||||
def expected
|
||||
[@events[@current]]
|
||||
end
|
||||
|
||||
def <<( event )
|
||||
add_event_to_arry( @events, event )
|
||||
end
|
||||
|
||||
|
||||
protected
|
||||
def expand_ref_in( arry, ind )
|
||||
new_events = []
|
||||
@references[ arry[ind].to_s ].each{ |evt|
|
||||
add_event_to_arry(new_events,evt)
|
||||
}
|
||||
arry[ind,1] = new_events
|
||||
end
|
||||
|
||||
def add_event_to_arry( arry, evt )
|
||||
evt = generate_event( evt )
|
||||
if evt.kind_of? String
|
||||
arry[-1].event_arg = evt if arry[-1].kind_of? Event and @value
|
||||
@value = false
|
||||
else
|
||||
arry << evt
|
||||
end
|
||||
end
|
||||
|
||||
def generate_event( event )
|
||||
return event if event.kind_of? State or event.class == Ref
|
||||
evt = nil
|
||||
arg = nil
|
||||
case event[0]
|
||||
when :start_element
|
||||
case event[1]
|
||||
when "element"
|
||||
evt = :start_element
|
||||
arg = event[2]["name"]
|
||||
when "attribute"
|
||||
evt = :start_attribute
|
||||
arg = event[2]["name"]
|
||||
when "text"
|
||||
evt = :text
|
||||
when "value"
|
||||
evt = :text
|
||||
@value = true
|
||||
end
|
||||
when :text
|
||||
return event[1]
|
||||
when :end_document
|
||||
return Event.new( event[0] )
|
||||
else # then :end_element
|
||||
case event[1]
|
||||
when "element"
|
||||
evt = :end_element
|
||||
when "attribute"
|
||||
evt = :end_attribute
|
||||
end
|
||||
end
|
||||
Event.new( evt, arg )
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
class Sequence < State
|
||||
def matches?(event)
|
||||
@events[@current].matches?( event )
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
class Optional < State
|
||||
def next( event )
|
||||
if @current == 0
|
||||
rv = super
|
||||
return rv if rv
|
||||
@prior = @previous.pop
|
||||
@prior.next( event )
|
||||
else
|
||||
super
|
||||
end
|
||||
end
|
||||
|
||||
def matches?(event)
|
||||
@events[@current].matches?(event) ||
|
||||
(@current == 0 and @previous[-1].matches?(event))
|
||||
end
|
||||
|
||||
def expected
|
||||
return [ @prior.expected, @events[0] ].flatten if @current == 0
|
||||
[@events[@current]]
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
class ZeroOrMore < Optional
|
||||
def next( event )
|
||||
expand_ref_in( @events, @current ) if @events[@current].class == Ref
|
||||
if ( @events[@current].matches?(event) )
|
||||
@current += 1
|
||||
if @events[@current].nil?
|
||||
@current = 0
|
||||
self
|
||||
elsif @events[@current].kind_of? State
|
||||
@current += 1
|
||||
@events[@current-1].previous = self
|
||||
@events[@current-1]
|
||||
else
|
||||
self
|
||||
end
|
||||
else
|
||||
@prior = @previous.pop
|
||||
return @prior.next( event ) if @current == 0
|
||||
nil
|
||||
end
|
||||
end
|
||||
|
||||
def expected
|
||||
return [ @prior.expected, @events[0] ].flatten if @current == 0
|
||||
[@events[@current]]
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
class OneOrMore < State
|
||||
def initialize context
|
||||
super
|
||||
@ord = 0
|
||||
end
|
||||
|
||||
def reset
|
||||
super
|
||||
@ord = 0
|
||||
end
|
||||
|
||||
def next( event )
|
||||
expand_ref_in( @events, @current ) if @events[@current].class == Ref
|
||||
if ( @events[@current].matches?(event) )
|
||||
@current += 1
|
||||
@ord += 1
|
||||
if @events[@current].nil?
|
||||
@current = 0
|
||||
self
|
||||
elsif @events[@current].kind_of? State
|
||||
@current += 1
|
||||
@events[@current-1].previous = self
|
||||
@events[@current-1]
|
||||
else
|
||||
self
|
||||
end
|
||||
else
|
||||
return @previous.pop.next( event ) if @current == 0 and @ord > 0
|
||||
nil
|
||||
end
|
||||
end
|
||||
|
||||
def matches?( event )
|
||||
@events[@current].matches?(event) ||
|
||||
(@current == 0 and @ord > 0 and @previous[-1].matches?(event))
|
||||
end
|
||||
|
||||
def expected
|
||||
if @current == 0 and @ord > 0
|
||||
[@previous[-1].expected, @events[0]].flatten
|
||||
else
|
||||
[@events[@current]]
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
class Choice < State
|
||||
def initialize context
|
||||
super
|
||||
@choices = []
|
||||
end
|
||||
|
||||
def reset
|
||||
super
|
||||
@events = []
|
||||
@choices.each { |c| c.each { |s| s.reset if s.kind_of? State } }
|
||||
end
|
||||
|
||||
def <<( event )
|
||||
add_event_to_arry( @choices, event )
|
||||
end
|
||||
|
||||
def next( event )
|
||||
# Make the choice if we haven't
|
||||
if @events.size == 0
|
||||
c = 0 ; max = @choices.size
|
||||
while c < max
|
||||
if @choices[c][0].class == Ref
|
||||
expand_ref_in( @choices[c], 0 )
|
||||
@choices += @choices[c]
|
||||
@choices.delete( @choices[c] )
|
||||
max -= 1
|
||||
else
|
||||
c += 1
|
||||
end
|
||||
end
|
||||
@events = @choices.find { |evt| evt[0].matches? event }
|
||||
# Remove the references
|
||||
# Find the events
|
||||
end
|
||||
unless @events
|
||||
@events = []
|
||||
return nil
|
||||
end
|
||||
super
|
||||
end
|
||||
|
||||
def matches?( event )
|
||||
return @events[@current].matches?( event ) if @events.size > 0
|
||||
!@choices.find{|evt| evt[0].matches?(event)}.nil?
|
||||
end
|
||||
|
||||
def expected
|
||||
return [@events[@current]] if @events.size > 0
|
||||
@choices.collect do |x|
|
||||
if x[0].kind_of? State
|
||||
x[0].expected
|
||||
else
|
||||
x[0]
|
||||
end
|
||||
end.flatten
|
||||
end
|
||||
|
||||
def inspect
|
||||
"< #{to_s} #{@choices.collect{|e| e.collect{|f|f.to_s}.join(', ')}.join(' or ')} >"
|
||||
end
|
||||
|
||||
protected
|
||||
def add_event_to_arry( arry, evt )
|
||||
if evt.kind_of? State or evt.class == Ref
|
||||
arry << [evt]
|
||||
elsif evt[0] == :text
|
||||
if arry[-1] and
|
||||
arry[-1][-1].kind_of?( Event ) and
|
||||
arry[-1][-1].event_type == :text and @value
|
||||
|
||||
arry[-1][-1].event_arg = evt[1]
|
||||
@value = false
|
||||
end
|
||||
else
|
||||
arry << [] if evt[0] == :start_element
|
||||
arry[-1] << generate_event( evt )
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
class Interleave < Choice
|
||||
def initialize context
|
||||
super
|
||||
@choice = 0
|
||||
end
|
||||
|
||||
def reset
|
||||
@choice = 0
|
||||
end
|
||||
|
||||
def next_current( event )
|
||||
# Expand references
|
||||
c = 0 ; max = @choices.size
|
||||
while c < max
|
||||
if @choices[c][0].class == Ref
|
||||
expand_ref_in( @choices[c], 0 )
|
||||
@choices += @choices[c]
|
||||
@choices.delete( @choices[c] )
|
||||
max -= 1
|
||||
else
|
||||
c += 1
|
||||
end
|
||||
end
|
||||
@events = @choices[@choice..-1].find { |evt| evt[0].matches? event }
|
||||
@current = 0
|
||||
if @events
|
||||
# reorder the choices
|
||||
old = @choices[@choice]
|
||||
idx = @choices.index( @events )
|
||||
@choices[@choice] = @events
|
||||
@choices[idx] = old
|
||||
@choice += 1
|
||||
end
|
||||
|
||||
@events = [] unless @events
|
||||
end
|
||||
|
||||
|
||||
def next( event )
|
||||
# Find the next series
|
||||
next_current(event) unless @events[@current]
|
||||
return nil unless @events[@current]
|
||||
|
||||
expand_ref_in( @events, @current ) if @events[@current].class == Ref
|
||||
if ( @events[@current].kind_of? State )
|
||||
@current += 1
|
||||
@events[@current-1].previous = self
|
||||
return @events[@current-1].next( event )
|
||||
end
|
||||
return @previous.pop.next( event ) if @events[@current].nil?
|
||||
if ( @events[@current].matches?(event) )
|
||||
@current += 1
|
||||
if @events[@current].nil?
|
||||
return self unless @choices[@choice].nil?
|
||||
@previous.pop
|
||||
elsif @events[@current].kind_of? State
|
||||
@current += 1
|
||||
@events[@current-1].previous = self
|
||||
@events[@current-1]
|
||||
else
|
||||
self
|
||||
end
|
||||
else
|
||||
nil
|
||||
end
|
||||
end
|
||||
|
||||
def matches?( event )
|
||||
return @events[@current].matches?( event ) if @events[@current]
|
||||
!@choices[@choice..-1].find{|evt| evt[0].matches?(event)}.nil?
|
||||
end
|
||||
|
||||
def expected
|
||||
return [@events[@current]] if @events[@current]
|
||||
@choices[@choice..-1].collect do |x|
|
||||
if x[0].kind_of? State
|
||||
x[0].expected
|
||||
else
|
||||
x[0]
|
||||
end
|
||||
end.flatten
|
||||
end
|
||||
|
||||
def inspect
|
||||
"< #{to_s} #{@choices.collect{|e| e.collect{|f|f.to_s}.join(', ')}.join(' and ')} >"
|
||||
end
|
||||
end
|
||||
|
||||
class Ref
|
||||
def initialize value
|
||||
@value = value
|
||||
end
|
||||
def to_s
|
||||
@value
|
||||
end
|
||||
def inspect
|
||||
"{#{to_s}}"
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,144 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative 'validationexception'
|
||||
|
||||
module REXML
|
||||
module Validation
|
||||
module Validator
|
||||
NILEVENT = [ nil ]
|
||||
def reset
|
||||
@current = @root
|
||||
@root.reset
|
||||
@root.previous = true
|
||||
@attr_stack = []
|
||||
self
|
||||
end
|
||||
def dump
|
||||
puts @root.inspect
|
||||
end
|
||||
def validate( event )
|
||||
@attr_stack = [] unless defined? @attr_stack
|
||||
match = @current.next(event)
|
||||
raise ValidationException.new( "Validation error. Expected: "+
|
||||
@current.expected.join( " or " )+" from #{@current.inspect} "+
|
||||
" but got #{Event.new( event[0], event[1] ).inspect}" ) unless match
|
||||
@current = match
|
||||
|
||||
# Check for attributes
|
||||
case event[0]
|
||||
when :start_element
|
||||
@attr_stack << event[2]
|
||||
begin
|
||||
sattr = [:start_attribute, nil]
|
||||
eattr = [:end_attribute]
|
||||
text = [:text, nil]
|
||||
k, = event[2].find { |key,value|
|
||||
sattr[1] = key
|
||||
m = @current.next( sattr )
|
||||
if m
|
||||
# If the state has text children...
|
||||
if m.matches?( eattr )
|
||||
@current = m
|
||||
else
|
||||
text[1] = value
|
||||
m = m.next( text )
|
||||
text[1] = nil
|
||||
return false unless m
|
||||
@current = m if m
|
||||
end
|
||||
m = @current.next( eattr )
|
||||
if m
|
||||
@current = m
|
||||
true
|
||||
else
|
||||
false
|
||||
end
|
||||
else
|
||||
false
|
||||
end
|
||||
}
|
||||
event[2].delete(k) if k
|
||||
end while k
|
||||
when :end_element
|
||||
attrs = @attr_stack.pop
|
||||
raise ValidationException.new( "Validation error. Illegal "+
|
||||
" attributes: #{attrs.inspect}") if attrs.length > 0
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
class Event
|
||||
def initialize(event_type, event_arg=nil )
|
||||
@event_type = event_type
|
||||
@event_arg = event_arg
|
||||
end
|
||||
|
||||
attr_reader :event_type
|
||||
attr_accessor :event_arg
|
||||
|
||||
def done?
|
||||
@done
|
||||
end
|
||||
|
||||
def single?
|
||||
(@event_type != :start_element and @event_type != :start_attribute)
|
||||
end
|
||||
|
||||
def matches?( event )
|
||||
return false unless event[0] == @event_type
|
||||
case event[0]
|
||||
when nil
|
||||
true
|
||||
when :start_element
|
||||
event[1] == @event_arg
|
||||
when :end_element
|
||||
true
|
||||
when :start_attribute
|
||||
event[1] == @event_arg
|
||||
when :end_attribute
|
||||
true
|
||||
when :end_document
|
||||
true
|
||||
when :text
|
||||
@event_arg.nil? || @event_arg == event[1]
|
||||
=begin
|
||||
when :processing_instruction
|
||||
false
|
||||
when :xmldecl
|
||||
false
|
||||
when :start_doctype
|
||||
false
|
||||
when :end_doctype
|
||||
false
|
||||
when :externalentity
|
||||
false
|
||||
when :elementdecl
|
||||
false
|
||||
when :entity
|
||||
false
|
||||
when :attlistdecl
|
||||
false
|
||||
when :notationdecl
|
||||
false
|
||||
when :end_doctype
|
||||
false
|
||||
=end
|
||||
else
|
||||
false
|
||||
end
|
||||
end
|
||||
|
||||
def ==( other )
|
||||
return false unless other.kind_of? Event
|
||||
@event_type == other.event_type and @event_arg == other.event_arg
|
||||
end
|
||||
|
||||
def to_s
|
||||
inspect
|
||||
end
|
||||
|
||||
def inspect
|
||||
"#{@event_type.inspect}( #@event_arg )"
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,10 @@
|
||||
# frozen_string_literal: false
|
||||
module REXML
|
||||
module Validation
|
||||
class ValidationException < RuntimeError
|
||||
def initialize msg
|
||||
super
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,130 @@
|
||||
# frozen_string_literal: false
|
||||
|
||||
require_relative 'encoding'
|
||||
require_relative 'source'
|
||||
|
||||
module REXML
|
||||
# NEEDS DOCUMENTATION
|
||||
class XMLDecl < Child
|
||||
include Encoding
|
||||
|
||||
DEFAULT_VERSION = "1.0"
|
||||
DEFAULT_ENCODING = "UTF-8"
|
||||
DEFAULT_STANDALONE = "no"
|
||||
START = "<?xml"
|
||||
STOP = "?>"
|
||||
|
||||
attr_accessor :version, :standalone
|
||||
attr_reader :writeencoding, :writethis
|
||||
|
||||
def initialize(version=DEFAULT_VERSION, encoding=nil, standalone=nil)
|
||||
@writethis = true
|
||||
@writeencoding = !encoding.nil?
|
||||
if version.kind_of? XMLDecl
|
||||
super()
|
||||
@version = version.version
|
||||
self.encoding = version.encoding
|
||||
@writeencoding = version.writeencoding
|
||||
@standalone = version.standalone
|
||||
@writethis = version.writethis
|
||||
else
|
||||
super()
|
||||
@version = version
|
||||
self.encoding = encoding
|
||||
@standalone = standalone
|
||||
end
|
||||
@version = DEFAULT_VERSION if @version.nil?
|
||||
end
|
||||
|
||||
def clone
|
||||
XMLDecl.new(self)
|
||||
end
|
||||
|
||||
# indent::
|
||||
# Ignored. There must be no whitespace before an XML declaration
|
||||
# transitive::
|
||||
# Ignored
|
||||
# ie_hack::
|
||||
# Ignored
|
||||
def write(writer, indent=-1, transitive=false, ie_hack=false)
|
||||
return nil unless @writethis or writer.kind_of? Output
|
||||
writer << START
|
||||
writer << " #{content encoding}"
|
||||
writer << STOP
|
||||
end
|
||||
|
||||
def ==( other )
|
||||
other.kind_of?(XMLDecl) and
|
||||
other.version == @version and
|
||||
other.encoding == self.encoding and
|
||||
other.standalone == @standalone
|
||||
end
|
||||
|
||||
def xmldecl version, encoding, standalone
|
||||
@version = version
|
||||
self.encoding = encoding
|
||||
@standalone = standalone
|
||||
end
|
||||
|
||||
def node_type
|
||||
:xmldecl
|
||||
end
|
||||
|
||||
alias :stand_alone? :standalone
|
||||
alias :old_enc= :encoding=
|
||||
|
||||
def encoding=( enc )
|
||||
if enc.nil?
|
||||
self.old_enc = "UTF-8"
|
||||
@writeencoding = false
|
||||
else
|
||||
self.old_enc = enc
|
||||
@writeencoding = true
|
||||
end
|
||||
self.dowrite
|
||||
end
|
||||
|
||||
# Only use this if you do not want the XML declaration to be written;
|
||||
# this object is ignored by the XML writer. Otherwise, instantiate your
|
||||
# own XMLDecl and add it to the document.
|
||||
#
|
||||
# Note that XML 1.1 documents *must* include an XML declaration
|
||||
def XMLDecl.default
|
||||
rv = XMLDecl.new( "1.0" )
|
||||
rv.nowrite
|
||||
rv
|
||||
end
|
||||
|
||||
def nowrite
|
||||
@writethis = false
|
||||
end
|
||||
|
||||
def dowrite
|
||||
@writethis = true
|
||||
end
|
||||
|
||||
def inspect
|
||||
"#{START} ... #{STOP}"
|
||||
end
|
||||
|
||||
private
|
||||
def content(enc)
|
||||
context = nil
|
||||
context = parent.context if parent
|
||||
if context and context[:prologue_quote] == :quote
|
||||
quote = "\""
|
||||
else
|
||||
quote = "'"
|
||||
end
|
||||
|
||||
rv = "version=#{quote}#{@version}#{quote}"
|
||||
if @writeencoding or enc !~ /\Autf-8\z/i
|
||||
rv << " encoding=#{quote}#{enc}#{quote}"
|
||||
end
|
||||
if @standalone
|
||||
rv << " standalone=#{quote}#{@standalone}#{quote}"
|
||||
end
|
||||
rv
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,85 @@
|
||||
# frozen_string_literal: false
|
||||
module REXML
|
||||
# Defines a number of tokens used for parsing XML. Not for general
|
||||
# consumption.
|
||||
module XMLTokens
|
||||
# From http://www.w3.org/TR/REC-xml/#sec-common-syn
|
||||
#
|
||||
# [4] NameStartChar ::=
|
||||
# ":" |
|
||||
# [A-Z] |
|
||||
# "_" |
|
||||
# [a-z] |
|
||||
# [#xC0-#xD6] |
|
||||
# [#xD8-#xF6] |
|
||||
# [#xF8-#x2FF] |
|
||||
# [#x370-#x37D] |
|
||||
# [#x37F-#x1FFF] |
|
||||
# [#x200C-#x200D] |
|
||||
# [#x2070-#x218F] |
|
||||
# [#x2C00-#x2FEF] |
|
||||
# [#x3001-#xD7FF] |
|
||||
# [#xF900-#xFDCF] |
|
||||
# [#xFDF0-#xFFFD] |
|
||||
# [#x10000-#xEFFFF]
|
||||
name_start_chars = [
|
||||
":",
|
||||
"A-Z",
|
||||
"_",
|
||||
"a-z",
|
||||
"\\u00C0-\\u00D6",
|
||||
"\\u00D8-\\u00F6",
|
||||
"\\u00F8-\\u02FF",
|
||||
"\\u0370-\\u037D",
|
||||
"\\u037F-\\u1FFF",
|
||||
"\\u200C-\\u200D",
|
||||
"\\u2070-\\u218F",
|
||||
"\\u2C00-\\u2FEF",
|
||||
"\\u3001-\\uD7FF",
|
||||
"\\uF900-\\uFDCF",
|
||||
"\\uFDF0-\\uFFFD",
|
||||
"\\u{10000}-\\u{EFFFF}",
|
||||
]
|
||||
# From http://www.w3.org/TR/REC-xml/#sec-common-syn
|
||||
#
|
||||
# [4a] NameChar ::=
|
||||
# NameStartChar |
|
||||
# "-" |
|
||||
# "." |
|
||||
# [0-9] |
|
||||
# #xB7 |
|
||||
# [#x0300-#x036F] |
|
||||
# [#x203F-#x2040]
|
||||
name_chars = name_start_chars + [
|
||||
"\\-",
|
||||
"\\.",
|
||||
"0-9",
|
||||
"\\u00B7",
|
||||
"\\u0300-\\u036F",
|
||||
"\\u203F-\\u2040",
|
||||
]
|
||||
NAME_START_CHAR = "[#{name_start_chars.join('')}]"
|
||||
NAME_CHAR = "[#{name_chars.join('')}]"
|
||||
NAMECHAR = NAME_CHAR # deprecated. Use NAME_CHAR instead.
|
||||
|
||||
# From http://www.w3.org/TR/xml-names11/#NT-NCName
|
||||
#
|
||||
# [6] NCNameStartChar ::= NameStartChar - ':'
|
||||
ncname_start_chars = name_start_chars - [":"]
|
||||
# From http://www.w3.org/TR/xml-names11/#NT-NCName
|
||||
#
|
||||
# [5] NCNameChar ::= NameChar - ':'
|
||||
ncname_chars = name_chars - [":"]
|
||||
NCNAME_STR = "[#{ncname_start_chars.join('')}][#{ncname_chars.join('')}]*"
|
||||
NAME_STR = "(?:#{NCNAME_STR}:)?#{NCNAME_STR}"
|
||||
|
||||
NAME = "(#{NAME_START_CHAR}#{NAME_CHAR}*)"
|
||||
NMTOKEN = "(?:#{NAME_CHAR})+"
|
||||
NMTOKENS = "#{NMTOKEN}(\\s+#{NMTOKEN})*"
|
||||
REFERENCE = "(?:&#{NAME};|&#\\d+;|&#x[0-9a-fA-F]+;)"
|
||||
|
||||
#REFERENCE = "(?:#{ENTITYREF}|#{CHARREF})"
|
||||
#ENTITYREF = "&#{NAME};"
|
||||
#CHARREF = "&#\\d+;|&#x[0-9a-fA-F]+;"
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,70 @@
|
||||
# frozen_string_literal: false
|
||||
require_relative 'functions'
|
||||
require_relative 'xpath_parser'
|
||||
|
||||
module REXML
|
||||
# Wrapper class. Use this class to access the XPath functions.
|
||||
class XPath
|
||||
include Functions
|
||||
# A base Hash object, supposing to be used when initializing a
|
||||
# default empty namespaces set, but is currently unused.
|
||||
# TODO: either set the namespaces=EMPTY_HASH, or deprecate this.
|
||||
EMPTY_HASH = {}
|
||||
|
||||
# Finds and returns the first node that matches the supplied xpath.
|
||||
# element::
|
||||
# The context element
|
||||
# path::
|
||||
# The xpath to search for. If not supplied or nil, returns the first
|
||||
# node matching '*'.
|
||||
# namespaces::
|
||||
# If supplied, a Hash which defines a namespace mapping.
|
||||
# variables::
|
||||
# If supplied, a Hash which maps $variables in the query
|
||||
# to values. This can be used to avoid XPath injection attacks
|
||||
# or to automatically handle escaping string values.
|
||||
#
|
||||
# XPath.first( node )
|
||||
# XPath.first( doc, "//b"} )
|
||||
# XPath.first( node, "a/x:b", { "x"=>"http://doofus" } )
|
||||
# XPath.first( node, '/book/publisher/text()=$publisher', {}, {"publisher"=>"O'Reilly"})
|
||||
def XPath::first(element, path=nil, namespaces=nil, variables={}, options={})
|
||||
raise "The namespaces argument, if supplied, must be a hash object." unless namespaces.nil? or namespaces.kind_of?(Hash)
|
||||
raise "The variables argument, if supplied, must be a hash object." unless variables.kind_of?(Hash)
|
||||
match(element, path, namespaces, variables, options).flatten[0]
|
||||
end
|
||||
|
||||
# Iterates over nodes that match the given path, calling the supplied
|
||||
# block with the match.
|
||||
# element::
|
||||
# The context element
|
||||
# path::
|
||||
# The xpath to search for. If not supplied or nil, defaults to '*'
|
||||
# namespaces::
|
||||
# If supplied, a Hash which defines a namespace mapping
|
||||
# variables::
|
||||
# If supplied, a Hash which maps $variables in the query
|
||||
# to values. This can be used to avoid XPath injection attacks
|
||||
# or to automatically handle escaping string values.
|
||||
#
|
||||
# XPath.each( node ) { |el| ... }
|
||||
# XPath.each( node, '/*[@attr='v']' ) { |el| ... }
|
||||
# XPath.each( node, 'ancestor::x' ) { |el| ... }
|
||||
# XPath.each( node, '/book/publisher/text()=$publisher', {}, {"publisher"=>"O'Reilly"}) \
|
||||
# {|el| ... }
|
||||
def XPath::each(element, path=nil, namespaces=nil, variables={}, options={}, &block)
|
||||
raise "The namespaces argument, if supplied, must be a hash object." unless namespaces.nil? or namespaces.kind_of?(Hash)
|
||||
raise "The variables argument, if supplied, must be a hash object." unless variables.kind_of?(Hash)
|
||||
match(element, path, namespaces, variables, options).each( &block )
|
||||
end
|
||||
|
||||
# Returns an array of nodes matching a given XPath.
|
||||
def XPath::match(element, path=nil, namespaces=nil, variables={}, options={})
|
||||
parser = XPathParser.new(**options)
|
||||
parser.namespaces = namespaces
|
||||
parser.variables = variables
|
||||
path = "*" unless path
|
||||
parser.parse(path,element)
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,980 @@
|
||||
# frozen_string_literal: false
|
||||
|
||||
require "pp"
|
||||
|
||||
require_relative 'namespace'
|
||||
require_relative 'xmltokens'
|
||||
require_relative 'attribute'
|
||||
require_relative 'parsers/xpathparser'
|
||||
|
||||
module REXML
|
||||
module DClonable
|
||||
refine Object do
|
||||
# provides a unified +clone+ operation, for REXML::XPathParser
|
||||
# to use across multiple Object types
|
||||
def dclone
|
||||
clone
|
||||
end
|
||||
end
|
||||
refine Symbol do
|
||||
# provides a unified +clone+ operation, for REXML::XPathParser
|
||||
# to use across multiple Object types
|
||||
def dclone ; self ; end
|
||||
end
|
||||
refine Integer do
|
||||
# provides a unified +clone+ operation, for REXML::XPathParser
|
||||
# to use across multiple Object types
|
||||
def dclone ; self ; end
|
||||
end
|
||||
refine Float do
|
||||
# provides a unified +clone+ operation, for REXML::XPathParser
|
||||
# to use across multiple Object types
|
||||
def dclone ; self ; end
|
||||
end
|
||||
refine Array do
|
||||
# provides a unified +clone+ operation, for REXML::XPathParser
|
||||
# to use across multiple Object+ types
|
||||
def dclone
|
||||
klone = self.clone
|
||||
klone.clear
|
||||
self.each{|v| klone << v.dclone}
|
||||
klone
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
using REXML::DClonable
|
||||
|
||||
module REXML
|
||||
# You don't want to use this class. Really. Use XPath, which is a wrapper
|
||||
# for this class. Believe me. You don't want to poke around in here.
|
||||
# There is strange, dark magic at work in this code. Beware. Go back! Go
|
||||
# back while you still can!
|
||||
class XPathParser
|
||||
include XMLTokens
|
||||
LITERAL = /^'([^']*)'|^"([^"]*)"/u
|
||||
|
||||
DEBUG = (ENV["REXML_XPATH_PARSER_DEBUG"] == "true")
|
||||
|
||||
def initialize(strict: false)
|
||||
@debug = DEBUG
|
||||
@parser = REXML::Parsers::XPathParser.new
|
||||
@namespaces = nil
|
||||
@variables = {}
|
||||
@nest = 0
|
||||
@strict = strict
|
||||
end
|
||||
|
||||
def namespaces=( namespaces={} )
|
||||
Functions::namespace_context = namespaces
|
||||
@namespaces = namespaces
|
||||
end
|
||||
|
||||
def variables=( vars={} )
|
||||
Functions::variables = vars
|
||||
@variables = vars
|
||||
end
|
||||
|
||||
def parse path, node
|
||||
path_stack = @parser.parse( path )
|
||||
if node.is_a?(Array)
|
||||
Kernel.warn("REXML::XPath.each, REXML::XPath.first, REXML::XPath.match dropped support for nodeset...", uplevel: 1)
|
||||
return [] if node.empty?
|
||||
node = node.first
|
||||
end
|
||||
|
||||
document = node.document
|
||||
if document
|
||||
document.__send__(:enable_cache) do
|
||||
match( path_stack, node )
|
||||
end
|
||||
else
|
||||
match( path_stack, node )
|
||||
end
|
||||
end
|
||||
|
||||
def get_first path, node
|
||||
path_stack = @parser.parse( path )
|
||||
first( path_stack, node )
|
||||
end
|
||||
|
||||
def predicate path, node
|
||||
path_stack = @parser.parse( path )
|
||||
match( path_stack, node )
|
||||
end
|
||||
|
||||
def []=( variable_name, value )
|
||||
@variables[ variable_name ] = value
|
||||
end
|
||||
|
||||
|
||||
# Performs a depth-first (document order) XPath search, and returns the
|
||||
# first match. This is the fastest, lightest way to return a single result.
|
||||
#
|
||||
# FIXME: This method is incomplete!
|
||||
def first( path_stack, node )
|
||||
return nil if path.size == 0
|
||||
|
||||
case path[0]
|
||||
when :document
|
||||
# do nothing
|
||||
first( path[1..-1], node )
|
||||
when :child
|
||||
for c in node.children
|
||||
r = first( path[1..-1], c )
|
||||
return r if r
|
||||
end
|
||||
when :qname
|
||||
name = path[2]
|
||||
if node.name == name
|
||||
return node if path.size == 3
|
||||
first( path[3..-1], node )
|
||||
else
|
||||
nil
|
||||
end
|
||||
when :descendant_or_self
|
||||
r = first( path[1..-1], node )
|
||||
return r if r
|
||||
for c in node.children
|
||||
r = first( path, c )
|
||||
return r if r
|
||||
end
|
||||
when :node
|
||||
first( path[1..-1], node )
|
||||
when :any
|
||||
first( path[1..-1], node )
|
||||
else
|
||||
nil
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
def match(path_stack, node)
|
||||
nodeset = [XPathNode.new(node, position: 1)]
|
||||
result = expr(path_stack, nodeset)
|
||||
case result
|
||||
when Array # nodeset
|
||||
unnode(result).uniq
|
||||
else
|
||||
[result]
|
||||
end
|
||||
end
|
||||
|
||||
private
|
||||
def strict?
|
||||
@strict
|
||||
end
|
||||
|
||||
# Returns a String namespace for a node, given a prefix
|
||||
# The rules are:
|
||||
#
|
||||
# 1. Use the supplied namespace mapping first.
|
||||
# 2. If no mapping was supplied, use the context node to look up the namespace
|
||||
def get_namespace( node, prefix )
|
||||
if @namespaces
|
||||
@namespaces[prefix] || ''
|
||||
else
|
||||
return node.namespace( prefix ) if node.node_type == :element
|
||||
''
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
# Expr takes a stack of path elements and a set of nodes (either a Parent
|
||||
# or an Array and returns an Array of matching nodes
|
||||
def expr( path_stack, nodeset, context=nil )
|
||||
enter(:expr, path_stack, nodeset) if @debug
|
||||
return nodeset if path_stack.length == 0 || nodeset.length == 0
|
||||
while path_stack.length > 0
|
||||
trace(:while, path_stack, nodeset) if @debug
|
||||
if nodeset.length == 0
|
||||
path_stack.clear
|
||||
return []
|
||||
end
|
||||
op = path_stack.shift
|
||||
case op
|
||||
when :document
|
||||
first_raw_node = nodeset.first.raw_node
|
||||
nodeset = [XPathNode.new(first_raw_node.root_node, position: 1)]
|
||||
when :self
|
||||
nodeset = step(path_stack) do
|
||||
[nodeset]
|
||||
end
|
||||
when :child
|
||||
nodeset = step(path_stack) do
|
||||
child(nodeset)
|
||||
end
|
||||
when :literal
|
||||
trace(:literal, path_stack, nodeset) if @debug
|
||||
return path_stack.shift
|
||||
when :attribute
|
||||
nodeset = step(path_stack, any_type: :attribute) do
|
||||
nodesets = []
|
||||
nodeset.each do |node|
|
||||
raw_node = node.raw_node
|
||||
next unless raw_node.node_type == :element
|
||||
attributes = raw_node.attributes
|
||||
next if attributes.empty?
|
||||
nodesets << attributes.each_attribute.collect.with_index do |attribute, i|
|
||||
XPathNode.new(attribute, position: i + 1)
|
||||
end
|
||||
end
|
||||
nodesets
|
||||
end
|
||||
when :namespace
|
||||
pre_defined_namespaces = {
|
||||
"xml" => "http://www.w3.org/XML/1998/namespace",
|
||||
}
|
||||
nodeset = step(path_stack, any_type: :namespace) do
|
||||
nodesets = []
|
||||
nodeset.each do |node|
|
||||
raw_node = node.raw_node
|
||||
case raw_node.node_type
|
||||
when :element
|
||||
if @namespaces
|
||||
nodesets << pre_defined_namespaces.merge(@namespaces)
|
||||
else
|
||||
nodesets << pre_defined_namespaces.merge(raw_node.namespaces)
|
||||
end
|
||||
when :attribute
|
||||
if @namespaces
|
||||
nodesets << pre_defined_namespaces.merge(@namespaces)
|
||||
else
|
||||
nodesets << pre_defined_namespaces.merge(raw_node.element.namespaces)
|
||||
end
|
||||
end
|
||||
end
|
||||
nodesets
|
||||
end
|
||||
when :parent
|
||||
nodeset = step(path_stack) do
|
||||
nodesets = []
|
||||
nodeset.each do |node|
|
||||
raw_node = node.raw_node
|
||||
if raw_node.node_type == :attribute
|
||||
parent = raw_node.element
|
||||
else
|
||||
parent = raw_node.parent
|
||||
end
|
||||
nodesets << [XPathNode.new(parent, position: 1)] if parent
|
||||
end
|
||||
nodesets
|
||||
end
|
||||
when :ancestor
|
||||
nodeset = step(path_stack) do
|
||||
nodesets = []
|
||||
# new_nodes = {}
|
||||
nodeset.each do |node|
|
||||
raw_node = node.raw_node
|
||||
new_nodeset = []
|
||||
while raw_node.parent
|
||||
raw_node = raw_node.parent
|
||||
# next if new_nodes.key?(node)
|
||||
new_nodeset << XPathNode.new(raw_node,
|
||||
position: new_nodeset.size + 1)
|
||||
# new_nodes[node] = true
|
||||
end
|
||||
nodesets << new_nodeset unless new_nodeset.empty?
|
||||
end
|
||||
nodesets
|
||||
end
|
||||
when :ancestor_or_self
|
||||
nodeset = step(path_stack) do
|
||||
nodesets = []
|
||||
# new_nodes = {}
|
||||
nodeset.each do |node|
|
||||
raw_node = node.raw_node
|
||||
next unless raw_node.node_type == :element
|
||||
new_nodeset = [XPathNode.new(raw_node, position: 1)]
|
||||
# new_nodes[node] = true
|
||||
while raw_node.parent
|
||||
raw_node = raw_node.parent
|
||||
# next if new_nodes.key?(node)
|
||||
new_nodeset << XPathNode.new(raw_node,
|
||||
position: new_nodeset.size + 1)
|
||||
# new_nodes[node] = true
|
||||
end
|
||||
nodesets << new_nodeset unless new_nodeset.empty?
|
||||
end
|
||||
nodesets
|
||||
end
|
||||
when :descendant_or_self
|
||||
nodeset = step(path_stack) do
|
||||
descendant(nodeset, true)
|
||||
end
|
||||
when :descendant
|
||||
nodeset = step(path_stack) do
|
||||
descendant(nodeset, false)
|
||||
end
|
||||
when :following_sibling
|
||||
nodeset = step(path_stack) do
|
||||
nodesets = []
|
||||
nodeset.each do |node|
|
||||
raw_node = node.raw_node
|
||||
next unless raw_node.respond_to?(:parent)
|
||||
next if raw_node.parent.nil?
|
||||
all_siblings = raw_node.parent.children
|
||||
current_index = all_siblings.index(raw_node)
|
||||
following_siblings = all_siblings[(current_index + 1)..-1]
|
||||
next if following_siblings.empty?
|
||||
nodesets << following_siblings.collect.with_index do |sibling, i|
|
||||
XPathNode.new(sibling, position: i + 1)
|
||||
end
|
||||
end
|
||||
nodesets
|
||||
end
|
||||
when :preceding_sibling
|
||||
nodeset = step(path_stack, order: :reverse) do
|
||||
nodesets = []
|
||||
nodeset.each do |node|
|
||||
raw_node = node.raw_node
|
||||
next unless raw_node.respond_to?(:parent)
|
||||
next if raw_node.parent.nil?
|
||||
all_siblings = raw_node.parent.children
|
||||
current_index = all_siblings.index(raw_node)
|
||||
preceding_siblings = all_siblings[0, current_index].reverse
|
||||
next if preceding_siblings.empty?
|
||||
nodesets << preceding_siblings.collect.with_index do |sibling, i|
|
||||
XPathNode.new(sibling, position: i + 1)
|
||||
end
|
||||
end
|
||||
nodesets
|
||||
end
|
||||
when :preceding
|
||||
nodeset = step(path_stack, order: :reverse) do
|
||||
unnode(nodeset) do |node|
|
||||
preceding(node)
|
||||
end
|
||||
end
|
||||
when :following
|
||||
nodeset = step(path_stack) do
|
||||
unnode(nodeset) do |node|
|
||||
following(node)
|
||||
end
|
||||
end
|
||||
when :variable
|
||||
var_name = path_stack.shift
|
||||
return [@variables[var_name]]
|
||||
|
||||
when :eq, :neq, :lt, :lteq, :gt, :gteq
|
||||
left = expr( path_stack.shift, nodeset.dup, context )
|
||||
right = expr( path_stack.shift, nodeset.dup, context )
|
||||
res = equality_relational_compare( left, op, right )
|
||||
trace(op, left, right, res) if @debug
|
||||
return res
|
||||
|
||||
when :or
|
||||
left = expr(path_stack.shift, nodeset.dup, context)
|
||||
return true if Functions.boolean(left)
|
||||
right = expr(path_stack.shift, nodeset.dup, context)
|
||||
return Functions.boolean(right)
|
||||
|
||||
when :and
|
||||
left = expr(path_stack.shift, nodeset.dup, context)
|
||||
return false unless Functions.boolean(left)
|
||||
right = expr(path_stack.shift, nodeset.dup, context)
|
||||
return Functions.boolean(right)
|
||||
|
||||
when :div, :mod, :mult, :plus, :minus
|
||||
left = expr(path_stack.shift, nodeset, context)
|
||||
right = expr(path_stack.shift, nodeset, context)
|
||||
left = unnode(left) if left.is_a?(Array)
|
||||
right = unnode(right) if right.is_a?(Array)
|
||||
left = Functions::number(left)
|
||||
right = Functions::number(right)
|
||||
case op
|
||||
when :div
|
||||
return left / right
|
||||
when :mod
|
||||
return left % right
|
||||
when :mult
|
||||
return left * right
|
||||
when :plus
|
||||
return left + right
|
||||
when :minus
|
||||
return left - right
|
||||
else
|
||||
raise "[BUG] Unexpected operator: <#{op.inspect}>"
|
||||
end
|
||||
when :union
|
||||
left = expr( path_stack.shift, nodeset, context )
|
||||
right = expr( path_stack.shift, nodeset, context )
|
||||
left = unnode(left) if left.is_a?(Array)
|
||||
right = unnode(right) if right.is_a?(Array)
|
||||
return (left | right)
|
||||
when :neg
|
||||
res = expr( path_stack, nodeset, context )
|
||||
res = unnode(res) if res.is_a?(Array)
|
||||
return -Functions.number(res)
|
||||
when :not
|
||||
when :function
|
||||
func_name = path_stack.shift.tr('-','_')
|
||||
arguments = path_stack.shift
|
||||
|
||||
if nodeset.size != 1
|
||||
message = "[BUG] Node set size must be 1 for function call: "
|
||||
message += "<#{func_name}>: <#{nodeset.inspect}>: "
|
||||
message += "<#{arguments.inspect}>"
|
||||
raise message
|
||||
end
|
||||
|
||||
node = nodeset.first
|
||||
if context
|
||||
target_context = context
|
||||
else
|
||||
target_context = {:size => nodeset.size}
|
||||
if node.is_a?(XPathNode)
|
||||
target_context[:node] = node.raw_node
|
||||
target_context[:index] = node.position
|
||||
else
|
||||
target_context[:node] = node
|
||||
target_context[:index] = 1
|
||||
end
|
||||
end
|
||||
args = arguments.dclone.collect do |arg|
|
||||
result = expr(arg, nodeset, target_context)
|
||||
result = unnode(result) if result.is_a?(Array)
|
||||
result
|
||||
end
|
||||
Functions.context = target_context
|
||||
return Functions.send(func_name, *args)
|
||||
|
||||
else
|
||||
raise "[BUG] Unexpected path: <#{op.inspect}>: <#{path_stack.inspect}>"
|
||||
end
|
||||
end # while
|
||||
return nodeset
|
||||
ensure
|
||||
leave(:expr, path_stack, nodeset) if @debug
|
||||
end
|
||||
|
||||
def step(path_stack, any_type: :element, order: :forward)
|
||||
nodesets = yield
|
||||
begin
|
||||
enter(:step, path_stack, nodesets) if @debug
|
||||
nodesets = node_test(path_stack, nodesets, any_type: any_type)
|
||||
while path_stack[0] == :predicate
|
||||
path_stack.shift # :predicate
|
||||
predicate_expression = path_stack.shift.dclone
|
||||
nodesets = evaluate_predicate(predicate_expression, nodesets)
|
||||
end
|
||||
if nodesets.size == 1
|
||||
ordered_nodeset = nodesets[0]
|
||||
else
|
||||
raw_nodes = []
|
||||
nodesets.each do |nodeset|
|
||||
nodeset.each do |node|
|
||||
if node.respond_to?(:raw_node)
|
||||
raw_nodes << node.raw_node
|
||||
else
|
||||
raw_nodes << node
|
||||
end
|
||||
end
|
||||
end
|
||||
ordered_nodeset = sort(raw_nodes, order)
|
||||
end
|
||||
new_nodeset = []
|
||||
ordered_nodeset.each do |node|
|
||||
# TODO: Remove duplicated
|
||||
new_nodeset << XPathNode.new(node, position: new_nodeset.size + 1)
|
||||
end
|
||||
new_nodeset
|
||||
ensure
|
||||
leave(:step, path_stack, new_nodeset) if @debug
|
||||
end
|
||||
end
|
||||
|
||||
def node_test(path_stack, nodesets, any_type: :element)
|
||||
enter(:node_test, path_stack, nodesets) if @debug
|
||||
operator = path_stack.shift
|
||||
case operator
|
||||
when :qname
|
||||
prefix = path_stack.shift
|
||||
name = path_stack.shift
|
||||
new_nodesets = nodesets.collect do |nodeset|
|
||||
filter_nodeset(nodeset) do |node|
|
||||
raw_node = node.raw_node
|
||||
case raw_node.node_type
|
||||
when :element
|
||||
if prefix.nil?
|
||||
raw_node.name == name
|
||||
elsif prefix.empty?
|
||||
if strict?
|
||||
raw_node.name == name and raw_node.namespace == ""
|
||||
else
|
||||
raw_node.name == name and raw_node.namespace == get_namespace(raw_node, prefix)
|
||||
end
|
||||
else
|
||||
raw_node.name == name and raw_node.namespace == get_namespace(raw_node, prefix)
|
||||
end
|
||||
when :attribute
|
||||
if prefix.nil?
|
||||
raw_node.name == name
|
||||
elsif prefix.empty?
|
||||
raw_node.name == name and raw_node.namespace == ""
|
||||
else
|
||||
raw_node.name == name and raw_node.namespace == get_namespace(raw_node.element, prefix)
|
||||
end
|
||||
else
|
||||
false
|
||||
end
|
||||
end
|
||||
end
|
||||
when :namespace
|
||||
prefix = path_stack.shift
|
||||
new_nodesets = nodesets.collect do |nodeset|
|
||||
filter_nodeset(nodeset) do |node|
|
||||
raw_node = node.raw_node
|
||||
case raw_node.node_type
|
||||
when :element
|
||||
namespaces = @namespaces || raw_node.namespaces
|
||||
raw_node.namespace == namespaces[prefix]
|
||||
when :attribute
|
||||
namespaces = @namespaces || raw_node.element.namespaces
|
||||
raw_node.namespace == namespaces[prefix]
|
||||
else
|
||||
false
|
||||
end
|
||||
end
|
||||
end
|
||||
when :any
|
||||
new_nodesets = nodesets.collect do |nodeset|
|
||||
filter_nodeset(nodeset) do |node|
|
||||
raw_node = node.raw_node
|
||||
raw_node.node_type == any_type
|
||||
end
|
||||
end
|
||||
when :comment
|
||||
new_nodesets = nodesets.collect do |nodeset|
|
||||
filter_nodeset(nodeset) do |node|
|
||||
raw_node = node.raw_node
|
||||
raw_node.node_type == :comment
|
||||
end
|
||||
end
|
||||
when :text
|
||||
new_nodesets = nodesets.collect do |nodeset|
|
||||
filter_nodeset(nodeset) do |node|
|
||||
raw_node = node.raw_node
|
||||
raw_node.node_type == :text
|
||||
end
|
||||
end
|
||||
when :processing_instruction
|
||||
target = path_stack.shift
|
||||
new_nodesets = nodesets.collect do |nodeset|
|
||||
filter_nodeset(nodeset) do |node|
|
||||
raw_node = node.raw_node
|
||||
(raw_node.node_type == :processing_instruction) and
|
||||
(target.empty? or (raw_node.target == target))
|
||||
end
|
||||
end
|
||||
when :node
|
||||
new_nodesets = nodesets.collect do |nodeset|
|
||||
filter_nodeset(nodeset) do |node|
|
||||
true
|
||||
end
|
||||
end
|
||||
else
|
||||
message = "[BUG] Unexpected node test: " +
|
||||
"<#{operator.inspect}>: <#{path_stack.inspect}>"
|
||||
raise message
|
||||
end
|
||||
new_nodesets
|
||||
ensure
|
||||
leave(:node_test, path_stack, new_nodesets) if @debug
|
||||
end
|
||||
|
||||
def filter_nodeset(nodeset)
|
||||
new_nodeset = []
|
||||
nodeset.each do |node|
|
||||
next unless yield(node)
|
||||
new_nodeset << XPathNode.new(node, position: new_nodeset.size + 1)
|
||||
end
|
||||
new_nodeset
|
||||
end
|
||||
|
||||
def evaluate_predicate(expression, nodesets)
|
||||
enter(:predicate, expression, nodesets) if @debug
|
||||
new_nodeset_count = 0
|
||||
new_nodesets = nodesets.collect do |nodeset|
|
||||
new_nodeset = []
|
||||
subcontext = { :size => nodeset.size }
|
||||
nodeset.each_with_index do |node, index|
|
||||
if node.is_a?(XPathNode)
|
||||
subcontext[:node] = node.raw_node
|
||||
subcontext[:index] = node.position
|
||||
else
|
||||
subcontext[:node] = node
|
||||
subcontext[:index] = index + 1
|
||||
end
|
||||
result = expr(expression.dclone, [node], subcontext)
|
||||
trace(:predicate_evaluate, expression, node, subcontext, result) if @debug
|
||||
result = result[0] if result.kind_of? Array and result.length == 1
|
||||
if result.kind_of? Numeric
|
||||
if result == node.position
|
||||
new_nodeset_count += 1
|
||||
new_nodeset << XPathNode.new(node, position: new_nodeset_count)
|
||||
end
|
||||
elsif result.instance_of? Array
|
||||
if result.size > 0 and result.inject(false) {|k,s| s or k}
|
||||
if result.size > 0
|
||||
new_nodeset_count += 1
|
||||
new_nodeset << XPathNode.new(node, position: new_nodeset_count)
|
||||
end
|
||||
end
|
||||
else
|
||||
if result
|
||||
new_nodeset_count += 1
|
||||
new_nodeset << XPathNode.new(node, position: new_nodeset_count)
|
||||
end
|
||||
end
|
||||
end
|
||||
new_nodeset
|
||||
end
|
||||
new_nodesets
|
||||
ensure
|
||||
leave(:predicate, new_nodesets) if @debug
|
||||
end
|
||||
|
||||
def trace(*args)
|
||||
indent = " " * @nest
|
||||
PP.pp(args, "").each_line do |line|
|
||||
puts("#{indent}#{line}")
|
||||
end
|
||||
end
|
||||
|
||||
def enter(tag, *args)
|
||||
trace(:enter, tag, *args)
|
||||
@nest += 1
|
||||
end
|
||||
|
||||
def leave(tag, *args)
|
||||
@nest -= 1
|
||||
trace(:leave, tag, *args)
|
||||
end
|
||||
|
||||
# Reorders an array of nodes so that they are in document order
|
||||
# It tries to do this efficiently.
|
||||
#
|
||||
# FIXME: I need to get rid of this, but the issue is that most of the XPath
|
||||
# interpreter functions as a filter, which means that we lose context going
|
||||
# in and out of function calls. If I knew what the index of the nodes was,
|
||||
# I wouldn't have to do this. Maybe add a document IDX for each node?
|
||||
# Problems with mutable documents. Or, rewrite everything.
|
||||
def sort(array_of_nodes, order)
|
||||
new_arry = []
|
||||
array_of_nodes.each { |node|
|
||||
node_idx = []
|
||||
np = node.node_type == :attribute ? node.element : node
|
||||
while np.parent and np.parent.node_type == :element
|
||||
node_idx << np.parent.index( np )
|
||||
np = np.parent
|
||||
end
|
||||
new_arry << [ node_idx.reverse, node ]
|
||||
}
|
||||
ordered = new_arry.sort_by do |index, node|
|
||||
if order == :forward
|
||||
index
|
||||
else
|
||||
index.map(&:-@)
|
||||
end
|
||||
end
|
||||
ordered.collect do |_index, node|
|
||||
node
|
||||
end
|
||||
end
|
||||
|
||||
def descendant(nodeset, include_self)
|
||||
nodesets = []
|
||||
nodeset.each do |node|
|
||||
new_nodeset = []
|
||||
new_nodes = {}
|
||||
descendant_recursive(node.raw_node, new_nodeset, new_nodes, include_self)
|
||||
nodesets << new_nodeset unless new_nodeset.empty?
|
||||
end
|
||||
nodesets
|
||||
end
|
||||
|
||||
def descendant_recursive(raw_node, new_nodeset, new_nodes, include_self)
|
||||
if include_self
|
||||
return if new_nodes.key?(raw_node)
|
||||
new_nodeset << XPathNode.new(raw_node, position: new_nodeset.size + 1)
|
||||
new_nodes[raw_node] = true
|
||||
end
|
||||
|
||||
node_type = raw_node.node_type
|
||||
if node_type == :element or node_type == :document
|
||||
raw_node.children.each do |child|
|
||||
descendant_recursive(child, new_nodeset, new_nodes, true)
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# Builds a nodeset of all of the preceding nodes of the supplied node,
|
||||
# in reverse document order
|
||||
# preceding:: includes every element in the document that precedes this node,
|
||||
# except for ancestors
|
||||
def preceding(node)
|
||||
ancestors = []
|
||||
parent = node.parent
|
||||
while parent
|
||||
ancestors << parent
|
||||
parent = parent.parent
|
||||
end
|
||||
|
||||
precedings = []
|
||||
preceding_node = preceding_node_of(node)
|
||||
while preceding_node
|
||||
if ancestors.include?(preceding_node)
|
||||
ancestors.delete(preceding_node)
|
||||
else
|
||||
precedings << XPathNode.new(preceding_node,
|
||||
position: precedings.size + 1)
|
||||
end
|
||||
preceding_node = preceding_node_of(preceding_node)
|
||||
end
|
||||
precedings
|
||||
end
|
||||
|
||||
def preceding_node_of( node )
|
||||
psn = node.previous_sibling_node
|
||||
if psn.nil?
|
||||
if node.parent.nil? or node.parent.class == Document
|
||||
return nil
|
||||
end
|
||||
return node.parent
|
||||
#psn = preceding_node_of( node.parent )
|
||||
end
|
||||
while psn and psn.kind_of? Element and psn.children.size > 0
|
||||
psn = psn.children[-1]
|
||||
end
|
||||
psn
|
||||
end
|
||||
|
||||
def following(node)
|
||||
followings = []
|
||||
following_node = next_sibling_node(node)
|
||||
while following_node
|
||||
followings << XPathNode.new(following_node,
|
||||
position: followings.size + 1)
|
||||
following_node = following_node_of(following_node)
|
||||
end
|
||||
followings
|
||||
end
|
||||
|
||||
def following_node_of( node )
|
||||
return node.children[0] if node.kind_of?(Element) and node.children.size > 0
|
||||
|
||||
next_sibling_node(node)
|
||||
end
|
||||
|
||||
def next_sibling_node(node)
|
||||
psn = node.next_sibling_node
|
||||
while psn.nil?
|
||||
return nil if node.parent.nil? or node.parent.class == Document
|
||||
node = node.parent
|
||||
psn = node.next_sibling_node
|
||||
end
|
||||
psn
|
||||
end
|
||||
|
||||
def child(nodeset)
|
||||
nodesets = []
|
||||
nodeset.each do |node|
|
||||
raw_node = node.raw_node
|
||||
node_type = raw_node.node_type
|
||||
# trace(:child, node_type, node)
|
||||
case node_type
|
||||
when :element
|
||||
nodesets << raw_node.children.collect.with_index do |child_node, i|
|
||||
XPathNode.new(child_node, position: i + 1)
|
||||
end
|
||||
when :document
|
||||
new_nodeset = []
|
||||
raw_node.children.each do |child|
|
||||
case child
|
||||
when XMLDecl, Text
|
||||
# Ignore
|
||||
else
|
||||
new_nodeset << XPathNode.new(child, position: new_nodeset.size + 1)
|
||||
end
|
||||
end
|
||||
nodesets << new_nodeset unless new_nodeset.empty?
|
||||
end
|
||||
end
|
||||
nodesets
|
||||
end
|
||||
|
||||
def norm b
|
||||
case b
|
||||
when true, false
|
||||
b
|
||||
when 'true', 'false'
|
||||
Functions::boolean( b )
|
||||
when /^\d+(\.\d+)?$/, Numeric
|
||||
Functions::number( b )
|
||||
else
|
||||
Functions::string( b )
|
||||
end
|
||||
end
|
||||
|
||||
def equality_relational_compare(set1, op, set2)
|
||||
set1 = unnode(set1) if set1.is_a?(Array)
|
||||
set2 = unnode(set2) if set2.is_a?(Array)
|
||||
|
||||
if set1.kind_of? Array and set2.kind_of? Array
|
||||
# If both objects to be compared are node-sets, then the
|
||||
# comparison will be true if and only if there is a node in the
|
||||
# first node-set and a node in the second node-set such that the
|
||||
# result of performing the comparison on the string-values of
|
||||
# the two nodes is true.
|
||||
set1.product(set2).any? do |node1, node2|
|
||||
node_string1 = Functions.string(node1)
|
||||
node_string2 = Functions.string(node2)
|
||||
compare(node_string1, op, node_string2)
|
||||
end
|
||||
elsif set1.kind_of? Array or set2.kind_of? Array
|
||||
# If one is nodeset and other is number, compare number to each item
|
||||
# in nodeset s.t. number op number(string(item))
|
||||
# If one is nodeset and other is string, compare string to each item
|
||||
# in nodeset s.t. string op string(item)
|
||||
# If one is nodeset and other is boolean, compare boolean to each item
|
||||
# in nodeset s.t. boolean op boolean(item)
|
||||
if set1.kind_of? Array
|
||||
a = set1
|
||||
b = set2
|
||||
else
|
||||
a = set2
|
||||
b = set1
|
||||
end
|
||||
|
||||
case b
|
||||
when true, false
|
||||
each_unnode(a).any? do |unnoded|
|
||||
compare(Functions.boolean(unnoded), op, b)
|
||||
end
|
||||
when Numeric
|
||||
each_unnode(a).any? do |unnoded|
|
||||
compare(Functions.number(unnoded), op, b)
|
||||
end
|
||||
when /\A\d+(\.\d+)?\z/
|
||||
b = Functions.number(b)
|
||||
each_unnode(a).any? do |unnoded|
|
||||
compare(Functions.number(unnoded), op, b)
|
||||
end
|
||||
else
|
||||
b = Functions::string(b)
|
||||
each_unnode(a).any? do |unnoded|
|
||||
compare(Functions::string(unnoded), op, b)
|
||||
end
|
||||
end
|
||||
else
|
||||
# If neither is nodeset,
|
||||
# If op is = or !=
|
||||
# If either boolean, convert to boolean
|
||||
# If either number, convert to number
|
||||
# Else, convert to string
|
||||
# Else
|
||||
# Convert both to numbers and compare
|
||||
compare(set1, op, set2)
|
||||
end
|
||||
end
|
||||
|
||||
def value_type(value)
|
||||
case value
|
||||
when true, false
|
||||
:boolean
|
||||
when Numeric
|
||||
:number
|
||||
when String
|
||||
:string
|
||||
else
|
||||
raise "[BUG] Unexpected value type: <#{value.inspect}>"
|
||||
end
|
||||
end
|
||||
|
||||
def normalize_compare_values(a, operator, b)
|
||||
a_type = value_type(a)
|
||||
b_type = value_type(b)
|
||||
case operator
|
||||
when :eq, :neq
|
||||
if a_type == :boolean or b_type == :boolean
|
||||
a = Functions.boolean(a) unless a_type == :boolean
|
||||
b = Functions.boolean(b) unless b_type == :boolean
|
||||
elsif a_type == :number or b_type == :number
|
||||
a = Functions.number(a) unless a_type == :number
|
||||
b = Functions.number(b) unless b_type == :number
|
||||
else
|
||||
a = Functions.string(a) unless a_type == :string
|
||||
b = Functions.string(b) unless b_type == :string
|
||||
end
|
||||
when :lt, :lteq, :gt, :gteq
|
||||
a = Functions.number(a) unless a_type == :number
|
||||
b = Functions.number(b) unless b_type == :number
|
||||
else
|
||||
message = "[BUG] Unexpected compare operator: " +
|
||||
"<#{operator.inspect}>: <#{a.inspect}>: <#{b.inspect}>"
|
||||
raise message
|
||||
end
|
||||
[a, b]
|
||||
end
|
||||
|
||||
def compare(a, operator, b)
|
||||
a, b = normalize_compare_values(a, operator, b)
|
||||
case operator
|
||||
when :eq
|
||||
a == b
|
||||
when :neq
|
||||
a != b
|
||||
when :lt
|
||||
a < b
|
||||
when :lteq
|
||||
a <= b
|
||||
when :gt
|
||||
a > b
|
||||
when :gteq
|
||||
a >= b
|
||||
else
|
||||
message = "[BUG] Unexpected compare operator: " +
|
||||
"<#{operator.inspect}>: <#{a.inspect}>: <#{b.inspect}>"
|
||||
raise message
|
||||
end
|
||||
end
|
||||
|
||||
def each_unnode(nodeset)
|
||||
return to_enum(__method__, nodeset) unless block_given?
|
||||
nodeset.each do |node|
|
||||
if node.is_a?(XPathNode)
|
||||
unnoded = node.raw_node
|
||||
else
|
||||
unnoded = node
|
||||
end
|
||||
yield(unnoded)
|
||||
end
|
||||
end
|
||||
|
||||
def unnode(nodeset)
|
||||
each_unnode(nodeset).collect do |unnoded|
|
||||
unnoded = yield(unnoded) if block_given?
|
||||
unnoded
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# @private
|
||||
class XPathNode
|
||||
attr_reader :raw_node, :context
|
||||
def initialize(node, context=nil)
|
||||
if node.is_a?(XPathNode)
|
||||
@raw_node = node.raw_node
|
||||
else
|
||||
@raw_node = node
|
||||
end
|
||||
@context = context || {}
|
||||
end
|
||||
|
||||
def position
|
||||
@context[:position]
|
||||
end
|
||||
end
|
||||
end
|
||||
Reference in New Issue
Block a user