Add bin and edit workflow
Gitea Actions Demo / Explore-Gitea-Actions (push) Failing after 9s

This commit is contained in:
2026-09-16 13:11:16 -06:00
parent c8ac4fcae5
commit 4cee170d66
17576 changed files with 895740 additions and 2 deletions
@@ -0,0 +1,66 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
class Attr < Node
alias_method :value, :content
alias_method :to_s, :content
alias_method :content=, :value=
#
# :call-seq: deconstruct_keys(array_of_names) → Hash
#
# Returns a hash describing the Attr, to use in pattern matching.
#
# Valid keys and their values:
# - +name+ → (String) The name of the attribute.
# - +value+ → (String) The value of the attribute.
# - +namespace+ → (Namespace, nil) The Namespace of the attribute, or +nil+ if there is no namespace.
#
# *Example*
#
# doc = Nokogiri::XML.parse(<<~XML)
# <?xml version="1.0"?>
# <root xmlns="http://nokogiri.org/ns/default" xmlns:noko="http://nokogiri.org/ns/noko">
# <child1 foo="abc" noko:bar="def"/>
# </root>
# XML
#
# attributes = doc.root.elements.first.attribute_nodes
# # => [#(Attr:0x35c { name = "foo", value = "abc" }),
# # #(Attr:0x370 {
# # name = "bar",
# # namespace = #(Namespace:0x384 {
# # prefix = "noko",
# # href = "http://nokogiri.org/ns/noko"
# # }),
# # value = "def"
# # })]
#
# attributes.first.deconstruct_keys([:name, :value, :namespace])
# # => {:name=>"foo", :value=>"abc", :namespace=>nil}
#
# attributes.last.deconstruct_keys([:name, :value, :namespace])
# # => {:name=>"bar",
# # :value=>"def",
# # :namespace=>
# # #(Namespace:0x384 {
# # prefix = "noko",
# # href = "http://nokogiri.org/ns/noko"
# # })}
#
# Since v1.14.0
#
def deconstruct_keys(keys)
{ name: name, value: value, namespace: namespace }
end
private
def inspect_attributes
[:name, :namespace, :value]
end
end
end
end
@@ -0,0 +1,22 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# Represents an attribute declaration in a DTD
class AttributeDecl < Nokogiri::XML::Node
undef_method :attribute_nodes
undef_method :attributes
undef_method :content
undef_method :namespace
undef_method :namespace_definitions
undef_method :line if method_defined?(:line)
private
def inspect_attributes
[:to_s]
end
end
end
end
@@ -0,0 +1,494 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# Nokogiri builder can be used for building XML and HTML documents.
#
# == Synopsis:
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.root {
# xml.products {
# xml.widget {
# xml.id_ "10"
# xml.name "Awesome widget"
# }
# }
# }
# end
# puts builder.to_xml
#
# Will output:
#
# <?xml version="1.0"?>
# <root>
# <products>
# <widget>
# <id>10</id>
# <name>Awesome widget</name>
# </widget>
# </products>
# </root>
#
#
# === Builder scope
#
# The builder allows two forms. When the builder is supplied with a block
# that has a parameter, the outside scope is maintained. This means you
# can access variables that are outside your builder. If you don't need
# outside scope, you can use the builder without the "xml" prefix like
# this:
#
# builder = Nokogiri::XML::Builder.new do
# root {
# products {
# widget {
# id_ "10"
# name "Awesome widget"
# }
# }
# }
# end
#
# == Special Tags
#
# The builder works by taking advantage of method_missing. Unfortunately
# some methods are defined in ruby that are difficult or dangerous to
# remove. You may want to create tags with the name "type", "class", and
# "id" for example. In that case, you can use an underscore to
# disambiguate your tag name from the method call.
#
# Here is an example of using the underscore to disambiguate tag names from
# ruby methods:
#
# @objects = [Object.new, Object.new, Object.new]
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.root {
# xml.objects {
# @objects.each do |o|
# xml.object {
# xml.type_ o.type
# xml.class_ o.class.name
# xml.id_ o.id
# }
# end
# }
# }
# end
# puts builder.to_xml
#
# The underscore may be used with any tag name, and the last underscore
# will just be removed. This code will output the following XML:
#
# <?xml version="1.0"?>
# <root>
# <objects>
# <object>
# <type>Object</type>
# <class>Object</class>
# <id>48390</id>
# </object>
# <object>
# <type>Object</type>
# <class>Object</class>
# <id>48380</id>
# </object>
# <object>
# <type>Object</type>
# <class>Object</class>
# <id>48370</id>
# </object>
# </objects>
# </root>
#
# == Tag Attributes
#
# Tag attributes may be supplied as method arguments. Here is our
# previous example, but using attributes rather than tags:
#
# @objects = [Object.new, Object.new, Object.new]
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.root {
# xml.objects {
# @objects.each do |o|
# xml.object(:type => o.type, :class => o.class, :id => o.id)
# end
# }
# }
# end
# puts builder.to_xml
#
# === Tag Attribute Short Cuts
#
# A couple attribute short cuts are available when building tags. The
# short cuts are available by special method calls when building a tag.
#
# This example builds an "object" tag with the class attribute "classy"
# and the id of "thing":
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.root {
# xml.objects {
# xml.object.classy.thing!
# }
# }
# end
# puts builder.to_xml
#
# Which will output:
#
# <?xml version="1.0"?>
# <root>
# <objects>
# <object class="classy" id="thing"/>
# </objects>
# </root>
#
# All other options are still supported with this syntax, including
# blocks and extra tag attributes.
#
# == Namespaces
#
# Namespaces are added similarly to attributes. Nokogiri::XML::Builder
# assumes that when an attribute starts with "xmlns", it is meant to be
# a namespace:
#
# builder = Nokogiri::XML::Builder.new { |xml|
# xml.root('xmlns' => 'default', 'xmlns:foo' => 'bar') do
# xml.tenderlove
# end
# }
# puts builder.to_xml
#
# Will output XML like this:
#
# <?xml version="1.0"?>
# <root xmlns:foo="bar" xmlns="default">
# <tenderlove/>
# </root>
#
# === Referencing declared namespaces
#
# Tags that reference non-default namespaces (i.e. a tag "foo:bar") can be
# built by using the Nokogiri::XML::Builder#[] method.
#
# For example:
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.root('xmlns:foo' => 'bar') {
# xml.objects {
# xml['foo'].object.classy.thing!
# }
# }
# end
# puts builder.to_xml
#
# Will output this XML:
#
# <?xml version="1.0"?>
# <root xmlns:foo="bar">
# <objects>
# <foo:object class="classy" id="thing"/>
# </objects>
# </root>
#
# Note the "foo:object" tag.
#
# === Namespace inheritance
#
# In the Builder context, children will inherit their parent's namespace. This is the same
# behavior as if the underlying {XML::Document} set +namespace_inheritance+ to +true+:
#
# result = Nokogiri::XML::Builder.new do |xml|
# xml["soapenv"].Envelope("xmlns:soapenv" => "http://schemas.xmlsoap.org/soap/envelope/") do
# xml.Header
# end
# end
# result.doc.to_xml
# # => <?xml version="1.0" encoding="utf-8"?>
# # <soapenv:Envelope xmlns:soapenv="http://schemas.xmlsoap.org/soap/envelope/">
# # <soapenv:Header/>
# # </soapenv:Envelope>
#
# Users may turn this behavior off by passing a keyword argument +namespace_inheritance:false+
# to the initializer:
#
# result = Nokogiri::XML::Builder.new(namespace_inheritance: false) do |xml|
# xml["soapenv"].Envelope("xmlns:soapenv" => "http://schemas.xmlsoap.org/soap/envelope/") do
# xml.Header
# xml["soapenv"].Body # users may explicitly opt into the namespace
# end
# end
# result.doc.to_xml
# # => <?xml version="1.0" encoding="utf-8"?>
# # <soapenv:Envelope xmlns:soapenv="http://schemas.xmlsoap.org/soap/envelope/">
# # <Header/>
# # <soapenv:Body/>
# # </soapenv:Envelope>
#
# For more information on namespace inheritance, please see {XML::Document#namespace_inheritance}
#
#
# == Document Types
#
# To create a document type (DTD), use the Builder#doc method to get
# the current context document. Then call Node#create_internal_subset to
# create the DTD node.
#
# For example, this Ruby:
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.doc.create_internal_subset(
# 'html',
# "-//W3C//DTD HTML 4.01 Transitional//EN",
# "http://www.w3.org/TR/html4/loose.dtd"
# )
# xml.root do
# xml.foo
# end
# end
#
# puts builder.to_xml
#
# Will output this xml:
#
# <?xml version="1.0"?>
# <!DOCTYPE html PUBLIC "-//W3C//DTD HTML 4.01 Transitional//EN" "http://www.w3.org/TR/html4/loose.dtd">
# <root>
# <foo/>
# </root>
#
class Builder
include Nokogiri::ClassResolver
DEFAULT_DOCUMENT_OPTIONS = { namespace_inheritance: true }
# The current Document object being built
attr_accessor :doc
# The parent of the current node being built
attr_accessor :parent
# A context object for use when the block has no arguments
attr_accessor :context
attr_accessor :arity # :nodoc:
###
# Create a builder with an existing root object. This is for use when
# you have an existing document that you would like to augment with
# builder methods. The builder context created will start with the
# given +root+ node.
#
# For example:
#
# doc = Nokogiri::XML(File.read('somedoc.xml'))
# Nokogiri::XML::Builder.with(doc.at_css('some_tag')) do |xml|
# # ... Use normal builder methods here ...
# xml.awesome # add the "awesome" tag below "some_tag"
# end
#
def self.with(root, &block)
new({}, root, &block)
end
###
# Create a new Builder object. +options+ are sent to the top level
# Document that is being built.
#
# Building a document with a particular encoding for example:
#
# Nokogiri::XML::Builder.new(:encoding => 'UTF-8') do |xml|
# ...
# end
def initialize(options = {}, root = nil, &block)
if root
@doc = root.document
@parent = root
else
@parent = @doc = related_class("Document").new
end
@context = nil
@arity = nil
@ns = nil
options = DEFAULT_DOCUMENT_OPTIONS.merge(options)
options.each do |k, v|
@doc.send(:"#{k}=", v)
end
return unless block
@arity = block.arity
if @arity <= 0
@context = eval("self", block.binding)
instance_eval(&block)
else
yield self
end
@parent = @doc
end
###
# Create a Text Node with content of +string+
def text(string)
insert(@doc.create_text_node(string))
end
###
# Create a CDATA Node with content of +string+
def cdata(string)
insert(doc.create_cdata(string))
end
###
# Create a Comment Node with content of +string+
def comment(string)
insert(doc.create_comment(string))
end
###
# Build a tag that is associated with namespace +ns+. Raises an
# ArgumentError if +ns+ has not been defined higher in the tree.
def [](ns)
if @parent != @doc
@ns = @parent.namespace_definitions.find { |x| x.prefix == ns.to_s }
end
return self if @ns
@parent.ancestors.each do |a|
next if a == doc
@ns = a.namespace_definitions.find { |x| x.prefix == ns.to_s }
return self if @ns
end
@ns = { pending: ns.to_s }
self
end
###
# Convert this Builder object to XML
def to_xml(*args)
if Nokogiri.jruby?
options = args.first.is_a?(Hash) ? args.shift : {}
unless options[:save_with]
options[:save_with] = Node::SaveOptions::AS_BUILDER
end
args.insert(0, options)
end
@doc.to_xml(*args)
end
###
# Append the given raw XML +string+ to the document
def <<(string)
@doc.fragment(string).children.each { |x| insert(x) }
end
def method_missing(method, *args, &block) # :nodoc:
if @context&.respond_to?(method)
@context.send(method, *args, &block)
else
node = @doc.create_element(method.to_s.sub(/[_!]$/, ""), *args) do |n|
# Set up the namespace
if @ns.is_a?(Nokogiri::XML::Namespace)
n.namespace = @ns
@ns = nil
end
end
if @ns.is_a?(Hash)
node.namespace = node.namespace_definitions.find { |x| x.prefix == @ns[:pending] }
if node.namespace.nil?
raise ArgumentError, "Namespace #{@ns[:pending]} has not been defined"
end
@ns = nil
end
insert(node, &block)
end
end
private
###
# Insert +node+ as a child of the current Node
def insert(node, &block)
node = @parent.add_child(node)
if block
begin
old_parent = @parent
@parent = node
@arity ||= block.arity
if @arity <= 0
instance_eval(&block)
else
yield(self)
end
ensure
@parent = old_parent
end
end
NodeBuilder.new(node, self)
end
class NodeBuilder # :nodoc:
def initialize(node, doc_builder)
@node = node
@doc_builder = doc_builder
end
def []=(k, v)
@node[k] = v
end
def [](k)
@node[k]
end
def method_missing(method, *args, &block)
opts = args.last.is_a?(Hash) ? args.pop : {}
case method.to_s
when /^(.*)!$/
@node["id"] = Regexp.last_match(1)
@node.content = args.first if args.first
when /^(.*)=/
@node[Regexp.last_match(1)] = args.first
else
@node["class"] =
((@node["class"] || "").split(/\s/) + [method.to_s]).join(" ")
@node.content = args.first if args.first
end
# Assign any extra options
opts.each do |k, v|
@node[k.to_s] = ((@node[k.to_s] || "").split(/\s/) + [v]).join(" ")
end
if block
old_parent = @doc_builder.parent
@doc_builder.parent = @node
arity = @doc_builder.arity || block.arity
value = if arity <= 0
@doc_builder.instance_eval(&block)
else
yield(@doc_builder)
end
@doc_builder.parent = old_parent
return value
end
self
end
end
end
end
end
@@ -0,0 +1,13 @@
# frozen_string_literal: true
module Nokogiri
module XML
class CDATA < Nokogiri::XML::Text
###
# Get the name of this CDATA node
def name
"#cdata-section"
end
end
end
end
@@ -0,0 +1,9 @@
# frozen_string_literal: true
module Nokogiri
module XML
class CharacterData < Nokogiri::XML::Node
include Nokogiri::XML::PP::CharacterData
end
end
end
@@ -0,0 +1,514 @@
# coding: utf-8
# frozen_string_literal: true
require "pathname"
module Nokogiri
module XML
# Nokogiri::XML::Document is the main entry point for dealing with \XML documents. The Document
# is created by parsing \XML content from a String or an IO object. See
# Nokogiri::XML::Document.parse for more information on parsing.
#
# Document inherits a great deal of functionality from its superclass Nokogiri::XML::Node, so
# please read that class's documentation as well.
class Document < Nokogiri::XML::Node
# See http://www.w3.org/TR/REC-xml-names/#ns-decl for more details. Note that we're not
# attempting to handle unicode characters partly because libxml2 doesn't handle unicode
# characters in NCNAMEs.
NCNAME_START_CHAR = "A-Za-z_"
NCNAME_CHAR = NCNAME_START_CHAR + "\\-\\.0-9"
NCNAME_RE = /^xmlns(?::([#{NCNAME_START_CHAR}][#{NCNAME_CHAR}]*))?$/
OBJECT_DUP_METHOD = Object.instance_method(:dup)
OBJECT_CLONE_METHOD = Object.instance_method(:clone)
private_constant :OBJECT_DUP_METHOD, :OBJECT_CLONE_METHOD
class << self
# call-seq:
# parse(input) { |options| ... } => Nokogiri::XML::Document
# parse(input, url:, encoding:, options:) => Nokogiri::XML::Document
#
# Parse \XML input from a String or IO object, and return a new XML::Document.
#
# 🛡 By default, Nokogiri treats documents as untrusted, and so does not attempt to load DTDs
# or access the network. See Nokogiri::XML::ParseOptions for a complete list of options; and
# that module's DEFAULT_XML constant for what's set (and not set) by default.
#
# [Required Parameters]
# - +input+ (String | IO) The content to be parsed.
#
# [Optional Keyword Arguments]
# - +url:+ (String) The base URI for this document.
#
# - +encoding:+ (String) The name of the encoding that should be used when processing the
# document. When not provided, the encoding will be determined based on the document
# content.
#
# - +options:+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
# behaviors during parsing. See ParseOptions for more information. The default value is
# +ParseOptions::DEFAULT_XML+.
#
# [Yields]
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
# can be configured before parsing. See Nokogiri::XML::ParseOptions for more information.
#
# [Returns] Nokogiri::XML::Document
def parse(
string_or_io,
url_ = nil, encoding_ = nil, options_ = XML::ParseOptions::DEFAULT_XML,
url: url_, encoding: encoding_, options: options_
)
options = Nokogiri::XML::ParseOptions.new(options) if Integer === options
yield options if block_given?
url ||= string_or_io.respond_to?(:path) ? string_or_io.path : nil
if empty_doc?(string_or_io)
if options.strict?
raise Nokogiri::XML::SyntaxError, "Empty document"
else
return encoding ? new.tap { |i| i.encoding = encoding } : new
end
end
doc = if string_or_io.respond_to?(:read)
# TODO: should we instead check for respond_to?(:to_path) ?
if string_or_io.is_a?(Pathname)
# resolve the Pathname to the file and open it as an IO object, see #2110
string_or_io = string_or_io.expand_path.open
url ||= string_or_io.path
end
read_io(string_or_io, url, encoding, options.to_i)
else
# read_memory pukes on empty docs
read_memory(string_or_io, url, encoding, options.to_i)
end
# do xinclude processing
doc.do_xinclude(options) if options.xinclude?
doc
end
private
def empty_doc?(string_or_io)
string_or_io.nil? ||
(string_or_io.respond_to?(:empty?) && string_or_io.empty?) ||
(string_or_io.respond_to?(:eof?) && string_or_io.eof?)
end
end
##
# :singleton-method: wrap
# :call-seq: wrap(java_document) → Nokogiri::XML::Document
#
# ⚠ This method is only available when running JRuby.
#
# Create a Document using an existing Java DOM document object.
#
# The returned Document shares the same underlying data structure as the Java object, so
# changes in one are reflected in the other.
#
# [Parameters]
# - `java_document` (Java::OrgW3cDom::Document)
# (The class `Java::OrgW3cDom::Document` is also accessible as `org.w3c.dom.Document`.)
#
# [Returns] Nokogiri::XML::Document
#
# See also \#to_java
# :method: to_java
# :call-seq: to_java() → Java::OrgW3cDom::Document
#
# ⚠ This method is only available when running JRuby.
#
# Returns the underlying Java DOM document object for this document.
#
# The returned Java object shares the same underlying data structure as this document, so
# changes in one are reflected in the other.
#
# [Returns]
# Java::OrgW3cDom::Document
# (The class `Java::OrgW3cDom::Document` is also accessible as `org.w3c.dom.Document`.)
#
# See also Document.wrap
# The errors found while parsing a document.
#
# [Returns] Array<Nokogiri::XML::SyntaxError>
attr_accessor :errors
# When `true`, reparented elements without a namespace will inherit their new parent's
# namespace (if one exists). Defaults to `false`.
#
# [Returns] Boolean
#
# *Example:* Default behavior of namespace inheritance
#
# xml = <<~EOF
# <root xmlns:foo="http://nokogiri.org/default_ns/test/foo">
# <foo:parent>
# </foo:parent>
# </root>
# EOF
# doc = Nokogiri::XML(xml)
# parent = doc.at_xpath("//foo:parent", "foo" => "http://nokogiri.org/default_ns/test/foo")
# parent.add_child("<child></child>")
# doc.to_xml
# # => <?xml version="1.0"?>
# # <root xmlns:foo="http://nokogiri.org/default_ns/test/foo">
# # <foo:parent>
# # <child/>
# # </foo:parent>
# # </root>
#
# *Example:* Setting namespace inheritance to `true`
#
# xml = <<~EOF
# <root xmlns:foo="http://nokogiri.org/default_ns/test/foo">
# <foo:parent>
# </foo:parent>
# </root>
# EOF
# doc = Nokogiri::XML(xml)
# doc.namespace_inheritance = true
# parent = doc.at_xpath("//foo:parent", "foo" => "http://nokogiri.org/default_ns/test/foo")
# parent.add_child("<child></child>")
# doc.to_xml
# # => <?xml version="1.0"?>
# # <root xmlns:foo="http://nokogiri.org/default_ns/test/foo">
# # <foo:parent>
# # <foo:child/>
# # </foo:parent>
# # </root>
#
# Since v1.12.4
attr_accessor :namespace_inheritance
def initialize(*args) # :nodoc: # rubocop:disable Lint/MissingSuper
@errors = []
@decorators = nil
@namespace_inheritance = false
end
#
# :call-seq:
# dup → Nokogiri::XML::Document
# dup(level) → Nokogiri::XML::Document
#
# Duplicate this node.
#
# [Parameters]
# - +level+ (optional Integer). 0 is a shallow copy, 1 (the default) is a deep copy.
# [Returns] The new Nokogiri::XML::Document
#
def dup(level = 1)
copy = OBJECT_DUP_METHOD.bind_call(self)
copy.initialize_copy_with_args(self, level)
end
#
# :call-seq:
# clone → Nokogiri::XML::Document
# clone(level) → Nokogiri::XML::Document
#
# Clone this node.
#
# [Parameters]
# - +level+ (optional Integer). 0 is a shallow copy, 1 (the default) is a deep copy.
# [Returns] The new Nokogiri::XML::Document
#
def clone(level = 1)
copy = OBJECT_CLONE_METHOD.bind_call(self)
copy.initialize_copy_with_args(self, level)
end
# :call-seq:
# create_element(name, *contents_or_attrs, &block) → Nokogiri::XML::Element
#
# Create a new Element with `name` belonging to this document, optionally setting contents or
# attributes.
#
# This method is _not_ the most user-friendly option if your intention is to add a node to the
# document tree. Prefer one of the Nokogiri::XML::Node methods like Node#add_child,
# Node#add_next_sibling, Node#replace, etc. which will both create an element (or subtree) and
# place it in the document tree.
#
# Arguments may be passed to initialize the element:
#
# - a Hash argument will be used to set attributes
# - a non-Hash object that responds to \#to_s will be used to set the new node's contents
#
# A block may be passed to mutate the node.
#
# [Parameters]
# - `name` (String)
# - `contents_or_attrs` (\#to_s, Hash)
# [Yields] `node` (Nokogiri::XML::Element)
# [Returns] Nokogiri::XML::Element
#
# *Example:* An empty element without attributes
#
# doc.create_element("div")
# # => <div></div>
#
# *Example:* An element with contents
#
# doc.create_element("div", "contents")
# # => <div>contents</div>
#
# *Example:* An element with attributes
#
# doc.create_element("div", {"class" => "container"})
# # => <div class='container'></div>
#
# *Example:* An element with contents and attributes
#
# doc.create_element("div", "contents", {"class" => "container"})
# # => <div class='container'>contents</div>
#
# *Example:* Passing a block to mutate the element
#
# doc.create_element("div") { |node| node["class"] = "blue" if before_noon? }
#
def create_element(name, *contents_or_attrs, &block)
elm = Nokogiri::XML::Element.new(name, self, &block)
contents_or_attrs.each do |arg|
case arg
when Hash
arg.each do |k, v|
key = k.to_s
if key =~ NCNAME_RE
ns_name = Regexp.last_match(1)
elm.add_namespace_definition(ns_name, v)
else
elm[k.to_s] = v.to_s
end
end
else
elm.content = arg
end
end
if (ns = elm.namespace_definitions.find { |n| n.prefix.nil? || (n.prefix == "") })
elm.namespace = ns
end
elm
end
# Create a Text Node with +string+
def create_text_node(string, &block)
Nokogiri::XML::Text.new(string.to_s, self, &block)
end
# Create a CDATA Node containing +string+
def create_cdata(string, &block)
Nokogiri::XML::CDATA.new(self, string.to_s, &block)
end
# Create a Comment Node containing +string+
def create_comment(string, &block)
Nokogiri::XML::Comment.new(self, string.to_s, &block)
end
# The name of this document. Always returns "document"
def name
"document"
end
# A reference to +self+
def document
self
end
# :call-seq:
# collect_namespaces() → Hash<String(Namespace#prefix) ⇒ String(Namespace#href)>
#
# Recursively get all namespaces from this node and its subtree and return them as a
# hash.
#
# ⚠ This method will not handle duplicate namespace prefixes, since the return value is a hash.
#
# Note that this method does an xpath lookup for nodes with namespaces, and as a result the
# order (and which duplicate prefix "wins") may be dependent on the implementation of the
# underlying XML library.
#
# *Example:* Basic usage
#
# Given this document:
#
# <root xmlns="default" xmlns:foo="bar">
# <bar xmlns:hello="world" />
# </root>
#
# This method will return:
#
# {"xmlns:foo"=>"bar", "xmlns"=>"default", "xmlns:hello"=>"world"}
#
# *Example:* Duplicate prefixes
#
# Given this document:
#
# <root xmlns:foo="bar">
# <bar xmlns:foo="baz" />
# </root>
#
# The hash returned will be something like:
#
# {"xmlns:foo" => "baz"}
#
def collect_namespaces
xpath("//namespace::*").each_with_object({}) do |ns, hash|
hash[["xmlns", ns.prefix].compact.join(":")] = ns.href if ns.prefix != "xml"
end
end
# Get the list of decorators given +key+
def decorators(key)
@decorators ||= {}
@decorators[key] ||= []
end
##
# Validate this Document against its DTD. Returns a list of errors on
# the document or +nil+ when there is no DTD.
def validate
return unless internal_subset
internal_subset.validate(self)
end
##
# Explore a document with shortcut methods. See Nokogiri::Slop for details.
#
# Note that any nodes that have been instantiated before #slop!
# is called will not be decorated with sloppy behavior. So, if you're in
# irb, the preferred idiom is:
#
# irb> doc = Nokogiri::Slop my_markup
#
# and not
#
# irb> doc = Nokogiri::HTML my_markup
# ... followed by irb's implicit inspect (and therefore instantiation of every node) ...
# irb> doc.slop!
# ... which does absolutely nothing.
#
def slop!
unless decorators(XML::Node).include?(Nokogiri::Decorators::Slop)
decorators(XML::Node) << Nokogiri::Decorators::Slop
decorate!
end
self
end
##
# Apply any decorators to +node+
def decorate(node)
return unless @decorators
@decorators.each do |klass, list|
next unless node.is_a?(klass)
list.each { |mod| node.extend(mod) }
end
end
alias_method :to_xml, :serialize
# Get the hash of namespaces on the root Nokogiri::XML::Node
def namespaces
root ? root.namespaces : {}
end
##
# Create a Nokogiri::XML::DocumentFragment from +tags+
# Returns an empty fragment if +tags+ is nil.
def fragment(tags = nil)
DocumentFragment.new(self, tags, root)
end
undef_method :swap, :parent, :namespace, :default_namespace=
undef_method :add_namespace_definition, :attributes
undef_method :namespace_definitions, :line, :add_namespace
def add_child(node_or_tags)
raise "A document may not have multiple root nodes." if (root && root.name != "nokogiri_text_wrapper") && !(node_or_tags.comment? || node_or_tags.processing_instruction?)
node_or_tags = coerce(node_or_tags)
if node_or_tags.is_a?(XML::NodeSet)
raise "A document may not have multiple root nodes." if node_or_tags.size > 1
super(node_or_tags.first)
else
super
end
end
alias_method :<<, :add_child
# :call-seq:
# xpath_doctype() → Nokogiri::CSS::XPathVisitor::DoctypeConfig
#
# [Returns] The document type which determines CSS-to-XPath translation.
#
# See XPathVisitor for more information.
def xpath_doctype
Nokogiri::CSS::XPathVisitor::DoctypeConfig::XML
end
#
# :call-seq: deconstruct_keys(array_of_names) → Hash
#
# Returns a hash describing the Document, to use in pattern matching.
#
# Valid keys and their values:
# - +root+ → (Node, nil) The root node of the Document, or +nil+ if the document is empty.
#
# In the future, other keys may allow accessing things like doctype and processing
# instructions. If you have a use case and would like this functionality, please let us know
# by opening an issue or a discussion on the github project.
#
# *Example*
#
# doc = Nokogiri::XML.parse(<<~XML)
# <?xml version="1.0"?>
# <root>
# <child>
# </root>
# XML
#
# doc.deconstruct_keys([:root])
# # => {:root=>
# # #(Element:0x35c {
# # name = "root",
# # children = [
# # #(Text "\n" + " "),
# # #(Element:0x370 { name = "child", children = [ #(Text "\n")] }),
# # #(Text "\n")]
# # })}
#
# *Example* of an empty document
#
# doc = Nokogiri::XML::Document.new
#
# doc.deconstruct_keys([:root])
# # => {:root=>nil}
#
# Since v1.14.0
#
def deconstruct_keys(keys)
{ root: root }
end
private
IMPLIED_XPATH_CONTEXTS = ["//"].freeze # :nodoc:
def inspect_attributes
[:name, :children]
end
end
end
end
@@ -0,0 +1,276 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
# DocumentFragment represents a fragment of an \XML document. It provides the same functionality
# exposed by XML::Node and can be used to contain one or more \XML subtrees.
class DocumentFragment < Nokogiri::XML::Node
# The options used to parse the document fragment. Returns the value of any options that were
# passed into the constructor as a parameter or set in a config block, else the default
# options for the specific subclass.
attr_reader :parse_options
class << self
# :call-seq:
# parse(input) { |options| ... } → XML::DocumentFragment
# parse(input, options:) → XML::DocumentFragment
#
# Parse \XML fragment input from a String, and return a new XML::DocumentFragment. This
# method creates a new, empty XML::Document to contain the fragment.
#
# [Required Parameters]
# - +input+ (String) The content to be parsed.
#
# [Optional Keyword Arguments]
# - +options+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
# behaviors during parsing. See ParseOptions for more information. The default value is
# +ParseOptions::DEFAULT_XML+.
#
# [Yields]
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
# can be configured before parsing. See Nokogiri::XML::ParseOptions for more information.
#
# [Returns] Nokogiri::XML::DocumentFragment
def parse(tags, options_ = ParseOptions::DEFAULT_XML, options: options_, &block)
new(XML::Document.new, tags, options: options, &block)
end
# Wrapper method to separate the concerns of:
# - the native object allocator's parameter (it only requires `document`)
# - the initializer's parameters
def new(document, ...) # :nodoc:
instance = native_new(document)
instance.send(:initialize, document, ...)
instance
end
end
# :call-seq:
# new(document, input=nil) { |options| ... } → DocumentFragment
# new(document, input=nil, context:, options:) → DocumentFragment
#
# Parse \XML fragment input from a String, and return a new DocumentFragment that is
# associated with the given +document+.
#
# 💡 It's recommended to use either XML::DocumentFragment.parse or Node#parse rather than call
# this method directly.
#
# [Required Parameters]
# - +document+ (XML::Document) The parent document to associate the returned fragment with.
#
# [Optional Parameters]
# - +input+ (String) The content to be parsed.
#
# [Optional Keyword Arguments]
# - +context:+ (Nokogiri::XML::Node) The <b>context node</b> for the subtree created. See
# below for more information.
#
# - +options:+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
# behaviors during parsing. See ParseOptions for more information. The default value is
# +ParseOptions::DEFAULT_XML+.
#
# [Yields]
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
# can be configured before parsing. See ParseOptions for more information.
#
# [Returns] XML::DocumentFragment
#
# === Context \Node
#
# If a context node is specified using +context:+, then the fragment will be created by
# calling Node#parse on that node, so the parser will behave as if that Node is the parent of
# the fragment subtree, and will resolve namespaces relative to that node.
#
def initialize(
document, tags = nil,
context_ = nil, options_ = ParseOptions::DEFAULT_XML,
context: context_, options: options_
) # rubocop:disable Lint/MissingSuper
return self unless tags
options = Nokogiri::XML::ParseOptions.new(options) if Integer === options
@parse_options = options
yield options if block_given?
children = if context
# Fix for issue#490
if Nokogiri.jruby?
# fix for issue #770
context.parse("<root #{namespace_declarations(context)}>#{tags}</root>", options).children
else
context.parse(tags, options)
end
else
wrapper_doc = XML::Document.parse("<root>#{tags}</root>", nil, nil, options)
self.errors = wrapper_doc.errors
wrapper_doc.xpath("/root/node()")
end
children.each { |child| child.parent = self }
end
if Nokogiri.uses_libxml?
def dup
new_document = document.dup
new_fragment = self.class.new(new_document)
children.each do |child|
child.dup(1, new_document).parent = new_fragment
end
new_fragment
end
end
###
# return the name for DocumentFragment
def name
"#document-fragment"
end
###
# Convert this DocumentFragment to a string
def to_s
children.to_s
end
###
# Convert this DocumentFragment to html
# See Nokogiri::XML::NodeSet#to_html
def to_html(*args)
if Nokogiri.jruby?
options = args.first.is_a?(Hash) ? args.shift : {}
options[:save_with] ||= Node::SaveOptions::DEFAULT_HTML
args.insert(0, options)
end
children.to_html(*args)
end
###
# Convert this DocumentFragment to xhtml
# See Nokogiri::XML::NodeSet#to_xhtml
def to_xhtml(*args)
if Nokogiri.jruby?
options = args.first.is_a?(Hash) ? args.shift : {}
options[:save_with] ||= Node::SaveOptions::DEFAULT_XHTML
args.insert(0, options)
end
children.to_xhtml(*args)
end
###
# Convert this DocumentFragment to xml
# See Nokogiri::XML::NodeSet#to_xml
def to_xml(*args)
children.to_xml(*args)
end
###
# call-seq: css *rules, [namespace-bindings, custom-pseudo-class]
#
# Search this fragment for CSS +rules+. +rules+ must be one or more CSS
# selectors. For example:
#
# For more information see Nokogiri::XML::Searchable#css
def css(*args)
if children.any?
children.css(*args) # 'children' is a smell here
else
NodeSet.new(document)
end
end
#
# NOTE that we don't delegate #xpath to children ... another smell.
# def xpath ; end
#
###
# call-seq: search *paths, [namespace-bindings, xpath-variable-bindings, custom-handler-class]
#
# Search this fragment for +paths+. +paths+ must be one or more XPath or CSS queries.
#
# For more information see Nokogiri::XML::Searchable#search
def search(*rules)
rules, handler, ns, binds = extract_params(rules)
rules.inject(NodeSet.new(document)) do |set, rule|
set + if Searchable::LOOKS_LIKE_XPATH.match?(rule)
xpath(*[rule, ns, handler, binds].compact)
else
children.css(*[rule, ns, handler].compact) # 'children' is a smell here
end
end
end
alias_method :serialize, :to_s
# A list of Nokogiri::XML::SyntaxError found when parsing a document
def errors
document.errors
end
def errors=(things) # :nodoc:
document.errors = things
end
def fragment(data)
document.fragment(data)
end
#
# :call-seq: deconstruct() → Array
#
# Returns the root nodes of this document fragment as an array, to use in pattern matching.
#
# 💡 Note that text nodes are returned as well as elements. If you wish to operate only on
# root elements, you should deconstruct the array returned by
# <tt>DocumentFragment#elements</tt>.
#
# *Example*
#
# frag = Nokogiri::HTML5.fragment(<<~HTML)
# <div>Start</div>
# This is a <a href="#jump">shortcut</a> for you.
# <div>End</div>
# HTML
#
# frag.deconstruct
# # => [#(Element:0x35c { name = "div", children = [ #(Text "Start")] }),
# # #(Text "\n" + "This is a "),
# # #(Element:0x370 {
# # name = "a",
# # attributes = [ #(Attr:0x384 { name = "href", value = "#jump" })],
# # children = [ #(Text "shortcut")]
# # }),
# # #(Text " for you.\n"),
# # #(Element:0x398 { name = "div", children = [ #(Text "End")] }),
# # #(Text "\n")]
#
# *Example* only the elements, not the text nodes.
#
# frag.elements.deconstruct
# # => [#(Element:0x35c { name = "div", children = [ #(Text "Start")] }),
# # #(Element:0x370 {
# # name = "a",
# # attributes = [ #(Attr:0x384 { name = "href", value = "#jump" })],
# # children = [ #(Text "shortcut")]
# # }),
# # #(Element:0x398 { name = "div", children = [ #(Text "End")] })]
#
# Since v1.14.0
#
def deconstruct
children.to_a
end
private
# fix for issue 770
def namespace_declarations(ctx)
ctx.namespace_scopes.map do |namespace|
prefix = namespace.prefix.nil? ? "" : ":#{namespace.prefix}"
%{xmlns#{prefix}="#{namespace.href}"}
end.join(" ")
end
end
end
end
@@ -0,0 +1,34 @@
# frozen_string_literal: true
module Nokogiri
module XML
class DTD < Nokogiri::XML::Node
undef_method :attribute_nodes
undef_method :values
undef_method :content
undef_method :namespace
undef_method :namespace_definitions
undef_method :line if method_defined?(:line)
def keys
attributes.keys
end
def each
attributes.each do |key, value|
yield([key, value])
end
end
def html_dtd?
name.casecmp("html").zero?
end
def html5_dtd?
html_dtd? &&
external_id.nil? &&
(system_id.nil? || system_id == "about:legacy-compat")
end
end
end
end
@@ -0,0 +1,46 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# Represents the allowed content in an Element Declaration inside a DTD:
#
# <?xml version="1.0"?><?TEST-STYLE PIDATA?>
# <!DOCTYPE staff SYSTEM "staff.dtd" [
# <!ELEMENT div1 (head, (p | list | note)*, div2*)>
# ]>
# </root>
#
# ElementContent represents the binary tree inside the <!ELEMENT> tag shown above that lists the
# possible content for the div1 tag.
class ElementContent
include Nokogiri::XML::PP::Node
# Possible definitions of type
PCDATA = 1
ELEMENT = 2
SEQ = 3
OR = 4
# Possible content occurrences
ONCE = 1
OPT = 2
MULT = 3
PLUS = 4
attr_reader :document
###
# Get the children of this ElementContent node
def children
[c1, c2].compact
end
private
def inspect_attributes
[:prefix, :name, :type, :occur, :children]
end
end
end
end
@@ -0,0 +1,17 @@
# frozen_string_literal: true
module Nokogiri
module XML
class ElementDecl < Nokogiri::XML::Node
undef_method :namespace
undef_method :namespace_definitions
undef_method :line if method_defined?(:line)
private
def inspect_attributes
[:to_s]
end
end
end
end
@@ -0,0 +1,23 @@
# frozen_string_literal: true
module Nokogiri
module XML
class EntityDecl < Nokogiri::XML::Node
undef_method :attribute_nodes
undef_method :attributes
undef_method :namespace
undef_method :namespace_definitions
undef_method :line if method_defined?(:line)
def self.new(name, doc, *args)
doc.create_entity(name, *args)
end
private
def inspect_attributes
[:to_s]
end
end
end
end
@@ -0,0 +1,20 @@
# frozen_string_literal: true
module Nokogiri
module XML
class EntityReference < Nokogiri::XML::Node
def children
# libxml2 will create a malformed child node for predefined
# entities. because any use of that child is likely to cause a
# segfault, we shall pretend that it doesn't exist.
#
# see https://github.com/sparklemotion/nokogiri/issues/1238 for details
NodeSet.new(document)
end
def inspect_attributes
[:name]
end
end
end
end
@@ -0,0 +1,57 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
class Namespace
include Nokogiri::XML::PP::Node
attr_reader :document
#
# :call-seq: deconstruct_keys(array_of_names) → Hash
#
# Returns a hash describing the Namespace, to use in pattern matching.
#
# Valid keys and their values:
# - +prefix+ → (String, nil) The namespace's prefix, or +nil+ if there is no prefix (e.g., default namespace).
# - +href+ → (String) The namespace's URI
#
# *Example*
#
# doc = Nokogiri::XML.parse(<<~XML)
# <?xml version="1.0"?>
# <root xmlns="http://nokogiri.org/ns/default" xmlns:noko="http://nokogiri.org/ns/noko">
# <child1 foo="abc" noko:bar="def"/>
# <noko:child2 foo="qwe" noko:bar="rty"/>
# </root>
# XML
#
# doc.root.elements.first.namespace
# # => #(Namespace:0x35c { href = "http://nokogiri.org/ns/default" })
#
# doc.root.elements.first.namespace.deconstruct_keys([:prefix, :href])
# # => {:prefix=>nil, :href=>"http://nokogiri.org/ns/default"}
#
# doc.root.elements.last.namespace
# # => #(Namespace:0x370 {
# # prefix = "noko",
# # href = "http://nokogiri.org/ns/noko"
# # })
#
# doc.root.elements.last.namespace.deconstruct_keys([:prefix, :href])
# # => {:prefix=>"noko", :href=>"http://nokogiri.org/ns/noko"}
#
# Since v1.14.0
#
def deconstruct_keys(keys)
{ prefix: prefix, href: href }
end
private
def inspect_attributes
[:prefix, :href]
end
end
end
end
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,76 @@
# frozen_string_literal: true
module Nokogiri
module XML
class Node
###
# Save options for serializing nodes.
# See the method group entitled Node@Serialization+and+Generating+Output for usage.
class SaveOptions
# Format serialized xml
FORMAT = 1
# Do not include declarations
NO_DECLARATION = 2
# Do not include empty tags
NO_EMPTY_TAGS = 4
# Do not save XHTML
NO_XHTML = 8
# Save as XHTML
AS_XHTML = 16
# Save as XML
AS_XML = 32
# Save as HTML
AS_HTML = 64
if Nokogiri.jruby?
# Save builder created document
AS_BUILDER = 128
# the default for XML documents
DEFAULT_XML = AS_XML # https://github.com/sparklemotion/nokogiri/issues/#issue/415
# the default for HTML document
DEFAULT_HTML = NO_DECLARATION | NO_EMPTY_TAGS | AS_HTML
# the default for XHTML document
DEFAULT_XHTML = NO_DECLARATION | AS_XHTML
else
# the default for XML documents
DEFAULT_XML = FORMAT | AS_XML
# the default for HTML document
DEFAULT_HTML = FORMAT | NO_DECLARATION | NO_EMPTY_TAGS | AS_HTML
# the default for XHTML document
DEFAULT_XHTML = FORMAT | NO_DECLARATION | AS_XHTML
end
# Integer representation of the SaveOptions
attr_reader :options
# Create a new SaveOptions object with +options+
def initialize(options = 0)
@options = options
end
constants.each do |constant|
class_eval <<~RUBY, __FILE__, __LINE__ + 1
def #{constant.downcase}
@options |= #{constant}
self
end
def #{constant.downcase}?
#{constant} & @options == #{constant}
end
RUBY
end
alias_method :to_i, :options
def inspect
options = []
self.class.constants.each do |k|
options << k.downcase if send(:"#{k.downcase}?")
end
super.sub(/>$/, " " + options.join(", ") + ">")
end
end
end
end
end
@@ -0,0 +1,449 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
####
# A NodeSet is an Enumerable that contains a list of Nokogiri::XML::Node objects.
#
# Typically a NodeSet is returned as a result of searching a Document via
# Nokogiri::XML::Searchable#css or Nokogiri::XML::Searchable#xpath.
#
# Note that the `#dup` and `#clone` methods perform shallow copies; these methods do not copy
# the Nodes contained in the NodeSet (similar to how Array and other Enumerable classes work).
class NodeSet
include Nokogiri::XML::Searchable
include Enumerable
# The Document this NodeSet is associated with
attr_accessor :document
# Create a NodeSet with +document+ defaulting to +list+
def initialize(document, list = [])
@document = document
document.decorate(self)
list.each { |x| self << x }
yield self if block_given?
end
###
# Get the first element of the NodeSet.
def first(n = nil)
return self[0] unless n
list = []
[n, length].min.times { |i| list << self[i] }
list
end
###
# Get the last element of the NodeSet.
def last
self[-1]
end
###
# Is this NodeSet empty?
def empty?
length == 0
end
###
# Returns the index of the first node in self that is == to +node+ or meets the given block. Returns nil if no match is found.
def index(node = nil)
if node
warn("given block not used") if block_given?
each_with_index { |member, j| return j if member == node }
elsif block_given?
each_with_index { |member, j| return j if yield(member) }
end
nil
end
###
# Insert +datum+ before the first Node in this NodeSet
def before(datum)
first.before(datum)
end
###
# Insert +datum+ after the last Node in this NodeSet
def after(datum)
last.after(datum)
end
alias_method :<<, :push
alias_method :remove, :unlink
###
# call-seq: css *rules, [namespace-bindings, custom-pseudo-class]
#
# Search this node set for CSS +rules+. +rules+ must be one or more CSS
# selectors. For example:
#
# For more information see Nokogiri::XML::Searchable#css
def css(*args)
rules, handler, ns, _ = extract_params(args)
paths = css_rules_to_xpath(rules, ns)
inject(NodeSet.new(document)) do |set, node|
set + xpath_internal(node, paths, handler, ns, nil)
end
end
###
# call-seq: xpath *paths, [namespace-bindings, variable-bindings, custom-handler-class]
#
# Search this node set for XPath +paths+. +paths+ must be one or more XPath
# queries.
#
# For more information see Nokogiri::XML::Searchable#xpath
def xpath(*args)
paths, handler, ns, binds = extract_params(args)
inject(NodeSet.new(document)) do |set, node|
set + xpath_internal(node, paths, handler, ns, binds)
end
end
###
# call-seq: search *paths, [namespace-bindings, xpath-variable-bindings, custom-handler-class]
#
# Search this object for +paths+, and return only the first
# result. +paths+ must be one or more XPath or CSS queries.
#
# See Searchable#search for more information.
#
# Or, if passed an integer, index into the NodeSet:
#
# node_set.at(3) # same as node_set[3]
#
def at(*args)
if args.length == 1 && args.first.is_a?(Numeric)
return self[args.first]
end
super
end
alias_method :%, :at
###
# Filter this list for nodes that match +expr+
def filter(expr)
find_all { |node| node.matches?(expr) }
end
###
# Add the class attribute +name+ to all Node objects in the
# NodeSet.
#
# See Nokogiri::XML::Node#add_class for more information.
def add_class(name)
each do |el|
el.add_class(name)
end
self
end
###
# Append the class attribute +name+ to all Node objects in the
# NodeSet.
#
# See Nokogiri::XML::Node#append_class for more information.
def append_class(name)
each do |el|
el.append_class(name)
end
self
end
###
# Remove the class attribute +name+ from all Node objects in the
# NodeSet.
#
# See Nokogiri::XML::Node#remove_class for more information.
def remove_class(name = nil)
each do |el|
el.remove_class(name)
end
self
end
###
# Set attributes on each Node in the NodeSet, or get an
# attribute from the first Node in the NodeSet.
#
# To get an attribute from the first Node in a NodeSet:
#
# node_set.attr("href") # => "https://www.nokogiri.org"
#
# Note that an empty NodeSet will return nil when +#attr+ is called as a getter.
#
# To set an attribute on each node, +key+ can either be an
# attribute name, or a Hash of attribute names and values. When
# called as a setter, +#attr+ returns the NodeSet.
#
# If +key+ is an attribute name, then either +value+ or +block+
# must be passed.
#
# If +key+ is a Hash then attributes will be set for each
# key/value pair:
#
# node_set.attr("href" => "https://www.nokogiri.org", "class" => "member")
#
# If +value+ is passed, it will be used as the attribute value
# for all nodes:
#
# node_set.attr("href", "https://www.nokogiri.org")
#
# If +block+ is passed, it will be called on each Node object in
# the NodeSet and the return value used as the attribute value
# for that node:
#
# node_set.attr("class") { |node| node.name }
#
def attr(key, value = nil, &block)
unless key.is_a?(Hash) || (key && (value || block))
return first&.attribute(key)
end
hash = key.is_a?(Hash) ? key : { key => value }
hash.each do |k, v|
each do |node|
node[k] = v || yield(node)
end
end
self
end
alias_method :set, :attr
alias_method :attribute, :attr
###
# Remove the attributed named +name+ from all Node objects in the NodeSet
def remove_attr(name)
each { |el| el.delete(name) }
self
end
alias_method :remove_attribute, :remove_attr
###
# Iterate over each node, yielding to +block+
def each
return to_enum unless block_given?
0.upto(length - 1) do |x|
yield self[x]
end
self
end
###
# Get the inner text of all contained Node objects
#
# Note: This joins the text of all Node objects in the NodeSet:
#
# doc = Nokogiri::XML('<xml><a><d>foo</d><d>bar</d></a></xml>')
# doc.css('d').text # => "foobar"
#
# Instead, if you want to return the text of all nodes in the NodeSet:
#
# doc.css('d').map(&:text) # => ["foo", "bar"]
#
# See Nokogiri::XML::Node#content for more information.
def inner_text
collect(&:inner_text).join("")
end
alias_method :text, :inner_text
###
# Get the inner html of all contained Node objects
def inner_html(*args)
collect { |j| j.inner_html(*args) }.join("")
end
# :call-seq:
# wrap(markup) -> self
# wrap(node) -> self
#
# Wrap each member of this NodeSet with the node parsed from +markup+ or a dup of the +node+.
#
# [Parameters]
# - *markup* (String)
# Markup that is parsed, once per member of the NodeSet, and used as the wrapper. Each
# node's parent, if it exists, is used as the context node for parsing; otherwise the
# associated document is used. If the parsed fragment has multiple roots, the first root
# node is used as the wrapper.
# - *node* (Nokogiri::XML::Node)
# An element that is `#dup`ed and used as the wrapper.
#
# [Returns] +self+, to support chaining.
#
# ⚠ Note that if a +String+ is passed, the markup will be parsed <b>once per node</b> in the
# NodeSet. You can avoid this overhead in cases where you know exactly the wrapper you wish to
# use by passing a +Node+ instead.
#
# Also see Node#wrap
#
# *Example* with a +String+ argument:
#
# doc = Nokogiri::HTML5(<<~HTML)
# <html><body>
# <a>a</a>
# <a>b</a>
# <a>c</a>
# <a>d</a>
# </body></html>
# HTML
# doc.css("a").wrap("<div></div>")
# doc.to_html
# # => <html><head></head><body>
# # <div><a>a</a></div>
# # <div><a>b</a></div>
# # <div><a>c</a></div>
# # <div><a>d</a></div>
# # </body></html>
#
# *Example* with a +Node+ argument
#
# 💡 Note that this is faster than the equivalent call passing a +String+ because it avoids
# having to reparse the wrapper markup for each node.
#
# doc = Nokogiri::HTML5(<<~HTML)
# <html><body>
# <a>a</a>
# <a>b</a>
# <a>c</a>
# <a>d</a>
# </body></html>
# HTML
# doc.css("a").wrap(doc.create_element("div"))
# doc.to_html
# # => <html><head></head><body>
# # <div><a>a</a></div>
# # <div><a>b</a></div>
# # <div><a>c</a></div>
# # <div><a>d</a></div>
# # </body></html>
#
def wrap(node_or_tags)
map { |node| node.wrap(node_or_tags) }
self
end
###
# Convert this NodeSet to a string.
def to_s
map(&:to_s).join
end
###
# Convert this NodeSet to HTML
def to_html(*args)
if Nokogiri.jruby?
options = args.first.is_a?(Hash) ? args.shift : {}
options[:save_with] ||= Node::SaveOptions::DEFAULT_HTML
args.insert(0, options)
end
if empty?
encoding = (args.first.is_a?(Hash) ? args.first[:encoding] : nil)
encoding ||= document.encoding
encoding.nil? ? "" : "".encode(encoding)
else
map { |x| x.to_html(*args) }.join
end
end
###
# Convert this NodeSet to XHTML
def to_xhtml(*args)
map { |x| x.to_xhtml(*args) }.join
end
###
# Convert this NodeSet to XML
def to_xml(*args)
map { |x| x.to_xml(*args) }.join
end
alias_method :size, :length
alias_method :to_ary, :to_a
###
# Removes the last element from set and returns it, or +nil+ if
# the set is empty
def pop
return if length == 0
delete(last)
end
###
# Returns the first element of the NodeSet and removes it. Returns
# +nil+ if the set is empty.
def shift
return if length == 0
delete(first)
end
###
# Equality -- Two NodeSets are equal if the contain the same number
# of elements and if each element is equal to the corresponding
# element in the other NodeSet
def ==(other)
return false unless other.is_a?(Nokogiri::XML::NodeSet)
return false unless length == other.length
each_with_index do |node, i|
return false unless node == other[i]
end
true
end
###
# Returns a new NodeSet containing all the children of all the nodes in
# the NodeSet
def children
node_set = NodeSet.new(document)
each do |node|
node.children.each { |n| node_set.push(n) }
end
node_set
end
###
# Returns a new NodeSet containing all the nodes in the NodeSet
# in reverse order
def reverse
node_set = NodeSet.new(document)
(length - 1).downto(0) do |x|
node_set.push(self[x])
end
node_set
end
###
# Return a nicely formatted string representation
def inspect
"[#{map(&:inspect).join(", ")}]"
end
alias_method :+, :|
#
# :call-seq: deconstruct() → Array
#
# Returns the members of this NodeSet as an array, to use in pattern matching.
#
# Since v1.14.0
#
def deconstruct
to_a
end
IMPLIED_XPATH_CONTEXTS = [".//", "self::"].freeze # :nodoc:
end
end
end
@@ -0,0 +1,19 @@
# frozen_string_literal: true
module Nokogiri
module XML
# Struct representing an {XML Schema Notation}[https://www.w3.org/TR/xml/#Notations]
class Notation < Struct.new(:name, :public_id, :system_id)
# dead comment to ensure rdoc processing
# :attr: name (String)
# The name for the element.
# :attr: public_id (String)
# The URI corresponding to the public identifier
# :attr: system_id (String,nil)
# The URI corresponding to the system identifier
end
end
end
@@ -0,0 +1,213 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
# Options that control the parsing behavior for XML::Document, XML::DocumentFragment,
# HTML4::Document, HTML4::DocumentFragment, XSLT::Stylesheet, and XML::Schema.
#
# These options directly expose libxml2's parse options, which are all boolean in the sense that
# an option is "on" or "off".
#
# 💡 Note that HTML5 parsing has a separate, orthogonal set of options due to the nature of the
# HTML5 specification. See Nokogiri::HTML5.
#
# ⚠ Not all parse options are supported on JRuby. Nokogiri will attempt to invoke the equivalent
# behavior in Xerces/NekoHTML on JRuby when it's possible.
#
# == Setting and unsetting parse options
#
# You can build your own combinations of parse options by using any of the following methods:
#
# [ParseOptions method chaining]
#
# Every option has an equivalent method in lowercase. You can chain these methods together to
# set various combinations.
#
# # Set the HUGE & PEDANTIC options
# po = Nokogiri::XML::ParseOptions.new.huge.pedantic
# doc = Nokogiri::XML::Document.parse(xml, nil, nil, po)
#
# Every option has an equivalent <code>no{option}</code> method in lowercase. You can call these
# methods on an instance of ParseOptions to unset the option.
#
# # Set the HUGE & PEDANTIC options
# po = Nokogiri::XML::ParseOptions.new.huge.pedantic
#
# # later we want to modify the options
# po.nohuge # Unset the HUGE option
# po.nopedantic # Unset the PEDANTIC option
#
# 💡 Note that some options begin with "no" leading to the logical but perhaps unintuitive
# double negative:
#
# po.nocdata # Set the NOCDATA parse option
# po.nonocdata # Unset the NOCDATA parse option
#
# 💡 Note that negation is not available for STRICT, which is itself a negation of all other
# features.
#
#
# [Using Ruby Blocks]
#
# Most parsing methods will accept a block for configuration of parse options, and we
# recommend chaining the setter methods:
#
# doc = Nokogiri::XML::Document.parse(xml) { |config| config.huge.pedantic }
#
#
# [ParseOptions constants]
#
# You can also use the constants declared under Nokogiri::XML::ParseOptions to set various
# combinations. They are bits in a bitmask, and so can be combined with bitwise operators:
#
# po = Nokogiri::XML::ParseOptions.new(Nokogiri::XML::ParseOptions::HUGE | Nokogiri::XML::ParseOptions::PEDANTIC)
# doc = Nokogiri::XML::Document.parse(xml, nil, nil, po)
#
class ParseOptions
# Strict parsing
STRICT = 0
# Recover from errors. On by default for XML::Document, XML::DocumentFragment,
# HTML4::Document, HTML4::DocumentFragment, XSLT::Stylesheet, and XML::Schema.
RECOVER = 1 << 0
# Substitute entities. Off by default.
#
# ⚠ This option enables entity substitution, contrary to what the name implies.
#
# ⚠ <b>It is UNSAFE to set this option</b> when parsing untrusted documents.
NOENT = 1 << 1
# Load external subsets. On by default for XSLT::Stylesheet.
#
# ⚠ <b>It is UNSAFE to set this option</b> when parsing untrusted documents.
DTDLOAD = 1 << 2
# Default DTD attributes. On by default for XSLT::Stylesheet.
DTDATTR = 1 << 3
# Validate with the DTD. Off by default.
DTDVALID = 1 << 4
# Suppress error reports. On by default for HTML4::Document and HTML4::DocumentFragment
NOERROR = 1 << 5
# Suppress warning reports. On by default for HTML4::Document and HTML4::DocumentFragment
NOWARNING = 1 << 6
# Enable pedantic error reporting. Off by default.
PEDANTIC = 1 << 7
# Remove blank nodes. Off by default.
NOBLANKS = 1 << 8
# Use the SAX1 interface internally. Off by default.
SAX1 = 1 << 9
# Implement XInclude substitution. Off by default.
XINCLUDE = 1 << 10
# Forbid network access. On by default for XML::Document, XML::DocumentFragment,
# HTML4::Document, HTML4::DocumentFragment, XSLT::Stylesheet, and XML::Schema.
#
# ⚠ <b>It is UNSAFE to unset this option</b> when parsing untrusted documents.
NONET = 1 << 11
# Do not reuse the context dictionary. Off by default.
NODICT = 1 << 12
# Remove redundant namespaces declarations. Off by default.
NSCLEAN = 1 << 13
# Merge CDATA as text nodes. On by default for XSLT::Stylesheet.
NOCDATA = 1 << 14
# Do not generate XInclude START/END nodes. Off by default.
NOXINCNODE = 1 << 15
# Compact small text nodes. Off by default.
#
# ⚠ No modification of the DOM tree is allowed after parsing. libxml2 may crash if you try to
# modify the tree.
COMPACT = 1 << 16
# Parse using XML-1.0 before update 5. Off by default
OLD10 = 1 << 17
# Do not fixup XInclude xml:base uris. Off by default
NOBASEFIX = 1 << 18
# Relax any hardcoded limit from the parser. Off by default.
#
# ⚠ <b>It is UNSAFE to set this option</b> when parsing untrusted documents.
HUGE = 1 << 19
# Support line numbers up to <code>long int</code> (default is a <code>short int</code>). On
# by default for for XML::Document, XML::DocumentFragment, HTML4::Document,
# HTML4::DocumentFragment, XSLT::Stylesheet, and XML::Schema.
BIG_LINES = 1 << 22
# The options mask used by default for parsing XML::Document and XML::DocumentFragment
DEFAULT_XML = RECOVER | NONET | BIG_LINES
# The options mask used by default used for parsing XSLT::Stylesheet
DEFAULT_XSLT = RECOVER | NONET | NOENT | DTDLOAD | DTDATTR | NOCDATA | BIG_LINES
# The options mask used by default used for parsing HTML4::Document and HTML4::DocumentFragment
DEFAULT_HTML = RECOVER | NOERROR | NOWARNING | NONET | BIG_LINES
# The options mask used by default used for parsing XML::Schema
DEFAULT_SCHEMA = NONET | BIG_LINES
attr_accessor :options
def initialize(options = STRICT)
@options = options
end
constants.each do |constant|
next if constant.to_sym == :STRICT
class_eval <<~RUBY, __FILE__, __LINE__ + 1
def #{constant.downcase}
@options |= #{constant}
self
end
def no#{constant.downcase}
@options &= ~#{constant}
self
end
def #{constant.downcase}?
#{constant} & @options == #{constant}
end
RUBY
end
def strict
@options &= ~RECOVER
self
end
def strict?
@options & RECOVER == STRICT
end
def ==(other)
other.to_i == to_i
end
alias_method :to_i, :options
def inspect
options = []
self.class.constants.each do |k|
options << k.downcase if send(:"#{k.downcase}?")
end
super.sub(/>$/, " " + options.join(", ") + ">")
end
end
end
end
@@ -0,0 +1,4 @@
# frozen_string_literal: true
require_relative "pp/node"
require_relative "pp/character_data"
@@ -0,0 +1,21 @@
# frozen_string_literal: true
module Nokogiri
module XML
# :nodoc: all
module PP
module CharacterData
def pretty_print(pp)
nice_name = self.class.name.split("::").last
pp.group(2, "#(#{nice_name} ", ")") do
pp.pp(text)
end
end
def inspect
"#<#{self.class.name}:#{format("0x%x", object_id)} #{text.inspect}>"
end
end
end
end
end
@@ -0,0 +1,73 @@
# frozen_string_literal: true
module Nokogiri
module XML
# :nodoc: all
module PP
module Node
COLLECTIONS = [:attribute_nodes, :children]
def inspect
# handle the case where an exception is thrown during object construction
if respond_to?(:data_ptr?) && !data_ptr?
return "#<#{self.class}:#{format("0x%x", object_id)} (no data)>"
end
attributes = inspect_attributes.reject do |x|
attribute = send(x)
!attribute || (attribute.respond_to?(:empty?) && attribute.empty?)
rescue NoMethodError
true
end
attributes = if inspect_attributes.length == 1
send(attributes.first).inspect
else
attributes.map do |attribute|
"#{attribute}=#{send(attribute).inspect}"
end.join(" ")
end
"#<#{self.class}:#{format("0x%x", object_id)} #{attributes}>"
end
def pretty_print(pp)
nice_name = self.class.name.split("::").last
pp.group(2, "#(#{nice_name}:#{format("0x%x", object_id)} {", "})") do
pp.breakable
attrs = inspect_attributes.filter_map do |t|
[t, send(t)] if respond_to?(t)
end.find_all do |x|
if x.last
if COLLECTIONS.include?(x.first)
!x.last.empty?
else
true
end
end
end
if inspect_attributes.length == 1
pp.pp(attrs.first.last)
else
pp.seplist(attrs) do |v|
if COLLECTIONS.include?(v.first)
pp.group(2, "#{v.first} = [", "]") do
pp.breakable
pp.seplist(v.last) do |item|
pp.pp(item)
end
end
else
pp.text("#{v.first} = ")
pp.pp(v.last)
end
end
end
pp.breakable
end
end
end
end
end
end
@@ -0,0 +1,11 @@
# frozen_string_literal: true
module Nokogiri
module XML
class ProcessingInstruction < Node
def initialize(document, name, content)
super(document, name)
end
end
end
end
@@ -0,0 +1,139 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# The Reader parser allows you to effectively pull parse an \XML document. Once instantiated,
# call Nokogiri::XML::Reader#each to iterate over each node.
#
# Nokogiri::XML::Reader parses an \XML document similar to the way a cursor would move. The
# Reader is given an \XML document, and yields nodes to an each block.
#
# The Reader parser might be good for when you need the speed and low memory usage of a \SAX
# parser, but do not want to write a SAX::Document handler.
#
# Here is an example of usage:
#
# reader = Nokogiri::XML::Reader.new <<~XML
# <x xmlns:tenderlove='http://tenderlovemaking.com/'>
# <tenderlove:foo awesome='true'>snuggles!</tenderlove:foo>
# </x>
# XML
#
# reader.each do |node|
# # node is an instance of Nokogiri::XML::Reader
# puts node.name
# end
#
# ⚠ Nokogiri::XML::Reader#each can only be called once! Once the cursor moves through the entire
# document, you must parse the document again. It may be better to capture all information you
# need during a single iteration.
#
# ⚠ libxml2 does not support error recovery in the Reader parser. The +RECOVER+ ParseOption is
# ignored. If a syntax error is encountered during parsing, an exception will be raised.
class Reader
include Enumerable
TYPE_NONE = 0
# Element node type
TYPE_ELEMENT = 1
# Attribute node type
TYPE_ATTRIBUTE = 2
# Text node type
TYPE_TEXT = 3
# CDATA node type
TYPE_CDATA = 4
# Entity Reference node type
TYPE_ENTITY_REFERENCE = 5
# Entity node type
TYPE_ENTITY = 6
# PI node type
TYPE_PROCESSING_INSTRUCTION = 7
# Comment node type
TYPE_COMMENT = 8
# Document node type
TYPE_DOCUMENT = 9
# Document Type node type
TYPE_DOCUMENT_TYPE = 10
# Document Fragment node type
TYPE_DOCUMENT_FRAGMENT = 11
# Notation node type
TYPE_NOTATION = 12
# Whitespace node type
TYPE_WHITESPACE = 13
# Significant Whitespace node type
TYPE_SIGNIFICANT_WHITESPACE = 14
# Element end node type
TYPE_END_ELEMENT = 15
# Entity end node type
TYPE_END_ENTITY = 16
# \XML Declaration node type
TYPE_XML_DECLARATION = 17
# A list of errors encountered while parsing
attr_accessor :errors
# The \XML source
attr_reader :source
alias_method :self_closing?, :empty_element?
# :call-seq:
# Reader.new(input) { |options| ... } → Reader
# Reader.new(input, url:, encoding:, options:) { |options| ... } → Reader
#
# Create a new Reader to parse an \XML document.
#
# [Required Parameters]
# - +input+ (String | IO): The \XML document to parse.
#
# [Optional Parameters]
# - +url:+ (String) The base URL of the document.
# - +encoding:+ (String) The name of the encoding of the document.
# - +options:+ (Integer | ParseOptions) Options to control the parser behavior.
# Defaults to +ParseOptions::STRICT+.
#
# [Yields]
# If present, the block will be passed a Nokogiri::XML::ParseOptions object to modify before
# the fragment is parsed. See Nokogiri::XML::ParseOptions for more information.
def self.new(
string_or_io,
url_ = nil, encoding_ = nil, options_ = ParseOptions::STRICT,
url: url_, encoding: encoding_, options: options_
)
options = Nokogiri::XML::ParseOptions.new(options) if Integer === options
yield options if block_given?
if string_or_io.respond_to?(:read)
return Reader.from_io(string_or_io, url, encoding, options.to_i)
end
Reader.from_memory(string_or_io, url, encoding, options.to_i)
end
private def initialize(source, url = nil, encoding = nil) # :nodoc:
@source = source
@errors = []
@encoding = encoding
end
# Get the attributes and namespaces of the current node as a Hash.
#
# This is the union of Reader#attribute_hash and Reader#namespaces
#
# [Returns]
# (Hash<String, String>) Attribute names and values, and namespace prefixes and hrefs.
def attributes
attribute_hash.merge(namespaces)
end
###
# Move the cursor through the document yielding the cursor to the block
def each
while (cursor = read)
yield cursor
end
end
end
end
end
@@ -0,0 +1,75 @@
# frozen_string_literal: true
module Nokogiri
module XML
class << self
# :call-seq:
# RelaxNG(input) → Nokogiri::XML::RelaxNG
# RelaxNG(input, options:) → Nokogiri::XML::RelaxNG
#
# Convenience method for Nokogiri::XML::RelaxNG.new
def RelaxNG(...)
RelaxNG.new(...)
end
end
# Nokogiri::XML::RelaxNG is used for validating \XML against a RELAX NG schema definition.
#
# 🛡 <b>Do not use this class for untrusted schema documents.</b> RELAX NG input is always
# treated as *trusted*, meaning that the underlying parsing libraries <b>will access network
# resources</b>. This is counter to Nokogiri's "untrusted by default" security policy, but is an
# unfortunate limitation of the underlying libraries.
#
# *Example:* Determine whether an \XML document is valid.
#
# schema = Nokogiri::XML::RelaxNG.new(File.read(RELAX_NG_FILE))
# doc = Nokogiri::XML::Document.parse(File.read(XML_FILE))
# schema.valid?(doc) # Boolean
#
# *Example:* Validate an \XML document against a \RelaxNG schema, and capture any errors that are found.
#
# schema = Nokogiri::XML::RelaxNG.new(File.open(RELAX_NG_FILE))
# doc = Nokogiri::XML::Document.parse(File.open(XML_FILE))
# errors = schema.validate(doc) # Array<SyntaxError>
#
# *Example:* Validate an \XML document using a Document containing a RELAX NG schema definition.
#
# schema_doc = Nokogiri::XML::Document.parse(File.read(RELAX_NG_FILE))
# schema = Nokogiri::XML::RelaxNG.from_document(schema_doc)
# doc = Nokogiri::XML::Document.parse(File.open(XML_FILE))
# schema.valid?(doc) # Boolean
#
class RelaxNG < Nokogiri::XML::Schema
# :call-seq:
# new(input) → Nokogiri::XML::RelaxNG
# new(input, options:) → Nokogiri::XML::RelaxNG
#
# Parse a RELAX NG schema definition from a String or IO to create a new Nokogiri::XML::RelaxNG.
#
# [Parameters]
# - +input+ (String | IO) RELAX NG schema definition
# - +options:+ (Nokogiri::XML::ParseOptions)
# Defaults to Nokogiri::XML::ParseOptions::DEFAULT_SCHEMA ⚠ Unused
#
# [Returns] Nokogiri::XML::RelaxNG
#
# ⚠ +parse_options+ is currently unused by this method and is present only as a placeholder for
# future functionality.
#
# Also see convenience method Nokogiri::XML::RelaxNG()
def self.new(input, parse_options_ = ParseOptions::DEFAULT_SCHEMA, options: parse_options_)
from_document(Nokogiri::XML::Document.parse(input), options)
end
# :call-seq:
# read_memory(input) → Nokogiri::XML::RelaxNG
# read_memory(input, options:) → Nokogiri::XML::RelaxNG
#
# Convenience method for Nokogiri::XML::RelaxNG.new.
def self.read_memory(...)
# TODO deprecate this method
new(...)
end
end
end
end
@@ -0,0 +1,54 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# SAX Parsers are event-driven parsers.
#
# Two SAX parsers for XML are available, a parser that reads from a string or IO object as it
# feels necessary, and a parser that you explicitly feed XML in chunks. If you want to let
# Nokogiri deal with reading your XML, use the Nokogiri::XML::SAX::Parser. If you want to have
# fine grain control over the XML input, use the Nokogiri::XML::SAX::PushParser.
#
# If you want to do SAX style parsing of HTML, check out Nokogiri::HTML4::SAX.
#
# The basic way a SAX style parser works is by creating a parser, telling the parser about the
# events we're interested in, then giving the parser some XML to process. The parser will notify
# you when it encounters events you said you would like to know about.
#
# To register for events, subclass Nokogiri::XML::SAX::Document and implement the methods for
# which you would like notification.
#
# For example, if I want to be notified when a document ends, and when an element starts, I
# would write a class like this:
#
# class MyHandler < Nokogiri::XML::SAX::Document
# def end_document
# puts "the document has ended"
# end
#
# def start_element name, attributes = []
# puts "#{name} started"
# end
# end
#
# Then I would instantiate a SAX parser with this document, and feed the parser some XML
#
# # Create a new parser
# parser = Nokogiri::XML::SAX::Parser.new(MyHandler.new)
#
# # Feed the parser some XML
# parser.parse(File.open(ARGV[0]))
#
# Now my document handler will be called when each node starts, and when then document ends. To
# see what kinds of events are available, take a look at Nokogiri::XML::SAX::Document.
#
module SAX
end
end
end
require_relative "sax/document"
require_relative "sax/parser_context"
require_relative "sax/parser"
require_relative "sax/push_parser"
@@ -0,0 +1,258 @@
# frozen_string_literal: true
module Nokogiri
module XML
module SAX
# :markup: markdown
#
# The SAX::Document class is used for registering types of events you are interested in
# handling. All of the methods on this class are available as possible events while parsing an
# \XML document. To register for any particular event, subclass this class and implement the
# methods you are interested in knowing about.
#
# To only be notified about start and end element events, write a class like this:
#
# class MyHandler < Nokogiri::XML::SAX::Document
# def start_element name, attrs = []
# puts "#{name} started!"
# end
#
# def end_element name
# puts "#{name} ended"
# end
# end
#
# You can use this event handler for any SAX-style parser included with Nokogiri.
#
# See also:
#
# - Nokogiri::XML::SAX
# - Nokogiri::HTML4::SAX
#
# ### Entity Handling
#
# ⚠ Entity handling is complicated in a SAX parser! Please read this section carefully if
# you're not getting the behavior you expect.
#
# Entities will be reported to the user via callbacks to #characters, to #reference, or
# possibly to both. The behavior is determined by a combination of _entity type_ and the value
# of ParserContext#replace_entities. (Recall that the default value of
# ParserContext#replace_entities is `false`.)
#
# ⚠ <b>It is UNSAFE to set ParserContext#replace_entities to `true`</b> when parsing untrusted
# documents.
#
# 💡 For more information on entity types, see [Wikipedia's page on
# DTDs](https://en.wikipedia.org/wiki/Document_type_definition#Entity_declarations).
#
# | Entity type | #characters | #reference |
# |--------------------------------------|------------------------------------|-------------------------------------|
# | Char ref (e.g., <tt>&#146;</tt>) | always | never |
# | Predefined (e.g., <tt>&amp;</tt>) | always | never |
# | Undeclared † | never | <tt>#replace_entities == false</tt> |
# | Internal | always | <tt>#replace_entities == false</tt> |
# | External † | <tt>#replace_entities == true</tt> | <tt>#replace_entities == false</tt> |
#
# &nbsp;
#
# † In the case where the replacement text for the entity is unknown (e.g., an undeclared entity
# or an external entity that could not be resolved because of network issues), then the
# replacement text will not be reported. If ParserContext#replace_entities is `true`, this
# means the #characters callback will not be invoked. If ParserContext#replace_entities is
# `false`, then the #reference callback will be invoked, but with `nil` for the `content`
# argument.
#
class Document
###
# Called when an \XML declaration is parsed.
#
# [Parameters]
# - +version+ (String) the version attribute
# - +encoding+ (String, nil) the encoding of the document if present, else +nil+
# - +standalone+ ("yes", "no", nil) the standalone attribute if present, else +nil+
def xmldecl(version, encoding, standalone)
end
###
# Called when document starts parsing.
def start_document
end
###
# Called when document ends parsing.
def end_document
end
###
# Called at the beginning of an element.
#
# [Parameters]
# - +name+ (String) the name of the element
# - +attrs+ (Array<Array<String>>) an assoc list of namespace declarations and attributes, e.g.:
# [ ["xmlns:foo", "http://sample.net"], ["size", "large"] ]
#
# 💡If you're dealing with XML and need to handle namespaces, use the
# #start_element_namespace method instead.
#
# Note that the element namespace and any attribute namespaces are not provided, and so any
# namespaced elements or attributes will be returned as strings including the prefix:
#
# parser.parse(<<~XML)
# <root xmlns:foo='http://foo.example.com/' xmlns='http://example.com/'>
# <foo:bar foo:quux="xxx">hello world</foo:bar>
# </root>
# XML
#
# assert_pattern do
# parser.document.start_elements => [
# ["root", [["xmlns:foo", "http://foo.example.com/"], ["xmlns", "http://example.com/"]]],
# ["foo:bar", [["foo:quux", "xxx"]]],
# ]
# end
#
def start_element(name, attrs = [])
end
###
# Called at the end of an element.
#
# [Parameters]
# - +name+ (String) the name of the element being closed
#
def end_element(name)
end
###
# Called at the beginning of an element.
#
# [Parameters]
# - +name+ (String) is the name of the element
# - +attrs+ (Array<Attribute>) is an array of structs with the following properties:
# - +localname+ (String) the local name of the attribute
# - +value+ (String) the value of the attribute
# - +prefix+ (String, nil) the namespace prefix of the attribute
# - +uri+ (String, nil) the namespace URI of the attribute
# - +prefix+ (String, nil) is the namespace prefix for the element
# - +uri+ (String, nil) is the associated URI for the element's namespace
# - +ns+ (Array<Array<String, String>>) is an assoc list of namespace declarations on the element
#
# 💡If you're dealing with HTML or don't care about namespaces, try #start_element instead.
#
# [Example]
# it "start_elements_namespace is called with namespaced attributes" do
# parser.parse(<<~XML)
# <root xmlns:foo='http://foo.example.com/'>
# <foo:a foo:bar='hello' />
# </root>
# XML
#
# assert_pattern do
# parser.document.start_elements_namespace => [
# [
# "root",
# [],
# nil, nil,
# [["foo", "http://foo.example.com/"]], # namespace declarations
# ], [
# "a",
# [Nokogiri::XML::SAX::Parser::Attribute(localname: "bar", prefix: "foo", uri: "http://foo.example.com/", value: "hello")], # prefixed attribute
# "foo", "http://foo.example.com/", # prefix and uri for the "a" element
# [],
# ]
# ]
# end
# end
#
def start_element_namespace(name, attrs = [], prefix = nil, uri = nil, ns = []) # rubocop:disable Metrics/ParameterLists
# Deal with SAX v1 interface
name = [prefix, name].compact.join(":")
attributes = ns.map do |ns_prefix, ns_uri|
[["xmlns", ns_prefix].compact.join(":"), ns_uri]
end + attrs.map do |attr|
[[attr.prefix, attr.localname].compact.join(":"), attr.value]
end
start_element(name, attributes)
end
###
# Called at the end of an element.
#
# [Parameters]
# - +name+ (String) is the name of the element
# - +prefix+ (String, nil) is the namespace prefix for the element
# - +uri+ (String, nil) is the associated URI for the element's namespace
#
def end_element_namespace(name, prefix = nil, uri = nil)
# Deal with SAX v1 interface
end_element([prefix, name].compact.join(":"))
end
###
# Called when character data is parsed, and for parsed entities when
# ParserContext#replace_entities is +true+.
#
# [Parameters]
# - +string+ contains the character data or entity replacement text
#
# ⚠ Please see Document@Entity+Handling for important information about how entities are handled.
#
# ⚠ This method might be called multiple times for a contiguous string of characters.
#
def characters(string)
end
###
# Called when a parsed entity is referenced and not replaced.
#
# [Parameters]
# - +name+ (String) is the name of the entity
# - +content+ (String, nil) is the replacement text for the entity, if known
#
# ⚠ Please see Document@Entity+Handling for important information about how entities are handled.
#
# ⚠ An internal entity may result in a call to both #characters and #reference.
#
# Since v1.17.0
#
def reference(name, content)
end
###
# Called when comments are encountered
# [Parameters]
# - +string+ contains the comment data
def comment(string)
end
###
# Called on document warnings
# [Parameters]
# - +string+ contains the warning
def warning(string)
end
###
# Called on document errors
# [Parameters]
# - +string+ contains the error
def error(string)
end
###
# Called when cdata blocks are found
# [Parameters]
# - +string+ contains the cdata content
def cdata_block(string)
end
###
# Called when processing instructions are found
# [Parameters]
# - +name+ is the target of the instruction
# - +content+ is the value of the instruction
def processing_instruction(name, content)
end
end
end
end
end
@@ -0,0 +1,199 @@
# frozen_string_literal: true
module Nokogiri
module XML
module SAX
###
# This parser is a SAX style parser that reads its input as it deems necessary. The parser
# takes a Nokogiri::XML::SAX::Document, an optional encoding, then given an XML input, sends
# messages to the Nokogiri::XML::SAX::Document.
#
# Here is an example of using this parser:
#
# # Create a subclass of Nokogiri::XML::SAX::Document and implement
# # the events we care about:
# class MyHandler < Nokogiri::XML::SAX::Document
# def start_element name, attrs = []
# puts "starting: #{name}"
# end
#
# def end_element name
# puts "ending: #{name}"
# end
# end
#
# parser = Nokogiri::XML::SAX::Parser.new(MyHandler.new)
#
# # Hand an IO object to the parser, which will read the XML from the IO.
# File.open(path_to_xml) do |f|
# parser.parse(f)
# end
#
# For more information about \SAX parsers, see Nokogiri::XML::SAX.
#
# Also see Nokogiri::XML::SAX::Document for the available events.
#
# For \HTML documents, use the subclass Nokogiri::HTML4::SAX::Parser.
#
class Parser
# to dynamically resolve ParserContext in inherited methods
include Nokogiri::ClassResolver
# Structure used for marshalling attributes for some callbacks in XML::SAX::Document.
class Attribute < Struct.new(:localname, :prefix, :uri, :value)
end
ENCODINGS = { # :nodoc:
"NONE" => 0, # No char encoding detected
"UTF-8" => 1, # UTF-8
"UTF16LE" => 2, # UTF-16 little endian
"UTF16BE" => 3, # UTF-16 big endian
"UCS4LE" => 4, # UCS-4 little endian
"UCS4BE" => 5, # UCS-4 big endian
"EBCDIC" => 6, # EBCDIC uh!
"UCS4-2143" => 7, # UCS-4 unusual ordering
"UCS4-3412" => 8, # UCS-4 unusual ordering
"UCS2" => 9, # UCS-2
"ISO-8859-1" => 10, # ISO-8859-1 ISO Latin 1
"ISO-8859-2" => 11, # ISO-8859-2 ISO Latin 2
"ISO-8859-3" => 12, # ISO-8859-3
"ISO-8859-4" => 13, # ISO-8859-4
"ISO-8859-5" => 14, # ISO-8859-5
"ISO-8859-6" => 15, # ISO-8859-6
"ISO-8859-7" => 16, # ISO-8859-7
"ISO-8859-8" => 17, # ISO-8859-8
"ISO-8859-9" => 18, # ISO-8859-9
"ISO-2022-JP" => 19, # ISO-2022-JP
"SHIFT-JIS" => 20, # Shift_JIS
"EUC-JP" => 21, # EUC-JP
"ASCII" => 22, # pure ASCII
}
REVERSE_ENCODINGS = ENCODINGS.invert # :nodoc:
deprecate_constant :ENCODINGS
# The Nokogiri::XML::SAX::Document where events will be sent.
attr_accessor :document
# The encoding beings used for this document.
attr_accessor :encoding
###
# :call-seq:
# new ⇒ SAX::Parser
# new(handler) ⇒ SAX::Parser
# new(handler, encoding) ⇒ SAX::Parser
#
# Create a new Parser.
#
# [Parameters]
# - +handler+ (optional Nokogiri::XML::SAX::Document) The document that will receive
# events. Will create a new Nokogiri::XML::SAX::Document if not given, which is accessible
# through the #document attribute.
# - +encoding+ (optional Encoding, String, nil) An Encoding or encoding name to use when
# parsing the input. (default +nil+ for auto-detection)
#
def initialize(doc = Nokogiri::XML::SAX::Document.new, encoding = nil)
@encoding = encoding
@document = doc
@warned = false
initialize_native unless Nokogiri.jruby?
end
###
# :call-seq:
# parse(input) { |parser_context| ... }
#
# Parse the input, sending events to the SAX::Document at #document.
#
# [Parameters]
# - +input+ (String, IO) The input to parse.
#
# If +input+ quacks like a readable IO object, this method forwards to Parser.parse_io,
# otherwise it forwards to Parser.parse_memory.
#
# [Yields]
# If a block is given, the underlying ParserContext object will be yielded. This can be used
# to set options on the parser context before parsing begins.
#
def parse(input, &block)
if input.respond_to?(:read) && input.respond_to?(:close)
parse_io(input, &block)
else
parse_memory(input, &block)
end
end
###
# :call-seq:
# parse_io(io) { |parser_context| ... }
# parse_io(io, encoding) { |parser_context| ... }
#
# Parse an input stream.
#
# [Parameters]
# - +io+ (IO) The readable IO object from which to read input
# - +encoding+ (optional Encoding, String, nil) An Encoding or encoding name to use when
# parsing the input, or +nil+ for auto-detection. (default #encoding)
#
# [Yields]
# If a block is given, the underlying ParserContext object will be yielded. This can be used
# to set options on the parser context before parsing begins.
#
def parse_io(io, encoding = @encoding)
ctx = related_class("ParserContext").io(io, encoding)
yield ctx if block_given?
ctx.parse_with(self)
end
###
# :call-seq:
# parse_memory(input) { |parser_context| ... }
# parse_memory(input, encoding) { |parser_context| ... }
#
# Parse an input string.
#
# [Parameters]
# - +input+ (String) The input string to be parsed.
# - +encoding+ (optional Encoding, String, nil) An Encoding or encoding name to use when
# parsing the input, or +nil+ for auto-detection. (default #encoding)
#
# [Yields]
# If a block is given, the underlying ParserContext object will be yielded. This can be used
# to set options on the parser context before parsing begins.
#
def parse_memory(input, encoding = @encoding)
ctx = related_class("ParserContext").memory(input, encoding)
yield ctx if block_given?
ctx.parse_with(self)
end
###
# :call-seq:
# parse_file(filename) { |parser_context| ... }
# parse_file(filename, encoding) { |parser_context| ... }
#
# Parse a file.
#
# [Parameters]
# - +filename+ (String) The path to the file to be parsed.
# - +encoding+ (optional Encoding, String, nil) An Encoding or encoding name to use when
# parsing the input, or +nil+ for auto-detection. (default #encoding)
#
# [Yields]
# If a block is given, the underlying ParserContext object will be yielded. This can be used
# to set options on the parser context before parsing begins.
#
def parse_file(filename, encoding = @encoding)
raise ArgumentError, "no filename provided" unless filename
raise Errno::ENOENT unless File.exist?(filename)
raise Errno::EISDIR if File.directory?(filename)
ctx = related_class("ParserContext").file(filename, encoding)
yield ctx if block_given?
ctx.parse_with(self)
end
end
end
end
end
@@ -0,0 +1,129 @@
# frozen_string_literal: true
module Nokogiri
module XML
module SAX
###
# Context object to invoke the XML SAX parser on the SAX::Document handler.
#
# 💡 This class is usually not instantiated by the user. Use Nokogiri::XML::SAX::Parser
# instead.
class ParserContext
class << self
###
# :call-seq:
# new(input)
# new(input, encoding)
#
# Create a parser context for an IO or a String. This is a shorthand method for
# ParserContext.io and ParserContext.memory.
#
# [Parameters]
# - +input+ (IO, String) A String or a readable IO object
# - +encoding+ (optional) (Encoding) The +Encoding+ to use, or the name of an
# encoding to use (default +nil+, encoding will be autodetected)
#
# If +input+ quacks like a readable IO object, this method forwards to ParserContext.io,
# otherwise it forwards to ParserContext.memory.
#
# [Returns] Nokogiri::XML::SAX::ParserContext
#
def new(input, encoding = nil)
if [:read, :close].all? { |x| input.respond_to?(x) }
io(input, encoding)
else
memory(input, encoding)
end
end
###
# :call-seq:
# io(input)
# io(input, encoding)
#
# Create a parser context for an +input+ IO which will assume +encoding+
#
# [Parameters]
# - +io+ (IO) The readable IO object from which to read input
# - +encoding+ (optional) (Encoding) The +Encoding+ to use, or the name of an
# encoding to use (default +nil+, encoding will be autodetected)
#
# [Returns] Nokogiri::XML::SAX::ParserContext
#
# 💡 Calling this method directly is discouraged. Use Nokogiri::XML::SAX::Parser parse
# methods which are more convenient for most use cases.
#
def io(input, encoding = nil)
native_io(input, resolve_encoding(encoding))
end
###
# :call-seq:
# memory(input)
# memory(input, encoding)
#
# Create a parser context for the +input+ String.
#
# [Parameters]
# - +input+ (String) The input string to be parsed.
# - +encoding+ (optional) (Encoding, String) The +Encoding+ to use, or the name of an encoding to
# use (default +nil+, encoding will be autodetected)
#
# [Returns] Nokogiri::XML::SAX::ParserContext
#
# 💡 Calling this method directly is discouraged. Use Nokogiri::XML::SAX::Parser parse methods
# which are more convenient for most use cases.
#
def memory(input, encoding = nil)
native_memory(input, resolve_encoding(encoding))
end
###
# :call-seq:
# file(path)
# file(path, encoding)
#
# Create a parser context for the file at +path+.
#
# [Parameters]
# - +path+ (String) The path to the input file
# - +encoding+ (optional) (Encoding, String) The +Encoding+ to use, or the name of an encoding to
# use (default +nil+, encoding will be autodetected)
#
# [Returns] Nokogiri::XML::SAX::ParserContext
#
# 💡 Calling this method directly is discouraged. Use Nokogiri::XML::SAX::Parser.parse_file which
# is more convenient for most use cases.
def file(input, encoding = nil)
native_file(input, resolve_encoding(encoding))
end
private def resolve_encoding(encoding)
case encoding
when Encoding
encoding
when nil
nil # totally fine, parser will guess encoding
when Integer
warn("Passing an integer to Nokogiri::XML::SAX::ParserContext.io is deprecated. Use an Encoding object instead. This will become an error in a future release.", uplevel: 2, category: :deprecated)
return nil if encoding == Parser::ENCODINGS["NONE"]
encoding = Parser::REVERSE_ENCODINGS[encoding]
raise ArgumentError, "Invalid libxml2 encoding id #{encoding}" if encoding.nil?
Encoding.find(encoding)
when String
Encoding.find(encoding)
else
raise ArgumentError, "Cannot resolve #{encoding.inspect} to an Encoding"
end
end
end
end
end
end
end
@@ -0,0 +1,64 @@
# frozen_string_literal: true
module Nokogiri
module XML
module SAX
###
# PushParser can parse a document that is fed to it manually. It
# must be given a SAX::Document object which will be called with
# SAX events as the document is being parsed.
#
# Calling PushParser#<< writes XML to the parser, calling any SAX
# callbacks it can.
#
# PushParser#finish tells the parser that the document is finished
# and calls the end_document SAX method.
#
# Example:
#
# parser = PushParser.new(Class.new(XML::SAX::Document) {
# def start_document
# puts "start document called"
# end
# }.new)
# parser << "<div>hello<"
# parser << "/div>"
# parser.finish
class PushParser
# The Nokogiri::XML::SAX::Document on which the PushParser will be
# operating
attr_accessor :document
###
# Create a new PushParser with +doc+ as the SAX Document, providing
# an optional +file_name+ and +encoding+
def initialize(doc = XML::SAX::Document.new, file_name = nil, encoding = "UTF-8")
@document = doc
@encoding = encoding
@sax_parser = XML::SAX::Parser.new(doc)
## Create our push parser context
initialize_native(@sax_parser, file_name)
end
###
# Write a +chunk+ of XML to the PushParser. Any callback methods
# that can be called will be called immediately.
def write(chunk, last_chunk = false)
native_write(chunk, last_chunk)
end
alias_method :<<, :write
###
# Finish the parsing. This method is only necessary for
# Nokogiri::XML::SAX::Document#end_document to be called.
#
# ⚠ Note that empty documents are treated as an error when using the libxml2-based
# implementation (CRuby), but are fine when using the Xerces-based implementation (JRuby).
def finish
write("", true)
end
end
end
end
end
@@ -0,0 +1,140 @@
# frozen_string_literal: true
module Nokogiri
module XML
class << self
# :call-seq:
# Schema(input) → Nokogiri::XML::Schema
# Schema(input, parse_options) → Nokogiri::XML::Schema
#
# Convenience method for Nokogiri::XML::Schema.new
def Schema(...)
Schema.new(...)
end
end
# Nokogiri::XML::Schema is used for validating \XML against an \XSD schema definition.
#
# ⚠ Since v1.11.0, Schema treats inputs as *untrusted* by default, and so external entities are
# not resolved from the network (+http://+ or +ftp://+). When parsing a trusted document, the
# caller may turn off the +NONET+ option via the ParseOptions to (re-)enable external entity
# resolution over a network connection.
#
# 🛡 Before v1.11.0, documents were "trusted" by default during schema parsing which was counter
# to Nokogiri's "untrusted by default" security policy.
#
# *Example:* Determine whether an \XML document is valid.
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# doc = Nokogiri::XML::Document.parse(File.read(XML_FILE))
# schema.valid?(doc) # Boolean
#
# *Example:* Validate an \XML document against an \XSD schema, and capture any errors that are found.
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# doc = Nokogiri::XML::Document.parse(File.read(XML_FILE))
# errors = schema.validate(doc) # Array<SyntaxError>
#
# *Example:* Validate an \XML document using a Document containing an \XSD schema definition.
#
# schema_doc = Nokogiri::XML::Document.parse(File.read(RELAX_NG_FILE))
# schema = Nokogiri::XML::Schema.from_document(schema_doc)
# doc = Nokogiri::XML::Document.parse(File.read(XML_FILE))
# schema.valid?(doc) # Boolean
#
class Schema
# The errors found while parsing the \XSD
#
# [Returns] Array<Nokogiri::XML::SyntaxError>
attr_accessor :errors
# The options used to parse the schema
#
# [Returns] Nokogiri::XML::ParseOptions
attr_accessor :parse_options
# :call-seq:
# new(input) → Nokogiri::XML::Schema
# new(input, parse_options) → Nokogiri::XML::Schema
#
# Parse an \XSD schema definition from a String or IO to create a new Nokogiri::XML::Schema
#
# [Parameters]
# - +input+ (String | IO) \XSD schema definition
# - +parse_options+ (Nokogiri::XML::ParseOptions)
# Defaults to Nokogiri::XML::ParseOptions::DEFAULT_SCHEMA
#
# [Returns] Nokogiri::XML::Schema
#
def self.new(input, parse_options_ = ParseOptions::DEFAULT_SCHEMA, parse_options: parse_options_)
from_document(Nokogiri::XML::Document.parse(input), parse_options)
end
# :call-seq:
# read_memory(input) → Nokogiri::XML::Schema
# read_memory(input, parse_options) → Nokogiri::XML::Schema
#
# Convenience method for Nokogiri::XML::Schema.new
def self.read_memory(...)
# TODO deprecate this method
new(...)
end
#
# :call-seq: validate(input) → Array<SyntaxError>
#
# Validate +input+ and return any errors that are found.
#
# [Parameters]
# - +input+ (Nokogiri::XML::Document | String)
# A parsed document, or a string containing a local filename.
#
# [Returns] Array<SyntaxError>
#
# *Example:* Validate an existing XML::Document, and capture any errors that are found.
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# errors = schema.validate(document)
#
# *Example:* Validate an \XML document on disk, and capture any errors that are found.
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# errors = schema.validate("/path/to/file.xml")
#
def validate(input)
if input.is_a?(Nokogiri::XML::Document)
validate_document(input)
elsif File.file?(input)
validate_file(input)
else
raise ArgumentError, "Must provide Nokogiri::XML::Document or the name of an existing file"
end
end
#
# :call-seq: valid?(input) → Boolean
#
# Validate +input+ and return a Boolean indicating whether the document is valid
#
# [Parameters]
# - +input+ (Nokogiri::XML::Document | String)
# A parsed document, or a string containing a local filename.
#
# [Returns] Boolean
#
# *Example:* Validate an existing XML::Document
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# return unless schema.valid?(document)
#
# *Example:* Validate an \XML document on disk
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# return unless schema.valid?("/path/to/file.xml")
#
def valid?(input)
validate(input).empty?
end
end
end
end
@@ -0,0 +1,274 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
#
# The Searchable module declares the interface used for searching your DOM.
#
# It implements the public methods #search, #css, and #xpath,
# as well as allowing specific implementations to specialize some
# of the important behaviors.
#
module Searchable
# Regular expression used by Searchable#search to determine if a query
# string is CSS or XPath
LOOKS_LIKE_XPATH = %r{^(\./|/|\.\.|\.$)}
# :section: Searching via XPath or CSS Queries
###
# call-seq:
# search(*paths, [namespace-bindings, xpath-variable-bindings, custom-handler-class])
#
# Search this object for +paths+. +paths+ must be one or more XPath or CSS queries:
#
# node.search("div.employee", ".//title")
#
# A hash of namespace bindings may be appended:
#
# node.search('.//bike:tire', {'bike' => 'http://schwinn.com/'})
# node.search('bike|tire', {'bike' => 'http://schwinn.com/'})
#
# For XPath queries, a hash of variable bindings may also be appended to the namespace
# bindings. For example:
#
# node.search('.//address[@domestic=$value]', nil, {:value => 'Yes'})
#
# 💡 Custom XPath functions and CSS pseudo-selectors may also be defined. To define custom
# functions create a class and implement the function you want to define, which will be in the
# `nokogiri` namespace in XPath queries.
#
# The first argument to the method will be the current matching NodeSet. Any other arguments
# are ones that you pass in. Note that this class may appear anywhere in the argument
# list. For example:
#
# handler = Class.new {
# def regex node_set, regex
# node_set.find_all { |node| node['some_attribute'] =~ /#{regex}/ }
# end
# }.new
# node.search('.//title[nokogiri:regex(., "\w+")]', 'div.employee:regex("[0-9]+")', handler)
#
# See Searchable#xpath and Searchable#css for further usage help.
def search(*args)
paths, handler, ns, binds = extract_params(args)
xpaths = paths.map(&:to_s).map do |path|
LOOKS_LIKE_XPATH.match?(path) ? path : xpath_query_from_css_rule(path, ns)
end.flatten.uniq
xpath(*(xpaths + [ns, handler, binds].compact))
end
alias_method :/, :search
###
# call-seq:
# at(*paths, [namespace-bindings, xpath-variable-bindings, custom-handler-class])
#
# Search this object for +paths+, and return only the first
# result. +paths+ must be one or more XPath or CSS queries.
#
# See Searchable#search for more information.
def at(*args)
search(*args).first
end
alias_method :%, :at
###
# call-seq:
# css(*rules, [namespace-bindings, custom-pseudo-class])
#
# Search this object for CSS +rules+. +rules+ must be one or more CSS
# selectors. For example:
#
# node.css('title')
# node.css('body h1.bold')
# node.css('div + p.green', 'div#one')
#
# A hash of namespace bindings may be appended. For example:
#
# node.css('bike|tire', {'bike' => 'http://schwinn.com/'})
#
# 💡 Custom CSS pseudo classes may also be defined which are mapped to a custom XPath
# function. To define custom pseudo classes, create a class and implement the custom pseudo
# class you want defined. The first argument to the method will be the matching context
# NodeSet. Any other arguments are ones that you pass in. For example:
#
# handler = Class.new {
# def regex(node_set, regex)
# node_set.find_all { |node| node['some_attribute'] =~ /#{regex}/ }
# end
# }.new
# node.css('title:regex("\w+")', handler)
#
# 💡 Some XPath syntax is supported in CSS queries. For example, to query for an attribute:
#
# node.css('img > @href') # returns all +href+ attributes on an +img+ element
# node.css('img / @href') # same
#
# # ⚠ this returns +class+ attributes from all +div+ elements AND THEIR CHILDREN!
# node.css('div @class')
#
# node.css
#
# 💡 Array-like syntax is supported in CSS queries as an alternative to using +:nth-child()+.
#
# ⚠ NOTE that indices are 1-based like +:nth-child+ and not 0-based like Ruby Arrays. For
# example:
#
# # equivalent to 'li:nth-child(2)'
# node.css('li[2]') # retrieve the second li element in a list
#
# ⚠ NOTE that the CSS query string is case-sensitive with regards to your document type. HTML
# tags will match only lowercase CSS queries, so if you search for "H1" in an HTML document,
# you'll never find anything. However, "H1" might be found in an XML document, where tags
# names are case-sensitive (e.g., "H1" is distinct from "h1").
def css(*args)
rules, handler, ns, _ = extract_params(args)
css_internal(self, rules, handler, ns)
end
##
# call-seq:
# at_css(*rules, [namespace-bindings, custom-pseudo-class])
#
# Search this object for CSS +rules+, and return only the first
# match. +rules+ must be one or more CSS selectors.
#
# See Searchable#css for more information.
def at_css(*args)
css(*args).first
end
###
# call-seq:
# xpath(*paths, [namespace-bindings, variable-bindings, custom-handler-class])
#
# Search this node for XPath +paths+. +paths+ must be one or more XPath
# queries.
#
# node.xpath('.//title')
#
# A hash of namespace bindings may be appended. For example:
#
# node.xpath('.//foo:name', {'foo' => 'http://example.org/'})
# node.xpath('.//xmlns:name', node.root.namespaces)
#
# A hash of variable bindings may also be appended to the namespace bindings. For example:
#
# node.xpath('.//address[@domestic=$value]', nil, {:value => 'Yes'})
#
# 💡 Custom XPath functions may also be defined. To define custom functions create a class and
# implement the function you want to define, which will be in the `nokogiri` namespace.
#
# The first argument to the method will be the current matching NodeSet. Any other arguments
# are ones that you pass in. Note that this class may appear anywhere in the argument
# list. For example:
#
# handler = Class.new {
# def regex(node_set, regex)
# node_set.find_all { |node| node['some_attribute'] =~ /#{regex}/ }
# end
# }.new
# node.xpath('.//title[nokogiri:regex(., "\w+")]', handler)
#
def xpath(*args)
paths, handler, ns, binds = extract_params(args)
xpath_internal(self, paths, handler, ns, binds)
end
##
# call-seq:
# at_xpath(*paths, [namespace-bindings, variable-bindings, custom-handler-class])
#
# Search this node for XPath +paths+, and return only the first
# match. +paths+ must be one or more XPath queries.
#
# See Searchable#xpath for more information.
def at_xpath(*args)
xpath(*args).first
end
# :call-seq:
# >(selector) → NodeSet
#
# Search this node's immediate children using CSS selector +selector+
def >(selector) # rubocop:disable Naming/BinaryOperatorParameterName
ns = document.root&.namespaces || {}
xpath(CSS.xpath_for(selector, prefix: "./", ns: ns).first)
end
# :section:
private
def extract_params(params) # :nodoc:
handler = params.find do |param|
![Hash, String, Symbol].include?(param.class)
end
params -= [handler] if handler
hashes = []
while Hash === params.last || params.last.nil?
hashes << params.pop
break if params.empty?
end
ns, binds = hashes.reverse
ns ||= document.root&.namespaces || {}
[params, handler, ns, binds]
end
def css_internal(node, rules, handler, ns)
xpath_internal(node, css_rules_to_xpath(rules, ns), handler, ns, nil)
end
def css_rules_to_xpath(rules, ns)
rules.map { |rule| xpath_query_from_css_rule(rule, ns) }
end
def xpath_query_from_css_rule(rule, ns)
self.class::IMPLIED_XPATH_CONTEXTS.map do |implied_xpath_context|
visitor = Nokogiri::CSS::XPathVisitor.new(
builtins: Nokogiri::CSS::XPathVisitor::BuiltinsConfig::OPTIMAL,
doctype: document.xpath_doctype,
prefix: implied_xpath_context,
namespaces: ns,
)
CSS.xpath_for(rule.to_s, visitor: visitor)
end.join(" | ")
end
def xpath_internal(node, paths, handler, ns, binds)
document = node.document
return NodeSet.new(document) unless document
if paths.length == 1
return xpath_impl(node, paths.first, handler, ns, binds)
end
NodeSet.new(document) do |combined|
paths.each do |path|
xpath_impl(node, path, handler, ns, binds).each { |set| combined << set }
end
end
end
def xpath_impl(node, path, handler, ns, binds)
context = XPathContext.new(node)
context.register_namespaces(ns)
context.register_variables(binds)
path = path.gsub("xmlns:", " :") unless Nokogiri.uses_libxml?
context.evaluate(path, handler)
end
end
end
end
@@ -0,0 +1,94 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# This class provides information about XML SyntaxErrors. These
# exceptions are typically stored on Nokogiri::XML::Document#errors.
class SyntaxError < ::Nokogiri::SyntaxError
class << self
def aggregate(errors)
return nil if errors.empty?
return errors.first if errors.length == 1
messages = ["Multiple errors encountered:"]
errors.each do |error|
messages << error.to_s
end
new(messages.join("\n"))
end
end
attr_reader :domain
attr_reader :code
attr_reader :level
attr_reader :file
attr_reader :line
# The XPath path of the node that caused the error when validating a `Nokogiri::XML::Document`.
#
# This attribute will only be non-nil when the error is emitted by `Schema#validate` on
# Document objects. It will return `nil` for DOM parsing errors and for errors emitted during
# Schema validation of files.
#
# ⚠ `#path` is not supported on JRuby, where it will always return `nil`.
attr_reader :path
attr_reader :str1
attr_reader :str2
attr_reader :str3
attr_reader :int1
attr_reader :column
###
# return true if this is a non error
def none?
level == 0
end
###
# return true if this is a warning
def warning?
level == 1
end
###
# return true if this is an error
def error?
level == 2
end
###
# return true if this error is fatal
def fatal?
level == 3
end
def to_s
message = super.chomp
[location_to_s, level_to_s, message]
.compact.join(": ")
.force_encoding(message.encoding)
end
private
def level_to_s
case level
when 3 then "FATAL"
when 2 then "ERROR"
when 1 then "WARNING"
end
end
def nil_or_zero?(attribute)
attribute.nil? || attribute.zero?
end
def location_to_s
return if nil_or_zero?(line) && nil_or_zero?(column)
"#{line}:#{column}"
end
end
end
end
@@ -0,0 +1,11 @@
# frozen_string_literal: true
module Nokogiri
module XML
class Text < Nokogiri::XML::CharacterData
def content=(string)
self.native_content = string.to_s
end
end
end
end
@@ -0,0 +1,21 @@
# frozen_string_literal: true
module Nokogiri
module XML
module XPath
# The XPath search prefix to search globally, +//+
GLOBAL_SEARCH_PREFIX = "//"
# The XPath search prefix to search direct descendants of the root element, +/+
ROOT_SEARCH_PREFIX = "/"
# The XPath search prefix to search direct descendants of the current element, +./+
CURRENT_SEARCH_PREFIX = "./"
# The XPath search prefix to search anywhere in the current element's subtree, +.//+
SUBTREE_SEARCH_PREFIX = ".//"
end
end
end
require_relative "xpath/syntax_error"
@@ -0,0 +1,13 @@
# frozen_string_literal: true
module Nokogiri
module XML
module XPath
class SyntaxError < XML::SyntaxError
def to_s
[super.chomp, str1].compact.join(": ")
end
end
end
end
end
@@ -0,0 +1,27 @@
# frozen_string_literal: true
module Nokogiri
module XML
class XPathContext
###
# Register namespaces in +namespaces+
def register_namespaces(namespaces)
namespaces.each do |key, value|
key = key.to_s.gsub(/.*:/, "") # strip off 'xmlns:' or 'xml:'
register_ns(key, value)
end
end
def register_variables(binds)
return if binds.nil?
binds.each do |key, value|
key = key.to_s
register_variable(key, value)
end
end
end
end
end