Add bin and edit workflow
Gitea Actions Demo / Explore-Gitea-Actions (push) Failing after 9s

This commit is contained in:
2026-09-16 13:11:16 -06:00
parent c8ac4fcae5
commit 4cee170d66
17576 changed files with 895740 additions and 2 deletions
@@ -0,0 +1,67 @@
# coding: utf-8
# frozen_string_literal: true
require "set"
module Nokogiri
#
# Some classes in Nokogiri are namespaced as a group, for example
# Document, DocumentFragment, and Builder.
#
# It's sometimes necessary to look up the related class, e.g.:
#
# XML::Builder → XML::Document
# HTML4::Builder → HTML4::Document
# HTML5::Document → HTML5::DocumentFragment
#
# This module is included into those key classes who need to do this.
#
module ClassResolver
# #related_class restricts matching namespaces to those matching this set.
VALID_NAMESPACES = Set.new(["HTML", "HTML4", "HTML5", "XML", "SAX"])
# :call-seq:
# related_class(class_name) → Class
#
# Find a class constant within the
#
# Some examples:
#
# Nokogiri::XML::Document.new.related_class("DocumentFragment")
# # => Nokogiri::XML::DocumentFragment
# Nokogiri::HTML4::Document.new.related_class("DocumentFragment")
# # => Nokogiri::HTML4::DocumentFragment
#
# Note this will also work for subclasses that follow the same convention, e.g.:
#
# Loofah::HTML::Document.new.related_class("DocumentFragment")
# # => Loofah::HTML::DocumentFragment
#
# And even if it's a subclass, this will iterate through the superclasses:
#
# class ThisIsATopLevelClass < Nokogiri::HTML4::Builder ; end
# ThisIsATopLevelClass.new.related_class("Document")
# # => Nokogiri::HTML4::Document
#
def related_class(class_name)
klass = nil
inspecting = self.class
while inspecting
namespace_path = inspecting.name.split("::")[0..-2]
inspecting = inspecting.superclass
next unless VALID_NAMESPACES.include?(namespace_path.last)
related_class_name = (namespace_path << class_name).join("::")
klass = begin
Object.const_get(related_class_name)
rescue NameError
nil
end
break if klass
end
klass
end
end
end
@@ -0,0 +1,132 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
# Translate a CSS selector into an XPath 1.0 query
module CSS
class << self
# TODO: Deprecate this method ahead of 2.0 and delete it in 2.0.
# It is not used by Nokogiri and shouldn't be part of the public API.
def parse(selector) # :nodoc:
warn("Nokogiri::CSS.parse is deprecated and will be removed in a future version of Nokogiri. Use Nokogiri::CSS::Parser#parse instead.", uplevel: 1, category: :deprecated)
Parser.new.parse(selector)
end
# :call-seq:
# xpath_for(selector_list) → Array<String>
# xpath_for(selector_list [, prefix:] [, ns:] [, visitor:] [, cache:]) → Array<String>
#
# Translate a CSS selector list to the equivalent XPath expressions.
#
# 💡 Note that translated queries are cached by default for performance concerns.
#
# ⚠ Users should prefer Nokogiri::XML::Searchable#css, which is mixed into all document and
# node classes, for querying documents with CSS selectors. This method is the underlying
# mechanism used by XML::Searchable and is provided solely for advanced users to translate
# \CSS selectors to XPath directly.
#
# Also see Nokogiri::XML::Searchable#css for documentation on supported CSS selector features,
# some extended syntax that Nokogiri supports, and advanced CSS features like pseudo-class
# functions.
#
# [Parameters]
# - +selector_list+ (String)
#
# The CSS selector to be translated into XPath. This is always a String, but that string
# value may be a {selector list}[https://www.w3.org/TR/selectors-4/#grouping] (see
# examples).
#
# [Keyword arguments]
# - +prefix:+ (String)
#
# The XPath expression prefix which determines the search context. See Nokogiri::XML::XPath
# for standard options. Default is +XPath::GLOBAL_SEARCH_PREFIX+.
#
# - +ns:+ (Hash<String ⇒ String>, nil)
#
# Namespaces that are referenced in the query, if any. This is a hash where the keys are the
# namespace prefix and the values are the namespace URIs. Default is +nil+ indicating an
# empty set of namespaces.
#
# - +visitor:+ (Nokogiri::CSS::XPathVisitor)
#
# Use this XPathVisitor object to transform the CSS AST into XPath expressions. See
# Nokogiri::CSS::XPathVisitor for more information on some of the complex behavior that can
# be customized for your document type. Default is +Nokogiri::CSS::XPathVisitor.new+.
#
# ⚠ Note that this option is mutually exclusive with +prefix+ and +ns+. If +visitor+ is
# provided, +prefix+ and +ns+ must not be present.
#
# - +cache:+ (Boolean)
#
# Whether to use the SelectorCache for the translated query to ensure that repeated queries
# don't incur the overhead of re-parsing the selector. Default is +true+.
#
# [Returns] (Array<String>) The equivalent set of XPath expressions for +selector_list+
#
# *Example* with a simple selector:
#
# Nokogiri::CSS.xpath_for("div") # => ["//div"]
#
# *Example* with a compound selector:
#
# Nokogiri::CSS.xpath_for("div.xl") # => ["//div[contains(concat(' ',normalize-space(@class),' '),' xl ')]"]
#
# *Example* with a complex selector:
#
# Nokogiri::CSS.xpath_for("h1 + div") # => ["//h1/following-sibling::*[1]/self::div"]
#
# *Example* with a selector list:
#
# Nokogiri::CSS.xpath_for("h1, h2, h3") # => ["//h1", "//h2", "//h3"]
#
def xpath_for(
selector, options = nil,
prefix: options&.delete(:prefix),
visitor: options&.delete(:visitor),
ns: options&.delete(:ns),
cache: true
)
unless options.nil?
warn("Nokogiri::CSS.xpath_for: Passing options as an explicit hash is deprecated. Use keyword arguments instead. This will become an error in a future release.", uplevel: 1, category: :deprecated)
end
raise(TypeError, "no implicit conversion of #{selector.inspect} to String") unless selector.respond_to?(:to_str)
selector = selector.to_str
raise(Nokogiri::CSS::SyntaxError, "empty CSS selector") if selector.empty?
if visitor
raise ArgumentError, "cannot provide both :prefix and :visitor" if prefix
raise ArgumentError, "cannot provide both :ns and :visitor" if ns
end
visitor ||= begin
visitor_kw = {}
visitor_kw[:prefix] = prefix if prefix
visitor_kw[:namespaces] = ns if ns
Nokogiri::CSS::XPathVisitor.new(**visitor_kw)
end
if cache
key = SelectorCache.key(selector: selector, visitor: visitor)
SelectorCache[key] ||= Parser.new.xpath_for(selector, visitor)
else
Parser.new.xpath_for(selector, visitor)
end
end
end
end
end
require_relative "css/selector_cache"
require_relative "css/node"
require_relative "css/xpath_visitor"
x = $-w
$-w = false
require_relative "css/parser"
$-w = x
require_relative "css/tokenizer"
require_relative "css/syntax_error"
@@ -0,0 +1,58 @@
# frozen_string_literal: true
module Nokogiri
module CSS
class Node # :nodoc:
ALLOW_COMBINATOR_ON_SELF = [:DIRECT_ADJACENT_SELECTOR, :FOLLOWING_SELECTOR, :CHILD_SELECTOR]
# Get the type of this node
attr_accessor :type
# Get the value of this node
attr_accessor :value
# Create a new Node with +type+ and +value+
def initialize(type, value)
@type = type
@value = value
end
# Accept +visitor+
def accept(visitor)
visitor.send(:"visit_#{type.to_s.downcase}", self)
end
###
# Convert this CSS node to xpath with +prefix+ using +visitor+
def to_xpath(visitor)
prefix = if ALLOW_COMBINATOR_ON_SELF.include?(type) && value.first.nil?
"."
else
visitor.prefix
end
prefix + visitor.accept(self)
end
# Find a node by type using +types+
def find_by_type(types)
matches = []
matches << self if to_type == types
@value.each do |v|
matches += v.find_by_type(types) if v.respond_to?(:find_by_type)
end
matches
end
# Convert to_type
def to_type
[@type] + @value.filter_map do |n|
n.to_type if n.respond_to?(:to_type)
end
end
# Convert to array
def to_a
[@type] + @value.map { |n| n.respond_to?(:to_a) ? n.to_a : [n] }
end
end
end
end
@@ -0,0 +1,772 @@
# frozen_string_literal: true
#
# DO NOT MODIFY!!!!
# This file is automatically generated by Racc 1.8.0
# from Racc grammar file "parser.y".
#
require 'racc/parser.rb'
require_relative "parser_extras"
module Nokogiri
module CSS
# :nodoc: all
class Parser < Racc::Parser
end
end
end
module Nokogiri
module CSS
class Parser < Racc::Parser
def unescape_css_identifier(identifier)
identifier.gsub(/\\(?:([^0-9a-fA-F])|([0-9a-fA-F]{1,6})\s?)/){ |m| $1 || [$2.hex].pack('U') }
end
def unescape_css_string(str)
str.gsub(/\\(?:([^0-9a-fA-F])|([0-9a-fA-F]{1,6})\s?)/) do |m|
if $1=="\n"
''
else
$1 || [$2.hex].pack('U')
end
end
end
##### State transition tables begin ###
racc_action_table = [
27, 11, 38, 99, 36, 12, 40, 26, 48, 25,
49, 27, 100, 12, 30, 36, 105, 99, -26, 28,
25, -26, 26, 27, 29, 14, 21, 23, 80, 30,
28, 36, 72, 26, -26, 29, 14, 21, 23, 27,
30, 91, 56, 36, 97, 96, 43, 29, 25, 26,
27, 92, 94, 21, 36, 95, 30, 98, 28, 25,
101, 26, 102, 29, 14, 21, 23, 96, 30, 28,
36, 36, 26, 103, 29, 14, 21, 23, 27, 30,
108, 107, 36, 109, 106, 43, 43, 25, 26, 26,
27, 110, 21, 21, 111, 30, 30, 28, 99, 50,
26, 53, 29, 14, 21, 23, 36, 30, 36, 56,
61, 64, 113, 66, 29, 14, 116, 36, 118, 36,
nil, 43, nil, 43, 26, nil, 26, 14, 21, 23,
21, 30, 43, 30, 43, 26, nil, 26, 36, 21,
36, 21, 30, 25, 30, nil, nil, nil, nil, nil,
nil, 61, 62, 43, 60, 43, 26, nil, 26, nil,
21, 23, 21, 30, 57, 30, 88, 89, 14, nil,
nil, 88, 89, nil, nil, nil, nil, 84, 85, 86,
nil, 87, 84, 85, 86, 83, 87, nil, 61, 93,
83, 66, 61, 93, nil, 66, 61, 93, nil, 66,
61, 93, nil, 66, nil, 14, nil, 61, 93, 14,
66, nil, nil, 14, nil, nil, nil, 14, 4, 5,
10, nil, nil, nil, 14, 4, 5, 47, 6, nil,
8, 7, 4, 5, 10, 6, nil, 8, 7, nil,
nil, nil, 6, nil, 8, 7 ]
racc_action_check = [
3, 1, 11, 64, 3, 70, 14, 17, 21, 3,
24, 9, 62, 1, 17, 9, 70, 62, 25, 3,
9, 64, 3, 30, 3, 3, 3, 3, 49, 3,
9, 16, 30, 9, 50, 9, 9, 9, 9, 12,
9, 53, 30, 12, 60, 60, 16, 30, 12, 16,
46, 54, 58, 16, 46, 59, 16, 61, 12, 46,
63, 12, 65, 12, 12, 12, 12, 66, 12, 46,
31, 32, 46, 67, 46, 46, 46, 46, 47, 46,
82, 82, 47, 82, 81, 31, 32, 47, 31, 32,
26, 90, 31, 32, 92, 31, 32, 47, 93, 26,
47, 26, 47, 47, 47, 47, 28, 47, 33, 26,
28, 28, 97, 28, 26, 26, 100, 34, 113, 35,
nil, 28, nil, 33, 28, nil, 33, 28, 28, 28,
33, 28, 34, 33, 35, 34, nil, 35, 43, 34,
68, 35, 34, 43, 35, nil, nil, nil, nil, nil,
nil, 27, 27, 43, 27, 68, 43, nil, 68, nil,
43, 43, 68, 43, 27, 68, 51, 51, 27, nil,
nil, 52, 52, nil, nil, nil, nil, 51, 51, 51,
nil, 51, 52, 52, 52, 51, 52, nil, 56, 56,
52, 56, 96, 96, nil, 96, 98, 98, nil, 98,
99, 99, nil, 99, nil, 56, nil, 101, 101, 96,
101, nil, nil, 98, nil, nil, nil, 99, 0, 0,
0, nil, nil, nil, 101, 20, 20, 20, 0, nil,
0, 0, 29, 29, 29, 20, nil, 20, 20, nil,
nil, nil, 29, nil, 29, 29 ]
racc_action_pointer = [
211, 1, nil, -2, nil, nil, nil, nil, nil, 9,
nil, 2, 37, nil, -5, nil, 25, -17, nil, nil,
218, -3, nil, nil, -20, -12, 88, 141, 100, 225,
21, 64, 65, 102, 111, 113, nil, nil, nil, nil,
nil, nil, nil, 132, nil, nil, 48, 76, nil, 17,
4, 163, 168, 16, 21, nil, 178, nil, 29, 32,
33, 45, 5, 48, -9, 39, 55, 50, 134, nil,
-7, nil, nil, nil, nil, nil, nil, nil, nil, nil,
nil, 59, 70, nil, nil, nil, nil, nil, nil, nil,
66, nil, 83, 86, nil, nil, 182, 105, 186, 190,
103, 197, nil, nil, nil, nil, nil, nil, nil, nil,
nil, nil, nil, 105, nil, nil, nil, nil, nil ]
racc_action_default = [
-81, -82, -2, -27, -4, -5, -6, -7, -8, -27,
-80, -82, -27, -3, -82, -10, -53, -12, -15, -16,
-20, -82, -22, -23, -82, -25, -27, -82, -27, -81,
-82, -59, -60, -61, -62, -63, -64, -17, 119, -1,
-9, -11, -52, -27, -13, -14, -27, -27, -21, -82,
-32, -68, -68, -82, -82, -33, -82, -34, -82, -82,
-43, -44, -45, -46, -25, -82, -43, -82, -77, -79,
-82, -50, -51, -54, -55, -56, -57, -58, -18, -19,
-24, -82, -82, -69, -70, -71, -72, -73, -74, -75,
-82, -30, -82, -45, -35, -36, -82, -49, -82, -82,
-82, -82, -37, -76, -78, -38, -28, -65, -66, -67,
-29, -31, -39, -82, -40, -41, -48, -42, -47 ]
racc_goto_table = [
58, 42, 13, 1, 46, 52, 19, 68, 37, 71,
41, 39, 19, 69, 44, 19, 73, 74, 75, 76,
77, 45, 68, 81, 90, 54, 51, 59, 69, 55,
nil, nil, 70, nil, nil, nil, nil, nil, nil, nil,
nil, nil, nil, nil, nil, 78, 79, nil, nil, 19,
19, nil, nil, 104, nil, nil, nil, nil, nil, nil,
nil, nil, nil, nil, nil, nil, nil, nil, nil, 112,
nil, 114, 115, nil, 117 ]
racc_goto_check = [
20, 14, 2, 1, 5, 11, 7, 9, 2, 11,
10, 2, 7, 14, 12, 7, 14, 14, 14, 14,
14, 13, 9, 19, 19, 17, 18, 21, 14, 7,
nil, nil, 1, nil, nil, nil, nil, nil, nil, nil,
nil, nil, nil, nil, nil, 2, 2, nil, nil, 7,
7, nil, nil, 14, nil, nil, nil, nil, nil, nil,
nil, nil, nil, nil, nil, nil, nil, nil, nil, 20,
nil, 20, 20, nil, 20 ]
racc_goto_pointer = [
nil, 3, -1, nil, nil, -16, nil, 3, nil, -21,
-6, -21, -3, 4, -15, nil, nil, -1, 0, -28,
-27, 0, nil, nil, nil, nil ]
racc_goto_default = [
nil, nil, nil, 2, 3, 9, 15, 63, 20, 16,
nil, 17, 34, 33, 18, 32, 22, 24, nil, nil,
65, nil, 31, 35, 82, 67 ]
racc_reduce_table = [
0, 0, :racc_error,
3, 33, :_reduce_1,
1, 33, :_reduce_2,
2, 33, :_reduce_3,
1, 37, :_reduce_4,
1, 37, :_reduce_5,
1, 37, :_reduce_6,
1, 37, :_reduce_7,
1, 37, :_reduce_8,
2, 38, :_reduce_9,
1, 39, :_reduce_10,
2, 40, :_reduce_11,
1, 40, :_reduce_none,
2, 40, :_reduce_13,
2, 40, :_reduce_14,
1, 40, :_reduce_15,
1, 40, :_reduce_none,
2, 35, :_reduce_17,
3, 34, :_reduce_18,
3, 34, :_reduce_19,
1, 34, :_reduce_none,
2, 47, :_reduce_21,
1, 41, :_reduce_none,
1, 41, :_reduce_23,
3, 48, :_reduce_24,
1, 48, :_reduce_25,
1, 49, :_reduce_26,
0, 49, :_reduce_none,
4, 45, :_reduce_28,
4, 45, :_reduce_29,
3, 45, :_reduce_30,
3, 50, :_reduce_31,
1, 50, :_reduce_32,
1, 50, :_reduce_none,
2, 43, :_reduce_34,
3, 43, :_reduce_35,
3, 43, :_reduce_36,
3, 43, :_reduce_37,
3, 43, :_reduce_38,
3, 52, :_reduce_39,
3, 52, :_reduce_40,
3, 52, :_reduce_41,
3, 52, :_reduce_42,
1, 52, :_reduce_none,
1, 52, :_reduce_none,
1, 52, :_reduce_45,
1, 52, :_reduce_none,
4, 53, :_reduce_47,
3, 53, :_reduce_48,
2, 53, :_reduce_49,
2, 44, :_reduce_50,
2, 44, :_reduce_51,
1, 42, :_reduce_none,
0, 42, :_reduce_none,
2, 46, :_reduce_54,
2, 46, :_reduce_55,
2, 46, :_reduce_56,
2, 46, :_reduce_57,
2, 46, :_reduce_58,
1, 46, :_reduce_none,
1, 46, :_reduce_none,
1, 46, :_reduce_none,
1, 46, :_reduce_none,
1, 46, :_reduce_none,
1, 54, :_reduce_64,
2, 51, :_reduce_65,
2, 51, :_reduce_66,
2, 51, :_reduce_67,
0, 51, :_reduce_none,
1, 56, :_reduce_69,
1, 56, :_reduce_70,
1, 56, :_reduce_71,
1, 56, :_reduce_72,
1, 56, :_reduce_73,
1, 56, :_reduce_74,
1, 56, :_reduce_75,
3, 55, :_reduce_76,
1, 57, :_reduce_none,
2, 57, :_reduce_none,
1, 57, :_reduce_none,
1, 36, :_reduce_none,
0, 36, :_reduce_none ]
racc_reduce_n = 82
racc_shift_n = 119
racc_token_table = {
false => 0,
:error => 1,
:FUNCTION => 2,
:INCLUDES => 3,
:DASHMATCH => 4,
:LBRACE => 5,
:HASH => 6,
:PLUS => 7,
:GREATER => 8,
:S => 9,
:STRING => 10,
:IDENT => 11,
:COMMA => 12,
:NUMBER => 13,
:PREFIXMATCH => 14,
:SUFFIXMATCH => 15,
:SUBSTRINGMATCH => 16,
:TILDE => 17,
:NOT_EQUAL => 18,
:SLASH => 19,
:DOUBLESLASH => 20,
:NOT => 21,
:EQUAL => 22,
:RPAREN => 23,
:LSQUARE => 24,
:RSQUARE => 25,
:HAS => 26,
"@" => 27,
"." => 28,
"*" => 29,
"|" => 30,
":" => 31 }
racc_nt_base = 32
racc_use_result_var = true
Racc_arg = [
racc_action_table,
racc_action_check,
racc_action_default,
racc_action_pointer,
racc_goto_table,
racc_goto_check,
racc_goto_default,
racc_goto_pointer,
racc_nt_base,
racc_reduce_table,
racc_token_table,
racc_shift_n,
racc_reduce_n,
racc_use_result_var ]
Ractor.make_shareable(Racc_arg) if defined?(Ractor)
Racc_token_to_s_table = [
"$end",
"error",
"FUNCTION",
"INCLUDES",
"DASHMATCH",
"LBRACE",
"HASH",
"PLUS",
"GREATER",
"S",
"STRING",
"IDENT",
"COMMA",
"NUMBER",
"PREFIXMATCH",
"SUFFIXMATCH",
"SUBSTRINGMATCH",
"TILDE",
"NOT_EQUAL",
"SLASH",
"DOUBLESLASH",
"NOT",
"EQUAL",
"RPAREN",
"LSQUARE",
"RSQUARE",
"HAS",
"\"@\"",
"\".\"",
"\"*\"",
"\"|\"",
"\":\"",
"$start",
"selector",
"simple_selector_1toN",
"prefixless_combinator_selector",
"optional_S",
"combinator",
"xpath_attribute_name",
"xpath_attribute",
"simple_selector",
"element_name",
"hcap_0toN",
"function",
"pseudo",
"attrib",
"hcap_1toN",
"class",
"namespaced_ident",
"namespace",
"attrib_name",
"attrib_val_0or1",
"expr",
"nth",
"attribute_id",
"negation",
"eql_incl_dash",
"negation_arg" ]
Ractor.make_shareable(Racc_token_to_s_table) if defined?(Ractor)
Racc_debug_parser = false
##### State transition tables end #####
# reduce 0 omitted
def _reduce_1(val, _values, result)
result = [val[0], val[2]].flatten
result
end
def _reduce_2(val, _values, result)
result = val.flatten
result
end
def _reduce_3(val, _values, result)
result = [val[1]].flatten
result
end
def _reduce_4(val, _values, result)
result = :DIRECT_ADJACENT_SELECTOR
result
end
def _reduce_5(val, _values, result)
result = :CHILD_SELECTOR
result
end
def _reduce_6(val, _values, result)
result = :FOLLOWING_SELECTOR
result
end
def _reduce_7(val, _values, result)
result = :DESCENDANT_SELECTOR
result
end
def _reduce_8(val, _values, result)
result = :CHILD_SELECTOR
result
end
def _reduce_9(val, _values, result)
result = val[1]
result
end
def _reduce_10(val, _values, result)
result = Node.new(:ATTRIB_NAME, [val[0]])
result
end
def _reduce_11(val, _values, result)
result = if val[1].nil?
val[0]
else
Node.new(:CONDITIONAL_SELECTOR, [val[0], val[1]])
end
result
end
# reduce 12 omitted
def _reduce_13(val, _values, result)
result = Node.new(:CONDITIONAL_SELECTOR, val)
result
end
def _reduce_14(val, _values, result)
result = Node.new(:CONDITIONAL_SELECTOR, val)
result
end
def _reduce_15(val, _values, result)
result = Node.new(:CONDITIONAL_SELECTOR, [Node.new(:ELEMENT_NAME, ['*']), val[0]])
result
end
# reduce 16 omitted
def _reduce_17(val, _values, result)
result = Node.new(val[0], [nil, val[1]])
result
end
def _reduce_18(val, _values, result)
result = Node.new(val[1], [val[0], val[2]])
result
end
def _reduce_19(val, _values, result)
result = Node.new(:DESCENDANT_SELECTOR, [val[0], val[2]])
result
end
# reduce 20 omitted
def _reduce_21(val, _values, result)
result = Node.new(:CLASS_CONDITION, [unescape_css_identifier(val[1])])
result
end
# reduce 22 omitted
def _reduce_23(val, _values, result)
result = Node.new(:ELEMENT_NAME, val)
result
end
def _reduce_24(val, _values, result)
result = Node.new(:ELEMENT_NAME, [val[0], val[2]])
result
end
def _reduce_25(val, _values, result)
name = val[0]
result = Node.new(:ELEMENT_NAME, [name])
result
end
def _reduce_26(val, _values, result)
result = val[0]
result
end
# reduce 27 omitted
def _reduce_28(val, _values, result)
result = Node.new(:ATTRIBUTE_CONDITION, [val[1]] + (val[2] || []))
result
end
def _reduce_29(val, _values, result)
result = Node.new(:ATTRIBUTE_CONDITION, [val[1]] + (val[2] || []))
result
end
def _reduce_30(val, _values, result)
result = Node.new(:PSEUDO_CLASS, [Node.new(:FUNCTION, ['nth-child(', val[1]])])
result
end
def _reduce_31(val, _values, result)
result = Node.new(:ATTRIB_NAME, [[val[0], val[2]].compact.join(':')])
result
end
def _reduce_32(val, _values, result)
result = Node.new(:ATTRIB_NAME, [val[0]])
result
end
# reduce 33 omitted
def _reduce_34(val, _values, result)
result = Node.new(:FUNCTION, [val[0].strip])
result
end
def _reduce_35(val, _values, result)
result = Node.new(:FUNCTION, [val[0].strip, val[1]].flatten)
result
end
def _reduce_36(val, _values, result)
result = Node.new(:FUNCTION, [val[0].strip, val[1]].flatten)
result
end
def _reduce_37(val, _values, result)
result = Node.new(:FUNCTION, [val[0].strip, val[1]].flatten)
result
end
def _reduce_38(val, _values, result)
result = Node.new(:FUNCTION, [val[0].strip, val[1]].flatten)
result
end
def _reduce_39(val, _values, result)
result = [val[0], val[2]]
result
end
def _reduce_40(val, _values, result)
result = [val[0], val[2]]
result
end
def _reduce_41(val, _values, result)
result = [val[0], val[2]]
result
end
def _reduce_42(val, _values, result)
result = [val[0], val[2]]
result
end
# reduce 43 omitted
# reduce 44 omitted
def _reduce_45(val, _values, result)
case val[0]
when 'even'
result = Node.new(:NTH, ['2','n','+','0'])
when 'odd'
result = Node.new(:NTH, ['2','n','+','1'])
when 'n'
result = Node.new(:NTH, ['1','n','+','0'])
else
result = val
end
result
end
# reduce 46 omitted
def _reduce_47(val, _values, result)
if val[1] == 'n'
result = Node.new(:NTH, val)
else
raise Racc::ParseError, "parse error on IDENT '#{val[1]}'"
end
result
end
def _reduce_48(val, _values, result)
# n+3, -n+3
if val[0] == 'n'
val.unshift("1")
result = Node.new(:NTH, val)
elsif val[0] == '-n'
val[0] = 'n'
val.unshift("-1")
result = Node.new(:NTH, val)
else
raise Racc::ParseError, "parse error on IDENT '#{val[1]}'"
end
result
end
def _reduce_49(val, _values, result)
# 5n, -5n, 10n-1
n = val[1]
if n[0, 2] == 'n-'
val[1] = 'n'
val << "-"
# b is contained in n as n is the string "n-b"
val << n[2, n.size]
result = Node.new(:NTH, val)
elsif n == 'n'
val << "+"
val << "0"
result = Node.new(:NTH, val)
else
raise Racc::ParseError, "parse error on IDENT '#{val[1]}'"
end
result
end
def _reduce_50(val, _values, result)
result = Node.new(:PSEUDO_CLASS, [val[1]])
result
end
def _reduce_51(val, _values, result)
result = Node.new(:PSEUDO_CLASS, [val[1]])
result
end
# reduce 52 omitted
# reduce 53 omitted
def _reduce_54(val, _values, result)
result = Node.new(:COMBINATOR, val)
result
end
def _reduce_55(val, _values, result)
result = Node.new(:COMBINATOR, val)
result
end
def _reduce_56(val, _values, result)
result = Node.new(:COMBINATOR, val)
result
end
def _reduce_57(val, _values, result)
result = Node.new(:COMBINATOR, val)
result
end
def _reduce_58(val, _values, result)
result = Node.new(:COMBINATOR, val)
result
end
# reduce 59 omitted
# reduce 60 omitted
# reduce 61 omitted
# reduce 62 omitted
# reduce 63 omitted
def _reduce_64(val, _values, result)
result = Node.new(:ID, [unescape_css_identifier(val[0])])
result
end
def _reduce_65(val, _values, result)
result = [val[0], unescape_css_identifier(val[1])]
result
end
def _reduce_66(val, _values, result)
result = [val[0], unescape_css_string(val[1])]
result
end
def _reduce_67(val, _values, result)
result = [val[0], val[1]]
result
end
# reduce 68 omitted
def _reduce_69(val, _values, result)
result = :equal
result
end
def _reduce_70(val, _values, result)
result = :prefix_match
result
end
def _reduce_71(val, _values, result)
result = :suffix_match
result
end
def _reduce_72(val, _values, result)
result = :substring_match
result
end
def _reduce_73(val, _values, result)
result = :not_equal
result
end
def _reduce_74(val, _values, result)
result = :includes
result
end
def _reduce_75(val, _values, result)
result = :dash_match
result
end
def _reduce_76(val, _values, result)
result = Node.new(:NOT, [val[1]])
result
end
# reduce 77 omitted
# reduce 78 omitted
# reduce 79 omitted
# reduce 80 omitted
# reduce 81 omitted
def _reduce_none(val, _values, result)
val[0]
end
end # class Parser
end # module CSS
end # module Nokogiri
@@ -0,0 +1,277 @@
class Nokogiri::CSS::Parser
token FUNCTION INCLUDES DASHMATCH LBRACE HASH PLUS GREATER S STRING IDENT
token COMMA NUMBER PREFIXMATCH SUFFIXMATCH SUBSTRINGMATCH TILDE NOT_EQUAL
token SLASH DOUBLESLASH NOT EQUAL RPAREN LSQUARE RSQUARE HAS
rule
selector:
selector COMMA simple_selector_1toN {
result = [val[0], val[2]].flatten
}
| prefixless_combinator_selector { result = val.flatten }
| optional_S simple_selector_1toN { result = [val[1]].flatten }
;
combinator:
PLUS { result = :DIRECT_ADJACENT_SELECTOR }
| GREATER { result = :CHILD_SELECTOR }
| TILDE { result = :FOLLOWING_SELECTOR }
| DOUBLESLASH { result = :DESCENDANT_SELECTOR }
| SLASH { result = :CHILD_SELECTOR }
;
xpath_attribute_name:
'@' IDENT { result = val[1] }
;
xpath_attribute:
xpath_attribute_name { result = Node.new(:ATTRIB_NAME, [val[0]]) }
;
simple_selector:
element_name hcap_0toN {
result = if val[1].nil?
val[0]
else
Node.new(:CONDITIONAL_SELECTOR, [val[0], val[1]])
end
}
| function
| function pseudo { result = Node.new(:CONDITIONAL_SELECTOR, val) }
| function attrib { result = Node.new(:CONDITIONAL_SELECTOR, val) }
| hcap_1toN { result = Node.new(:CONDITIONAL_SELECTOR, [Node.new(:ELEMENT_NAME, ['*']), val[0]]) }
| xpath_attribute
;
prefixless_combinator_selector:
combinator simple_selector_1toN { result = Node.new(val[0], [nil, val[1]]) }
;
simple_selector_1toN:
simple_selector combinator simple_selector_1toN { result = Node.new(val[1], [val[0], val[2]]) }
| simple_selector S simple_selector_1toN { result = Node.new(:DESCENDANT_SELECTOR, [val[0], val[2]]) }
| simple_selector
;
class:
'.' IDENT { result = Node.new(:CLASS_CONDITION, [unescape_css_identifier(val[1])]) }
;
element_name:
namespaced_ident
| '*' { result = Node.new(:ELEMENT_NAME, val) }
;
namespaced_ident:
namespace '|' IDENT { result = Node.new(:ELEMENT_NAME, [val[0], val[2]]) }
| IDENT {
name = val[0]
result = Node.new(:ELEMENT_NAME, [name])
}
;
namespace:
IDENT { result = val[0] }
|
;
attrib:
LSQUARE attrib_name attrib_val_0or1 RSQUARE {
result = Node.new(:ATTRIBUTE_CONDITION, [val[1]] + (val[2] || []))
}
| LSQUARE function attrib_val_0or1 RSQUARE {
result = Node.new(:ATTRIBUTE_CONDITION, [val[1]] + (val[2] || []))
}
| LSQUARE NUMBER RSQUARE {
result = Node.new(:PSEUDO_CLASS, [Node.new(:FUNCTION, ['nth-child(', val[1]])])
}
;
attrib_name:
namespace '|' IDENT { result = Node.new(:ATTRIB_NAME, [[val[0], val[2]].compact.join(':')]) }
| IDENT { result = Node.new(:ATTRIB_NAME, [val[0]]) }
| xpath_attribute
;
function:
FUNCTION RPAREN {
result = Node.new(:FUNCTION, [val[0].strip])
}
| FUNCTION expr RPAREN {
result = Node.new(:FUNCTION, [val[0].strip, val[1]].flatten)
}
| FUNCTION nth RPAREN {
result = Node.new(:FUNCTION, [val[0].strip, val[1]].flatten)
}
| NOT expr RPAREN {
result = Node.new(:FUNCTION, [val[0].strip, val[1]].flatten)
}
| HAS selector RPAREN {
result = Node.new(:FUNCTION, [val[0].strip, val[1]].flatten)
}
;
expr:
NUMBER COMMA expr { result = [val[0], val[2]] }
| STRING COMMA expr { result = [val[0], val[2]] }
| IDENT COMMA expr { result = [val[0], val[2]] }
| xpath_attribute COMMA expr { result = [val[0], val[2]] }
| NUMBER
| STRING
| IDENT {
case val[0]
when 'even'
result = Node.new(:NTH, ['2','n','+','0'])
when 'odd'
result = Node.new(:NTH, ['2','n','+','1'])
when 'n'
result = Node.new(:NTH, ['1','n','+','0'])
else
result = val
end
}
| xpath_attribute
;
nth:
NUMBER IDENT PLUS NUMBER # 5n+3 -5n+3
{
if val[1] == 'n'
result = Node.new(:NTH, val)
else
raise Racc::ParseError, "parse error on IDENT '#{val[1]}'"
end
}
| IDENT PLUS NUMBER { # n+3, -n+3
if val[0] == 'n'
val.unshift("1")
result = Node.new(:NTH, val)
elsif val[0] == '-n'
val[0] = 'n'
val.unshift("-1")
result = Node.new(:NTH, val)
else
raise Racc::ParseError, "parse error on IDENT '#{val[1]}'"
end
}
| NUMBER IDENT { # 5n, -5n, 10n-1
n = val[1]
if n[0, 2] == 'n-'
val[1] = 'n'
val << "-"
# b is contained in n as n is the string "n-b"
val << n[2, n.size]
result = Node.new(:NTH, val)
elsif n == 'n'
val << "+"
val << "0"
result = Node.new(:NTH, val)
else
raise Racc::ParseError, "parse error on IDENT '#{val[1]}'"
end
}
;
pseudo:
':' function {
result = Node.new(:PSEUDO_CLASS, [val[1]])
}
| ':' IDENT { result = Node.new(:PSEUDO_CLASS, [val[1]]) }
;
hcap_0toN:
hcap_1toN
|
;
hcap_1toN:
attribute_id hcap_1toN {
result = Node.new(:COMBINATOR, val)
}
| class hcap_1toN {
result = Node.new(:COMBINATOR, val)
}
| attrib hcap_1toN {
result = Node.new(:COMBINATOR, val)
}
| pseudo hcap_1toN {
result = Node.new(:COMBINATOR, val)
}
| negation hcap_1toN {
result = Node.new(:COMBINATOR, val)
}
| attribute_id
| class
| attrib
| pseudo
| negation
;
attribute_id:
HASH { result = Node.new(:ID, [unescape_css_identifier(val[0])]) }
;
attrib_val_0or1:
eql_incl_dash IDENT { result = [val[0], unescape_css_identifier(val[1])] }
| eql_incl_dash STRING { result = [val[0], unescape_css_string(val[1])] }
| eql_incl_dash NUMBER { result = [val[0], val[1]] }
|
;
eql_incl_dash:
EQUAL { result = :equal }
| PREFIXMATCH { result = :prefix_match }
| SUFFIXMATCH { result = :suffix_match }
| SUBSTRINGMATCH { result = :substring_match }
| NOT_EQUAL { result = :not_equal }
| INCLUDES { result = :includes }
| DASHMATCH { result = :dash_match }
;
negation:
NOT negation_arg RPAREN {
result = Node.new(:NOT, [val[1]])
}
;
negation_arg:
element_name
| element_name hcap_1toN
| hcap_1toN
;
optional_S:
S
|
;
end
---- header
require_relative "parser_extras"
module Nokogiri
module CSS
# :nodoc: all
class Parser < Racc::Parser
end
end
end
---- inner
def unescape_css_identifier(identifier)
identifier.gsub(/\\(?:([^0-9a-fA-F])|([0-9a-fA-F]{1,6})\s?)/){ |m| $1 || [$2.hex].pack('U') }
end
def unescape_css_string(str)
str.gsub(/\\(?:([^0-9a-fA-F])|([0-9a-fA-F]{1,6})\s?)/) do |m|
if $1=="\n"
''
else
$1 || [$2.hex].pack('U')
end
end
end
@@ -0,0 +1,36 @@
# frozen_string_literal: true
require "thread"
module Nokogiri
module CSS
class Parser < Racc::Parser # :nodoc:
def initialize
@tokenizer = Tokenizer.new
super
end
def parse(string)
@tokenizer.scan_setup(string)
do_parse
end
def next_token
@tokenizer.next_token
end
# Get the xpath for +selector+ using +visitor+
def xpath_for(selector, visitor)
parse(selector).map do |ast|
ast.to_xpath(visitor)
end
end
# On CSS parser error, raise an exception
def on_error(error_token_id, error_value, value_stack)
after = value_stack.compact.last
raise SyntaxError, "unexpected '#{error_value}' after '#{after}'"
end
end
end
end
@@ -0,0 +1,38 @@
# frozen_string_literal: true
module Nokogiri
module CSS
module SelectorCache # :nodoc:
@cache = {}
@mutex = Mutex.new
class << self
# Retrieve the cached XPath expressions for the key
def [](key)
@mutex.synchronize { @cache[key] }
end
# Insert the XPath expressions `value` at the cache key
def []=(key, value)
@mutex.synchronize { @cache[key] = value }
end
# Clear the cache
def clear_cache(create_new_object = false)
@mutex.synchronize do
if create_new_object # used in tests to avoid 'method redefined' warnings when injecting spies
@cache = {}
else
@cache.clear
end
end
end
# Construct a unique key cache key
def key(selector:, visitor:)
[selector, visitor.config]
end
end
end
end
end
@@ -0,0 +1,9 @@
# frozen_string_literal: true
require_relative "../syntax_error"
module Nokogiri
module CSS
class SyntaxError < ::Nokogiri::SyntaxError
end
end
end
@@ -0,0 +1,155 @@
# frozen_string_literal: true
#--
# DO NOT MODIFY!!!!
# This file is automatically generated by rex 1.0.7
# from lexical definition file "lib/nokogiri/css/tokenizer.rex".
#++
module Nokogiri
module CSS
# :nodoc: all
class Tokenizer
require 'strscan'
class ScanError < StandardError ; end
attr_reader :lineno
attr_reader :filename
attr_accessor :state
def scan_setup(str)
@ss = StringScanner.new(str)
@lineno = 1
@state = nil
end
def action
yield
end
def scan_str(str)
scan_setup(str)
do_parse
end
alias :scan :scan_str
def load_file( filename )
@filename = filename
File.open(filename, "r") do |f|
scan_setup(f.read)
end
end
def scan_file( filename )
load_file(filename)
do_parse
end
def next_token
return if @ss.eos?
# skips empty actions
until token = _next_token or @ss.eos?; end
token
end
def _next_token
text = @ss.peek(1)
@lineno += 1 if text == "\n"
token = case @state
when nil
case
when (text = @ss.scan(/has\([\s]*/))
action { [:HAS, text] }
when (text = @ss.scan(/-?([_A-Za-z]|[^\0-\177]|(\\[0-9A-Fa-f]{1,6}(\r\n|[\s])?|\\[^\n\r\f0-9A-Fa-f]))([_A-Za-z0-9-]|[^\0-\177]|(\\[0-9A-Fa-f]{1,6}(\r\n|[\s])?|\\[^\n\r\f0-9A-Fa-f]))*\([\s]*/))
action { [:FUNCTION, text] }
when (text = @ss.scan(/-?([_A-Za-z]|[^\0-\177]|(\\[0-9A-Fa-f]{1,6}(\r\n|[\s])?|\\[^\n\r\f0-9A-Fa-f]))([_A-Za-z0-9-]|[^\0-\177]|(\\[0-9A-Fa-f]{1,6}(\r\n|[\s])?|\\[^\n\r\f0-9A-Fa-f]))*/))
action { [:IDENT, text] }
when (text = @ss.scan(/\#([_A-Za-z0-9-]|[^\0-\177]|(\\[0-9A-Fa-f]{1,6}(\r\n|[\s])?|\\[^\n\r\f0-9A-Fa-f]))+/))
action { [:HASH, text] }
when (text = @ss.scan(/[\s]*~=[\s]*/))
action { [:INCLUDES, text] }
when (text = @ss.scan(/[\s]*\|=[\s]*/))
action { [:DASHMATCH, text] }
when (text = @ss.scan(/[\s]*\^=[\s]*/))
action { [:PREFIXMATCH, text] }
when (text = @ss.scan(/[\s]*\$=[\s]*/))
action { [:SUFFIXMATCH, text] }
when (text = @ss.scan(/[\s]*\*=[\s]*/))
action { [:SUBSTRINGMATCH, text] }
when (text = @ss.scan(/[\s]*!=[\s]*/))
action { [:NOT_EQUAL, text] }
when (text = @ss.scan(/[\s]*=[\s]*/))
action { [:EQUAL, text] }
when (text = @ss.scan(/[\s]*\)/))
action { [:RPAREN, text] }
when (text = @ss.scan(/\[[\s]*/))
action { [:LSQUARE, text] }
when (text = @ss.scan(/[\s]*\]/))
action { [:RSQUARE, text] }
when (text = @ss.scan(/[\s]*\+[\s]*/))
action { [:PLUS, text] }
when (text = @ss.scan(/[\s]*>[\s]*/))
action { [:GREATER, text] }
when (text = @ss.scan(/[\s]*,[\s]*/))
action { [:COMMA, text] }
when (text = @ss.scan(/[\s]*~[\s]*/))
action { [:TILDE, text] }
when (text = @ss.scan(/\:not\([\s]*/))
action { [:NOT, text] }
when (text = @ss.scan(/-?([0-9]+|[0-9]*\.[0-9]+)/))
action { [:NUMBER, text] }
when (text = @ss.scan(/[\s]*\/\/[\s]*/))
action { [:DOUBLESLASH, text] }
when (text = @ss.scan(/[\s]*\/[\s]*/))
action { [:SLASH, text] }
when (text = @ss.scan(/U\+[0-9a-f?]{1,6}(-[0-9a-f]{1,6})?/))
action {[:UNICODE_RANGE, text] }
when (text = @ss.scan(/[\s]+/))
action { [:S, text] }
when (text = @ss.scan(/("([^\n\r\f"]|(\n|\r\n|\r|\f)|[^\0-\177]|(\\[0-9A-Fa-f]{1,6}(\r\n|[\s])?|\\[^\n\r\f0-9A-Fa-f]))*(?<!\\)(?:\\{2})*"|'([^\n\r\f']|(\n|\r\n|\r|\f)|[^\0-\177]|(\\[0-9A-Fa-f]{1,6}(\r\n|[\s])?|\\[^\n\r\f0-9A-Fa-f]))*(?<!\\)(?:\\{2})*')/))
action { [:STRING, text] }
when (text = @ss.scan(/./))
action { [text, text] }
else
text = @ss.string[@ss.pos .. -1]
raise ScanError, "can not match: '" + text + "'"
end # if
else
raise ScanError, "undefined state: '" + state.to_s + "'"
end # case state
token
end # def _next_token
end # class
end
end
@@ -0,0 +1,57 @@
module Nokogiri
module CSS
# :nodoc: all
class Tokenizer
macro
nl (\n|\r\n|\r|\f)
w [\s]*
nonascii [^\0-\177]
num -?([0-9]+|[0-9]*\.[0-9]+)
unicode \\[0-9A-Fa-f]{1,6}(\r\n|[\s])?
escape ({unicode}|\\[^\n\r\f0-9A-Fa-f])
nmchar ([_A-Za-z0-9-]|{nonascii}|{escape})
nmstart ([_A-Za-z]|{nonascii}|{escape})
name {nmstart}{nmchar}*
ident -?{name}
charref {nmchar}+
string1 "([^\n\r\f"]|{nl}|{nonascii}|{escape})*(?<!\\)(?:\\{2})*"
string2 '([^\n\r\f']|{nl}|{nonascii}|{escape})*(?<!\\)(?:\\{2})*'
string ({string1}|{string2})
rule
# [:state] pattern [actions]
has\({w} { [:HAS, text] }
{ident}\({w} { [:FUNCTION, text] }
{ident} { [:IDENT, text] }
\#{charref} { [:HASH, text] }
{w}~={w} { [:INCLUDES, text] }
{w}\|={w} { [:DASHMATCH, text] }
{w}\^={w} { [:PREFIXMATCH, text] }
{w}\$={w} { [:SUFFIXMATCH, text] }
{w}\*={w} { [:SUBSTRINGMATCH, text] }
{w}!={w} { [:NOT_EQUAL, text] }
{w}={w} { [:EQUAL, text] }
{w}\) { [:RPAREN, text] }
\[{w} { [:LSQUARE, text] }
{w}\] { [:RSQUARE, text] }
{w}\+{w} { [:PLUS, text] }
{w}>{w} { [:GREATER, text] }
{w},{w} { [:COMMA, text] }
{w}~{w} { [:TILDE, text] }
\:not\({w} { [:NOT, text] }
{num} { [:NUMBER, text] }
{w}\/\/{w} { [:DOUBLESLASH, text] }
{w}\/{w} { [:SLASH, text] }
U\+[0-9a-f?]{1,6}(-[0-9a-f]{1,6})? {[:UNICODE_RANGE, text] }
[\s]+ { [:S, text] }
{string} { [:STRING, text] }
. { [text, text] }
end
end
end
@@ -0,0 +1,376 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module CSS
# When translating CSS selectors to XPath queries with Nokogiri::CSS.xpath_for, the XPathVisitor
# class allows for changing some of the behaviors related to builtin xpath functions and quirks
# of HTML5.
class XPathVisitor
WILDCARD_NAMESPACES = Nokogiri.libxml2_patches.include?("0009-allow-wildcard-namespaces.patch") # :nodoc:
# Enum to direct XPathVisitor when to use Nokogiri builtin XPath functions.
module BuiltinsConfig
# Never use Nokogiri builtin functions, always generate vanilla XPath 1.0 queries. This is
# the default when calling Nokogiri::CSS.xpath_for directly.
NEVER = :never
# Always use Nokogiri builtin functions whenever possible. This is probably only useful for testing.
ALWAYS = :always
# Only use Nokogiri builtin functions when they will be faster than vanilla XPath. This is
# the behavior chosen when searching for CSS selectors on a Nokogiri document, fragment, or
# node.
OPTIMAL = :optimal
# :nodoc: array of values for validation
VALUES = [NEVER, ALWAYS, OPTIMAL]
end
# Enum to direct XPathVisitor when to tweak the XPath query to suit the nature of the document
# being searched. Note that searches for CSS selectors from a Nokogiri document, fragment, or
# node will choose the correct option automatically.
module DoctypeConfig
# The document being searched is an XML document. This is the default.
XML = :xml
# The document being searched is an HTML4 document.
HTML4 = :html4
# The document being searched is an HTML5 document.
HTML5 = :html5
# :nodoc: array of values for validation
VALUES = [XML, HTML4, HTML5]
end
# The visitor configuration set via the +builtins:+ keyword argument to XPathVisitor.new.
attr_reader :builtins
# The visitor configuration set via the +doctype:+ keyword argument to XPathVisitor.new.
attr_reader :doctype
# The visitor configuration set via the +prefix:+ keyword argument to XPathVisitor.new.
attr_reader :prefix
# The visitor configuration set via the +namespaces:+ keyword argument to XPathVisitor.new.
attr_reader :namespaces
# :call-seq:
# new() → XPathVisitor
# new(builtins:, doctype:) → XPathVisitor
#
# [Parameters]
# - +builtins:+ (BuiltinsConfig) Determine when to use Nokogiri's built-in xpath functions for performance improvements.
# - +doctype:+ (DoctypeConfig) Make document-type-specific accommodations for CSS queries.
#
# [Returns] XPathVisitor
#
def initialize(
builtins: BuiltinsConfig::NEVER,
doctype: DoctypeConfig::XML,
prefix: Nokogiri::XML::XPath::GLOBAL_SEARCH_PREFIX,
namespaces: nil
)
unless BuiltinsConfig::VALUES.include?(builtins)
raise(ArgumentError, "Invalid values #{builtins.inspect} for builtins: keyword parameter")
end
unless DoctypeConfig::VALUES.include?(doctype)
raise(ArgumentError, "Invalid values #{doctype.inspect} for doctype: keyword parameter")
end
@builtins = builtins
@doctype = doctype
@prefix = prefix
@namespaces = namespaces
end
# :call-seq: config() → Hash
#
# [Returns]
# a Hash representing the configuration of the XPathVisitor, suitable for use as
# part of the CSS cache key.
def config
{ builtins: @builtins, doctype: @doctype, prefix: @prefix, namespaces: @namespaces }
end
# :stopdoc:
def visit_function(node)
msg = :"visit_function_#{node.value.first.gsub(/[(]/, "")}"
return send(msg, node) if respond_to?(msg)
case node.value.first
when /^text\(/
"child::text()"
when /^self\(/
"self::#{node.value[1]}"
when /^eq\(/
"position()=#{node.value[1]}"
when /^(nth|nth-of-type)\(/
if node.value[1].is_a?(Nokogiri::CSS::Node) && (node.value[1].type == :NTH)
nth(node.value[1])
else
"position()=#{node.value[1]}"
end
when /^nth-child\(/
if node.value[1].is_a?(Nokogiri::CSS::Node) && (node.value[1].type == :NTH)
nth(node.value[1], child: true)
else
"count(preceding-sibling::*)=#{node.value[1].to_i - 1}"
end
when /^nth-last-of-type\(/
if node.value[1].is_a?(Nokogiri::CSS::Node) && (node.value[1].type == :NTH)
nth(node.value[1], last: true)
else
index = node.value[1].to_i - 1
index == 0 ? "position()=last()" : "position()=last()-#{index}"
end
when /^nth-last-child\(/
if node.value[1].is_a?(Nokogiri::CSS::Node) && (node.value[1].type == :NTH)
nth(node.value[1], last: true, child: true)
else
"count(following-sibling::*)=#{node.value[1].to_i - 1}"
end
when /^(first|first-of-type)\(/
"position()=1"
when /^(last|last-of-type)\(/
"position()=last()"
when /^contains\(/
"contains(.,#{node.value[1]})"
when /^gt\(/
"position()>#{node.value[1]}"
when /^only-child\(/
"last()=1"
when /^comment\(/
"comment()"
when /^has\(/
is_direct = node.value[1].value[0].nil? # e.g. "has(> a)", "has(~ a)", "has(+ a)"
".#{"//" unless is_direct}#{node.value[1].accept(self)}"
else
validate_xpath_function_name(node.value.first)
# xpath function call, let's marshal those arguments
args = ["."]
args += node.value[1..-1].map do |n|
n.is_a?(Nokogiri::CSS::Node) ? n.accept(self) : n
end
"nokogiri:#{node.value.first}#{args.join(",")})"
end
end
def visit_not(node)
child = node.value.first
if :ELEMENT_NAME == child.type
"not(self::#{child.accept(self)})"
else
"not(#{child.accept(self)})"
end
end
def visit_id(node)
node.value.first =~ /^#(.*)$/
"@id='#{Regexp.last_match(1)}'"
end
def visit_attribute_condition(node)
attribute = node.value.first.accept(self)
return attribute if node.value.length == 1
value = node.value.last
value = "'#{value}'" unless /^['"]/.match?(value)
# quoted values - see test_attribute_value_with_quotes in test/css/test_parser.rb
if (value[0] == value[-1]) && %q{"'}.include?(value[0])
str_value = value[1..-2]
if str_value.include?(value[0])
value = 'concat("' + str_value.split('"', -1).join(%q{",'"',"}) + '","")'
end
end
case node.value[1]
when :equal
attribute + "=" + value.to_s
when :not_equal
attribute + "!=" + value.to_s
when :substring_match
"contains(#{attribute},#{value})"
when :prefix_match
"starts-with(#{attribute},#{value})"
when :dash_match
"#{attribute}=#{value} or starts-with(#{attribute},concat(#{value},'-'))"
when :includes
value = value[1..-2] # strip quotes
css_class(attribute, value)
when :suffix_match
"substring(#{attribute},string-length(#{attribute})-string-length(#{value})+1,string-length(#{value}))=#{value}"
else
attribute + " #{node.value[1]} " + value.to_s
end
end
def visit_pseudo_class(node)
if node.value.first.is_a?(Nokogiri::CSS::Node) && (node.value.first.type == :FUNCTION)
node.value.first.accept(self)
else
msg = :"visit_pseudo_class_#{node.value.first.gsub(/[(]/, "")}"
return send(msg, node) if respond_to?(msg)
case node.value.first
when "first" then "position()=1"
when "first-child" then "count(preceding-sibling::*)=0"
when "last" then "position()=last()"
when "last-child" then "count(following-sibling::*)=0"
when "first-of-type" then "position()=1"
when "last-of-type" then "position()=last()"
when "only-child" then "count(preceding-sibling::*)=0 and count(following-sibling::*)=0"
when "only-of-type" then "last()=1"
when "empty" then "not(node())"
when "parent" then "node()"
when "root" then "not(parent::*)"
else
validate_xpath_function_name(node.value.first)
"nokogiri:#{node.value.first}(.)"
end
end
end
def visit_class_condition(node)
css_class("@class", node.value.first)
end
def visit_combinator(node)
if is_of_type_pseudo_class?(node.value.last)
"#{node.value.first&.accept(self)}][#{node.value.last.accept(self)}"
else
"#{node.value.first&.accept(self)} and #{node.value.last.accept(self)}"
end
end
{
"direct_adjacent_selector" => "/following-sibling::*[1]/self::",
"following_selector" => "/following-sibling::",
"descendant_selector" => "//",
"child_selector" => "/",
}.each do |k, v|
class_eval <<~RUBY, __FILE__, __LINE__ + 1
def visit_#{k} node
"\#{node.value.first.accept(self) if node.value.first}#{v}\#{node.value.last.accept(self)}"
end
RUBY
end
def visit_conditional_selector(node)
node.value.first.accept(self) + "[" +
node.value.last.accept(self) + "]"
end
def visit_element_name(node)
if @doctype == DoctypeConfig::HTML5 && html5_element_name_needs_namespace_handling(node)
# HTML5 has namespaces that should be ignored in CSS queries
# https://github.com/sparklemotion/nokogiri/issues/2376
if @builtins == BuiltinsConfig::ALWAYS || (@builtins == BuiltinsConfig::OPTIMAL && Nokogiri.uses_libxml?)
if WILDCARD_NAMESPACES
"*:#{node.value.first}"
else
"*[nokogiri-builtin:local-name-is('#{node.value.first}')]"
end
else
"*[local-name()='#{node.value.first}']"
end
elsif node.value.length == 2 # has a namespace prefix
if node.value.first.nil? # namespace prefix is empty
node.value.last
else
node.value.join(":")
end
elsif node.value.first != "*" && @namespaces&.key?("xmlns")
# apply the default namespace (if one is present) to a non-wildcard selector
"xmlns:#{node.value.first}"
else
node.value.first
end
end
def visit_attrib_name(node)
"@#{node.value.first}"
end
def accept(node)
node.accept(self)
end
private
def validate_xpath_function_name(name)
if name.start_with?("-")
raise Nokogiri::CSS::SyntaxError, "Invalid XPath function name '#{name}'"
end
end
def html5_element_name_needs_namespace_handling(node)
# if there is already a namespace (i.e., it is a prefixed QName), use it as normal
node.value.length == 1 &&
# if this is the wildcard selector "*", use it as normal
node.value.first != "*"
end
def nth(node, options = {})
unless node.value.size == 4
raise(ArgumentError, "expected an+b node to contain 4 tokens, but is #{node.value.inspect}")
end
a, b = read_a_and_positive_b(node.value)
position = if options[:child]
options[:last] ? "(count(following-sibling::*)+1)" : "(count(preceding-sibling::*)+1)"
else
options[:last] ? "(last()-position()+1)" : "position()"
end
if b.zero?
"(#{position} mod #{a})=0"
else
compare = a < 0 ? "<=" : ">="
if a.abs == 1
"#{position}#{compare}#{b}"
else
"(#{position}#{compare}#{b}) and (((#{position}-#{b}) mod #{a.abs})=0)"
end
end
end
def read_a_and_positive_b(values)
op = values[2].strip
if op == "+"
a = values[0].to_i
b = values[3].to_i
elsif op == "-"
a = values[0].to_i
b = a - (values[3].to_i % a)
else
raise ArgumentError, "expected an+b node to have either + or - as the operator, but is #{op.inspect}"
end
[a, b]
end
def is_of_type_pseudo_class?(node) # rubocop:disable Naming/PredicateName
if node.type == :PSEUDO_CLASS
if node.value[0].is_a?(Nokogiri::CSS::Node) && (node.value[0].type == :FUNCTION)
node.value[0].value[0]
else
node.value[0]
end =~ /(nth|first|last|only)-of-type(\()?/
end
end
def css_class(hay, needle)
if @builtins == BuiltinsConfig::ALWAYS || (@builtins == BuiltinsConfig::OPTIMAL && Nokogiri.uses_libxml?)
# use the builtin implementation
"nokogiri-builtin:css-class(#{hay},'#{needle}')"
else
# use only ordinary xpath functions
"contains(concat(' ',normalize-space(#{hay}),' '),' #{needle} ')"
end
end
end
end
end
@@ -0,0 +1,42 @@
# frozen_string_literal: true
module Nokogiri
module Decorators
###
# The Slop decorator implements method missing such that a methods may be
# used instead of XPath or CSS. See Nokogiri.Slop
module Slop
# The default XPath search context for Slop
XPATH_PREFIX = "./"
###
# look for node with +name+. See Nokogiri.Slop
def method_missing(name, *args, &block)
if args.empty?
list = xpath("#{XPATH_PREFIX}#{name.to_s.sub(/^_/, "")}")
elsif args.first.is_a?(Hash)
hash = args.first
if hash[:css]
list = css("#{name}#{hash[:css]}")
elsif hash[:xpath]
conds = Array(hash[:xpath]).join(" and ")
list = xpath("#{XPATH_PREFIX}#{name}[#{conds}]")
end
else
list = xpath(
*CSS.xpath_for("#{name}#{args.first}", prefix: XPATH_PREFIX, cache: false),
)
end
super if list.empty?
list.length == 1 ? list.first : list
end
def respond_to_missing?(name, include_private = false)
list = xpath("#{XPATH_PREFIX}#{name.to_s.sub(/^_/, "")}")
!list.empty?
end
end
end
end
@@ -0,0 +1,57 @@
# encoding: utf-8
# frozen_string_literal: true
module Nokogiri
class EncodingHandler
# Popular encoding aliases not known by all iconv implementations that Nokogiri should support.
USEFUL_ALIASES = {
# alias_name => true_name
"ISO-2022-JP" => "ISO-2022-JP", # only for JRuby tests, this is a no-op in CRuby
"NOKOGIRI-SENTINEL" => "ISO-2022-JP", # indicating the Nokogiri has installed aliases
"Windows-31J" => "CP932", # Windows-31J is the IANA registered name of CP932.
}
class << self
def install_default_aliases
USEFUL_ALIASES.each do |alias_name, name|
EncodingHandler.alias(name, alias_name) if EncodingHandler[alias_name].nil?
end
end
end
# :stopdoc:
if Nokogiri.jruby?
class << self
def [](name)
storage.key?(name) ? new(storage[name]) : nil
end
def alias(name, alias_name)
storage[alias_name] = name
end
def delete(name)
storage.delete(name)
end
def clear_aliases!
storage.clear
end
private
def storage
@storage ||= {}
end
end
def initialize(name)
@name = name
end
attr_reader :name
end
end
end
Nokogiri::EncodingHandler.install_default_aliases
@@ -0,0 +1,32 @@
# frozen_string_literal: true
# load the C or Java extension
begin
# native precompiled gems package shared libraries in <gem_dir>/lib/nokogiri/<ruby_version>
RUBY_VERSION =~ /(\d+\.\d+)/
require_relative "#{Regexp.last_match(1)}/nokogiri"
rescue LoadError => e
if e.message.include?("GLIBC")
warn(<<~EOM)
ERROR: It looks like you're trying to use Nokogiri as a precompiled native gem on a system
with an unsupported version of glibc.
#{e.message}
If that's the case, then please install Nokogiri via the `ruby` platform gem:
gem install nokogiri --platform=ruby
or:
bundle config set force_ruby_platform true
Please visit https://nokogiri.org/tutorials/installing_nokogiri.html for more help.
EOM
raise e
end
# use "require" instead of "require_relative" because non-native gems will place C extension files
# in Gem::BasicSpecification#extension_dir after compilation (during normal installation), which
# is in $LOAD_PATH but not necessarily relative to this file (see #2300)
require "nokogiri/nokogiri"
end
@@ -0,0 +1,15 @@
# frozen_string_literal: true
module Nokogiri
module Gumbo
# The default maximum number of attributes per element.
DEFAULT_MAX_ATTRIBUTES = 400
# The default maximum number of errors for parsing a document or a fragment.
DEFAULT_MAX_ERRORS = 0
# The default maximum depth of the DOM tree produced by parsing a document
# or fragment.
DEFAULT_MAX_TREE_DEPTH = 400
end
end
@@ -0,0 +1,48 @@
# coding: utf-8
# frozen_string_literal: true
require_relative "html4"
module Nokogiri
# Alias for Nokogiri::HTML4
HTML = Nokogiri::HTML4
# :singleton-method: HTML
# :call-seq: HTML(input, url = nil, encoding = nil, options = XML::ParseOptions::DEFAULT_HTML, &block) → Nokogiri::HTML4::Document
#
# Parse HTML. Convenience method for Nokogiri::HTML4::Document.parse
# :nodoc:
define_singleton_method(:HTML, Nokogiri.method(:HTML4))
# 💡 This module/namespace is an alias for Nokogiri::HTML4 as of v1.12.0. Before v1.12.0,
# Nokogiri::HTML4 did not exist, and this was the module/namespace for all HTML-related
# classes.
module HTML
# 💡 This class is an alias for Nokogiri::HTML4::Document as of v1.12.0.
class Document < Nokogiri::XML::Document
end
# 💡 This class is an alias for Nokogiri::HTML4::DocumentFragment as of v1.12.0.
class DocumentFragment < Nokogiri::XML::DocumentFragment
end
# 💡 This class is an alias for Nokogiri::HTML4::Builder as of v1.12.0.
class Builder < Nokogiri::XML::Builder
end
module SAX
# 💡 This class is an alias for Nokogiri::HTML4::SAX::Parser as of v1.12.0.
class Parser < Nokogiri::XML::SAX::Parser
end
# 💡 This class is an alias for Nokogiri::HTML4::SAX::ParserContext as of v1.12.0.
class ParserContext < Nokogiri::XML::SAX::ParserContext
end
# 💡 This class is an alias for Nokogiri::HTML4::SAX::PushParser as of v1.12.0.
class PushParser
end
end
end
end
@@ -0,0 +1,42 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
class << self
# Convenience method for Nokogiri::HTML4::Document.parse
def HTML4(...)
Nokogiri::HTML4::Document.parse(...)
end
end
# Since v1.12.0
#
# 💡 Before v1.12.0, Nokogiri::HTML4 did not exist, and Nokogiri::HTML was the module/namespace
# for parsing HTML.
module HTML4
class << self
# Convenience method for Nokogiri::HTML4::Document.parse
def parse(...)
Document.parse(...)
end
# Convenience method for Nokogiri::HTML4::DocumentFragment.parse
def fragment(...)
HTML4::DocumentFragment.parse(...)
end
end
# Instance of Nokogiri::HTML4::EntityLookup
NamedCharacters = EntityLookup.new
end
end
require_relative "html4/entity_lookup"
require_relative "html4/document"
require_relative "html4/document_fragment"
require_relative "html4/encoding_reader"
require_relative "html4/sax/parser_context"
require_relative "html4/sax/parser"
require_relative "html4/sax/push_parser"
require_relative "html4/element_description"
require_relative "html4/element_description_defaults"
@@ -0,0 +1,37 @@
# frozen_string_literal: true
module Nokogiri
module HTML4
###
# Nokogiri HTML builder is used for building HTML documents. It is very
# similar to the Nokogiri::XML::Builder. In fact, you should go read the
# documentation for Nokogiri::XML::Builder before reading this
# documentation.
#
# == Synopsis:
#
# Create an HTML document with a body that has an onload attribute, and a
# span tag with a class of "bold" that has content of "Hello world".
#
# builder = Nokogiri::HTML4::Builder.new do |doc|
# doc.html {
# doc.body(:onload => 'some_func();') {
# doc.span.bold {
# doc.text "Hello world"
# }
# }
# }
# end
# puts builder.to_html
#
# The HTML builder inherits from the XML builder, so make sure to read the
# Nokogiri::XML::Builder documentation.
class Builder < Nokogiri::XML::Builder
###
# Convert the builder to HTML
def to_html
@doc.to_html
end
end
end
end
@@ -0,0 +1,235 @@
# coding: utf-8
# frozen_string_literal: true
require "pathname"
module Nokogiri
module HTML4
class Document < Nokogiri::XML::Document
###
# Get the meta tag encoding for this document. If there is no meta tag,
# then nil is returned.
def meta_encoding
if (meta = at_xpath("//meta[@charset]"))
meta[:charset]
elsif (meta = meta_content_type)
meta["content"][/charset\s*=\s*([\w-]+)/i, 1]
end
end
###
# Set the meta tag encoding for this document.
#
# If an meta encoding tag is already present, its content is
# replaced with the given text.
#
# Otherwise, this method tries to create one at an appropriate
# place supplying head and/or html elements as necessary, which
# is inside a head element if any, and before any text node or
# content element (typically <body>) if any.
#
# The result when trying to set an encoding that is different
# from the document encoding is undefined.
#
# Beware in CRuby, that libxml2 automatically inserts a meta tag
# into a head element.
def meta_encoding=(encoding)
if (meta = meta_content_type)
meta["content"] = format("text/html; charset=%s", encoding)
encoding
elsif (meta = at_xpath("//meta[@charset]"))
meta["charset"] = encoding
else
meta = XML::Node.new("meta", self)
if (dtd = internal_subset) && dtd.html5_dtd?
meta["charset"] = encoding
else
meta["http-equiv"] = "Content-Type"
meta["content"] = format("text/html; charset=%s", encoding)
end
if (head = at_xpath("//head"))
head.prepend_child(meta)
else
set_metadata_element(meta)
end
encoding
end
end
def meta_content_type
xpath("//meta[@http-equiv and boolean(@content)]").find do |node|
node["http-equiv"] =~ /\AContent-Type\z/i
end
end
private :meta_content_type
###
# Get the title string of this document. Return nil if there is
# no title tag.
def title
(title = at_xpath("//title")) && title.inner_text
end
###
# Set the title string of this document.
#
# If a title element is already present, its content is replaced
# with the given text.
#
# Otherwise, this method tries to create one at an appropriate
# place supplying head and/or html elements as necessary, which
# is inside a head element if any, right after a meta
# encoding/charset tag if any, and before any text node or
# content element (typically <body>) if any.
def title=(text)
tnode = XML::Text.new(text, self)
if (title = at_xpath("//title"))
title.children = tnode
return text
end
title = XML::Node.new("title", self) << tnode
if (head = at_xpath("//head"))
head << title
elsif (meta = at_xpath("//meta[@charset]") || meta_content_type)
# better put after charset declaration
meta.add_next_sibling(title)
else
set_metadata_element(title)
end
end
def set_metadata_element(element) # rubocop:disable Naming/AccessorMethodName
if (head = at_xpath("//head"))
head << element
elsif (html = at_xpath("//html"))
head = html.prepend_child(XML::Node.new("head", self))
head.prepend_child(element)
elsif (first = children.find do |node|
case node
when XML::Element, XML::Text
true
end
end)
# We reach here only if the underlying document model
# allows <html>/<head> elements to be omitted and does not
# automatically supply them.
first.add_previous_sibling(element)
else
html = add_child(XML::Node.new("html", self))
head = html.add_child(XML::Node.new("head", self))
head.prepend_child(element)
end
end
private :set_metadata_element
####
# Serialize Node using +options+. Save options can also be set using a block.
#
# See also Nokogiri::XML::Node::SaveOptions and Node@Serialization+and+Generating+Output.
#
# These two statements are equivalent:
#
# node.serialize(:encoding => 'UTF-8', :save_with => FORMAT | AS_XML)
#
# or
#
# node.serialize(:encoding => 'UTF-8') do |config|
# config.format.as_xml
# end
#
def serialize(options = {})
options[:save_with] ||= XML::Node::SaveOptions::DEFAULT_HTML
super
end
####
# Create a Nokogiri::XML::DocumentFragment from +tags+
def fragment(tags = nil)
DocumentFragment.new(self, tags, root)
end
# :call-seq:
# xpath_doctype() → Nokogiri::CSS::XPathVisitor::DoctypeConfig
#
# [Returns] The document type which determines CSS-to-XPath translation.
#
# See XPathVisitor for more information.
def xpath_doctype
Nokogiri::CSS::XPathVisitor::DoctypeConfig::HTML4
end
class << self
# :call-seq:
# parse(input) { |options| ... } => Nokogiri::HTML4::Document
# parse(input, url:, encoding:, options:) => Nokogiri::HTML4::Document
#
# Parse \HTML4 input from a String or IO object, and return a new HTML4::Document.
#
# [Required Parameters]
# - +input+ (String | IO) The content to be parsed.
#
# [Optional Keyword Arguments]
# - +url:+ (String) The base URI for this document.
#
# - +encoding:+ (String) The name of the encoding that should be used when processing the
# document. When not provided, the encoding will be determined based on the document
# content.
#
# - +options:+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
# behaviors during parsing. See ParseOptions for more information. The default value is
# +ParseOptions::DEFAULT_HTML+.
#
# [Yields]
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
# can be configured before parsing. See Nokogiri::XML::ParseOptions for more information.
#
# [Returns] Nokogiri::HTML4::Document
def parse(
input,
url_ = nil, encoding_ = nil, options_ = XML::ParseOptions::DEFAULT_HTML,
url: url_, encoding: encoding_, options: options_
)
options = Nokogiri::XML::ParseOptions.new(options) if Integer === options
yield options if block_given?
url ||= input.respond_to?(:path) ? input.path : nil
if input.respond_to?(:encoding)
unless input.encoding == Encoding::ASCII_8BIT
encoding ||= input.encoding.name
end
end
if input.respond_to?(:read)
if input.is_a?(Pathname)
# resolve the Pathname to the file and open it as an IO object, see #2110
input = input.expand_path.open
url ||= input.path
end
unless encoding
input = EncodingReader.new(input)
begin
return read_io(input, url, encoding, options.to_i)
rescue EncodingReader::EncodingFound => e
encoding = e.found_encoding
end
end
return read_io(input, url, encoding, options.to_i)
end
# read_memory pukes on empty docs
if input.nil? || input.empty?
return encoding ? new.tap { |i| i.encoding = encoding } : new
end
encoding ||= EncodingReader.detect_encoding(input)
read_memory(input, url, encoding, options.to_i)
end
end
end
end
end
@@ -0,0 +1,166 @@
# frozen_string_literal: true
module Nokogiri
module HTML4
class DocumentFragment < Nokogiri::XML::DocumentFragment
#
# :call-seq:
# parse(input) { |options| ... } → HTML4::DocumentFragment
# parse(input, encoding:, options:) { |options| ... } → HTML4::DocumentFragment
#
# Parse \HTML4 fragment input from a String, and return a new HTML4::DocumentFragment. This
# method creates a new, empty HTML4::Document to contain the fragment.
#
# [Required Parameters]
# - +input+ (String | IO) The content to be parsed.
#
# [Optional Keyword Arguments]
# - +encoding:+ (String) The name of the encoding that should be used when processing the
# document. When not provided, the encoding will be determined based on the document
# content.
#
# - +options:+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
# behaviors during parsing. See ParseOptions for more information. The default value is
# +ParseOptions::DEFAULT_HTML+.
#
# [Yields]
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
# can be configured before parsing. See ParseOptions for more information.
#
# [Returns] HTML4::DocumentFragment
#
# *Example:* Parsing a string
#
# fragment = HTML4::DocumentFragment.parse("<div>Hello World</div>")
#
# *Example:* Parsing an IO
#
# fragment = File.open("fragment.html") do |file|
# HTML4::DocumentFragment.parse(file)
# end
#
# *Example:* Specifying encoding
#
# fragment = HTML4::DocumentFragment.parse(input, encoding: "EUC-JP")
#
# *Example:* Setting parse options dynamically
#
# HTML4::DocumentFragment.parse("<div>Hello World") do |options|
# options.huge.pedantic
# end
#
def self.parse(
input,
encoding_ = nil, options_ = XML::ParseOptions::DEFAULT_HTML,
encoding: encoding_, options: options_,
&block
)
# TODO: this method should take a context node.
doc = HTML4::Document.new
if input.respond_to?(:read)
# Handle IO-like objects (IO, File, StringIO, etc.)
# The _read_ method of these objects doesn't accept an +encoding+ parameter.
# Encoding is usually set when the IO object is created or opened,
# or by using the _set_encoding_ method.
#
# 1. If +encoding+ is provided and the object supports _set_encoding_,
# set the encoding before reading.
# 2. Read the content from the IO-like object.
#
# Note: After reading, the content's encoding will be:
# - The encoding set by _set_encoding_ if it was called
# - The default encoding of the IO object otherwise
#
# For StringIO specifically, _set_encoding_ affects only the internal string,
# not how the data is read out.
input.set_encoding(encoding) if encoding && input.respond_to?(:set_encoding)
input = input.read
end
encoding ||= if input.respond_to?(:encoding)
encoding = input.encoding
if encoding == ::Encoding::ASCII_8BIT
"UTF-8"
else
encoding.name
end
else
"UTF-8"
end
doc.encoding = encoding
new(doc, input, options: options, &block)
end
#
# :call-seq:
# new(document) { |options| ... } → HTML4::DocumentFragment
# new(document, input) { |options| ... } → HTML4::DocumentFragment
# new(document, input, context:, options:) { |options| ... } → HTML4::DocumentFragment
#
# Parse \HTML4 fragment input from a String, and return a new HTML4::DocumentFragment.
#
# 💡 It's recommended to use either HTML4::DocumentFragment.parse or XML::Node#parse rather
# than call this method directly.
#
# [Required Parameters]
# - +document+ (HTML4::Document) The parent document to associate the returned fragment with.
#
# [Optional Parameters]
# - +input+ (String) The content to be parsed.
#
# [Optional Keyword Arguments]
# - +context:+ (Nokogiri::XML::Node) The <b>context node</b> for the subtree created. See
# below for more information.
#
# - +options:+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
# behaviors during parsing. See ParseOptions for more information. The default value is
# +ParseOptions::DEFAULT_HTML+.
#
# [Yields]
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
# can be configured before parsing. See ParseOptions for more information.
#
# [Returns] HTML4::DocumentFragment
#
# === Context \Node
#
# If a context node is specified using +context:+, then the fragment will be created by
# calling XML::Node#parse on that node, so the parser will behave as if that Node is the
# parent of the fragment subtree.
#
def initialize(
document, input = nil,
context_ = nil, options_ = XML::ParseOptions::DEFAULT_HTML,
context: context_, options: options_
) # rubocop:disable Lint/MissingSuper
return self unless input
options = Nokogiri::XML::ParseOptions.new(options) if Integer === options
@parse_options = options
yield options if block_given?
if context
preexisting_errors = document.errors.dup
node_set = context.parse("<div>#{input}</div>", options)
node_set.first.children.each { |child| child.parent = self } unless node_set.empty?
self.errors = document.errors - preexisting_errors
else
# This is a horrible hack, but I don't care
path = if /^\s*?<body/i.match?(input)
"/html/body"
else
"/html/body/node()"
end
temp_doc = HTML4::Document.parse("<html><body>#{input}", nil, document.encoding, options)
temp_doc.xpath(path).each { |child| child.parent = self }
self.errors = temp_doc.errors
end
children
end
end
end
end
@@ -0,0 +1,25 @@
# frozen_string_literal: true
module Nokogiri
module HTML4
class ElementDescription
###
# Is this element a block element?
def block?
!inline?
end
###
# Convert this description to a string
def to_s
"#{name}: #{description}"
end
###
# Inspection information
def inspect
"#<#{self.class.name}: #{name} #{description}>"
end
end
end
end
@@ -0,0 +1,121 @@
# frozen_string_literal: true
module Nokogiri
module HTML4
# Libxml2's parser has poor support for encoding detection. First, it does not recognize the
# HTML5 style meta charset declaration. Secondly, even if it successfully detects an encoding
# hint, it does not re-decode or re-parse the preceding part which may be garbled.
#
# EncodingReader aims to perform advanced encoding detection beyond what Libxml2 does, and to
# emulate rewinding of a stream and make Libxml2 redo parsing from the start when an encoding
# hint is found.
# :nodoc: all
class EncodingReader
class EncodingFound < StandardError
attr_reader :found_encoding
def initialize(encoding)
@found_encoding = encoding
super(format("encoding found: %s", encoding))
end
end
class SAXHandler < Nokogiri::XML::SAX::Document
attr_reader :encoding
def initialize
@encoding = nil
super
end
def start_element(name, attrs = [])
return unless name == "meta"
attr = Hash[attrs]
(charset = attr["charset"]) &&
(@encoding = charset)
(http_equiv = attr["http-equiv"]) &&
http_equiv.match(/\AContent-Type\z/i) &&
(content = attr["content"]) &&
(m = content.match(/;\s*charset\s*=\s*([\w-]+)/)) &&
(@encoding = m[1])
end
end
class JumpSAXHandler < SAXHandler
def initialize(jumptag)
@jumptag = jumptag
super()
end
def start_element(name, attrs = [])
super
throw(@jumptag, @encoding) if @encoding
throw(@jumptag, nil) if /\A(?:div|h1|img|p|br)\z/.match?(name)
end
end
def self.detect_encoding(chunk)
(m = chunk.match(/\A(<\?xml[ \t\r\n][^>]*>)/)) &&
(return Nokogiri.XML(m[1]).encoding)
if Nokogiri.jruby?
(m = chunk.match(/(<meta\s)(.*)(charset\s*=\s*([\w-]+))(.*)/i)) &&
(return m[4])
catch(:encoding_found) do
Nokogiri::HTML4::SAX::Parser.new(JumpSAXHandler.new(:encoding_found)).parse(chunk)
nil
end
else
handler = SAXHandler.new
parser = Nokogiri::HTML4::SAX::PushParser.new(handler)
begin
parser << chunk
rescue
Nokogiri::SyntaxError
end
handler.encoding
end
end
def initialize(io)
@io = io
@firstchunk = nil
@encoding_found = nil
end
# This method is used by the C extension so that
# Nokogiri::HTML4::Document#read_io() does not leak memory when
# EncodingFound is raised.
attr_reader :encoding_found
def read(len)
# no support for a call without len
unless @firstchunk
(@firstchunk = @io.read(len)) || return
# This implementation expects that the first call from
# htmlReadIO() is made with a length long enough (~1KB) to
# achieve advanced encoding detection.
if (encoding = EncodingReader.detect_encoding(@firstchunk))
# The first chunk is stored for the next read in retry.
raise @encoding_found = EncodingFound.new(encoding)
end
end
@encoding_found = nil
ret = @firstchunk.slice!(0, len)
if (len -= ret.length) > 0
(rest = @io.read(len)) && ret << (rest)
end
if ret.empty?
nil
else
ret
end
end
end
end
end
@@ -0,0 +1,15 @@
# frozen_string_literal: true
module Nokogiri
module HTML4
class EntityDescription < Struct.new(:value, :name, :description); end
class EntityLookup
###
# Look up entity with +name+
def [](name)
(val = get(name)) && val.value
end
end
end
end
@@ -0,0 +1,48 @@
# frozen_string_literal: true
module Nokogiri
module HTML4
###
# Nokogiri provides a SAX parser to process HTML4 which will provide HTML recovery
# ("autocorrection") features.
#
# See Nokogiri::HTML4::SAX::Parser for a basic example of using a SAX parser with HTML.
#
# For more information on SAX parsers, see Nokogiri::XML::SAX
#
module SAX
###
# This parser is a SAX style parser that reads its input as it deems necessary. The parser
# takes a Nokogiri::XML::SAX::Document, an optional encoding, then given an HTML input, sends
# messages to the Nokogiri::XML::SAX::Document.
#
# ⚠ This is an HTML4 parser and so may not support some HTML5 features and behaviors.
#
# Here is a basic usage example:
#
# class MyHandler < Nokogiri::XML::SAX::Document
# def start_element name, attributes = []
# puts "found a #{name}"
# end
# end
#
# parser = Nokogiri::HTML4::SAX::Parser.new(MyHandler.new)
#
# # Hand an IO object to the parser, which will read the HTML from the IO.
# File.open(path_to_html) do |f|
# parser.parse(f)
# end
#
# For more information on \SAX parsers, see Nokogiri::XML::SAX or the parent class
# Nokogiri::XML::SAX::Parser.
#
# Also see Nokogiri::XML::SAX::Document for the available events.
#
class Parser < Nokogiri::XML::SAX::Parser
# this class inherits its behavior from Nokogiri::XML::SAX::Parser, but note that superclass
# uses Nokogiri::ClassResolver to use HTML4::SAX::ParserContext as the context class for
# this class, which is where the real behavioral differences are implemented.
end
end
end
end
@@ -0,0 +1,15 @@
# frozen_string_literal: true
module Nokogiri
module HTML4
module SAX
###
# Context object to invoke the HTML4 SAX parser on the SAX::Document handler.
#
# 💡 This class is usually not instantiated by the user. Use Nokogiri::HTML4::SAX::Parser
# instead.
class ParserContext < Nokogiri::XML::SAX::ParserContext
end
end
end
end
@@ -0,0 +1,37 @@
# frozen_string_literal: true
module Nokogiri
module HTML4
module SAX
class PushParser
# The Nokogiri::HTML4::SAX::Document on which the PushParser will be
# operating
attr_accessor :document
def initialize(doc = HTML4::SAX::Document.new, file_name = nil, encoding = "UTF-8")
@document = doc
@encoding = encoding
@sax_parser = HTML4::SAX::Parser.new(doc, @encoding)
## Create our push parser context
initialize_native(@sax_parser, file_name, encoding)
end
###
# Write a +chunk+ of HTML to the PushParser. Any callback methods
# that can be called will be called immediately.
def write(chunk, last_chunk = false)
native_write(chunk, last_chunk)
end
alias_method :<<, :write
###
# Finish the parsing. This method is only necessary for
# Nokogiri::HTML4::SAX::Document#end_document to be called.
def finish
write("", true)
end
end
end
end
end
@@ -0,0 +1,368 @@
# coding: utf-8
# frozen_string_literal: true
# This file includes code from the Nokogumbo project, whose license follows.
#
# Copyright 2013-2021 Sam Ruby, Stephen Checkoway
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
require_relative "html5/document"
require_relative "html5/document_fragment"
require_relative "html5/node"
require_relative "html5/builder"
module Nokogiri
# Convenience method for Nokogiri::HTML5::Document.parse
def self.HTML5(...)
Nokogiri::HTML5::Document.parse(...)
end
# == Usage
#
# Parse an HTML5 document:
#
# doc = Nokogiri.HTML5(input)
#
# Parse an HTML5 fragment:
#
# fragment = Nokogiri::HTML5.fragment(input)
#
# ⚠ HTML5 functionality is not available when running JRuby.
#
# == Parsing options
#
# The document and fragment parsing methods support options that are different from
# Nokogiri::HTML4::Document or Nokogiri::XML::Document.
#
# - <tt>Nokogiri.HTML5(input, url:, encoding:, **parse_options)</tt>
# - <tt>Nokogiri::HTML5.parse(input, url:, encoding:, **parse_options)</tt>
# - <tt>Nokogiri::HTML5::Document.parse(input, url:, encoding:, **parse_options)</tt>
# - <tt>Nokogiri::HTML5.fragment(input, encoding:, **parse_options)</tt>
# - <tt>Nokogiri::HTML5::DocumentFragment.parse(input, encoding:, **parse_options)</tt>
#
# The four currently supported parse options are
#
# - +max_errors:+ (Integer, default 0) Maximum number of parse errors to report in HTML5::Document#errors.
# - +max_tree_depth:+ (Integer, default +Nokogiri::Gumbo::DEFAULT_MAX_TREE_DEPTH+) Maximum tree depth to parse.
# - +max_attributes:+ (Integer, default +Nokogiri::Gumbo::DEFAULT_MAX_ATTRIBUTES+) Maximum number of attributes to parse per element.
# - +parse_noscript_content_as_text:+ (Boolean, default false) When enabled, parse +noscript+ tag content as text, mimicking the behavior of web browsers.
#
# These options are explained in the following sections.
#
# === Error reporting: +max_errors:+
#
# Nokogiri contains an experimental HTML5 parse error reporting facility. By default, no parse
# errors are reported but this can be configured by passing the +:max_errors+ option to
# HTML5.parse or HTML5.fragment.
#
# For example, this script:
#
# doc = Nokogiri::HTML5.parse('<span/>Hi there!</span foo=bar />', max_errors: 10)
# doc.errors.each do |err|
# puts(err)
# end
#
# Emits:
#
# 1:1: ERROR: Expected a doctype token
# <span/>Hi there!</span foo=bar />
# ^
# 1:1: ERROR: Start tag of nonvoid HTML element ends with '/>', use '>'.
# <span/>Hi there!</span foo=bar />
# ^
# 1:17: ERROR: End tag ends with '/>', use '>'.
# <span/>Hi there!</span foo=bar />
# ^
# 1:17: ERROR: End tag contains attributes.
# <span/>Hi there!</span foo=bar />
# ^
#
# Using <tt>max_errors: -1</tt> results in an unlimited number of errors being returned.
#
# The errors returned by HTML5::Document#errors are instances of Nokogiri::XML::SyntaxError.
#
# The {HTML standard}[https://html.spec.whatwg.org/multipage/parsing.html#parse-errors] defines a
# number of standard parse error codes. These error codes only cover the "tokenization" stage of
# parsing HTML. The parse errors in the "tree construction" stage do not have standardized error
# codes (yet).
#
# As a convenience to Nokogiri users, the defined error codes are available
# via Nokogiri::XML::SyntaxError#str1 method.
#
# doc = Nokogiri::HTML5.parse('<span/>Hi there!</span foo=bar />', max_errors: 10)
# doc.errors.each do |err|
# puts("#{err.line}:#{err.column}: #{err.str1}")
# end
# doc = Nokogiri::HTML5.parse('<span/>Hi there!</span foo=bar />',
# # => 1:1: generic-parser
# # 1:1: non-void-html-element-start-tag-with-trailing-solidus
# # 1:17: end-tag-with-trailing-solidus
# # 1:17: end-tag-with-attributes
#
# Note that the first error is +generic-parser+ because it's an error from the tree construction
# stage and doesn't have a standardized error code.
#
# For the purposes of semantic versioning, the error messages, error locations, and error codes
# are not part of Nokogiri's public API. That is, these are subject to change without Nokogiri's
# major version number changing. These may be stabilized in the future.
#
# === Maximum tree depth: +max_tree_depth:+
#
# The maximum depth of the DOM tree parsed by the various parsing methods is configurable by the
# +:max_tree_depth+ option. If the depth of the tree would exceed this limit, then an
# +ArgumentError+ is thrown.
#
# This limit (which defaults to +Nokogiri::Gumbo::DEFAULT_MAX_TREE_DEPTH+) can be removed by
# giving the option <tt>max_tree_depth: -1</tt>.
#
# html = '<!DOCTYPE html>' + '<div>' * 1000
# doc = Nokogiri.HTML5(html)
# # raises ArgumentError: Document tree depth limit exceeded
# doc = Nokogiri.HTML5(html, max_tree_depth: -1)
#
# === Attribute limit per element: +max_attributes:+
#
# The maximum number of attributes per DOM element is configurable by the +:max_attributes+
# option. If a given element would exceed this limit, then an +ArgumentError+ is thrown.
#
# This limit (which defaults to +Nokogiri::Gumbo::DEFAULT_MAX_ATTRIBUTES+) can be removed by
# giving the option <tt>max_attributes: -1</tt>.
#
# html = '<!DOCTYPE html><div ' + (1..1000).map { |x| "attr-#{x}" }.join(' # ') + '>'
# # "<!DOCTYPE html><div attr-1 attr-2 attr-3 ... attr-1000>"
# doc = Nokogiri.HTML5(html)
# # raises ArgumentError: Attributes per element limit exceeded
#
# doc = Nokogiri.HTML5(html, max_attributes: -1)
# # parses successfully
#
# === Parse +noscript+ elements' content as text: +parse_noscript_content_as_text:+
#
# By default, the content of +noscript+ elements is parsed as HTML elements. Browsers that
# support scripting parse the content of +noscript+ elements as raw text.
#
# The +:parse_noscript_content_as_text+ option causes Nokogiri to parse the content of +noscript+
# elements as a single text node.
#
# html = "<!DOCTYPE html><noscript><meta charset='UTF-8'><link rel=stylesheet href=!></noscript>"
# doc = Nokogiri::HTML5.parse(html, parse_noscript_content_as_text: true)
# pp doc.at_xpath("/html/head/noscript")
# # => #(Element:0x878c {
# # name = "noscript",
# # children = [ #(Text "<meta charset='UTF-8'><link rel=stylesheet href=!>")]
# # })
#
# In contrast, <tt>parse_noscript_content_as_text: false</tt> (the default) causes the +noscript+
# element in the previous example to have two children, a +meta+ element and a +link+ element.
#
# doc = Nokogiri::HTML5.parse(html)
# puts doc.at_xpath("/html/head/noscript")
# # => #(Element:0x96b4 {
# # name = "noscript",
# # children = [
# # #(Element:0x97e0 { name = "meta", attribute_nodes = [ #(Attr:0x990c { name = "charset", value = "UTF-8" })] }),
# # #(Element:0x9b00 {
# # name = "link",
# # attribute_nodes = [
# # #(Attr:0x9c2c { name = "rel", value = "stylesheet" }),
# # #(Attr:0x9dd0 { name = "href", value = "!" })]
# # })]
# # })
#
# == HTML Serialization
#
# After parsing HTML, it may be serialized using any of the Nokogiri::XML::Node serialization
# methods. In particular, XML::Node#serialize, XML::Node#to_html, and XML::Node#to_s will
# serialize a given node and its children. (This is the equivalent of JavaScript's
# +Element.outerHTML+.) Similarly, XML::Node#inner_html will serialize the children of a given
# node. (This is the equivalent of JavaScript's +Element.innerHTML+.)
#
# doc = Nokogiri::HTML5("<!DOCTYPE html><span>Hello world!</span>")
# puts doc.serialize
# # => <!DOCTYPE html><html><head></head><body><span>Hello world!</span></body></html>
#
# Due to quirks in how HTML is parsed and serialized, it's possible for a DOM tree to be
# serialized and then re-parsed, resulting in a different DOM. Mostly, this happens with DOMs
# produced from invalid HTML. Unfortunately, even valid HTML may not survive serialization and
# re-parsing.
#
# In particular, a newline at the start of +pre+, +listing+, and +textarea+
# elements is ignored by the parser.
#
# doc = Nokogiri::HTML5(<<-EOF)
# <!DOCTYPE html>
# <pre>
# Content</pre>
# EOF
# puts doc.at('/html/body/pre').serialize
# # => <pre>Content</pre>
#
# In this case, the original HTML is semantically equivalent to the serialized version. If the
# +pre+, +listing+, or +textarea+ content starts with two newlines, the first newline will be
# stripped on the first parse and the second newline will be stripped on the second, leading to
# semantically different DOMs. Passing the parameter <tt>preserve_newline: true</tt> will cause
# two or more newlines to be preserved. (A single leading newline will still be removed.)
#
# doc = Nokogiri::HTML5(<<-EOF)
# <!DOCTYPE html>
# <listing>
#
# Content</listing>
# EOF
# puts doc.at('/html/body/listing').serialize(preserve_newline: true)
# # => <listing>
# #
# # Content</listing>
#
# == Encodings
#
# Nokogiri always parses HTML5 using {UTF-8}[https://en.wikipedia.org/wiki/UTF-8]; however, the
# encoding of the input can be explicitly selected via the optional +encoding+ parameter. This is
# most useful when the input comes not from a string but from an IO object.
#
# When serializing a document or node, the encoding of the output string can be specified via the
# +:encoding+ options. Characters that cannot be encoded in the selected encoding will be encoded
# as {HTML numeric
# entities}[https://en.wikipedia.org/wiki/List_of_XML_and_HTML_character_entity_references].
#
# frag = Nokogiri::HTML5.fragment('<span>아는 길도 물어가라</span>')
# html = frag.serialize(encoding: 'US-ASCII')
# puts html
# # => <span>&#xc544;&#xb294; &#xae38;&#xb3c4; &#xbb3c;&#xc5b4;&#xac00;&#xb77c;</span>
#
# frag = Nokogiri::HTML5.fragment(html)
# puts frag.serialize
# # => <span>아는 길도 물어가라</span>
#
# (There's a {bug}[https://bugs.ruby-lang.org/issues/15033] in all current versions of Ruby that
# can cause the entity encoding to fail. Of the mandated supported encodings for HTML, the only
# encoding I'm aware of that has this bug is <tt>'ISO-2022-JP'</tt>. We recommend avoiding this
# encoding.)
#
# == Notes
#
# * The Nokogiri::HTML5.fragment function takes a String or IO and parses it as a HTML5 document
# in a +body+ context. As a result, the +html+, +head+, and +body+ elements are removed from
# this document, and any children of these elements that remain are returned as a
# Nokogiri::HTML5::DocumentFragment; but you can pass in a different context (e.g., "html" to
# get +head+ and +body+ tags in the result).
#
# * The Nokogiri::HTML5.parse function takes a String or IO and passes it to the
# <code>gumbo_parse_with_options</code> method, using the default options. The resulting Gumbo
# parse tree is then walked.
#
# * Instead of uppercase element names, lowercase element names are produced.
#
# * Instead of returning +unknown+ as the element name for unknown tags, the original tag name is
# returned verbatim.
#
# Since v1.12.0
module HTML5
class << self
# Convenience method for Nokogiri::HTML5::Document.parse
def parse(...)
Document.parse(...)
end
# Convenience method for Nokogiri::HTML5::DocumentFragment.parse
def fragment(...)
DocumentFragment.parse(...)
end
# :nodoc:
def read_and_encode(string, encoding)
# Read the string with the given encoding.
if string.respond_to?(:read)
string = if encoding.nil?
string.read
else
string.read(encoding: encoding)
end
else
# Otherwise the string has the given encoding.
string = string.to_s
if encoding
string = string.dup
string.force_encoding(encoding)
end
end
# convert to UTF-8
if string.encoding != Encoding::UTF_8
string = reencode(string)
end
string
end
private
# Charset sniffing is a complex and controversial topic that understandably isn't done _by
# default_ by the Ruby Net::HTTP library. This being said, it is a very real problem for
# consumers of HTML as the default for HTML is iso-8859-1, most "good" producers use utf-8, and
# the Gumbo parser *only* supports utf-8.
#
# Accordingly, Nokogiri::HTML4::Document.parse provides limited encoding detection. Following
# this lead, Nokogiri::HTML5 attempts to do likewise, while attempting to more closely follow
# the HTML5 standard.
#
# http://bugs.ruby-lang.org/issues/2567
# http://www.w3.org/TR/html5/syntax.html#determining-the-character-encoding
#
def reencode(body, content_type = nil)
if body.encoding == Encoding::ASCII_8BIT
encoding = nil
# look for a Byte Order Mark (BOM)
initial_bytes = body[0..2].bytes
if initial_bytes[0..2] == [0xEF, 0xBB, 0xBF]
encoding = Encoding::UTF_8
elsif initial_bytes[0..1] == [0xFE, 0xFF]
encoding = Encoding::UTF_16BE
elsif initial_bytes[0..1] == [0xFF, 0xFE]
encoding = Encoding::UTF_16LE
end
# look for a charset in a content-encoding header
if content_type
encoding ||= content_type[/charset=["']?(.*?)($|["';\s])/i, 1]
end
# look for a charset in a meta tag in the first 1024 bytes
unless encoding
data = body[0..1023].gsub(/<!--.*?(-->|\Z)/m, "")
data.scan(/<meta.*?>/im).each do |meta|
encoding ||= meta[/charset=["']?([^>]*?)($|["'\s>])/im, 1]
end
end
# if all else fails, default to the official default encoding for HTML
encoding ||= Encoding::ISO_8859_1
# change the encoding to match the detected or inferred encoding
body = body.dup
begin
body.force_encoding(encoding)
rescue ArgumentError
body.force_encoding(Encoding::ISO_8859_1)
end
end
body.encode(Encoding::UTF_8)
end
end
end
end
require_relative "gumbo"
@@ -0,0 +1,40 @@
# frozen_string_literal: true
module Nokogiri
module HTML5
###
# Nokogiri HTML5 builder is used for building HTML documents. It is very similar to the
# Nokogiri::XML::Builder. In fact, you should go read the documentation for
# Nokogiri::XML::Builder before reading this documentation.
#
# The construction behavior is identical to HTML4::Builder, but HTML5 documents implement the
# [HTML5 standard's serialization
# algorithm](https://www.w3.org/TR/2008/WD-html5-20080610/serializing.html).
#
# == Synopsis:
#
# Create an HTML5 document with a body that has an onload attribute, and a
# span tag with a class of "bold" that has content of "Hello world".
#
# builder = Nokogiri::HTML5::Builder.new do |doc|
# doc.html {
# doc.body(:onload => 'some_func();') {
# doc.span.bold {
# doc.text "Hello world"
# }
# }
# }
# end
# puts builder.to_html
#
# The HTML5 builder inherits from the XML builder, so make sure to read the
# Nokogiri::XML::Builder documentation.
class Builder < Nokogiri::XML::Builder
###
# Convert the builder to HTML
def to_html
@doc.to_html
end
end
end
end
@@ -0,0 +1,199 @@
# coding: utf-8
# frozen_string_literal: true
#
# Copyright 2013-2021 Sam Ruby, Stephen Checkoway
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
require_relative "../html4/document"
module Nokogiri
module HTML5
# Enum for the HTML5 parser quirks mode values. Values returned by HTML5::Document#quirks_mode
#
# See https://dom.spec.whatwg.org/#concept-document-quirks for more information on HTML5 quirks
# mode.
#
# Since v1.14.0
module QuirksMode
NO_QUIRKS = 0 # The document was parsed in "no-quirks" mode
QUIRKS = 1 # The document was parsed in "quirks" mode
LIMITED_QUIRKS = 2 # The document was parsed in "limited-quirks" mode
end
# Since v1.12.0
#
# 💡 HTML5 functionality is not available when running JRuby.
class Document < Nokogiri::HTML4::Document
# Get the url name for this document, as passed into Document.parse, Document.read_io, or
# Document.read_memory
attr_reader :url
# Get the parser's quirks mode value. See HTML5::QuirksMode.
#
# This method returns +nil+ if the parser was not invoked (e.g., Nokogiri::HTML5::Document.new).
#
# Since v1.14.0
attr_reader :quirks_mode
class << self
# :call-seq:
# parse(input) { |options| ... } → HTML5::Document
# parse(input, url: encoding:) { |options| ... } → HTML5::Document
# parse(input, **options) → HTML5::Document
#
# Parse \HTML input with a parser compliant with the HTML5 spec. This method uses the
# encoding of +input+ if it can be determined, or else falls back to the +encoding:+
# parameter.
#
# [Required Parameters]
# - +input+ (String | IO) the \HTML content to be parsed.
#
# [Optional Parameters]
# - +url:+ (String) the base URI of the document.
#
# [Optional Keyword Arguments]
# - +encoding:+ (Encoding) The name of the encoding that should be used when processing the
# document. When not provided, the encoding will be determined based on the document
# content.
#
# - +max_errors:+ (Integer) The maximum number of parse errors to record. (default
# +Nokogiri::Gumbo::DEFAULT_MAX_ERRORS+ which is currently 0)
#
# - +max_tree_depth:+ (Integer) The maximum depth of the parse tree. (default
# +Nokogiri::Gumbo::DEFAULT_MAX_TREE_DEPTH+)
#
# - +max_attributes:+ (Integer) The maximum number of attributes allowed on an
# element. (default +Nokogiri::Gumbo::DEFAULT_MAX_ATTRIBUTES+)
#
# - +parse_noscript_content_as_text:+ (Boolean) Whether to parse the content of +noscript+
# elements as text. (default +false+)
#
# See rdoc-ref:HTML5@Parsing+options for a complete description of these parsing options.
#
# [Yields]
# If present, the block will be passed a Hash object to modify with parse options before the
# input is parsed. See rdoc-ref:HTML5@Parsing+options for a list of available options.
#
# ⚠ Note that +url:+ and +encoding:+ cannot be set by the configuration block.
#
# [Returns] Nokogiri::HTML5::Document
#
# *Example:* Parse a string with a specific encoding and custom max errors limit.
#
# Nokogiri::HTML5::Document.parse(socket, encoding: "ISO-8859-1", max_errors: 10)
#
# *Example:* Parse a string setting the +:parse_noscript_content_as_text+ option using the
# configuration block parameter.
#
# Nokogiri::HTML5::Document.parse(input) { |c| c[:parse_noscript_content_as_text] = true }
#
def parse(
string_or_io,
url_ = nil, encoding_ = nil,
url: url_, encoding: encoding_,
**options, &block
)
yield options if block
string_or_io = "" unless string_or_io
if string_or_io.respond_to?(:encoding) && string_or_io.encoding != Encoding::ASCII_8BIT
encoding ||= string_or_io.encoding.name
end
if string_or_io.respond_to?(:read) && string_or_io.respond_to?(:path)
url ||= string_or_io.path
end
unless string_or_io.respond_to?(:read) || string_or_io.respond_to?(:to_str)
raise ArgumentError, "not a string or IO object"
end
do_parse(string_or_io, url, encoding, **options)
end
# Create a new document from an IO object.
#
# 💡 Most users should prefer Document.parse to this method.
def read_io(io, url_ = nil, encoding_ = nil, url: url_, encoding: encoding_, **options)
raise ArgumentError, "io object doesn't respond to :read" unless io.respond_to?(:read)
do_parse(io, url, encoding, **options)
end
# Create a new document from a String.
#
# 💡 Most users should prefer Document.parse to this method.
def read_memory(string, url_ = nil, encoding_ = nil, url: url_, encoding: encoding_, **options)
raise ArgumentError, "string object doesn't respond to :to_str" unless string.respond_to?(:to_str)
do_parse(string, url, encoding, **options)
end
private
def do_parse(string_or_io, url, encoding, **options)
string = HTML5.read_and_encode(string_or_io, encoding)
options[:max_attributes] ||= Nokogiri::Gumbo::DEFAULT_MAX_ATTRIBUTES
options[:max_errors] ||= options.delete(:max_parse_errors) || Nokogiri::Gumbo::DEFAULT_MAX_ERRORS
options[:max_tree_depth] ||= Nokogiri::Gumbo::DEFAULT_MAX_TREE_DEPTH
doc = Nokogiri::Gumbo.parse(string, url, self, **options)
doc.encoding = "UTF-8"
doc
end
end
def initialize(*args) # :nodoc:
super
@url = nil
@quirks_mode = nil
end
# :call-seq:
# fragment() → Nokogiri::HTML5::DocumentFragment
# fragment(markup) → Nokogiri::HTML5::DocumentFragment
#
# Parse a HTML5 document fragment from +markup+, returning a Nokogiri::HTML5::DocumentFragment.
#
# [Properties]
# - +markup+ (String) The HTML5 markup fragment to be parsed
#
# [Returns]
# Nokogiri::HTML5::DocumentFragment. This object's children will be empty if +markup+ is not
# passed, is empty, or is +nil+.
#
def fragment(markup = nil)
DocumentFragment.new(self, markup)
end
def to_xml(options = {}, &block) # :nodoc:
# Bypass XML::Document#to_xml which doesn't add
# XML::Node::SaveOptions::AS_XML like XML::Node#to_xml does.
XML::Node.instance_method(:to_xml).bind_call(self, options, &block)
end
# :call-seq:
# xpath_doctype() → Nokogiri::CSS::XPathVisitor::DoctypeConfig
#
# [Returns] The document type which determines CSS-to-XPath translation.
#
# See CSS::XPathVisitor for more information.
def xpath_doctype
Nokogiri::CSS::XPathVisitor::DoctypeConfig::HTML5
end
end
end
end
@@ -0,0 +1,200 @@
# coding: utf-8
# frozen_string_literal: true
#
# Copyright 2013-2021 Sam Ruby, Stephen Checkoway
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
require_relative "../html4/document_fragment"
module Nokogiri
module HTML5
# Since v1.12.0
#
# 💡 HTML5 functionality is not available when running JRuby.
class DocumentFragment < Nokogiri::HTML4::DocumentFragment
class << self
# :call-seq:
# parse(input, **options) → HTML5::DocumentFragment
#
# Parse \HTML5 fragment input from a String, and return a new HTML5::DocumentFragment. This
# method creates a new, empty HTML5::Document to contain the fragment.
#
# [Parameters]
# - +input+ (String | IO) The HTML5 document fragment to parse.
#
# [Optional Keyword Arguments]
# - +encoding:+ (String | Encoding) The encoding, or name of the encoding, that should be
# used when processing the document. When not provided, the encoding will be determined
# based on the document content. Also see Nokogiri::HTML5 for a longer explanation of how
# encoding is handled by the parser.
#
# - +context:+ (String | Nokogiri::XML::Node) The node, or the name of an HTML5 element, "in
# context" of which to parse the document fragment. See below for more
# information. (default +"body"+)
#
# - +max_errors:+ (Integer) The maximum number of parse errors to record. (default
# +Nokogiri::Gumbo::DEFAULT_MAX_ERRORS+ which is currently 0)
#
# - +max_tree_depth:+ (Integer) The maximum depth of the parse tree. (default
# +Nokogiri::Gumbo::DEFAULT_MAX_TREE_DEPTH+)
#
# - +max_attributes:+ (Integer) The maximum number of attributes allowed on an
# element. (default +Nokogiri::Gumbo::DEFAULT_MAX_ATTRIBUTES+)
#
# - +parse_noscript_content_as_text:+ (Boolean) Whether to parse the content of +noscript+
# elements as text. (default +false+)
#
# See rdoc-ref:HTML5@Parsing+options for a complete description of these parsing options.
#
# [Returns] Nokogiri::HTML5::DocumentFragment
#
# === Context \Node
#
# If a context node is specified using +context:+, then the parser will behave as if that
# Node, or a hypothetical tag named as specified, is the parent of the fragment subtree.
#
def parse(
input,
encoding_ = nil, positional_options_hash = nil,
encoding: encoding_, **options
)
unless positional_options_hash.nil? || positional_options_hash.empty?
options.merge!(positional_options_hash)
end
context = options.delete(:context)
document = HTML5::Document.new
document.encoding = "UTF-8"
input = HTML5.read_and_encode(input, encoding)
new(document, input, context, options)
end
end
attr_accessor :document
attr_accessor :errors
# Get the parser's quirks mode value. See HTML5::QuirksMode.
#
# This method returns `nil` if the parser was not invoked (e.g.,
# `Nokogiri::HTML5::DocumentFragment.new(doc)`).
#
# Since v1.14.0
attr_reader :quirks_mode
#
# :call-seq:
# new(document, input, **options) → HTML5::DocumentFragment
#
# Parse \HTML5 fragment input from a String, and return a new HTML5::DocumentFragment.
#
# 💡 It's recommended to use either HTML5::DocumentFragment.parse or HTML5::Node#fragment
# rather than call this method directly.
#
# [Required Parameters]
# - +document+ (HTML5::Document) The parent document to associate the returned fragment with.
#
# [Optional Parameters]
# - +input+ (String) The content to be parsed.
#
# [Optional Keyword Arguments]
# - +encoding:+ (String | Encoding) The encoding, or name of the encoding, that should be
# used when processing the document. When not provided, the encoding will be determined
# based on the document content. Also see Nokogiri::HTML5 for a longer explanation of how
# encoding is handled by the parser.
#
# - +context:+ (String | Nokogiri::XML::Node) The node, or the name of an HTML5 element, in
# which to parse the document fragment. (default +"body"+)
#
# - +max_errors:+ (Integer) The maximum number of parse errors to record. (default
# +Nokogiri::Gumbo::DEFAULT_MAX_ERRORS+ which is currently 0)
#
# - +max_tree_depth:+ (Integer) The maximum depth of the parse tree. (default
# +Nokogiri::Gumbo::DEFAULT_MAX_TREE_DEPTH+)
#
# - +max_attributes:+ (Integer) The maximum number of attributes allowed on an
# element. (default +Nokogiri::Gumbo::DEFAULT_MAX_ATTRIBUTES+)
#
# - +parse_noscript_content_as_text:+ (Boolean) Whether to parse the content of +noscript+
# elements as text. (default +false+)
#
# See rdoc-ref:HTML5@Parsing+options for a complete description of these parsing options.
#
# [Returns] HTML5::DocumentFragment
#
# === Context \Node
#
# If a context node is specified using +context:+, then the parser will behave as if that
# Node, or a hypothetical tag named as specified, is the parent of the fragment subtree.
#
def initialize(
doc, input = nil,
context_ = nil, positional_options_hash = nil,
context: context_,
**options
) # rubocop:disable Lint/MissingSuper
unless positional_options_hash.nil? || positional_options_hash.empty?
options.merge!(positional_options_hash)
end
@document = doc
@errors = []
return self unless input
input = Nokogiri::HTML5.read_and_encode(input, nil)
context = options.delete(:context) if options.key?(:context)
options[:max_attributes] ||= Nokogiri::Gumbo::DEFAULT_MAX_ATTRIBUTES
options[:max_errors] ||= options.delete(:max_parse_errors) || Nokogiri::Gumbo::DEFAULT_MAX_ERRORS
options[:max_tree_depth] ||= Nokogiri::Gumbo::DEFAULT_MAX_TREE_DEPTH
Nokogiri::Gumbo.fragment(self, input, context, **options)
end
def serialize(options = {}, &block) # :nodoc:
# Bypass XML::Document.serialize which doesn't support options even
# though XML::Node.serialize does!
XML::Node.instance_method(:serialize).bind_call(self, options, &block)
end
def extract_params(params) # :nodoc:
handler = params.find do |param|
![Hash, String, Symbol].include?(param.class)
end
params -= [handler] if handler
hashes = []
while Hash === params.last || params.last.nil?
hashes << params.pop
break if params.empty?
end
ns, binds = hashes.reverse
ns ||=
begin
ns = {}
children.each { |child| ns.merge!(child.namespaces) }
ns
end
[params, handler, ns, binds]
end
end
end
end
# vim: set shiftwidth=2 softtabstop=2 tabstop=8 expandtab:
@@ -0,0 +1,103 @@
# coding: utf-8
# frozen_string_literal: true
#
# Copyright 2013-2021 Sam Ruby, Stephen Checkoway
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
#
# TODO: this whole file should go away. maybe make it a decorator?
#
require_relative "../xml/node"
module Nokogiri
module HTML5
# Since v1.12.0
#
# 💡 HTML5 functionality is not available when running JRuby.
module Node
def inner_html(options = {})
return super unless document.is_a?(HTML5::Document)
result = options[:preserve_newline] && prepend_newline? ? +"\n" : +""
result << children.map { |child| child.to_html(options) }.join
result
end
def write_to(io, *options)
return super unless document.is_a?(HTML5::Document)
options = options.first.is_a?(Hash) ? options.shift : {}
encoding = options[:encoding] || options[0]
if Nokogiri.jruby?
save_options = options[:save_with] || options[1]
indent_times = options[:indent] || 0
else
save_options = options[:save_with] || options[1] || XML::Node::SaveOptions::FORMAT
indent_times = options[:indent] || 2
end
indent_string = (options[:indent_text] || " ") * indent_times
config = XML::Node::SaveOptions.new(save_options.to_i)
yield config if block_given?
encoding = encoding.is_a?(Encoding) ? encoding.name : encoding
config_options = config.options
if config_options & (XML::Node::SaveOptions::AS_XML | XML::Node::SaveOptions::AS_XHTML) != 0
# Use Nokogiri's serializing code.
native_write_to(io, encoding, indent_string, config_options)
else
# Serialize including the current node.
html = html_standard_serialize(options[:preserve_newline] || false)
encoding ||= document.encoding || Encoding::UTF_8
io << html.encode(encoding, fallback: lambda { |c| "&#x#{c.ord.to_s(16)};" })
end
end
def fragment(tags)
return super unless document.is_a?(HTML5::Document)
DocumentFragment.new(document, tags, self)
end
private
# HTML elements can have attributes that contain colons.
# Nokogiri::XML::Node#[]= treats names with colons as a prefixed QName
# and tries to create an attribute in a namespace. This is especially
# annoying with attribute names like xml:lang since libxml2 will
# actually create the xml namespace if it doesn't exist already.
def add_child_node_and_reparent_attrs(node)
return super unless document.is_a?(HTML5::Document)
# I'm not sure what this method is supposed to do. Reparenting
# namespaces is handled by libxml2, including child namespaces which
# this method wouldn't handle.
# https://github.com/sparklemotion/nokogiri/issues/1790
add_child_node(node)
# node.attribute_nodes.find_all { |a| a.namespace }.each do |attr|
# attr.remove
# ns = attr.namespace
# a["#{ns.prefix}:#{attr.name}"] = attr.value
# end
end
end
# Monkey patch
XML::Node.prepend(HTML5::Node)
end
end
# vim: set shiftwidth=2 softtabstop=2 tabstop=8 expandtab:
@@ -0,0 +1,3 @@
# frozen_string_literal: true
require_relative "nokogiri_jars"
@@ -0,0 +1,43 @@
# this is a generated file, to avoid over-writing it just delete this comment
begin
require 'jar_dependencies'
rescue LoadError
require 'xalan/serializer/2.7.3/serializer-2.7.3.jar'
require 'net/sourceforge/htmlunit/neko-htmlunit/2.63.0/neko-htmlunit-2.63.0.jar'
require 'nu/validator/jing/20200702VNU/jing-20200702VNU.jar'
require 'xerces/xercesImpl/2.12.2/xercesImpl-2.12.2.jar'
require 'net/sf/saxon/Saxon-HE/9.6.0-4/Saxon-HE-9.6.0-4.jar'
require 'xalan/xalan/2.7.3/xalan-2.7.3.jar'
require 'xml-apis/xml-apis/1.4.01/xml-apis-1.4.01.jar'
require 'org/nokogiri/nekodtd/0.1.11.noko2/nekodtd-0.1.11.noko2.jar'
require 'isorelax/isorelax/20030108/isorelax-20030108.jar'
end
if defined? Jars
require_jar 'xalan', 'serializer', '2.7.3'
require_jar 'net.sourceforge.htmlunit', 'neko-htmlunit', '2.63.0'
require_jar 'nu.validator', 'jing', '20200702VNU'
require_jar 'xerces', 'xercesImpl', '2.12.2'
require_jar 'net.sf.saxon', 'Saxon-HE', '9.6.0-4'
require_jar 'xalan', 'xalan', '2.7.3'
require_jar 'xml-apis', 'xml-apis', '1.4.01'
require_jar 'org.nokogiri', 'nekodtd', '0.1.11.noko2'
require_jar 'isorelax', 'isorelax', '20030108'
end
module Nokogiri
# generated by the :vendor_jars rake task
JAR_DEPENDENCIES = {
"isorelax:isorelax" => "20030108",
"net.sf.saxon:Saxon-HE" => "9.6.0-4",
"net.sourceforge.htmlunit:neko-htmlunit" => "2.63.0",
"nu.validator:jing" => "20200702VNU",
"org.nokogiri:nekodtd" => "0.1.11.noko2",
"xalan:serializer" => "2.7.3",
"xalan:xalan" => "2.7.3",
"xerces:xercesImpl" => "2.12.2",
"xml-apis:xml-apis" => "1.4.01",
}.freeze
XERCES_VERSION = JAR_DEPENDENCIES["xerces:xercesImpl"]
NEKO_VERSION = JAR_DEPENDENCIES["net.sourceforge.htmlunit:neko-htmlunit"]
end
@@ -0,0 +1,6 @@
# frozen_string_literal: true
module Nokogiri
class SyntaxError < ::StandardError
end
end
@@ -0,0 +1,4 @@
# frozen_string_literal: true
require_relative "version/constant"
require_relative "version/info"
@@ -0,0 +1,6 @@
# frozen_string_literal: true
module Nokogiri
# The version of Nokogiri you are using
VERSION = "1.18.3"
end
@@ -0,0 +1,224 @@
# frozen_string_literal: true
require "singleton"
require "shellwords"
module Nokogiri
class VersionInfo # :nodoc:
include Singleton
def jruby?
::JRUBY_VERSION if ::RUBY_PLATFORM == "java"
end
def windows?
::RUBY_PLATFORM =~ /mingw|mswin/
end
def ruby_minor
Gem::Version.new(::RUBY_VERSION).segments[0..1].join(".")
end
def engine
defined?(::RUBY_ENGINE) ? ::RUBY_ENGINE : "mri"
end
def loaded_libxml_version
Gem::Version.new(Nokogiri::LIBXML_LOADED_VERSION
.scan(/^(\d+)(\d\d)(\d\d)(?!\d)/).first
.collect(&:to_i)
.join("."))
end
def compiled_libxml_version
Gem::Version.new(Nokogiri::LIBXML_COMPILED_VERSION)
end
def loaded_libxslt_version
Gem::Version.new(Nokogiri::LIBXSLT_LOADED_VERSION
.scan(/^(\d+)(\d\d)(\d\d)(?!\d)/).first
.collect(&:to_i)
.join("."))
end
def compiled_libxslt_version
Gem::Version.new(Nokogiri::LIBXSLT_COMPILED_VERSION)
end
def libxml2?
defined?(Nokogiri::LIBXML_COMPILED_VERSION)
end
def libxml2_has_iconv?
defined?(Nokogiri::LIBXML_ICONV_ENABLED) && Nokogiri::LIBXML_ICONV_ENABLED
end
def libxslt_has_datetime?
defined?(Nokogiri::LIBXSLT_DATETIME_ENABLED) && Nokogiri::LIBXSLT_DATETIME_ENABLED
end
def libxml2_using_packaged?
libxml2? && Nokogiri::PACKAGED_LIBRARIES
end
def libxml2_using_system?
libxml2? && !libxml2_using_packaged?
end
def libxml2_precompiled?
libxml2_using_packaged? && Nokogiri::PRECOMPILED_LIBRARIES
end
def warnings
warnings = []
if libxml2?
if compiled_libxml_version != loaded_libxml_version
warnings << "Nokogiri was built against libxml version #{compiled_libxml_version}, but has dynamically loaded #{loaded_libxml_version}"
end
if compiled_libxslt_version != loaded_libxslt_version
warnings << "Nokogiri was built against libxslt version #{compiled_libxslt_version}, but has dynamically loaded #{loaded_libxslt_version}"
end
end
warnings
end
def to_hash
header_directory = File.expand_path(File.join(File.dirname(__FILE__), "../../../ext/nokogiri"))
{}.tap do |vi|
vi["warnings"] = []
vi["nokogiri"] = {}.tap do |nokogiri|
nokogiri["version"] = Nokogiri::VERSION
unless jruby?
# enable gems to build against Nokogiri with the following in their extconf.rb:
#
# append_cflags(Nokogiri::VERSION_INFO["nokogiri"]["cppflags"])
# append_ldflags(Nokogiri::VERSION_INFO["nokogiri"]["ldflags"])
#
# though, this won't work on all platform and versions of Ruby, and won't be supported
# forever, see https://github.com/sparklemotion/nokogiri/discussions/2746 for context.
#
cppflags = ["-I#{header_directory.shellescape}"]
ldflags = []
if libxml2_using_packaged?
cppflags << "-I#{File.join(header_directory, "include").shellescape}"
cppflags << "-I#{File.join(header_directory, "include/libxml2").shellescape}"
end
if windows?
# on windows, third party libraries that wish to link against nokogiri
# should link against nokogiri.so to resolve symbols. see #2167
lib_directory = File.expand_path(File.join(File.dirname(__FILE__), "../#{ruby_minor}"))
unless File.exist?(lib_directory)
lib_directory = File.expand_path(File.join(File.dirname(__FILE__), ".."))
end
ldflags << "-L#{lib_directory.shellescape}"
ldflags << "-l:nokogiri.so"
end
nokogiri["cppflags"] = cppflags
nokogiri["ldflags"] = ldflags
end
end
vi["ruby"] = {}.tap do |ruby|
ruby["version"] = ::RUBY_VERSION
ruby["platform"] = ::RUBY_PLATFORM
ruby["gem_platform"] = ::Gem::Platform.local.to_s
ruby["description"] = ::RUBY_DESCRIPTION
ruby["engine"] = engine
ruby["jruby"] = jruby? if jruby?
end
if libxml2?
vi["libxml"] = {}.tap do |libxml|
if libxml2_using_packaged?
libxml["source"] = "packaged"
libxml["precompiled"] = libxml2_precompiled?
libxml["patches"] = Nokogiri::LIBXML2_PATCHES
else
libxml["source"] = "system"
end
libxml["memory_management"] = Nokogiri::LIBXML_MEMORY_MANAGEMENT
libxml["iconv_enabled"] = libxml2_has_iconv?
libxml["compiled"] = compiled_libxml_version.to_s
libxml["loaded"] = loaded_libxml_version.to_s
end
vi["libxslt"] = {}.tap do |libxslt|
if libxml2_using_packaged?
libxslt["source"] = "packaged"
libxslt["precompiled"] = libxml2_precompiled?
libxslt["patches"] = Nokogiri::LIBXSLT_PATCHES
else
libxslt["source"] = "system"
end
libxslt["datetime_enabled"] = libxslt_has_datetime?
libxslt["compiled"] = compiled_libxslt_version.to_s
libxslt["loaded"] = loaded_libxslt_version.to_s
end
vi["warnings"] = warnings
end
if defined?(Nokogiri::OTHER_LIBRARY_VERSIONS)
# see extconf for how this string is assembled: "lib1name:lib1version,lib2name:lib2version"
vi["other_libraries"] = Hash[*Nokogiri::OTHER_LIBRARY_VERSIONS.split(/[,:]/)]
elsif jruby?
vi["other_libraries"] = {}.tap do |ol|
Nokogiri::JAR_DEPENDENCIES.each do |k, v|
ol[k] = v
end
end
end
end
end
def to_markdown
require "yaml"
"# Nokogiri (#{Nokogiri::VERSION})\n" +
YAML.dump(to_hash).each_line.map { |line| " #{line}" }.join
end
instance.warnings.each do |warning|
warn "WARNING: #{warning}"
end
end
# :nodoc:
def self.uses_libxml?(requirement = nil)
return false unless VersionInfo.instance.libxml2?
return true unless requirement
Gem::Requirement.new(requirement).satisfied_by?(VersionInfo.instance.loaded_libxml_version)
end
# :nodoc:
def self.uses_gumbo?
uses_libxml? # TODO: replace with Gumbo functionality
end
# :nodoc:
def self.jruby?
VersionInfo.instance.jruby?
end
# :nodoc:
def self.libxml2_patches
if VersionInfo.instance.libxml2_using_packaged?
Nokogiri::VERSION_INFO["libxml"]["patches"]
else
[]
end
end
require_relative "../jruby/dependencies" if Nokogiri.jruby?
require_relative "../extension"
# Detailed version info about Nokogiri and the installed extension dependencies.
VERSION_INFO = VersionInfo.instance.to_hash
end
@@ -0,0 +1,65 @@
# frozen_string_literal: true
module Nokogiri
class << self
# Convenience method for Nokogiri::XML::Document.parse
def XML(...)
Nokogiri::XML::Document.parse(...)
end
end
module XML
# Original C14N 1.0 spec canonicalization
XML_C14N_1_0 = 0
# Exclusive C14N 1.0 spec canonicalization
XML_C14N_EXCLUSIVE_1_0 = 1
# C14N 1.1 spec canonicalization
XML_C14N_1_1 = 2
class << self
# Convenience method for Nokogiri::XML::Reader.new
def Reader(...)
Reader.new(...)
end
# Convenience method for Nokogiri::XML::Document.parse
def parse(...)
Document.parse(...)
end
# Convenience method for Nokogiri::XML::DocumentFragment.parse
def fragment(...)
XML::DocumentFragment.parse(...)
end
end
end
end
require_relative "xml/pp"
require_relative "xml/parse_options"
require_relative "xml/sax"
require_relative "xml/searchable"
require_relative "xml/node"
require_relative "xml/attribute_decl"
require_relative "xml/element_decl"
require_relative "xml/element_content"
require_relative "xml/character_data"
require_relative "xml/namespace"
require_relative "xml/attr"
require_relative "xml/dtd"
require_relative "xml/cdata"
require_relative "xml/text"
require_relative "xml/document"
require_relative "xml/document_fragment"
require_relative "xml/processing_instruction"
require_relative "xml/node_set"
require_relative "xml/syntax_error"
require_relative "xml/xpath"
require_relative "xml/xpath_context"
require_relative "xml/builder"
require_relative "xml/reader"
require_relative "xml/notation"
require_relative "xml/entity_decl"
require_relative "xml/entity_reference"
require_relative "xml/schema"
require_relative "xml/relax_ng"
@@ -0,0 +1,66 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
class Attr < Node
alias_method :value, :content
alias_method :to_s, :content
alias_method :content=, :value=
#
# :call-seq: deconstruct_keys(array_of_names) → Hash
#
# Returns a hash describing the Attr, to use in pattern matching.
#
# Valid keys and their values:
# - +name+ → (String) The name of the attribute.
# - +value+ → (String) The value of the attribute.
# - +namespace+ → (Namespace, nil) The Namespace of the attribute, or +nil+ if there is no namespace.
#
# *Example*
#
# doc = Nokogiri::XML.parse(<<~XML)
# <?xml version="1.0"?>
# <root xmlns="http://nokogiri.org/ns/default" xmlns:noko="http://nokogiri.org/ns/noko">
# <child1 foo="abc" noko:bar="def"/>
# </root>
# XML
#
# attributes = doc.root.elements.first.attribute_nodes
# # => [#(Attr:0x35c { name = "foo", value = "abc" }),
# # #(Attr:0x370 {
# # name = "bar",
# # namespace = #(Namespace:0x384 {
# # prefix = "noko",
# # href = "http://nokogiri.org/ns/noko"
# # }),
# # value = "def"
# # })]
#
# attributes.first.deconstruct_keys([:name, :value, :namespace])
# # => {:name=>"foo", :value=>"abc", :namespace=>nil}
#
# attributes.last.deconstruct_keys([:name, :value, :namespace])
# # => {:name=>"bar",
# # :value=>"def",
# # :namespace=>
# # #(Namespace:0x384 {
# # prefix = "noko",
# # href = "http://nokogiri.org/ns/noko"
# # })}
#
# Since v1.14.0
#
def deconstruct_keys(keys)
{ name: name, value: value, namespace: namespace }
end
private
def inspect_attributes
[:name, :namespace, :value]
end
end
end
end
@@ -0,0 +1,22 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# Represents an attribute declaration in a DTD
class AttributeDecl < Nokogiri::XML::Node
undef_method :attribute_nodes
undef_method :attributes
undef_method :content
undef_method :namespace
undef_method :namespace_definitions
undef_method :line if method_defined?(:line)
private
def inspect_attributes
[:to_s]
end
end
end
end
@@ -0,0 +1,494 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# Nokogiri builder can be used for building XML and HTML documents.
#
# == Synopsis:
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.root {
# xml.products {
# xml.widget {
# xml.id_ "10"
# xml.name "Awesome widget"
# }
# }
# }
# end
# puts builder.to_xml
#
# Will output:
#
# <?xml version="1.0"?>
# <root>
# <products>
# <widget>
# <id>10</id>
# <name>Awesome widget</name>
# </widget>
# </products>
# </root>
#
#
# === Builder scope
#
# The builder allows two forms. When the builder is supplied with a block
# that has a parameter, the outside scope is maintained. This means you
# can access variables that are outside your builder. If you don't need
# outside scope, you can use the builder without the "xml" prefix like
# this:
#
# builder = Nokogiri::XML::Builder.new do
# root {
# products {
# widget {
# id_ "10"
# name "Awesome widget"
# }
# }
# }
# end
#
# == Special Tags
#
# The builder works by taking advantage of method_missing. Unfortunately
# some methods are defined in ruby that are difficult or dangerous to
# remove. You may want to create tags with the name "type", "class", and
# "id" for example. In that case, you can use an underscore to
# disambiguate your tag name from the method call.
#
# Here is an example of using the underscore to disambiguate tag names from
# ruby methods:
#
# @objects = [Object.new, Object.new, Object.new]
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.root {
# xml.objects {
# @objects.each do |o|
# xml.object {
# xml.type_ o.type
# xml.class_ o.class.name
# xml.id_ o.id
# }
# end
# }
# }
# end
# puts builder.to_xml
#
# The underscore may be used with any tag name, and the last underscore
# will just be removed. This code will output the following XML:
#
# <?xml version="1.0"?>
# <root>
# <objects>
# <object>
# <type>Object</type>
# <class>Object</class>
# <id>48390</id>
# </object>
# <object>
# <type>Object</type>
# <class>Object</class>
# <id>48380</id>
# </object>
# <object>
# <type>Object</type>
# <class>Object</class>
# <id>48370</id>
# </object>
# </objects>
# </root>
#
# == Tag Attributes
#
# Tag attributes may be supplied as method arguments. Here is our
# previous example, but using attributes rather than tags:
#
# @objects = [Object.new, Object.new, Object.new]
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.root {
# xml.objects {
# @objects.each do |o|
# xml.object(:type => o.type, :class => o.class, :id => o.id)
# end
# }
# }
# end
# puts builder.to_xml
#
# === Tag Attribute Short Cuts
#
# A couple attribute short cuts are available when building tags. The
# short cuts are available by special method calls when building a tag.
#
# This example builds an "object" tag with the class attribute "classy"
# and the id of "thing":
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.root {
# xml.objects {
# xml.object.classy.thing!
# }
# }
# end
# puts builder.to_xml
#
# Which will output:
#
# <?xml version="1.0"?>
# <root>
# <objects>
# <object class="classy" id="thing"/>
# </objects>
# </root>
#
# All other options are still supported with this syntax, including
# blocks and extra tag attributes.
#
# == Namespaces
#
# Namespaces are added similarly to attributes. Nokogiri::XML::Builder
# assumes that when an attribute starts with "xmlns", it is meant to be
# a namespace:
#
# builder = Nokogiri::XML::Builder.new { |xml|
# xml.root('xmlns' => 'default', 'xmlns:foo' => 'bar') do
# xml.tenderlove
# end
# }
# puts builder.to_xml
#
# Will output XML like this:
#
# <?xml version="1.0"?>
# <root xmlns:foo="bar" xmlns="default">
# <tenderlove/>
# </root>
#
# === Referencing declared namespaces
#
# Tags that reference non-default namespaces (i.e. a tag "foo:bar") can be
# built by using the Nokogiri::XML::Builder#[] method.
#
# For example:
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.root('xmlns:foo' => 'bar') {
# xml.objects {
# xml['foo'].object.classy.thing!
# }
# }
# end
# puts builder.to_xml
#
# Will output this XML:
#
# <?xml version="1.0"?>
# <root xmlns:foo="bar">
# <objects>
# <foo:object class="classy" id="thing"/>
# </objects>
# </root>
#
# Note the "foo:object" tag.
#
# === Namespace inheritance
#
# In the Builder context, children will inherit their parent's namespace. This is the same
# behavior as if the underlying {XML::Document} set +namespace_inheritance+ to +true+:
#
# result = Nokogiri::XML::Builder.new do |xml|
# xml["soapenv"].Envelope("xmlns:soapenv" => "http://schemas.xmlsoap.org/soap/envelope/") do
# xml.Header
# end
# end
# result.doc.to_xml
# # => <?xml version="1.0" encoding="utf-8"?>
# # <soapenv:Envelope xmlns:soapenv="http://schemas.xmlsoap.org/soap/envelope/">
# # <soapenv:Header/>
# # </soapenv:Envelope>
#
# Users may turn this behavior off by passing a keyword argument +namespace_inheritance:false+
# to the initializer:
#
# result = Nokogiri::XML::Builder.new(namespace_inheritance: false) do |xml|
# xml["soapenv"].Envelope("xmlns:soapenv" => "http://schemas.xmlsoap.org/soap/envelope/") do
# xml.Header
# xml["soapenv"].Body # users may explicitly opt into the namespace
# end
# end
# result.doc.to_xml
# # => <?xml version="1.0" encoding="utf-8"?>
# # <soapenv:Envelope xmlns:soapenv="http://schemas.xmlsoap.org/soap/envelope/">
# # <Header/>
# # <soapenv:Body/>
# # </soapenv:Envelope>
#
# For more information on namespace inheritance, please see {XML::Document#namespace_inheritance}
#
#
# == Document Types
#
# To create a document type (DTD), use the Builder#doc method to get
# the current context document. Then call Node#create_internal_subset to
# create the DTD node.
#
# For example, this Ruby:
#
# builder = Nokogiri::XML::Builder.new do |xml|
# xml.doc.create_internal_subset(
# 'html',
# "-//W3C//DTD HTML 4.01 Transitional//EN",
# "http://www.w3.org/TR/html4/loose.dtd"
# )
# xml.root do
# xml.foo
# end
# end
#
# puts builder.to_xml
#
# Will output this xml:
#
# <?xml version="1.0"?>
# <!DOCTYPE html PUBLIC "-//W3C//DTD HTML 4.01 Transitional//EN" "http://www.w3.org/TR/html4/loose.dtd">
# <root>
# <foo/>
# </root>
#
class Builder
include Nokogiri::ClassResolver
DEFAULT_DOCUMENT_OPTIONS = { namespace_inheritance: true }
# The current Document object being built
attr_accessor :doc
# The parent of the current node being built
attr_accessor :parent
# A context object for use when the block has no arguments
attr_accessor :context
attr_accessor :arity # :nodoc:
###
# Create a builder with an existing root object. This is for use when
# you have an existing document that you would like to augment with
# builder methods. The builder context created will start with the
# given +root+ node.
#
# For example:
#
# doc = Nokogiri::XML(File.read('somedoc.xml'))
# Nokogiri::XML::Builder.with(doc.at_css('some_tag')) do |xml|
# # ... Use normal builder methods here ...
# xml.awesome # add the "awesome" tag below "some_tag"
# end
#
def self.with(root, &block)
new({}, root, &block)
end
###
# Create a new Builder object. +options+ are sent to the top level
# Document that is being built.
#
# Building a document with a particular encoding for example:
#
# Nokogiri::XML::Builder.new(:encoding => 'UTF-8') do |xml|
# ...
# end
def initialize(options = {}, root = nil, &block)
if root
@doc = root.document
@parent = root
else
@parent = @doc = related_class("Document").new
end
@context = nil
@arity = nil
@ns = nil
options = DEFAULT_DOCUMENT_OPTIONS.merge(options)
options.each do |k, v|
@doc.send(:"#{k}=", v)
end
return unless block
@arity = block.arity
if @arity <= 0
@context = eval("self", block.binding)
instance_eval(&block)
else
yield self
end
@parent = @doc
end
###
# Create a Text Node with content of +string+
def text(string)
insert(@doc.create_text_node(string))
end
###
# Create a CDATA Node with content of +string+
def cdata(string)
insert(doc.create_cdata(string))
end
###
# Create a Comment Node with content of +string+
def comment(string)
insert(doc.create_comment(string))
end
###
# Build a tag that is associated with namespace +ns+. Raises an
# ArgumentError if +ns+ has not been defined higher in the tree.
def [](ns)
if @parent != @doc
@ns = @parent.namespace_definitions.find { |x| x.prefix == ns.to_s }
end
return self if @ns
@parent.ancestors.each do |a|
next if a == doc
@ns = a.namespace_definitions.find { |x| x.prefix == ns.to_s }
return self if @ns
end
@ns = { pending: ns.to_s }
self
end
###
# Convert this Builder object to XML
def to_xml(*args)
if Nokogiri.jruby?
options = args.first.is_a?(Hash) ? args.shift : {}
unless options[:save_with]
options[:save_with] = Node::SaveOptions::AS_BUILDER
end
args.insert(0, options)
end
@doc.to_xml(*args)
end
###
# Append the given raw XML +string+ to the document
def <<(string)
@doc.fragment(string).children.each { |x| insert(x) }
end
def method_missing(method, *args, &block) # :nodoc:
if @context&.respond_to?(method)
@context.send(method, *args, &block)
else
node = @doc.create_element(method.to_s.sub(/[_!]$/, ""), *args) do |n|
# Set up the namespace
if @ns.is_a?(Nokogiri::XML::Namespace)
n.namespace = @ns
@ns = nil
end
end
if @ns.is_a?(Hash)
node.namespace = node.namespace_definitions.find { |x| x.prefix == @ns[:pending] }
if node.namespace.nil?
raise ArgumentError, "Namespace #{@ns[:pending]} has not been defined"
end
@ns = nil
end
insert(node, &block)
end
end
private
###
# Insert +node+ as a child of the current Node
def insert(node, &block)
node = @parent.add_child(node)
if block
begin
old_parent = @parent
@parent = node
@arity ||= block.arity
if @arity <= 0
instance_eval(&block)
else
yield(self)
end
ensure
@parent = old_parent
end
end
NodeBuilder.new(node, self)
end
class NodeBuilder # :nodoc:
def initialize(node, doc_builder)
@node = node
@doc_builder = doc_builder
end
def []=(k, v)
@node[k] = v
end
def [](k)
@node[k]
end
def method_missing(method, *args, &block)
opts = args.last.is_a?(Hash) ? args.pop : {}
case method.to_s
when /^(.*)!$/
@node["id"] = Regexp.last_match(1)
@node.content = args.first if args.first
when /^(.*)=/
@node[Regexp.last_match(1)] = args.first
else
@node["class"] =
((@node["class"] || "").split(/\s/) + [method.to_s]).join(" ")
@node.content = args.first if args.first
end
# Assign any extra options
opts.each do |k, v|
@node[k.to_s] = ((@node[k.to_s] || "").split(/\s/) + [v]).join(" ")
end
if block
old_parent = @doc_builder.parent
@doc_builder.parent = @node
arity = @doc_builder.arity || block.arity
value = if arity <= 0
@doc_builder.instance_eval(&block)
else
yield(@doc_builder)
end
@doc_builder.parent = old_parent
return value
end
self
end
end
end
end
end
@@ -0,0 +1,13 @@
# frozen_string_literal: true
module Nokogiri
module XML
class CDATA < Nokogiri::XML::Text
###
# Get the name of this CDATA node
def name
"#cdata-section"
end
end
end
end
@@ -0,0 +1,9 @@
# frozen_string_literal: true
module Nokogiri
module XML
class CharacterData < Nokogiri::XML::Node
include Nokogiri::XML::PP::CharacterData
end
end
end
@@ -0,0 +1,514 @@
# coding: utf-8
# frozen_string_literal: true
require "pathname"
module Nokogiri
module XML
# Nokogiri::XML::Document is the main entry point for dealing with \XML documents. The Document
# is created by parsing \XML content from a String or an IO object. See
# Nokogiri::XML::Document.parse for more information on parsing.
#
# Document inherits a great deal of functionality from its superclass Nokogiri::XML::Node, so
# please read that class's documentation as well.
class Document < Nokogiri::XML::Node
# See http://www.w3.org/TR/REC-xml-names/#ns-decl for more details. Note that we're not
# attempting to handle unicode characters partly because libxml2 doesn't handle unicode
# characters in NCNAMEs.
NCNAME_START_CHAR = "A-Za-z_"
NCNAME_CHAR = NCNAME_START_CHAR + "\\-\\.0-9"
NCNAME_RE = /^xmlns(?::([#{NCNAME_START_CHAR}][#{NCNAME_CHAR}]*))?$/
OBJECT_DUP_METHOD = Object.instance_method(:dup)
OBJECT_CLONE_METHOD = Object.instance_method(:clone)
private_constant :OBJECT_DUP_METHOD, :OBJECT_CLONE_METHOD
class << self
# call-seq:
# parse(input) { |options| ... } => Nokogiri::XML::Document
# parse(input, url:, encoding:, options:) => Nokogiri::XML::Document
#
# Parse \XML input from a String or IO object, and return a new XML::Document.
#
# 🛡 By default, Nokogiri treats documents as untrusted, and so does not attempt to load DTDs
# or access the network. See Nokogiri::XML::ParseOptions for a complete list of options; and
# that module's DEFAULT_XML constant for what's set (and not set) by default.
#
# [Required Parameters]
# - +input+ (String | IO) The content to be parsed.
#
# [Optional Keyword Arguments]
# - +url:+ (String) The base URI for this document.
#
# - +encoding:+ (String) The name of the encoding that should be used when processing the
# document. When not provided, the encoding will be determined based on the document
# content.
#
# - +options:+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
# behaviors during parsing. See ParseOptions for more information. The default value is
# +ParseOptions::DEFAULT_XML+.
#
# [Yields]
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
# can be configured before parsing. See Nokogiri::XML::ParseOptions for more information.
#
# [Returns] Nokogiri::XML::Document
def parse(
string_or_io,
url_ = nil, encoding_ = nil, options_ = XML::ParseOptions::DEFAULT_XML,
url: url_, encoding: encoding_, options: options_
)
options = Nokogiri::XML::ParseOptions.new(options) if Integer === options
yield options if block_given?
url ||= string_or_io.respond_to?(:path) ? string_or_io.path : nil
if empty_doc?(string_or_io)
if options.strict?
raise Nokogiri::XML::SyntaxError, "Empty document"
else
return encoding ? new.tap { |i| i.encoding = encoding } : new
end
end
doc = if string_or_io.respond_to?(:read)
# TODO: should we instead check for respond_to?(:to_path) ?
if string_or_io.is_a?(Pathname)
# resolve the Pathname to the file and open it as an IO object, see #2110
string_or_io = string_or_io.expand_path.open
url ||= string_or_io.path
end
read_io(string_or_io, url, encoding, options.to_i)
else
# read_memory pukes on empty docs
read_memory(string_or_io, url, encoding, options.to_i)
end
# do xinclude processing
doc.do_xinclude(options) if options.xinclude?
doc
end
private
def empty_doc?(string_or_io)
string_or_io.nil? ||
(string_or_io.respond_to?(:empty?) && string_or_io.empty?) ||
(string_or_io.respond_to?(:eof?) && string_or_io.eof?)
end
end
##
# :singleton-method: wrap
# :call-seq: wrap(java_document) → Nokogiri::XML::Document
#
# ⚠ This method is only available when running JRuby.
#
# Create a Document using an existing Java DOM document object.
#
# The returned Document shares the same underlying data structure as the Java object, so
# changes in one are reflected in the other.
#
# [Parameters]
# - `java_document` (Java::OrgW3cDom::Document)
# (The class `Java::OrgW3cDom::Document` is also accessible as `org.w3c.dom.Document`.)
#
# [Returns] Nokogiri::XML::Document
#
# See also \#to_java
# :method: to_java
# :call-seq: to_java() → Java::OrgW3cDom::Document
#
# ⚠ This method is only available when running JRuby.
#
# Returns the underlying Java DOM document object for this document.
#
# The returned Java object shares the same underlying data structure as this document, so
# changes in one are reflected in the other.
#
# [Returns]
# Java::OrgW3cDom::Document
# (The class `Java::OrgW3cDom::Document` is also accessible as `org.w3c.dom.Document`.)
#
# See also Document.wrap
# The errors found while parsing a document.
#
# [Returns] Array<Nokogiri::XML::SyntaxError>
attr_accessor :errors
# When `true`, reparented elements without a namespace will inherit their new parent's
# namespace (if one exists). Defaults to `false`.
#
# [Returns] Boolean
#
# *Example:* Default behavior of namespace inheritance
#
# xml = <<~EOF
# <root xmlns:foo="http://nokogiri.org/default_ns/test/foo">
# <foo:parent>
# </foo:parent>
# </root>
# EOF
# doc = Nokogiri::XML(xml)
# parent = doc.at_xpath("//foo:parent", "foo" => "http://nokogiri.org/default_ns/test/foo")
# parent.add_child("<child></child>")
# doc.to_xml
# # => <?xml version="1.0"?>
# # <root xmlns:foo="http://nokogiri.org/default_ns/test/foo">
# # <foo:parent>
# # <child/>
# # </foo:parent>
# # </root>
#
# *Example:* Setting namespace inheritance to `true`
#
# xml = <<~EOF
# <root xmlns:foo="http://nokogiri.org/default_ns/test/foo">
# <foo:parent>
# </foo:parent>
# </root>
# EOF
# doc = Nokogiri::XML(xml)
# doc.namespace_inheritance = true
# parent = doc.at_xpath("//foo:parent", "foo" => "http://nokogiri.org/default_ns/test/foo")
# parent.add_child("<child></child>")
# doc.to_xml
# # => <?xml version="1.0"?>
# # <root xmlns:foo="http://nokogiri.org/default_ns/test/foo">
# # <foo:parent>
# # <foo:child/>
# # </foo:parent>
# # </root>
#
# Since v1.12.4
attr_accessor :namespace_inheritance
def initialize(*args) # :nodoc: # rubocop:disable Lint/MissingSuper
@errors = []
@decorators = nil
@namespace_inheritance = false
end
#
# :call-seq:
# dup → Nokogiri::XML::Document
# dup(level) → Nokogiri::XML::Document
#
# Duplicate this node.
#
# [Parameters]
# - +level+ (optional Integer). 0 is a shallow copy, 1 (the default) is a deep copy.
# [Returns] The new Nokogiri::XML::Document
#
def dup(level = 1)
copy = OBJECT_DUP_METHOD.bind_call(self)
copy.initialize_copy_with_args(self, level)
end
#
# :call-seq:
# clone → Nokogiri::XML::Document
# clone(level) → Nokogiri::XML::Document
#
# Clone this node.
#
# [Parameters]
# - +level+ (optional Integer). 0 is a shallow copy, 1 (the default) is a deep copy.
# [Returns] The new Nokogiri::XML::Document
#
def clone(level = 1)
copy = OBJECT_CLONE_METHOD.bind_call(self)
copy.initialize_copy_with_args(self, level)
end
# :call-seq:
# create_element(name, *contents_or_attrs, &block) → Nokogiri::XML::Element
#
# Create a new Element with `name` belonging to this document, optionally setting contents or
# attributes.
#
# This method is _not_ the most user-friendly option if your intention is to add a node to the
# document tree. Prefer one of the Nokogiri::XML::Node methods like Node#add_child,
# Node#add_next_sibling, Node#replace, etc. which will both create an element (or subtree) and
# place it in the document tree.
#
# Arguments may be passed to initialize the element:
#
# - a Hash argument will be used to set attributes
# - a non-Hash object that responds to \#to_s will be used to set the new node's contents
#
# A block may be passed to mutate the node.
#
# [Parameters]
# - `name` (String)
# - `contents_or_attrs` (\#to_s, Hash)
# [Yields] `node` (Nokogiri::XML::Element)
# [Returns] Nokogiri::XML::Element
#
# *Example:* An empty element without attributes
#
# doc.create_element("div")
# # => <div></div>
#
# *Example:* An element with contents
#
# doc.create_element("div", "contents")
# # => <div>contents</div>
#
# *Example:* An element with attributes
#
# doc.create_element("div", {"class" => "container"})
# # => <div class='container'></div>
#
# *Example:* An element with contents and attributes
#
# doc.create_element("div", "contents", {"class" => "container"})
# # => <div class='container'>contents</div>
#
# *Example:* Passing a block to mutate the element
#
# doc.create_element("div") { |node| node["class"] = "blue" if before_noon? }
#
def create_element(name, *contents_or_attrs, &block)
elm = Nokogiri::XML::Element.new(name, self, &block)
contents_or_attrs.each do |arg|
case arg
when Hash
arg.each do |k, v|
key = k.to_s
if key =~ NCNAME_RE
ns_name = Regexp.last_match(1)
elm.add_namespace_definition(ns_name, v)
else
elm[k.to_s] = v.to_s
end
end
else
elm.content = arg
end
end
if (ns = elm.namespace_definitions.find { |n| n.prefix.nil? || (n.prefix == "") })
elm.namespace = ns
end
elm
end
# Create a Text Node with +string+
def create_text_node(string, &block)
Nokogiri::XML::Text.new(string.to_s, self, &block)
end
# Create a CDATA Node containing +string+
def create_cdata(string, &block)
Nokogiri::XML::CDATA.new(self, string.to_s, &block)
end
# Create a Comment Node containing +string+
def create_comment(string, &block)
Nokogiri::XML::Comment.new(self, string.to_s, &block)
end
# The name of this document. Always returns "document"
def name
"document"
end
# A reference to +self+
def document
self
end
# :call-seq:
# collect_namespaces() → Hash<String(Namespace#prefix) ⇒ String(Namespace#href)>
#
# Recursively get all namespaces from this node and its subtree and return them as a
# hash.
#
# ⚠ This method will not handle duplicate namespace prefixes, since the return value is a hash.
#
# Note that this method does an xpath lookup for nodes with namespaces, and as a result the
# order (and which duplicate prefix "wins") may be dependent on the implementation of the
# underlying XML library.
#
# *Example:* Basic usage
#
# Given this document:
#
# <root xmlns="default" xmlns:foo="bar">
# <bar xmlns:hello="world" />
# </root>
#
# This method will return:
#
# {"xmlns:foo"=>"bar", "xmlns"=>"default", "xmlns:hello"=>"world"}
#
# *Example:* Duplicate prefixes
#
# Given this document:
#
# <root xmlns:foo="bar">
# <bar xmlns:foo="baz" />
# </root>
#
# The hash returned will be something like:
#
# {"xmlns:foo" => "baz"}
#
def collect_namespaces
xpath("//namespace::*").each_with_object({}) do |ns, hash|
hash[["xmlns", ns.prefix].compact.join(":")] = ns.href if ns.prefix != "xml"
end
end
# Get the list of decorators given +key+
def decorators(key)
@decorators ||= {}
@decorators[key] ||= []
end
##
# Validate this Document against its DTD. Returns a list of errors on
# the document or +nil+ when there is no DTD.
def validate
return unless internal_subset
internal_subset.validate(self)
end
##
# Explore a document with shortcut methods. See Nokogiri::Slop for details.
#
# Note that any nodes that have been instantiated before #slop!
# is called will not be decorated with sloppy behavior. So, if you're in
# irb, the preferred idiom is:
#
# irb> doc = Nokogiri::Slop my_markup
#
# and not
#
# irb> doc = Nokogiri::HTML my_markup
# ... followed by irb's implicit inspect (and therefore instantiation of every node) ...
# irb> doc.slop!
# ... which does absolutely nothing.
#
def slop!
unless decorators(XML::Node).include?(Nokogiri::Decorators::Slop)
decorators(XML::Node) << Nokogiri::Decorators::Slop
decorate!
end
self
end
##
# Apply any decorators to +node+
def decorate(node)
return unless @decorators
@decorators.each do |klass, list|
next unless node.is_a?(klass)
list.each { |mod| node.extend(mod) }
end
end
alias_method :to_xml, :serialize
# Get the hash of namespaces on the root Nokogiri::XML::Node
def namespaces
root ? root.namespaces : {}
end
##
# Create a Nokogiri::XML::DocumentFragment from +tags+
# Returns an empty fragment if +tags+ is nil.
def fragment(tags = nil)
DocumentFragment.new(self, tags, root)
end
undef_method :swap, :parent, :namespace, :default_namespace=
undef_method :add_namespace_definition, :attributes
undef_method :namespace_definitions, :line, :add_namespace
def add_child(node_or_tags)
raise "A document may not have multiple root nodes." if (root && root.name != "nokogiri_text_wrapper") && !(node_or_tags.comment? || node_or_tags.processing_instruction?)
node_or_tags = coerce(node_or_tags)
if node_or_tags.is_a?(XML::NodeSet)
raise "A document may not have multiple root nodes." if node_or_tags.size > 1
super(node_or_tags.first)
else
super
end
end
alias_method :<<, :add_child
# :call-seq:
# xpath_doctype() → Nokogiri::CSS::XPathVisitor::DoctypeConfig
#
# [Returns] The document type which determines CSS-to-XPath translation.
#
# See XPathVisitor for more information.
def xpath_doctype
Nokogiri::CSS::XPathVisitor::DoctypeConfig::XML
end
#
# :call-seq: deconstruct_keys(array_of_names) → Hash
#
# Returns a hash describing the Document, to use in pattern matching.
#
# Valid keys and their values:
# - +root+ → (Node, nil) The root node of the Document, or +nil+ if the document is empty.
#
# In the future, other keys may allow accessing things like doctype and processing
# instructions. If you have a use case and would like this functionality, please let us know
# by opening an issue or a discussion on the github project.
#
# *Example*
#
# doc = Nokogiri::XML.parse(<<~XML)
# <?xml version="1.0"?>
# <root>
# <child>
# </root>
# XML
#
# doc.deconstruct_keys([:root])
# # => {:root=>
# # #(Element:0x35c {
# # name = "root",
# # children = [
# # #(Text "\n" + " "),
# # #(Element:0x370 { name = "child", children = [ #(Text "\n")] }),
# # #(Text "\n")]
# # })}
#
# *Example* of an empty document
#
# doc = Nokogiri::XML::Document.new
#
# doc.deconstruct_keys([:root])
# # => {:root=>nil}
#
# Since v1.14.0
#
def deconstruct_keys(keys)
{ root: root }
end
private
IMPLIED_XPATH_CONTEXTS = ["//"].freeze # :nodoc:
def inspect_attributes
[:name, :children]
end
end
end
end
@@ -0,0 +1,276 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
# DocumentFragment represents a fragment of an \XML document. It provides the same functionality
# exposed by XML::Node and can be used to contain one or more \XML subtrees.
class DocumentFragment < Nokogiri::XML::Node
# The options used to parse the document fragment. Returns the value of any options that were
# passed into the constructor as a parameter or set in a config block, else the default
# options for the specific subclass.
attr_reader :parse_options
class << self
# :call-seq:
# parse(input) { |options| ... } → XML::DocumentFragment
# parse(input, options:) → XML::DocumentFragment
#
# Parse \XML fragment input from a String, and return a new XML::DocumentFragment. This
# method creates a new, empty XML::Document to contain the fragment.
#
# [Required Parameters]
# - +input+ (String) The content to be parsed.
#
# [Optional Keyword Arguments]
# - +options+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
# behaviors during parsing. See ParseOptions for more information. The default value is
# +ParseOptions::DEFAULT_XML+.
#
# [Yields]
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
# can be configured before parsing. See Nokogiri::XML::ParseOptions for more information.
#
# [Returns] Nokogiri::XML::DocumentFragment
def parse(tags, options_ = ParseOptions::DEFAULT_XML, options: options_, &block)
new(XML::Document.new, tags, options: options, &block)
end
# Wrapper method to separate the concerns of:
# - the native object allocator's parameter (it only requires `document`)
# - the initializer's parameters
def new(document, ...) # :nodoc:
instance = native_new(document)
instance.send(:initialize, document, ...)
instance
end
end
# :call-seq:
# new(document, input=nil) { |options| ... } → DocumentFragment
# new(document, input=nil, context:, options:) → DocumentFragment
#
# Parse \XML fragment input from a String, and return a new DocumentFragment that is
# associated with the given +document+.
#
# 💡 It's recommended to use either XML::DocumentFragment.parse or Node#parse rather than call
# this method directly.
#
# [Required Parameters]
# - +document+ (XML::Document) The parent document to associate the returned fragment with.
#
# [Optional Parameters]
# - +input+ (String) The content to be parsed.
#
# [Optional Keyword Arguments]
# - +context:+ (Nokogiri::XML::Node) The <b>context node</b> for the subtree created. See
# below for more information.
#
# - +options:+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
# behaviors during parsing. See ParseOptions for more information. The default value is
# +ParseOptions::DEFAULT_XML+.
#
# [Yields]
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
# can be configured before parsing. See ParseOptions for more information.
#
# [Returns] XML::DocumentFragment
#
# === Context \Node
#
# If a context node is specified using +context:+, then the fragment will be created by
# calling Node#parse on that node, so the parser will behave as if that Node is the parent of
# the fragment subtree, and will resolve namespaces relative to that node.
#
def initialize(
document, tags = nil,
context_ = nil, options_ = ParseOptions::DEFAULT_XML,
context: context_, options: options_
) # rubocop:disable Lint/MissingSuper
return self unless tags
options = Nokogiri::XML::ParseOptions.new(options) if Integer === options
@parse_options = options
yield options if block_given?
children = if context
# Fix for issue#490
if Nokogiri.jruby?
# fix for issue #770
context.parse("<root #{namespace_declarations(context)}>#{tags}</root>", options).children
else
context.parse(tags, options)
end
else
wrapper_doc = XML::Document.parse("<root>#{tags}</root>", nil, nil, options)
self.errors = wrapper_doc.errors
wrapper_doc.xpath("/root/node()")
end
children.each { |child| child.parent = self }
end
if Nokogiri.uses_libxml?
def dup
new_document = document.dup
new_fragment = self.class.new(new_document)
children.each do |child|
child.dup(1, new_document).parent = new_fragment
end
new_fragment
end
end
###
# return the name for DocumentFragment
def name
"#document-fragment"
end
###
# Convert this DocumentFragment to a string
def to_s
children.to_s
end
###
# Convert this DocumentFragment to html
# See Nokogiri::XML::NodeSet#to_html
def to_html(*args)
if Nokogiri.jruby?
options = args.first.is_a?(Hash) ? args.shift : {}
options[:save_with] ||= Node::SaveOptions::DEFAULT_HTML
args.insert(0, options)
end
children.to_html(*args)
end
###
# Convert this DocumentFragment to xhtml
# See Nokogiri::XML::NodeSet#to_xhtml
def to_xhtml(*args)
if Nokogiri.jruby?
options = args.first.is_a?(Hash) ? args.shift : {}
options[:save_with] ||= Node::SaveOptions::DEFAULT_XHTML
args.insert(0, options)
end
children.to_xhtml(*args)
end
###
# Convert this DocumentFragment to xml
# See Nokogiri::XML::NodeSet#to_xml
def to_xml(*args)
children.to_xml(*args)
end
###
# call-seq: css *rules, [namespace-bindings, custom-pseudo-class]
#
# Search this fragment for CSS +rules+. +rules+ must be one or more CSS
# selectors. For example:
#
# For more information see Nokogiri::XML::Searchable#css
def css(*args)
if children.any?
children.css(*args) # 'children' is a smell here
else
NodeSet.new(document)
end
end
#
# NOTE that we don't delegate #xpath to children ... another smell.
# def xpath ; end
#
###
# call-seq: search *paths, [namespace-bindings, xpath-variable-bindings, custom-handler-class]
#
# Search this fragment for +paths+. +paths+ must be one or more XPath or CSS queries.
#
# For more information see Nokogiri::XML::Searchable#search
def search(*rules)
rules, handler, ns, binds = extract_params(rules)
rules.inject(NodeSet.new(document)) do |set, rule|
set + if Searchable::LOOKS_LIKE_XPATH.match?(rule)
xpath(*[rule, ns, handler, binds].compact)
else
children.css(*[rule, ns, handler].compact) # 'children' is a smell here
end
end
end
alias_method :serialize, :to_s
# A list of Nokogiri::XML::SyntaxError found when parsing a document
def errors
document.errors
end
def errors=(things) # :nodoc:
document.errors = things
end
def fragment(data)
document.fragment(data)
end
#
# :call-seq: deconstruct() → Array
#
# Returns the root nodes of this document fragment as an array, to use in pattern matching.
#
# 💡 Note that text nodes are returned as well as elements. If you wish to operate only on
# root elements, you should deconstruct the array returned by
# <tt>DocumentFragment#elements</tt>.
#
# *Example*
#
# frag = Nokogiri::HTML5.fragment(<<~HTML)
# <div>Start</div>
# This is a <a href="#jump">shortcut</a> for you.
# <div>End</div>
# HTML
#
# frag.deconstruct
# # => [#(Element:0x35c { name = "div", children = [ #(Text "Start")] }),
# # #(Text "\n" + "This is a "),
# # #(Element:0x370 {
# # name = "a",
# # attributes = [ #(Attr:0x384 { name = "href", value = "#jump" })],
# # children = [ #(Text "shortcut")]
# # }),
# # #(Text " for you.\n"),
# # #(Element:0x398 { name = "div", children = [ #(Text "End")] }),
# # #(Text "\n")]
#
# *Example* only the elements, not the text nodes.
#
# frag.elements.deconstruct
# # => [#(Element:0x35c { name = "div", children = [ #(Text "Start")] }),
# # #(Element:0x370 {
# # name = "a",
# # attributes = [ #(Attr:0x384 { name = "href", value = "#jump" })],
# # children = [ #(Text "shortcut")]
# # }),
# # #(Element:0x398 { name = "div", children = [ #(Text "End")] })]
#
# Since v1.14.0
#
def deconstruct
children.to_a
end
private
# fix for issue 770
def namespace_declarations(ctx)
ctx.namespace_scopes.map do |namespace|
prefix = namespace.prefix.nil? ? "" : ":#{namespace.prefix}"
%{xmlns#{prefix}="#{namespace.href}"}
end.join(" ")
end
end
end
end
@@ -0,0 +1,34 @@
# frozen_string_literal: true
module Nokogiri
module XML
class DTD < Nokogiri::XML::Node
undef_method :attribute_nodes
undef_method :values
undef_method :content
undef_method :namespace
undef_method :namespace_definitions
undef_method :line if method_defined?(:line)
def keys
attributes.keys
end
def each
attributes.each do |key, value|
yield([key, value])
end
end
def html_dtd?
name.casecmp("html").zero?
end
def html5_dtd?
html_dtd? &&
external_id.nil? &&
(system_id.nil? || system_id == "about:legacy-compat")
end
end
end
end
@@ -0,0 +1,46 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# Represents the allowed content in an Element Declaration inside a DTD:
#
# <?xml version="1.0"?><?TEST-STYLE PIDATA?>
# <!DOCTYPE staff SYSTEM "staff.dtd" [
# <!ELEMENT div1 (head, (p | list | note)*, div2*)>
# ]>
# </root>
#
# ElementContent represents the binary tree inside the <!ELEMENT> tag shown above that lists the
# possible content for the div1 tag.
class ElementContent
include Nokogiri::XML::PP::Node
# Possible definitions of type
PCDATA = 1
ELEMENT = 2
SEQ = 3
OR = 4
# Possible content occurrences
ONCE = 1
OPT = 2
MULT = 3
PLUS = 4
attr_reader :document
###
# Get the children of this ElementContent node
def children
[c1, c2].compact
end
private
def inspect_attributes
[:prefix, :name, :type, :occur, :children]
end
end
end
end
@@ -0,0 +1,17 @@
# frozen_string_literal: true
module Nokogiri
module XML
class ElementDecl < Nokogiri::XML::Node
undef_method :namespace
undef_method :namespace_definitions
undef_method :line if method_defined?(:line)
private
def inspect_attributes
[:to_s]
end
end
end
end
@@ -0,0 +1,23 @@
# frozen_string_literal: true
module Nokogiri
module XML
class EntityDecl < Nokogiri::XML::Node
undef_method :attribute_nodes
undef_method :attributes
undef_method :namespace
undef_method :namespace_definitions
undef_method :line if method_defined?(:line)
def self.new(name, doc, *args)
doc.create_entity(name, *args)
end
private
def inspect_attributes
[:to_s]
end
end
end
end
@@ -0,0 +1,20 @@
# frozen_string_literal: true
module Nokogiri
module XML
class EntityReference < Nokogiri::XML::Node
def children
# libxml2 will create a malformed child node for predefined
# entities. because any use of that child is likely to cause a
# segfault, we shall pretend that it doesn't exist.
#
# see https://github.com/sparklemotion/nokogiri/issues/1238 for details
NodeSet.new(document)
end
def inspect_attributes
[:name]
end
end
end
end
@@ -0,0 +1,57 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
class Namespace
include Nokogiri::XML::PP::Node
attr_reader :document
#
# :call-seq: deconstruct_keys(array_of_names) → Hash
#
# Returns a hash describing the Namespace, to use in pattern matching.
#
# Valid keys and their values:
# - +prefix+ → (String, nil) The namespace's prefix, or +nil+ if there is no prefix (e.g., default namespace).
# - +href+ → (String) The namespace's URI
#
# *Example*
#
# doc = Nokogiri::XML.parse(<<~XML)
# <?xml version="1.0"?>
# <root xmlns="http://nokogiri.org/ns/default" xmlns:noko="http://nokogiri.org/ns/noko">
# <child1 foo="abc" noko:bar="def"/>
# <noko:child2 foo="qwe" noko:bar="rty"/>
# </root>
# XML
#
# doc.root.elements.first.namespace
# # => #(Namespace:0x35c { href = "http://nokogiri.org/ns/default" })
#
# doc.root.elements.first.namespace.deconstruct_keys([:prefix, :href])
# # => {:prefix=>nil, :href=>"http://nokogiri.org/ns/default"}
#
# doc.root.elements.last.namespace
# # => #(Namespace:0x370 {
# # prefix = "noko",
# # href = "http://nokogiri.org/ns/noko"
# # })
#
# doc.root.elements.last.namespace.deconstruct_keys([:prefix, :href])
# # => {:prefix=>"noko", :href=>"http://nokogiri.org/ns/noko"}
#
# Since v1.14.0
#
def deconstruct_keys(keys)
{ prefix: prefix, href: href }
end
private
def inspect_attributes
[:prefix, :href]
end
end
end
end
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,76 @@
# frozen_string_literal: true
module Nokogiri
module XML
class Node
###
# Save options for serializing nodes.
# See the method group entitled Node@Serialization+and+Generating+Output for usage.
class SaveOptions
# Format serialized xml
FORMAT = 1
# Do not include declarations
NO_DECLARATION = 2
# Do not include empty tags
NO_EMPTY_TAGS = 4
# Do not save XHTML
NO_XHTML = 8
# Save as XHTML
AS_XHTML = 16
# Save as XML
AS_XML = 32
# Save as HTML
AS_HTML = 64
if Nokogiri.jruby?
# Save builder created document
AS_BUILDER = 128
# the default for XML documents
DEFAULT_XML = AS_XML # https://github.com/sparklemotion/nokogiri/issues/#issue/415
# the default for HTML document
DEFAULT_HTML = NO_DECLARATION | NO_EMPTY_TAGS | AS_HTML
# the default for XHTML document
DEFAULT_XHTML = NO_DECLARATION | AS_XHTML
else
# the default for XML documents
DEFAULT_XML = FORMAT | AS_XML
# the default for HTML document
DEFAULT_HTML = FORMAT | NO_DECLARATION | NO_EMPTY_TAGS | AS_HTML
# the default for XHTML document
DEFAULT_XHTML = FORMAT | NO_DECLARATION | AS_XHTML
end
# Integer representation of the SaveOptions
attr_reader :options
# Create a new SaveOptions object with +options+
def initialize(options = 0)
@options = options
end
constants.each do |constant|
class_eval <<~RUBY, __FILE__, __LINE__ + 1
def #{constant.downcase}
@options |= #{constant}
self
end
def #{constant.downcase}?
#{constant} & @options == #{constant}
end
RUBY
end
alias_method :to_i, :options
def inspect
options = []
self.class.constants.each do |k|
options << k.downcase if send(:"#{k.downcase}?")
end
super.sub(/>$/, " " + options.join(", ") + ">")
end
end
end
end
end
@@ -0,0 +1,449 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
####
# A NodeSet is an Enumerable that contains a list of Nokogiri::XML::Node objects.
#
# Typically a NodeSet is returned as a result of searching a Document via
# Nokogiri::XML::Searchable#css or Nokogiri::XML::Searchable#xpath.
#
# Note that the `#dup` and `#clone` methods perform shallow copies; these methods do not copy
# the Nodes contained in the NodeSet (similar to how Array and other Enumerable classes work).
class NodeSet
include Nokogiri::XML::Searchable
include Enumerable
# The Document this NodeSet is associated with
attr_accessor :document
# Create a NodeSet with +document+ defaulting to +list+
def initialize(document, list = [])
@document = document
document.decorate(self)
list.each { |x| self << x }
yield self if block_given?
end
###
# Get the first element of the NodeSet.
def first(n = nil)
return self[0] unless n
list = []
[n, length].min.times { |i| list << self[i] }
list
end
###
# Get the last element of the NodeSet.
def last
self[-1]
end
###
# Is this NodeSet empty?
def empty?
length == 0
end
###
# Returns the index of the first node in self that is == to +node+ or meets the given block. Returns nil if no match is found.
def index(node = nil)
if node
warn("given block not used") if block_given?
each_with_index { |member, j| return j if member == node }
elsif block_given?
each_with_index { |member, j| return j if yield(member) }
end
nil
end
###
# Insert +datum+ before the first Node in this NodeSet
def before(datum)
first.before(datum)
end
###
# Insert +datum+ after the last Node in this NodeSet
def after(datum)
last.after(datum)
end
alias_method :<<, :push
alias_method :remove, :unlink
###
# call-seq: css *rules, [namespace-bindings, custom-pseudo-class]
#
# Search this node set for CSS +rules+. +rules+ must be one or more CSS
# selectors. For example:
#
# For more information see Nokogiri::XML::Searchable#css
def css(*args)
rules, handler, ns, _ = extract_params(args)
paths = css_rules_to_xpath(rules, ns)
inject(NodeSet.new(document)) do |set, node|
set + xpath_internal(node, paths, handler, ns, nil)
end
end
###
# call-seq: xpath *paths, [namespace-bindings, variable-bindings, custom-handler-class]
#
# Search this node set for XPath +paths+. +paths+ must be one or more XPath
# queries.
#
# For more information see Nokogiri::XML::Searchable#xpath
def xpath(*args)
paths, handler, ns, binds = extract_params(args)
inject(NodeSet.new(document)) do |set, node|
set + xpath_internal(node, paths, handler, ns, binds)
end
end
###
# call-seq: search *paths, [namespace-bindings, xpath-variable-bindings, custom-handler-class]
#
# Search this object for +paths+, and return only the first
# result. +paths+ must be one or more XPath or CSS queries.
#
# See Searchable#search for more information.
#
# Or, if passed an integer, index into the NodeSet:
#
# node_set.at(3) # same as node_set[3]
#
def at(*args)
if args.length == 1 && args.first.is_a?(Numeric)
return self[args.first]
end
super
end
alias_method :%, :at
###
# Filter this list for nodes that match +expr+
def filter(expr)
find_all { |node| node.matches?(expr) }
end
###
# Add the class attribute +name+ to all Node objects in the
# NodeSet.
#
# See Nokogiri::XML::Node#add_class for more information.
def add_class(name)
each do |el|
el.add_class(name)
end
self
end
###
# Append the class attribute +name+ to all Node objects in the
# NodeSet.
#
# See Nokogiri::XML::Node#append_class for more information.
def append_class(name)
each do |el|
el.append_class(name)
end
self
end
###
# Remove the class attribute +name+ from all Node objects in the
# NodeSet.
#
# See Nokogiri::XML::Node#remove_class for more information.
def remove_class(name = nil)
each do |el|
el.remove_class(name)
end
self
end
###
# Set attributes on each Node in the NodeSet, or get an
# attribute from the first Node in the NodeSet.
#
# To get an attribute from the first Node in a NodeSet:
#
# node_set.attr("href") # => "https://www.nokogiri.org"
#
# Note that an empty NodeSet will return nil when +#attr+ is called as a getter.
#
# To set an attribute on each node, +key+ can either be an
# attribute name, or a Hash of attribute names and values. When
# called as a setter, +#attr+ returns the NodeSet.
#
# If +key+ is an attribute name, then either +value+ or +block+
# must be passed.
#
# If +key+ is a Hash then attributes will be set for each
# key/value pair:
#
# node_set.attr("href" => "https://www.nokogiri.org", "class" => "member")
#
# If +value+ is passed, it will be used as the attribute value
# for all nodes:
#
# node_set.attr("href", "https://www.nokogiri.org")
#
# If +block+ is passed, it will be called on each Node object in
# the NodeSet and the return value used as the attribute value
# for that node:
#
# node_set.attr("class") { |node| node.name }
#
def attr(key, value = nil, &block)
unless key.is_a?(Hash) || (key && (value || block))
return first&.attribute(key)
end
hash = key.is_a?(Hash) ? key : { key => value }
hash.each do |k, v|
each do |node|
node[k] = v || yield(node)
end
end
self
end
alias_method :set, :attr
alias_method :attribute, :attr
###
# Remove the attributed named +name+ from all Node objects in the NodeSet
def remove_attr(name)
each { |el| el.delete(name) }
self
end
alias_method :remove_attribute, :remove_attr
###
# Iterate over each node, yielding to +block+
def each
return to_enum unless block_given?
0.upto(length - 1) do |x|
yield self[x]
end
self
end
###
# Get the inner text of all contained Node objects
#
# Note: This joins the text of all Node objects in the NodeSet:
#
# doc = Nokogiri::XML('<xml><a><d>foo</d><d>bar</d></a></xml>')
# doc.css('d').text # => "foobar"
#
# Instead, if you want to return the text of all nodes in the NodeSet:
#
# doc.css('d').map(&:text) # => ["foo", "bar"]
#
# See Nokogiri::XML::Node#content for more information.
def inner_text
collect(&:inner_text).join("")
end
alias_method :text, :inner_text
###
# Get the inner html of all contained Node objects
def inner_html(*args)
collect { |j| j.inner_html(*args) }.join("")
end
# :call-seq:
# wrap(markup) -> self
# wrap(node) -> self
#
# Wrap each member of this NodeSet with the node parsed from +markup+ or a dup of the +node+.
#
# [Parameters]
# - *markup* (String)
# Markup that is parsed, once per member of the NodeSet, and used as the wrapper. Each
# node's parent, if it exists, is used as the context node for parsing; otherwise the
# associated document is used. If the parsed fragment has multiple roots, the first root
# node is used as the wrapper.
# - *node* (Nokogiri::XML::Node)
# An element that is `#dup`ed and used as the wrapper.
#
# [Returns] +self+, to support chaining.
#
# ⚠ Note that if a +String+ is passed, the markup will be parsed <b>once per node</b> in the
# NodeSet. You can avoid this overhead in cases where you know exactly the wrapper you wish to
# use by passing a +Node+ instead.
#
# Also see Node#wrap
#
# *Example* with a +String+ argument:
#
# doc = Nokogiri::HTML5(<<~HTML)
# <html><body>
# <a>a</a>
# <a>b</a>
# <a>c</a>
# <a>d</a>
# </body></html>
# HTML
# doc.css("a").wrap("<div></div>")
# doc.to_html
# # => <html><head></head><body>
# # <div><a>a</a></div>
# # <div><a>b</a></div>
# # <div><a>c</a></div>
# # <div><a>d</a></div>
# # </body></html>
#
# *Example* with a +Node+ argument
#
# 💡 Note that this is faster than the equivalent call passing a +String+ because it avoids
# having to reparse the wrapper markup for each node.
#
# doc = Nokogiri::HTML5(<<~HTML)
# <html><body>
# <a>a</a>
# <a>b</a>
# <a>c</a>
# <a>d</a>
# </body></html>
# HTML
# doc.css("a").wrap(doc.create_element("div"))
# doc.to_html
# # => <html><head></head><body>
# # <div><a>a</a></div>
# # <div><a>b</a></div>
# # <div><a>c</a></div>
# # <div><a>d</a></div>
# # </body></html>
#
def wrap(node_or_tags)
map { |node| node.wrap(node_or_tags) }
self
end
###
# Convert this NodeSet to a string.
def to_s
map(&:to_s).join
end
###
# Convert this NodeSet to HTML
def to_html(*args)
if Nokogiri.jruby?
options = args.first.is_a?(Hash) ? args.shift : {}
options[:save_with] ||= Node::SaveOptions::DEFAULT_HTML
args.insert(0, options)
end
if empty?
encoding = (args.first.is_a?(Hash) ? args.first[:encoding] : nil)
encoding ||= document.encoding
encoding.nil? ? "" : "".encode(encoding)
else
map { |x| x.to_html(*args) }.join
end
end
###
# Convert this NodeSet to XHTML
def to_xhtml(*args)
map { |x| x.to_xhtml(*args) }.join
end
###
# Convert this NodeSet to XML
def to_xml(*args)
map { |x| x.to_xml(*args) }.join
end
alias_method :size, :length
alias_method :to_ary, :to_a
###
# Removes the last element from set and returns it, or +nil+ if
# the set is empty
def pop
return if length == 0
delete(last)
end
###
# Returns the first element of the NodeSet and removes it. Returns
# +nil+ if the set is empty.
def shift
return if length == 0
delete(first)
end
###
# Equality -- Two NodeSets are equal if the contain the same number
# of elements and if each element is equal to the corresponding
# element in the other NodeSet
def ==(other)
return false unless other.is_a?(Nokogiri::XML::NodeSet)
return false unless length == other.length
each_with_index do |node, i|
return false unless node == other[i]
end
true
end
###
# Returns a new NodeSet containing all the children of all the nodes in
# the NodeSet
def children
node_set = NodeSet.new(document)
each do |node|
node.children.each { |n| node_set.push(n) }
end
node_set
end
###
# Returns a new NodeSet containing all the nodes in the NodeSet
# in reverse order
def reverse
node_set = NodeSet.new(document)
(length - 1).downto(0) do |x|
node_set.push(self[x])
end
node_set
end
###
# Return a nicely formatted string representation
def inspect
"[#{map(&:inspect).join(", ")}]"
end
alias_method :+, :|
#
# :call-seq: deconstruct() → Array
#
# Returns the members of this NodeSet as an array, to use in pattern matching.
#
# Since v1.14.0
#
def deconstruct
to_a
end
IMPLIED_XPATH_CONTEXTS = [".//", "self::"].freeze # :nodoc:
end
end
end
@@ -0,0 +1,19 @@
# frozen_string_literal: true
module Nokogiri
module XML
# Struct representing an {XML Schema Notation}[https://www.w3.org/TR/xml/#Notations]
class Notation < Struct.new(:name, :public_id, :system_id)
# dead comment to ensure rdoc processing
# :attr: name (String)
# The name for the element.
# :attr: public_id (String)
# The URI corresponding to the public identifier
# :attr: system_id (String,nil)
# The URI corresponding to the system identifier
end
end
end
@@ -0,0 +1,213 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
# Options that control the parsing behavior for XML::Document, XML::DocumentFragment,
# HTML4::Document, HTML4::DocumentFragment, XSLT::Stylesheet, and XML::Schema.
#
# These options directly expose libxml2's parse options, which are all boolean in the sense that
# an option is "on" or "off".
#
# 💡 Note that HTML5 parsing has a separate, orthogonal set of options due to the nature of the
# HTML5 specification. See Nokogiri::HTML5.
#
# ⚠ Not all parse options are supported on JRuby. Nokogiri will attempt to invoke the equivalent
# behavior in Xerces/NekoHTML on JRuby when it's possible.
#
# == Setting and unsetting parse options
#
# You can build your own combinations of parse options by using any of the following methods:
#
# [ParseOptions method chaining]
#
# Every option has an equivalent method in lowercase. You can chain these methods together to
# set various combinations.
#
# # Set the HUGE & PEDANTIC options
# po = Nokogiri::XML::ParseOptions.new.huge.pedantic
# doc = Nokogiri::XML::Document.parse(xml, nil, nil, po)
#
# Every option has an equivalent <code>no{option}</code> method in lowercase. You can call these
# methods on an instance of ParseOptions to unset the option.
#
# # Set the HUGE & PEDANTIC options
# po = Nokogiri::XML::ParseOptions.new.huge.pedantic
#
# # later we want to modify the options
# po.nohuge # Unset the HUGE option
# po.nopedantic # Unset the PEDANTIC option
#
# 💡 Note that some options begin with "no" leading to the logical but perhaps unintuitive
# double negative:
#
# po.nocdata # Set the NOCDATA parse option
# po.nonocdata # Unset the NOCDATA parse option
#
# 💡 Note that negation is not available for STRICT, which is itself a negation of all other
# features.
#
#
# [Using Ruby Blocks]
#
# Most parsing methods will accept a block for configuration of parse options, and we
# recommend chaining the setter methods:
#
# doc = Nokogiri::XML::Document.parse(xml) { |config| config.huge.pedantic }
#
#
# [ParseOptions constants]
#
# You can also use the constants declared under Nokogiri::XML::ParseOptions to set various
# combinations. They are bits in a bitmask, and so can be combined with bitwise operators:
#
# po = Nokogiri::XML::ParseOptions.new(Nokogiri::XML::ParseOptions::HUGE | Nokogiri::XML::ParseOptions::PEDANTIC)
# doc = Nokogiri::XML::Document.parse(xml, nil, nil, po)
#
class ParseOptions
# Strict parsing
STRICT = 0
# Recover from errors. On by default for XML::Document, XML::DocumentFragment,
# HTML4::Document, HTML4::DocumentFragment, XSLT::Stylesheet, and XML::Schema.
RECOVER = 1 << 0
# Substitute entities. Off by default.
#
# ⚠ This option enables entity substitution, contrary to what the name implies.
#
# ⚠ <b>It is UNSAFE to set this option</b> when parsing untrusted documents.
NOENT = 1 << 1
# Load external subsets. On by default for XSLT::Stylesheet.
#
# ⚠ <b>It is UNSAFE to set this option</b> when parsing untrusted documents.
DTDLOAD = 1 << 2
# Default DTD attributes. On by default for XSLT::Stylesheet.
DTDATTR = 1 << 3
# Validate with the DTD. Off by default.
DTDVALID = 1 << 4
# Suppress error reports. On by default for HTML4::Document and HTML4::DocumentFragment
NOERROR = 1 << 5
# Suppress warning reports. On by default for HTML4::Document and HTML4::DocumentFragment
NOWARNING = 1 << 6
# Enable pedantic error reporting. Off by default.
PEDANTIC = 1 << 7
# Remove blank nodes. Off by default.
NOBLANKS = 1 << 8
# Use the SAX1 interface internally. Off by default.
SAX1 = 1 << 9
# Implement XInclude substitution. Off by default.
XINCLUDE = 1 << 10
# Forbid network access. On by default for XML::Document, XML::DocumentFragment,
# HTML4::Document, HTML4::DocumentFragment, XSLT::Stylesheet, and XML::Schema.
#
# ⚠ <b>It is UNSAFE to unset this option</b> when parsing untrusted documents.
NONET = 1 << 11
# Do not reuse the context dictionary. Off by default.
NODICT = 1 << 12
# Remove redundant namespaces declarations. Off by default.
NSCLEAN = 1 << 13
# Merge CDATA as text nodes. On by default for XSLT::Stylesheet.
NOCDATA = 1 << 14
# Do not generate XInclude START/END nodes. Off by default.
NOXINCNODE = 1 << 15
# Compact small text nodes. Off by default.
#
# ⚠ No modification of the DOM tree is allowed after parsing. libxml2 may crash if you try to
# modify the tree.
COMPACT = 1 << 16
# Parse using XML-1.0 before update 5. Off by default
OLD10 = 1 << 17
# Do not fixup XInclude xml:base uris. Off by default
NOBASEFIX = 1 << 18
# Relax any hardcoded limit from the parser. Off by default.
#
# ⚠ <b>It is UNSAFE to set this option</b> when parsing untrusted documents.
HUGE = 1 << 19
# Support line numbers up to <code>long int</code> (default is a <code>short int</code>). On
# by default for for XML::Document, XML::DocumentFragment, HTML4::Document,
# HTML4::DocumentFragment, XSLT::Stylesheet, and XML::Schema.
BIG_LINES = 1 << 22
# The options mask used by default for parsing XML::Document and XML::DocumentFragment
DEFAULT_XML = RECOVER | NONET | BIG_LINES
# The options mask used by default used for parsing XSLT::Stylesheet
DEFAULT_XSLT = RECOVER | NONET | NOENT | DTDLOAD | DTDATTR | NOCDATA | BIG_LINES
# The options mask used by default used for parsing HTML4::Document and HTML4::DocumentFragment
DEFAULT_HTML = RECOVER | NOERROR | NOWARNING | NONET | BIG_LINES
# The options mask used by default used for parsing XML::Schema
DEFAULT_SCHEMA = NONET | BIG_LINES
attr_accessor :options
def initialize(options = STRICT)
@options = options
end
constants.each do |constant|
next if constant.to_sym == :STRICT
class_eval <<~RUBY, __FILE__, __LINE__ + 1
def #{constant.downcase}
@options |= #{constant}
self
end
def no#{constant.downcase}
@options &= ~#{constant}
self
end
def #{constant.downcase}?
#{constant} & @options == #{constant}
end
RUBY
end
def strict
@options &= ~RECOVER
self
end
def strict?
@options & RECOVER == STRICT
end
def ==(other)
other.to_i == to_i
end
alias_method :to_i, :options
def inspect
options = []
self.class.constants.each do |k|
options << k.downcase if send(:"#{k.downcase}?")
end
super.sub(/>$/, " " + options.join(", ") + ">")
end
end
end
end
@@ -0,0 +1,4 @@
# frozen_string_literal: true
require_relative "pp/node"
require_relative "pp/character_data"
@@ -0,0 +1,21 @@
# frozen_string_literal: true
module Nokogiri
module XML
# :nodoc: all
module PP
module CharacterData
def pretty_print(pp)
nice_name = self.class.name.split("::").last
pp.group(2, "#(#{nice_name} ", ")") do
pp.pp(text)
end
end
def inspect
"#<#{self.class.name}:#{format("0x%x", object_id)} #{text.inspect}>"
end
end
end
end
end
@@ -0,0 +1,73 @@
# frozen_string_literal: true
module Nokogiri
module XML
# :nodoc: all
module PP
module Node
COLLECTIONS = [:attribute_nodes, :children]
def inspect
# handle the case where an exception is thrown during object construction
if respond_to?(:data_ptr?) && !data_ptr?
return "#<#{self.class}:#{format("0x%x", object_id)} (no data)>"
end
attributes = inspect_attributes.reject do |x|
attribute = send(x)
!attribute || (attribute.respond_to?(:empty?) && attribute.empty?)
rescue NoMethodError
true
end
attributes = if inspect_attributes.length == 1
send(attributes.first).inspect
else
attributes.map do |attribute|
"#{attribute}=#{send(attribute).inspect}"
end.join(" ")
end
"#<#{self.class}:#{format("0x%x", object_id)} #{attributes}>"
end
def pretty_print(pp)
nice_name = self.class.name.split("::").last
pp.group(2, "#(#{nice_name}:#{format("0x%x", object_id)} {", "})") do
pp.breakable
attrs = inspect_attributes.filter_map do |t|
[t, send(t)] if respond_to?(t)
end.find_all do |x|
if x.last
if COLLECTIONS.include?(x.first)
!x.last.empty?
else
true
end
end
end
if inspect_attributes.length == 1
pp.pp(attrs.first.last)
else
pp.seplist(attrs) do |v|
if COLLECTIONS.include?(v.first)
pp.group(2, "#{v.first} = [", "]") do
pp.breakable
pp.seplist(v.last) do |item|
pp.pp(item)
end
end
else
pp.text("#{v.first} = ")
pp.pp(v.last)
end
end
end
pp.breakable
end
end
end
end
end
end
@@ -0,0 +1,11 @@
# frozen_string_literal: true
module Nokogiri
module XML
class ProcessingInstruction < Node
def initialize(document, name, content)
super(document, name)
end
end
end
end
@@ -0,0 +1,139 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# The Reader parser allows you to effectively pull parse an \XML document. Once instantiated,
# call Nokogiri::XML::Reader#each to iterate over each node.
#
# Nokogiri::XML::Reader parses an \XML document similar to the way a cursor would move. The
# Reader is given an \XML document, and yields nodes to an each block.
#
# The Reader parser might be good for when you need the speed and low memory usage of a \SAX
# parser, but do not want to write a SAX::Document handler.
#
# Here is an example of usage:
#
# reader = Nokogiri::XML::Reader.new <<~XML
# <x xmlns:tenderlove='http://tenderlovemaking.com/'>
# <tenderlove:foo awesome='true'>snuggles!</tenderlove:foo>
# </x>
# XML
#
# reader.each do |node|
# # node is an instance of Nokogiri::XML::Reader
# puts node.name
# end
#
# ⚠ Nokogiri::XML::Reader#each can only be called once! Once the cursor moves through the entire
# document, you must parse the document again. It may be better to capture all information you
# need during a single iteration.
#
# ⚠ libxml2 does not support error recovery in the Reader parser. The +RECOVER+ ParseOption is
# ignored. If a syntax error is encountered during parsing, an exception will be raised.
class Reader
include Enumerable
TYPE_NONE = 0
# Element node type
TYPE_ELEMENT = 1
# Attribute node type
TYPE_ATTRIBUTE = 2
# Text node type
TYPE_TEXT = 3
# CDATA node type
TYPE_CDATA = 4
# Entity Reference node type
TYPE_ENTITY_REFERENCE = 5
# Entity node type
TYPE_ENTITY = 6
# PI node type
TYPE_PROCESSING_INSTRUCTION = 7
# Comment node type
TYPE_COMMENT = 8
# Document node type
TYPE_DOCUMENT = 9
# Document Type node type
TYPE_DOCUMENT_TYPE = 10
# Document Fragment node type
TYPE_DOCUMENT_FRAGMENT = 11
# Notation node type
TYPE_NOTATION = 12
# Whitespace node type
TYPE_WHITESPACE = 13
# Significant Whitespace node type
TYPE_SIGNIFICANT_WHITESPACE = 14
# Element end node type
TYPE_END_ELEMENT = 15
# Entity end node type
TYPE_END_ENTITY = 16
# \XML Declaration node type
TYPE_XML_DECLARATION = 17
# A list of errors encountered while parsing
attr_accessor :errors
# The \XML source
attr_reader :source
alias_method :self_closing?, :empty_element?
# :call-seq:
# Reader.new(input) { |options| ... } → Reader
# Reader.new(input, url:, encoding:, options:) { |options| ... } → Reader
#
# Create a new Reader to parse an \XML document.
#
# [Required Parameters]
# - +input+ (String | IO): The \XML document to parse.
#
# [Optional Parameters]
# - +url:+ (String) The base URL of the document.
# - +encoding:+ (String) The name of the encoding of the document.
# - +options:+ (Integer | ParseOptions) Options to control the parser behavior.
# Defaults to +ParseOptions::STRICT+.
#
# [Yields]
# If present, the block will be passed a Nokogiri::XML::ParseOptions object to modify before
# the fragment is parsed. See Nokogiri::XML::ParseOptions for more information.
def self.new(
string_or_io,
url_ = nil, encoding_ = nil, options_ = ParseOptions::STRICT,
url: url_, encoding: encoding_, options: options_
)
options = Nokogiri::XML::ParseOptions.new(options) if Integer === options
yield options if block_given?
if string_or_io.respond_to?(:read)
return Reader.from_io(string_or_io, url, encoding, options.to_i)
end
Reader.from_memory(string_or_io, url, encoding, options.to_i)
end
private def initialize(source, url = nil, encoding = nil) # :nodoc:
@source = source
@errors = []
@encoding = encoding
end
# Get the attributes and namespaces of the current node as a Hash.
#
# This is the union of Reader#attribute_hash and Reader#namespaces
#
# [Returns]
# (Hash<String, String>) Attribute names and values, and namespace prefixes and hrefs.
def attributes
attribute_hash.merge(namespaces)
end
###
# Move the cursor through the document yielding the cursor to the block
def each
while (cursor = read)
yield cursor
end
end
end
end
end
@@ -0,0 +1,75 @@
# frozen_string_literal: true
module Nokogiri
module XML
class << self
# :call-seq:
# RelaxNG(input) → Nokogiri::XML::RelaxNG
# RelaxNG(input, options:) → Nokogiri::XML::RelaxNG
#
# Convenience method for Nokogiri::XML::RelaxNG.new
def RelaxNG(...)
RelaxNG.new(...)
end
end
# Nokogiri::XML::RelaxNG is used for validating \XML against a RELAX NG schema definition.
#
# 🛡 <b>Do not use this class for untrusted schema documents.</b> RELAX NG input is always
# treated as *trusted*, meaning that the underlying parsing libraries <b>will access network
# resources</b>. This is counter to Nokogiri's "untrusted by default" security policy, but is an
# unfortunate limitation of the underlying libraries.
#
# *Example:* Determine whether an \XML document is valid.
#
# schema = Nokogiri::XML::RelaxNG.new(File.read(RELAX_NG_FILE))
# doc = Nokogiri::XML::Document.parse(File.read(XML_FILE))
# schema.valid?(doc) # Boolean
#
# *Example:* Validate an \XML document against a \RelaxNG schema, and capture any errors that are found.
#
# schema = Nokogiri::XML::RelaxNG.new(File.open(RELAX_NG_FILE))
# doc = Nokogiri::XML::Document.parse(File.open(XML_FILE))
# errors = schema.validate(doc) # Array<SyntaxError>
#
# *Example:* Validate an \XML document using a Document containing a RELAX NG schema definition.
#
# schema_doc = Nokogiri::XML::Document.parse(File.read(RELAX_NG_FILE))
# schema = Nokogiri::XML::RelaxNG.from_document(schema_doc)
# doc = Nokogiri::XML::Document.parse(File.open(XML_FILE))
# schema.valid?(doc) # Boolean
#
class RelaxNG < Nokogiri::XML::Schema
# :call-seq:
# new(input) → Nokogiri::XML::RelaxNG
# new(input, options:) → Nokogiri::XML::RelaxNG
#
# Parse a RELAX NG schema definition from a String or IO to create a new Nokogiri::XML::RelaxNG.
#
# [Parameters]
# - +input+ (String | IO) RELAX NG schema definition
# - +options:+ (Nokogiri::XML::ParseOptions)
# Defaults to Nokogiri::XML::ParseOptions::DEFAULT_SCHEMA ⚠ Unused
#
# [Returns] Nokogiri::XML::RelaxNG
#
# ⚠ +parse_options+ is currently unused by this method and is present only as a placeholder for
# future functionality.
#
# Also see convenience method Nokogiri::XML::RelaxNG()
def self.new(input, parse_options_ = ParseOptions::DEFAULT_SCHEMA, options: parse_options_)
from_document(Nokogiri::XML::Document.parse(input), options)
end
# :call-seq:
# read_memory(input) → Nokogiri::XML::RelaxNG
# read_memory(input, options:) → Nokogiri::XML::RelaxNG
#
# Convenience method for Nokogiri::XML::RelaxNG.new.
def self.read_memory(...)
# TODO deprecate this method
new(...)
end
end
end
end
@@ -0,0 +1,54 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# SAX Parsers are event-driven parsers.
#
# Two SAX parsers for XML are available, a parser that reads from a string or IO object as it
# feels necessary, and a parser that you explicitly feed XML in chunks. If you want to let
# Nokogiri deal with reading your XML, use the Nokogiri::XML::SAX::Parser. If you want to have
# fine grain control over the XML input, use the Nokogiri::XML::SAX::PushParser.
#
# If you want to do SAX style parsing of HTML, check out Nokogiri::HTML4::SAX.
#
# The basic way a SAX style parser works is by creating a parser, telling the parser about the
# events we're interested in, then giving the parser some XML to process. The parser will notify
# you when it encounters events you said you would like to know about.
#
# To register for events, subclass Nokogiri::XML::SAX::Document and implement the methods for
# which you would like notification.
#
# For example, if I want to be notified when a document ends, and when an element starts, I
# would write a class like this:
#
# class MyHandler < Nokogiri::XML::SAX::Document
# def end_document
# puts "the document has ended"
# end
#
# def start_element name, attributes = []
# puts "#{name} started"
# end
# end
#
# Then I would instantiate a SAX parser with this document, and feed the parser some XML
#
# # Create a new parser
# parser = Nokogiri::XML::SAX::Parser.new(MyHandler.new)
#
# # Feed the parser some XML
# parser.parse(File.open(ARGV[0]))
#
# Now my document handler will be called when each node starts, and when then document ends. To
# see what kinds of events are available, take a look at Nokogiri::XML::SAX::Document.
#
module SAX
end
end
end
require_relative "sax/document"
require_relative "sax/parser_context"
require_relative "sax/parser"
require_relative "sax/push_parser"
@@ -0,0 +1,258 @@
# frozen_string_literal: true
module Nokogiri
module XML
module SAX
# :markup: markdown
#
# The SAX::Document class is used for registering types of events you are interested in
# handling. All of the methods on this class are available as possible events while parsing an
# \XML document. To register for any particular event, subclass this class and implement the
# methods you are interested in knowing about.
#
# To only be notified about start and end element events, write a class like this:
#
# class MyHandler < Nokogiri::XML::SAX::Document
# def start_element name, attrs = []
# puts "#{name} started!"
# end
#
# def end_element name
# puts "#{name} ended"
# end
# end
#
# You can use this event handler for any SAX-style parser included with Nokogiri.
#
# See also:
#
# - Nokogiri::XML::SAX
# - Nokogiri::HTML4::SAX
#
# ### Entity Handling
#
# ⚠ Entity handling is complicated in a SAX parser! Please read this section carefully if
# you're not getting the behavior you expect.
#
# Entities will be reported to the user via callbacks to #characters, to #reference, or
# possibly to both. The behavior is determined by a combination of _entity type_ and the value
# of ParserContext#replace_entities. (Recall that the default value of
# ParserContext#replace_entities is `false`.)
#
# ⚠ <b>It is UNSAFE to set ParserContext#replace_entities to `true`</b> when parsing untrusted
# documents.
#
# 💡 For more information on entity types, see [Wikipedia's page on
# DTDs](https://en.wikipedia.org/wiki/Document_type_definition#Entity_declarations).
#
# | Entity type | #characters | #reference |
# |--------------------------------------|------------------------------------|-------------------------------------|
# | Char ref (e.g., <tt>&#146;</tt>) | always | never |
# | Predefined (e.g., <tt>&amp;</tt>) | always | never |
# | Undeclared † | never | <tt>#replace_entities == false</tt> |
# | Internal | always | <tt>#replace_entities == false</tt> |
# | External † | <tt>#replace_entities == true</tt> | <tt>#replace_entities == false</tt> |
#
# &nbsp;
#
# † In the case where the replacement text for the entity is unknown (e.g., an undeclared entity
# or an external entity that could not be resolved because of network issues), then the
# replacement text will not be reported. If ParserContext#replace_entities is `true`, this
# means the #characters callback will not be invoked. If ParserContext#replace_entities is
# `false`, then the #reference callback will be invoked, but with `nil` for the `content`
# argument.
#
class Document
###
# Called when an \XML declaration is parsed.
#
# [Parameters]
# - +version+ (String) the version attribute
# - +encoding+ (String, nil) the encoding of the document if present, else +nil+
# - +standalone+ ("yes", "no", nil) the standalone attribute if present, else +nil+
def xmldecl(version, encoding, standalone)
end
###
# Called when document starts parsing.
def start_document
end
###
# Called when document ends parsing.
def end_document
end
###
# Called at the beginning of an element.
#
# [Parameters]
# - +name+ (String) the name of the element
# - +attrs+ (Array<Array<String>>) an assoc list of namespace declarations and attributes, e.g.:
# [ ["xmlns:foo", "http://sample.net"], ["size", "large"] ]
#
# 💡If you're dealing with XML and need to handle namespaces, use the
# #start_element_namespace method instead.
#
# Note that the element namespace and any attribute namespaces are not provided, and so any
# namespaced elements or attributes will be returned as strings including the prefix:
#
# parser.parse(<<~XML)
# <root xmlns:foo='http://foo.example.com/' xmlns='http://example.com/'>
# <foo:bar foo:quux="xxx">hello world</foo:bar>
# </root>
# XML
#
# assert_pattern do
# parser.document.start_elements => [
# ["root", [["xmlns:foo", "http://foo.example.com/"], ["xmlns", "http://example.com/"]]],
# ["foo:bar", [["foo:quux", "xxx"]]],
# ]
# end
#
def start_element(name, attrs = [])
end
###
# Called at the end of an element.
#
# [Parameters]
# - +name+ (String) the name of the element being closed
#
def end_element(name)
end
###
# Called at the beginning of an element.
#
# [Parameters]
# - +name+ (String) is the name of the element
# - +attrs+ (Array<Attribute>) is an array of structs with the following properties:
# - +localname+ (String) the local name of the attribute
# - +value+ (String) the value of the attribute
# - +prefix+ (String, nil) the namespace prefix of the attribute
# - +uri+ (String, nil) the namespace URI of the attribute
# - +prefix+ (String, nil) is the namespace prefix for the element
# - +uri+ (String, nil) is the associated URI for the element's namespace
# - +ns+ (Array<Array<String, String>>) is an assoc list of namespace declarations on the element
#
# 💡If you're dealing with HTML or don't care about namespaces, try #start_element instead.
#
# [Example]
# it "start_elements_namespace is called with namespaced attributes" do
# parser.parse(<<~XML)
# <root xmlns:foo='http://foo.example.com/'>
# <foo:a foo:bar='hello' />
# </root>
# XML
#
# assert_pattern do
# parser.document.start_elements_namespace => [
# [
# "root",
# [],
# nil, nil,
# [["foo", "http://foo.example.com/"]], # namespace declarations
# ], [
# "a",
# [Nokogiri::XML::SAX::Parser::Attribute(localname: "bar", prefix: "foo", uri: "http://foo.example.com/", value: "hello")], # prefixed attribute
# "foo", "http://foo.example.com/", # prefix and uri for the "a" element
# [],
# ]
# ]
# end
# end
#
def start_element_namespace(name, attrs = [], prefix = nil, uri = nil, ns = []) # rubocop:disable Metrics/ParameterLists
# Deal with SAX v1 interface
name = [prefix, name].compact.join(":")
attributes = ns.map do |ns_prefix, ns_uri|
[["xmlns", ns_prefix].compact.join(":"), ns_uri]
end + attrs.map do |attr|
[[attr.prefix, attr.localname].compact.join(":"), attr.value]
end
start_element(name, attributes)
end
###
# Called at the end of an element.
#
# [Parameters]
# - +name+ (String) is the name of the element
# - +prefix+ (String, nil) is the namespace prefix for the element
# - +uri+ (String, nil) is the associated URI for the element's namespace
#
def end_element_namespace(name, prefix = nil, uri = nil)
# Deal with SAX v1 interface
end_element([prefix, name].compact.join(":"))
end
###
# Called when character data is parsed, and for parsed entities when
# ParserContext#replace_entities is +true+.
#
# [Parameters]
# - +string+ contains the character data or entity replacement text
#
# ⚠ Please see Document@Entity+Handling for important information about how entities are handled.
#
# ⚠ This method might be called multiple times for a contiguous string of characters.
#
def characters(string)
end
###
# Called when a parsed entity is referenced and not replaced.
#
# [Parameters]
# - +name+ (String) is the name of the entity
# - +content+ (String, nil) is the replacement text for the entity, if known
#
# ⚠ Please see Document@Entity+Handling for important information about how entities are handled.
#
# ⚠ An internal entity may result in a call to both #characters and #reference.
#
# Since v1.17.0
#
def reference(name, content)
end
###
# Called when comments are encountered
# [Parameters]
# - +string+ contains the comment data
def comment(string)
end
###
# Called on document warnings
# [Parameters]
# - +string+ contains the warning
def warning(string)
end
###
# Called on document errors
# [Parameters]
# - +string+ contains the error
def error(string)
end
###
# Called when cdata blocks are found
# [Parameters]
# - +string+ contains the cdata content
def cdata_block(string)
end
###
# Called when processing instructions are found
# [Parameters]
# - +name+ is the target of the instruction
# - +content+ is the value of the instruction
def processing_instruction(name, content)
end
end
end
end
end
@@ -0,0 +1,199 @@
# frozen_string_literal: true
module Nokogiri
module XML
module SAX
###
# This parser is a SAX style parser that reads its input as it deems necessary. The parser
# takes a Nokogiri::XML::SAX::Document, an optional encoding, then given an XML input, sends
# messages to the Nokogiri::XML::SAX::Document.
#
# Here is an example of using this parser:
#
# # Create a subclass of Nokogiri::XML::SAX::Document and implement
# # the events we care about:
# class MyHandler < Nokogiri::XML::SAX::Document
# def start_element name, attrs = []
# puts "starting: #{name}"
# end
#
# def end_element name
# puts "ending: #{name}"
# end
# end
#
# parser = Nokogiri::XML::SAX::Parser.new(MyHandler.new)
#
# # Hand an IO object to the parser, which will read the XML from the IO.
# File.open(path_to_xml) do |f|
# parser.parse(f)
# end
#
# For more information about \SAX parsers, see Nokogiri::XML::SAX.
#
# Also see Nokogiri::XML::SAX::Document for the available events.
#
# For \HTML documents, use the subclass Nokogiri::HTML4::SAX::Parser.
#
class Parser
# to dynamically resolve ParserContext in inherited methods
include Nokogiri::ClassResolver
# Structure used for marshalling attributes for some callbacks in XML::SAX::Document.
class Attribute < Struct.new(:localname, :prefix, :uri, :value)
end
ENCODINGS = { # :nodoc:
"NONE" => 0, # No char encoding detected
"UTF-8" => 1, # UTF-8
"UTF16LE" => 2, # UTF-16 little endian
"UTF16BE" => 3, # UTF-16 big endian
"UCS4LE" => 4, # UCS-4 little endian
"UCS4BE" => 5, # UCS-4 big endian
"EBCDIC" => 6, # EBCDIC uh!
"UCS4-2143" => 7, # UCS-4 unusual ordering
"UCS4-3412" => 8, # UCS-4 unusual ordering
"UCS2" => 9, # UCS-2
"ISO-8859-1" => 10, # ISO-8859-1 ISO Latin 1
"ISO-8859-2" => 11, # ISO-8859-2 ISO Latin 2
"ISO-8859-3" => 12, # ISO-8859-3
"ISO-8859-4" => 13, # ISO-8859-4
"ISO-8859-5" => 14, # ISO-8859-5
"ISO-8859-6" => 15, # ISO-8859-6
"ISO-8859-7" => 16, # ISO-8859-7
"ISO-8859-8" => 17, # ISO-8859-8
"ISO-8859-9" => 18, # ISO-8859-9
"ISO-2022-JP" => 19, # ISO-2022-JP
"SHIFT-JIS" => 20, # Shift_JIS
"EUC-JP" => 21, # EUC-JP
"ASCII" => 22, # pure ASCII
}
REVERSE_ENCODINGS = ENCODINGS.invert # :nodoc:
deprecate_constant :ENCODINGS
# The Nokogiri::XML::SAX::Document where events will be sent.
attr_accessor :document
# The encoding beings used for this document.
attr_accessor :encoding
###
# :call-seq:
# new ⇒ SAX::Parser
# new(handler) ⇒ SAX::Parser
# new(handler, encoding) ⇒ SAX::Parser
#
# Create a new Parser.
#
# [Parameters]
# - +handler+ (optional Nokogiri::XML::SAX::Document) The document that will receive
# events. Will create a new Nokogiri::XML::SAX::Document if not given, which is accessible
# through the #document attribute.
# - +encoding+ (optional Encoding, String, nil) An Encoding or encoding name to use when
# parsing the input. (default +nil+ for auto-detection)
#
def initialize(doc = Nokogiri::XML::SAX::Document.new, encoding = nil)
@encoding = encoding
@document = doc
@warned = false
initialize_native unless Nokogiri.jruby?
end
###
# :call-seq:
# parse(input) { |parser_context| ... }
#
# Parse the input, sending events to the SAX::Document at #document.
#
# [Parameters]
# - +input+ (String, IO) The input to parse.
#
# If +input+ quacks like a readable IO object, this method forwards to Parser.parse_io,
# otherwise it forwards to Parser.parse_memory.
#
# [Yields]
# If a block is given, the underlying ParserContext object will be yielded. This can be used
# to set options on the parser context before parsing begins.
#
def parse(input, &block)
if input.respond_to?(:read) && input.respond_to?(:close)
parse_io(input, &block)
else
parse_memory(input, &block)
end
end
###
# :call-seq:
# parse_io(io) { |parser_context| ... }
# parse_io(io, encoding) { |parser_context| ... }
#
# Parse an input stream.
#
# [Parameters]
# - +io+ (IO) The readable IO object from which to read input
# - +encoding+ (optional Encoding, String, nil) An Encoding or encoding name to use when
# parsing the input, or +nil+ for auto-detection. (default #encoding)
#
# [Yields]
# If a block is given, the underlying ParserContext object will be yielded. This can be used
# to set options on the parser context before parsing begins.
#
def parse_io(io, encoding = @encoding)
ctx = related_class("ParserContext").io(io, encoding)
yield ctx if block_given?
ctx.parse_with(self)
end
###
# :call-seq:
# parse_memory(input) { |parser_context| ... }
# parse_memory(input, encoding) { |parser_context| ... }
#
# Parse an input string.
#
# [Parameters]
# - +input+ (String) The input string to be parsed.
# - +encoding+ (optional Encoding, String, nil) An Encoding or encoding name to use when
# parsing the input, or +nil+ for auto-detection. (default #encoding)
#
# [Yields]
# If a block is given, the underlying ParserContext object will be yielded. This can be used
# to set options on the parser context before parsing begins.
#
def parse_memory(input, encoding = @encoding)
ctx = related_class("ParserContext").memory(input, encoding)
yield ctx if block_given?
ctx.parse_with(self)
end
###
# :call-seq:
# parse_file(filename) { |parser_context| ... }
# parse_file(filename, encoding) { |parser_context| ... }
#
# Parse a file.
#
# [Parameters]
# - +filename+ (String) The path to the file to be parsed.
# - +encoding+ (optional Encoding, String, nil) An Encoding or encoding name to use when
# parsing the input, or +nil+ for auto-detection. (default #encoding)
#
# [Yields]
# If a block is given, the underlying ParserContext object will be yielded. This can be used
# to set options on the parser context before parsing begins.
#
def parse_file(filename, encoding = @encoding)
raise ArgumentError, "no filename provided" unless filename
raise Errno::ENOENT unless File.exist?(filename)
raise Errno::EISDIR if File.directory?(filename)
ctx = related_class("ParserContext").file(filename, encoding)
yield ctx if block_given?
ctx.parse_with(self)
end
end
end
end
end
@@ -0,0 +1,129 @@
# frozen_string_literal: true
module Nokogiri
module XML
module SAX
###
# Context object to invoke the XML SAX parser on the SAX::Document handler.
#
# 💡 This class is usually not instantiated by the user. Use Nokogiri::XML::SAX::Parser
# instead.
class ParserContext
class << self
###
# :call-seq:
# new(input)
# new(input, encoding)
#
# Create a parser context for an IO or a String. This is a shorthand method for
# ParserContext.io and ParserContext.memory.
#
# [Parameters]
# - +input+ (IO, String) A String or a readable IO object
# - +encoding+ (optional) (Encoding) The +Encoding+ to use, or the name of an
# encoding to use (default +nil+, encoding will be autodetected)
#
# If +input+ quacks like a readable IO object, this method forwards to ParserContext.io,
# otherwise it forwards to ParserContext.memory.
#
# [Returns] Nokogiri::XML::SAX::ParserContext
#
def new(input, encoding = nil)
if [:read, :close].all? { |x| input.respond_to?(x) }
io(input, encoding)
else
memory(input, encoding)
end
end
###
# :call-seq:
# io(input)
# io(input, encoding)
#
# Create a parser context for an +input+ IO which will assume +encoding+
#
# [Parameters]
# - +io+ (IO) The readable IO object from which to read input
# - +encoding+ (optional) (Encoding) The +Encoding+ to use, or the name of an
# encoding to use (default +nil+, encoding will be autodetected)
#
# [Returns] Nokogiri::XML::SAX::ParserContext
#
# 💡 Calling this method directly is discouraged. Use Nokogiri::XML::SAX::Parser parse
# methods which are more convenient for most use cases.
#
def io(input, encoding = nil)
native_io(input, resolve_encoding(encoding))
end
###
# :call-seq:
# memory(input)
# memory(input, encoding)
#
# Create a parser context for the +input+ String.
#
# [Parameters]
# - +input+ (String) The input string to be parsed.
# - +encoding+ (optional) (Encoding, String) The +Encoding+ to use, or the name of an encoding to
# use (default +nil+, encoding will be autodetected)
#
# [Returns] Nokogiri::XML::SAX::ParserContext
#
# 💡 Calling this method directly is discouraged. Use Nokogiri::XML::SAX::Parser parse methods
# which are more convenient for most use cases.
#
def memory(input, encoding = nil)
native_memory(input, resolve_encoding(encoding))
end
###
# :call-seq:
# file(path)
# file(path, encoding)
#
# Create a parser context for the file at +path+.
#
# [Parameters]
# - +path+ (String) The path to the input file
# - +encoding+ (optional) (Encoding, String) The +Encoding+ to use, or the name of an encoding to
# use (default +nil+, encoding will be autodetected)
#
# [Returns] Nokogiri::XML::SAX::ParserContext
#
# 💡 Calling this method directly is discouraged. Use Nokogiri::XML::SAX::Parser.parse_file which
# is more convenient for most use cases.
def file(input, encoding = nil)
native_file(input, resolve_encoding(encoding))
end
private def resolve_encoding(encoding)
case encoding
when Encoding
encoding
when nil
nil # totally fine, parser will guess encoding
when Integer
warn("Passing an integer to Nokogiri::XML::SAX::ParserContext.io is deprecated. Use an Encoding object instead. This will become an error in a future release.", uplevel: 2, category: :deprecated)
return nil if encoding == Parser::ENCODINGS["NONE"]
encoding = Parser::REVERSE_ENCODINGS[encoding]
raise ArgumentError, "Invalid libxml2 encoding id #{encoding}" if encoding.nil?
Encoding.find(encoding)
when String
Encoding.find(encoding)
else
raise ArgumentError, "Cannot resolve #{encoding.inspect} to an Encoding"
end
end
end
end
end
end
end
@@ -0,0 +1,64 @@
# frozen_string_literal: true
module Nokogiri
module XML
module SAX
###
# PushParser can parse a document that is fed to it manually. It
# must be given a SAX::Document object which will be called with
# SAX events as the document is being parsed.
#
# Calling PushParser#<< writes XML to the parser, calling any SAX
# callbacks it can.
#
# PushParser#finish tells the parser that the document is finished
# and calls the end_document SAX method.
#
# Example:
#
# parser = PushParser.new(Class.new(XML::SAX::Document) {
# def start_document
# puts "start document called"
# end
# }.new)
# parser << "<div>hello<"
# parser << "/div>"
# parser.finish
class PushParser
# The Nokogiri::XML::SAX::Document on which the PushParser will be
# operating
attr_accessor :document
###
# Create a new PushParser with +doc+ as the SAX Document, providing
# an optional +file_name+ and +encoding+
def initialize(doc = XML::SAX::Document.new, file_name = nil, encoding = "UTF-8")
@document = doc
@encoding = encoding
@sax_parser = XML::SAX::Parser.new(doc)
## Create our push parser context
initialize_native(@sax_parser, file_name)
end
###
# Write a +chunk+ of XML to the PushParser. Any callback methods
# that can be called will be called immediately.
def write(chunk, last_chunk = false)
native_write(chunk, last_chunk)
end
alias_method :<<, :write
###
# Finish the parsing. This method is only necessary for
# Nokogiri::XML::SAX::Document#end_document to be called.
#
# ⚠ Note that empty documents are treated as an error when using the libxml2-based
# implementation (CRuby), but are fine when using the Xerces-based implementation (JRuby).
def finish
write("", true)
end
end
end
end
end
@@ -0,0 +1,140 @@
# frozen_string_literal: true
module Nokogiri
module XML
class << self
# :call-seq:
# Schema(input) → Nokogiri::XML::Schema
# Schema(input, parse_options) → Nokogiri::XML::Schema
#
# Convenience method for Nokogiri::XML::Schema.new
def Schema(...)
Schema.new(...)
end
end
# Nokogiri::XML::Schema is used for validating \XML against an \XSD schema definition.
#
# ⚠ Since v1.11.0, Schema treats inputs as *untrusted* by default, and so external entities are
# not resolved from the network (+http://+ or +ftp://+). When parsing a trusted document, the
# caller may turn off the +NONET+ option via the ParseOptions to (re-)enable external entity
# resolution over a network connection.
#
# 🛡 Before v1.11.0, documents were "trusted" by default during schema parsing which was counter
# to Nokogiri's "untrusted by default" security policy.
#
# *Example:* Determine whether an \XML document is valid.
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# doc = Nokogiri::XML::Document.parse(File.read(XML_FILE))
# schema.valid?(doc) # Boolean
#
# *Example:* Validate an \XML document against an \XSD schema, and capture any errors that are found.
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# doc = Nokogiri::XML::Document.parse(File.read(XML_FILE))
# errors = schema.validate(doc) # Array<SyntaxError>
#
# *Example:* Validate an \XML document using a Document containing an \XSD schema definition.
#
# schema_doc = Nokogiri::XML::Document.parse(File.read(RELAX_NG_FILE))
# schema = Nokogiri::XML::Schema.from_document(schema_doc)
# doc = Nokogiri::XML::Document.parse(File.read(XML_FILE))
# schema.valid?(doc) # Boolean
#
class Schema
# The errors found while parsing the \XSD
#
# [Returns] Array<Nokogiri::XML::SyntaxError>
attr_accessor :errors
# The options used to parse the schema
#
# [Returns] Nokogiri::XML::ParseOptions
attr_accessor :parse_options
# :call-seq:
# new(input) → Nokogiri::XML::Schema
# new(input, parse_options) → Nokogiri::XML::Schema
#
# Parse an \XSD schema definition from a String or IO to create a new Nokogiri::XML::Schema
#
# [Parameters]
# - +input+ (String | IO) \XSD schema definition
# - +parse_options+ (Nokogiri::XML::ParseOptions)
# Defaults to Nokogiri::XML::ParseOptions::DEFAULT_SCHEMA
#
# [Returns] Nokogiri::XML::Schema
#
def self.new(input, parse_options_ = ParseOptions::DEFAULT_SCHEMA, parse_options: parse_options_)
from_document(Nokogiri::XML::Document.parse(input), parse_options)
end
# :call-seq:
# read_memory(input) → Nokogiri::XML::Schema
# read_memory(input, parse_options) → Nokogiri::XML::Schema
#
# Convenience method for Nokogiri::XML::Schema.new
def self.read_memory(...)
# TODO deprecate this method
new(...)
end
#
# :call-seq: validate(input) → Array<SyntaxError>
#
# Validate +input+ and return any errors that are found.
#
# [Parameters]
# - +input+ (Nokogiri::XML::Document | String)
# A parsed document, or a string containing a local filename.
#
# [Returns] Array<SyntaxError>
#
# *Example:* Validate an existing XML::Document, and capture any errors that are found.
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# errors = schema.validate(document)
#
# *Example:* Validate an \XML document on disk, and capture any errors that are found.
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# errors = schema.validate("/path/to/file.xml")
#
def validate(input)
if input.is_a?(Nokogiri::XML::Document)
validate_document(input)
elsif File.file?(input)
validate_file(input)
else
raise ArgumentError, "Must provide Nokogiri::XML::Document or the name of an existing file"
end
end
#
# :call-seq: valid?(input) → Boolean
#
# Validate +input+ and return a Boolean indicating whether the document is valid
#
# [Parameters]
# - +input+ (Nokogiri::XML::Document | String)
# A parsed document, or a string containing a local filename.
#
# [Returns] Boolean
#
# *Example:* Validate an existing XML::Document
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# return unless schema.valid?(document)
#
# *Example:* Validate an \XML document on disk
#
# schema = Nokogiri::XML::Schema.new(File.read(XSD_FILE))
# return unless schema.valid?("/path/to/file.xml")
#
def valid?(input)
validate(input).empty?
end
end
end
end
@@ -0,0 +1,274 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
module XML
#
# The Searchable module declares the interface used for searching your DOM.
#
# It implements the public methods #search, #css, and #xpath,
# as well as allowing specific implementations to specialize some
# of the important behaviors.
#
module Searchable
# Regular expression used by Searchable#search to determine if a query
# string is CSS or XPath
LOOKS_LIKE_XPATH = %r{^(\./|/|\.\.|\.$)}
# :section: Searching via XPath or CSS Queries
###
# call-seq:
# search(*paths, [namespace-bindings, xpath-variable-bindings, custom-handler-class])
#
# Search this object for +paths+. +paths+ must be one or more XPath or CSS queries:
#
# node.search("div.employee", ".//title")
#
# A hash of namespace bindings may be appended:
#
# node.search('.//bike:tire', {'bike' => 'http://schwinn.com/'})
# node.search('bike|tire', {'bike' => 'http://schwinn.com/'})
#
# For XPath queries, a hash of variable bindings may also be appended to the namespace
# bindings. For example:
#
# node.search('.//address[@domestic=$value]', nil, {:value => 'Yes'})
#
# 💡 Custom XPath functions and CSS pseudo-selectors may also be defined. To define custom
# functions create a class and implement the function you want to define, which will be in the
# `nokogiri` namespace in XPath queries.
#
# The first argument to the method will be the current matching NodeSet. Any other arguments
# are ones that you pass in. Note that this class may appear anywhere in the argument
# list. For example:
#
# handler = Class.new {
# def regex node_set, regex
# node_set.find_all { |node| node['some_attribute'] =~ /#{regex}/ }
# end
# }.new
# node.search('.//title[nokogiri:regex(., "\w+")]', 'div.employee:regex("[0-9]+")', handler)
#
# See Searchable#xpath and Searchable#css for further usage help.
def search(*args)
paths, handler, ns, binds = extract_params(args)
xpaths = paths.map(&:to_s).map do |path|
LOOKS_LIKE_XPATH.match?(path) ? path : xpath_query_from_css_rule(path, ns)
end.flatten.uniq
xpath(*(xpaths + [ns, handler, binds].compact))
end
alias_method :/, :search
###
# call-seq:
# at(*paths, [namespace-bindings, xpath-variable-bindings, custom-handler-class])
#
# Search this object for +paths+, and return only the first
# result. +paths+ must be one or more XPath or CSS queries.
#
# See Searchable#search for more information.
def at(*args)
search(*args).first
end
alias_method :%, :at
###
# call-seq:
# css(*rules, [namespace-bindings, custom-pseudo-class])
#
# Search this object for CSS +rules+. +rules+ must be one or more CSS
# selectors. For example:
#
# node.css('title')
# node.css('body h1.bold')
# node.css('div + p.green', 'div#one')
#
# A hash of namespace bindings may be appended. For example:
#
# node.css('bike|tire', {'bike' => 'http://schwinn.com/'})
#
# 💡 Custom CSS pseudo classes may also be defined which are mapped to a custom XPath
# function. To define custom pseudo classes, create a class and implement the custom pseudo
# class you want defined. The first argument to the method will be the matching context
# NodeSet. Any other arguments are ones that you pass in. For example:
#
# handler = Class.new {
# def regex(node_set, regex)
# node_set.find_all { |node| node['some_attribute'] =~ /#{regex}/ }
# end
# }.new
# node.css('title:regex("\w+")', handler)
#
# 💡 Some XPath syntax is supported in CSS queries. For example, to query for an attribute:
#
# node.css('img > @href') # returns all +href+ attributes on an +img+ element
# node.css('img / @href') # same
#
# # ⚠ this returns +class+ attributes from all +div+ elements AND THEIR CHILDREN!
# node.css('div @class')
#
# node.css
#
# 💡 Array-like syntax is supported in CSS queries as an alternative to using +:nth-child()+.
#
# ⚠ NOTE that indices are 1-based like +:nth-child+ and not 0-based like Ruby Arrays. For
# example:
#
# # equivalent to 'li:nth-child(2)'
# node.css('li[2]') # retrieve the second li element in a list
#
# ⚠ NOTE that the CSS query string is case-sensitive with regards to your document type. HTML
# tags will match only lowercase CSS queries, so if you search for "H1" in an HTML document,
# you'll never find anything. However, "H1" might be found in an XML document, where tags
# names are case-sensitive (e.g., "H1" is distinct from "h1").
def css(*args)
rules, handler, ns, _ = extract_params(args)
css_internal(self, rules, handler, ns)
end
##
# call-seq:
# at_css(*rules, [namespace-bindings, custom-pseudo-class])
#
# Search this object for CSS +rules+, and return only the first
# match. +rules+ must be one or more CSS selectors.
#
# See Searchable#css for more information.
def at_css(*args)
css(*args).first
end
###
# call-seq:
# xpath(*paths, [namespace-bindings, variable-bindings, custom-handler-class])
#
# Search this node for XPath +paths+. +paths+ must be one or more XPath
# queries.
#
# node.xpath('.//title')
#
# A hash of namespace bindings may be appended. For example:
#
# node.xpath('.//foo:name', {'foo' => 'http://example.org/'})
# node.xpath('.//xmlns:name', node.root.namespaces)
#
# A hash of variable bindings may also be appended to the namespace bindings. For example:
#
# node.xpath('.//address[@domestic=$value]', nil, {:value => 'Yes'})
#
# 💡 Custom XPath functions may also be defined. To define custom functions create a class and
# implement the function you want to define, which will be in the `nokogiri` namespace.
#
# The first argument to the method will be the current matching NodeSet. Any other arguments
# are ones that you pass in. Note that this class may appear anywhere in the argument
# list. For example:
#
# handler = Class.new {
# def regex(node_set, regex)
# node_set.find_all { |node| node['some_attribute'] =~ /#{regex}/ }
# end
# }.new
# node.xpath('.//title[nokogiri:regex(., "\w+")]', handler)
#
def xpath(*args)
paths, handler, ns, binds = extract_params(args)
xpath_internal(self, paths, handler, ns, binds)
end
##
# call-seq:
# at_xpath(*paths, [namespace-bindings, variable-bindings, custom-handler-class])
#
# Search this node for XPath +paths+, and return only the first
# match. +paths+ must be one or more XPath queries.
#
# See Searchable#xpath for more information.
def at_xpath(*args)
xpath(*args).first
end
# :call-seq:
# >(selector) → NodeSet
#
# Search this node's immediate children using CSS selector +selector+
def >(selector) # rubocop:disable Naming/BinaryOperatorParameterName
ns = document.root&.namespaces || {}
xpath(CSS.xpath_for(selector, prefix: "./", ns: ns).first)
end
# :section:
private
def extract_params(params) # :nodoc:
handler = params.find do |param|
![Hash, String, Symbol].include?(param.class)
end
params -= [handler] if handler
hashes = []
while Hash === params.last || params.last.nil?
hashes << params.pop
break if params.empty?
end
ns, binds = hashes.reverse
ns ||= document.root&.namespaces || {}
[params, handler, ns, binds]
end
def css_internal(node, rules, handler, ns)
xpath_internal(node, css_rules_to_xpath(rules, ns), handler, ns, nil)
end
def css_rules_to_xpath(rules, ns)
rules.map { |rule| xpath_query_from_css_rule(rule, ns) }
end
def xpath_query_from_css_rule(rule, ns)
self.class::IMPLIED_XPATH_CONTEXTS.map do |implied_xpath_context|
visitor = Nokogiri::CSS::XPathVisitor.new(
builtins: Nokogiri::CSS::XPathVisitor::BuiltinsConfig::OPTIMAL,
doctype: document.xpath_doctype,
prefix: implied_xpath_context,
namespaces: ns,
)
CSS.xpath_for(rule.to_s, visitor: visitor)
end.join(" | ")
end
def xpath_internal(node, paths, handler, ns, binds)
document = node.document
return NodeSet.new(document) unless document
if paths.length == 1
return xpath_impl(node, paths.first, handler, ns, binds)
end
NodeSet.new(document) do |combined|
paths.each do |path|
xpath_impl(node, path, handler, ns, binds).each { |set| combined << set }
end
end
end
def xpath_impl(node, path, handler, ns, binds)
context = XPathContext.new(node)
context.register_namespaces(ns)
context.register_variables(binds)
path = path.gsub("xmlns:", " :") unless Nokogiri.uses_libxml?
context.evaluate(path, handler)
end
end
end
end
@@ -0,0 +1,94 @@
# frozen_string_literal: true
module Nokogiri
module XML
###
# This class provides information about XML SyntaxErrors. These
# exceptions are typically stored on Nokogiri::XML::Document#errors.
class SyntaxError < ::Nokogiri::SyntaxError
class << self
def aggregate(errors)
return nil if errors.empty?
return errors.first if errors.length == 1
messages = ["Multiple errors encountered:"]
errors.each do |error|
messages << error.to_s
end
new(messages.join("\n"))
end
end
attr_reader :domain
attr_reader :code
attr_reader :level
attr_reader :file
attr_reader :line
# The XPath path of the node that caused the error when validating a `Nokogiri::XML::Document`.
#
# This attribute will only be non-nil when the error is emitted by `Schema#validate` on
# Document objects. It will return `nil` for DOM parsing errors and for errors emitted during
# Schema validation of files.
#
# ⚠ `#path` is not supported on JRuby, where it will always return `nil`.
attr_reader :path
attr_reader :str1
attr_reader :str2
attr_reader :str3
attr_reader :int1
attr_reader :column
###
# return true if this is a non error
def none?
level == 0
end
###
# return true if this is a warning
def warning?
level == 1
end
###
# return true if this is an error
def error?
level == 2
end
###
# return true if this error is fatal
def fatal?
level == 3
end
def to_s
message = super.chomp
[location_to_s, level_to_s, message]
.compact.join(": ")
.force_encoding(message.encoding)
end
private
def level_to_s
case level
when 3 then "FATAL"
when 2 then "ERROR"
when 1 then "WARNING"
end
end
def nil_or_zero?(attribute)
attribute.nil? || attribute.zero?
end
def location_to_s
return if nil_or_zero?(line) && nil_or_zero?(column)
"#{line}:#{column}"
end
end
end
end
@@ -0,0 +1,11 @@
# frozen_string_literal: true
module Nokogiri
module XML
class Text < Nokogiri::XML::CharacterData
def content=(string)
self.native_content = string.to_s
end
end
end
end
@@ -0,0 +1,21 @@
# frozen_string_literal: true
module Nokogiri
module XML
module XPath
# The XPath search prefix to search globally, +//+
GLOBAL_SEARCH_PREFIX = "//"
# The XPath search prefix to search direct descendants of the root element, +/+
ROOT_SEARCH_PREFIX = "/"
# The XPath search prefix to search direct descendants of the current element, +./+
CURRENT_SEARCH_PREFIX = "./"
# The XPath search prefix to search anywhere in the current element's subtree, +.//+
SUBTREE_SEARCH_PREFIX = ".//"
end
end
end
require_relative "xpath/syntax_error"
@@ -0,0 +1,13 @@
# frozen_string_literal: true
module Nokogiri
module XML
module XPath
class SyntaxError < XML::SyntaxError
def to_s
[super.chomp, str1].compact.join(": ")
end
end
end
end
end
@@ -0,0 +1,27 @@
# frozen_string_literal: true
module Nokogiri
module XML
class XPathContext
###
# Register namespaces in +namespaces+
def register_namespaces(namespaces)
namespaces.each do |key, value|
key = key.to_s.gsub(/.*:/, "") # strip off 'xmlns:' or 'xml:'
register_ns(key, value)
end
end
def register_variables(binds)
return if binds.nil?
binds.each do |key, value|
key = key.to_s
register_variable(key, value)
end
end
end
end
end
@@ -0,0 +1,129 @@
# coding: utf-8
# frozen_string_literal: true
module Nokogiri
class << self
# Convenience method for Nokogiri::XSLT.parse
def XSLT(...)
XSLT.parse(...)
end
end
###
# See Nokogiri::XSLT::Stylesheet for creating and manipulating
# Stylesheet object.
module XSLT
class << self
# :call-seq:
# parse(xsl) → Nokogiri::XSLT::Stylesheet
# parse(xsl, modules) → Nokogiri::XSLT::Stylesheet
#
# Parse the stylesheet in +xsl+, registering optional +modules+ as custom class handlers.
#
# [Parameters]
# - +xsl+ (String) XSL content to be parsed into a stylesheet
# - +modules+ (Hash<String ⇒ Class>) A hash of URI-to-handler relations for linking a
# namespace to a custom function handler.
#
# ⚠ The XSLT handler classes are registered *globally*.
#
# Also see Nokogiri::XSLT.register
#
# *Example*
#
# xml = Nokogiri.XML(<<~XML)
# <nodes>
# <node>Foo</node>
# <node>Bar</node>
# </nodes>
# XML
#
# handler = Class.new do
# def reverse(node)
# node.text.reverse
# end
# end
#
# xsl = <<~XSL
# <xsl:stylesheet version="1.0"
# xmlns:xsl="http://www.w3.org/1999/XSL/Transform"
# xmlns:myfuncs="http://nokogiri.org/xslt/myfuncs"
# extension-element-prefixes="myfuncs">
# <xsl:template match="/">
# <reversed>
# <xsl:for-each select="nodes/node">
# <reverse><xsl:copy-of select="myfuncs:reverse(.)"/></reverse>
# </xsl:for-each>
# </reversed>
# </xsl:template>
# </xsl:stylesheet>
# XSL
#
# xsl = Nokogiri.XSLT(xsl, "http://nokogiri.org/xslt/myfuncs" => handler)
# xsl.transform(xml).to_xml
# # => "<?xml version=\"1.0\"?>\n" +
# # "<reversed>\n" +
# # " <reverse>ooF</reverse>\n" +
# # " <reverse>raB</reverse>\n" +
# # "</reversed>\n"
#
def parse(string, modules = {})
modules.each do |url, klass|
XSLT.register(url, klass)
end
doc = XML::Document.parse(string, nil, nil, XML::ParseOptions::DEFAULT_XSLT)
if Nokogiri.jruby?
Stylesheet.parse_stylesheet_doc(doc, string)
else
Stylesheet.parse_stylesheet_doc(doc)
end
end
# :call-seq:
# quote_params(params) → Array
#
# Quote parameters in +params+ for stylesheet safety.
# See Nokogiri::XSLT::Stylesheet.transform for example usage.
#
# [Parameters]
# - +params+ (Hash, Array) XSLT parameters (key->value, or tuples of [key, value])
#
# [Returns] Array of string parameters, with quotes correctly escaped for use with XSLT::Stylesheet.transform
#
def quote_params(params)
params.flatten.each_slice(2).with_object([]) do |kv, quoted_params|
key, value = kv.map(&:to_s)
value = if value.include?("'")
"concat('#{value.gsub("'", %q{', "'", '})}')"
else
"'#{value}'"
end
quoted_params << key
quoted_params << value
end
end
# call-seq:
# register(uri, custom_handler_class)
#
# Register a class that implements custom XSLT transformation functions.
#
# ⚠ The XSLT handler classes are registered *globally*.
#
# [Parameters}
# - +uri+ (String) The namespace for the custom handlers
# - +custom_handler_class+ (Class) A class with ruby methods that can be called during
# transformation
#
# See Nokogiri::XSLT.parse for usage.
#
def register(uri, custom_handler_class)
# NOTE: this is implemented in the C extension, see ext/nokogiri/xslt_stylesheet.c
raise NotImplementedError, "Nokogiri::XSLT.register is not implemented on JRuby"
end if Nokogiri.jruby?
end
end
end
require_relative "xslt/stylesheet"
@@ -0,0 +1,49 @@
# frozen_string_literal: true
module Nokogiri
module XSLT
###
# A Stylesheet represents an XSLT Stylesheet object. Stylesheet creation
# is done through Nokogiri.XSLT. Here is an example of transforming
# an XML::Document with a Stylesheet:
#
# doc = Nokogiri::XML(File.read('some_file.xml'))
# xslt = Nokogiri::XSLT(File.read('some_transformer.xslt'))
#
# xslt.transform(doc) # => Nokogiri::XML::Document
#
# Many XSLT transformations include serialization behavior to emit a non-XML document. For these
# cases, please take care to invoke the #serialize method on the result of the transformation:
#
# doc = Nokogiri::XML(File.read('some_file.xml'))
# xslt = Nokogiri::XSLT(File.read('some_transformer.xslt'))
# xslt.serialize(xslt.transform(doc)) # => String
#
# or use the #apply_to method, which is a shortcut for `serialize(transform(document))`:
#
# doc = Nokogiri::XML(File.read('some_file.xml'))
# xslt = Nokogiri::XSLT(File.read('some_transformer.xslt'))
# xslt.apply_to(doc) # => String
#
# See Nokogiri::XSLT::Stylesheet#transform for more information and examples.
class Stylesheet
# :call-seq:
# apply_to(document, params = []) -> String
#
# Apply an XSLT stylesheet to an XML::Document and serialize it properly. This method is
# equivalent to calling #serialize on the result of #transform.
#
# [Parameters]
# - +document+ is an instance of XML::Document to transform
# - +params+ is an array of strings used as XSLT parameters, passed into #transform
#
# [Returns]
# A string containing the serialized result of the transformation.
#
# See Nokogiri::XSLT::Stylesheet#transform for more information and examples.
def apply_to(document, params = [])
serialize(transform(document, params))
end
end
end
end