This commit is contained in:
@@ -0,0 +1,37 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
module Nokogiri
|
||||
module HTML4
|
||||
###
|
||||
# Nokogiri HTML builder is used for building HTML documents. It is very
|
||||
# similar to the Nokogiri::XML::Builder. In fact, you should go read the
|
||||
# documentation for Nokogiri::XML::Builder before reading this
|
||||
# documentation.
|
||||
#
|
||||
# == Synopsis:
|
||||
#
|
||||
# Create an HTML document with a body that has an onload attribute, and a
|
||||
# span tag with a class of "bold" that has content of "Hello world".
|
||||
#
|
||||
# builder = Nokogiri::HTML4::Builder.new do |doc|
|
||||
# doc.html {
|
||||
# doc.body(:onload => 'some_func();') {
|
||||
# doc.span.bold {
|
||||
# doc.text "Hello world"
|
||||
# }
|
||||
# }
|
||||
# }
|
||||
# end
|
||||
# puts builder.to_html
|
||||
#
|
||||
# The HTML builder inherits from the XML builder, so make sure to read the
|
||||
# Nokogiri::XML::Builder documentation.
|
||||
class Builder < Nokogiri::XML::Builder
|
||||
###
|
||||
# Convert the builder to HTML
|
||||
def to_html
|
||||
@doc.to_html
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,235 @@
|
||||
# coding: utf-8
|
||||
# frozen_string_literal: true
|
||||
|
||||
require "pathname"
|
||||
|
||||
module Nokogiri
|
||||
module HTML4
|
||||
class Document < Nokogiri::XML::Document
|
||||
###
|
||||
# Get the meta tag encoding for this document. If there is no meta tag,
|
||||
# then nil is returned.
|
||||
def meta_encoding
|
||||
if (meta = at_xpath("//meta[@charset]"))
|
||||
meta[:charset]
|
||||
elsif (meta = meta_content_type)
|
||||
meta["content"][/charset\s*=\s*([\w-]+)/i, 1]
|
||||
end
|
||||
end
|
||||
|
||||
###
|
||||
# Set the meta tag encoding for this document.
|
||||
#
|
||||
# If an meta encoding tag is already present, its content is
|
||||
# replaced with the given text.
|
||||
#
|
||||
# Otherwise, this method tries to create one at an appropriate
|
||||
# place supplying head and/or html elements as necessary, which
|
||||
# is inside a head element if any, and before any text node or
|
||||
# content element (typically <body>) if any.
|
||||
#
|
||||
# The result when trying to set an encoding that is different
|
||||
# from the document encoding is undefined.
|
||||
#
|
||||
# Beware in CRuby, that libxml2 automatically inserts a meta tag
|
||||
# into a head element.
|
||||
def meta_encoding=(encoding)
|
||||
if (meta = meta_content_type)
|
||||
meta["content"] = format("text/html; charset=%s", encoding)
|
||||
encoding
|
||||
elsif (meta = at_xpath("//meta[@charset]"))
|
||||
meta["charset"] = encoding
|
||||
else
|
||||
meta = XML::Node.new("meta", self)
|
||||
if (dtd = internal_subset) && dtd.html5_dtd?
|
||||
meta["charset"] = encoding
|
||||
else
|
||||
meta["http-equiv"] = "Content-Type"
|
||||
meta["content"] = format("text/html; charset=%s", encoding)
|
||||
end
|
||||
|
||||
if (head = at_xpath("//head"))
|
||||
head.prepend_child(meta)
|
||||
else
|
||||
set_metadata_element(meta)
|
||||
end
|
||||
encoding
|
||||
end
|
||||
end
|
||||
|
||||
def meta_content_type
|
||||
xpath("//meta[@http-equiv and boolean(@content)]").find do |node|
|
||||
node["http-equiv"] =~ /\AContent-Type\z/i
|
||||
end
|
||||
end
|
||||
private :meta_content_type
|
||||
|
||||
###
|
||||
# Get the title string of this document. Return nil if there is
|
||||
# no title tag.
|
||||
def title
|
||||
(title = at_xpath("//title")) && title.inner_text
|
||||
end
|
||||
|
||||
###
|
||||
# Set the title string of this document.
|
||||
#
|
||||
# If a title element is already present, its content is replaced
|
||||
# with the given text.
|
||||
#
|
||||
# Otherwise, this method tries to create one at an appropriate
|
||||
# place supplying head and/or html elements as necessary, which
|
||||
# is inside a head element if any, right after a meta
|
||||
# encoding/charset tag if any, and before any text node or
|
||||
# content element (typically <body>) if any.
|
||||
def title=(text)
|
||||
tnode = XML::Text.new(text, self)
|
||||
if (title = at_xpath("//title"))
|
||||
title.children = tnode
|
||||
return text
|
||||
end
|
||||
|
||||
title = XML::Node.new("title", self) << tnode
|
||||
if (head = at_xpath("//head"))
|
||||
head << title
|
||||
elsif (meta = at_xpath("//meta[@charset]") || meta_content_type)
|
||||
# better put after charset declaration
|
||||
meta.add_next_sibling(title)
|
||||
else
|
||||
set_metadata_element(title)
|
||||
end
|
||||
end
|
||||
|
||||
def set_metadata_element(element) # rubocop:disable Naming/AccessorMethodName
|
||||
if (head = at_xpath("//head"))
|
||||
head << element
|
||||
elsif (html = at_xpath("//html"))
|
||||
head = html.prepend_child(XML::Node.new("head", self))
|
||||
head.prepend_child(element)
|
||||
elsif (first = children.find do |node|
|
||||
case node
|
||||
when XML::Element, XML::Text
|
||||
true
|
||||
end
|
||||
end)
|
||||
# We reach here only if the underlying document model
|
||||
# allows <html>/<head> elements to be omitted and does not
|
||||
# automatically supply them.
|
||||
first.add_previous_sibling(element)
|
||||
else
|
||||
html = add_child(XML::Node.new("html", self))
|
||||
head = html.add_child(XML::Node.new("head", self))
|
||||
head.prepend_child(element)
|
||||
end
|
||||
end
|
||||
private :set_metadata_element
|
||||
|
||||
####
|
||||
# Serialize Node using +options+. Save options can also be set using a block.
|
||||
#
|
||||
# See also Nokogiri::XML::Node::SaveOptions and Node@Serialization+and+Generating+Output.
|
||||
#
|
||||
# These two statements are equivalent:
|
||||
#
|
||||
# node.serialize(:encoding => 'UTF-8', :save_with => FORMAT | AS_XML)
|
||||
#
|
||||
# or
|
||||
#
|
||||
# node.serialize(:encoding => 'UTF-8') do |config|
|
||||
# config.format.as_xml
|
||||
# end
|
||||
#
|
||||
def serialize(options = {})
|
||||
options[:save_with] ||= XML::Node::SaveOptions::DEFAULT_HTML
|
||||
super
|
||||
end
|
||||
|
||||
####
|
||||
# Create a Nokogiri::XML::DocumentFragment from +tags+
|
||||
def fragment(tags = nil)
|
||||
DocumentFragment.new(self, tags, root)
|
||||
end
|
||||
|
||||
# :call-seq:
|
||||
# xpath_doctype() → Nokogiri::CSS::XPathVisitor::DoctypeConfig
|
||||
#
|
||||
# [Returns] The document type which determines CSS-to-XPath translation.
|
||||
#
|
||||
# See XPathVisitor for more information.
|
||||
def xpath_doctype
|
||||
Nokogiri::CSS::XPathVisitor::DoctypeConfig::HTML4
|
||||
end
|
||||
|
||||
class << self
|
||||
# :call-seq:
|
||||
# parse(input) { |options| ... } => Nokogiri::HTML4::Document
|
||||
# parse(input, url:, encoding:, options:) => Nokogiri::HTML4::Document
|
||||
#
|
||||
# Parse \HTML4 input from a String or IO object, and return a new HTML4::Document.
|
||||
#
|
||||
# [Required Parameters]
|
||||
# - +input+ (String | IO) The content to be parsed.
|
||||
#
|
||||
# [Optional Keyword Arguments]
|
||||
# - +url:+ (String) The base URI for this document.
|
||||
#
|
||||
# - +encoding:+ (String) The name of the encoding that should be used when processing the
|
||||
# document. When not provided, the encoding will be determined based on the document
|
||||
# content.
|
||||
#
|
||||
# - +options:+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
|
||||
# behaviors during parsing. See ParseOptions for more information. The default value is
|
||||
# +ParseOptions::DEFAULT_HTML+.
|
||||
#
|
||||
# [Yields]
|
||||
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
|
||||
# can be configured before parsing. See Nokogiri::XML::ParseOptions for more information.
|
||||
#
|
||||
# [Returns] Nokogiri::HTML4::Document
|
||||
def parse(
|
||||
input,
|
||||
url_ = nil, encoding_ = nil, options_ = XML::ParseOptions::DEFAULT_HTML,
|
||||
url: url_, encoding: encoding_, options: options_
|
||||
)
|
||||
options = Nokogiri::XML::ParseOptions.new(options) if Integer === options
|
||||
yield options if block_given?
|
||||
|
||||
url ||= input.respond_to?(:path) ? input.path : nil
|
||||
|
||||
if input.respond_to?(:encoding)
|
||||
unless input.encoding == Encoding::ASCII_8BIT
|
||||
encoding ||= input.encoding.name
|
||||
end
|
||||
end
|
||||
|
||||
if input.respond_to?(:read)
|
||||
if input.is_a?(Pathname)
|
||||
# resolve the Pathname to the file and open it as an IO object, see #2110
|
||||
input = input.expand_path.open
|
||||
url ||= input.path
|
||||
end
|
||||
|
||||
unless encoding
|
||||
input = EncodingReader.new(input)
|
||||
begin
|
||||
return read_io(input, url, encoding, options.to_i)
|
||||
rescue EncodingReader::EncodingFound => e
|
||||
encoding = e.found_encoding
|
||||
end
|
||||
end
|
||||
return read_io(input, url, encoding, options.to_i)
|
||||
end
|
||||
|
||||
# read_memory pukes on empty docs
|
||||
if input.nil? || input.empty?
|
||||
return encoding ? new.tap { |i| i.encoding = encoding } : new
|
||||
end
|
||||
|
||||
encoding ||= EncodingReader.detect_encoding(input)
|
||||
|
||||
read_memory(input, url, encoding, options.to_i)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,166 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
module Nokogiri
|
||||
module HTML4
|
||||
class DocumentFragment < Nokogiri::XML::DocumentFragment
|
||||
#
|
||||
# :call-seq:
|
||||
# parse(input) { |options| ... } → HTML4::DocumentFragment
|
||||
# parse(input, encoding:, options:) { |options| ... } → HTML4::DocumentFragment
|
||||
#
|
||||
# Parse \HTML4 fragment input from a String, and return a new HTML4::DocumentFragment. This
|
||||
# method creates a new, empty HTML4::Document to contain the fragment.
|
||||
#
|
||||
# [Required Parameters]
|
||||
# - +input+ (String | IO) The content to be parsed.
|
||||
#
|
||||
# [Optional Keyword Arguments]
|
||||
# - +encoding:+ (String) The name of the encoding that should be used when processing the
|
||||
# document. When not provided, the encoding will be determined based on the document
|
||||
# content.
|
||||
#
|
||||
# - +options:+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
|
||||
# behaviors during parsing. See ParseOptions for more information. The default value is
|
||||
# +ParseOptions::DEFAULT_HTML+.
|
||||
#
|
||||
# [Yields]
|
||||
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
|
||||
# can be configured before parsing. See ParseOptions for more information.
|
||||
#
|
||||
# [Returns] HTML4::DocumentFragment
|
||||
#
|
||||
# *Example:* Parsing a string
|
||||
#
|
||||
# fragment = HTML4::DocumentFragment.parse("<div>Hello World</div>")
|
||||
#
|
||||
# *Example:* Parsing an IO
|
||||
#
|
||||
# fragment = File.open("fragment.html") do |file|
|
||||
# HTML4::DocumentFragment.parse(file)
|
||||
# end
|
||||
#
|
||||
# *Example:* Specifying encoding
|
||||
#
|
||||
# fragment = HTML4::DocumentFragment.parse(input, encoding: "EUC-JP")
|
||||
#
|
||||
# *Example:* Setting parse options dynamically
|
||||
#
|
||||
# HTML4::DocumentFragment.parse("<div>Hello World") do |options|
|
||||
# options.huge.pedantic
|
||||
# end
|
||||
#
|
||||
def self.parse(
|
||||
input,
|
||||
encoding_ = nil, options_ = XML::ParseOptions::DEFAULT_HTML,
|
||||
encoding: encoding_, options: options_,
|
||||
&block
|
||||
)
|
||||
# TODO: this method should take a context node.
|
||||
doc = HTML4::Document.new
|
||||
|
||||
if input.respond_to?(:read)
|
||||
# Handle IO-like objects (IO, File, StringIO, etc.)
|
||||
# The _read_ method of these objects doesn't accept an +encoding+ parameter.
|
||||
# Encoding is usually set when the IO object is created or opened,
|
||||
# or by using the _set_encoding_ method.
|
||||
#
|
||||
# 1. If +encoding+ is provided and the object supports _set_encoding_,
|
||||
# set the encoding before reading.
|
||||
# 2. Read the content from the IO-like object.
|
||||
#
|
||||
# Note: After reading, the content's encoding will be:
|
||||
# - The encoding set by _set_encoding_ if it was called
|
||||
# - The default encoding of the IO object otherwise
|
||||
#
|
||||
# For StringIO specifically, _set_encoding_ affects only the internal string,
|
||||
# not how the data is read out.
|
||||
input.set_encoding(encoding) if encoding && input.respond_to?(:set_encoding)
|
||||
input = input.read
|
||||
end
|
||||
|
||||
encoding ||= if input.respond_to?(:encoding)
|
||||
encoding = input.encoding
|
||||
if encoding == ::Encoding::ASCII_8BIT
|
||||
"UTF-8"
|
||||
else
|
||||
encoding.name
|
||||
end
|
||||
else
|
||||
"UTF-8"
|
||||
end
|
||||
|
||||
doc.encoding = encoding
|
||||
|
||||
new(doc, input, options: options, &block)
|
||||
end
|
||||
|
||||
#
|
||||
# :call-seq:
|
||||
# new(document) { |options| ... } → HTML4::DocumentFragment
|
||||
# new(document, input) { |options| ... } → HTML4::DocumentFragment
|
||||
# new(document, input, context:, options:) { |options| ... } → HTML4::DocumentFragment
|
||||
#
|
||||
# Parse \HTML4 fragment input from a String, and return a new HTML4::DocumentFragment.
|
||||
#
|
||||
# 💡 It's recommended to use either HTML4::DocumentFragment.parse or XML::Node#parse rather
|
||||
# than call this method directly.
|
||||
#
|
||||
# [Required Parameters]
|
||||
# - +document+ (HTML4::Document) The parent document to associate the returned fragment with.
|
||||
#
|
||||
# [Optional Parameters]
|
||||
# - +input+ (String) The content to be parsed.
|
||||
#
|
||||
# [Optional Keyword Arguments]
|
||||
# - +context:+ (Nokogiri::XML::Node) The <b>context node</b> for the subtree created. See
|
||||
# below for more information.
|
||||
#
|
||||
# - +options:+ (Nokogiri::XML::ParseOptions) Configuration object that determines some
|
||||
# behaviors during parsing. See ParseOptions for more information. The default value is
|
||||
# +ParseOptions::DEFAULT_HTML+.
|
||||
#
|
||||
# [Yields]
|
||||
# If a block is given, a Nokogiri::XML::ParseOptions object is yielded to the block which
|
||||
# can be configured before parsing. See ParseOptions for more information.
|
||||
#
|
||||
# [Returns] HTML4::DocumentFragment
|
||||
#
|
||||
# === Context \Node
|
||||
#
|
||||
# If a context node is specified using +context:+, then the fragment will be created by
|
||||
# calling XML::Node#parse on that node, so the parser will behave as if that Node is the
|
||||
# parent of the fragment subtree.
|
||||
#
|
||||
def initialize(
|
||||
document, input = nil,
|
||||
context_ = nil, options_ = XML::ParseOptions::DEFAULT_HTML,
|
||||
context: context_, options: options_
|
||||
) # rubocop:disable Lint/MissingSuper
|
||||
return self unless input
|
||||
|
||||
options = Nokogiri::XML::ParseOptions.new(options) if Integer === options
|
||||
@parse_options = options
|
||||
yield options if block_given?
|
||||
|
||||
if context
|
||||
preexisting_errors = document.errors.dup
|
||||
node_set = context.parse("<div>#{input}</div>", options)
|
||||
node_set.first.children.each { |child| child.parent = self } unless node_set.empty?
|
||||
self.errors = document.errors - preexisting_errors
|
||||
else
|
||||
# This is a horrible hack, but I don't care
|
||||
path = if /^\s*?<body/i.match?(input)
|
||||
"/html/body"
|
||||
else
|
||||
"/html/body/node()"
|
||||
end
|
||||
|
||||
temp_doc = HTML4::Document.parse("<html><body>#{input}", nil, document.encoding, options)
|
||||
temp_doc.xpath(path).each { |child| child.parent = self }
|
||||
self.errors = temp_doc.errors
|
||||
end
|
||||
children
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,25 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
module Nokogiri
|
||||
module HTML4
|
||||
class ElementDescription
|
||||
###
|
||||
# Is this element a block element?
|
||||
def block?
|
||||
!inline?
|
||||
end
|
||||
|
||||
###
|
||||
# Convert this description to a string
|
||||
def to_s
|
||||
"#{name}: #{description}"
|
||||
end
|
||||
|
||||
###
|
||||
# Inspection information
|
||||
def inspect
|
||||
"#<#{self.class.name}: #{name} #{description}>"
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
+2040
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,121 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
module Nokogiri
|
||||
module HTML4
|
||||
# Libxml2's parser has poor support for encoding detection. First, it does not recognize the
|
||||
# HTML5 style meta charset declaration. Secondly, even if it successfully detects an encoding
|
||||
# hint, it does not re-decode or re-parse the preceding part which may be garbled.
|
||||
#
|
||||
# EncodingReader aims to perform advanced encoding detection beyond what Libxml2 does, and to
|
||||
# emulate rewinding of a stream and make Libxml2 redo parsing from the start when an encoding
|
||||
# hint is found.
|
||||
|
||||
# :nodoc: all
|
||||
class EncodingReader
|
||||
class EncodingFound < StandardError
|
||||
attr_reader :found_encoding
|
||||
|
||||
def initialize(encoding)
|
||||
@found_encoding = encoding
|
||||
super(format("encoding found: %s", encoding))
|
||||
end
|
||||
end
|
||||
|
||||
class SAXHandler < Nokogiri::XML::SAX::Document
|
||||
attr_reader :encoding
|
||||
|
||||
def initialize
|
||||
@encoding = nil
|
||||
super
|
||||
end
|
||||
|
||||
def start_element(name, attrs = [])
|
||||
return unless name == "meta"
|
||||
|
||||
attr = Hash[attrs]
|
||||
(charset = attr["charset"]) &&
|
||||
(@encoding = charset)
|
||||
(http_equiv = attr["http-equiv"]) &&
|
||||
http_equiv.match(/\AContent-Type\z/i) &&
|
||||
(content = attr["content"]) &&
|
||||
(m = content.match(/;\s*charset\s*=\s*([\w-]+)/)) &&
|
||||
(@encoding = m[1])
|
||||
end
|
||||
end
|
||||
|
||||
class JumpSAXHandler < SAXHandler
|
||||
def initialize(jumptag)
|
||||
@jumptag = jumptag
|
||||
super()
|
||||
end
|
||||
|
||||
def start_element(name, attrs = [])
|
||||
super
|
||||
throw(@jumptag, @encoding) if @encoding
|
||||
throw(@jumptag, nil) if /\A(?:div|h1|img|p|br)\z/.match?(name)
|
||||
end
|
||||
end
|
||||
|
||||
def self.detect_encoding(chunk)
|
||||
(m = chunk.match(/\A(<\?xml[ \t\r\n][^>]*>)/)) &&
|
||||
(return Nokogiri.XML(m[1]).encoding)
|
||||
|
||||
if Nokogiri.jruby?
|
||||
(m = chunk.match(/(<meta\s)(.*)(charset\s*=\s*([\w-]+))(.*)/i)) &&
|
||||
(return m[4])
|
||||
catch(:encoding_found) do
|
||||
Nokogiri::HTML4::SAX::Parser.new(JumpSAXHandler.new(:encoding_found)).parse(chunk)
|
||||
nil
|
||||
end
|
||||
else
|
||||
handler = SAXHandler.new
|
||||
parser = Nokogiri::HTML4::SAX::PushParser.new(handler)
|
||||
begin
|
||||
parser << chunk
|
||||
rescue
|
||||
Nokogiri::SyntaxError
|
||||
end
|
||||
handler.encoding
|
||||
end
|
||||
end
|
||||
|
||||
def initialize(io)
|
||||
@io = io
|
||||
@firstchunk = nil
|
||||
@encoding_found = nil
|
||||
end
|
||||
|
||||
# This method is used by the C extension so that
|
||||
# Nokogiri::HTML4::Document#read_io() does not leak memory when
|
||||
# EncodingFound is raised.
|
||||
attr_reader :encoding_found
|
||||
|
||||
def read(len)
|
||||
# no support for a call without len
|
||||
|
||||
unless @firstchunk
|
||||
(@firstchunk = @io.read(len)) || return
|
||||
|
||||
# This implementation expects that the first call from
|
||||
# htmlReadIO() is made with a length long enough (~1KB) to
|
||||
# achieve advanced encoding detection.
|
||||
if (encoding = EncodingReader.detect_encoding(@firstchunk))
|
||||
# The first chunk is stored for the next read in retry.
|
||||
raise @encoding_found = EncodingFound.new(encoding)
|
||||
end
|
||||
end
|
||||
@encoding_found = nil
|
||||
|
||||
ret = @firstchunk.slice!(0, len)
|
||||
if (len -= ret.length) > 0
|
||||
(rest = @io.read(len)) && ret << (rest)
|
||||
end
|
||||
if ret.empty?
|
||||
nil
|
||||
else
|
||||
ret
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,15 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
module Nokogiri
|
||||
module HTML4
|
||||
class EntityDescription < Struct.new(:value, :name, :description); end
|
||||
|
||||
class EntityLookup
|
||||
###
|
||||
# Look up entity with +name+
|
||||
def [](name)
|
||||
(val = get(name)) && val.value
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,48 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
module Nokogiri
|
||||
module HTML4
|
||||
###
|
||||
# Nokogiri provides a SAX parser to process HTML4 which will provide HTML recovery
|
||||
# ("autocorrection") features.
|
||||
#
|
||||
# See Nokogiri::HTML4::SAX::Parser for a basic example of using a SAX parser with HTML.
|
||||
#
|
||||
# For more information on SAX parsers, see Nokogiri::XML::SAX
|
||||
#
|
||||
module SAX
|
||||
###
|
||||
# This parser is a SAX style parser that reads its input as it deems necessary. The parser
|
||||
# takes a Nokogiri::XML::SAX::Document, an optional encoding, then given an HTML input, sends
|
||||
# messages to the Nokogiri::XML::SAX::Document.
|
||||
#
|
||||
# ⚠ This is an HTML4 parser and so may not support some HTML5 features and behaviors.
|
||||
#
|
||||
# Here is a basic usage example:
|
||||
#
|
||||
# class MyHandler < Nokogiri::XML::SAX::Document
|
||||
# def start_element name, attributes = []
|
||||
# puts "found a #{name}"
|
||||
# end
|
||||
# end
|
||||
#
|
||||
# parser = Nokogiri::HTML4::SAX::Parser.new(MyHandler.new)
|
||||
#
|
||||
# # Hand an IO object to the parser, which will read the HTML from the IO.
|
||||
# File.open(path_to_html) do |f|
|
||||
# parser.parse(f)
|
||||
# end
|
||||
#
|
||||
# For more information on \SAX parsers, see Nokogiri::XML::SAX or the parent class
|
||||
# Nokogiri::XML::SAX::Parser.
|
||||
#
|
||||
# Also see Nokogiri::XML::SAX::Document for the available events.
|
||||
#
|
||||
class Parser < Nokogiri::XML::SAX::Parser
|
||||
# this class inherits its behavior from Nokogiri::XML::SAX::Parser, but note that superclass
|
||||
# uses Nokogiri::ClassResolver to use HTML4::SAX::ParserContext as the context class for
|
||||
# this class, which is where the real behavioral differences are implemented.
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,15 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
module Nokogiri
|
||||
module HTML4
|
||||
module SAX
|
||||
###
|
||||
# Context object to invoke the HTML4 SAX parser on the SAX::Document handler.
|
||||
#
|
||||
# 💡 This class is usually not instantiated by the user. Use Nokogiri::HTML4::SAX::Parser
|
||||
# instead.
|
||||
class ParserContext < Nokogiri::XML::SAX::ParserContext
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,37 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
module Nokogiri
|
||||
module HTML4
|
||||
module SAX
|
||||
class PushParser
|
||||
# The Nokogiri::HTML4::SAX::Document on which the PushParser will be
|
||||
# operating
|
||||
attr_accessor :document
|
||||
|
||||
def initialize(doc = HTML4::SAX::Document.new, file_name = nil, encoding = "UTF-8")
|
||||
@document = doc
|
||||
@encoding = encoding
|
||||
@sax_parser = HTML4::SAX::Parser.new(doc, @encoding)
|
||||
|
||||
## Create our push parser context
|
||||
initialize_native(@sax_parser, file_name, encoding)
|
||||
end
|
||||
|
||||
###
|
||||
# Write a +chunk+ of HTML to the PushParser. Any callback methods
|
||||
# that can be called will be called immediately.
|
||||
def write(chunk, last_chunk = false)
|
||||
native_write(chunk, last_chunk)
|
||||
end
|
||||
alias_method :<<, :write
|
||||
|
||||
###
|
||||
# Finish the parsing. This method is only necessary for
|
||||
# Nokogiri::HTML4::SAX::Document#end_document to be called.
|
||||
def finish
|
||||
write("", true)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
Reference in New Issue
Block a user