This commit is contained in:
@@ -0,0 +1,177 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
# = Public Suffix
|
||||
#
|
||||
# Domain name parser based on the Public Suffix List.
|
||||
#
|
||||
# Copyright (c) 2009-2026 Simone Carletti <weppos@weppos.net>
|
||||
|
||||
require_relative "public_suffix/domain"
|
||||
require_relative "public_suffix/version"
|
||||
require_relative "public_suffix/errors"
|
||||
require_relative "public_suffix/rule"
|
||||
require_relative "public_suffix/list"
|
||||
|
||||
# PublicSuffix is a Ruby domain name parser based on the Public Suffix List.
|
||||
#
|
||||
# The [Public Suffix List](https://publicsuffix.org) is a cross-vendor initiative
|
||||
# to provide an accurate list of domain name suffixes.
|
||||
#
|
||||
# The Public Suffix List is an initiative of the Mozilla Project,
|
||||
# but is maintained as a community resource. It is available for use in any software,
|
||||
# but was originally created to meet the needs of browser manufacturers.
|
||||
module PublicSuffix
|
||||
|
||||
DOT = "."
|
||||
BANG = "!"
|
||||
STAR = "*"
|
||||
|
||||
# Parses +name+ and returns the {PublicSuffix::Domain} instance.
|
||||
#
|
||||
# @example Parse a valid domain
|
||||
# PublicSuffix.parse("google.com")
|
||||
# # => #<PublicSuffix::Domain:0x007fec2e51e588 @sld="google", @tld="com", @trd=nil>
|
||||
#
|
||||
# @example Parse a valid subdomain
|
||||
# PublicSuffix.parse("www.google.com")
|
||||
# # => #<PublicSuffix::Domain:0x007fec276d4cf8 @sld="google", @tld="com", @trd="www">
|
||||
#
|
||||
# @example Parse a fully qualified domain
|
||||
# PublicSuffix.parse("google.com.")
|
||||
# # => #<PublicSuffix::Domain:0x007fec257caf38 @sld="google", @tld="com", @trd=nil>
|
||||
#
|
||||
# @example Parse a fully qualified domain (subdomain)
|
||||
# PublicSuffix.parse("www.google.com.")
|
||||
# # => #<PublicSuffix::Domain:0x007fec27b6bca8 @sld="google", @tld="com", @trd="www">
|
||||
#
|
||||
# @example Parse an invalid (unlisted) domain
|
||||
# PublicSuffix.parse("x.yz")
|
||||
# # => #<PublicSuffix::Domain:0x007fec2f49bec0 @sld="x", @tld="yz", @trd=nil>
|
||||
#
|
||||
# @example Parse an invalid (unlisted) domain with strict checking (without applying the default * rule)
|
||||
# PublicSuffix.parse("x.yz", default_rule: nil)
|
||||
# # => PublicSuffix::DomainInvalid: `x.yz` is not a valid domain
|
||||
#
|
||||
# @example Parse an URL (not supported, only domains)
|
||||
# PublicSuffix.parse("http://www.google.com")
|
||||
# # => PublicSuffix::DomainInvalid: http://www.google.com is not expected to contain a scheme
|
||||
#
|
||||
#
|
||||
# @param name [#to_s] The domain name or fully qualified domain name to parse.
|
||||
# @param list [PublicSuffix::List] The rule list to search, defaults to the default {PublicSuffix::List}
|
||||
# @param ignore_private [Boolean]
|
||||
# @return [PublicSuffix::Domain]
|
||||
#
|
||||
# @raise [PublicSuffix::DomainInvalid] If domain is not a valid domain.
|
||||
# @raise [PublicSuffix::DomainNotAllowed] If a rule for +domain+ is found, but the rule doesn't allow +domain+.
|
||||
def self.parse(name, list: List.default, default_rule: list.default_rule, ignore_private: false)
|
||||
what = normalize(name)
|
||||
raise what if what.is_a?(DomainInvalid)
|
||||
|
||||
rule = list.find(what, default: default_rule, ignore_private: ignore_private)
|
||||
|
||||
# rubocop:disable Style/IfUnlessModifier
|
||||
if rule.nil?
|
||||
raise DomainInvalid, "`#{what}` is not a valid domain"
|
||||
end
|
||||
if rule.decompose(what).last.nil?
|
||||
raise DomainNotAllowed, "`#{what}` is not allowed according to Registry policy"
|
||||
end
|
||||
|
||||
# rubocop:enable Style/IfUnlessModifier
|
||||
|
||||
decompose(what, rule)
|
||||
end
|
||||
|
||||
# Checks whether +domain+ is assigned and allowed, without actually parsing it.
|
||||
#
|
||||
# This method doesn't care whether domain is a domain or subdomain.
|
||||
# The validation is performed using the default {PublicSuffix::List}.
|
||||
#
|
||||
# @example Validate a valid domain
|
||||
# PublicSuffix.valid?("example.com")
|
||||
# # => true
|
||||
#
|
||||
# @example Validate a valid subdomain
|
||||
# PublicSuffix.valid?("www.example.com")
|
||||
# # => true
|
||||
#
|
||||
# @example Validate a not-listed domain
|
||||
# PublicSuffix.valid?("example.tldnotlisted")
|
||||
# # => true
|
||||
#
|
||||
# @example Validate a not-listed domain with strict checking (without applying the default * rule)
|
||||
# PublicSuffix.valid?("example.tldnotlisted")
|
||||
# # => true
|
||||
# PublicSuffix.valid?("example.tldnotlisted", default_rule: nil)
|
||||
# # => false
|
||||
#
|
||||
# @example Validate a fully qualified domain
|
||||
# PublicSuffix.valid?("google.com.")
|
||||
# # => true
|
||||
# PublicSuffix.valid?("www.google.com.")
|
||||
# # => true
|
||||
#
|
||||
# @example Check an URL (which is not a valid domain)
|
||||
# PublicSuffix.valid?("http://www.example.com")
|
||||
# # => false
|
||||
#
|
||||
#
|
||||
# @param name [#to_s] The domain name or fully qualified domain name to validate.
|
||||
# @param ignore_private [Boolean]
|
||||
# @return [Boolean]
|
||||
def self.valid?(name, list: List.default, default_rule: list.default_rule, ignore_private: false)
|
||||
what = normalize(name)
|
||||
return false if what.is_a?(DomainInvalid)
|
||||
|
||||
rule = list.find(what, default: default_rule, ignore_private: ignore_private)
|
||||
|
||||
!rule.nil? && !rule.decompose(what).last.nil?
|
||||
end
|
||||
|
||||
# Attempt to parse the name and returns the domain, if valid.
|
||||
#
|
||||
# This method doesn't raise. Instead, it returns nil if the domain is not valid for whatever reason.
|
||||
#
|
||||
# @param name [#to_s] The domain name or fully qualified domain name to parse.
|
||||
# @param list [PublicSuffix::List] The rule list to search, defaults to the default {PublicSuffix::List}
|
||||
# @param ignore_private [Boolean]
|
||||
# @return [String]
|
||||
def self.domain(name, **options)
|
||||
parse(name, **options).domain
|
||||
rescue PublicSuffix::Error
|
||||
nil
|
||||
end
|
||||
|
||||
|
||||
# private
|
||||
|
||||
def self.decompose(name, rule)
|
||||
left, right = rule.decompose(name)
|
||||
|
||||
parts = left.split(DOT)
|
||||
# If we have 0 parts left, there is just a tld and no domain or subdomain
|
||||
# If we have 1 part left, there is just a tld, domain and not subdomain
|
||||
# If we have 2 parts left, the last part is the domain, the other parts (combined) are the subdomain
|
||||
tld = right
|
||||
sld = parts.empty? ? nil : parts.pop
|
||||
trd = parts.empty? ? nil : parts.join(DOT)
|
||||
|
||||
Domain.new(tld, sld, trd)
|
||||
end
|
||||
|
||||
# Pretend we know how to deal with user input.
|
||||
def self.normalize(name)
|
||||
name = name.to_s.dup
|
||||
name.strip!
|
||||
name.chomp!(DOT)
|
||||
name.downcase!
|
||||
|
||||
return DomainInvalid.new("Name is blank") if name.empty?
|
||||
return DomainInvalid.new("Name starts with a dot") if name.start_with?(DOT)
|
||||
return DomainInvalid.new(format("%s is not expected to contain a scheme", name)) if name.include?("://")
|
||||
|
||||
name
|
||||
end
|
||||
|
||||
end
|
||||
@@ -0,0 +1,235 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
# = Public Suffix
|
||||
#
|
||||
# Domain name parser based on the Public Suffix List.
|
||||
#
|
||||
# Copyright (c) 2009-2026 Simone Carletti <weppos@weppos.net>
|
||||
|
||||
module PublicSuffix
|
||||
|
||||
# Domain represents a domain name, composed by a TLD, SLD and TRD.
|
||||
class Domain
|
||||
|
||||
# Splits a string into the labels, that is the dot-separated parts.
|
||||
#
|
||||
# The input is not validated, but it is assumed to be a valid domain name.
|
||||
#
|
||||
# @example
|
||||
#
|
||||
# name_to_labels('example.com')
|
||||
# # => ['example', 'com']
|
||||
#
|
||||
# name_to_labels('example.co.uk')
|
||||
# # => ['example', 'co', 'uk']
|
||||
#
|
||||
# @param name [String, #to_s] The domain name to split.
|
||||
# @return [Array<String>]
|
||||
def self.name_to_labels(name)
|
||||
name.to_s.split(DOT)
|
||||
end
|
||||
|
||||
|
||||
attr_reader :tld, :sld, :trd
|
||||
|
||||
# Creates and returns a new {PublicSuffix::Domain} instance.
|
||||
#
|
||||
# @overload initialize(tld)
|
||||
# Initializes with a +tld+.
|
||||
# @param [String] tld The TLD (extension)
|
||||
# @overload initialize(tld, sld)
|
||||
# Initializes with a +tld+ and +sld+.
|
||||
# @param [String] tld The TLD (extension)
|
||||
# @param [String] sld The TRD (domain)
|
||||
# @overload initialize(tld, sld, trd)
|
||||
# Initializes with a +tld+, +sld+ and +trd+.
|
||||
# @param [String] tld The TLD (extension)
|
||||
# @param [String] sld The SLD (domain)
|
||||
# @param [String] trd The TRD (subdomain)
|
||||
#
|
||||
# @yield [self] Yields on self.
|
||||
# @yieldparam [PublicSuffix::Domain] self The newly creates instance
|
||||
#
|
||||
# @example Initialize with a TLD
|
||||
# PublicSuffix::Domain.new("com")
|
||||
# # => #<PublicSuffix::Domain @tld="com">
|
||||
#
|
||||
# @example Initialize with a TLD and SLD
|
||||
# PublicSuffix::Domain.new("com", "example")
|
||||
# # => #<PublicSuffix::Domain @tld="com", @trd=nil>
|
||||
#
|
||||
# @example Initialize with a TLD, SLD and TRD
|
||||
# PublicSuffix::Domain.new("com", "example", "wwww")
|
||||
# # => #<PublicSuffix::Domain @tld="com", @trd=nil, @sld="example">
|
||||
#
|
||||
def initialize(*args)
|
||||
@tld, @sld, @trd = args
|
||||
yield(self) if block_given?
|
||||
end
|
||||
|
||||
# Returns a string representation of this object.
|
||||
#
|
||||
# @return [String]
|
||||
def to_s
|
||||
name
|
||||
end
|
||||
|
||||
# Returns an array containing the domain parts.
|
||||
#
|
||||
# @return [Array<String, nil>]
|
||||
#
|
||||
# @example
|
||||
#
|
||||
# PublicSuffix::Domain.new("google.com").to_a
|
||||
# # => [nil, "google", "com"]
|
||||
#
|
||||
# PublicSuffix::Domain.new("www.google.com").to_a
|
||||
# # => [nil, "google", "com"]
|
||||
#
|
||||
def to_a
|
||||
[@trd, @sld, @tld]
|
||||
end
|
||||
|
||||
# Returns the full domain name.
|
||||
#
|
||||
# @return [String]
|
||||
#
|
||||
# @example Gets the domain name of a domain
|
||||
# PublicSuffix::Domain.new("com", "google").name
|
||||
# # => "google.com"
|
||||
#
|
||||
# @example Gets the domain name of a subdomain
|
||||
# PublicSuffix::Domain.new("com", "google", "www").name
|
||||
# # => "www.google.com"
|
||||
#
|
||||
def name
|
||||
[@trd, @sld, @tld].compact.join(DOT)
|
||||
end
|
||||
|
||||
# Returns a domain-like representation of this object
|
||||
# if the object is a {#domain?}, <tt>nil</tt> otherwise.
|
||||
#
|
||||
# PublicSuffix::Domain.new("com").domain
|
||||
# # => nil
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google").domain
|
||||
# # => "google.com"
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google", "www").domain
|
||||
# # => "www.google.com"
|
||||
#
|
||||
# This method doesn't validate the input. It handles the domain
|
||||
# as a valid domain name and simply applies the necessary transformations.
|
||||
#
|
||||
# This method returns a FQD, not just the domain part.
|
||||
# To get the domain part, use <tt>#sld</tt> (aka second level domain).
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google", "www").domain
|
||||
# # => "google.com"
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google", "www").sld
|
||||
# # => "google"
|
||||
#
|
||||
# @see #domain?
|
||||
# @see #subdomain
|
||||
#
|
||||
# @return [String]
|
||||
def domain
|
||||
[@sld, @tld].join(DOT) if domain?
|
||||
end
|
||||
|
||||
# Returns a subdomain-like representation of this object
|
||||
# if the object is a {#subdomain?}, <tt>nil</tt> otherwise.
|
||||
#
|
||||
# PublicSuffix::Domain.new("com").subdomain
|
||||
# # => nil
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google").subdomain
|
||||
# # => nil
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google", "www").subdomain
|
||||
# # => "www.google.com"
|
||||
#
|
||||
# This method doesn't validate the input. It handles the domain
|
||||
# as a valid domain name and simply applies the necessary transformations.
|
||||
#
|
||||
# This method returns a FQD, not just the subdomain part.
|
||||
# To get the subdomain part, use <tt>#trd</tt> (aka third level domain).
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google", "www").subdomain
|
||||
# # => "www.google.com"
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google", "www").trd
|
||||
# # => "www"
|
||||
#
|
||||
# @see #subdomain?
|
||||
# @see #domain
|
||||
#
|
||||
# @return [String]
|
||||
def subdomain
|
||||
[@trd, @sld, @tld].join(DOT) if subdomain?
|
||||
end
|
||||
|
||||
# Checks whether <tt>self</tt> looks like a domain.
|
||||
#
|
||||
# This method doesn't actually validate the domain.
|
||||
# It only checks whether the instance contains
|
||||
# a value for the {#tld} and {#sld} attributes.
|
||||
#
|
||||
# @example
|
||||
#
|
||||
# PublicSuffix::Domain.new("com").domain?
|
||||
# # => false
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google").domain?
|
||||
# # => true
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google", "www").domain?
|
||||
# # => true
|
||||
#
|
||||
# # This is an invalid domain, but returns true
|
||||
# # because this method doesn't validate the content.
|
||||
# PublicSuffix::Domain.new("com", nil).domain?
|
||||
# # => true
|
||||
#
|
||||
# @see #subdomain?
|
||||
#
|
||||
# @return [Boolean]
|
||||
def domain?
|
||||
!(@tld.nil? || @sld.nil?)
|
||||
end
|
||||
|
||||
# Checks whether <tt>self</tt> looks like a subdomain.
|
||||
#
|
||||
# This method doesn't actually validate the subdomain.
|
||||
# It only checks whether the instance contains
|
||||
# a value for the {#tld}, {#sld} and {#trd} attributes.
|
||||
# If you also want to validate the domain,
|
||||
# use {#valid_subdomain?} instead.
|
||||
#
|
||||
# @example
|
||||
#
|
||||
# PublicSuffix::Domain.new("com").subdomain?
|
||||
# # => false
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google").subdomain?
|
||||
# # => false
|
||||
#
|
||||
# PublicSuffix::Domain.new("com", "google", "www").subdomain?
|
||||
# # => true
|
||||
#
|
||||
# # This is an invalid domain, but returns true
|
||||
# # because this method doesn't validate the content.
|
||||
# PublicSuffix::Domain.new("com", "example", nil).subdomain?
|
||||
# # => true
|
||||
#
|
||||
# @see #domain?
|
||||
#
|
||||
# @return [Boolean]
|
||||
def subdomain?
|
||||
!(@tld.nil? || @sld.nil? || @trd.nil?)
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
end
|
||||
@@ -0,0 +1,41 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
# = Public Suffix
|
||||
#
|
||||
# Domain name parser based on the Public Suffix List.
|
||||
#
|
||||
# Copyright (c) 2009-2026 Simone Carletti <weppos@weppos.net>
|
||||
|
||||
module PublicSuffix
|
||||
|
||||
class Error < StandardError
|
||||
end
|
||||
|
||||
# Raised when trying to parse an invalid name.
|
||||
# A name is considered invalid when no rule is found in the definition list.
|
||||
#
|
||||
# @example
|
||||
#
|
||||
# PublicSuffix.parse("nic.test")
|
||||
# # => PublicSuffix::DomainInvalid
|
||||
#
|
||||
# PublicSuffix.parse("http://www.nic.it")
|
||||
# # => PublicSuffix::DomainInvalid
|
||||
#
|
||||
class DomainInvalid < Error
|
||||
end
|
||||
|
||||
# Raised when trying to parse a name that matches a suffix.
|
||||
#
|
||||
# @example
|
||||
#
|
||||
# PublicSuffix.parse("nic.do")
|
||||
# # => PublicSuffix::DomainNotAllowed
|
||||
#
|
||||
# PublicSuffix.parse("www.nic.do")
|
||||
# # => PublicSuffix::Domain
|
||||
#
|
||||
class DomainNotAllowed < DomainInvalid
|
||||
end
|
||||
|
||||
end
|
||||
@@ -0,0 +1,247 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
# = Public Suffix
|
||||
#
|
||||
# Domain name parser based on the Public Suffix List.
|
||||
#
|
||||
# Copyright (c) 2009-2026 Simone Carletti <weppos@weppos.net>
|
||||
|
||||
module PublicSuffix
|
||||
|
||||
# A {PublicSuffix::List} is a collection of one
|
||||
# or more {PublicSuffix::Rule}.
|
||||
#
|
||||
# Given a {PublicSuffix::List},
|
||||
# you can add or remove {PublicSuffix::Rule},
|
||||
# iterate all items in the list or search for the first rule
|
||||
# which matches a specific domain name.
|
||||
#
|
||||
# # Create a new list
|
||||
# list = PublicSuffix::List.new
|
||||
#
|
||||
# # Push two rules to the list
|
||||
# list << PublicSuffix::Rule.factory("it")
|
||||
# list << PublicSuffix::Rule.factory("com")
|
||||
#
|
||||
# # Get the size of the list
|
||||
# list.size
|
||||
# # => 2
|
||||
#
|
||||
# # Search for the rule matching given domain
|
||||
# list.find("example.com")
|
||||
# # => #<PublicSuffix::Rule::Normal>
|
||||
# list.find("example.org")
|
||||
# # => nil
|
||||
#
|
||||
# You can create as many {PublicSuffix::List} you want.
|
||||
# The {PublicSuffix::List.default} rule list is used
|
||||
# to tokenize and validate a domain.
|
||||
#
|
||||
class List
|
||||
|
||||
DEFAULT_LIST_PATH = File.expand_path("../../data/list.txt", __dir__)
|
||||
|
||||
# Gets the default rule list.
|
||||
#
|
||||
# Initializes a new {PublicSuffix::List} parsing the content
|
||||
# of {PublicSuffix::List.default_list_content}, if required.
|
||||
#
|
||||
# @return [PublicSuffix::List]
|
||||
def self.default(**options)
|
||||
@default ||= parse(File.read(DEFAULT_LIST_PATH), **options)
|
||||
end
|
||||
|
||||
# Sets the default rule list to +value+.
|
||||
#
|
||||
# @param value [PublicSuffix::List] the new list
|
||||
# @return [PublicSuffix::List]
|
||||
def self.default=(value)
|
||||
@default = value
|
||||
end
|
||||
|
||||
# Parse given +input+ treating the content as Public Suffix List.
|
||||
#
|
||||
# See http://publicsuffix.org/format/ for more details about input format.
|
||||
#
|
||||
# @param input [#each_line] the list to parse
|
||||
# @param private_domains [Boolean] whether to ignore the private domains section
|
||||
# @return [PublicSuffix::List]
|
||||
def self.parse(input, private_domains: true)
|
||||
comment_token = "//"
|
||||
private_token = "===BEGIN PRIVATE DOMAINS==="
|
||||
section = nil # 1 == ICANN, 2 == PRIVATE
|
||||
|
||||
new do |list|
|
||||
input.each_line do |line|
|
||||
line.strip!
|
||||
case # rubocop:disable Style/EmptyCaseCondition
|
||||
|
||||
# skip blank lines
|
||||
when line.empty?
|
||||
next
|
||||
|
||||
# include private domains or stop scanner
|
||||
when line.include?(private_token)
|
||||
break if !private_domains
|
||||
|
||||
section = 2
|
||||
|
||||
# skip comments
|
||||
when line.start_with?(comment_token) # rubocop:disable Lint/DuplicateBranch
|
||||
next
|
||||
|
||||
else
|
||||
list.add(Rule.factory(line, private: section == 2))
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
# Initializes an empty {PublicSuffix::List}.
|
||||
#
|
||||
# @yield [self] Yields on self.
|
||||
# @yieldparam [PublicSuffix::List] self The newly created instance.
|
||||
def initialize
|
||||
@rules = {}
|
||||
yield(self) if block_given?
|
||||
end
|
||||
|
||||
|
||||
# Checks whether two lists are equal.
|
||||
#
|
||||
# List <tt>one</tt> is equal to <tt>two</tt>, if <tt>two</tt> is an instance of
|
||||
# {PublicSuffix::List} and each +PublicSuffix::Rule::*+
|
||||
# in list <tt>one</tt> is available in list <tt>two</tt>, in the same order.
|
||||
#
|
||||
# @param other [PublicSuffix::List] the List to compare
|
||||
# @return [Boolean]
|
||||
def ==(other)
|
||||
return false unless other.is_a?(List)
|
||||
|
||||
equal?(other) || @rules == other.rules
|
||||
end
|
||||
alias eql? ==
|
||||
|
||||
# Iterates each rule in the list.
|
||||
def each(&block)
|
||||
Enumerator.new do |y|
|
||||
@rules.each do |key, node|
|
||||
y << entry_to_rule(node, key)
|
||||
end
|
||||
end.each(&block)
|
||||
end
|
||||
|
||||
|
||||
# Adds the given object to the list and optionally refreshes the rule index.
|
||||
#
|
||||
# @param rule [PublicSuffix::Rule::*] the rule to add to the list
|
||||
# @return [self]
|
||||
def add(rule)
|
||||
@rules[rule.value] = rule_to_entry(rule)
|
||||
self
|
||||
end
|
||||
alias << add
|
||||
|
||||
# Gets the number of rules in the list.
|
||||
#
|
||||
# @return [Integer]
|
||||
def size
|
||||
@rules.size
|
||||
end
|
||||
|
||||
# Checks whether the list is empty.
|
||||
#
|
||||
# @return [Boolean]
|
||||
def empty?
|
||||
@rules.empty?
|
||||
end
|
||||
|
||||
# Removes all rules.
|
||||
#
|
||||
# @return [self]
|
||||
def clear
|
||||
@rules.clear
|
||||
self
|
||||
end
|
||||
|
||||
# Finds and returns the rule corresponding to the longest public suffix for the hostname.
|
||||
#
|
||||
# @param name [#to_s] the hostname
|
||||
# @param default [PublicSuffix::Rule::*] the default rule to return in case no rule matches
|
||||
# @return [PublicSuffix::Rule::*]
|
||||
def find(name, default: default_rule, **options)
|
||||
rule = select(name, **options).inject do |l, r|
|
||||
return r if r.instance_of?(Rule::Exception)
|
||||
|
||||
l.length > r.length ? l : r
|
||||
end
|
||||
rule || default
|
||||
end
|
||||
|
||||
# Selects all the rules matching given hostame.
|
||||
#
|
||||
# If `ignore_private` is set to true, the algorithm will skip the rules that are flagged as
|
||||
# private domain. Note that the rules will still be part of the loop.
|
||||
# If you frequently need to access lists ignoring the private domains,
|
||||
# you should create a list that doesn't include these domains setting the
|
||||
# `private_domains: false` option when calling {.parse}.
|
||||
#
|
||||
# Note that this method is currently private, as you should not rely on it. Instead,
|
||||
# the public interface is {#find}. The current internal algorithm allows to return all
|
||||
# matching rules, but different data structures may not be able to do it, and instead would
|
||||
# return only the match. For this reason, you should rely on {#find}.
|
||||
#
|
||||
# @param name [#to_s] the hostname
|
||||
# @param ignore_private [Boolean]
|
||||
# @return [Array<PublicSuffix::Rule::*>]
|
||||
def select(name, ignore_private: false)
|
||||
name = name.to_s
|
||||
|
||||
parts = name.split(DOT).reverse!
|
||||
index = 0
|
||||
query = parts[index]
|
||||
rules = []
|
||||
|
||||
loop do
|
||||
match = @rules[query]
|
||||
rules << entry_to_rule(match, query) if !match.nil? && (ignore_private == false || match.private == false)
|
||||
|
||||
index += 1
|
||||
break if index >= parts.size
|
||||
|
||||
query = parts[index] + DOT + query
|
||||
end
|
||||
|
||||
rules
|
||||
end
|
||||
private :select
|
||||
|
||||
# Gets the default rule.
|
||||
#
|
||||
# @see PublicSuffix::Rule.default_rule
|
||||
#
|
||||
# @return [PublicSuffix::Rule::*]
|
||||
def default_rule
|
||||
PublicSuffix::Rule.default
|
||||
end
|
||||
|
||||
|
||||
protected
|
||||
|
||||
attr_reader :rules
|
||||
|
||||
|
||||
private
|
||||
|
||||
def entry_to_rule(entry, value)
|
||||
entry.type.new(value: value, length: entry.length, private: entry.private)
|
||||
end
|
||||
|
||||
def rule_to_entry(rule)
|
||||
Rule::Entry.new(rule.class, rule.length, rule.private)
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,350 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
# = Public Suffix
|
||||
#
|
||||
# Domain name parser based on the Public Suffix List.
|
||||
#
|
||||
# Copyright (c) 2009-2026 Simone Carletti <weppos@weppos.net>
|
||||
|
||||
module PublicSuffix
|
||||
|
||||
# A Rule is a special object which holds a single definition
|
||||
# of the Public Suffix List.
|
||||
#
|
||||
# There are 3 types of rules, each one represented by a specific
|
||||
# subclass within the +PublicSuffix::Rule+ namespace.
|
||||
#
|
||||
# To create a new Rule, use the {PublicSuffix::Rule#factory} method.
|
||||
#
|
||||
# PublicSuffix::Rule.factory("ar")
|
||||
# # => #<PublicSuffix::Rule::Normal>
|
||||
#
|
||||
module Rule
|
||||
|
||||
# @api internal
|
||||
Entry = Struct.new(:type, :length, :private) # rubocop:disable Lint/StructNewOverride
|
||||
|
||||
# = Abstract rule class
|
||||
#
|
||||
# This represent the base class for a Rule definition
|
||||
# in the {Public Suffix List}[https://publicsuffix.org].
|
||||
#
|
||||
# This is intended to be an Abstract class
|
||||
# and you shouldn't create a direct instance. The only purpose
|
||||
# of this class is to expose a common interface
|
||||
# for all the available subclasses.
|
||||
#
|
||||
# * {PublicSuffix::Rule::Normal}
|
||||
# * {PublicSuffix::Rule::Exception}
|
||||
# * {PublicSuffix::Rule::Wildcard}
|
||||
#
|
||||
# ## Properties
|
||||
#
|
||||
# A rule is composed by 4 properties:
|
||||
#
|
||||
# value - A normalized version of the rule name.
|
||||
# The normalization process depends on rule tpe.
|
||||
#
|
||||
# Here's an example
|
||||
#
|
||||
# PublicSuffix::Rule.factory("*.google.com")
|
||||
# #<PublicSuffix::Rule::Wildcard:0x1015c14b0
|
||||
# @value="google.com"
|
||||
# >
|
||||
#
|
||||
# ## Rule Creation
|
||||
#
|
||||
# The best way to create a new rule is passing the rule name
|
||||
# to the <tt>PublicSuffix::Rule.factory</tt> method.
|
||||
#
|
||||
# PublicSuffix::Rule.factory("com")
|
||||
# # => PublicSuffix::Rule::Normal
|
||||
#
|
||||
# PublicSuffix::Rule.factory("*.com")
|
||||
# # => PublicSuffix::Rule::Wildcard
|
||||
#
|
||||
# This method will detect the rule type and create an instance
|
||||
# from the proper rule class.
|
||||
#
|
||||
# ## Rule Usage
|
||||
#
|
||||
# A rule describes the composition of a domain name and explains how to tokenize
|
||||
# the name into tld, sld and trd.
|
||||
#
|
||||
# To use a rule, you first need to be sure the name you want to tokenize
|
||||
# can be handled by the current rule.
|
||||
# You can use the <tt>#match?</tt> method.
|
||||
#
|
||||
# rule = PublicSuffix::Rule.factory("com")
|
||||
#
|
||||
# rule.match?("google.com")
|
||||
# # => true
|
||||
#
|
||||
# rule.match?("google.com")
|
||||
# # => false
|
||||
#
|
||||
# Rule order is significant. A name can match more than one rule.
|
||||
# See the {Public Suffix Documentation}[http://publicsuffix.org/format/]
|
||||
# to learn more about rule priority.
|
||||
#
|
||||
# When you have the right rule, you can use it to tokenize the domain name.
|
||||
#
|
||||
# rule = PublicSuffix::Rule.factory("com")
|
||||
#
|
||||
# rule.decompose("google.com")
|
||||
# # => ["google", "com"]
|
||||
#
|
||||
# rule.decompose("www.google.com")
|
||||
# # => ["www.google", "com"]
|
||||
#
|
||||
# @abstract
|
||||
#
|
||||
class Base
|
||||
|
||||
# @return [String] the rule definition
|
||||
attr_reader :value
|
||||
|
||||
# @return [String] the length of the rule
|
||||
attr_reader :length
|
||||
|
||||
# @return [Boolean] true if the rule is a private domain
|
||||
attr_reader :private
|
||||
|
||||
|
||||
# Initializes a new rule from the content.
|
||||
#
|
||||
# @param content [String] the content of the rule
|
||||
# @param private [Boolean]
|
||||
def self.build(content, private: false)
|
||||
new(value: content, private: private)
|
||||
end
|
||||
|
||||
# Initializes a new rule.
|
||||
#
|
||||
# @param value [String]
|
||||
# @param private [Boolean]
|
||||
def initialize(value:, length: nil, private: false)
|
||||
@value = value.to_s
|
||||
@length = length || (@value.count(DOT) + 1)
|
||||
@private = private
|
||||
end
|
||||
|
||||
# Checks whether this rule is equal to <tt>other</tt>.
|
||||
#
|
||||
# @param other [PublicSuffix::Rule::*] The rule to compare
|
||||
# @return [Boolean] true if this rule and other are instances of the same class
|
||||
# and has the same value, false otherwise.
|
||||
def ==(other)
|
||||
equal?(other) || (self.class == other.class && value == other.value)
|
||||
end
|
||||
alias eql? ==
|
||||
|
||||
# Checks if this rule matches +name+.
|
||||
#
|
||||
# A domain name is said to match a rule if and only if
|
||||
# all of the following conditions are met:
|
||||
#
|
||||
# - When the domain and rule are split into corresponding labels,
|
||||
# that the domain contains as many or more labels than the rule.
|
||||
# - Beginning with the right-most labels of both the domain and the rule,
|
||||
# and continuing for all labels in the rule, one finds that for every pair,
|
||||
# either they are identical, or that the label from the rule is "*".
|
||||
#
|
||||
# @see https://publicsuffix.org/list/
|
||||
#
|
||||
# @example
|
||||
# PublicSuffix::Rule.factory("com").match?("example.com")
|
||||
# # => true
|
||||
# PublicSuffix::Rule.factory("com").match?("example.net")
|
||||
# # => false
|
||||
#
|
||||
# @param name [String] the domain name to check
|
||||
# @return [Boolean]
|
||||
def match?(name)
|
||||
# NOTE: it works because of the assumption there are no
|
||||
# rules like foo.*.com. If the assumption is incorrect,
|
||||
# we need to properly walk the input and skip parts according
|
||||
# to wildcard component.
|
||||
diff = name.chomp(value)
|
||||
diff.empty? || diff.end_with?(DOT)
|
||||
end
|
||||
|
||||
# @abstract
|
||||
def parts
|
||||
raise NotImplementedError
|
||||
end
|
||||
|
||||
# @abstract
|
||||
# @param domain [#to_s] The domain name to decompose
|
||||
# @return [Array<String, nil>]
|
||||
def decompose(*)
|
||||
raise NotImplementedError
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
# Normal represents a standard rule (e.g. com).
|
||||
class Normal < Base
|
||||
|
||||
# Gets the original rule definition.
|
||||
#
|
||||
# @return [String] The rule definition.
|
||||
def rule
|
||||
value
|
||||
end
|
||||
|
||||
# Decomposes the domain name according to rule properties.
|
||||
#
|
||||
# @param domain [#to_s] The domain name to decompose
|
||||
# @return [Array<String>] The array with [trd + sld, tld].
|
||||
def decompose(domain)
|
||||
suffix = parts.join('\.')
|
||||
matches = domain.to_s.match(/^(.*)\.(#{suffix})$/)
|
||||
matches ? matches[1..2] : [nil, nil]
|
||||
end
|
||||
|
||||
# dot-split rule value and returns all rule parts
|
||||
# in the order they appear in the value.
|
||||
#
|
||||
# @return [Array<String>]
|
||||
def parts
|
||||
@value.split(DOT)
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
# Wildcard represents a wildcard rule (e.g. *.co.uk).
|
||||
class Wildcard < Base
|
||||
|
||||
# Initializes a new rule from the content.
|
||||
#
|
||||
# @param content [String] the content of the rule
|
||||
# @param private [Boolean]
|
||||
def self.build(content, private: false)
|
||||
new(value: content.to_s[2..], private: private)
|
||||
end
|
||||
|
||||
# Initializes a new rule.
|
||||
#
|
||||
# @param value [String]
|
||||
# @param length [Integer]
|
||||
# @param private [Boolean]
|
||||
def initialize(value:, length: nil, private: false)
|
||||
super
|
||||
length or @length += 1 # * counts as 1
|
||||
end
|
||||
|
||||
# Gets the original rule definition.
|
||||
#
|
||||
# @return [String] The rule definition.
|
||||
def rule
|
||||
value == "" ? STAR : STAR + DOT + value
|
||||
end
|
||||
|
||||
# Decomposes the domain name according to rule properties.
|
||||
#
|
||||
# @param domain [#to_s] The domain name to decompose
|
||||
# @return [Array<String>] The array with [trd + sld, tld].
|
||||
def decompose(domain)
|
||||
suffix = ([".*?"] + parts).join('\.')
|
||||
matches = domain.to_s.match(/^(.*)\.(#{suffix})$/)
|
||||
matches ? matches[1..2] : [nil, nil]
|
||||
end
|
||||
|
||||
# dot-split rule value and returns all rule parts
|
||||
# in the order they appear in the value.
|
||||
#
|
||||
# @return [Array<String>]
|
||||
def parts
|
||||
@value.split(DOT)
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
# Exception represents an exception rule (e.g. !parliament.uk).
|
||||
class Exception < Base
|
||||
|
||||
# Initializes a new rule from the content.
|
||||
#
|
||||
# @param content [#to_s] the content of the rule
|
||||
# @param private [Boolean]
|
||||
def self.build(content, private: false)
|
||||
new(value: content.to_s[1..], private: private)
|
||||
end
|
||||
|
||||
# Gets the original rule definition.
|
||||
#
|
||||
# @return [String] The rule definition.
|
||||
def rule
|
||||
BANG + value
|
||||
end
|
||||
|
||||
# Decomposes the domain name according to rule properties.
|
||||
#
|
||||
# @param domain [#to_s] The domain name to decompose
|
||||
# @return [Array<String>] The array with [trd + sld, tld].
|
||||
def decompose(domain)
|
||||
suffix = parts.join('\.')
|
||||
matches = domain.to_s.match(/^(.*)\.(#{suffix})$/)
|
||||
matches ? matches[1..2] : [nil, nil]
|
||||
end
|
||||
|
||||
# dot-split rule value and returns all rule parts
|
||||
# in the order they appear in the value.
|
||||
# The leftmost label is not considered a label.
|
||||
#
|
||||
# See http://publicsuffix.org/format/:
|
||||
# If the prevailing rule is a exception rule,
|
||||
# modify it by removing the leftmost label.
|
||||
#
|
||||
# @return [Array<String>]
|
||||
def parts
|
||||
@value.split(DOT)[1..]
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
|
||||
# Takes the +name+ of the rule, detects the specific rule class
|
||||
# and creates a new instance of that class.
|
||||
# The +name+ becomes the rule +value+.
|
||||
#
|
||||
# @example Creates a Normal rule
|
||||
# PublicSuffix::Rule.factory("ar")
|
||||
# # => #<PublicSuffix::Rule::Normal>
|
||||
#
|
||||
# @example Creates a Wildcard rule
|
||||
# PublicSuffix::Rule.factory("*.ar")
|
||||
# # => #<PublicSuffix::Rule::Wildcard>
|
||||
#
|
||||
# @example Creates an Exception rule
|
||||
# PublicSuffix::Rule.factory("!congresodelalengua3.ar")
|
||||
# # => #<PublicSuffix::Rule::Exception>
|
||||
#
|
||||
# @param content [#to_s] the content of the rule
|
||||
# @return [PublicSuffix::Rule::*] A rule instance.
|
||||
def self.factory(content, private: false)
|
||||
case content.to_s[0, 1]
|
||||
when STAR
|
||||
Wildcard
|
||||
when BANG
|
||||
Exception
|
||||
else
|
||||
Normal
|
||||
end.build(content, private: private)
|
||||
end
|
||||
|
||||
# The default rule to use if no rule match.
|
||||
#
|
||||
# The default rule is "*". From https://publicsuffix.org/list/:
|
||||
#
|
||||
# > If no rules match, the prevailing rule is "*".
|
||||
#
|
||||
# @return [PublicSuffix::Rule::Wildcard] The default rule.
|
||||
def self.default
|
||||
factory(STAR)
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
end
|
||||
@@ -0,0 +1,14 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
# = Public Suffix
|
||||
#
|
||||
# Domain name parser based on the Public Suffix List.
|
||||
#
|
||||
# Copyright (c) 2009-2026 Simone Carletti <weppos@weppos.net>
|
||||
|
||||
module PublicSuffix
|
||||
|
||||
# @return [String] the current library version
|
||||
VERSION = "7.0.5"
|
||||
|
||||
end
|
||||
Reference in New Issue
Block a user