This commit is contained in:
@@ -0,0 +1,161 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
require_relative '../table/cmap'
|
||||
require_relative '../table/glyf'
|
||||
require_relative '../table/head'
|
||||
require_relative '../table/hhea'
|
||||
require_relative '../table/hmtx'
|
||||
require_relative '../table/kern'
|
||||
require_relative '../table/loca'
|
||||
require_relative '../table/maxp'
|
||||
require_relative '../table/name'
|
||||
require_relative '../table/post'
|
||||
require_relative '../table/simple'
|
||||
|
||||
module TTFunk
|
||||
module Subset
|
||||
# Base subset.
|
||||
#
|
||||
# @api private
|
||||
class Base
|
||||
# Microsoft Platform ID
|
||||
MICROSOFT_PLATFORM_ID = 3
|
||||
|
||||
# Symbol Encoding ID for Microsoft Platform
|
||||
MS_SYMBOL_ENCODING_ID = 0
|
||||
|
||||
# Original font
|
||||
#
|
||||
# @return [TTFunk::File]
|
||||
attr_reader :original
|
||||
|
||||
# @param original [TTFunk::File]
|
||||
def initialize(original)
|
||||
@original = original
|
||||
end
|
||||
|
||||
# Is this Unicode-based subset?
|
||||
#
|
||||
# @return [Boolean]
|
||||
def unicode?
|
||||
false
|
||||
end
|
||||
|
||||
# Does this subset use Microsoft Symbolic encoding?
|
||||
#
|
||||
# @return [Boolean]
|
||||
def microsoft_symbol?
|
||||
new_cmap_table[:platform_id] == MICROSOFT_PLATFORM_ID &&
|
||||
new_cmap_table[:encoding_id] == MS_SYMBOL_ENCODING_ID
|
||||
end
|
||||
|
||||
# Get a mapping from this subset to Unicode.
|
||||
#
|
||||
# @return [Hash{Integer => Integer}]
|
||||
def to_unicode_map
|
||||
{}
|
||||
end
|
||||
|
||||
# Encode this subset into a binary font representation.
|
||||
#
|
||||
# @param options [Hash]
|
||||
# @return [String]
|
||||
def encode(options = {})
|
||||
encoder_klass.new(original, self, options).encode
|
||||
end
|
||||
|
||||
# Encoder class for this subset.
|
||||
#
|
||||
# @return [TTFunk::TTFEncoder, TTFunk::OTFEncoder]
|
||||
def encoder_klass
|
||||
original.cff.exists? ? OTFEncoder : TTFEncoder
|
||||
end
|
||||
|
||||
# Get the first Unicode cmap from the original font.
|
||||
#
|
||||
# @return [TTFunk::Table::Cmap::Subtable]
|
||||
def unicode_cmap
|
||||
@unicode_cmap ||= @original.cmap.unicode.first
|
||||
end
|
||||
|
||||
# Get glyphs in this subset.
|
||||
#
|
||||
# @return [Hash{Integer => TTFunk::Table::Glyf::Simple,
|
||||
# TTFunk::Table::Glyf::Compound}] if original is a TrueType font
|
||||
# @return [Hash{Integer => TTFunk::Table::Cff::Charstring] if original is
|
||||
# a CFF-based OpenType font
|
||||
def glyphs
|
||||
@glyphs ||= collect_glyphs(original_glyph_ids)
|
||||
end
|
||||
|
||||
# Get glyphs by their IDs in the original font.
|
||||
#
|
||||
# @param glyph_ids [Array<Integer>]
|
||||
# @return [Hash{Integer => TTFunk::Table::Glyf::Simple,
|
||||
# TTFunk::Table::Glyf::Compound>] if original is a TrueType font
|
||||
# @return [Hash{Integer => TTFunk::Table::Cff::Charstring}] if original is
|
||||
# a CFF-based OpenType font
|
||||
def collect_glyphs(glyph_ids)
|
||||
collected =
|
||||
glyph_ids.each_with_object({}) do |id, h|
|
||||
h[id] = glyph_for(id)
|
||||
end
|
||||
|
||||
additional_ids = collected.values
|
||||
.select { |g| g && g.compound? }
|
||||
.map(&:glyph_ids)
|
||||
.flatten
|
||||
|
||||
collected.update(collect_glyphs(additional_ids)) if additional_ids.any?
|
||||
|
||||
collected
|
||||
end
|
||||
|
||||
# Glyph ID mapping from the original font to this subset.
|
||||
#
|
||||
# @return [Hash{Integer => Integer}]
|
||||
def old_to_new_glyph
|
||||
@old_to_new_glyph ||=
|
||||
begin
|
||||
charmap = new_cmap_table[:charmap]
|
||||
old_to_new =
|
||||
charmap.each_with_object(0 => 0) do |(_, ids), map|
|
||||
map[ids[:old]] = ids[:new]
|
||||
end
|
||||
|
||||
next_glyph_id = new_cmap_table[:max_glyph_id]
|
||||
|
||||
glyphs.each_key do |old_id|
|
||||
unless old_to_new.key?(old_id)
|
||||
old_to_new[old_id] = next_glyph_id
|
||||
next_glyph_id += 1
|
||||
end
|
||||
end
|
||||
|
||||
old_to_new
|
||||
end
|
||||
end
|
||||
|
||||
# Glyph ID mapping from this subset to the original font.
|
||||
#
|
||||
# @return [Hash{Integer => Integer}]
|
||||
def new_to_old_glyph
|
||||
@new_to_old_glyph ||= old_to_new_glyph.invert
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def glyph_for(glyph_id)
|
||||
if original.cff.exists?
|
||||
original
|
||||
.cff
|
||||
.top_index[0]
|
||||
.charstrings_index[glyph_id]
|
||||
.glyph
|
||||
else
|
||||
original.glyph_outlines.for(glyph_id)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,133 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
require 'set'
|
||||
|
||||
require_relative 'base'
|
||||
|
||||
module TTFunk
|
||||
module Subset
|
||||
# A subset that uses standard code page encoding.
|
||||
class CodePage < Base
|
||||
class << self
|
||||
# Get a mapping from an encoding to Unicode
|
||||
#
|
||||
# @param encoding [Encoding, String, Symbol]
|
||||
# @return [Hash{Integer => Integer}]
|
||||
def unicode_mapping_for(encoding)
|
||||
mapping_cache[encoding] ||=
|
||||
(0..255).each_with_object({}) do |c, ret|
|
||||
codepoint =
|
||||
c.chr(encoding)
|
||||
.encode(Encoding::UTF_8, undef: :replace, replace: '')
|
||||
.codepoints
|
||||
.first
|
||||
ret[c] = codepoint if codepoint
|
||||
end
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def mapping_cache
|
||||
@mapping_cache ||= {}
|
||||
end
|
||||
end
|
||||
|
||||
# Code page used in this subset.
|
||||
# This is used for proper `OS/2` table encoding.
|
||||
# @return [Integer]
|
||||
attr_reader :code_page
|
||||
|
||||
# Encoding used in this subset.
|
||||
# @return [Encoding, String, Symbol]
|
||||
attr_reader :encoding
|
||||
|
||||
# @param original [TTFunk::File]
|
||||
# @param code_page [Integer]
|
||||
# @param encoding [Encoding, String, Symbol]
|
||||
def initialize(original, code_page, encoding)
|
||||
super(original)
|
||||
@code_page = code_page
|
||||
@encoding = encoding
|
||||
@subset = Array.new(256)
|
||||
@from_unicode_cache = {}
|
||||
use(space_char_code)
|
||||
end
|
||||
|
||||
# Get a mapping from this subset to Unicode.
|
||||
#
|
||||
# @return [Hash]
|
||||
def to_unicode_map
|
||||
self.class.unicode_mapping_for(encoding)
|
||||
.select { |codepoint, _unicode| @subset[codepoint] }
|
||||
end
|
||||
|
||||
# Add a character to subset.
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [void]
|
||||
def use(character)
|
||||
@subset[from_unicode(character)] = character
|
||||
end
|
||||
|
||||
# Can this subset include the character? This depends on the encoding used
|
||||
# in this subset.
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [Boolean]
|
||||
def covers?(character)
|
||||
!from_unicode(character).nil?
|
||||
end
|
||||
|
||||
# Does this subset actually has the character?
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [Boolean]
|
||||
def includes?(character)
|
||||
code = from_unicode(character)
|
||||
code && @subset[code]
|
||||
end
|
||||
|
||||
# Get character code for Unicode codepoint.
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [Integer, nil]
|
||||
def from_unicode(character)
|
||||
@from_unicode_cache[character] ||= (+'' << character).encode!(encoding).ord
|
||||
rescue Encoding::UndefinedConversionError
|
||||
nil
|
||||
end
|
||||
|
||||
# Get `cmap` table for this subset.
|
||||
#
|
||||
# @return [TTFunk::Table::Cmap]
|
||||
def new_cmap_table
|
||||
@new_cmap_table ||=
|
||||
begin
|
||||
mapping = {}
|
||||
|
||||
@subset.each_with_index do |unicode, roman|
|
||||
mapping[roman] = unicode_cmap[unicode]
|
||||
end
|
||||
|
||||
TTFunk::Table::Cmap.encode(mapping, :mac_roman)
|
||||
end
|
||||
end
|
||||
|
||||
# Get the list of Glyph IDs from the original font that are in this
|
||||
# subset.
|
||||
#
|
||||
# @return [Array<Integer>]
|
||||
def original_glyph_ids
|
||||
([0] + @subset.map { |unicode| unicode && unicode_cmap[unicode] })
|
||||
.compact.uniq.sort
|
||||
end
|
||||
|
||||
# Get a chacter code for Space in this subset
|
||||
#
|
||||
# @return [Integer, nil]
|
||||
def space_char_code
|
||||
@space_char_code ||= from_unicode(Unicode::SPACE_CHAR)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,17 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
require 'set'
|
||||
|
||||
require_relative 'code_page'
|
||||
|
||||
module TTFunk
|
||||
module Subset
|
||||
# Mac Roman subset. It uses code page 10,000 and Mac OS Roman encoding.
|
||||
class MacRoman < CodePage
|
||||
# @param original [TTFunk::File]
|
||||
def initialize(original)
|
||||
super(original, 10_000, Encoding::MACROMAN)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,90 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
require 'set'
|
||||
require_relative 'base'
|
||||
|
||||
module TTFunk
|
||||
module Subset
|
||||
# Unicode-based subset.
|
||||
class Unicode < Base
|
||||
# Space character code
|
||||
SPACE_CHAR = 0x20
|
||||
|
||||
# @param original [TTFunk::File]
|
||||
def initialize(original)
|
||||
super
|
||||
@subset = Set.new
|
||||
use(SPACE_CHAR)
|
||||
end
|
||||
|
||||
# Is this a Unicode-based subset?
|
||||
#
|
||||
# @return [true]
|
||||
def unicode?
|
||||
true
|
||||
end
|
||||
|
||||
# Get a mapping from this subset to Unicode.
|
||||
#
|
||||
# @return [Hash{Integer => Integer}]
|
||||
def to_unicode_map
|
||||
@subset.each_with_object({}) { |code, map| map[code] = code }
|
||||
end
|
||||
|
||||
# Add a character to subset.
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [void]
|
||||
def use(character)
|
||||
@subset << character
|
||||
end
|
||||
|
||||
# Can this subset include the character?
|
||||
#
|
||||
# @param _character [Integer] Unicode codepoint
|
||||
# @return [true]
|
||||
def covers?(_character)
|
||||
true
|
||||
end
|
||||
|
||||
# Does this subset actually has the character?
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [Boolean]
|
||||
def includes?(character)
|
||||
@subset.include?(character)
|
||||
end
|
||||
|
||||
# Get character code for Unicode codepoint.
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [Integer]
|
||||
def from_unicode(character)
|
||||
character
|
||||
end
|
||||
|
||||
# Get `cmap` table for this subset.
|
||||
#
|
||||
# @return [TTFunk::Table::Cmap]
|
||||
def new_cmap_table
|
||||
@new_cmap_table ||=
|
||||
begin
|
||||
mapping =
|
||||
@subset.each_with_object({}) do |code, map|
|
||||
map[code] = unicode_cmap[code]
|
||||
end
|
||||
|
||||
TTFunk::Table::Cmap.encode(mapping, :unicode)
|
||||
end
|
||||
end
|
||||
|
||||
# Get the list of Glyph IDs from the original font that are in this
|
||||
# subset.
|
||||
#
|
||||
# @return [Array<Integer>]
|
||||
def original_glyph_ids
|
||||
([0] + @subset.map { |code| unicode_cmap[code] }).uniq.sort
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,99 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
require 'set'
|
||||
require_relative 'base'
|
||||
|
||||
module TTFunk
|
||||
module Subset
|
||||
# An 8-bit Unicode-based subset. It can include any Unicode character but
|
||||
# limits number of characters so that the could be encoded by a single byte.
|
||||
class Unicode8Bit < Base
|
||||
# @param original [TTFunk::File]
|
||||
def initialize(original)
|
||||
super
|
||||
@subset = { 0x20 => 0x20 }
|
||||
@unicodes = { 0x20 => 0x20 }
|
||||
@next = 0x21 # apparently, PDF's don't like to use chars between 0-31
|
||||
end
|
||||
|
||||
# Is this a Unicode-based subset?
|
||||
#
|
||||
# @return [true]
|
||||
def unicode?
|
||||
true
|
||||
end
|
||||
|
||||
# Get a mapping from this subset to Unicode.
|
||||
#
|
||||
# @return [Hash{Integer => Integer}]
|
||||
def to_unicode_map
|
||||
@subset.dup
|
||||
end
|
||||
|
||||
# Add a character to subset.
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [void]
|
||||
def use(character)
|
||||
unless @unicodes.key?(character)
|
||||
@subset[@next] = character
|
||||
@unicodes[character] = @next
|
||||
@next += 1
|
||||
end
|
||||
end
|
||||
|
||||
# Can this subset include the character?
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [Boolean]
|
||||
def covers?(character)
|
||||
@unicodes.key?(character) || @next < 256
|
||||
end
|
||||
|
||||
# Does this subset actually has the character?
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [Boolean]
|
||||
def includes?(character)
|
||||
@unicodes.key?(character)
|
||||
end
|
||||
|
||||
# Get character code for Unicode codepoint.
|
||||
#
|
||||
# @param character [Integer] Unicode codepoint
|
||||
# @return [Integer]
|
||||
def from_unicode(character)
|
||||
@unicodes[character]
|
||||
end
|
||||
|
||||
# Get `cmap` table for this subset.
|
||||
#
|
||||
# @return [TTFunk::Table::Cmap]
|
||||
def new_cmap_table
|
||||
@new_cmap_table ||=
|
||||
begin
|
||||
mapping =
|
||||
@subset.each_with_object({}) do |(code, unicode), map|
|
||||
map[code] = unicode_cmap[unicode]
|
||||
map
|
||||
end
|
||||
|
||||
# since we're mapping a subset of the unicode glyphs into an
|
||||
# arbitrary 256-character space, the actual encoding we're
|
||||
# using is irrelevant. We choose MacRoman because it's a 256-character
|
||||
# encoding that happens to be well-supported in both TTF and
|
||||
# PDF formats.
|
||||
TTFunk::Table::Cmap.encode(mapping, :mac_roman)
|
||||
end
|
||||
end
|
||||
|
||||
# Get the list of Glyph IDs from the original font that are in this
|
||||
# subset.
|
||||
#
|
||||
# @return [Array<Integer>]
|
||||
def original_glyph_ids
|
||||
([0] + @unicodes.keys.map { |unicode| unicode_cmap[unicode] }).uniq.sort
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,17 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
require 'set'
|
||||
|
||||
require_relative 'code_page'
|
||||
|
||||
module TTFunk
|
||||
module Subset
|
||||
# Windows 1252 sbset. It uses code page 1252 and Windows-1252 encoding.
|
||||
class Windows1252 < CodePage
|
||||
# @param original [TTFunk::File]
|
||||
def initialize(original)
|
||||
super(original, 1252, Encoding::CP1252)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
Reference in New Issue
Block a user