This commit is contained in:
@@ -0,0 +1,23 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
|
||||
# List all callbacks generated by each page
|
||||
#
|
||||
# WARNING: this will generate a *lot* of output, so you probably want to pipe
|
||||
# it through less or to a text file.
|
||||
|
||||
require 'rubygems'
|
||||
require 'pdf/reader'
|
||||
|
||||
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cairo-basic.pdf"
|
||||
|
||||
PDF::Reader.open(filename) do |reader|
|
||||
reader.pages.each do |page|
|
||||
receiver = PDF::Reader::RegisterReceiver.new
|
||||
page.walk(receiver)
|
||||
|
||||
receiver.callbacks.each do |cb|
|
||||
puts cb
|
||||
end
|
||||
end
|
||||
end
|
||||
+50
@@ -0,0 +1,50 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
|
||||
# A sample script that attempts to extract bates numbers from a PDF file.
|
||||
# Bates numbers are often used to markup documents being used in legal
|
||||
# cases. For more info, see http://en.wikipedia.org/wiki/Bates_numbering
|
||||
#
|
||||
# Acrobat 9 introduced a markup syntax that directly specifies the bates
|
||||
# number for each page. For earlier versions, the easiest way to find
|
||||
# the number is to look for words that match a pattern.
|
||||
#
|
||||
# This example attempts to extract numbers using the Acrobat 9 syntax.
|
||||
# As a fall back, you can use a regular expression to look for words
|
||||
# that match the numbers you expect in the page content.
|
||||
|
||||
require 'rubygems'
|
||||
require 'pdf/reader'
|
||||
|
||||
class BatesReceiver
|
||||
|
||||
attr_reader :numbers
|
||||
|
||||
def initialize
|
||||
@numbers = []
|
||||
end
|
||||
|
||||
def begin_marked_content(*args)
|
||||
return unless args.size >= 2
|
||||
return unless args.first == :Artifact
|
||||
return unless args[1][:Subtype] == :BatesN
|
||||
|
||||
@numbers << args[1][:Contents]
|
||||
end
|
||||
alias :begin_marked_content_with_pl :begin_marked_content
|
||||
|
||||
end
|
||||
|
||||
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cairo-basic.pdf"
|
||||
|
||||
PDF::Reader.open(filename) do |reader|
|
||||
reader.pages.each do |page|
|
||||
receiver = BatesReceiver.new
|
||||
page.walk(receiver)
|
||||
if receiver.numbers.empty?
|
||||
puts page.text.scan(/CC.+/)
|
||||
else
|
||||
puts receiver.numbers.inspect
|
||||
end
|
||||
end
|
||||
end
|
||||
+82
@@ -0,0 +1,82 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
|
||||
# This demonstrates a way to extract TTF fonts from a PDF. It could be expanded
|
||||
# to support extra font formats if required. Be aware that many PDFs subset
|
||||
# fonts before they're embedded so glyphs may be missing or re-arranged.
|
||||
|
||||
require 'pdf/reader'
|
||||
|
||||
module ExtractFonts
|
||||
|
||||
class Extractor
|
||||
|
||||
def page(page)
|
||||
count = 0
|
||||
|
||||
return count if page.fonts.nil? || page.fonts.empty?
|
||||
|
||||
page.fonts.each do |label, font|
|
||||
next if complete_refs[label]
|
||||
complete_refs[label] = true
|
||||
|
||||
process_font(page, font)
|
||||
|
||||
count += 1
|
||||
end
|
||||
|
||||
count
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def process_font(page, font)
|
||||
font = page.objects.deref(font)
|
||||
|
||||
case font[:Subtype]
|
||||
when :Type0 then
|
||||
font[:DescendantFonts].each { |f| process_font(page, f) }
|
||||
when :TrueType, :CIDFontType2 then
|
||||
ExtractFonts::TTF.new(page.objects, font).save("#{font[:BaseFont]}.ttf")
|
||||
else
|
||||
$stderr.puts "unsupported font type #{font[:Subtype]} for #{font[:BaseFont]}"
|
||||
end
|
||||
end
|
||||
|
||||
def complete_refs
|
||||
@complete_refs ||= {}
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
class TTF
|
||||
|
||||
def initialize(objects, font)
|
||||
@objects, @font = objects, font
|
||||
@descriptor = @objects.deref(@font[:FontDescriptor])
|
||||
end
|
||||
|
||||
def save(filename)
|
||||
puts "#{filename}"
|
||||
if @descriptor && @descriptor[:FontFile2]
|
||||
stream = @objects.deref(@descriptor[:FontFile2])
|
||||
File.open(filename, "wb") { |file| file.write stream.unfiltered_data }
|
||||
else
|
||||
$stderr.puts "- TTF font not embedded"
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
if ARGV.size == 0 # default file name
|
||||
ARGV << File.expand_path(File.join(File.dirname(__dir__), "spec", "data", "cairo-unicode.pdf"))
|
||||
end
|
||||
|
||||
extractor = ExtractFonts::Extractor.new
|
||||
|
||||
ARGV.each do |arg|
|
||||
PDF::Reader.open(arg) do |reader|
|
||||
page = reader.page(1)
|
||||
extractor.page(page)
|
||||
end
|
||||
end
|
||||
+231
@@ -0,0 +1,231 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
|
||||
# This demonstrates a way to extract some images (those based on the JPG or
|
||||
# TIFF formats) from a PDF. There are other ways to store images, so
|
||||
# it may need to be expanded for real world usage, but it should serve
|
||||
# as a good guide.
|
||||
#
|
||||
# Thanks to Jack Rusher for the initial version of this example.
|
||||
|
||||
require 'pdf/reader'
|
||||
|
||||
module ExtractImages
|
||||
|
||||
class Extractor
|
||||
|
||||
def page(page)
|
||||
process_page(page, 0)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def complete_refs
|
||||
@complete_refs ||= {}
|
||||
end
|
||||
|
||||
def process_page(page, count)
|
||||
xobjects = page.xobjects
|
||||
return count if xobjects.empty?
|
||||
|
||||
xobjects.each do |name, stream|
|
||||
case stream.hash[:Subtype]
|
||||
when :Image then
|
||||
count += 1
|
||||
|
||||
case stream.hash[:Filter]
|
||||
when :CCITTFaxDecode then
|
||||
ExtractImages::Tiff.new(stream).save("#{page.number}-#{count}-#{name}.tif")
|
||||
when :DCTDecode then
|
||||
ExtractImages::Jpg.new(stream).save("#{page.number}-#{count}-#{name}.jpg")
|
||||
else
|
||||
ExtractImages::Raw.new(stream).save("#{page.number}-#{count}-#{name}.tif")
|
||||
end
|
||||
when :Form then
|
||||
count = process_page(PDF::Reader::FormXObject.new(page, stream), count)
|
||||
end
|
||||
end
|
||||
count
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
class Raw
|
||||
attr_reader :stream
|
||||
|
||||
def initialize(stream)
|
||||
@stream = stream
|
||||
end
|
||||
|
||||
def save(filename)
|
||||
case @stream.hash[:ColorSpace]
|
||||
when :DeviceCMYK then save_cmyk(filename)
|
||||
when :DeviceGray then save_gray(filename)
|
||||
when :DeviceRGB then save_rgb(filename)
|
||||
else
|
||||
$stderr.puts "unsupport color depth #{@stream.hash[:ColorSpace]} #{filename}"
|
||||
end
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def save_cmyk(filename)
|
||||
h = stream.hash[:Height]
|
||||
w = stream.hash[:Width]
|
||||
bpc = stream.hash[:BitsPerComponent]
|
||||
len = stream.hash[:Length]
|
||||
puts "#{filename}: h=#{h}, w=#{w}, bpc=#{bpc}, len=#{len}"
|
||||
|
||||
# Synthesize a TIFF header
|
||||
long_tag = lambda {|tag, count, value| [ tag, 4, count, value ].pack( "ssII" ) }
|
||||
short_tag = lambda {|tag, count, value| [ tag, 3, count, value ].pack( "ssII" ) }
|
||||
# header = byte order, version magic, offset of directory, directory count,
|
||||
# followed by a series of tags containing metadata.
|
||||
tag_count = 10
|
||||
header = [ 73, 73, 42, 8, tag_count ].pack("ccsIs")
|
||||
tiff = header.dup
|
||||
tiff << short_tag.call( 256, 1, w ) # image width
|
||||
tiff << short_tag.call( 257, 1, h ) # image height
|
||||
tiff << long_tag.call( 258, 4, (header.size + (tag_count*12) + 4)) # bits per pixel
|
||||
tiff << short_tag.call( 259, 1, 1 ) # compression
|
||||
tiff << short_tag.call( 262, 1, 5 ) # colorspace - separation
|
||||
tiff << long_tag.call( 273, 1, (10 + (tag_count*12) + 20) ) # data offset
|
||||
tiff << short_tag.call( 277, 1, 4 ) # samples per pixel
|
||||
tiff << long_tag.call( 279, 1, stream.unfiltered_data.size) # data byte size
|
||||
tiff << short_tag.call( 284, 1, 1 ) # planer config
|
||||
tiff << long_tag.call( 332, 1, 1) # inkset - CMYK
|
||||
tiff << [0].pack("I") # next IFD pointer
|
||||
tiff << [bpc, bpc, bpc, bpc].pack("IIII")
|
||||
tiff << stream.unfiltered_data
|
||||
File.open(filename, "wb") { |file| file.write tiff }
|
||||
end
|
||||
|
||||
def save_gray(filename)
|
||||
h = stream.hash[:Height]
|
||||
w = stream.hash[:Width]
|
||||
bpc = stream.hash[:BitsPerComponent]
|
||||
len = stream.hash[:Length]
|
||||
puts "#{filename}: h=#{h}, w=#{w}, bpc=#{bpc}, len=#{len}"
|
||||
|
||||
# Synthesize a TIFF header
|
||||
long_tag = lambda {|tag, count, value| [ tag, 4, count, value ].pack( "ssII" ) }
|
||||
short_tag = lambda {|tag, count, value| [ tag, 3, count, value ].pack( "ssII" ) }
|
||||
# header = byte order, version magic, offset of directory, directory count,
|
||||
# followed by a series of tags containing metadata.
|
||||
tag_count = 9
|
||||
header = [ 73, 73, 42, 8, tag_count ].pack("ccsIs")
|
||||
tiff = header.dup
|
||||
tiff << short_tag.call( 256, 1, w ) # image width
|
||||
tiff << short_tag.call( 257, 1, h ) # image height
|
||||
tiff << short_tag.call( 258, 1, 8 ) # bits per pixel
|
||||
tiff << short_tag.call( 259, 1, 1 ) # compression
|
||||
tiff << short_tag.call( 262, 1, 1 ) # colorspace - grayscale
|
||||
tiff << long_tag.call( 273, 1, (10 + (tag_count*12) + 4) ) # data offset
|
||||
tiff << short_tag.call( 277, 1, 1 ) # samples per pixel
|
||||
tiff << long_tag.call( 279, 1, stream.unfiltered_data.size) # data byte size
|
||||
tiff << short_tag.call( 284, 1, 1 ) # planer config
|
||||
tiff << [0].pack("I") # next IFD pointer
|
||||
p stream.unfiltered_data.size
|
||||
tiff << stream.unfiltered_data
|
||||
File.open(filename, "wb") { |file| file.write tiff }
|
||||
end
|
||||
|
||||
def save_rgb(filename)
|
||||
h = stream.hash[:Height]
|
||||
w = stream.hash[:Width]
|
||||
bpc = stream.hash[:BitsPerComponent]
|
||||
len = stream.hash[:Length]
|
||||
puts "#{filename}: h=#{h}, w=#{w}, bpc=#{bpc}, len=#{len}"
|
||||
|
||||
# Synthesize a TIFF header
|
||||
long_tag = lambda {|tag, count, value| [ tag, 4, count, value ].pack( "ssII" ) }
|
||||
short_tag = lambda {|tag, count, value| [ tag, 3, count, value ].pack( "ssII" ) }
|
||||
# header = byte order, version magic, offset of directory, directory count,
|
||||
# followed by a series of tags containing metadata.
|
||||
tag_count = 8
|
||||
header = [ 73, 73, 42, 8, tag_count ].pack("ccsIs")
|
||||
tiff = header.dup
|
||||
tiff << short_tag.call( 256, 1, w ) # image width
|
||||
tiff << short_tag.call( 257, 1, h ) # image height
|
||||
tiff << long_tag.call( 258, 3, (header.size + (tag_count*12) + 4)) # bits per pixel
|
||||
tiff << short_tag.call( 259, 1, 1 ) # compression
|
||||
tiff << short_tag.call( 262, 1, 2 ) # colorspace - RGB
|
||||
tiff << long_tag.call( 273, 1, (header.size + (tag_count*12) + 16) ) # data offset
|
||||
tiff << short_tag.call( 277, 1, 3 ) # samples per pixel
|
||||
tiff << long_tag.call( 279, 1, stream.unfiltered_data.size) # data byte size
|
||||
tiff << [0].pack("I") # next IFD pointer
|
||||
tiff << [bpc, bpc, bpc].pack("III")
|
||||
tiff << stream.unfiltered_data
|
||||
File.open(filename, "wb") { |file| file.write tiff }
|
||||
end
|
||||
end
|
||||
|
||||
class Jpg
|
||||
attr_reader :stream
|
||||
|
||||
def initialize(stream)
|
||||
@stream = stream
|
||||
end
|
||||
|
||||
def save(filename)
|
||||
w = stream.hash[:Width]
|
||||
h = stream.hash[:Height]
|
||||
puts "#{filename}: h=#{h}, w=#{w}"
|
||||
File.open(filename, "wb") { |file| file.write stream.data }
|
||||
end
|
||||
end
|
||||
|
||||
class Tiff
|
||||
attr_reader :stream
|
||||
|
||||
def initialize(stream)
|
||||
@stream = stream
|
||||
end
|
||||
|
||||
def save(filename)
|
||||
if stream.hash[:DecodeParms][:K] <= 0
|
||||
save_group_four(filename)
|
||||
else
|
||||
$stderr.puts "#{filename}: CCITT non-group 4/2D image."
|
||||
end
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
# Group 4, 2D
|
||||
def save_group_four(filename)
|
||||
k = stream.hash[:DecodeParms][:K]
|
||||
h = stream.hash[:Height]
|
||||
w = stream.hash[:Width]
|
||||
bpc = stream.hash[:BitsPerComponent]
|
||||
mask = stream.hash[:ImageMask]
|
||||
len = stream.hash[:Length]
|
||||
cols = stream.hash[:DecodeParms][:Columns]
|
||||
puts "#{filename}: h=#{h}, w=#{w}, bpc=#{bpc}, mask=#{mask}, len=#{len}, cols=#{cols}, k=#{k}"
|
||||
|
||||
# Synthesize a TIFF header
|
||||
long_tag = lambda {|tag, value| [ tag, 4, 1, value ].pack( "ssII" ) }
|
||||
short_tag = lambda {|tag, value| [ tag, 3, 1, value ].pack( "ssII" ) }
|
||||
# header = byte order, version magic, offset of directory, directory count,
|
||||
# followed by a series of tags containing metadata: 259 is a magic number for
|
||||
# the compression type; 273 is the offset of the image data.
|
||||
tiff = [ 73, 73, 42, 8, 5 ].pack("ccsIs") \
|
||||
+ short_tag.call( 256, cols ) \
|
||||
+ short_tag.call( 257, h ) \
|
||||
+ short_tag.call( 259, 4 ) \
|
||||
+ long_tag.call( 273, (10 + (5*12) + 4) ) \
|
||||
+ long_tag.call( 279, len) \
|
||||
+ [0].pack("I") \
|
||||
+ stream.data
|
||||
File.open(filename, "wb") { |file| file.write tiff }
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/adobe_sample.pdf"
|
||||
extractor = ExtractImages::Extractor.new
|
||||
|
||||
PDF::Reader.open(filename) do |reader|
|
||||
page = reader.page(1)
|
||||
extractor.page(page)
|
||||
end
|
||||
@@ -0,0 +1,24 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
|
||||
# Extract an (imperfect) array of paragraphs divided somewhat
|
||||
# arbitrarily on line length.
|
||||
|
||||
require 'pdf/reader'
|
||||
|
||||
reader = PDF::Reader.new('somefile.pdf')
|
||||
|
||||
paragraph = ""
|
||||
paragraphs = []
|
||||
reader.pages.each do |page|
|
||||
lines = page.text.scan(/^.+/)
|
||||
lines.each do |line|
|
||||
if line.length > 55
|
||||
paragraph += " #{line}"
|
||||
else
|
||||
paragraph += " #{line}"
|
||||
paragraphs << paragraph
|
||||
paragraph = ""
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,12 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
|
||||
# get direct access to PDF objects
|
||||
|
||||
require 'pdf/reader'
|
||||
|
||||
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cairo-unicode.pdf"
|
||||
|
||||
reader = PDF::Reader.new(filename)
|
||||
puts reader.objects[3]
|
||||
puts reader.objects[4]
|
||||
@@ -0,0 +1,14 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
|
||||
# Extract metadata only
|
||||
|
||||
require 'rubygems'
|
||||
require 'pdf/reader'
|
||||
|
||||
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cross_ref_stream.pdf"
|
||||
|
||||
PDF::Reader.open(filename) do |reader|
|
||||
puts reader.info.inspect
|
||||
puts reader.metadata.inspect
|
||||
end
|
||||
@@ -0,0 +1,13 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
|
||||
# A simple app to count the number of pages in a PDF File.
|
||||
|
||||
require 'rubygems'
|
||||
require 'pdf/reader'
|
||||
|
||||
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cross_ref_stream.pdf"
|
||||
|
||||
PDF::Reader.open(filename) do |reader|
|
||||
puts "#{reader.page_count} page(s)"
|
||||
end
|
||||
@@ -0,0 +1,34 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
# typed: ignore
|
||||
|
||||
# Basic RSpec of a generated PDF
|
||||
#
|
||||
# USAGE: rspec -c examples/rspec.rb
|
||||
|
||||
require 'rubygems'
|
||||
require 'pdf/reader'
|
||||
require 'rspec'
|
||||
require 'prawn'
|
||||
require 'stringio'
|
||||
|
||||
describe "My generated PDF" do
|
||||
it "should have the correct text on 2 pages" do
|
||||
|
||||
# generate our PDF
|
||||
pdf = Prawn::Document.new
|
||||
pdf.text "Chunky"
|
||||
pdf.start_new_page
|
||||
pdf.text "Bacon"
|
||||
io = StringIO.new(pdf.render)
|
||||
|
||||
# process the PDF
|
||||
PDF::Reader.open(io) do |reader|
|
||||
reader.page_count.should eql(2) # correct page count
|
||||
|
||||
reader.page(1).text.should eql("Chunky") # correct content
|
||||
reader.page(2).text.should eql("Bacon") # correct content
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,15 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
|
||||
# Extract all text from a single PDF
|
||||
|
||||
require 'rubygems'
|
||||
require 'pdf/reader'
|
||||
|
||||
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cairo-unicode.pdf"
|
||||
|
||||
PDF::Reader.open(filename) do |reader|
|
||||
reader.pages.each do |page|
|
||||
puts page.text
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,13 @@
|
||||
#!/usr/bin/env ruby
|
||||
# coding: utf-8
|
||||
|
||||
# Determine the PDF version of a file
|
||||
|
||||
require 'rubygems'
|
||||
require 'pdf/reader'
|
||||
|
||||
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cairo-basic.pdf"
|
||||
|
||||
PDF::Reader.open(filename) do |reader|
|
||||
puts reader.pdf_version
|
||||
end
|
||||
Reference in New Issue
Block a user