Add bin and edit workflow
Gitea Actions Demo / Explore-Gitea-Actions (push) Failing after 9s

This commit is contained in:
2026-09-16 13:11:16 -06:00
parent c8ac4fcae5
commit 4cee170d66
17576 changed files with 895740 additions and 2 deletions
@@ -0,0 +1,23 @@
#!/usr/bin/env ruby
# coding: utf-8
# List all callbacks generated by each page
#
# WARNING: this will generate a *lot* of output, so you probably want to pipe
# it through less or to a text file.
require 'rubygems'
require 'pdf/reader'
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cairo-basic.pdf"
PDF::Reader.open(filename) do |reader|
reader.pages.each do |page|
receiver = PDF::Reader::RegisterReceiver.new
page.walk(receiver)
receiver.callbacks.each do |cb|
puts cb
end
end
end
+50
View File
@@ -0,0 +1,50 @@
#!/usr/bin/env ruby
# coding: utf-8
# A sample script that attempts to extract bates numbers from a PDF file.
# Bates numbers are often used to markup documents being used in legal
# cases. For more info, see http://en.wikipedia.org/wiki/Bates_numbering
#
# Acrobat 9 introduced a markup syntax that directly specifies the bates
# number for each page. For earlier versions, the easiest way to find
# the number is to look for words that match a pattern.
#
# This example attempts to extract numbers using the Acrobat 9 syntax.
# As a fall back, you can use a regular expression to look for words
# that match the numbers you expect in the page content.
require 'rubygems'
require 'pdf/reader'
class BatesReceiver
attr_reader :numbers
def initialize
@numbers = []
end
def begin_marked_content(*args)
return unless args.size >= 2
return unless args.first == :Artifact
return unless args[1][:Subtype] == :BatesN
@numbers << args[1][:Contents]
end
alias :begin_marked_content_with_pl :begin_marked_content
end
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cairo-basic.pdf"
PDF::Reader.open(filename) do |reader|
reader.pages.each do |page|
receiver = BatesReceiver.new
page.walk(receiver)
if receiver.numbers.empty?
puts page.text.scan(/CC.+/)
else
puts receiver.numbers.inspect
end
end
end
+82
View File
@@ -0,0 +1,82 @@
#!/usr/bin/env ruby
# coding: utf-8
# This demonstrates a way to extract TTF fonts from a PDF. It could be expanded
# to support extra font formats if required. Be aware that many PDFs subset
# fonts before they're embedded so glyphs may be missing or re-arranged.
require 'pdf/reader'
module ExtractFonts
class Extractor
def page(page)
count = 0
return count if page.fonts.nil? || page.fonts.empty?
page.fonts.each do |label, font|
next if complete_refs[label]
complete_refs[label] = true
process_font(page, font)
count += 1
end
count
end
private
def process_font(page, font)
font = page.objects.deref(font)
case font[:Subtype]
when :Type0 then
font[:DescendantFonts].each { |f| process_font(page, f) }
when :TrueType, :CIDFontType2 then
ExtractFonts::TTF.new(page.objects, font).save("#{font[:BaseFont]}.ttf")
else
$stderr.puts "unsupported font type #{font[:Subtype]} for #{font[:BaseFont]}"
end
end
def complete_refs
@complete_refs ||= {}
end
end
class TTF
def initialize(objects, font)
@objects, @font = objects, font
@descriptor = @objects.deref(@font[:FontDescriptor])
end
def save(filename)
puts "#{filename}"
if @descriptor && @descriptor[:FontFile2]
stream = @objects.deref(@descriptor[:FontFile2])
File.open(filename, "wb") { |file| file.write stream.unfiltered_data }
else
$stderr.puts "- TTF font not embedded"
end
end
end
end
if ARGV.size == 0 # default file name
ARGV << File.expand_path(File.join(File.dirname(__dir__), "spec", "data", "cairo-unicode.pdf"))
end
extractor = ExtractFonts::Extractor.new
ARGV.each do |arg|
PDF::Reader.open(arg) do |reader|
page = reader.page(1)
extractor.page(page)
end
end
+231
View File
@@ -0,0 +1,231 @@
#!/usr/bin/env ruby
# coding: utf-8
# This demonstrates a way to extract some images (those based on the JPG or
# TIFF formats) from a PDF. There are other ways to store images, so
# it may need to be expanded for real world usage, but it should serve
# as a good guide.
#
# Thanks to Jack Rusher for the initial version of this example.
require 'pdf/reader'
module ExtractImages
class Extractor
def page(page)
process_page(page, 0)
end
private
def complete_refs
@complete_refs ||= {}
end
def process_page(page, count)
xobjects = page.xobjects
return count if xobjects.empty?
xobjects.each do |name, stream|
case stream.hash[:Subtype]
when :Image then
count += 1
case stream.hash[:Filter]
when :CCITTFaxDecode then
ExtractImages::Tiff.new(stream).save("#{page.number}-#{count}-#{name}.tif")
when :DCTDecode then
ExtractImages::Jpg.new(stream).save("#{page.number}-#{count}-#{name}.jpg")
else
ExtractImages::Raw.new(stream).save("#{page.number}-#{count}-#{name}.tif")
end
when :Form then
count = process_page(PDF::Reader::FormXObject.new(page, stream), count)
end
end
count
end
end
class Raw
attr_reader :stream
def initialize(stream)
@stream = stream
end
def save(filename)
case @stream.hash[:ColorSpace]
when :DeviceCMYK then save_cmyk(filename)
when :DeviceGray then save_gray(filename)
when :DeviceRGB then save_rgb(filename)
else
$stderr.puts "unsupport color depth #{@stream.hash[:ColorSpace]} #{filename}"
end
end
private
def save_cmyk(filename)
h = stream.hash[:Height]
w = stream.hash[:Width]
bpc = stream.hash[:BitsPerComponent]
len = stream.hash[:Length]
puts "#{filename}: h=#{h}, w=#{w}, bpc=#{bpc}, len=#{len}"
# Synthesize a TIFF header
long_tag = lambda {|tag, count, value| [ tag, 4, count, value ].pack( "ssII" ) }
short_tag = lambda {|tag, count, value| [ tag, 3, count, value ].pack( "ssII" ) }
# header = byte order, version magic, offset of directory, directory count,
# followed by a series of tags containing metadata.
tag_count = 10
header = [ 73, 73, 42, 8, tag_count ].pack("ccsIs")
tiff = header.dup
tiff << short_tag.call( 256, 1, w ) # image width
tiff << short_tag.call( 257, 1, h ) # image height
tiff << long_tag.call( 258, 4, (header.size + (tag_count*12) + 4)) # bits per pixel
tiff << short_tag.call( 259, 1, 1 ) # compression
tiff << short_tag.call( 262, 1, 5 ) # colorspace - separation
tiff << long_tag.call( 273, 1, (10 + (tag_count*12) + 20) ) # data offset
tiff << short_tag.call( 277, 1, 4 ) # samples per pixel
tiff << long_tag.call( 279, 1, stream.unfiltered_data.size) # data byte size
tiff << short_tag.call( 284, 1, 1 ) # planer config
tiff << long_tag.call( 332, 1, 1) # inkset - CMYK
tiff << [0].pack("I") # next IFD pointer
tiff << [bpc, bpc, bpc, bpc].pack("IIII")
tiff << stream.unfiltered_data
File.open(filename, "wb") { |file| file.write tiff }
end
def save_gray(filename)
h = stream.hash[:Height]
w = stream.hash[:Width]
bpc = stream.hash[:BitsPerComponent]
len = stream.hash[:Length]
puts "#{filename}: h=#{h}, w=#{w}, bpc=#{bpc}, len=#{len}"
# Synthesize a TIFF header
long_tag = lambda {|tag, count, value| [ tag, 4, count, value ].pack( "ssII" ) }
short_tag = lambda {|tag, count, value| [ tag, 3, count, value ].pack( "ssII" ) }
# header = byte order, version magic, offset of directory, directory count,
# followed by a series of tags containing metadata.
tag_count = 9
header = [ 73, 73, 42, 8, tag_count ].pack("ccsIs")
tiff = header.dup
tiff << short_tag.call( 256, 1, w ) # image width
tiff << short_tag.call( 257, 1, h ) # image height
tiff << short_tag.call( 258, 1, 8 ) # bits per pixel
tiff << short_tag.call( 259, 1, 1 ) # compression
tiff << short_tag.call( 262, 1, 1 ) # colorspace - grayscale
tiff << long_tag.call( 273, 1, (10 + (tag_count*12) + 4) ) # data offset
tiff << short_tag.call( 277, 1, 1 ) # samples per pixel
tiff << long_tag.call( 279, 1, stream.unfiltered_data.size) # data byte size
tiff << short_tag.call( 284, 1, 1 ) # planer config
tiff << [0].pack("I") # next IFD pointer
p stream.unfiltered_data.size
tiff << stream.unfiltered_data
File.open(filename, "wb") { |file| file.write tiff }
end
def save_rgb(filename)
h = stream.hash[:Height]
w = stream.hash[:Width]
bpc = stream.hash[:BitsPerComponent]
len = stream.hash[:Length]
puts "#{filename}: h=#{h}, w=#{w}, bpc=#{bpc}, len=#{len}"
# Synthesize a TIFF header
long_tag = lambda {|tag, count, value| [ tag, 4, count, value ].pack( "ssII" ) }
short_tag = lambda {|tag, count, value| [ tag, 3, count, value ].pack( "ssII" ) }
# header = byte order, version magic, offset of directory, directory count,
# followed by a series of tags containing metadata.
tag_count = 8
header = [ 73, 73, 42, 8, tag_count ].pack("ccsIs")
tiff = header.dup
tiff << short_tag.call( 256, 1, w ) # image width
tiff << short_tag.call( 257, 1, h ) # image height
tiff << long_tag.call( 258, 3, (header.size + (tag_count*12) + 4)) # bits per pixel
tiff << short_tag.call( 259, 1, 1 ) # compression
tiff << short_tag.call( 262, 1, 2 ) # colorspace - RGB
tiff << long_tag.call( 273, 1, (header.size + (tag_count*12) + 16) ) # data offset
tiff << short_tag.call( 277, 1, 3 ) # samples per pixel
tiff << long_tag.call( 279, 1, stream.unfiltered_data.size) # data byte size
tiff << [0].pack("I") # next IFD pointer
tiff << [bpc, bpc, bpc].pack("III")
tiff << stream.unfiltered_data
File.open(filename, "wb") { |file| file.write tiff }
end
end
class Jpg
attr_reader :stream
def initialize(stream)
@stream = stream
end
def save(filename)
w = stream.hash[:Width]
h = stream.hash[:Height]
puts "#{filename}: h=#{h}, w=#{w}"
File.open(filename, "wb") { |file| file.write stream.data }
end
end
class Tiff
attr_reader :stream
def initialize(stream)
@stream = stream
end
def save(filename)
if stream.hash[:DecodeParms][:K] <= 0
save_group_four(filename)
else
$stderr.puts "#{filename}: CCITT non-group 4/2D image."
end
end
private
# Group 4, 2D
def save_group_four(filename)
k = stream.hash[:DecodeParms][:K]
h = stream.hash[:Height]
w = stream.hash[:Width]
bpc = stream.hash[:BitsPerComponent]
mask = stream.hash[:ImageMask]
len = stream.hash[:Length]
cols = stream.hash[:DecodeParms][:Columns]
puts "#{filename}: h=#{h}, w=#{w}, bpc=#{bpc}, mask=#{mask}, len=#{len}, cols=#{cols}, k=#{k}"
# Synthesize a TIFF header
long_tag = lambda {|tag, value| [ tag, 4, 1, value ].pack( "ssII" ) }
short_tag = lambda {|tag, value| [ tag, 3, 1, value ].pack( "ssII" ) }
# header = byte order, version magic, offset of directory, directory count,
# followed by a series of tags containing metadata: 259 is a magic number for
# the compression type; 273 is the offset of the image data.
tiff = [ 73, 73, 42, 8, 5 ].pack("ccsIs") \
+ short_tag.call( 256, cols ) \
+ short_tag.call( 257, h ) \
+ short_tag.call( 259, 4 ) \
+ long_tag.call( 273, (10 + (5*12) + 4) ) \
+ long_tag.call( 279, len) \
+ [0].pack("I") \
+ stream.data
File.open(filename, "wb") { |file| file.write tiff }
end
end
end
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/adobe_sample.pdf"
extractor = ExtractImages::Extractor.new
PDF::Reader.open(filename) do |reader|
page = reader.page(1)
extractor.page(page)
end
@@ -0,0 +1,24 @@
#!/usr/bin/env ruby
# coding: utf-8
# Extract an (imperfect) array of paragraphs divided somewhat
# arbitrarily on line length.
require 'pdf/reader'
reader = PDF::Reader.new('somefile.pdf')
paragraph = ""
paragraphs = []
reader.pages.each do |page|
lines = page.text.scan(/^.+/)
lines.each do |line|
if line.length > 55
paragraph += " #{line}"
else
paragraph += " #{line}"
paragraphs << paragraph
paragraph = ""
end
end
end
@@ -0,0 +1,12 @@
#!/usr/bin/env ruby
# coding: utf-8
# get direct access to PDF objects
require 'pdf/reader'
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cairo-unicode.pdf"
reader = PDF::Reader.new(filename)
puts reader.objects[3]
puts reader.objects[4]
@@ -0,0 +1,14 @@
#!/usr/bin/env ruby
# coding: utf-8
# Extract metadata only
require 'rubygems'
require 'pdf/reader'
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cross_ref_stream.pdf"
PDF::Reader.open(filename) do |reader|
puts reader.info.inspect
puts reader.metadata.inspect
end
@@ -0,0 +1,13 @@
#!/usr/bin/env ruby
# coding: utf-8
# A simple app to count the number of pages in a PDF File.
require 'rubygems'
require 'pdf/reader'
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cross_ref_stream.pdf"
PDF::Reader.open(filename) do |reader|
puts "#{reader.page_count} page(s)"
end
@@ -0,0 +1,34 @@
#!/usr/bin/env ruby
# coding: utf-8
# typed: ignore
# Basic RSpec of a generated PDF
#
# USAGE: rspec -c examples/rspec.rb
require 'rubygems'
require 'pdf/reader'
require 'rspec'
require 'prawn'
require 'stringio'
describe "My generated PDF" do
it "should have the correct text on 2 pages" do
# generate our PDF
pdf = Prawn::Document.new
pdf.text "Chunky"
pdf.start_new_page
pdf.text "Bacon"
io = StringIO.new(pdf.render)
# process the PDF
PDF::Reader.open(io) do |reader|
reader.page_count.should eql(2) # correct page count
reader.page(1).text.should eql("Chunky") # correct content
reader.page(2).text.should eql("Bacon") # correct content
end
end
end
@@ -0,0 +1,15 @@
#!/usr/bin/env ruby
# coding: utf-8
# Extract all text from a single PDF
require 'rubygems'
require 'pdf/reader'
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cairo-unicode.pdf"
PDF::Reader.open(filename) do |reader|
reader.pages.each do |page|
puts page.text
end
end
@@ -0,0 +1,13 @@
#!/usr/bin/env ruby
# coding: utf-8
# Determine the PDF version of a file
require 'rubygems'
require 'pdf/reader'
filename = File.expand_path(File.dirname(__FILE__)) + "/../spec/data/cairo-basic.pdf"
PDF::Reader.open(filename) do |reader|
puts reader.pdf_version
end