Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion .github/workflows/Build.yml
Original file line number Diff line number Diff line change
Expand Up @@ -11,13 +11,14 @@ jobs:
test:
name: Julia ${{ matrix.version }} - ${{ matrix.os }} - ${{ matrix.arch }} - ${{ github.event_name }}
runs-on: ${{ matrix.os }}
continue-on-error: ${{ matrix.version == 'nightly' }}
strategy:
fail-fast: false
matrix:
version:
- '1.6'
- '1' # automatically expands to the latest stable 1.x release of Julia
- nightly
- 'nightly'
os:
- ubuntu-latest
arch:
Expand Down
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@ test/*.pdf
test/*.res
test/pvt
test/PDFTest*
test/pdftest*
file.txt
data/fonts/Arial.afm
test/sample01.pem
Expand Down
3 changes: 2 additions & 1 deletion Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@ julia = "1.6"
[extras]
Test = "8dfed614-e22c-5e08-85e1-65c5234f0b40"
ZipFile = "a5390f91-8eb1-5f08-bee0-b1d1ffed6cea"
Downloads = "f43a241f-c20a-4ad4-852c-f6b1247861c6"

[targets]
test = ["Test", "ZipFile"]
test = ["Test", "ZipFile", "Downloads"]
2 changes: 1 addition & 1 deletion src/CosDoc.jl
Original file line number Diff line number Diff line change
Expand Up @@ -391,7 +391,7 @@ function read_trailer(ps::IOStream, lookahead::Int)
end

function doc_trailer_update(ps::IOStream, doc::CosDocImpl)
TRAILER_REWIND = 50
TRAILER_REWIND = 256

seek(ps, doc.size-TRAILER_REWIND)

Expand Down
8 changes: 1 addition & 7 deletions src/Inflate.jl
Original file line number Diff line number Diff line change
Expand Up @@ -45,13 +45,7 @@ const Z_MEM_ERROR = -4
const Z_BUF_ERROR = -5
const Z_VERSION_ERROR = -6

@static if Base.VERSION > v"1.3-"
using Zlib_jll: libz
else
isfile(joinpath(dirname(@__FILE__),"..","deps","deps.jl")) ||
error("PDFIO not properly installed. Please run Pkg.build(\"PDFIO\")")
include("../deps/deps.jl")
end
using Zlib_jll: libz

_zlibVersion() = ccall((:zlibVersion, libz), Ptr{Cstring}, ())

Expand Down
8 changes: 1 addition & 7 deletions src/LibCrypto.jl
Original file line number Diff line number Diff line change
@@ -1,10 +1,4 @@
@static if Base.VERSION > v"1.3-"
using OpenSSL_jll: libcrypto
else
isfile(joinpath(dirname(@__FILE__), "..", "deps", "deps.jl")) ||
error("PDFIO not properly installed. Please run Pkg.build(\"PDFIO\")")
include("../deps/deps.jl")
end
using OpenSSL_jll: libcrypto

using Base: SecretBuffer, SecretBuffer!
import Base: copy
Expand Down
6 changes: 4 additions & 2 deletions src/PDFontTables.jl
Original file line number Diff line number Diff line change
Expand Up @@ -55,15 +55,17 @@ const GlyphName_to_ZAPEncoding = reverse_dict(ZAPEncoding_to_GlyphName)

using AdobeGlyphList

function agl_mapping_to_dict(m)
function agl_mapping_to_dict(m; fn=false)
dict = Dict{CosName, Char}()
map((@view m[:,1]), (@view m[:,2])) do x, y
v1, v2 = fn ? (2, 1) : (1, 2)
map((@view m[:,v1]), (@view m[:,v2])) do x, y
dict[CosName(strip(x))] = y
end
return dict
end

const AGL_Glyph_to_Unicode = agl_mapping_to_dict(agl())
const AGLFN_Glyph_to_Unicode = agl_mapping_to_dict(aglfn(), fn=true)
const AGL_ZAP_to_Unicode = agl_mapping_to_dict(zapfdingbats())
const AGL_Unicode_to_Glyph = reverse_dict(AGL_Glyph_to_Unicode)
const AGL_Unicode_to_ZAP = reverse_dict(AGL_ZAP_to_Unicode)
Expand Down
163 changes: 141 additions & 22 deletions src/PDFonts.jl
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,14 @@ const endbfrange = b"endbfrange"
const begincodespacerange = b"begincodespacerange"
const endcodespacerange = b"endcodespacerange"

# EXTRACTION_MODE determines how text extraction of textboxes is handled
# - `:spaces` (default)
# all white spaces are handled as a single space character
# - `:tabs`
# non-space white spaces are handled as tab characters
# - `:boxes`
# text is split into several textboxes with respective coordinates
const EXTRACTION_MODE = Ref(:spaces) # :spaces, :tabs, :boxes

mutable struct CMap
code_space::IntervalTree{UInt8,
Expand All @@ -45,7 +53,7 @@ function show(io::IO, cmap::CMap)
end


const FontUnicodeMapping = Union{Dict{UInt8, Char}, CMap, Nothing}
const FontUnicodeMapping = Union{Dict{UInt8, Vector{Char}}, CMap, Nothing}

#=
mutable struct FontUnicodeMapping
Expand All @@ -56,7 +64,56 @@ mutable struct FontUnicodeMapping
end
=#

function merge_encoding!(fum::Dict{UInt8, Char}, encoding::CosName,
function Base.merge!(fum::Dict{UInt8, Vector{Char}}, enc::Dict{UInt8, Char})
for (k, v) in enc
fum[k] = [v]
end
end

function get_agl_unicode(g::AbstractString)::Union{Vector{Char}, Char}
r = r"u(?'u'[[:xdigit:]]+$)|uni(?'uni'[[:xdigit:]]{4,6}$)"
m = match(r, g)
if m !== nothing
u, uni = m["u"], m["uni"]
if u !== nothing
l = length(u)
if l > 3 && mod(l, 4) == 0
ret = Char[]
for i = 1:4:l
c = parse(UInt16, SubString(u, i, i+3), base=16)
0xE000 > c > 0xD7FF && break
push!(ret, Char(c))
end
length(ret)*4 == l && return(ret)
end
else
c = parse(UInt32, uni, base=16)
0x0000 <= c <= 0xD7FF && 0xE000 <= c <= 0x10FFFF && return Char(c)
end
end
cg = CosName(g)
return get(AGL_Glyph_to_Unicode, cg, get(AGLFN_Glyph_to_Unicode, cg, zero(Char)))
end

function get_unicodes_from_glyph_name(s::String)
n = split(s, ".")
nf = n[1]
isempty(nf) && return [zero(Char)]
gs = split(nf, "_")
u = Char[]
for g in gs
append!(u, get_agl_unicode(g))
end
return u
end

function merge_agl!(fum::Dict{UInt8, Vector{Char}}, d::Dict{UInt8, CosName})
for (k, v) in d
fum[k] = get_unicodes_from_glyph_name(String(v))
end
end

function merge_encoding!(fum::Dict{UInt8, Vector{Char}}, encoding::CosName,
doc::CosDoc, font::IDDRef{CosDict})
encoding_mapping =
encoding == cn"WinAnsiEncoding" ? WINEncoding_to_Unicode :
Expand All @@ -82,10 +139,12 @@ function FontType(subtype::CosName)
return FontDefType()
end

# Entry point if someone wants to handle encoding based on subtype
# By default maps to the default font unicode mapping.
merge_encoding!(fum::FontUnicodeMapping, ftype::FontType,
doc::CosDoc, font::IDDRef{CosDict}) = fum

function merge_encoding!(fum::Dict{UInt8, Char},
function merge_encoding!(fum::Dict{UInt8, Vector{Char}},
ftype::Union{FontType1, FontMMType1},
doc::CosDoc, font::IDDRef{CosDict})
basefont = cosDocGetObject(doc, font, cn"BaseFont")
Expand All @@ -104,14 +163,14 @@ end
# Reading encoding from the font files in case of Symbolic fonts are not
# supported.
# Font subset is addressed with font name identification.
function merge_encoding!(fum::Dict{UInt8, Char}, encoding::CosNullType,
function merge_encoding!(fum::Dict{UInt8, Vector{Char}}, encoding::CosNullType,
doc::CosDoc, font::IDDRef{CosDict})
subtype = cosDocGetObject(doc, font, cn"Subtype")
subtype === CosNull && return fum
return merge_encoding!(fum, FontType(subtype), doc, font)
end

function merge_encoding!(fum::Dict{UInt8, Char},
function merge_encoding!(fum::Dict{UInt8, Vector{Char}},
encoding::IDD{CosDict},
doc::CosDoc, font::IDDRef{CosDict})
baseenc = cosDocGetObject(doc, encoding, cn"BaseEncoding")
Expand All @@ -133,8 +192,7 @@ function merge_encoding!(fum::Dict{UInt8, Char},
end
end

dict_to_unicode = dict_remap(d, AGL_Glyph_to_Unicode)
merge!(fum, dict_to_unicode)
merge_agl!(fum, d)
return fum
end

Expand All @@ -143,7 +201,7 @@ function get_unicode_mapping(doc::CosDoc, font::IDDRef{CosDict})
toUnicode !== CosNull &&
return get_unicode_mapping(toUnicode)
encoding = cosDocGetObject(doc, font, cn"Encoding")
d = merge_encoding!(Dict{UInt8, Char}(), encoding, doc, font)
d = merge_encoding!(Dict{UInt8, Vector{Char}}(), encoding, doc, font)
return length(d) == 0 ? nothing : d
end

Expand Down Expand Up @@ -218,11 +276,11 @@ function get_glyph_id_mapping(cosdoc::CosDoc, cosfont::IDD{CosDict})
return glyph_name_to_cid, cid_to_glyph_name
end

get_encoded_string(s::CosString, fum::Union{Dict{UInt8, Char}, CMap}) =
get_encoded_string(s::CosString, fum::FontUnicodeMapping) =
get_encoded_string(Vector{UInt8}(s), fum)

function get_encoded_string(v::Union{Vector{UInt8}, NTuple{N, UInt8}},
fum::Dict{UInt8, Char}) where N
fum::Dict{UInt8, Vector{Char}}) where N
length(v) == 0 && return ""
return String(NativeEncodingToUnicode(v, fum))
end
Expand Down Expand Up @@ -334,8 +392,17 @@ cmap_command(b::Vector{UInt8}) =
length(b), b != beginbfchar && b != beginbfrange && b != begincodespacerange ?
nothing : Symbol(String(b))

function _offset(obj::CosXString, offset)
da = Vector{UInt8}(obj)
db = UInt16(da[1]*256+da[2]+offset)
da[1], da[2] = UInt8(div(db, 256)), UInt8(mod(db, 256))
io = IOBuffer()
bytes2hex(io, da)
return CosXString(take!(io))
end

function on_cmap_command!(stm::IO, command::Symbol,
params::Vector{CosInt}, cmap::CMap)
params::Vector{CosInt}, cmap::CMap)
n = get(pop!(params))
o1, o2, o3 = CosNull, CosNull, CosNull
for i = 1:n
Expand All @@ -352,18 +419,57 @@ function on_cmap_command!(stm::IO, command::Symbol,
if l == 1
cmap.range_map[Interval(d1[1], d2[1])] = o3
else
imap = get!(cmap.range_map, Interval(d1[1], d2[1]),
IntervalTree{UInt8, CosObject}())
imap[Interval(d1[2], d2[2])] = o3
if d1[2] <= d2[2]
imap = get!(cmap.range_map, Interval(d1[1], d2[1]),
IntervalTree{UInt8, CosObject}())
imap[Interval(d1[2], d2[2])] = o3
else
@warn "Corrupt CMap file. Repairing... Some encodings may not map properly."
imap = get!(cmap.range_map, Interval(d1[1], d1[1]),
IntervalTree{UInt8, CosObject}())
imap[Interval(d1[2], 0xff)] = o3
o3 = _offset(o3, 0xff - d1[2] + 1)

if d2[1] - d1[1] > 1
i1, i2 = d1[1]+0x1, d2[1]-0x1
imap = get!(cmap.range_map, Interval(i1, i2),
IntervalTree{UInt8, CosObject}())
imap[Interval(0x00, 0xff)] = o3
o3 = _offset(o3, (d2[1] - d1[1] - 1)*0x100)
end
imap = get!(cmap.range_map, Interval(d2[1], d2[1]),
IntervalTree{UInt8, CosObject}())
imap[Interval(0x00, d2[2])] = o3
end
end
else
l = length(d1)
@assert (d1[1] <= d2[1]) E_INVALID_CODESPACERANGE
if l == 1
cmap.code_space[Interval(d1[1], d2[1])] = CosNull
else
imap = IntervalTree{UInt8, CosNullType}()
imap[Interval(d1[2], d2[2])] = CosNull
cmap.code_space[Interval(d1[1], d2[1])] = imap
if d1[2] <= d2[2]
imap = IntervalTree{UInt8, CosNullType}()
imap[Interval(d1[2], d2[2])] = CosNull
cmap.code_space[Interval(d1[1], d2[1])] = imap
else
@warn "Corrupt CMap file. Repairing... Some encodings may not map properly."
imap = IntervalTree{UInt8, CosNullType}()
imap[Interval(d1[2], 0xff)] = CosNull
cmap.code_space[Interval(d1[1], d1[1])] = imap

imap = get!(cmap.code_space, Interval(d1[1], d1[1]), IntervalTree{UInt8, CosNullType}())
imap[Interval(d1[2], 0xff)] = CosNull

imap = get!(cmap.code_space, Interval(d2[1], d2[1]), IntervalTree{UInt8, CosNullType}())
imap[Interval(0x00, d2[2])] = CosNull

if d2[1] - d1[1] > 1
i1, i2 = d1[1]+0x1, d2[1]-0x1
imap = get!(cmap.code_space, Interval(i1, i2), IntervalTree{UInt8, CosNullType}())
imap[Interval(0x00, 0xff)] = CosNull
end
end
end
end
end
Expand Down Expand Up @@ -571,13 +677,26 @@ function get_TextBox(ss::Vector{Union{CosXString, CosLiteralString,
totalw = 0f0
tj = 0f0
text = ""
offset = 0f0
params = Tuple{String, Float32, Float32, Float32}[]
for s in ss
if s isa CosXString || s isa CosLiteralString
prev_char = INIT_CODE(pdfont.widths)
t = String(get_encoded_string(s, pdfont))
if (-tj) > 180 && length(t) > 0 && t[1] != ' ' &&
length(text) > 0 && text[end] != ' '
text *= " "
if (-tj) > 180
if EXTRACTION_MODE[] == :spaces
if length(t) > 0 && t[1] != ' ' && length(text) > 0 && text[end] != ' '
text *= " "
end
elseif EXTRACTION_MODE[] == :tabs
text *= "\t"
elseif EXTRACTION_MODE[] == :boxes
push!(params, (text, totalw * th, tfs, offset * th))
offset += totalw - tj * tfs / 1000f0
text = ""
tj = 0f0
totalw = 0f0
end
end
text *= t
barr = Vector{UInt8}(s)
Expand All @@ -588,8 +707,8 @@ function get_TextBox(ss::Vector{Union{CosXString, CosLiteralString,
tj = s |> get |> Float32
end
end
totalw *= th
return text, totalw, tfs
push!(params, (text, totalw * th, tfs, offset * th))
return params
end

function get_character_width(cid::UInt16, w::CIDWidth)
Expand Down
25 changes: 16 additions & 9 deletions src/PDPageElement.jl
Original file line number Diff line number Diff line change
Expand Up @@ -693,20 +693,27 @@ end
fontname, font = eval_unicode_mapping(tr, state)

heap = get(state, :text_layout, Vector{TextLayout})
text, w, h = get_TextBox(tr.ss, font, tfs, tc, tw, th)
boxparams = get_TextBox(tr.ss, font, tfs, tc, tw, th)
h = boxparams[1][3]

d = get(state, :h_profile, Dict{Int, Int})
ih = round(Int, h*10)
d[ih] = get(d, ih, 0) + length(text)

tb = [0f0 0f0 1f0; w 0f0 1f0; w h 1f0; 0f0 h 1f0]*trm

if !get(state, :in_artifact, false)
tl = TextLayout(tb[1,1], tb[1,2], tb[2,1], tb[2,2],
tb[3,1], tb[3,2], tb[4,1], tb[4,2],
text, fontname, font.flags)
push!(heap, tl)
for params in boxparams
text = params[1]
w = params[2]
offset = params[4]
d[ih] = get(d, ih, 0) + length(text)
tb = [offset 0f0 1f0; offset + w 0f0 1f0; offset + w h 1f0; offset h 1f0]*trm
tl = TextLayout(tb[1,1], tb[1,2], tb[2,1], tb[2,2],
tb[3,1], tb[3,2], tb[4,1], tb[4,2],
text, fontname, font.flags)
push!(heap, tl)
end
end
offset_text_pos!(w, 0f0, state)
totalw = boxparams[end][4] + boxparams[end][2]
offset_text_pos!(totalw, 0f0, state)
return state
end

Expand Down
Loading