diff --git a/.badges/main/coverage.svg b/.badges/main/coverage.svg
new file mode 100644
index 0000000..5b27c36
--- /dev/null
+++ b/.badges/main/coverage.svg
@@ -0,0 +1 @@
+
\ No newline at end of file
diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml
deleted file mode 100644
index b65afa1..0000000
--- a/.github/workflows/test.yml
+++ /dev/null
@@ -1,39 +0,0 @@
-name: Test
-
-on:
- push:
- branches: [main]
- pull_request:
- workflow_dispatch:
-
-permissions:
- contents: write # push the coverage badge to the `badges` branch (main only)
-
-concurrency:
- group: ${{ github.workflow }}-${{ github.ref }}
- cancel-in-progress: ${{ github.event_name == 'pull_request' }}
-
-jobs:
- test:
- name: Test
- runs-on: ubuntu-latest
- steps:
- - name: Checkout
- uses: actions/checkout@v6
-
- - name: Set up Go
- uses: actions/setup-go@v6
- with:
- go-version-file: go.mod
-
- - name: Run tests
- run: go test -race -coverprofile=cover.out ./...
-
- # Root (Docker) action, not /action/source: the source variant compiles the
- # tool, which needs a newer Go than go.mod pins — and fails the build.
- - name: Coverage badge
- uses: vladopajic/go-test-coverage@v2
- with:
- profile: cover.out
- git-token: ${{ github.ref_name == 'main' && secrets.GITHUB_TOKEN || '' }}
- git-branch: badges
diff --git a/.gitignore b/.gitignore
deleted file mode 100644
index e4b2eef..0000000
--- a/.gitignore
+++ /dev/null
@@ -1,5 +0,0 @@
-*.test
-*.out
-coverage.txt
-.idea/
-.vscode/
diff --git a/LICENSE b/LICENSE
deleted file mode 100644
index 1249904..0000000
--- a/LICENSE
+++ /dev/null
@@ -1,21 +0,0 @@
-MIT License
-
-Copyright (c) 2026 Patrick Gundlach
-
-Permission is hereby granted, free of charge, to any person obtaining a copy
-of this software and associated documentation files (the "Software"), to deal
-in the Software without restriction, including without limitation the rights
-to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
-copies of the Software, and to permit persons to whom the Software is
-furnished to do so, subject to the following conditions:
-
-The above copyright notice and this permission notice shall be included in
-all copies or substantial portions of the Software.
-
-THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
-IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
-FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
-AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
-LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
-OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
-THE SOFTWARE.
diff --git a/README.md b/README.md
deleted file mode 100644
index f36baa3..0000000
--- a/README.md
+++ /dev/null
@@ -1,124 +0,0 @@
-# pdfdisassembler
-
-[](https://github.com/speedata/pdfdisassembler/actions/workflows/test.yml)
-[](https://github.com/speedata/pdfdisassembler/actions/workflows/test.yml)
-[](https://pkg.go.dev/github.com/speedata/pdfdisassembler)
-
-
-A focused, read-only PDF parser for Go. Built for tooling that **inspects**
-PDFs — accessibility checkers, validators, debuggers — without dragging in
-the writing, optimisation, signing and image-rendering machinery that
-general-purpose PDF libraries carry.
-
-Full API documentation:
-
-## Status
-
-Pre-1.0. The API may break between minor releases.
-
-## Why
-
-The Go PDF ecosystem has a real gap for read-only structural inspection.
-Existing libraries are either too large (pdfcpu: ~50 kLOC, multi-MB WASM
-overhead), licensed restrictively (unipdf: AGPL/commercial), CGo (go-fitz),
-or too thin (rsc/pdf, ledongthuc/pdf). pdfdisassembler targets PDF 1.x and
-2.0 reading in pure Go, WASM-friendly by construction. The only external
-dependency is `github.com/andybalholm/brotli` (pure Go, MIT), which backs
-the BrotliDecode filter; only its decoder side gets linked into consumers.
-
-## Scope
-
-In scope: PDF 1.x and 2.0 reading, classical xref and xref streams,
-indirect-object resolution, stream filters (FlateDecode, ASCII85, ASCIIHex,
-LZW, RunLength, and BrotliDecode — a PDF Association extension to PDF 2.0,
-pending ISO 32000 inclusion), text-string decoding (PDFDocEncoding,
-UTF-16BE BOM, UTF-8 BOM), catalog + page-tree navigation (page boxes,
-rotation, resources and
-content streams, with inherited attributes resolved along the `/Parent`
-chain), DocumentInfo, XMP metadata access, structure tree traversal,
-`/Standard` security handler (V2, V4, V5), defensive parsing.
-
-Out of scope: writing PDFs, image filters (DCTDecode/JBIG2/JPX/CCITTFax),
-image rendering, font internals, XFA, public-key encryption, signature
-verification, content-stream graphics-state interpretation, LTV.
-
-## Usage
-
-Open a file and read top-level metadata:
-
-```go
-import "github.com/speedata/pdfdisassembler"
-
-r, err := pdfdisassembler.OpenFile("doc.pdf")
-if err != nil {
- return err
-}
-defer r.Close()
-
-fmt.Println("PDF version:", r.Version())
-info := r.DocumentInfo()
-fmt.Println("Title:", info.Title)
-```
-
-Walk every live indirect object and decode any streams that carry one of
-the supported filters:
-
-```go
-r, err := pdfdisassembler.OpenFile("doc.pdf")
-if err != nil {
- return err
-}
-defer r.Close()
-
-for entry := range r.Objects() {
- s, ok := entry.Object.(*pdfdisassembler.Stream)
- if !ok {
- continue
- }
- ref := entry.Reference
- data, err := r.DecodeStream(ref)
- if err != nil {
- fmt.Printf("%d %d R: %v\n", ref.Number, ref.Generation, err)
- continue
- }
- fmt.Printf("%d %d R: %d bytes raw, %d bytes decoded\n",
- ref.Number, ref.Generation, s.RawLength(), len(data))
-}
-```
-
-More complete examples live under [`examples/`](examples): `inspect`
-prints a summary of a PDF, `structtree` walks the `/StructTreeRoot` as a
-starting point for accessibility tooling, and `pageinfo` reports per-page
-boxes, rotation, resources and content size via the page-tree API.
-
-## Testing
-
-Snapshot tests live under `testdata/fixtures//`. Each fixture has an
-`input.pdf` and a committed `golden.json`. `TestFixtures` opens every
-fixture, runs `Dump`, and compares against the golden — a byte-stable
-JSON snapshot of the object graph.
-
-Adding a fixture:
-
-1. Drop `input.pdf` into `testdata/fixtures//`
-2. `go test -update -run TestFixtures/` — generates `golden.json`
-3. **Inspect the golden manually**: does it match what the PDF spec says
- should happen? The golden is *the expected behaviour*, not "what the
- parser currently does"
-4. Commit the PDF, the golden, and an optional `README.md` explaining
- what the fixture proves
-
-For synthetic fixtures, see `testdata/fixtures/generate.go`. Run it from
-the repo root to (re)create the in-code fixture PDFs.
-
-The same dump format is exposed as a CLI:
-
-```
-go install github.com/speedata/pdfdisassembler/cmd/pdfdump@latest
-pdfdump doc.pdf > doc.json
-diff <(pdfdump a.pdf) <(pdfdump b.pdf)
-```
-
-## License
-
-MIT. See [LICENSE](LICENSE).
diff --git a/cmd/pdfdump/main.go b/cmd/pdfdump/main.go
deleted file mode 100644
index b9e9152..0000000
--- a/cmd/pdfdump/main.go
+++ /dev/null
@@ -1,71 +0,0 @@
-// Command pdfdump emits a JSON snapshot of a PDF in the same format used
-// by pdfdisassembler's snapshot-test harness.
-//
-// Usage:
-//
-// pdfdump [-stream-content] [-no-preview]
-//
-// Diff two PDFs structurally:
-//
-// diff <(pdfdump a.pdf) <(pdfdump b.pdf)
-//
-// Reproduce a parser bug:
-//
-// pdfdump broken.pdf > broken.json
-// # attach broken.pdf and broken.json to the issue
-package main
-
-import (
- "errors"
- "flag"
- "fmt"
- "io"
- "log"
- "os"
-
- "github.com/speedata/pdfdisassembler"
-)
-
-func main() {
- if err := run(os.Args[1:], os.Stdout); err != nil {
- if errors.Is(err, flag.ErrHelp) {
- return
- }
- log.Fatal(err)
- }
-}
-
-func run(args []string, stdout io.Writer) error {
- fs := flag.NewFlagSet("pdfdump", flag.ContinueOnError)
- inlineContent := fs.Bool("stream-content", false, "embed decoded stream bytes as hex under decoded.hex")
- noPreview := fs.Bool("no-preview", false, "omit the preview_utf8 field on streams")
- fs.Usage = func() {
- fmt.Fprintln(os.Stderr, "usage: pdfdump [flags] ")
- fs.PrintDefaults()
- }
- if err := fs.Parse(args); err != nil {
- return err
- }
- if fs.NArg() != 1 {
- fs.Usage()
- return errors.New("pdfdump: exactly one input file is required")
- }
- r, err := pdfdisassembler.OpenFile(fs.Arg(0))
- if err != nil {
- return err
- }
- defer r.Close()
-
- opts := pdfdisassembler.DumpOptions{
- InlineStreamContent: *inlineContent,
- }
- if *noPreview {
- opts.PreviewMaxBytes = -1
- }
- data, err := pdfdisassembler.Dump(r, opts)
- if err != nil {
- return err
- }
- _, err = stdout.Write(data)
- return err
-}
diff --git a/cmd/pdfdump/main_test.go b/cmd/pdfdump/main_test.go
deleted file mode 100644
index c172573..0000000
--- a/cmd/pdfdump/main_test.go
+++ /dev/null
@@ -1,64 +0,0 @@
-package main
-
-import (
- "bytes"
- "io"
- "os"
- "path/filepath"
- "testing"
-)
-
-// pdfdump is run on untrusted files, so hostile input must yield a graceful
-// error or valid JSON — never a panic (which the test framework would surface).
-func TestRunHostileInputsNoPanic(t *testing.T) {
- cases := map[string][]byte{
- "empty": {},
- "garbage": []byte("not a pdf at all"),
- "header only": []byte("%PDF-1.7\n"),
- "broken xref": []byte("%PDF-1.7\n1 0 obj\n<< /Type /Catalog >>\nendobj\nstartxref\n999999\n%%EOF"),
- "nul bytes": bytes.Repeat([]byte{0}, 256),
- "unclosed dict": []byte("%PDF-1.7\n1 0 obj\n<< /Type /Catalog\nendobj\ntrailer\n<< /Root 1 0 R >>\nstartxref\n9\n%%EOF"),
- }
- dir := t.TempDir()
- for name, body := range cases {
- t.Run(name, func(t *testing.T) {
- p := filepath.Join(dir, "in.pdf")
- if err := os.WriteFile(p, body, 0o600); err != nil {
- t.Fatal(err)
- }
- var out bytes.Buffer
- if err := run([]string{p}, &out); err == nil {
- if out.Len() == 0 || out.Bytes()[0] != '{' {
- t.Fatalf("succeeded but produced non-JSON output (%d bytes)", out.Len())
- }
- }
- })
- }
-}
-
-func TestRunValidFixtures(t *testing.T) {
- fixtures, err := filepath.Glob("../../testdata/fixtures/*/input.pdf")
- if err != nil || len(fixtures) == 0 {
- t.Fatalf("no fixtures found: %v", err)
- }
- for _, fx := range fixtures {
- t.Run(filepath.Base(filepath.Dir(fx)), func(t *testing.T) {
- var out bytes.Buffer
- if err := run([]string{"-stream-content", fx}, &out); err != nil {
- t.Fatalf("run: %v", err)
- }
- if out.Len() == 0 || out.Bytes()[0] != '{' {
- t.Fatalf("expected JSON output, got %d bytes", out.Len())
- }
- })
- }
-}
-
-func TestRunArgErrors(t *testing.T) {
- if err := run(nil, io.Discard); err == nil {
- t.Error("expected an error with no file argument")
- }
- if err := run([]string{"/nonexistent/nope.pdf"}, io.Discard); err == nil {
- t.Error("expected an error for a missing file")
- }
-}
diff --git a/contentstream/doc.go b/contentstream/doc.go
deleted file mode 100644
index 0ca5f0f..0000000
--- a/contentstream/doc.go
+++ /dev/null
@@ -1,36 +0,0 @@
-// Package contentstream tokenises PDF content streams into a sequence
-// of operations. Content streams are the postfix-notation graphics
-// instructions that paint each page (text-showing operators, path
-// operators, graphics-state ops, marked-content tags, …).
-//
-// The scanner is operand-aware: operands are collected up to each
-// operator keyword and surfaced together as one Op. Inline images
-// (BI/ID/EI) are folded into a single synthetic EI op so the binary
-// image bytes between ID and EI do not derail tokenisation.
-//
-// The scanner does NOT interpret the operations: it does not track
-// graphics state, does not render glyphs, does not resolve XObjects.
-// Higher-level consumers (e.g. tagged-PDF validators) layer that logic
-// on top.
-//
-// # Usage
-//
-// for op, err := range contentstream.New(decoded).All() {
-// if err != nil { ... }
-// switch op.Operator {
-// case "Tf":
-// // op.Operands[0].Name is the font resource key
-// case "BDC":
-// // op.Operands[0].Name is the structure tag
-// // op.Operands[1] is either a Name (ref into /Properties)
-// // or a Dict (inline properties)
-// }
-// }
-//
-// # Scope
-//
-// The scanner accepts the subset of PDF object syntax that can appear
-// in content streams: numbers, names, strings (literal and hex),
-// arrays, dictionaries, and operator keywords. Indirect references and
-// stream objects do not occur in content streams and are not handled.
-package contentstream
diff --git a/contentstream/operand.go b/contentstream/operand.go
deleted file mode 100644
index 88d6a49..0000000
--- a/contentstream/operand.go
+++ /dev/null
@@ -1,81 +0,0 @@
-package contentstream
-
-import "strconv"
-
-// Kind identifies the type of an operand value.
-type Kind int
-
-const (
- // KindUnknown is the zero value; not produced by the scanner.
- KindUnknown Kind = iota
- // KindNumber covers both PDF integers and reals. Use Operand.Int()
- // to recover an int64 when the producer wrote an integer literal.
- KindNumber
- // KindName is a PDF name without the leading slash.
- KindName
- // KindString is a literal or hex string. The raw decoded bytes are
- // in Operand.Bytes; the scanner does not apply text-string decoding
- // (UTF-16BE BOM, PDFDocEncoding, …) because content-stream strings
- // are text shown to the reader and their semantic encoding depends
- // on the active font, not on the PDF text-string convention.
- KindString
- // KindArray holds operands of a PDF array, in source order. The
- // most common occurrence is the operand of TJ: a mix of strings
- // and number adjustments.
- KindArray
- // KindDict holds the entries of an inline dictionary. The most
- // common occurrence is the property dictionary that follows BDC.
- KindDict
- // KindBool is rare in content streams but appears in BDC property
- // dictionaries occasionally.
- KindBool
- // KindNull is rare in content streams but appears in BDC property
- // dictionaries occasionally.
- KindNull
-)
-
-// Operand is a single value pushed onto the operand stack before an
-// operator keyword. It is a tagged union: which field is meaningful
-// depends on Kind.
-type Operand struct {
- Kind Kind
- // Number carries the parsed numeric value when Kind == KindNumber.
- // numStr preserves the original literal so Int() can decide whether
- // the producer wrote an integer.
- Number float64
- numStr string
- // Name is the name body (no leading slash) when Kind == KindName.
- Name string
- // Bytes is the decoded string payload when Kind == KindString.
- Bytes []byte
- // Array is the element list when Kind == KindArray.
- Array []Operand
- // Dict is the entry map when Kind == KindDict. Iteration order is
- // not preserved; use Dict.Keys / parse separately if order matters.
- Dict Dict
- // Bool is the boolean value when Kind == KindBool.
- Bool bool
-}
-
-// Int reports the operand as an int64 if the producer wrote an integer
-// literal (no decimal point, no exponent). The ok flag is false for
-// real-number literals and for non-number operands.
-func (o Operand) Int() (int64, bool) {
- if o.Kind != KindNumber {
- return 0, false
- }
- for _, c := range o.numStr {
- if c == '.' || c == 'e' || c == 'E' {
- return 0, false
- }
- }
- v, err := strconv.ParseInt(o.numStr, 10, 64)
- if err != nil {
- return 0, false
- }
- return v, true
-}
-
-// Dict is a small key→Operand map for inline content-stream
-// dictionaries. Nested dictionaries are supported.
-type Dict map[string]Operand
diff --git a/contentstream/scanner.go b/contentstream/scanner.go
deleted file mode 100644
index bb2cc9a..0000000
--- a/contentstream/scanner.go
+++ /dev/null
@@ -1,366 +0,0 @@
-package contentstream
-
-import (
- "errors"
- "fmt"
- "io"
- "iter"
- "strconv"
-
- "github.com/speedata/pdfdisassembler/internal/lex"
-)
-
-// Op is one content-stream operation: zero or more operands followed
-// by an operator keyword (e.g. "Tf", "Tj", "BDC").
-//
-// For inline-image runs, Operator is "EI" and Image carries the raw
-// bytes between ID and EI; the BI dictionary is in Operands[0] as a
-// KindDict (or empty if BI carried no entries).
-type Op struct {
- Operator string
- Operands []Operand
- Image []byte
- // Offset is the byte position of the operator keyword in the
- // source slice. Useful for error messages and source ranges.
- Offset int64
-}
-
-// Scanner walks a decoded content stream and yields one Op at a time.
-// It is not safe for concurrent use.
-type Scanner struct {
- lx *lex.Lexer
- stack []Operand
- depth int
- done bool
-}
-
-// New returns a Scanner over the decoded content-stream bytes src.
-// src is not copied. For pages whose /Contents is an array of streams,
-// concatenate the decoded payloads with a single whitespace byte (per
-// PDF 32000-1:2008 §7.8.2) before passing them in.
-func New(src []byte) *Scanner {
- return &Scanner{lx: lex.New(src)}
-}
-
-// ErrUnexpectedEOF indicates that the scanner ran out of bytes mid-
-// operation (e.g. inside a dictionary, or while looking for EI).
-var ErrUnexpectedEOF = errors.New("pdfdisassembler/contentstream: unexpected EOF")
-
-// maxNestDepth bounds array/dict nesting so a hostile content stream can't
-// recurse the scanner into a stack overflow.
-const maxNestDepth = 1000
-
-// maxOperands caps operands accumulated per operation (and per array/dict) so
-// a flood of operands can't pin large amounts of memory.
-const maxOperands = 100000
-
-// Next returns the next operation. At end of stream it returns io.EOF.
-// Any other error indicates malformed input; the scanner is not safe
-// to keep using after an error.
-func (s *Scanner) Next() (Op, error) {
- if s.done {
- return Op{}, io.EOF
- }
- for {
- if len(s.stack) > maxOperands {
- return Op{}, fmt.Errorf("pdfdisassembler/contentstream: too many operands (> %d)", maxOperands)
- }
- tok, err := s.nextToken()
- if err != nil {
- return Op{}, err
- }
- switch tok.Kind {
- case lex.EOF:
- s.done = true
- if len(s.stack) != 0 {
- // Trailing operands without an operator — common in
- // the wild. Drop them silently.
- s.stack = s.stack[:0]
- }
- return Op{}, io.EOF
- case lex.Integer, lex.Real:
- n, _ := strconv.ParseFloat(string(tok.Bytes), 64)
- s.stack = append(s.stack, Operand{
- Kind: KindNumber,
- Number: n,
- numStr: string(tok.Bytes),
- })
- case lex.Name:
- s.stack = append(s.stack, Operand{Kind: KindName, Name: string(tok.Bytes)})
- case lex.LitString, lex.HexString:
- s.stack = append(s.stack, Operand{Kind: KindString, Bytes: append([]byte(nil), tok.Bytes...)})
- case lex.ArrayStart:
- arr, err := s.readArray()
- if err != nil {
- return Op{}, err
- }
- s.stack = append(s.stack, Operand{Kind: KindArray, Array: arr})
- case lex.DictStart:
- d, err := s.readDict()
- if err != nil {
- return Op{}, err
- }
- s.stack = append(s.stack, Operand{Kind: KindDict, Dict: d})
- case lex.Keyword:
- kw := string(tok.Bytes)
- switch kw {
- case "true":
- s.stack = append(s.stack, Operand{Kind: KindBool, Bool: true})
- continue
- case "false":
- s.stack = append(s.stack, Operand{Kind: KindBool, Bool: false})
- continue
- case "null":
- s.stack = append(s.stack, Operand{Kind: KindNull})
- continue
- case "BI":
- img, err := s.readInlineImage()
- if err != nil {
- return Op{}, err
- }
- op := Op{Operator: "EI", Operands: s.takeStack(), Image: img, Offset: tok.Offset}
- return op, nil
- }
- op := Op{Operator: kw, Operands: s.takeStack(), Offset: tok.Offset}
- return op, nil
- default:
- return Op{}, fmt.Errorf("pdfdisassembler/contentstream: unexpected token %s at %d", tok.Kind, tok.Offset)
- }
- }
-}
-
-// All returns a range-over-func iterator that yields each Op until EOF
-// or the first error. The error is delivered through the second loop
-// variable as on the final iteration.
-func (s *Scanner) All() iter.Seq2[Op, error] {
- return func(yield func(Op, error) bool) {
- for {
- op, err := s.Next()
- if err == io.EOF {
- return
- }
- if !yield(op, err) {
- return
- }
- if err != nil {
- return
- }
- }
- }
-}
-
-func (s *Scanner) takeStack() []Operand {
- out := s.stack
- s.stack = nil
- return out
-}
-
-func (s *Scanner) nextToken() (lex.Token, error) {
- return s.lx.Next()
-}
-
-func (s *Scanner) readArray() ([]Operand, error) {
- s.depth++
- defer func() { s.depth-- }()
- if s.depth > maxNestDepth {
- return nil, fmt.Errorf("pdfdisassembler/contentstream: nesting too deep (> %d)", maxNestDepth)
- }
- var out []Operand
- for {
- if len(out) > maxOperands {
- return nil, fmt.Errorf("pdfdisassembler/contentstream: array too large (> %d)", maxOperands)
- }
- tok, err := s.nextToken()
- if err != nil {
- return nil, err
- }
- switch tok.Kind {
- case lex.ArrayEnd:
- return out, nil
- case lex.EOF:
- return nil, ErrUnexpectedEOF
- case lex.Integer, lex.Real:
- n, _ := strconv.ParseFloat(string(tok.Bytes), 64)
- out = append(out, Operand{Kind: KindNumber, Number: n, numStr: string(tok.Bytes)})
- case lex.Name:
- out = append(out, Operand{Kind: KindName, Name: string(tok.Bytes)})
- case lex.LitString, lex.HexString:
- out = append(out, Operand{Kind: KindString, Bytes: append([]byte(nil), tok.Bytes...)})
- case lex.ArrayStart:
- nested, err := s.readArray()
- if err != nil {
- return nil, err
- }
- out = append(out, Operand{Kind: KindArray, Array: nested})
- case lex.DictStart:
- d, err := s.readDict()
- if err != nil {
- return nil, err
- }
- out = append(out, Operand{Kind: KindDict, Dict: d})
- case lex.Keyword:
- switch string(tok.Bytes) {
- case "true":
- out = append(out, Operand{Kind: KindBool, Bool: true})
- case "false":
- out = append(out, Operand{Kind: KindBool, Bool: false})
- case "null":
- out = append(out, Operand{Kind: KindNull})
- default:
- return nil, fmt.Errorf("pdfdisassembler/contentstream: unexpected keyword %q inside array at %d", tok.Bytes, tok.Offset)
- }
- default:
- return nil, fmt.Errorf("pdfdisassembler/contentstream: unexpected token %s inside array at %d", tok.Kind, tok.Offset)
- }
- }
-}
-
-func (s *Scanner) readDict() (Dict, error) {
- s.depth++
- defer func() { s.depth-- }()
- if s.depth > maxNestDepth {
- return nil, fmt.Errorf("pdfdisassembler/contentstream: nesting too deep (> %d)", maxNestDepth)
- }
- out := Dict{}
- for {
- if len(out) > maxOperands {
- return nil, fmt.Errorf("pdfdisassembler/contentstream: dict too large (> %d)", maxOperands)
- }
- tok, err := s.nextToken()
- if err != nil {
- return nil, err
- }
- if tok.Kind == lex.DictEnd {
- return out, nil
- }
- if tok.Kind == lex.EOF {
- return nil, ErrUnexpectedEOF
- }
- if tok.Kind != lex.Name {
- return nil, fmt.Errorf("pdfdisassembler/contentstream: expected name as dict key at %d, got %s", tok.Offset, tok.Kind)
- }
- key := string(tok.Bytes)
- val, err := s.readValue()
- if err != nil {
- return nil, err
- }
- out[key] = val
- }
-}
-
-func (s *Scanner) readValue() (Operand, error) {
- tok, err := s.nextToken()
- if err != nil {
- return Operand{}, err
- }
- switch tok.Kind {
- case lex.Integer, lex.Real:
- n, _ := strconv.ParseFloat(string(tok.Bytes), 64)
- return Operand{Kind: KindNumber, Number: n, numStr: string(tok.Bytes)}, nil
- case lex.Name:
- return Operand{Kind: KindName, Name: string(tok.Bytes)}, nil
- case lex.LitString, lex.HexString:
- return Operand{Kind: KindString, Bytes: append([]byte(nil), tok.Bytes...)}, nil
- case lex.ArrayStart:
- arr, err := s.readArray()
- if err != nil {
- return Operand{}, err
- }
- return Operand{Kind: KindArray, Array: arr}, nil
- case lex.DictStart:
- d, err := s.readDict()
- if err != nil {
- return Operand{}, err
- }
- return Operand{Kind: KindDict, Dict: d}, nil
- case lex.Keyword:
- switch string(tok.Bytes) {
- case "true":
- return Operand{Kind: KindBool, Bool: true}, nil
- case "false":
- return Operand{Kind: KindBool, Bool: false}, nil
- case "null":
- return Operand{Kind: KindNull}, nil
- }
- return Operand{}, fmt.Errorf("pdfdisassembler/contentstream: unexpected keyword %q where value expected at %d", tok.Bytes, tok.Offset)
- case lex.EOF:
- return Operand{}, ErrUnexpectedEOF
- default:
- return Operand{}, fmt.Errorf("pdfdisassembler/contentstream: unexpected token %s where value expected at %d", tok.Kind, tok.Offset)
- }
-}
-
-// readInlineImage handles BI…ID…EI. The BI token has already been
-// consumed. We read a dictionary body until we see the "ID" keyword,
-// then scan the raw source for the "EI" terminator and return the
-// bytes in between as the image payload.
-func (s *Scanner) readInlineImage() ([]byte, error) {
- // BI has no '<<' — entries follow directly until ID.
- dict := Dict{}
- for {
- tok, err := s.nextToken()
- if err != nil {
- return nil, err
- }
- if tok.Kind == lex.Keyword && string(tok.Bytes) == "ID" {
- break
- }
- if tok.Kind == lex.EOF {
- return nil, ErrUnexpectedEOF
- }
- if tok.Kind != lex.Name {
- return nil, fmt.Errorf("pdfdisassembler/contentstream: expected name inside BI block at %d, got %s", tok.Offset, tok.Kind)
- }
- key := string(tok.Bytes)
- val, err := s.readValue()
- if err != nil {
- return nil, err
- }
- dict[key] = val
- }
- // We stashed the dict in stack so it surfaces as Operand of EI.
- s.stack = append(s.stack, Operand{Kind: KindDict, Dict: dict})
-
- // Per PDF 32000-1:2008 §8.9.7, ID is followed by exactly one
- // whitespace byte, then the raw image data, then EI preceded by
- // whitespace. We approximate "preceded by whitespace" rather than
- // strictly enforcing exactly-one — producers vary.
- src := s.lx.Source()
- pos := s.lx.Pos()
- if pos < len(src) && (src[pos] == ' ' || src[pos] == '\t' || src[pos] == '\n' || src[pos] == '\r') {
- pos++
- }
- imgStart := pos
- // Scan for "EI" preceded by whitespace and followed by whitespace/EOF.
- for pos < len(src) {
- // Look for 'E' first.
- if src[pos] != 'E' {
- pos++
- continue
- }
- if pos+1 >= len(src) || src[pos+1] != 'I' {
- pos++
- continue
- }
- // Check leading boundary.
- if pos == 0 {
- pos++
- continue
- }
- if !lex.IsWhitespace(src[pos-1]) {
- pos++
- continue
- }
- // Check trailing boundary.
- if pos+2 == len(src) || lex.IsWhitespace(src[pos+2]) || lex.IsDelimiter(src[pos+2]) {
- imgEnd := pos - 1 // strip the whitespace separator
- if imgEnd < imgStart {
- imgEnd = imgStart // empty image: no data between ID and EI
- }
- s.lx.SetPos(pos + 2)
- return append([]byte(nil), src[imgStart:imgEnd]...), nil
- }
- pos++
- }
- return nil, ErrUnexpectedEOF
-}
diff --git a/contentstream/scanner_test.go b/contentstream/scanner_test.go
deleted file mode 100644
index a598588..0000000
--- a/contentstream/scanner_test.go
+++ /dev/null
@@ -1,466 +0,0 @@
-package contentstream_test
-
-import (
- "errors"
- "io"
- "reflect"
- "strings"
- "testing"
-
- "github.com/speedata/pdfdisassembler/contentstream"
-)
-
-func collect(t *testing.T, src string) []contentstream.Op {
- t.Helper()
- var out []contentstream.Op
- sc := contentstream.New([]byte(src))
- for {
- op, err := sc.Next()
- if errors.Is(err, io.EOF) {
- return out
- }
- if err != nil {
- t.Fatalf("scan error: %v", err)
- }
- out = append(out, op)
- }
-}
-
-func TestEmpty(t *testing.T) {
- if got := collect(t, ""); len(got) != 0 {
- t.Fatalf("want 0 ops, got %d", len(got))
- }
- if got := collect(t, " \n\t "); len(got) != 0 {
- t.Fatalf("want 0 ops on whitespace-only, got %d", len(got))
- }
-}
-
-func TestSimpleOperators(t *testing.T) {
- src := "q 1 0 0 1 100 200 cm Q"
- ops := collect(t, src)
- if len(ops) != 3 {
- t.Fatalf("want 3 ops, got %d (%+v)", len(ops), ops)
- }
- if ops[0].Operator != "q" || len(ops[0].Operands) != 0 {
- t.Errorf("op[0] = %+v, want q with no operands", ops[0])
- }
- if ops[1].Operator != "cm" || len(ops[1].Operands) != 6 {
- t.Errorf("op[1] = %+v, want cm with 6 operands", ops[1])
- }
- if ops[2].Operator != "Q" {
- t.Errorf("op[2].Operator = %q, want Q", ops[2].Operator)
- }
-}
-
-func TestTfOperator(t *testing.T) {
- src := "BT /F1 12 Tf (Hello) Tj ET"
- ops := collect(t, src)
- if len(ops) != 4 {
- t.Fatalf("want 4 ops, got %d", len(ops))
- }
- if ops[0].Operator != "BT" || ops[3].Operator != "ET" {
- t.Errorf("BT/ET framing missing: %+v", ops)
- }
- if ops[1].Operator != "Tf" {
- t.Fatalf("op[1].Operator = %q, want Tf", ops[1].Operator)
- }
- if ops[1].Operands[0].Kind != contentstream.KindName || ops[1].Operands[0].Name != "F1" {
- t.Errorf("Tf font operand = %+v, want name F1", ops[1].Operands[0])
- }
- if ops[1].Operands[1].Kind != contentstream.KindNumber || ops[1].Operands[1].Number != 12 {
- t.Errorf("Tf size operand = %+v, want number 12", ops[1].Operands[1])
- }
- if ops[2].Operator != "Tj" {
- t.Fatalf("op[2].Operator = %q, want Tj", ops[2].Operator)
- }
- if string(ops[2].Operands[0].Bytes) != "Hello" {
- t.Errorf("Tj string = %q, want Hello", ops[2].Operands[0].Bytes)
- }
-}
-
-func TestTJArray(t *testing.T) {
- src := "[(He) -10 (l) -5 (lo)] TJ"
- ops := collect(t, src)
- if len(ops) != 1 {
- t.Fatalf("want 1 op, got %d", len(ops))
- }
- if ops[0].Operator != "TJ" {
- t.Fatalf("op.Operator = %q, want TJ", ops[0].Operator)
- }
- arr := ops[0].Operands[0]
- if arr.Kind != contentstream.KindArray {
- t.Fatalf("TJ operand kind = %v, want Array", arr.Kind)
- }
- if len(arr.Array) != 5 {
- t.Fatalf("TJ array len = %d, want 5", len(arr.Array))
- }
- if string(arr.Array[0].Bytes) != "He" || arr.Array[1].Number != -10 {
- t.Errorf("TJ contents off: %+v", arr.Array)
- }
-}
-
-func TestBDCInlineDict(t *testing.T) {
- src := "/Span << /MCID 7 /Lang (en-US) >> BDC (text) Tj EMC"
- ops := collect(t, src)
- if len(ops) != 3 {
- t.Fatalf("want 3 ops, got %d", len(ops))
- }
- if ops[0].Operator != "BDC" {
- t.Fatalf("op[0].Operator = %q, want BDC", ops[0].Operator)
- }
- if ops[0].Operands[0].Kind != contentstream.KindName || ops[0].Operands[0].Name != "Span" {
- t.Errorf("BDC tag = %+v, want name Span", ops[0].Operands[0])
- }
- props := ops[0].Operands[1]
- if props.Kind != contentstream.KindDict {
- t.Fatalf("BDC props kind = %v, want Dict", props.Kind)
- }
- mcid, ok := props.Dict["MCID"]
- if !ok {
- t.Fatalf("MCID missing from %+v", props.Dict)
- }
- if n, ok := mcid.Int(); !ok || n != 7 {
- t.Errorf("MCID = %v (intOk=%v), want 7", n, ok)
- }
- if ops[2].Operator != "EMC" {
- t.Errorf("op[2].Operator = %q, want EMC", ops[2].Operator)
- }
-}
-
-func TestBDCPropertyNameRef(t *testing.T) {
- src := "/Artifact /P1 BDC (x) Tj EMC"
- ops := collect(t, src)
- if len(ops) != 3 {
- t.Fatalf("want 3 ops, got %d", len(ops))
- }
- if ops[0].Operator != "BDC" {
- t.Fatalf("op[0].Operator = %q, want BDC", ops[0].Operator)
- }
- if ops[0].Operands[0].Name != "Artifact" {
- t.Errorf("tag = %q, want Artifact", ops[0].Operands[0].Name)
- }
- if ops[0].Operands[1].Kind != contentstream.KindName || ops[0].Operands[1].Name != "P1" {
- t.Errorf("properties ref = %+v, want name P1", ops[0].Operands[1])
- }
-}
-
-func TestHexString(t *testing.T) {
- src := "<48656C6C6F> Tj"
- ops := collect(t, src)
- if len(ops) != 1 {
- t.Fatalf("want 1 op, got %d", len(ops))
- }
- if string(ops[0].Operands[0].Bytes) != "Hello" {
- t.Errorf("hex Tj = %q, want Hello", ops[0].Operands[0].Bytes)
- }
-}
-
-func TestInlineImage(t *testing.T) {
- // BI /W 2 /H 2 /CS /G /BPC 8 ID
- // 4 raw bytes (\x00\x01\x02\x03) then EI
- src := "BI /W 2 /H 2 /CS /G /BPC 8 ID \x00\x01\x02\x03\nEI Q"
- ops := collect(t, src)
- if len(ops) != 2 {
- t.Fatalf("want 2 ops, got %d (%+v)", len(ops), ops)
- }
- if ops[0].Operator != "EI" {
- t.Fatalf("op[0].Operator = %q, want EI", ops[0].Operator)
- }
- if !reflect.DeepEqual(ops[0].Image, []byte{0, 1, 2, 3}) {
- t.Errorf("inline image bytes = % x, want 00 01 02 03", ops[0].Image)
- }
- if ops[1].Operator != "Q" {
- t.Errorf("op[1].Operator = %q, want Q", ops[1].Operator)
- }
-}
-
-func TestNumberInt(t *testing.T) {
- src := "42 3.14 0 Tr"
- ops := collect(t, src)
- if len(ops) != 1 {
- t.Fatalf("want 1 op, got %d", len(ops))
- }
- if n, ok := ops[0].Operands[0].Int(); !ok || n != 42 {
- t.Errorf("int %v ok=%v, want 42", n, ok)
- }
- if _, ok := ops[0].Operands[1].Int(); ok {
- t.Errorf("real should not yield Int()")
- }
-}
-
-func TestAllIteratorStopsOnError(t *testing.T) {
- src := "<>", 5000) + " BDC"
- sc := contentstream.New([]byte(src))
- if _, err := sc.Next(); err == nil {
- t.Fatal("expected a nesting-depth error, got nil")
- }
-}
-
-// Control: moderate nesting must still resolve, proving the limit doesn't
-// reject legitimate content.
-func TestModeratelyNestedArrayResolves(t *testing.T) {
- const depth = 100
- src := strings.Repeat("[", depth) + strings.Repeat("]", depth) + " n"
- sc := contentstream.New([]byte(src))
- op, err := sc.Next()
- if err != nil {
- t.Fatalf("unexpected error at depth %d: %v", depth, err)
- }
- if op.Operator != "n" || len(op.Operands) != 1 || op.Operands[0].Kind != contentstream.KindArray {
- t.Fatalf("want n op with one array operand, got %+v", op)
- }
-}
-
-// A flood of operands before an operator — directly or inside one array —
-// must be rejected rather than accumulated unboundedly.
-func TestScannerOperandFloodRejected(t *testing.T) {
- for name, src := range map[string]string{
- "bare": strings.Repeat("1 ", 200000) + "n",
- "array": "[" + strings.Repeat("1 ", 200000) + "] n",
- } {
- t.Run(name, func(t *testing.T) {
- if _, err := contentstream.New([]byte(src)).Next(); err == nil {
- t.Fatal("expected an operand-flood error, got nil")
- }
- })
- }
-}
-
-// FuzzScanner asserts the content-stream scanner never panics: Next may error,
-// but must not crash on arbitrary input.
-func FuzzScanner(f *testing.F) {
- f.Add([]byte("q 1 0 0 1 0 0 cm /F1 12 Tf (hi) Tj [(a) -5 (b)] TJ BI /W 1 ID xx EI Q"))
- f.Add([]byte("/Span << /MCID 7 >> BDC EMC"))
- f.Fuzz(func(t *testing.T, data []byte) {
- sc := contentstream.New(data)
- for i := 0; i <= len(data); i++ {
- if _, err := sc.Next(); err != nil {
- break
- }
- }
- })
-}
-
-// One BDC property dict, deliberately built to hit every readValue value kind.
-func TestBDCPropertyValueKinds(t *testing.T) {
- src := `/P << /B true /F false /N null /Num 1 /Nm /X /S (s) /A [1 2] /D << /Inner 9 >> >> BDC`
- ops := collect(t, src)
- if len(ops) != 1 || ops[0].Operator != "BDC" {
- t.Fatalf("want one BDC op, got %+v", ops)
- }
- d := ops[0].Operands[1].Dict
- wantKind := map[string]contentstream.Kind{
- "B": contentstream.KindBool,
- "F": contentstream.KindBool,
- "N": contentstream.KindNull,
- "Num": contentstream.KindNumber,
- "Nm": contentstream.KindName,
- "S": contentstream.KindString,
- "A": contentstream.KindArray,
- "D": contentstream.KindDict,
- }
- for k, want := range wantKind {
- if d[k].Kind != want {
- t.Errorf("%s.Kind = %v, want %v", k, d[k].Kind, want)
- }
- }
- if !d["B"].Bool || d["F"].Bool {
- t.Errorf("bool values wrong: B=%v F=%v, want true/false", d["B"].Bool, d["F"].Bool)
- }
- if string(d["S"].Bytes) != "s" {
- t.Errorf("string value = %q, want s", d["S"].Bytes)
- }
- if len(d["A"].Array) != 2 {
- t.Errorf("array value len = %d, want 2", len(d["A"].Array))
- }
- if d["D"].Dict["Inner"].Kind != contentstream.KindNumber {
- t.Errorf("nested dict /Inner kind = %v, want Number", d["D"].Dict["Inner"].Kind)
- }
-}
-
-func TestReadDictValueErrors(t *testing.T) {
- for _, src := range []string{
- "/P << /K", // EOF mid-value
- "/P << /K ] >> BDC", // delimiter where a value is expected
- "/P << /K foo >> BDC", // unexpected keyword where a value is expected
- } {
- t.Run(src, func(t *testing.T) {
- if _, err := contentstream.New([]byte(src)).Next(); err == nil {
- t.Fatal("expected an error, got nil")
- }
- })
- }
-}
-
-// Every readArray element kind, including the ones TJ rarely uses (bool, null,
-// name, nested array/dict).
-func TestArrayMixedElementKinds(t *testing.T) {
- ops := collect(t, "[42 3.14 /Nm (s) <41> true false null [1 2] << /K 9 >>] TJ")
- if len(ops) != 1 || ops[0].Operator != "TJ" {
- t.Fatalf("want one TJ op, got %+v", ops)
- }
- got := ops[0].Operands[0].Array
- wantKinds := []contentstream.Kind{
- contentstream.KindNumber, contentstream.KindNumber, contentstream.KindName,
- contentstream.KindString, contentstream.KindString, contentstream.KindBool,
- contentstream.KindBool, contentstream.KindNull, contentstream.KindArray,
- contentstream.KindDict,
- }
- if len(got) != len(wantKinds) {
- t.Fatalf("array len = %d, want %d", len(got), len(wantKinds))
- }
- for i, want := range wantKinds {
- if got[i].Kind != want {
- t.Errorf("element %d Kind = %v, want %v", i, got[i].Kind, want)
- }
- }
-}
-
-func TestArrayElementErrors(t *testing.T) {
- for _, src := range []string{"[ foo ] n", "[1 2"} { // bad keyword in array; unterminated
- t.Run(src, func(t *testing.T) {
- if _, err := contentstream.New([]byte(src)).Next(); err == nil {
- t.Fatal("expected an error, got nil")
- }
- })
- }
-}
-
-// Int must report ok=false on int64 overflow (not silently wrap) and for non-numbers.
-func TestOperandIntEdgeCases(t *testing.T) {
- if _, ok := collect(t, "/Name n")[0].Operands[0].Int(); ok {
- t.Error("name operand yielded an int")
- }
- if _, ok := collect(t, "99999999999999999999999999 n")[0].Operands[0].Int(); ok {
- t.Error("overflowing integer literal yielded an int")
- }
-}
-
-// readInlineImage must error, not panic, on a malformed BI block.
-func TestInlineImageErrors(t *testing.T) {
- for _, src := range []string{
- "BI /W 1", // EOF before ID
- "BI 5 ID x EI", // non-name key before ID
- "BI /W ] ID x EI", // bad entry value
- "BI /W 1 ID abcdefg", // no EI terminator before EOF
- } {
- t.Run(src, func(t *testing.T) {
- if _, err := contentstream.New([]byte(src)).Next(); err == nil {
- t.Fatal("expected an error, got nil")
- }
- })
- }
-}
-
-func TestNextStrayDelimiter(t *testing.T) {
- for _, src := range []string{"]", ">>"} {
- if _, err := contentstream.New([]byte(src)).Next(); err == nil {
- t.Errorf("%q: expected an error, got nil", src)
- }
- }
-}
-
-// Breaking out of the All range mid-iteration must stop cleanly (the
-// yield-returned-false path).
-func TestAllEarlyBreak(t *testing.T) {
- sc := contentstream.New([]byte("q Q q"))
- n := 0
- for op, err := range sc.All() {
- if err != nil {
- t.Fatalf("unexpected error: %v", err)
- }
- _ = op
- n++
- break
- }
- if n != 1 {
- t.Fatalf("iterated %d ops, want 1 before break", n)
- }
-}
-
-// Operands left on the stack at EOF with no trailing operator are dropped, not
-// emitted as a bogus op.
-func TestTrailingOperandsDropped(t *testing.T) {
- if ops := collect(t, "1 2 3"); len(ops) != 0 {
- t.Fatalf("got %d ops, want 0 (operands without an operator are dropped)", len(ops))
- }
-}
-
-func TestReadDictStructuralErrors(t *testing.T) {
- for _, src := range []string{
- "/X << 1 2 >> BDC", // key is not a name
- "/X << /K 1", // EOF before '>>'
- } {
- t.Run(src, func(t *testing.T) {
- if _, err := contentstream.New([]byte(src)).Next(); err == nil {
- t.Fatal("expected an error, got nil")
- }
- })
- }
-}
-
-// An "EI" that appears in the image data without a whitespace boundary must be
-// skipped; scanning continues to the real, whitespace-delimited terminator.
-func TestInlineImageFakeEIInData(t *testing.T) {
- ops := collect(t, "BI ID aEIb EI Q")
- if len(ops) < 1 || ops[0].Operator != "EI" {
- t.Fatalf("want EI op first, got %+v", ops)
- }
- if string(ops[0].Image) != "aEIb" {
- t.Errorf("image = %q, want aEIb", ops[0].Image)
- }
-}
diff --git a/crypt.go b/crypt.go
deleted file mode 100644
index 5624c4c..0000000
--- a/crypt.go
+++ /dev/null
@@ -1,121 +0,0 @@
-package pdfdisassembler
-
-import (
- "fmt"
-
- "github.com/speedata/pdfdisassembler/internal/crypt"
-)
-
-// encryptCtx wraps the document-level encryption state. nil if the PDF is
-// unencrypted.
-type encryptCtx struct {
- handler *crypt.Handler
-}
-
-// initEncrypt reads the trailer /Encrypt entry, builds the crypt.Handler
-// using the empty user password. Documents secured with a non-empty user
-// password fail to open via the public API; callers will need an explicit
-// password hook (TODO: expose).
-func (r *Reader) initEncrypt() error {
- if r.trailer == nil {
- return nil
- }
- encObj, ok := r.trailer.Get("Encrypt")
- if !ok {
- return nil
- }
- encDict, err := r.ResolveDict(encObj)
- if err != nil {
- return fmt.Errorf("pdfdisassembler: /Encrypt: %w", err)
- }
-
- filter, _ := encDict.Name("Filter")
- if filter != "Standard" {
- return fmt.Errorf("pdfdisassembler: encryption filter %q not supported", filter)
- }
-
- params, err := encryptParamsFromDict(r, encDict)
- if err != nil {
- return err
- }
-
- h, err := crypt.New(params, nil)
- if err != nil {
- return fmt.Errorf("pdfdisassembler: encryption: %w", err)
- }
- r.encrypt = &encryptCtx{handler: h}
- return nil
-}
-
-func encryptParamsFromDict(r *Reader, d *Dict) (crypt.Params, error) {
- var p crypt.Params
- if v, ok := d.Int("V"); ok {
- p.V = int(v)
- }
- if v, ok := d.Int("R"); ok {
- p.R = int(v)
- }
- if v, ok := d.Int("Length"); ok {
- p.Length = int(v)
- } else {
- p.Length = 40 // V1 default
- }
- if v, ok := d.Int("P"); ok {
- p.P = int32(v)
- }
- if o, ok := d.Bytes("O"); ok {
- p.OwnerEntry = o
- }
- if u, ok := d.Bytes("U"); ok {
- p.UserEntry = u
- }
- if oe, ok := d.Bytes("OE"); ok {
- p.OE = oe
- }
- if ue, ok := d.Bytes("UE"); ok {
- p.UE = ue
- }
- if perms, ok := d.Bytes("Perms"); ok {
- p.Perms = perms
- }
- p.EncryptMeta = true
- if v, ok := d.Bool("EncryptMetadata"); ok {
- p.EncryptMeta = v
- }
- if n, ok := d.Name("StmF"); ok {
- p.StmF = string(n)
- }
- if n, ok := d.Name("StrF"); ok {
- p.StrF = string(n)
- }
- if n, ok := d.Name("EFF"); ok {
- p.EFF = string(n)
- }
- // File ID first element comes from trailer.
- if id, ok := r.trailer.Array("ID"); ok && len(id) >= 1 {
- if s, ok := id[0].(String); ok {
- p.ID0 = []byte(s)
- }
- }
-
- // /CF dictionary: name → CFM string.
- p.CryptFilters = map[string]string{}
- if cf, ok := d.Dict("CF"); ok {
- for name, v := range cf.Iter() {
- cd, ok := v.(*Dict)
- if !ok {
- continue
- }
- if cfm, ok := cd.Name("CFM"); ok {
- p.CryptFilters[name] = string(cfm)
- }
- }
- }
- return p, nil
-}
-
-func (e *encryptCtx) decryptStream(data []byte, objNum, objGen int) ([]byte, error) {
- // V4 streams may carry an inline /Crypt filter overriding the cipher; it
- // is not yet honored — the default stream cipher is always used.
- return e.handler.DecryptStream(data, objNum, objGen, "")
-}
diff --git a/crypt_test.go b/crypt_test.go
deleted file mode 100644
index ba3aea2..0000000
--- a/crypt_test.go
+++ /dev/null
@@ -1,287 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "crypto/md5"
- "crypto/rc4"
- "encoding/hex"
- "fmt"
- "strings"
- "testing"
-)
-
-// buildEncryptedPDF constructs a PDF secured with the /Standard handler
-// (V2/R3 RC4) whose /Encrypt dict declares the given /Length in bits. /O and
-// /U are 32-byte placeholders; the empty-password key derivation runs during
-// Open regardless of whether they validate.
-func buildEncryptedPDF(t *testing.T, length int) []byte {
- t.Helper()
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
-
- offsets := make([]int, 4) // index 1..3
-
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
-
- o := strings.Repeat("ab", 32) // 32 bytes, hex-encoded
- u := strings.Repeat("cd", 32)
- offsets[3] = off()
- fmt.Fprintf(&buf,
- "3 0 obj\n<< /Filter /Standard /V 2 /R 3 /Length %d /O <%s> /U <%s> /P -44 >>\nendobj\n",
- length, o, u)
-
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 4\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 3; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- id := "<00112233445566778899aabbccddeeff>"
- fmt.Fprintf(&buf,
- "trailer\n<< /Size 4 /Root 1 0 R /Encrypt 3 0 R /ID [%s %s] >>\n", id, id)
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-// A malicious /Encrypt dict can declare a /Length whose key size exceeds the
-// 16-byte MD5 digest (or is negative). Open must surface an error, not panic.
-func TestEncryptHostileKeyLengthNoPanic(t *testing.T) {
- for _, length := range []int{256, 4096, -8} {
- t.Run(fmt.Sprintf("length_%d", length), func(t *testing.T) {
- data := buildEncryptedPDF(t, length)
- if _, err := Open(bytes.NewReader(data)); err == nil {
- t.Fatal("expected an error for hostile /Length, got nil")
- }
- })
- }
-}
-
-// stdPassPad is the 32-byte padding string from PDF 32000-1:2008 algorithm 2,
-// used to build an empty-password V2/R3 fixture.
-var stdPassPad = []byte{
- 0x28, 0xbf, 0x4e, 0x5e, 0x4e, 0x75, 0x8a, 0x41,
- 0x64, 0x00, 0x4e, 0x56, 0xff, 0xfa, 0x01, 0x08,
- 0x2e, 0x2e, 0x00, 0xb6, 0xd0, 0x68, 0x3e, 0x80,
- 0x2f, 0x0c, 0xa9, 0xfe, 0x64, 0x53, 0x69, 0x7a,
-}
-
-// emptyPwRC4Key derives the V2/R3 file key for the empty user password.
-func emptyPwRC4Key(owner, id0 []byte, p int32, bits int) []byte {
- h := md5.New()
- h.Write(stdPassPad)
- h.Write(owner)
- h.Write([]byte{byte(uint32(p)), byte(uint32(p) >> 8), byte(uint32(p) >> 16), byte(uint32(p) >> 24)})
- h.Write(id0)
- sum := h.Sum(nil)
- keyLen := bits / 8
- for i := 0; i < 50; i++ {
- s := md5.Sum(sum[:keyLen])
- sum = s[:]
- }
- key := make([]byte, keyLen)
- copy(key, sum[:keyLen])
- return key
-}
-
-// emptyPwU computes the /U value (algorithm 5, R>=3) for the empty password,
-// so Open's password validation accepts the fixture.
-func emptyPwU(key, id0 []byte) []byte {
- h := md5.New()
- h.Write(stdPassPad)
- h.Write(id0)
- digest := h.Sum(nil)
- out := make([]byte, 16)
- c, _ := rc4.NewCipher(key)
- c.XORKeyStream(out, digest)
- for i := 1; i <= 19; i++ {
- tweaked := make([]byte, len(key))
- for j, b := range key {
- tweaked[j] = b ^ byte(i)
- }
- c2, _ := rc4.NewCipher(tweaked)
- c2.XORKeyStream(out, out)
- }
- u := make([]byte, 32)
- copy(u, out)
- return u
-}
-
-// objKeyRC4 derives the per-object RC4 key (algorithm 1).
-func objKeyRC4(fileKey []byte, num, gen int) []byte {
- buf := append([]byte{}, fileKey...)
- buf = append(buf, byte(num), byte(num>>8), byte(num>>16), byte(gen), byte(gen>>8))
- sum := md5.Sum(buf)
- n := len(fileKey) + 5
- if n > 16 {
- n = 16
- }
- return sum[:n]
-}
-
-func rc4Crypt(key, data []byte) []byte {
- out := make([]byte, len(data))
- c, _ := rc4.NewCipher(key)
- c.XORKeyStream(out, data)
- return out
-}
-
-// assembleEncryptedPDF builds a classical-xref PDF — catalog (1), pages (2),
-// the given /Encrypt dict body (3), and an already-encrypted stream (4) — with
-// the trailer wired to /Encrypt 3 0 R and /ID [id0 id0].
-func assembleEncryptedPDF(encryptBody string, encStream, id0 []byte) []byte {
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
- offsets := make([]int, 5) // 1..4
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
- offsets[3] = off()
- fmt.Fprintf(&buf, "3 0 obj\n%s\nendobj\n", encryptBody)
- offsets[4] = off()
- fmt.Fprintf(&buf, "4 0 obj\n<< /Length %d >>\nstream\n", len(encStream))
- buf.Write(encStream)
- fmt.Fprint(&buf, "\nendstream\nendobj\n")
-
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 5\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 4; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- id := hex.EncodeToString(id0)
- fmt.Fprintf(&buf,
- "trailer\n<< /Size 5 /Root 1 0 R /Encrypt 3 0 R /ID [<%s> <%s>] >>\n", id, id)
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-// buildRC4EncryptedStreamPDF builds a V2/R3 RC4-encrypted PDF (empty password)
-// whose object 4 is a stream carrying RC4-encrypted plaintext.
-func buildRC4EncryptedStreamPDF(t *testing.T, plaintext []byte) []byte {
- t.Helper()
- owner := bytes.Repeat([]byte{0x5a}, 32)
- id0 := bytes.Repeat([]byte{0x7c}, 16)
- const bits = 128
- var p int32 = -44
- fileKey := emptyPwRC4Key(owner, id0, p, bits)
- u := emptyPwU(fileKey, id0)
- enc := rc4Crypt(objKeyRC4(fileKey, 4, 0), plaintext)
- body := fmt.Sprintf("<< /Filter /Standard /V 2 /R 3 /Length %d /O <%s> /U <%s> /P %d >>",
- bits, hex.EncodeToString(owner), hex.EncodeToString(u), p)
- return assembleEncryptedPDF(body, enc, id0)
-}
-
-// buildV4RC4EncryptedStreamPDF builds a V4/R4 PDF whose StdCF crypt filter uses
-// CFM /V2 (RC4) — same empty-password key derivation as V2/R3, reached through
-// the V4 /CF + /StmF parsing path.
-func buildV4RC4EncryptedStreamPDF(t *testing.T, plaintext []byte) []byte {
- t.Helper()
- owner := bytes.Repeat([]byte{0x5a}, 32)
- id0 := bytes.Repeat([]byte{0x7c}, 16)
- const bits = 128
- var p int32 = -44
- fileKey := emptyPwRC4Key(owner, id0, p, bits)
- u := emptyPwU(fileKey, id0)
- enc := rc4Crypt(objKeyRC4(fileKey, 4, 0), plaintext)
- body := fmt.Sprintf("<< /Filter /Standard /V 4 /R 4 /Length %d /O <%s> /U <%s> /P %d "+
- "/CF << /StdCF << /CFM /V2 /Length 16 >> >> /StmF /StdCF /StrF /StdCF /EncryptMetadata true >>",
- bits, hex.EncodeToString(owner), hex.EncodeToString(u), p)
- return assembleEncryptedPDF(body, enc, id0)
-}
-
-// Open must accept an RC4-encrypted PDF secured with the empty user password
-// and decrypt its stream content end-to-end.
-func TestOpenDecryptsRC4Stream(t *testing.T) {
- plaintext := []byte("BT (top secret invoice) Tj ET")
- data := buildRC4EncryptedStreamPDF(t, plaintext)
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- got, err := r.DecodeStream(Reference{Number: 4, Generation: 0})
- if err != nil {
- t.Fatalf("DecodeStream: %v", err)
- }
- if !bytes.Equal(got, plaintext) {
- t.Fatalf("decrypted stream mismatch:\n got %q\nwant %q", got, plaintext)
- }
-}
-
-// The V4 path resolves the stream cipher through /CF + /StmF rather than /V
-// directly, so it must be exercised end-to-end too.
-func TestOpenDecryptsV4RC4Stream(t *testing.T) {
- plaintext := []byte("BT (V4 crypt filter) Tj ET")
- data := buildV4RC4EncryptedStreamPDF(t, plaintext)
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- got, err := r.DecodeStream(Reference{Number: 4, Generation: 0})
- if err != nil {
- t.Fatalf("DecodeStream: %v", err)
- }
- if !bytes.Equal(got, plaintext) {
- t.Fatalf("V4 decrypted stream mismatch:\n got %q\nwant %q", got, plaintext)
- }
-}
-
-// buildPDFWithEncryptObj wraps an arbitrary /Encrypt dict body as object 3 of a
-// classical-xref PDF, with the trailer pointing /Encrypt at it.
-func buildPDFWithEncryptObj(t *testing.T, encryptBody string) []byte {
- t.Helper()
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
- offsets := make([]int, 4)
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
- offsets[3] = off()
- fmt.Fprintf(&buf, "3 0 obj\n%s\nendobj\n", encryptBody)
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 4\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 3; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- id := "<00112233445566778899aabbccddeeff>"
- fmt.Fprintf(&buf, "trailer\n<< /Size 4 /Root 1 0 R /Encrypt 3 0 R /ID [%s %s] >>\n", id, id)
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-// Only the /Standard security handler is supported; any other /Filter must be
-// rejected rather than silently treated as unencrypted.
-func TestEncryptNonStandardFilterRejected(t *testing.T) {
- data := buildPDFWithEncryptObj(t, "<< /Filter /FooSecurity /V 2 /R 3 /Length 128 >>")
- if _, err := Open(bytes.NewReader(data)); err == nil {
- t.Fatal("expected an error for a non-Standard /Filter, got nil")
- }
-}
-
-// Malformed /Encrypt dictionaries (wrong type, missing fields, short V5 entries)
-// must surface as errors during Open, never panics.
-func TestEncryptMalformedNoPanic(t *testing.T) {
- cases := map[string]string{
- "encrypt not a dict": "42",
- "missing V and R": "<< /Filter /Standard >>",
- "short O and U": "<< /Filter /Standard /V 2 /R 3 /Length 128 /O <00> /U <00> /P 0 >>",
- "v5 short entries": "<< /Filter /Standard /V 5 /R 6 /Length 256 /O <00> /U <00> /OE <00> /UE <00> /Perms <00> /P 0 >>",
- }
- for name, body := range cases {
- t.Run(name, func(t *testing.T) {
- if _, err := Open(bytes.NewReader(buildPDFWithEncryptObj(t, body))); err == nil {
- t.Fatal("expected an error, got nil")
- }
- })
- }
-}
diff --git a/doc.go b/doc.go
deleted file mode 100644
index d1e3432..0000000
--- a/doc.go
+++ /dev/null
@@ -1,46 +0,0 @@
-// Package pdfdisassembler is a focused, read-only PDF parser for Go.
-//
-// It targets tooling that *inspects* PDFs — accessibility checkers,
-// validators, debuggers — without dragging in the writing, optimisation,
-// signing and image-rendering machinery that general-purpose PDF libraries
-// carry.
-//
-// # Scope
-//
-// In scope: PDF 1.x and 2.0 reading, classical xref and xref streams,
-// indirect-object resolution, stream filters (FlateDecode, ASCII85,
-// ASCIIHex, LZW, RunLength, and BrotliDecode — a PDF Association
-// extension to PDF 2.0, pending ISO 32000 inclusion), text-string
-// decoding (PDFDocEncoding, UTF-16BE BOM, UTF-8 BOM), catalog +
-// page-tree navigation (page boxes,
-// rotation, resources and content streams, with inherited attributes
-// resolved along the /Parent chain), DocumentInfo, XMP metadata access,
-// structure-tree traversal, the /Standard security handler (V2, V4, V5),
-// defensive xref recovery.
-//
-// Out of scope: writing PDFs, image filters (DCTDecode/JBIG2/JPX/CCITTFax),
-// image rendering, font internals, XFA, public-key encryption, signature
-// verification, content-stream graphics-state interpretation, LTV.
-//
-// # Usage
-//
-// r, err := pdfdisassembler.OpenFile("doc.pdf")
-// if err != nil {
-// return err
-// }
-// defer r.Close()
-//
-// fmt.Println("PDF version:", r.Version())
-// info := r.DocumentInfo()
-// fmt.Println("Title:", info.Title)
-//
-// for entry := range r.Objects() {
-// // inspect every live indirect object
-// _ = entry
-// }
-//
-// # API stability
-//
-// Pre-1.0. The API may break between minor releases but never within a
-// patch release.
-package pdfdisassembler
diff --git a/dump.go b/dump.go
deleted file mode 100644
index 4e2b59a..0000000
--- a/dump.go
+++ /dev/null
@@ -1,297 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "crypto/sha256"
- "encoding/hex"
- "encoding/json"
- "fmt"
-)
-
-// DumpOptions controls Dump's behaviour.
-type DumpOptions struct {
- // PreviewMaxBytes is the maximum number of decoded stream bytes shown
- // as the preview_utf8 field. Default 80. Set to -1 to disable.
- PreviewMaxBytes int
- // InlineStreamContent embeds the decoded stream as hex under the
- // "decoded.hex" field. Off by default — real PDFs produce huge
- // fixtures otherwise.
- InlineStreamContent bool
-}
-
-// Dump returns a deterministic JSON snapshot of r suitable for golden-file
-// snapshot tests and human inspection.
-//
-// The output is tagged so every PDF value kind is unambiguous (Name vs.
-// Text-String vs. Byte-String, Integer vs. Real, etc.). Indirect references
-// are preserved as references, so the object graph is acyclic and diffs
-// stay reviewable in PRs. Streams contribute metadata (raw_length, filter
-// chain, decoded length, SHA-256, optional preview) but never their full
-// content — see DumpOptions.InlineStreamContent if you need it.
-//
-// The output is intended to be byte-stable across runs; dictionary keys
-// are emitted in PDF insertion order, objects in (Number, Generation)
-// order.
-func Dump(r *Reader, opts DumpOptions) ([]byte, error) {
- if opts.PreviewMaxBytes == 0 {
- opts.PreviewMaxBytes = 80
- }
-
- top := orderedMap{}
- top = append(top, orderedKV{"version", r.Version()})
- top = append(top, orderedKV{"xref_format", r.xrefFormat()})
- top = append(top, orderedKV{"encrypted", r.encrypt != nil})
-
- if r.trailer != nil {
- top = append(top, orderedKV{"trailer", dumpDictTagged(r.trailer, opts)})
- }
-
- objs := orderedMap{}
- for entry := range r.Objects() {
- key := fmt.Sprintf("%d %d", entry.Reference.Number, entry.Reference.Generation)
- objs = append(objs, orderedKV{key, dumpValue(entry.Object, opts)})
- }
- top = append(top, orderedKV{"objects", objs})
-
- raw, err := json.Marshal(top)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler/dump: marshal: %w", err)
- }
- var pretty bytes.Buffer
- if err := json.Indent(&pretty, raw, "", " "); err != nil {
- return nil, fmt.Errorf("pdfdisassembler/dump: indent: %w", err)
- }
- pretty.WriteByte('\n')
- // Unescape the HTML-safe sequences that json.Marshal emits by default.
- // PDFs are full of '<<' and '>>' (dict delimiters, hex strings), and
- // '&' shows up in metadata XML — readable diffs win over byte-pedantic
- // HTML-safety, which is irrelevant here. The transform is safe: these
- // escape sequences only appear inside JSON string literals, where the
- // unescaped form is equivalent.
- return unescapeHTMLSafe(pretty.Bytes()), nil
-}
-
-// unescapeHTMLSafe rewrites the six-byte sequences <, >, &
-// back into their single-character forms <, >, &. These escapes only
-// appear inside JSON string literals (no backslashes outside strings),
-// so substitution is byte-safe and JSON remains valid.
-func unescapeHTMLSafe(b []byte) []byte {
- b = bytes.ReplaceAll(b, []byte{'\\', 'u', '0', '0', '3', 'c'}, []byte{'<'})
- b = bytes.ReplaceAll(b, []byte{'\\', 'u', '0', '0', '3', 'C'}, []byte{'<'})
- b = bytes.ReplaceAll(b, []byte{'\\', 'u', '0', '0', '3', 'e'}, []byte{'>'})
- b = bytes.ReplaceAll(b, []byte{'\\', 'u', '0', '0', '3', 'E'}, []byte{'>'})
- b = bytes.ReplaceAll(b, []byte{'\\', 'u', '0', '0', '2', '6'}, []byte{'&'})
- return b
-}
-
-// xrefFormat reports how the cross-reference table was stored.
-func (r *Reader) xrefFormat() string {
- if r.trailer == nil {
- return "unknown"
- }
- if t, ok := r.trailer.Name("Type"); ok && t == "XRef" {
- return "stream"
- }
- if _, ok := r.trailer.Get("XRefStm"); ok {
- return "hybrid"
- }
- return "classical"
-}
-
-// orderedKV is one entry of an orderedMap.
-type orderedKV struct {
- Key string
- Val any
-}
-
-// orderedMap is a JSON object that marshals in insertion order. Used for
-// both the top-level dump and every PDF dictionary, so that key order in
-// goldens matches the producer's writing order.
-type orderedMap []orderedKV
-
-func (m orderedMap) MarshalJSON() ([]byte, error) {
- var buf bytes.Buffer
- buf.WriteByte('{')
- for i, kv := range m {
- if i > 0 {
- buf.WriteByte(',')
- }
- kb, err := json.Marshal(kv.Key)
- if err != nil {
- return nil, err
- }
- buf.Write(kb)
- buf.WriteByte(':')
- vb, err := json.Marshal(kv.Val)
- if err != nil {
- return nil, err
- }
- buf.Write(vb)
- }
- buf.WriteByte('}')
- return buf.Bytes(), nil
-}
-
-// dumpValue produces the tagged JSON form of a PDF object.
-func dumpValue(v Object, opts DumpOptions) orderedMap {
- switch t := v.(type) {
- case Name:
- return orderedMap{{"name", string(t)}}
- case Integer:
- return orderedMap{{"int", int64(t)}}
- case Real:
- return orderedMap{{"real", float64(t)}}
- case Bool:
- return orderedMap{{"bool", bool(t)}}
- case Null:
- return orderedMap{{"null", nil}}
- case String:
- if looksLikeText(t) {
- return orderedMap{{"text", decodeTextString(t)}}
- }
- return orderedMap{{"hex", hex.EncodeToString(t)}}
- case Reference:
- return orderedMap{{"ref", fmt.Sprintf("%d %d", t.Number, t.Generation)}}
- case Array:
- out := make([]orderedMap, len(t))
- for i, e := range t {
- out[i] = dumpValue(e, opts)
- }
- return orderedMap{{"array", out}}
- case *Dict:
- return orderedMap{{"dict", dumpDictContents(t, opts)}}
- case *Stream:
- return orderedMap{{"stream", dumpStream(t, opts)}}
- case nil:
- return orderedMap{{"null", nil}}
- }
- return orderedMap{{"unknown", fmt.Sprintf("%T", v)}}
-}
-
-func dumpDictContents(d *Dict, opts DumpOptions) orderedMap {
- if d == nil {
- return orderedMap{}
- }
- out := orderedMap{}
- for k, v := range d.Iter() {
- out = append(out, orderedKV{k, dumpValue(v, opts)})
- }
- return out
-}
-
-func dumpDictTagged(d *Dict, opts DumpOptions) orderedMap {
- return orderedMap{{"dict", dumpDictContents(d, opts)}}
-}
-
-func dumpStream(s *Stream, opts DumpOptions) orderedMap {
- out := orderedMap{}
- out = append(out, orderedKV{"dict", dumpDictContents(s.Dict, opts)})
- out = append(out, orderedKV{"raw_length", s.rawLength})
-
- names, _, ferr := s.reader.streamFilterChain(s.Dict)
- if ferr == nil {
- if names == nil {
- names = []string{}
- }
- out = append(out, orderedKV{"filters", names})
- } else {
- out = append(out, orderedKV{"filters_error", ferr.Error()})
- }
-
- dec := orderedMap{}
- decoded, derr := s.Content()
- if derr != nil {
- dec = append(dec, orderedKV{"error", derr.Error()})
- } else {
- sum := sha256.Sum256(decoded)
- dec = append(dec, orderedKV{"length", int64(len(decoded))})
- dec = append(dec, orderedKV{"sha256", hex.EncodeToString(sum[:])})
- if opts.PreviewMaxBytes > 0 {
- if p := preview(decoded, opts.PreviewMaxBytes); p != "" {
- dec = append(dec, orderedKV{"preview_utf8", p})
- }
- }
- if opts.InlineStreamContent {
- dec = append(dec, orderedKV{"hex", hex.EncodeToString(decoded)})
- }
- }
- out = append(out, orderedKV{"decoded", dec})
- return out
-}
-
-// looksLikeText reports whether s is likely a PDF text string.
-//
-// The rule:
-//
-// 1. BOM-prefixed strings (UTF-16BE/LE, UTF-8) → text
-// 2. Strings with any C0 control byte (except \t, \r, \n) or DEL → hex
-// This catches file identifiers, hashes and encryption blobs, where
-// random bytes almost always include something in 0x00–0x1F.
-// 3. ASCII-only strings → text
-// 4. Strings with high bytes (0x80–0xFF) → decode via PDFDocEncoding;
-// if the decoded form contains no U+FFFD (undefined-slot marker)
-// and no control runes, treat as text. This catches PDFDocEncoded
-// content like ActualText with bullets, en-dashes, etc.
-func looksLikeText(s String) bool {
- if len(s) >= 2 {
- if s[0] == 0xFE && s[1] == 0xFF {
- return true
- }
- if s[0] == 0xFF && s[1] == 0xFE {
- return true
- }
- }
- if len(s) >= 3 && s[0] == 0xEF && s[1] == 0xBB && s[2] == 0xBF {
- return true
- }
- hasHigh := false
- for _, c := range s {
- if c == '\t' || c == '\r' || c == '\n' {
- continue
- }
- if c < 0x20 || c == 0x7F {
- return false
- }
- if c >= 0x80 {
- hasHigh = true
- }
- }
- if !hasHigh {
- return true
- }
- // Verify the PDFDocEncoded form is clean.
- for _, r := range decodePDFDocEncoding(s) {
- if r == 0xFFFD {
- return false
- }
- if r < 0x20 && r != '\t' && r != '\r' && r != '\n' {
- return false
- }
- }
- return true
-}
-
-// preview returns up to max bytes of b as a string, or "" if any byte is
-// non-printable. Truncated previews are suffixed with an ellipsis.
-func preview(b []byte, max int) string {
- n := len(b)
- truncated := false
- if n > max {
- n = max
- truncated = true
- }
- head := b[:n]
- for _, c := range head {
- if c == '\t' || c == '\r' || c == '\n' {
- continue
- }
- if c < 0x20 || c > 0x7E {
- return ""
- }
- }
- s := string(head)
- if truncated {
- s += "…"
- }
- return s
-}
diff --git a/examples/inspect/main.go b/examples/inspect/main.go
deleted file mode 100644
index 605af7e..0000000
--- a/examples/inspect/main.go
+++ /dev/null
@@ -1,60 +0,0 @@
-// Command inspect prints a summary of a PDF: version, document info,
-// catalog top-level keys, page count.
-package main
-
-import (
- "fmt"
- "log"
- "os"
-
- "github.com/speedata/pdfdisassembler"
-)
-
-func main() {
- if len(os.Args) < 2 {
- fmt.Fprintln(os.Stderr, "usage: inspect ")
- os.Exit(2)
- }
- r, err := pdfdisassembler.OpenFile(os.Args[1])
- if err != nil {
- log.Fatal(err)
- }
- defer r.Close()
-
- fmt.Printf("PDF version: %s\n", r.Version())
-
- info := r.DocumentInfo()
- if info.Title != "" {
- fmt.Printf("Title: %s\n", info.Title)
- }
- if info.Author != "" {
- fmt.Printf("Author: %s\n", info.Author)
- }
- if info.Producer != "" {
- fmt.Printf("Producer: %s\n", info.Producer)
- }
- if !info.CreationDate.IsZero() {
- fmt.Printf("Created: %s\n", info.CreationDate.Format("2006-01-02 15:04:05"))
- }
-
- cat, err := r.Catalog()
- if err != nil {
- log.Fatalf("catalog: %v", err)
- }
- fmt.Println("Catalog keys:")
- for k := range cat.Iter() {
- fmt.Printf(" /%s\n", k)
- }
-
- if pages, ok := cat.Dict("Pages"); ok {
- if n, ok := pages.Int("Count"); ok {
- fmt.Printf("Pages: %d\n", n)
- }
- }
-
- count := 0
- for range r.Objects() {
- count++
- }
- fmt.Printf("Live indirect objects: %d\n", count)
-}
diff --git a/examples/pageinfo/main.go b/examples/pageinfo/main.go
deleted file mode 100644
index 1db2ff9..0000000
--- a/examples/pageinfo/main.go
+++ /dev/null
@@ -1,71 +0,0 @@
-// Command pageinfo prints per-page geometry and content info for a PDF:
-// page count and, for each page, its boxes, rotation, resource categories,
-// and decoded content size. Intended as a starting point for page-importer
-// and layout tooling.
-package main
-
-import (
- "fmt"
- "io"
- "log"
- "os"
- "sort"
-
- "github.com/speedata/pdfdisassembler"
-)
-
-func main() {
- if len(os.Args) < 2 {
- fmt.Fprintln(os.Stderr, "usage: pageinfo ")
- os.Exit(2)
- }
- r, err := pdfdisassembler.OpenFile(os.Args[1])
- if err != nil {
- log.Fatal(err)
- }
- defer r.Close()
-
- if err := report(os.Stdout, r); err != nil {
- log.Fatal(err)
- }
-}
-
-// report writes a human-readable page summary for r to w. The page tree is
-// walked once via Reader.Pages; inherited attributes (boxes, rotation,
-// resources) are resolved by the Page accessors.
-func report(w io.Writer, r *pdfdisassembler.Reader) error {
- pages, err := r.Pages()
- if err != nil {
- return err
- }
- fmt.Fprintf(w, "Pages: %d\n", len(pages))
-
- for _, p := range pages {
- // Display pages 1-based, matching how readers number them; the API
- // itself is 0-based (Page.Index).
- fmt.Fprintf(w, "Page %d:\n", p.Index()+1)
-
- if box, ok := p.Box(pdfdisassembler.MediaBox); ok {
- fmt.Fprintf(w, " MediaBox: %g x %g pt\n", box.Width(), box.Height())
- }
- if box, ok := p.Box(pdfdisassembler.CropBox); ok {
- fmt.Fprintf(w, " CropBox: [%g %g %g %g]\n", box.LLX, box.LLY, box.URX, box.URY)
- }
- if rot := p.Rotation(); rot != 0 {
- fmt.Fprintf(w, " Rotation: %d\n", rot)
- }
- if res, ok := p.Resources(); ok {
- keys := res.Keys()
- sort.Strings(keys)
- fmt.Fprintf(w, " Resources: %v\n", keys)
- }
-
- content, err := p.Content()
- if err != nil {
- fmt.Fprintf(w, " Content: (error: %v)\n", err)
- continue
- }
- fmt.Fprintf(w, " Content: %d bytes decoded\n", len(content))
- }
- return nil
-}
diff --git a/examples/pageinfo/main_test.go b/examples/pageinfo/main_test.go
deleted file mode 100644
index 59bd043..0000000
--- a/examples/pageinfo/main_test.go
+++ /dev/null
@@ -1,108 +0,0 @@
-package main
-
-import (
- "bytes"
- "fmt"
- "strings"
- "testing"
-
- "github.com/speedata/pdfdisassembler"
-)
-
-// buildObjPDF assembles 1-based object bodies into a PDF with a classical xref
-// and the given trailer dictionary body (without the surrounding << >>).
-func buildObjPDF(t *testing.T, objs []string, trailer string) []byte {
- t.Helper()
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
- offsets := make([]int, len(objs)+1)
- for i, body := range objs {
- offsets[i+1] = off()
- fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", i+1, body)
- }
- xrefOff := off()
- fmt.Fprintf(&buf, "xref\n0 %d\n%010d %05d f \n", len(objs)+1, 0, 65535)
- for i := 1; i <= len(objs); i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprintf(&buf, "trailer\n<< /Size %d %s >>\nstartxref\n%d\n%%%%EOF\n",
- len(objs)+1, trailer, xrefOff)
- return buf.Bytes()
-}
-
-// buildPagesPDF builds a two-page document: page 1 inherits its MediaBox and
-// Resources from the /Pages root and carries a content stream; page 2 overrides
-// MediaBox, adds a CropBox and /Rotate, and has no content.
-func buildPagesPDF(t *testing.T) []byte {
- const content = "BT (Hi) Tj ET" // 13 bytes
- return buildObjPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 2 /Kids [ 3 0 R 4 0 R ] /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> >>",
- "<< /Type /Page /Parent 2 0 R /Contents 6 0 R >>",
- "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /CropBox [5 5 195 195] /Rotate 90 >>",
- "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
- fmt.Sprintf("<< /Length %d >>\nstream\n%s\nendstream", len(content), content),
- }, "/Root 1 0 R")
-}
-
-func TestReport(t *testing.T) {
- r, err := pdfdisassembler.Open(bytes.NewReader(buildPagesPDF(t)))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
-
- var out bytes.Buffer
- if err := report(&out, r); err != nil {
- t.Fatalf("report: %v", err)
- }
- got := out.String()
-
- wants := []string{
- "Pages: 2",
- "Page 1:",
- "MediaBox: 612 x 792 pt", // inherited from /Pages root
- "Resources: [Font]", // inherited
- "Content: 13 bytes decoded",
- "Page 2:",
- "MediaBox: 200 x 200 pt", // overridden locally
- "CropBox: [5 5 195 195]",
- "Rotation: 90",
- "Content: 0 bytes decoded", // no /Contents
- }
- for _, w := range wants {
- if !strings.Contains(got, w) {
- t.Errorf("output missing %q; got:\n%s", w, got)
- }
- }
-
- // Page 1 has no rotation, so no Rotation line should appear before "Page 2".
- page1 := got[strings.Index(got, "Page 1:"):strings.Index(got, "Page 2:")]
- if strings.Contains(page1, "Rotation:") {
- t.Errorf("page 1 should not print a Rotation line; got:\n%s", page1)
- }
-}
-
-// TestReportCyclicKidsTerminates feeds a page tree whose /Kids cycles back on
-// itself; report must return rather than recurse until the stack overflows.
-func TestReportCyclicKidsTerminates(t *testing.T) {
- data := buildObjPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Pages /Kids [ 2 0 R ] >>", // cycle back to obj 2
- }, "/Root 1 0 R")
- r, err := pdfdisassembler.Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
-
- var out bytes.Buffer
- if err := report(&out, r); err != nil {
- t.Fatalf("report: %v", err)
- }
- if got := out.String(); !strings.Contains(got, "Pages: 0") {
- t.Errorf("want \"Pages: 0\", got:\n%s", got)
- }
-}
diff --git a/examples/structtree/main.go b/examples/structtree/main.go
deleted file mode 100644
index fd2a6cc..0000000
--- a/examples/structtree/main.go
+++ /dev/null
@@ -1,100 +0,0 @@
-// Command structtree dumps the /StructTreeRoot of a PDF in indented form.
-// Intended as a starting point for accessibility-checker tooling.
-package main
-
-import (
- "fmt"
- "log"
- "os"
- "strings"
-
- "github.com/speedata/pdfdisassembler"
-)
-
-func main() {
- if len(os.Args) < 2 {
- fmt.Fprintln(os.Stderr, "usage: structtree ")
- os.Exit(2)
- }
- r, err := pdfdisassembler.OpenFile(os.Args[1])
- if err != nil {
- log.Fatal(err)
- }
- defer r.Close()
-
- cat, err := r.Catalog()
- if err != nil {
- log.Fatal(err)
- }
- root, ok := cat.Dict("StructTreeRoot")
- if !ok {
- fmt.Println("(no StructTreeRoot)")
- return
- }
-
- roleMap := map[string]string{}
- if rm, ok := root.Dict("RoleMap"); ok {
- for k, v := range rm.Iter() {
- if n, ok := v.(pdfdisassembler.Name); ok {
- roleMap[k] = string(n)
- }
- }
- }
-
- walk(r, root, roleMap, 0, map[pdfdisassembler.Reference]struct{}{})
-}
-
-// maxStructDepth bounds the recursion so a deeply nested or cyclic /K tree in a
-// hostile PDF can't overflow the stack; seen breaks reference cycles earlier.
-const maxStructDepth = 1000
-
-// visit reports whether ref is newly seen (false if already visited).
-func visit(seen map[pdfdisassembler.Reference]struct{}, ref pdfdisassembler.Reference) bool {
- if _, ok := seen[ref]; ok {
- return false
- }
- seen[ref] = struct{}{}
- return true
-}
-
-func walk(r *pdfdisassembler.Reader, node *pdfdisassembler.Dict, roleMap map[string]string, depth int, seen map[pdfdisassembler.Reference]struct{}) {
- if node == nil || depth > maxStructDepth {
- return
- }
- indent := strings.Repeat(" ", depth)
- typeName, _ := node.Name("S")
- if typeName == "" {
- typeName, _ = node.Name("Type")
- }
- role := string(typeName)
- if mapped, ok := roleMap[role]; ok {
- role = role + " -> " + mapped
- }
- fmt.Printf("%s%s\n", indent, role)
-
- k, ok := node.Get("K")
- if !ok {
- return
- }
- if ref, ok := k.(pdfdisassembler.Reference); ok {
- if !visit(seen, ref) {
- return
- }
- if v, err := r.Resolve(ref); err == nil {
- k = v
- }
- }
- switch t := k.(type) {
- case pdfdisassembler.Array:
- for _, child := range t {
- if ref, ok := child.(pdfdisassembler.Reference); ok && !visit(seen, ref) {
- continue
- }
- if d, err := r.ResolveDict(child); err == nil {
- walk(r, d, roleMap, depth+1, seen)
- }
- }
- case *pdfdisassembler.Dict:
- walk(r, t, roleMap, depth+1, seen)
- }
-}
diff --git a/examples/structtree/main_test.go b/examples/structtree/main_test.go
deleted file mode 100644
index 08a3eeb..0000000
--- a/examples/structtree/main_test.go
+++ /dev/null
@@ -1,86 +0,0 @@
-package main
-
-import (
- "bytes"
- "fmt"
- "io"
- "os"
- "strings"
- "testing"
-
- "github.com/speedata/pdfdisassembler"
-)
-
-// buildCyclicStructTreePDF builds a PDF whose /StructTreeRoot /K chain cycles
-// (obj 5's /K points back to obj 4).
-func buildCyclicStructTreePDF(t *testing.T) []byte {
- t.Helper()
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
- offsets := make([]int, 6) // 1..5
- obj := func(n int, body string) {
- offsets[n] = off()
- fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", n, body)
- }
- obj(1, "<< /Type /Catalog /Pages 2 0 R /StructTreeRoot 3 0 R >>")
- obj(2, "<< /Type /Pages /Count 0 /Kids [] >>")
- obj(3, "<< /Type /StructTreeRoot /K 4 0 R >>")
- obj(4, "<< /Type /StructElem /S /Document /K 5 0 R >>")
- obj(5, "<< /Type /StructElem /S /P /K 4 0 R >>") // cycle back to obj 4
-
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 6\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 5; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprintf(&buf, "trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-func captureStdout(t *testing.T, fn func()) string {
- t.Helper()
- old := os.Stdout
- rp, wp, err := os.Pipe()
- if err != nil {
- t.Fatalf("pipe: %v", err)
- }
- os.Stdout = wp
- done := make(chan string, 1)
- go func() {
- var b bytes.Buffer
- io.Copy(&b, rp)
- done <- b.String()
- }()
- fn()
- wp.Close()
- os.Stdout = old
- return <-done
-}
-
-func TestWalkCyclicStructTreeTerminates(t *testing.T) {
- r, err := pdfdisassembler.Open(bytes.NewReader(buildCyclicStructTreePDF(t)))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- cat, err := r.Catalog()
- if err != nil {
- t.Fatalf("Catalog: %v", err)
- }
- root, ok := cat.Dict("StructTreeRoot")
- if !ok {
- t.Fatal("no StructTreeRoot")
- }
- // Must return (the /K cycle would otherwise recurse until stack overflow);
- // the dump must show it descended the chain, proving the cycle is exercised.
- out := captureStdout(t, func() {
- walk(r, root, map[string]string{}, 0, map[pdfdisassembler.Reference]struct{}{})
- })
- for _, want := range []string{"Document", "P"} {
- if !strings.Contains(out, want) {
- t.Errorf("dump missing %q; got:\n%s", want, out)
- }
- }
-}
diff --git a/filter.go b/filter.go
deleted file mode 100644
index 14767de..0000000
--- a/filter.go
+++ /dev/null
@@ -1,161 +0,0 @@
-package pdfdisassembler
-
-import (
- "fmt"
-
- "github.com/speedata/pdfdisassembler/internal/filter"
-)
-
-// decodeStream is the Stream.Content backend. It reads the raw bytes from
-// the file, applies decryption if enabled, then runs the declared filter
-// chain.
-func (r *Reader) decodeStream(s *Stream) ([]byte, error) {
- raw, err := r.rawStreamSlice(s)
- if err != nil {
- return nil, err
- }
- return r.applyFilters(s, raw, true)
-}
-
-// rawStreamSlice returns the stream's raw bytes as a sub-slice of the file
-// buffer (no copy), after bounds-checking the declared extent.
-func (r *Reader) rawStreamSlice(s *Stream) ([]byte, error) {
- if s.rawOffset < 0 || s.rawOffset+s.rawLength > int64(len(r.buf)) {
- return nil, fmt.Errorf("pdfdisassembler: stream %d %d R: bytes out of range", s.objNumber, s.objGeneration)
- }
- return r.buf[s.rawOffset : s.rawOffset+s.rawLength], nil
-}
-
-// rawStreamBytes returns a copy of the stream's raw bytes (see Stream.RawBytes).
-func (r *Reader) rawStreamBytes(s *Stream) ([]byte, error) {
- raw, err := r.rawStreamSlice(s)
- if err != nil {
- return nil, err
- }
- out := make([]byte, len(raw))
- copy(out, raw)
- return out, nil
-}
-
-// applyFilters decrypts (if encrypted is true and an encryption context
-// exists) and runs the filter chain declared on the stream dict.
-func (r *Reader) applyFilters(s *Stream, raw []byte, encrypted bool) ([]byte, error) {
- data := raw
- if encrypted && r.encrypt != nil {
- // Cross-reference streams are themselves unencrypted; callers
- // must pass encrypted=false for those.
- dec, err := r.encrypt.decryptStream(data, s.objNumber, s.objGeneration)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: decrypt stream %d %d R: %w", s.objNumber, s.objGeneration, err)
- }
- data = dec
- }
-
- filters, params, err := r.streamFilterChain(s.Dict)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: stream %d %d R filter chain: %w", s.objNumber, s.objGeneration, err)
- }
- for i, name := range filters {
- // Skip image-only filters: return what we have and report.
- if filter.IsImageFilter(name) {
- return nil, fmt.Errorf("pdfdisassembler: stream %d %d R uses image-only filter %q (not decoded)", s.objNumber, s.objGeneration, name)
- }
- p := params[i]
- p.MaxOutput = r.MaxStreamSize
- out, err := filter.Decode(name, data, p)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: stream %d %d R filter %q: %w", s.objNumber, s.objGeneration, name, err)
- }
- data = out
- }
- return data, nil
-}
-
-// streamFilterChain returns the ordered filter names and per-filter params
-// for a stream dict. Both /Filter and /F (abbreviation) are accepted.
-func (r *Reader) streamFilterChain(d *Dict) ([]string, []filter.Params, error) {
- v, ok := d.Get("Filter")
- if !ok {
- v, ok = d.Get("F")
- }
- if !ok {
- return nil, nil, nil
- }
- v, err := r.Resolve(v)
- if err != nil {
- return nil, nil, err
- }
- var names []string
- switch t := v.(type) {
- case Name:
- names = []string{string(t)}
- case Array:
- for _, e := range t {
- e, err := r.Resolve(e)
- if err != nil {
- return nil, nil, err
- }
- n, ok := e.(Name)
- if !ok {
- return nil, nil, fmt.Errorf("/Filter entry is %T, want Name", e)
- }
- names = append(names, string(n))
- }
- default:
- return nil, nil, fmt.Errorf("/Filter is %T", v)
- }
-
- pv, _ := d.Get("DecodeParms")
- if pv == nil {
- pv, _ = d.Get("DP")
- }
- if pv != nil {
- pv, err = r.Resolve(pv)
- if err != nil {
- return nil, nil, err
- }
- }
- params := make([]filter.Params, len(names))
- switch t := pv.(type) {
- case nil, Null:
- // nothing
- case *Dict:
- if len(names) >= 1 {
- params[0] = paramsFromDict(t)
- }
- case Array:
- for i, e := range t {
- if i >= len(params) {
- break
- }
- e, err := r.Resolve(e)
- if err != nil {
- return nil, nil, err
- }
- if d, ok := e.(*Dict); ok {
- params[i] = paramsFromDict(d)
- }
- }
- }
- return names, params, nil
-}
-
-func paramsFromDict(d *Dict) filter.Params {
- var p filter.Params
- if n, ok := d.Int("Predictor"); ok {
- p.Predictor = int(n)
- }
- if n, ok := d.Int("Columns"); ok {
- p.Columns = int(n)
- }
- if n, ok := d.Int("Colors"); ok {
- p.Colors = int(n)
- }
- if n, ok := d.Int("BitsPerComponent"); ok {
- p.BitsPerComponent = int(n)
- }
- if n, ok := d.Int("EarlyChange"); ok && n == 0 {
- p.NoEarlyChange = true
- }
- return p
-}
diff --git a/filter_test.go b/filter_test.go
deleted file mode 100644
index 352a264..0000000
--- a/filter_test.go
+++ /dev/null
@@ -1,179 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "compress/lzw"
- "compress/zlib"
- "encoding/ascii85"
- "fmt"
- "testing"
-)
-
-// buildStreamObjectPDF wraps stream as indirect object 3 with the given stream
-// dict entries (e.g. "/Filter /LZWDecode ..."), reachable via a classical xref.
-func buildStreamObjectPDF(t *testing.T, dictEntries string, stream []byte) []byte {
- t.Helper()
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- buf.WriteString("%PDF-1.7\n%\xe2\xe3\xcf\xd3\n")
- offsets := make([]int, 4)
-
- offsets[1] = off()
- buf.WriteString("1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- offsets[2] = off()
- buf.WriteString("2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
- offsets[3] = off()
- fmt.Fprintf(&buf, "3 0 obj\n<< %s /Length %d >>\nstream\n", dictEntries, len(stream))
- buf.Write(stream)
- buf.WriteString("\nendstream\nendobj\n")
-
- xrefOff := off()
- buf.WriteString("xref\n0 4\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 3; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- buf.WriteString("trailer\n<< /Size 4 /Root 1 0 R >>\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-func streamObject3(t *testing.T, data []byte) []byte {
- t.Helper()
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- t.Cleanup(func() { r.Close() })
- v, err := r.Resolve(Reference{Number: 3, Generation: 0})
- if err != nil {
- t.Fatalf("Resolve: %v", err)
- }
- stm, ok := v.(*Stream)
- if !ok {
- t.Fatalf("object 3 is %T, want *Stream", v)
- }
- got, err := stm.Content()
- if err != nil {
- t.Fatalf("Content: %v", err)
- }
- return got
-}
-
-// A stream declaring /DecodeParms << /EarlyChange 0 >> must decode with early
-// change off. The stdlib LZW writer emits the non-early convention, so honouring
-// the parameter reproduces the input; ignoring it garbles a stream past the
-// first code-width boundary.
-func TestLZWStreamEarlyChangeZero(t *testing.T) {
- orig := make([]byte, 4096)
- x := uint32(99)
- for i := range orig {
- x = x*1664525 + 1013904223
- orig[i] = byte(x >> 24)
- }
- var enc bytes.Buffer
- w := lzw.NewWriter(&enc, lzw.MSB, 8)
- w.Write(orig)
- w.Close()
-
- got := streamObject3(t, buildStreamObjectPDF(t,
- "/Filter /LZWDecode /DecodeParms << /EarlyChange 0 >>", enc.Bytes()))
- if !bytes.Equal(got, orig) {
- t.Fatal("LZW stream with /EarlyChange 0 decoded incorrectly")
- }
-}
-
-// A /Filter array applies filters in order: the raw bytes are ASCII85 wrapping a
-// FlateDecode stream, so the chain must un-ASCII85 then inflate.
-func TestStreamFilterChainArray(t *testing.T) {
- orig := []byte("chained filters: ASCII85 over Flate over the original bytes")
- var fl bytes.Buffer
- zw := zlib.NewWriter(&fl)
- zw.Write(orig)
- zw.Close()
- a85 := make([]byte, ascii85.MaxEncodedLen(fl.Len()))
- n := ascii85.Encode(a85, fl.Bytes())
- stream := append(a85[:n:n], '~', '>')
-
- got := streamObject3(t, buildStreamObjectPDF(t,
- "/Filter [ /ASCII85Decode /FlateDecode ]", stream))
- if !bytes.Equal(got, orig) {
- t.Fatalf("chained decode = %q, want %q", got, orig)
- }
-}
-
-func TestParamsFromDict(t *testing.T) {
- d := newDict(nil)
- d.set("Predictor", Integer(12))
- d.set("Columns", Integer(5))
- d.set("Colors", Integer(3))
- d.set("BitsPerComponent", Integer(16))
- d.set("EarlyChange", Integer(0))
- p := paramsFromDict(d)
- if p.Predictor != 12 || p.Columns != 5 || p.Colors != 3 || p.BitsPerComponent != 16 {
- t.Errorf("predictor params = %+v, want Predictor=12 Columns=5 Colors=3 BitsPerComponent=16", p)
- }
- if !p.NoEarlyChange {
- t.Error("/EarlyChange 0 must set NoEarlyChange")
- }
-
- // 1 is the LZW default, so it must NOT set NoEarlyChange.
- d1 := newDict(nil)
- d1.set("EarlyChange", Integer(1))
- if paramsFromDict(d1).NoEarlyChange {
- t.Error("/EarlyChange 1 must not set NoEarlyChange")
- }
-
- empty := paramsFromDict(newDict(nil))
- if empty.Predictor != 0 || empty.Columns != 0 || empty.Colors != 0 ||
- empty.BitsPerComponent != 0 || empty.NoEarlyChange {
- t.Errorf("empty dict = %+v, want zero Params", empty)
- }
-}
-
-// /F and /DP are the /Filter and /DecodeParms abbreviations; a null /DP entry
-// leaves that filter with default params.
-func TestStreamFilterChainAbbreviations(t *testing.T) {
- orig := []byte("abbreviated filter keys decode the same")
- var fl bytes.Buffer
- zw := zlib.NewWriter(&fl)
- zw.Write(orig)
- zw.Close()
- a85 := make([]byte, ascii85.MaxEncodedLen(fl.Len()))
- n := ascii85.Encode(a85, fl.Bytes())
- stream := append(a85[:n:n], '~', '>')
-
- got := streamObject3(t, buildStreamObjectPDF(t,
- "/F [ /ASCII85Decode /FlateDecode ] /DP [ null null ]", stream))
- if !bytes.Equal(got, orig) {
- t.Fatalf("abbreviated /F+/DP decode = %q, want %q", got, orig)
- }
-}
-
-// A malformed or unsupported filter chain must surface an error from Content(),
-// not panic.
-func TestStreamFilterChainErrors(t *testing.T) {
- cases := map[string]string{
- "filter_wrong_type": "/Filter 42",
- "filter_array_bad_entry": "/Filter [ /FlateDecode 42 ]",
- "image_only_filter": "/Filter /DCTDecode",
- "undecodable_data": "/Filter /FlateDecode",
- }
- for name, dictEntries := range cases {
- t.Run(name, func(t *testing.T) {
- data := buildStreamObjectPDF(t, dictEntries, []byte("not valid filtered data"))
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- v, err := r.Resolve(Reference{Number: 3, Generation: 0})
- if err != nil {
- t.Fatalf("Resolve: %v", err)
- }
- if _, err := v.(*Stream).Content(); err == nil {
- t.Fatal("expected a Content() error, got nil")
- }
- })
- }
-}
diff --git a/fixtures_test.go b/fixtures_test.go
deleted file mode 100644
index 197f98c..0000000
--- a/fixtures_test.go
+++ /dev/null
@@ -1,109 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "flag"
- "fmt"
- "os"
- "path/filepath"
- "strings"
- "testing"
-)
-
-// updateGoldens regenerates fixture golden.json files instead of comparing.
-// Run as: go test -update -run TestFixtures.
-var updateGoldens = flag.Bool("update", false, "regenerate fixture golden.json files")
-
-// TestFixtures iterates every directory under testdata/fixtures, opens
-// its input.pdf, and compares Dump output against the committed
-// golden.json. A test failure prints the first few differing lines and
-// the command to regenerate the golden.
-//
-// Adding a fixture:
-//
-// 1. Create testdata/fixtures//input.pdf (real-world or generated)
-// 2. go test -update -run TestFixtures/
-// 3. Inspect golden.json — does it match what the spec says should happen?
-// 4. Commit input.pdf, golden.json, and (optionally) README.md describing
-// what the fixture proves.
-func TestFixtures(t *testing.T) {
- root := "testdata/fixtures"
- entries, err := os.ReadDir(root)
- if err != nil {
- t.Skipf("no testdata/fixtures: %v", err)
- return
- }
- for _, e := range entries {
- if !e.IsDir() {
- continue
- }
- name := e.Name()
- t.Run(name, func(t *testing.T) {
- runFixture(t, filepath.Join(root, name))
- })
- }
-}
-
-func runFixture(t *testing.T, dir string) {
- t.Helper()
- inputPath := filepath.Join(dir, "input.pdf")
- goldenPath := filepath.Join(dir, "golden.json")
-
- r, err := OpenFile(inputPath)
- if err != nil {
- t.Fatalf("Open %s: %v", inputPath, err)
- }
- defer r.Close()
-
- got, err := Dump(r, DumpOptions{})
- if err != nil {
- t.Fatalf("Dump: %v", err)
- }
-
- if *updateGoldens {
- if err := os.WriteFile(goldenPath, got, 0o644); err != nil {
- t.Fatalf("write golden: %v", err)
- }
- t.Logf("updated %s (%d bytes)", goldenPath, len(got))
- return
- }
-
- want, err := os.ReadFile(goldenPath)
- if err != nil {
- t.Fatalf("read golden %s: %v\n\tregenerate with: go test -update -run %s",
- goldenPath, err, t.Name())
- }
- if bytes.Equal(got, want) {
- return
- }
- t.Errorf("dump mismatch\n regenerate: go test -update -run %s\n first diffs:\n%s",
- t.Name(), firstDiffLines(want, got, 10))
-}
-
-// firstDiffLines returns at most maxLines of unified-style "-want / +got"
-// hints around line-level differences.
-func firstDiffLines(want, got []byte, maxLines int) string {
- wl := strings.Split(string(want), "\n")
- gl := strings.Split(string(got), "\n")
- max := len(wl)
- if len(gl) > max {
- max = len(gl)
- }
- var out strings.Builder
- shown := 0
- for i := 0; i < max && shown < maxLines; i++ {
- var w, g string
- if i < len(wl) {
- w = wl[i]
- }
- if i < len(gl) {
- g = gl[i]
- }
- if w == g {
- continue
- }
- fmt.Fprintf(&out, " line %d:\n - %s\n + %s\n", i+1, w, g)
- shown++
- }
- return out.String()
-}
diff --git a/go.mod b/go.mod
deleted file mode 100644
index 5271bc6..0000000
--- a/go.mod
+++ /dev/null
@@ -1,5 +0,0 @@
-module github.com/speedata/pdfdisassembler
-
-go 1.23
-
-require github.com/andybalholm/brotli v1.2.2
diff --git a/go.sum b/go.sum
deleted file mode 100644
index 80d4b3a..0000000
--- a/go.sum
+++ /dev/null
@@ -1,4 +0,0 @@
-github.com/andybalholm/brotli v1.2.2 h1:HzTuoo2ErYQqf5qvcJInB8uvqSVxRttzkFexPWtnceM=
-github.com/andybalholm/brotli v1.2.2/go.mod h1:rzTDkvFWvIrjDXZHkuS16NPggd91W3kUSvPlQ1pLaKY=
-github.com/xyproto/randomstring v1.0.5 h1:YtlWPoRdgMu3NZtP45drfy1GKoojuR7hmRcnhZqKjWU=
-github.com/xyproto/randomstring v1.0.5/go.mod h1:rgmS5DeNXLivK7YprL0pY+lTuhNQW3iGxZ18UQApw/E=
diff --git a/internal/crypt/crypt.go b/internal/crypt/crypt.go
deleted file mode 100644
index 595815a..0000000
--- a/internal/crypt/crypt.go
+++ /dev/null
@@ -1,470 +0,0 @@
-// Package crypt implements the PDF /Standard security handler for
-// versions V2 (RC4), V4 (RC4 or AES-128) and V5 (AES-256, PDF 1.7
-// Extension 3 and PDF 2.0).
-//
-// Only password-based access (the user password "empty string" path
-// included) is supported. Public-key encryption (/Adobe.PubSec) is
-// out of scope.
-package crypt
-
-import (
- "bytes"
- "crypto/aes"
- "crypto/cipher"
- "crypto/md5"
- "crypto/rc4"
- "crypto/sha256"
- "crypto/sha512"
- "errors"
- "fmt"
-)
-
-// Algorithm identifies a stream/string cipher.
-type Algorithm int
-
-const (
- AlgRC4 Algorithm = iota + 1
- AlgAES128
- AlgAES256
- AlgIdentity
-)
-
-// Handler holds the state needed to decrypt strings and streams in a PDF
-// once a password has been validated.
-type Handler struct {
- V int // /V version
- R int // /R revision
- Length int // key length in bits (V2/V4)
- FileKey []byte // file encryption key (computed from password)
- StringAlg Algorithm // algorithm for strings
- StreamAlg Algorithm // algorithm for streams
- EmbedAlg Algorithm // algorithm for embedded files
- CryptFilts map[string]filterDef
- StmF string // default stream filter (V4)
- StrF string // default string filter (V4)
- EFF string // embedded-file filter (V4)
-}
-
-type filterDef struct {
- CFM Algorithm
-}
-
-// Params is the inputs needed to instantiate a Handler from the PDF
-// /Encrypt dict.
-type Params struct {
- V int
- R int
- Length int
- P int32
- OwnerEntry []byte // /O
- UserEntry []byte // /U
- OE []byte // /OE (V5)
- UE []byte // /UE (V5)
- Perms []byte // /Perms (V5)
- ID0 []byte // first element of /ID
- EncryptMeta bool
- StmF string
- StrF string
- EFF string
- CryptFilters map[string]string // name → CFM
-}
-
-// New tries to instantiate a Handler given the encryption parameters and
-// the user password (empty string is the most common case).
-func New(p Params, password []byte) (*Handler, error) {
- h := &Handler{
- V: p.V,
- R: p.R,
- Length: p.Length,
- StmF: p.StmF,
- StrF: p.StrF,
- EFF: p.EFF,
- CryptFilts: map[string]filterDef{},
- }
- for name, cfm := range p.CryptFilters {
- alg, err := algFromCFM(cfm)
- if err != nil {
- return nil, err
- }
- h.CryptFilts[name] = filterDef{CFM: alg}
- }
-
- switch p.V {
- case 1, 2:
- h.StringAlg = AlgRC4
- h.StreamAlg = AlgRC4
- h.EmbedAlg = AlgRC4
- key, err := computeRC4Key(p, password)
- if err != nil {
- return nil, err
- }
- h.FileKey = key
- case 4:
- // V4 introduces /CF, /StmF, /StrF for per-stream/string cipher.
- h.StringAlg = h.algFor(p.StrF)
- h.StreamAlg = h.algFor(p.StmF)
- h.EmbedAlg = h.algFor(p.EFF)
- key, err := computeRC4Key(p, password)
- if err != nil {
- return nil, err
- }
- h.FileKey = key
- case 5:
- h.StringAlg = AlgAES256
- h.StreamAlg = AlgAES256
- h.EmbedAlg = AlgAES256
- key, err := computeV5Key(p, password)
- if err != nil {
- return nil, err
- }
- h.FileKey = key
- default:
- return nil, fmt.Errorf("crypt: unsupported /V %d", p.V)
- }
- return h, nil
-}
-
-func (h *Handler) algFor(filterName string) Algorithm {
- if filterName == "" || filterName == "Identity" {
- return AlgIdentity
- }
- if def, ok := h.CryptFilts[filterName]; ok {
- return def.CFM
- }
- return AlgIdentity
-}
-
-func algFromCFM(cfm string) (Algorithm, error) {
- switch cfm {
- case "V2":
- return AlgRC4, nil
- case "AESV2":
- return AlgAES128, nil
- case "AESV3":
- return AlgAES256, nil
- case "None":
- return AlgIdentity, nil
- }
- return 0, fmt.Errorf("crypt: unknown /CFM %q", cfm)
-}
-
-// DecryptString decrypts a string using the configured string algorithm
-// and object identity.
-func (h *Handler) DecryptString(data []byte, objNum, objGen int) ([]byte, error) {
- return h.decrypt(data, objNum, objGen, h.StringAlg)
-}
-
-// DecryptStream decrypts a stream. cryptFilterName, if non-empty, overrides
-// the default stream algorithm (V4 streams can carry an inline /Filter
-// chain containing /Crypt with parameters).
-func (h *Handler) DecryptStream(data []byte, objNum, objGen int, cryptFilterName string) ([]byte, error) {
- alg := h.StreamAlg
- if cryptFilterName != "" {
- alg = h.algFor(cryptFilterName)
- }
- return h.decrypt(data, objNum, objGen, alg)
-}
-
-func (h *Handler) decrypt(data []byte, objNum, objGen int, alg Algorithm) ([]byte, error) {
- switch alg {
- case AlgIdentity:
- return data, nil
- case AlgRC4:
- key := h.objKeyRC4orAES(objNum, objGen, false)
- out := make([]byte, len(data))
- c, _ := rc4.NewCipher(key)
- c.XORKeyStream(out, data)
- return out, nil
- case AlgAES128:
- key := h.objKeyRC4orAES(objNum, objGen, true)
- return aesCBCDecrypt(key, data)
- case AlgAES256:
- return aesCBCDecrypt(h.FileKey, data)
- }
- return nil, fmt.Errorf("crypt: unknown algorithm %d", alg)
-}
-
-// objKeyRC4orAES derives the per-object encryption key for V2/V4.
-//
-// PDF 32000-1:2008 §7.6.2: object key = MD5(fileKey || lo3(objNum) ||
-// lo2(objGen) || (for AES) "sAlT"). Truncate to min(len(fileKey)+5, 16).
-func (h *Handler) objKeyRC4orAES(objNum, objGen int, aes bool) []byte {
- buf := make([]byte, 0, len(h.FileKey)+9)
- buf = append(buf, h.FileKey...)
- buf = append(buf,
- byte(objNum),
- byte(objNum>>8),
- byte(objNum>>16),
- byte(objGen),
- byte(objGen>>8),
- )
- if aes {
- buf = append(buf, 's', 'A', 'l', 'T')
- }
- sum := md5.Sum(buf)
- n := len(h.FileKey) + 5
- if n > 16 {
- n = 16
- }
- return sum[:n]
-}
-
-// aesCBCDecrypt unwraps AES/CBC/PKCS#7 with a 16-byte IV prepended.
-func aesCBCDecrypt(key, data []byte) ([]byte, error) {
- if len(data) < aes.BlockSize {
- return nil, errors.New("crypt: AES data shorter than IV")
- }
- iv := data[:aes.BlockSize]
- body := data[aes.BlockSize:]
- if len(body)%aes.BlockSize != 0 {
- return nil, errors.New("crypt: AES body not block-aligned")
- }
- block, err := aes.NewCipher(key)
- if err != nil {
- return nil, err
- }
- mode := cipher.NewCBCDecrypter(block, iv)
- out := make([]byte, len(body))
- mode.CryptBlocks(out, body)
- // Strip PKCS#7 padding.
- if len(out) == 0 {
- return out, nil
- }
- pad := int(out[len(out)-1])
- if pad < 1 || pad > aes.BlockSize {
- return out, nil // tolerate broken padding
- }
- if pad > len(out) {
- return out, nil
- }
- return out[:len(out)-pad], nil
-}
-
-// computeRC4Key implements PDF 32000-1:2008 algorithm 2 for the file key
-// (V1/V2/V4 with RC4 or AESV2). The user password is the input; the empty
-// string is the default.
-func computeRC4Key(p Params, password []byte) ([]byte, error) {
- pad := padPassword(password)
- h := md5.New()
- h.Write(pad)
- h.Write(p.OwnerEntry)
- pBytes := []byte{
- byte(uint32(p.P)),
- byte(uint32(p.P) >> 8),
- byte(uint32(p.P) >> 16),
- byte(uint32(p.P) >> 24),
- }
- h.Write(pBytes)
- h.Write(p.ID0)
- if p.R >= 4 && !p.EncryptMeta {
- h.Write([]byte{0xff, 0xff, 0xff, 0xff})
- }
- sum := h.Sum(nil)
- keyLen := p.Length / 8
- if keyLen == 0 {
- keyLen = 5 // V1 default
- }
- // /Length is attacker-controlled; the key is sliced from a 16-byte MD5
- // digest, so anything outside [1, md5.Size] would slice/make out of range.
- if keyLen < 1 || keyLen > md5.Size {
- return nil, fmt.Errorf("crypt: invalid key length %d bits", p.Length)
- }
- if p.R >= 3 {
- for i := 0; i < 50; i++ {
- s := md5.Sum(sum[:keyLen])
- sum = s[:]
- }
- }
- key := make([]byte, keyLen)
- copy(key, sum[:keyLen])
-
- // Validate password by computing U and comparing.
- uExpected, err := computeU(p, key)
- if err != nil {
- return nil, err
- }
- if !validU(uExpected, p.UserEntry, p.R) {
- return nil, errors.New("crypt: password incorrect (V2/V4)")
- }
- return key, nil
-}
-
-var passPad = []byte{
- 0x28, 0xbf, 0x4e, 0x5e, 0x4e, 0x75, 0x8a, 0x41,
- 0x64, 0x00, 0x4e, 0x56, 0xff, 0xfa, 0x01, 0x08,
- 0x2e, 0x2e, 0x00, 0xb6, 0xd0, 0x68, 0x3e, 0x80,
- 0x2f, 0x0c, 0xa9, 0xfe, 0x64, 0x53, 0x69, 0x7a,
-}
-
-func padPassword(p []byte) []byte {
- if len(p) >= 32 {
- return p[:32]
- }
- out := make([]byte, 32)
- copy(out, p)
- copy(out[len(p):], passPad)
- return out
-}
-
-func computeU(p Params, key []byte) ([]byte, error) {
- if p.R == 2 {
- out := make([]byte, 32)
- c, _ := rc4.NewCipher(key)
- c.XORKeyStream(out, passPad)
- return out, nil
- }
- // R >= 3.
- h := md5.New()
- h.Write(passPad)
- h.Write(p.ID0)
- digest := h.Sum(nil)
- out := make([]byte, 16)
- c, _ := rc4.NewCipher(key)
- c.XORKeyStream(out, digest)
- for i := 1; i <= 19; i++ {
- tweaked := make([]byte, len(key))
- for j, b := range key {
- tweaked[j] = b ^ byte(i)
- }
- c2, _ := rc4.NewCipher(tweaked)
- c2.XORKeyStream(out, out)
- }
- final := make([]byte, 32)
- copy(final, out)
- // Trailing bytes are arbitrary per spec; pad with zeros.
- return final, nil
-}
-
-func validU(expected, actual []byte, r int) bool {
- if r == 2 {
- return bytes.Equal(expected, actual)
- }
- if len(actual) < 16 {
- return false
- }
- return bytes.Equal(expected[:16], actual[:16])
-}
-
-// computeV5Key implements PDF 32000-2:2020 §7.6.4 / PDF 1.7 Extension 3
-// §3.5.2 — the AES-256 key derivation.
-func computeV5Key(p Params, password []byte) ([]byte, error) {
- if len(p.UserEntry) < 48 || len(p.OwnerEntry) < 48 {
- return nil, errors.New("crypt: V5 entries too short")
- }
- if len(p.UE) < 32 || len(p.OE) < 32 {
- return nil, errors.New("crypt: V5 /UE or /OE missing")
- }
- // Limit password length to 127 bytes per spec.
- if len(password) > 127 {
- password = password[:127]
- }
- uValHash := p.UserEntry[:32]
- uVS := p.UserEntry[32:40]
- uKS := p.UserEntry[40:48]
- oValHash := p.OwnerEntry[:32]
- oVS := p.OwnerEntry[32:40]
- oKS := p.OwnerEntry[40:48]
- _ = uValHash
- _ = oValHash
-
- // Try user password first.
- if hash, err := v5Hash(password, uVS, nil, p.R); err == nil && bytes.Equal(hash, uValHash) {
- kHash, err := v5Hash(password, uKS, nil, p.R)
- if err != nil {
- return nil, err
- }
- return v5DecryptKey(kHash, p.UE)
- }
- // Try owner password.
- if hash, err := v5Hash(password, oVS, p.UserEntry[:48], p.R); err == nil && bytes.Equal(hash, oValHash) {
- kHash, err := v5Hash(password, oKS, p.UserEntry[:48], p.R)
- if err != nil {
- return nil, err
- }
- return v5DecryptKey(kHash, p.OE)
- }
- return nil, errors.New("crypt: password incorrect (V5)")
-}
-
-func v5DecryptKey(kHash, encryptedKey []byte) ([]byte, error) {
- if len(kHash) != 32 || len(encryptedKey) != 32 {
- return nil, errors.New("crypt: V5 key derivation: wrong sizes")
- }
- block, err := aes.NewCipher(kHash)
- if err != nil {
- return nil, err
- }
- iv := make([]byte, aes.BlockSize)
- mode := cipher.NewCBCDecrypter(block, iv)
- out := make([]byte, 32)
- mode.CryptBlocks(out, encryptedKey)
- return out, nil
-}
-
-// v5Hash implements the PDF 2.0 / R=6 password hashing function. For R=5
-// (PDF 1.7 Ext.3) it's just SHA-256.
-func v5Hash(password, salt, userKey []byte, R int) ([]byte, error) {
- switch R {
- case 5:
- h := sha256.New()
- h.Write(password)
- h.Write(salt)
- h.Write(userKey)
- return h.Sum(nil), nil
- case 6:
- return r6Hash(password, salt, userKey)
- }
- return nil, fmt.Errorf("crypt: unsupported R=%d", R)
-}
-
-// r6Hash is the iterated AES-128 hash from PDF 2.0 §7.6.4.3.4.
-func r6Hash(password, salt, userKey []byte) ([]byte, error) {
- h := sha256.New()
- h.Write(password)
- h.Write(salt)
- h.Write(userKey)
- K := h.Sum(nil)
- round := 0
- for {
- K1 := make([]byte, 0, 64*(len(password)+len(K)+len(userKey)))
- for i := 0; i < 64; i++ {
- K1 = append(K1, password...)
- K1 = append(K1, K...)
- K1 = append(K1, userKey...)
- }
- if len(K) < 32 {
- return nil, errors.New("r6Hash: short K")
- }
- block, err := aes.NewCipher(K[:16])
- if err != nil {
- return nil, err
- }
- mode := cipher.NewCBCEncrypter(block, K[16:32])
- E := make([]byte, len(K1))
- mode.CryptBlocks(E, K1)
-
- // Treat first 16 bytes as big-endian int and take mod 3.
- sum := 0
- for i := 0; i < 16; i++ {
- sum = (sum*256 + int(E[i])) % 3
- }
- switch sum {
- case 0:
- s := sha256.Sum256(E)
- K = s[:]
- case 1:
- s := sha512.Sum384(E)
- K = s[:]
- case 2:
- s := sha512.Sum512(E)
- K = s[:]
- }
- round++
- if round >= 64 && int(E[len(E)-1]) <= round-32 {
- return K[:32], nil
- }
- if round > 1000 {
- return nil, errors.New("r6Hash: too many rounds")
- }
- }
-}
diff --git a/internal/crypt/crypt_test.go b/internal/crypt/crypt_test.go
deleted file mode 100644
index bbcfdcd..0000000
--- a/internal/crypt/crypt_test.go
+++ /dev/null
@@ -1,433 +0,0 @@
-package crypt
-
-import (
- "bytes"
- "crypto/aes"
- "crypto/cipher"
- "crypto/md5"
- "testing"
-)
-
-// New must reject an /Encrypt /Length whose derived key size (Length/8) falls
-// outside [1, 16] — the RC4/AESV2 file key is sliced from a 16-byte MD5 digest,
-// so a hostile large or negative /Length would slice out of range and panic.
-func TestNewRejectsHostileKeyLength(t *testing.T) {
- for _, length := range []int{136, 256, 4096, -8} {
- base := Params{
- V: 2,
- R: 3,
- Length: length,
- OwnerEntry: make([]byte, 32),
- UserEntry: make([]byte, 32),
- ID0: make([]byte, 16),
- }
- if _, err := New(base, nil); err == nil {
- t.Fatalf("Length=%d: expected error, got nil", length)
- }
- }
-}
-
-// fixedIV is a deterministic 16-byte IV for reproducible AES test vectors.
-var fixedIV = []byte{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15}
-
-// aesCBCEncryptStream is the inverse of aesCBCDecrypt: PKCS#7-pad, CBC-encrypt,
-// and prepend the IV, producing a blob the handler should decrypt back.
-func aesCBCEncryptStream(t *testing.T, key, iv, plaintext []byte) []byte {
- t.Helper()
- padLen := aes.BlockSize - len(plaintext)%aes.BlockSize
- padded := append(append([]byte{}, plaintext...), bytes.Repeat([]byte{byte(padLen)}, padLen)...)
- block, err := aes.NewCipher(key)
- if err != nil {
- t.Fatalf("aes.NewCipher: %v", err)
- }
- body := make([]byte, len(padded))
- cipher.NewCBCEncrypter(block, iv).CryptBlocks(body, padded)
- return append(append([]byte{}, iv...), body...)
-}
-
-func TestDecryptRC4RoundTrip(t *testing.T) {
- h := &Handler{FileKey: bytes.Repeat([]byte{0x33}, 16), StreamAlg: AlgRC4, StringAlg: AlgRC4}
- plaintext := []byte("the quick brown fox / RC4")
- // RC4 is symmetric: decrypting plaintext yields ciphertext.
- ct, err := h.DecryptStream(plaintext, 12, 0, "")
- if err != nil {
- t.Fatalf("encrypt: %v", err)
- }
- if bytes.Equal(ct, plaintext) {
- t.Fatal("ciphertext equals plaintext")
- }
- got, err := h.DecryptStream(ct, 12, 0, "")
- if err != nil {
- t.Fatalf("decrypt: %v", err)
- }
- if !bytes.Equal(got, plaintext) {
- t.Fatalf("round-trip mismatch: %q", got)
- }
- // Per-object keying: the same bytes under a different object number must
- // not decrypt to the plaintext.
- if other, _ := h.DecryptStream(ct, 99, 0, ""); bytes.Equal(other, plaintext) {
- t.Fatal("ciphertext decrypted under wrong object number")
- }
-}
-
-func TestDecryptAES128RoundTrip(t *testing.T) {
- h := &Handler{FileKey: bytes.Repeat([]byte{0x11}, 16), StreamAlg: AlgAES128, StringAlg: AlgAES128}
- plaintext := []byte("attachment bytes under AESV2")
- key := h.objKeyRC4orAES(7, 0, true)
- ct := aesCBCEncryptStream(t, key, fixedIV, plaintext)
- got, err := h.DecryptStream(ct, 7, 0, "")
- if err != nil {
- t.Fatalf("decrypt: %v", err)
- }
- if !bytes.Equal(got, plaintext) {
- t.Fatalf("round-trip mismatch: %q", got)
- }
-}
-
-func TestDecryptAES256RoundTrip(t *testing.T) {
- // V5/AESV3 keys streams directly with the file key (no per-object key).
- h := &Handler{FileKey: bytes.Repeat([]byte{0x22}, 32), StreamAlg: AlgAES256, StringAlg: AlgAES256}
- plaintext := []byte("AES-256 stream content for V5")
- ct := aesCBCEncryptStream(t, h.FileKey, fixedIV, plaintext)
- got, err := h.DecryptString(ct, 5, 0)
- if err != nil {
- t.Fatalf("decrypt: %v", err)
- }
- if !bytes.Equal(got, plaintext) {
- t.Fatalf("round-trip mismatch: %q", got)
- }
-}
-
-// Attacker-supplied AES blobs (too short for the IV, not block-aligned, empty)
-// must surface an error or empty output — never panic.
-func TestDecryptAESMalformedNoPanic(t *testing.T) {
- h := &Handler{FileKey: bytes.Repeat([]byte{0x11}, 16), StreamAlg: AlgAES128}
- cases := map[string][]byte{
- "empty": {},
- "shorter_than_iv": make([]byte, aes.BlockSize-1),
- "iv_only": make([]byte, aes.BlockSize),
- "unaligned_body": make([]byte, aes.BlockSize+aes.BlockSize-1),
- "one_byte": {0x00},
- }
- for name, data := range cases {
- t.Run(name, func(t *testing.T) {
- // Must not panic; result is ignored, the point is robustness.
- _, _ = h.DecryptStream(data, 1, 0, "")
- })
- }
-}
-
-func TestDecryptIdentityPassthrough(t *testing.T) {
- h := &Handler{StreamAlg: AlgIdentity, StringAlg: AlgIdentity}
- data := []byte{0xde, 0xad, 0xbe, 0xef}
- got, err := h.DecryptStream(data, 1, 0, "")
- if err != nil {
- t.Fatalf("identity: %v", err)
- }
- if !bytes.Equal(got, data) {
- t.Fatal("identity altered data")
- }
-}
-
-// deriveRC4Key mirrors computeRC4Key's derivation (without the /U validation),
-// so a test can compute the matching /U for an empty-password fixture.
-func deriveRC4Key(p Params, password []byte) []byte {
- pad := padPassword(password)
- h := md5.New()
- h.Write(pad)
- h.Write(p.OwnerEntry)
- h.Write([]byte{
- byte(uint32(p.P)), byte(uint32(p.P) >> 8),
- byte(uint32(p.P) >> 16), byte(uint32(p.P) >> 24),
- })
- h.Write(p.ID0)
- if p.R >= 4 && !p.EncryptMeta {
- h.Write([]byte{0xff, 0xff, 0xff, 0xff})
- }
- sum := h.Sum(nil)
- keyLen := p.Length / 8
- if keyLen == 0 {
- keyLen = 5
- }
- if p.R >= 3 {
- for i := 0; i < 50; i++ {
- s := md5.Sum(sum[:keyLen])
- sum = s[:]
- }
- }
- key := make([]byte, keyLen)
- copy(key, sum[:keyLen])
- return key
-}
-
-// New must reconstruct the V2/V4 file key from a correct empty-password /U.
-func TestNewV2V4KeyDerivationRoundTrip(t *testing.T) {
- cases := []struct {
- name string
- V, R, bits int
- }{
- {"V2R2", 2, 2, 40},
- {"V2R3", 2, 3, 128},
- {"V4R4", 4, 4, 128},
- }
- for _, tc := range cases {
- t.Run(tc.name, func(t *testing.T) {
- password := []byte{}
- p := Params{
- V: tc.V, R: tc.R, Length: tc.bits,
- OwnerEntry: bytes.Repeat([]byte{0x5a}, 32),
- ID0: bytes.Repeat([]byte{0x7c}, 16),
- P: -3904,
- EncryptMeta: true,
- StmF: "StdCF", StrF: "StdCF",
- CryptFilters: map[string]string{"StdCF": "V2"},
- }
- key := deriveRC4Key(p, password)
- u, err := computeU(p, key)
- if err != nil {
- t.Fatalf("computeU: %v", err)
- }
- p.UserEntry = u
- h, err := New(p, password)
- if err != nil {
- t.Fatalf("New: %v", err)
- }
- if !bytes.Equal(h.FileKey, key) {
- t.Fatalf("file key mismatch:\n got %x\nwant %x", h.FileKey, key)
- }
- })
- }
-}
-
-// A wrong /U must be rejected, not accepted with a garbage key.
-func TestNewV2RejectsWrongUserEntry(t *testing.T) {
- p := Params{
- V: 2, R: 3, Length: 128,
- OwnerEntry: bytes.Repeat([]byte{0x5a}, 32),
- ID0: bytes.Repeat([]byte{0x7c}, 16),
- UserEntry: bytes.Repeat([]byte{0x00}, 32), // not the real /U
- }
- if _, err := New(p, []byte{}); err == nil {
- t.Fatal("expected password-incorrect error, got nil")
- }
-}
-
-// aesCBCEncryptRaw is the inverse of v5DecryptKey: CBC-encrypt block-aligned
-// data with a zero-prepend-free layout.
-func aesCBCEncryptRaw(t *testing.T, key, iv, plaintext []byte) []byte {
- t.Helper()
- block, err := aes.NewCipher(key)
- if err != nil {
- t.Fatalf("aes.NewCipher: %v", err)
- }
- out := make([]byte, len(plaintext))
- cipher.NewCBCEncrypter(block, iv).CryptBlocks(out, plaintext)
- return out
-}
-
-// computeV5Key must recover the AES-256 file key from a correct empty-password
-// /U, /UE for both R=5 (SHA-256) and R=6 (the iterated r6Hash).
-func TestComputeV5KeyRoundTrip(t *testing.T) {
- for _, R := range []int{5, 6} {
- t.Run(map[int]string{5: "R5", 6: "R6"}[R], func(t *testing.T) {
- password := []byte("user-pw")
- fileKey := bytes.Repeat([]byte{0x42}, 32)
- uVS := bytes.Repeat([]byte{0x01}, 8)
- uKS := bytes.Repeat([]byte{0x02}, 8)
-
- uValHash, err := v5Hash(password, uVS, nil, R)
- if err != nil {
- t.Fatalf("v5Hash(validation): %v", err)
- }
- kHash, err := v5Hash(password, uKS, nil, R)
- if err != nil {
- t.Fatalf("v5Hash(key): %v", err)
- }
- ue := aesCBCEncryptRaw(t, kHash, make([]byte, aes.BlockSize), fileKey)
-
- userEntry := append(append(append([]byte{}, uValHash...), uVS...), uKS...)
- p := Params{
- V: 5, R: R,
- UserEntry: userEntry, // 48 bytes
- OwnerEntry: make([]byte, 48), // present but unused (user path matches first)
- UE: ue, // 32 bytes
- OE: make([]byte, 32),
- }
- key, err := computeV5Key(p, password)
- if err != nil {
- t.Fatalf("computeV5Key: %v", err)
- }
- if !bytes.Equal(key, fileKey) {
- t.Fatalf("V5 key mismatch:\n got %x\nwant %x", key, fileKey)
- }
- })
- }
-}
-
-// New must map each V4 /CF crypt-filter method to a cipher and reject unknown
-// ones, for a valid empty-password setup.
-func TestNewV4CryptFilterMethods(t *testing.T) {
- cases := []struct {
- cfm string
- wantErr bool
- }{
- {"V2", false}, {"AESV2", false}, {"AESV3", false}, {"None", false}, {"Bogus", true},
- }
- for _, tc := range cases {
- t.Run(tc.cfm, func(t *testing.T) {
- p := Params{
- V: 4, R: 4, Length: 128,
- OwnerEntry: bytes.Repeat([]byte{0x5a}, 32),
- ID0: bytes.Repeat([]byte{0x7c}, 16),
- P: -3904,
- EncryptMeta: true,
- StmF: "StdCF", StrF: "StdCF",
- CryptFilters: map[string]string{"StdCF": tc.cfm},
- }
- key := deriveRC4Key(p, nil)
- u, err := computeU(p, key)
- if err != nil {
- t.Fatalf("computeU: %v", err)
- }
- p.UserEntry = u
- _, err = New(p, nil)
- if tc.wantErr != (err != nil) {
- t.Fatalf("CFM %q: wantErr=%v, got err=%v", tc.cfm, tc.wantErr, err)
- }
- })
- }
-}
-
-// computeV5Key must reject short /U, /O, /UE, /OE entries rather than slicing
-// out of range.
-func TestComputeV5KeyRejectsShortEntries(t *testing.T) {
- cases := map[string]Params{
- "short_user": {V: 5, R: 6, UserEntry: make([]byte, 47), OwnerEntry: make([]byte, 48), UE: make([]byte, 32), OE: make([]byte, 32)},
- "short_owner": {V: 5, R: 6, UserEntry: make([]byte, 48), OwnerEntry: make([]byte, 47), UE: make([]byte, 32), OE: make([]byte, 32)},
- "short_ue": {V: 5, R: 6, UserEntry: make([]byte, 48), OwnerEntry: make([]byte, 48), UE: make([]byte, 31), OE: make([]byte, 32)},
- "short_oe": {V: 5, R: 6, UserEntry: make([]byte, 48), OwnerEntry: make([]byte, 48), UE: make([]byte, 32), OE: make([]byte, 31)},
- }
- for name, p := range cases {
- t.Run(name, func(t *testing.T) {
- if _, err := computeV5Key(p, []byte{}); err == nil {
- t.Fatal("expected error for short entry, got nil")
- }
- })
- }
-}
-
-// Owner-password path. Unlike the user path, the owner hash mixes in the
-// 48-byte /U entry (§7.6.4.4.10) — so userEntry here must be well-formed.
-func TestComputeV5KeyOwnerPath(t *testing.T) {
- const R = 6
- password := []byte("owner-pw")
- fileKey := bytes.Repeat([]byte{0x37}, 32)
-
- // All-zero validation hash (first 32 bytes) can never equal v5Hash output,
- // forcing the user path to miss so the owner path is taken.
- uVS := bytes.Repeat([]byte{0x11}, 8)
- uKS := bytes.Repeat([]byte{0x22}, 8)
- userEntry := append(append(make([]byte, 32), uVS...), uKS...) // 48 bytes
-
- oVS := bytes.Repeat([]byte{0x33}, 8)
- oKS := bytes.Repeat([]byte{0x44}, 8)
- oValHash, err := v5Hash(password, oVS, userEntry, R)
- if err != nil {
- t.Fatalf("v5Hash(owner validation): %v", err)
- }
- oKeyHash, err := v5Hash(password, oKS, userEntry, R)
- if err != nil {
- t.Fatalf("v5Hash(owner key): %v", err)
- }
- oe := aesCBCEncryptRaw(t, oKeyHash, make([]byte, aes.BlockSize), fileKey)
- ownerEntry := append(append(append([]byte{}, oValHash...), oVS...), oKS...)
-
- p := Params{
- V: 5, R: R,
- UserEntry: userEntry,
- OwnerEntry: ownerEntry,
- UE: make([]byte, 32), // present but never reached
- OE: oe,
- }
- key, err := computeV5Key(p, password)
- if err != nil {
- t.Fatalf("computeV5Key: %v", err)
- }
- if !bytes.Equal(key, fileKey) {
- t.Fatalf("owner-path key mismatch:\n got %x\nwant %x", key, fileKey)
- }
-}
-
-func TestComputeV5KeyWrongPassword(t *testing.T) {
- p := Params{
- V: 5, R: 6,
- UserEntry: bytes.Repeat([]byte{0x01}, 48),
- OwnerEntry: bytes.Repeat([]byte{0x02}, 48),
- UE: bytes.Repeat([]byte{0x03}, 32),
- OE: bytes.Repeat([]byte{0x04}, 32),
- }
- overlong := bytes.Repeat([]byte{'z'}, 200) // > 127: exercises the spec truncation
- if _, err := computeV5Key(p, overlong); err == nil {
- t.Fatal("expected an error for a non-matching password, got nil")
- }
-}
-
-// New must reject an unsupported /V rather than returning a zero handler.
-func TestNewUnsupportedVersion(t *testing.T) {
- for _, v := range []int{0, 3, 99} {
- if _, err := New(Params{V: v}, nil); err == nil {
- t.Errorf("New(/V %d) should error", v)
- }
- }
-}
-
-// New drives the /V 5 branch end to end (AES-256 key derivation + algorithm
-// selection), not just computeV5Key in isolation.
-func TestNewV5(t *testing.T) {
- const R = 6
- password := []byte("v5-user")
- fileKey := bytes.Repeat([]byte{0x42}, 32)
- uVS := bytes.Repeat([]byte{0x01}, 8)
- uKS := bytes.Repeat([]byte{0x02}, 8)
- uValHash, err := v5Hash(password, uVS, nil, R)
- if err != nil {
- t.Fatalf("v5Hash(validation): %v", err)
- }
- kHash, err := v5Hash(password, uKS, nil, R)
- if err != nil {
- t.Fatalf("v5Hash(key): %v", err)
- }
- ue := aesCBCEncryptRaw(t, kHash, make([]byte, aes.BlockSize), fileKey)
- userEntry := append(append(append([]byte{}, uValHash...), uVS...), uKS...)
-
- h, err := New(Params{
- V: 5, R: R,
- UserEntry: userEntry,
- OwnerEntry: make([]byte, 48),
- UE: ue,
- OE: make([]byte, 32),
- }, password)
- if err != nil {
- t.Fatalf("New(V5): %v", err)
- }
- if !bytes.Equal(h.FileKey, fileKey) {
- t.Fatal("V5 file key mismatch")
- }
- if h.StreamAlg != AlgAES256 {
- t.Errorf("StreamAlg = %v, want AlgAES256", h.StreamAlg)
- }
-}
-
-// A per-stream Identity crypt filter overrides the default algorithm and passes
-// the bytes through unchanged.
-func TestDecryptStreamIdentityOverride(t *testing.T) {
- h := &Handler{StreamAlg: AlgRC4, FileKey: bytes.Repeat([]byte{1}, 16)}
- data := []byte("plaintext")
- out, err := h.DecryptStream(data, 1, 0, "Identity")
- if err != nil {
- t.Fatalf("DecryptStream: %v", err)
- }
- if !bytes.Equal(out, data) {
- t.Errorf("Identity override = %q, want %q", out, data)
- }
-}
diff --git a/internal/filter/filter.go b/internal/filter/filter.go
deleted file mode 100644
index 757c6e0..0000000
--- a/internal/filter/filter.go
+++ /dev/null
@@ -1,377 +0,0 @@
-// Package filter implements the PDF stream filters needed for read-only
-// inspection of document structure: FlateDecode (with predictors), LZW,
-// ASCII85, ASCIIHex, RunLength, and BrotliDecode (a PDF Association
-// extension to PDF 2.0, pending ISO 32000 inclusion).
-//
-// Image-only filters (DCTDecode, JBIG2Decode, JPXDecode, CCITTFaxDecode)
-// are intentionally not implemented — pdfdisassembler does not decode
-// image streams.
-package filter
-
-import (
- "bytes"
- "compress/zlib"
- "errors"
- "fmt"
- "io"
-
- "github.com/andybalholm/brotli"
-)
-
-// ErrUnsupported is returned for filters this package does not implement.
-type ErrUnsupported struct{ Name string }
-
-func (e ErrUnsupported) Error() string {
- return fmt.Sprintf("pdfdisassembler/filter: unsupported filter %q", e.Name)
-}
-
-// Params describes the decode-time parameters for a single filter.
-type Params struct {
- // FlateDecode/LZWDecode predictor parameters.
- Predictor int
- Columns int
- Colors int
- BitsPerComponent int
- // NoEarlyChange honours the rare /EarlyChange 0; the zero value keeps
- // LZWDecode's default early code-width change.
- NoEarlyChange bool
- // MaxOutput caps decoded bytes per filter; <= 0 means unlimited.
- MaxOutput int64
-}
-
-// Decode applies the named filter to in.
-func Decode(name string, in []byte, p Params) ([]byte, error) {
- switch name {
- case "FlateDecode", "Fl":
- return decodeFlate(in, p)
- case "LZWDecode", "LZW":
- return decodeLZW(in, p)
- case "ASCII85Decode", "A85":
- return decodeASCII85(in)
- case "ASCIIHexDecode", "AHx":
- return decodeASCIIHex(in)
- case "RunLengthDecode", "RL":
- return decodeRunLength(in, p.MaxOutput)
- case "BrotliDecode":
- // No abbreviation: the spec keeps BrotliDecode out of the
- // inline-image abbreviation table (it is banned there).
- return decodeBrotli(in, p)
- }
- return nil, ErrUnsupported{Name: name}
-}
-
-// IsImageFilter reports whether name designates one of the image-only
-// filters that pdfdisassembler intentionally skips.
-func IsImageFilter(name string) bool {
- switch name {
- case "DCTDecode", "DCT",
- "JBIG2Decode",
- "JPXDecode",
- "CCITTFaxDecode", "CCF",
- "Crypt":
- return true
- }
- return false
-}
-
-func decodeFlate(in []byte, p Params) ([]byte, error) {
- zr, err := zlib.NewReader(bytes.NewReader(in))
- if err != nil {
- return nil, fmt.Errorf("FlateDecode: %w", err)
- }
- defer zr.Close()
- dec, err := readAllLimited(zr, p.MaxOutput)
- if err != nil {
- return nil, fmt.Errorf("FlateDecode: %w", err)
- }
- if p.Predictor > 1 {
- return applyPredictor(dec, p)
- }
- return dec, nil
-}
-
-// decodeBrotli handles BrotliDecode ("Brotli compression in PDF 2.0",
-// PDF Association EXTN-BROTLI-1): stream data compressed per RFC 7932 with
-// no additional headers. The spec extends the FlateDecode/LZWDecode
-// DecodeParms table to BrotliDecode, so predictors apply exactly as in
-// decodeFlate. Large-window Brotli (RFC 9841) is required by the spec but
-// not reachable through the underlying decoder's API; such streams (rare,
-// encoder opt-in only) fail with a window-bits format error rather than
-// decoding incorrectly.
-func decodeBrotli(in []byte, p Params) ([]byte, error) {
- // brotli.NewReader cannot fail; errors surface on Read.
- dec, err := readAllLimited(brotli.NewReader(bytes.NewReader(in)), p.MaxOutput)
- if err != nil {
- return nil, fmt.Errorf("BrotliDecode: %w", err)
- }
- if p.Predictor > 1 {
- return applyPredictor(dec, p)
- }
- return dec, nil
-}
-
-// readAllLimited is io.ReadAll that errors once r yields more than max bytes.
-// max <= 0 disables the limit.
-func readAllLimited(r io.Reader, max int64) ([]byte, error) {
- if max <= 0 {
- return io.ReadAll(r)
- }
- out, err := io.ReadAll(io.LimitReader(r, max+1)) // +1 to tell "== max" from "> max"
- if err != nil {
- return nil, err
- }
- if int64(len(out)) > max {
- return nil, fmt.Errorf("decoded output exceeds %d-byte limit (possible decompression bomb)", max)
- }
- return out, nil
-}
-
-func decodeASCII85(in []byte) ([]byte, error) {
- // Trim "<~" prefix and "~>" suffix if present.
- if len(in) >= 2 && in[0] == '<' && in[1] == '~' {
- in = in[2:]
- }
- end := bytes.Index(in, []byte("~>"))
- if end >= 0 {
- in = in[:end]
- }
-
- var out []byte
- var group uint32
- n := 0
- for _, c := range in {
- switch {
- case c == 'z':
- if n != 0 {
- return nil, errors.New("ASCII85Decode: 'z' inside group")
- }
- out = append(out, 0, 0, 0, 0)
- continue
- case c >= '!' && c <= 'u':
- group = group*85 + uint32(c-'!')
- n++
- case c == ' ' || c == '\t' || c == '\r' || c == '\n' || c == '\f':
- continue
- default:
- return nil, fmt.Errorf("ASCII85Decode: invalid byte 0x%02x", c)
- }
- if n == 5 {
- out = append(out,
- byte(group>>24),
- byte(group>>16),
- byte(group>>8),
- byte(group),
- )
- group = 0
- n = 0
- }
- }
- if n > 0 {
- for i := n; i < 5; i++ {
- group = group*85 + 84
- }
- buf := []byte{
- byte(group >> 24),
- byte(group >> 16),
- byte(group >> 8),
- byte(group),
- }
- out = append(out, buf[:n-1]...)
- }
- return out, nil
-}
-
-func decodeASCIIHex(in []byte) ([]byte, error) {
- var out []byte
- var hi int
- have := false
- for _, c := range in {
- if c == '>' {
- break
- }
- if c == ' ' || c == '\t' || c == '\r' || c == '\n' || c == '\f' {
- continue
- }
- d, ok := hexDigit(c)
- if !ok {
- return nil, fmt.Errorf("ASCIIHexDecode: invalid hex byte 0x%02x", c)
- }
- if have {
- out = append(out, byte(hi<<4|d))
- have = false
- } else {
- hi = d
- have = true
- }
- }
- if have {
- out = append(out, byte(hi<<4))
- }
- return out, nil
-}
-
-func decodeRunLength(in []byte, maxOut int64) ([]byte, error) {
- var out []byte
- for i := 0; i < len(in); {
- b := in[i]
- i++
- switch {
- case b < 128:
- n := int(b) + 1
- if i+n > len(in) {
- return nil, errors.New("RunLengthDecode: truncated literal")
- }
- out = append(out, in[i:i+n]...)
- i += n
- case b > 128:
- n := 257 - int(b)
- if i >= len(in) {
- return nil, errors.New("RunLengthDecode: truncated run")
- }
- for k := 0; k < n; k++ {
- out = append(out, in[i])
- }
- i++
- default: // b == 128: EOD
- return out, nil
- }
- if maxOut > 0 && int64(len(out)) > maxOut {
- return nil, fmt.Errorf("RunLengthDecode: decoded output exceeds limit of %d bytes", maxOut)
- }
- }
- return out, nil
-}
-
-func hexDigit(c byte) (int, bool) {
- switch {
- case c >= '0' && c <= '9':
- return int(c - '0'), true
- case c >= 'a' && c <= 'f':
- return int(c-'a') + 10, true
- case c >= 'A' && c <= 'F':
- return int(c-'A') + 10, true
- }
- return 0, false
-}
-
-// applyPredictor reverses the PNG / TIFF predictor wrapping applied to
-// LZW/Flate data. See PDF 32000-1:2008 §7.4.4.4.
-func applyPredictor(in []byte, p Params) ([]byte, error) {
- if p.Predictor <= 1 {
- return in, nil
- }
- colors := p.Colors
- if colors == 0 {
- colors = 1
- }
- bpc := p.BitsPerComponent
- if bpc == 0 {
- bpc = 8
- }
- columns := p.Columns
- if columns == 0 {
- columns = 1
- }
- // /Colors, /BitsPerComponent, /Columns are attacker-controlled. Negative
- // values (or an overflowing product) drive rowBytes to zero or negative,
- // which would divide-by-zero or make a negative-length slice below.
- if colors < 1 || bpc < 1 || columns < 1 {
- return nil, fmt.Errorf("predictor: /Colors, /BitsPerComponent, /Columns must be positive")
- }
- bytesPerPixel := (colors*bpc + 7) / 8
- rowBytes := (columns*colors*bpc + 7) / 8
- if bytesPerPixel < 1 || rowBytes < 1 {
- return nil, fmt.Errorf("predictor: invalid row geometry (bytesPerPixel=%d, rowBytes=%d)", bytesPerPixel, rowBytes)
- }
-
- if p.Predictor == 2 {
- // TIFF predictor 2: per-row horizontal differences. Not commonly
- // used in our domain but supported for completeness.
- if len(in)%rowBytes != 0 {
- return nil, fmt.Errorf("predictor 2: %d bytes not divisible by row %d", len(in), rowBytes)
- }
- out := make([]byte, len(in))
- for r := 0; r < len(in); r += rowBytes {
- row := in[r : r+rowBytes]
- dst := out[r : r+rowBytes]
- copy(dst, row)
- for c := bytesPerPixel; c < rowBytes; c++ {
- dst[c] = byte(int(row[c]) + int(dst[c-bytesPerPixel]))
- }
- }
- return out, nil
- }
-
- // PNG predictors: rowBytes data preceded by a 1-byte filter tag.
- stride := rowBytes + 1
- if len(in)%stride != 0 {
- return nil, fmt.Errorf("predictor PNG: %d bytes not divisible by row %d", len(in), stride)
- }
- rows := len(in) / stride
- out := make([]byte, rows*rowBytes)
- prev := make([]byte, rowBytes)
- cur := make([]byte, rowBytes)
- for r := 0; r < rows; r++ {
- tag := in[r*stride]
- row := in[r*stride+1 : (r+1)*stride]
- switch tag {
- case 0: // None
- copy(cur, row)
- case 1: // Sub
- for c := 0; c < rowBytes; c++ {
- var left byte
- if c >= bytesPerPixel {
- left = cur[c-bytesPerPixel]
- }
- cur[c] = row[c] + left
- }
- case 2: // Up
- for c := 0; c < rowBytes; c++ {
- cur[c] = row[c] + prev[c]
- }
- case 3: // Average
- for c := 0; c < rowBytes; c++ {
- var left byte
- if c >= bytesPerPixel {
- left = cur[c-bytesPerPixel]
- }
- cur[c] = row[c] + byte((int(left)+int(prev[c]))/2)
- }
- case 4: // Paeth
- for c := 0; c < rowBytes; c++ {
- var left, upLeft byte
- if c >= bytesPerPixel {
- left = cur[c-bytesPerPixel]
- upLeft = prev[c-bytesPerPixel]
- }
- cur[c] = row[c] + paeth(left, prev[c], upLeft)
- }
- default:
- return nil, fmt.Errorf("predictor PNG: unknown tag %d", tag)
- }
- copy(out[r*rowBytes:], cur)
- copy(prev, cur)
- }
- return out, nil
-}
-
-func paeth(a, b, c byte) byte {
- p := int(a) + int(b) - int(c)
- pa := abs(p - int(a))
- pb := abs(p - int(b))
- pc := abs(p - int(c))
- switch {
- case pa <= pb && pa <= pc:
- return a
- case pb <= pc:
- return b
- }
- return c
-}
-
-func abs(x int) int {
- if x < 0 {
- return -x
- }
- return x
-}
diff --git a/internal/filter/filter_test.go b/internal/filter/filter_test.go
deleted file mode 100644
index b10594b..0000000
--- a/internal/filter/filter_test.go
+++ /dev/null
@@ -1,529 +0,0 @@
-package filter
-
-import (
- "bytes"
- "compress/lzw"
- "compress/zlib"
- "encoding/ascii85"
- "errors"
- "fmt"
- "strings"
- "testing"
-
- "github.com/andybalholm/brotli"
-)
-
-// lcg fills b with a deterministic pseudo-random byte stream — enough entropy
-// to grow the LZW dictionary through every code width and trigger a reset.
-func lcg(b []byte, seed uint32) {
- x := seed
- for i := range b {
- x = x*1664525 + 1013904223
- b[i] = byte(x >> 24)
- }
-}
-
-// TestLZWRoundTripStdlib decodes streams produced by the standard library's
-// MSB LZW writer. That writer uses the non-early code-width change, so decode
-// with NoEarlyChange; the varied inputs exercise dictionary reuse, the KwKwK
-// case, 9->12-bit width growth, and the dictionary-full reset.
-func TestLZWRoundTripStdlib(t *testing.T) {
- big := make([]byte, 64<<10)
- lcg(big, 1)
- for _, orig := range [][]byte{
- []byte("ABABABABABABABABABAB"),
- []byte("TOBEORNOTTOBEORTOBEORNOT"),
- bytes.Repeat([]byte("xyz "), 4096),
- big,
- {},
- {42},
- } {
- var buf bytes.Buffer
- w := lzw.NewWriter(&buf, lzw.MSB, 8)
- if _, err := w.Write(orig); err != nil {
- t.Fatal(err)
- }
- w.Close()
- got, err := Decode("LZWDecode", buf.Bytes(), Params{NoEarlyChange: true})
- if err != nil {
- t.Fatalf("decode (len %d): %v", len(orig), err)
- }
- if !bytes.Equal(got, orig) {
- t.Fatalf("round-trip mismatch for len %d input", len(orig))
- }
- }
-}
-
-// TestLZWEarlyChangeHonored proves NoEarlyChange actually selects the decode
-// convention: the stdlib stream (non-early) round-trips only with NoEarlyChange
-// set; under the early-change default the same bytes must not reproduce the input.
-func TestLZWEarlyChangeHonored(t *testing.T) {
- orig := make([]byte, 4096) // long enough to cross the first width boundary
- lcg(orig, 7)
- var buf bytes.Buffer
- w := lzw.NewWriter(&buf, lzw.MSB, 8)
- w.Write(orig)
- w.Close()
- enc := buf.Bytes()
-
- got, err := Decode("LZWDecode", enc, Params{NoEarlyChange: true})
- if err != nil || !bytes.Equal(got, orig) {
- t.Fatalf("NoEarlyChange decode of a non-early stream: err=%v equal=%v", err, bytes.Equal(got, orig))
- }
- if d, err := Decode("LZWDecode", enc, Params{}); err == nil && bytes.Equal(d, orig) {
- t.Fatal("early-change default reproduced a non-early stream: the flag is ignored")
- }
-}
-
-// A stream truncated mid-code must not panic: readBits pads the final partial
-// word with zeros and decoding stops cleanly (partial output or error).
-func TestLZWTruncatedNoPanic(t *testing.T) {
- orig := make([]byte, 600) // long enough to reach 10-bit codes
- lcg(orig, 3)
- var buf bytes.Buffer
- w := lzw.NewWriter(&buf, lzw.MSB, 8)
- w.Write(orig)
- w.Close()
- enc := buf.Bytes()
- for cut := 1; cut <= 4 && cut < len(enc); cut++ {
- _, _ = Decode("LZWDecode", enc[:len(enc)-cut], Params{NoEarlyChange: true})
- }
-}
-
-func TestASCII85RoundTripStdlib(t *testing.T) {
- for _, orig := range [][]byte{
- []byte("Hello World!"),
- {0, 0, 0, 0, 1, 2, 3}, // includes an all-zero group
- {1}, // 1-byte partial group
- {1, 2}, // 2-byte partial group
- {1, 2, 3}, // 3-byte partial group
- bytes.Repeat([]byte{255}, 17),
- {},
- } {
- enc := make([]byte, ascii85.MaxEncodedLen(len(orig)))
- n := ascii85.Encode(enc, orig)
- in := append(enc[:n:n], '~', '>')
- out, err := Decode("ASCII85Decode", in, Params{})
- if err != nil {
- t.Fatalf("decode %v: %v", orig, err)
- }
- if !bytes.Equal(out, orig) {
- t.Fatalf("round-trip mismatch: in=%v out=%v", orig, out)
- }
- }
-}
-
-func TestASCII85EdgeCases(t *testing.T) {
- // 'z' is shorthand for a full zero group; <~ is an optional opening marker;
- // whitespace between digits is ignored.
- out, err := Decode("ASCII85Decode", []byte("<~z 8 7 c U R D ] i , \" E b o 8 0 ~>"), Params{})
- if err != nil {
- t.Fatalf("decode: %v", err)
- }
- if want := append([]byte{0, 0, 0, 0}, "Hello World!"...); !bytes.Equal(out, want) {
- t.Fatalf("got %q, want %q", out, want)
- }
-
- if _, err := Decode("ASCII85Decode", []byte("abc!\x01def~>"), Params{}); err == nil {
- t.Fatal("expected an error on a byte outside the ASCII85 alphabet")
- }
- // 'z' may not appear mid-group (after a partial digit).
- if _, err := Decode("ASCII85Decode", []byte("87z~>"), Params{}); err == nil {
- t.Fatal("expected an error for 'z' inside a group")
- }
-}
-
-func TestASCIIHex(t *testing.T) {
- cases := map[string]string{
- "48656C6C6F>": "Hello",
- "4 86 56C 6C6F": "Hello",
- "48656c6c6f": "Hello",
- }
- for in, want := range cases {
- out, err := Decode("ASCIIHexDecode", []byte(in), Params{})
- if err != nil {
- t.Fatalf("%q: %v", in, err)
- }
- if string(out) != want {
- t.Fatalf("%q: got %q want %q", in, out, want)
- }
- }
-}
-
-func TestASCII85(t *testing.T) {
- in := []byte("87cURD]i,\"Ebo80~>")
- out, err := Decode("ASCII85Decode", in, Params{})
- if err != nil {
- t.Fatal(err)
- }
- if string(out) != "Hello World!" {
- t.Fatalf("got %q", out)
- }
-}
-
-func TestRunLength(t *testing.T) {
- // 3 literal "ABC" (length-1=2), then 3 copies of 'X' (257-3=254), EOD.
- in := []byte{2, 'A', 'B', 'C', 254, 'X', 128}
- out, err := Decode("RunLengthDecode", in, Params{})
- if err != nil {
- t.Fatal(err)
- }
- if string(out) != "ABCXXX" {
- t.Fatalf("got %q", out)
- }
-}
-
-func TestFlate(t *testing.T) {
- var buf bytes.Buffer
- zw := zlib.NewWriter(&buf)
- zw.Write([]byte("hello flate"))
- zw.Close()
- out, err := Decode("FlateDecode", buf.Bytes(), Params{})
- if err != nil {
- t.Fatal(err)
- }
- if string(out) != "hello flate" {
- t.Fatalf("got %q", out)
- }
-}
-
-func TestFlatePNGPredictor(t *testing.T) {
- // 2 rows, 4 bytes each. Predictor tag 0 = None.
- row1 := []byte{0, 1, 2, 3, 4}
- row2 := []byte{0, 5, 6, 7, 8}
- raw := append(append([]byte{}, row1...), row2...)
- var buf bytes.Buffer
- zw := zlib.NewWriter(&buf)
- zw.Write(raw)
- zw.Close()
- out, err := Decode("FlateDecode", buf.Bytes(), Params{
- Predictor: 12,
- Columns: 4,
- Colors: 1,
- BitsPerComponent: 8,
- })
- if err != nil {
- t.Fatal(err)
- }
- want := []byte{1, 2, 3, 4, 5, 6, 7, 8}
- if !bytes.Equal(out, want) {
- t.Fatalf("got % x want % x", out, want)
- }
-}
-
-func TestPredictorHostileParamsNoPanic(t *testing.T) {
- var buf bytes.Buffer
- zw := zlib.NewWriter(&buf)
- zw.Write([]byte("ABCDEFGH"))
- zw.Close()
- flate := buf.Bytes()
-
- cases := []struct {
- name string
- p Params
- }{
- {"tiff_rowbytes_zero", Params{Predictor: 2, Colors: -7, BitsPerComponent: 1, Columns: 1}},
- {"png_stride_zero", Params{Predictor: 12, Colors: -15, BitsPerComponent: 1, Columns: 1}},
- {"png_negative_make", Params{Predictor: 12, Colors: -23, BitsPerComponent: 1, Columns: 1}},
- {"negative_columns", Params{Predictor: 12, Colors: 1, BitsPerComponent: 8, Columns: -4}},
- }
- for _, tc := range cases {
- t.Run(tc.name, func(t *testing.T) {
- if _, err := Decode("FlateDecode", flate, tc.p); err == nil {
- t.Fatal("expected an error for hostile predictor params, got nil")
- }
- })
- }
-}
-
-// pngForwardFilter is the inverse of applyPredictor's PNG path: it applies row
-// filter tag to rows, so decode(encode(rows)) must recover rows exactly.
-func pngForwardFilter(tag byte, rows [][]byte, bpp int) []byte {
- var out []byte
- prev := make([]byte, len(rows[0]))
- for _, raw := range rows {
- out = append(out, tag)
- filt := make([]byte, len(raw))
- for c := range raw {
- var left, upLeft byte
- up := prev[c]
- if c >= bpp {
- left = raw[c-bpp]
- upLeft = prev[c-bpp]
- }
- switch tag {
- case 0:
- filt[c] = raw[c]
- case 1:
- filt[c] = raw[c] - left
- case 2:
- filt[c] = raw[c] - up
- case 3:
- filt[c] = raw[c] - byte((int(left)+int(up))/2)
- case 4:
- filt[c] = raw[c] - paeth(left, up, upLeft)
- }
- }
- out = append(out, filt...)
- prev = raw
- }
- return out
-}
-
-func flate(t *testing.T, b []byte) []byte {
- t.Helper()
- var buf bytes.Buffer
- zw := zlib.NewWriter(&buf)
- zw.Write(b)
- zw.Close()
- return buf.Bytes()
-}
-
-func TestPredictorPNGRoundTrip(t *testing.T) {
- rows := [][]byte{
- {10, 20, 30, 40},
- {15, 25, 35, 45},
- {200, 100, 50, 25},
- }
- var want []byte
- for _, r := range rows {
- want = append(want, r...)
- }
- for tag := byte(0); tag <= 4; tag++ {
- t.Run(fmt.Sprintf("tag%d", tag), func(t *testing.T) {
- filtered := pngForwardFilter(tag, rows, 1)
- out, err := Decode("FlateDecode", flate(t, filtered), Params{
- Predictor: 12, Columns: 4, Colors: 1, BitsPerComponent: 8,
- })
- if err != nil {
- t.Fatalf("decode: %v", err)
- }
- if !bytes.Equal(out, want) {
- t.Fatalf("tag %d round-trip: got % x want % x", tag, out, want)
- }
- })
- }
-}
-
-func TestPredictorTIFFRoundTrip(t *testing.T) {
- raw := []byte{10, 5, 250, 3}
- filt := make([]byte, len(raw))
- filt[0] = raw[0]
- for c := 1; c < len(raw); c++ {
- filt[c] = raw[c] - raw[c-1]
- }
- out, err := Decode("FlateDecode", flate(t, filt), Params{
- Predictor: 2, Columns: 4, Colors: 1, BitsPerComponent: 8,
- })
- if err != nil {
- t.Fatalf("decode: %v", err)
- }
- if !bytes.Equal(out, raw) {
- t.Fatalf("TIFF round-trip: got % x want % x", out, raw)
- }
-}
-
-func TestFlateBombRejected(t *testing.T) {
- // 1 MiB of zeros (compresses to ~1 KB) against a 4 KB cap.
- var buf bytes.Buffer
- zw := zlib.NewWriter(&buf)
- zw.Write(make([]byte, 1<<20))
- zw.Close()
- if _, err := Decode("FlateDecode", buf.Bytes(), Params{MaxOutput: 4096}); err == nil {
- t.Fatal("expected error for output exceeding MaxOutput, got nil")
- }
-}
-
-func TestFlateUnderLimit(t *testing.T) {
- var buf bytes.Buffer
- zw := zlib.NewWriter(&buf)
- zw.Write([]byte("hello flate"))
- zw.Close()
- out, err := Decode("FlateDecode", buf.Bytes(), Params{MaxOutput: 1 << 20})
- if err != nil {
- t.Fatal(err)
- }
- if string(out) != "hello flate" {
- t.Fatalf("got %q", out)
- }
-}
-
-// brotliCompress is the test-side encoder for the read-only BrotliDecode path.
-func brotliCompress(t *testing.T, b []byte) []byte {
- t.Helper()
- var buf bytes.Buffer
- bw := brotli.NewWriter(&buf)
- if _, err := bw.Write(b); err != nil {
- t.Fatal(err)
- }
- if err := bw.Close(); err != nil {
- t.Fatal(err)
- }
- return buf.Bytes()
-}
-
-func TestBrotliRoundTrip(t *testing.T) {
- big := make([]byte, 64<<10)
- lcg(big, 5)
- for _, orig := range [][]byte{
- []byte("hello brotli"),
- bytes.Repeat([]byte("xyz "), 4096),
- big,
- {},
- {42},
- } {
- got, err := Decode("BrotliDecode", brotliCompress(t, orig), Params{MaxOutput: 1 << 20})
- if err != nil {
- t.Fatalf("decode (len %d): %v", len(orig), err)
- }
- if !bytes.Equal(got, orig) {
- t.Fatalf("round-trip mismatch for len %d input", len(orig))
- }
- }
-}
-
-func TestBrotliCorruptRejected(t *testing.T) {
- enc := brotliCompress(t, bytes.Repeat([]byte("pdfdisassembler "), 512))
- cases := map[string][]byte{
- "empty": {},
- "garbage": {0xC1, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF},
- "truncated": enc[:len(enc)/2],
- }
- for name, in := range cases {
- t.Run(name, func(t *testing.T) {
- _, err := Decode("BrotliDecode", in, Params{})
- if err == nil {
- t.Fatal("expected an error, got nil")
- }
- if !strings.Contains(err.Error(), "BrotliDecode") {
- t.Fatalf("error %q does not name BrotliDecode", err)
- }
- })
- }
-}
-
-func TestBrotliBombRejected(t *testing.T) {
- // 1 MiB of zeros (compresses to a few bytes) against a 4 KB cap.
- if _, err := Decode("BrotliDecode", brotliCompress(t, make([]byte, 1<<20)), Params{MaxOutput: 4096}); err == nil {
- t.Fatal("expected error for output exceeding MaxOutput, got nil")
- }
-}
-
-// The Brotli extension spec extends the Flate/LZW predictor parameters to
-// BrotliDecode, so the PNG-predictor path must work identically.
-func TestBrotliPNGPredictor(t *testing.T) {
- // 2 rows, 4 bytes each. Predictor tag 0 = None.
- row1 := []byte{0, 1, 2, 3, 4}
- row2 := []byte{0, 5, 6, 7, 8}
- raw := append(append([]byte{}, row1...), row2...)
- out, err := Decode("BrotliDecode", brotliCompress(t, raw), Params{
- Predictor: 12,
- Columns: 4,
- Colors: 1,
- BitsPerComponent: 8,
- })
- if err != nil {
- t.Fatal(err)
- }
- want := []byte{1, 2, 3, 4, 5, 6, 7, 8}
- if !bytes.Equal(out, want) {
- t.Fatalf("got % x want % x", out, want)
- }
-}
-
-// The spec defines no abbreviation for BrotliDecode (it is excluded from the
-// inline-image abbreviation table), so "Br" must stay unsupported.
-func TestBrotliNoAbbreviation(t *testing.T) {
- var unsup ErrUnsupported
- if _, err := Decode("Br", []byte{0x3B}, Params{}); !errors.As(err, &unsup) {
- t.Fatalf("Decode(Br) error = %v, want ErrUnsupported", err)
- }
-}
-
-func TestRunLengthBombRejected(t *testing.T) {
- // {129,'X'} expands to 257-129 = 128 copies of 'X', past the 64-byte cap.
- if _, err := Decode("RunLengthDecode", []byte{129, 'X'}, Params{MaxOutput: 64}); err == nil {
- t.Fatal("expected error for output exceeding MaxOutput, got nil")
- }
-}
-
-func TestLZWDecodeAndBombRejected(t *testing.T) {
- // PDF-LZW (9-bit, MSB-first) codes 65,66,257 -> "AB" then EOD.
- in := []byte{0x20, 0x90, 0xA0, 0x20}
- out, err := Decode("LZWDecode", in, Params{})
- if err != nil || string(out) != "AB" {
- t.Fatalf("baseline decode: out=%q err=%v", out, err)
- }
- if _, err := Decode("LZWDecode", in, Params{MaxOutput: 1}); err == nil {
- t.Fatal("expected error for output exceeding MaxOutput, got nil")
- }
-}
-
-func TestImageFilterRejected(t *testing.T) {
- if !IsImageFilter("DCTDecode") {
- t.Fatal("DCTDecode should be image filter")
- }
- if !IsImageFilter("JPXDecode") {
- t.Fatal("JPXDecode should be image filter")
- }
- if IsImageFilter("FlateDecode") {
- t.Fatal("FlateDecode should not be image filter")
- }
-}
-
-// FuzzDecode asserts every filter and the predictor never panic on arbitrary
-// input or attacker-controlled predictor parameters.
-func FuzzDecode(f *testing.F) {
- var seed bytes.Buffer
- zw := zlib.NewWriter(&seed)
- zw.Write([]byte("seed data"))
- zw.Close()
- f.Add(seed.Bytes(), 12, 4, 1, 8)
- f.Fuzz(func(t *testing.T, data []byte, predictor, columns, colors, bpc int) {
- p := Params{
- Predictor: predictor,
- Columns: columns,
- Colors: colors,
- BitsPerComponent: bpc,
- MaxOutput: 1 << 20,
- }
- for _, name := range []string{"FlateDecode", "LZWDecode", "ASCII85Decode", "ASCIIHexDecode", "RunLengthDecode", "BrotliDecode"} {
- _, _ = Decode(name, data, p)
- }
- })
-}
-
-// Expected values hand-computed from PNG §6.6 (not copied from output).
-// paeth(0,0,10) makes p=a+b-c negative — the one vector here that reaches
-// abs's negative branch; don't drop it.
-func TestPaeth(t *testing.T) {
- cases := []struct {
- a, b, c, want byte
- }{
- {0, 0, 0, 0},
- {1, 2, 3, 1}, // p=0: pa=1 pb=2 pc=3 -> a
- {255, 0, 0, 255}, // p=255: pa=0 -> a
- {0, 0, 10, 0}, // p=-10: pa=10 pb=10 pc=20 -> a (negative estimate)
- {10, 20, 11, 20}, // p=19: pa=9 pb=1 pc=8 -> b
- {0, 10, 6, 6}, // p=4: pa=4 pb=6 pc=2 -> c
- }
- for _, tc := range cases {
- if got := paeth(tc.a, tc.b, tc.c); got != tc.want {
- t.Errorf("paeth(%d,%d,%d) = %d, want %d", tc.a, tc.b, tc.c, got, tc.want)
- }
- }
-}
-
-func TestErrUnsupported(t *testing.T) {
- _, err := Decode("DCTDecode", []byte("x"), Params{})
- var unsup ErrUnsupported
- if !errors.As(err, &unsup) {
- t.Fatalf("Decode(DCTDecode) error = %v, want ErrUnsupported", err)
- }
- if unsup.Name != "DCTDecode" {
- t.Errorf("ErrUnsupported.Name = %q, want DCTDecode", unsup.Name)
- }
- if want := `pdfdisassembler/filter: unsupported filter "DCTDecode"`; unsup.Error() != want {
- t.Errorf("Error() = %q, want %q", unsup.Error(), want)
- }
-}
diff --git a/internal/filter/lzw.go b/internal/filter/lzw.go
deleted file mode 100644
index 53dad19..0000000
--- a/internal/filter/lzw.go
+++ /dev/null
@@ -1,111 +0,0 @@
-package filter
-
-import "fmt"
-
-// decodeLZW decodes a PDF LZW stream. Code widths grow from 9 to 12 bits;
-// the early-change flag (default 1) shrinks the threshold at which each
-// width step happens.
-func decodeLZW(in []byte, p Params) ([]byte, error) {
- early := 1
- if p.NoEarlyChange {
- early = 0
- }
- const (
- clearCode = 256
- eodCode = 257
- )
- br := bitReader{src: in}
- codeWidth := 9
- dict := make([][]byte, 258, 4096)
- for i := 0; i < 256; i++ {
- dict[i] = []byte{byte(i)}
- }
-
- var out []byte
- prev := -1
-
- resize := func() {
- codeWidth = 9
- dict = dict[:258]
- }
-
- for {
- code, ok := br.readBits(codeWidth)
- if !ok {
- break
- }
- switch {
- case code == clearCode:
- resize()
- prev = -1
- continue
- case code == eodCode:
- return out, nil
- }
-
- var entry []byte
- switch {
- case int(code) < len(dict):
- entry = dict[code]
- case int(code) == len(dict) && prev >= 0:
- pe := dict[prev]
- entry = make([]byte, len(pe)+1)
- copy(entry, pe)
- entry[len(pe)] = pe[0]
- default:
- return nil, fmt.Errorf("LZWDecode: invalid code %d at width %d", code, codeWidth)
- }
- out = append(out, entry...)
- if p.MaxOutput > 0 && int64(len(out)) > p.MaxOutput {
- return nil, fmt.Errorf("LZWDecode: decoded output exceeds limit of %d bytes (possible decompression bomb)", p.MaxOutput)
- }
- if prev >= 0 && len(dict) < 4096 {
- pe := dict[prev]
- ne := make([]byte, len(pe)+1)
- copy(ne, pe)
- ne[len(pe)] = entry[0]
- dict = append(dict, ne)
- }
- prev = int(code)
-
- // Grow width: the new code's index will be len(dict). We need to
- // switch when the next code may not fit.
- threshold := (1 << uint(codeWidth)) - early
- if len(dict) >= threshold && codeWidth < 12 {
- codeWidth++
- }
- }
- return out, nil
-}
-
-type bitReader struct {
- src []byte
- bytePos int
- bitPos uint // 0 = MSB unread
- buf uint64
- have uint // number of bits buffered
-}
-
-func (b *bitReader) readBits(n int) (uint32, bool) {
- for b.have < uint(n) {
- if b.bytePos >= len(b.src) {
- if b.have == 0 {
- return 0, false
- }
- // Pad with zeros to flush trailing partial word.
- b.buf <<= 8
- b.have += 8
- b.bytePos++
- continue
- }
- b.buf = (b.buf << 8) | uint64(b.src[b.bytePos])
- b.have += 8
- b.bytePos++
- }
- shift := b.have - uint(n)
- mask := (uint64(1) << uint(n)) - 1
- v := uint32((b.buf >> shift) & mask)
- b.have -= uint(n)
- b.buf &= (uint64(1) << b.have) - 1
- return v, true
-}
diff --git a/internal/lex/lex.go b/internal/lex/lex.go
deleted file mode 100644
index 7f46b77..0000000
--- a/internal/lex/lex.go
+++ /dev/null
@@ -1,430 +0,0 @@
-// Package lex tokenises PDF input. It deals with the lexical layer of
-// PDF objects — whitespace, comments, names, numbers, strings, arrays,
-// dictionaries, the stream/endstream/obj/endobj/R/null/true/false keywords —
-// but does not assemble higher-level structures. The parser layered above
-// it turns token streams into Object trees.
-package lex
-
-import (
- "errors"
- "fmt"
-)
-
-// Kind identifies a token's lexical category.
-type Kind int
-
-const (
- // EOF marks end of input.
- EOF Kind = iota
- // Name is a PDF name without the leading slash.
- Name
- // Integer is a literal integer (no decimal point, optional sign).
- Integer
- // Real is a literal real number (has a decimal point or 'e' exponent —
- // PDF does not actually allow exponents but we accept them).
- Real
- // LitString is a parenthesised literal string with escapes already
- // resolved.
- LitString
- // HexString is an angle-bracketed hex string with hex pairs already
- // decoded to bytes.
- HexString
- // ArrayStart is the '[' token.
- ArrayStart
- // ArrayEnd is the ']' token.
- ArrayEnd
- // DictStart is the '<<' token.
- DictStart
- // DictEnd is the '>>' token.
- DictEnd
- // Keyword is any unquoted identifier: true, false, null, obj, endobj,
- // stream, endstream, R, xref, trailer, startxref, n, f.
- Keyword
-)
-
-func (k Kind) String() string {
- switch k {
- case EOF:
- return "EOF"
- case Name:
- return "Name"
- case Integer:
- return "Integer"
- case Real:
- return "Real"
- case LitString:
- return "LitString"
- case HexString:
- return "HexString"
- case ArrayStart:
- return "["
- case ArrayEnd:
- return "]"
- case DictStart:
- return "<<"
- case DictEnd:
- return ">>"
- case Keyword:
- return "Keyword"
- }
- return fmt.Sprintf("Kind(%d)", int(k))
-}
-
-// Token is a single lexical unit. Bytes carries the token payload; its
-// meaning depends on Kind:
-// - Name, Keyword: ASCII name body, no leading slash
-// - Integer, Real: literal digits
-// - LitString, HexString: decoded bytes
-// - ArrayStart, ArrayEnd, DictStart, DictEnd, EOF: empty
-type Token struct {
- Kind Kind
- Bytes []byte
- Offset int64 // byte offset in the input where this token started
-}
-
-// Lexer converts a byte slice into a stream of Tokens. It is not safe for
-// concurrent use.
-type Lexer struct {
- src []byte
- pos int
-}
-
-// New creates a Lexer over src. The src slice is not copied.
-func New(src []byte) *Lexer {
- return &Lexer{src: src}
-}
-
-// Pos returns the current byte offset.
-func (l *Lexer) Pos() int { return l.pos }
-
-// SetPos rewinds or fast-forwards the lexer.
-func (l *Lexer) SetPos(p int) { l.pos = p }
-
-// Remaining returns the unread portion of the source.
-func (l *Lexer) Remaining() []byte { return l.src[l.pos:] }
-
-// Source returns the underlying source slice.
-func (l *Lexer) Source() []byte { return l.src }
-
-// ErrUnexpectedEOF indicates that the lexer ran out of bytes mid-token.
-var ErrUnexpectedEOF = errors.New("pdfdisassembler/lex: unexpected EOF")
-
-// IsWhitespace reports whether c is a PDF whitespace character (§7.2.2).
-func IsWhitespace(c byte) bool {
- switch c {
- case 0, '\t', '\n', '\f', '\r', ' ':
- return true
- }
- return false
-}
-
-// IsDelimiter reports whether c is a PDF delimiter character (§7.2.2).
-func IsDelimiter(c byte) bool {
- switch c {
- case '(', ')', '<', '>', '[', ']', '{', '}', '/', '%':
- return true
- }
- return false
-}
-
-// IsRegular reports whether c is a regular character (neither whitespace
-// nor delimiter).
-func IsRegular(c byte) bool { return !IsWhitespace(c) && !IsDelimiter(c) }
-
-// SkipWhitespace advances over PDF whitespace and comments.
-func (l *Lexer) SkipWhitespace() {
- for l.pos < len(l.src) {
- c := l.src[l.pos]
- if IsWhitespace(c) {
- l.pos++
- continue
- }
- if c == '%' {
- // Comment to end of line.
- for l.pos < len(l.src) && l.src[l.pos] != '\n' && l.src[l.pos] != '\r' {
- l.pos++
- }
- continue
- }
- return
- }
-}
-
-// Next returns the next token. At EOF it returns a Token with Kind=EOF.
-func (l *Lexer) Next() (Token, error) {
- l.SkipWhitespace()
- if l.pos >= len(l.src) {
- return Token{Kind: EOF, Offset: int64(l.pos)}, nil
- }
- start := l.pos
- c := l.src[l.pos]
-
- switch {
- case c == '/':
- return l.readName(start)
- case c == '(':
- return l.readLiteralString(start)
- case c == '<':
- if l.pos+1 < len(l.src) && l.src[l.pos+1] == '<' {
- l.pos += 2
- return Token{Kind: DictStart, Offset: int64(start)}, nil
- }
- return l.readHexString(start)
- case c == '>':
- if l.pos+1 < len(l.src) && l.src[l.pos+1] == '>' {
- l.pos += 2
- return Token{Kind: DictEnd, Offset: int64(start)}, nil
- }
- return Token{}, fmt.Errorf("pdfdisassembler/lex: unexpected '>' at %d", l.pos)
- case c == '[':
- l.pos++
- return Token{Kind: ArrayStart, Offset: int64(start)}, nil
- case c == ']':
- l.pos++
- return Token{Kind: ArrayEnd, Offset: int64(start)}, nil
- case c == '+' || c == '-' || c == '.' || (c >= '0' && c <= '9'):
- return l.readNumber(start)
- default:
- return l.readKeyword(start)
- }
-}
-
-func (l *Lexer) readName(start int) (Token, error) {
- l.pos++ // skip '/'
- nameStart := l.pos
- var buf []byte
- for l.pos < len(l.src) {
- c := l.src[l.pos]
- if IsWhitespace(c) || IsDelimiter(c) {
- break
- }
- if c == '#' {
- if buf == nil {
- buf = append(buf, l.src[nameStart:l.pos]...)
- }
- if l.pos+2 >= len(l.src) {
- return Token{}, ErrUnexpectedEOF
- }
- hi, ok1 := hexDigit(l.src[l.pos+1])
- lo, ok2 := hexDigit(l.src[l.pos+2])
- if !ok1 || !ok2 {
- return Token{}, fmt.Errorf("pdfdisassembler/lex: invalid #XX escape in name at %d", l.pos)
- }
- buf = append(buf, byte(hi<<4|lo))
- l.pos += 3
- continue
- }
- if buf != nil {
- buf = append(buf, c)
- }
- l.pos++
- }
- if buf == nil {
- buf = l.src[nameStart:l.pos]
- }
- return Token{Kind: Name, Bytes: buf, Offset: int64(start)}, nil
-}
-
-func (l *Lexer) readLiteralString(start int) (Token, error) {
- l.pos++ // skip '('
- depth := 1
- var buf []byte
- for l.pos < len(l.src) {
- c := l.src[l.pos]
- switch c {
- case '(':
- depth++
- buf = append(buf, c)
- l.pos++
- case ')':
- depth--
- if depth == 0 {
- l.pos++
- return Token{Kind: LitString, Bytes: buf, Offset: int64(start)}, nil
- }
- buf = append(buf, c)
- l.pos++
- case '\\':
- if l.pos+1 >= len(l.src) {
- return Token{}, ErrUnexpectedEOF
- }
- next := l.src[l.pos+1]
- switch next {
- case 'n':
- buf = append(buf, '\n')
- l.pos += 2
- case 'r':
- buf = append(buf, '\r')
- l.pos += 2
- case 't':
- buf = append(buf, '\t')
- l.pos += 2
- case 'b':
- buf = append(buf, '\b')
- l.pos += 2
- case 'f':
- buf = append(buf, '\f')
- l.pos += 2
- case '(':
- buf = append(buf, '(')
- l.pos += 2
- case ')':
- buf = append(buf, ')')
- l.pos += 2
- case '\\':
- buf = append(buf, '\\')
- l.pos += 2
- case '\n':
- // line continuation
- l.pos += 2
- case '\r':
- l.pos += 2
- if l.pos < len(l.src) && l.src[l.pos] == '\n' {
- l.pos++
- }
- case '0', '1', '2', '3', '4', '5', '6', '7':
- // Octal: up to 3 digits.
- v := 0
- n := 0
- p := l.pos + 1
- for n < 3 && p < len(l.src) {
- d := l.src[p]
- if d < '0' || d > '7' {
- break
- }
- v = v*8 + int(d-'0')
- p++
- n++
- }
- buf = append(buf, byte(v&0xFF))
- l.pos = p
- default:
- // Unknown escape: drop the backslash, keep next byte.
- buf = append(buf, next)
- l.pos += 2
- }
- case '\r':
- // CR or CRLF inside literal becomes LF (per spec §7.3.4.2).
- buf = append(buf, '\n')
- l.pos++
- if l.pos < len(l.src) && l.src[l.pos] == '\n' {
- l.pos++
- }
- default:
- buf = append(buf, c)
- l.pos++
- }
- }
- return Token{}, ErrUnexpectedEOF
-}
-
-func (l *Lexer) readHexString(start int) (Token, error) {
- l.pos++ // skip '<'
- var buf []byte
- var hi int
- have := false
- for l.pos < len(l.src) {
- c := l.src[l.pos]
- if c == '>' {
- if have {
- buf = append(buf, byte(hi<<4))
- }
- l.pos++
- return Token{Kind: HexString, Bytes: buf, Offset: int64(start)}, nil
- }
- if IsWhitespace(c) {
- l.pos++
- continue
- }
- d, ok := hexDigit(c)
- if !ok {
- return Token{}, fmt.Errorf("pdfdisassembler/lex: invalid hex digit %q at %d", c, l.pos)
- }
- if have {
- buf = append(buf, byte(hi<<4|d))
- have = false
- } else {
- hi = d
- have = true
- }
- l.pos++
- }
- return Token{}, ErrUnexpectedEOF
-}
-
-func (l *Lexer) readNumber(start int) (Token, error) {
- isReal := false
- p := l.pos
- if p < len(l.src) && (l.src[p] == '+' || l.src[p] == '-') {
- p++
- }
- for p < len(l.src) {
- c := l.src[p]
- if c == '.' {
- isReal = true
- p++
- continue
- }
- if c >= '0' && c <= '9' {
- p++
- continue
- }
- break
- }
- tok := Token{Bytes: l.src[l.pos:p], Offset: int64(start)}
- if isReal {
- tok.Kind = Real
- } else {
- tok.Kind = Integer
- }
- l.pos = p
- return tok, nil
-}
-
-func (l *Lexer) readKeyword(start int) (Token, error) {
- p := l.pos
- for p < len(l.src) && IsRegular(l.src[p]) {
- p++
- }
- if p == l.pos {
- return Token{}, fmt.Errorf("pdfdisassembler/lex: stuck at byte 0x%02x at %d", l.src[l.pos], l.pos)
- }
- tok := Token{Kind: Keyword, Bytes: l.src[l.pos:p], Offset: int64(start)}
- l.pos = p
- return tok, nil
-}
-
-func hexDigit(c byte) (int, bool) {
- switch {
- case c >= '0' && c <= '9':
- return int(c - '0'), true
- case c >= 'a' && c <= 'f':
- return int(c-'a') + 10, true
- case c >= 'A' && c <= 'F':
- return int(c-'A') + 10, true
- }
- return 0, false
-}
-
-// ReadStreamData consumes raw stream bytes of the given length, starting
-// at the current position. It honours the spec's EOL handling: a single
-// LF or CRLF *immediately* after the "stream" keyword is part of the
-// keyword line, not the stream content. Callers should call this after
-// the "stream" keyword token has been consumed.
-func (l *Lexer) ReadStreamData(length int) ([]byte, error) {
- // Skip optional CR LF or single LF following "stream".
- if l.pos < len(l.src) && l.src[l.pos] == '\r' {
- l.pos++
- }
- if l.pos < len(l.src) && l.src[l.pos] == '\n' {
- l.pos++
- }
- // length is the attacker-controlled /Length; compare against the bytes
- // remaining rather than computing l.pos+length, which can overflow.
- if length < 0 || length > len(l.src)-l.pos {
- return nil, ErrUnexpectedEOF
- }
- out := l.src[l.pos : l.pos+length]
- l.pos += length
- return out, nil
-}
diff --git a/internal/lex/lex_test.go b/internal/lex/lex_test.go
deleted file mode 100644
index 0d551ab..0000000
--- a/internal/lex/lex_test.go
+++ /dev/null
@@ -1,285 +0,0 @@
-package lex
-
-import (
- "bytes"
- "testing"
-)
-
-func TestLexerNamesAndNumbers(t *testing.T) {
- src := []byte("/Length 12 -3 +4 5.6 .7 8. true false null")
- lx := New(src)
- want := []struct {
- kind Kind
- bytes string
- }{
- {Name, "Length"},
- {Integer, "12"},
- {Integer, "-3"},
- {Integer, "+4"},
- {Real, "5.6"},
- {Real, ".7"},
- {Real, "8."},
- {Keyword, "true"},
- {Keyword, "false"},
- {Keyword, "null"},
- {EOF, ""},
- }
- for i, w := range want {
- tok, err := lx.Next()
- if err != nil {
- t.Fatalf("tok %d: err %v", i, err)
- }
- if tok.Kind != w.kind {
- t.Fatalf("tok %d: kind %v want %v", i, tok.Kind, w.kind)
- }
- if string(tok.Bytes) != w.bytes {
- t.Fatalf("tok %d: bytes %q want %q", i, tok.Bytes, w.bytes)
- }
- }
-}
-
-func TestLexerNameHashEscape(t *testing.T) {
- src := []byte("/A#20B /ABC")
- lx := New(src)
- t1, _ := lx.Next()
- if string(t1.Bytes) != "A B" {
- t.Fatalf("got %q want %q", t1.Bytes, "A B")
- }
- t2, _ := lx.Next()
- if string(t2.Bytes) != "ABC" {
- t.Fatalf("got %q want %q", t2.Bytes, "ABC")
- }
-}
-
-func TestLexerLiteralString(t *testing.T) {
- cases := []struct {
- in string
- want string
- }{
- {"(hello)", "hello"},
- {"(a (nested) b)", "a (nested) b"},
- {"(line\\nbreak)", "line\nbreak"},
- {"(\\053\\053)", "++"},
- {"(\\\\)", "\\"},
- {"(a\\\nb)", "ab"},
- }
- for _, c := range cases {
- lx := New([]byte(c.in))
- tok, err := lx.Next()
- if err != nil {
- t.Fatalf("%q: %v", c.in, err)
- }
- if tok.Kind != LitString {
- t.Fatalf("%q: kind %v", c.in, tok.Kind)
- }
- if string(tok.Bytes) != c.want {
- t.Fatalf("%q: got %q want %q", c.in, tok.Bytes, c.want)
- }
- }
-}
-
-func TestLexerHexString(t *testing.T) {
- src := []byte("<48656C6C6F>")
- tok, err := New(src).Next()
- if err != nil {
- t.Fatal(err)
- }
- if tok.Kind != HexString {
- t.Fatalf("kind %v", tok.Kind)
- }
- if string(tok.Bytes) != "Hello" {
- t.Fatalf("got %q", tok.Bytes)
- }
-}
-
-func TestLexerHexStringOddNibble(t *testing.T) {
- src := []byte("")
- tok, err := New(src).Next()
- if err != nil {
- t.Fatal(err)
- }
- if len(tok.Bytes) != 1 || tok.Bytes[0] != 0xF0 {
- t.Fatalf("got % x", tok.Bytes)
- }
-}
-
-func TestLexerDictArrayDelims(t *testing.T) {
- src := []byte("<< /A 1 >> [ 1 2 3 ]")
- lx := New(src)
- kinds := []Kind{DictStart, Name, Integer, DictEnd, ArrayStart, Integer, Integer, Integer, ArrayEnd, EOF}
- for i, k := range kinds {
- tok, err := lx.Next()
- if err != nil {
- t.Fatalf("tok %d: %v", i, err)
- }
- if tok.Kind != k {
- t.Fatalf("tok %d: kind %v want %v", i, tok.Kind, k)
- }
- }
-}
-
-func TestLexerComment(t *testing.T) {
- src := []byte("% comment\n1 % trailing\n2")
- lx := New(src)
- t1, _ := lx.Next()
- t2, _ := lx.Next()
- t3, _ := lx.Next()
- if t1.Kind != Integer || string(t1.Bytes) != "1" {
- t.Fatalf("t1: %v %q", t1.Kind, t1.Bytes)
- }
- if t2.Kind != Integer || string(t2.Bytes) != "2" {
- t.Fatalf("t2: %v %q", t2.Kind, t2.Bytes)
- }
- if t3.Kind != EOF {
- t.Fatalf("t3 kind %v", t3.Kind)
- }
-}
-
-func TestReadStreamDataHostileLength(t *testing.T) {
- const maxInt = int(^uint(0) >> 1)
- for _, length := range []int{-1, -1000, maxInt} {
- // Leading "\n" makes the EOL skip advance pos, so pos+length overflows.
- l := New([]byte("\nstream body bytes"))
- if _, err := l.ReadStreamData(length); err == nil {
- t.Fatalf("length=%d: expected error, got nil", length)
- }
- }
-}
-
-func TestReadStreamDataValid(t *testing.T) {
- l := New([]byte("\nABCDEF"))
- out, err := l.ReadStreamData(6)
- if err != nil {
- t.Fatalf("ReadStreamData: %v", err)
- }
- if string(out) != "ABCDEF" {
- t.Fatalf("got %q want ABCDEF", out)
- }
-}
-
-func TestLexerTruncatedTokensError(t *testing.T) {
- cases := map[string]string{
- "name escape at eof": "/AB#",
- "name escape one digit": "/AB#F",
- "name escape bad hex": "/A#GG",
- "string backslash at eof": "(abc\\",
- "unterminated string": "(abc",
- "unterminated hex": "<48",
- "bad hex digit": "<4G>",
- }
- for name, src := range cases {
- t.Run(name, func(t *testing.T) {
- if _, err := New([]byte(src)).Next(); err == nil {
- t.Fatalf("%q: expected error, got nil", src)
- }
- })
- }
-}
-
-func TestLexerStringEscapes(t *testing.T) {
- lx := New([]byte(`(\101\n\)\(end)`)) // octal 'A', newline, literal ) and (
- tok, err := lx.Next()
- if err != nil {
- t.Fatalf("Next: %v", err)
- }
- if want := "A\n)(end"; string(tok.Bytes) != want {
- t.Fatalf("got %q want %q", tok.Bytes, want)
- }
-}
-
-// FuzzLexer asserts tokenisation never panics on arbitrary input.
-func FuzzLexer(f *testing.F) {
- f.Add([]byte("/Name 123 -4.5 (str\\n) <48656C> << /A 1 >> [ 1 2 ] true null %c"))
- f.Fuzz(func(t *testing.T, data []byte) {
- lx := New(data)
- for i := 0; i <= len(data); i++ {
- tok, err := lx.Next()
- if err != nil || tok.Kind == EOF {
- break
- }
- }
- })
-}
-
-// Position accessors that back the xref reader's manual seeking.
-func TestLexerAccessors(t *testing.T) {
- src := []byte("hello world")
- lx := New(src)
- if lx.Pos() != 0 {
- t.Errorf("initial Pos = %d, want 0", lx.Pos())
- }
- if !bytes.Equal(lx.Source(), src) {
- t.Errorf("Source = %q, want %q", lx.Source(), src)
- }
- lx.SetPos(6)
- if lx.Pos() != 6 {
- t.Errorf("Pos after SetPos(6) = %d, want 6", lx.Pos())
- }
- if got := lx.Remaining(); string(got) != "world" {
- t.Errorf("Remaining = %q, want world", got)
- }
-}
-
-// Literal-string escapes/newlines (§7.3.4.2) not already in TestLexerLiteralString.
-func TestLexerLiteralStringEscapes(t *testing.T) {
- cases := []struct {
- name, in, want string
- }{
- {"named_escapes", "(\\t\\b\\f\\r)", "\t\b\f\r"},
- {"cr_line_continuation", "(a\\\rb)", "ab"},
- {"crlf_line_continuation", "(a\\\r\nb)", "ab"},
- {"unknown_escape_keeps_char", "(\\q)", "q"},
- {"bare_cr_to_lf", "(a\rb)", "a\nb"},
- {"bare_crlf_to_lf", "(a\r\nb)", "a\nb"},
- }
- for _, c := range cases {
- t.Run(c.name, func(t *testing.T) {
- tok, err := New([]byte(c.in)).Next()
- if err != nil {
- t.Fatalf("Next: %v", err)
- }
- if tok.Kind != LitString {
- t.Fatalf("kind = %v, want LitString", tok.Kind)
- }
- if string(tok.Bytes) != c.want {
- t.Errorf("got %q, want %q", tok.Bytes, c.want)
- }
- })
- }
-}
-
-func TestLexerLiteralStringErrors(t *testing.T) {
- for _, in := range []string{"(\\", "(abc"} { // trailing backslash; unterminated
- if _, err := New([]byte(in)).Next(); err == nil {
- t.Errorf("%q: expected an error, got nil", in)
- }
- }
-}
-
-// A misclassified delimiter byte (§7.2.2) would break tokenisation, so pin the set.
-func TestIsDelimiter(t *testing.T) {
- for _, c := range []byte("()<>[]{}/%") {
- if !IsDelimiter(c) {
- t.Errorf("IsDelimiter(%q) = false, want true", c)
- }
- }
- for _, c := range []byte("aZ0 \t.\\") {
- if IsDelimiter(c) {
- t.Errorf("IsDelimiter(%q) = true, want false", c)
- }
- }
-}
-
-func TestHexDigit(t *testing.T) {
- valid := map[byte]int{'0': 0, '9': 9, 'a': 10, 'f': 15, 'A': 10, 'F': 15}
- for c, want := range valid {
- if v, ok := hexDigit(c); !ok || v != want {
- t.Errorf("hexDigit(%q) = (%d, %v), want (%d, true)", c, v, ok, want)
- }
- }
- for _, c := range []byte("gG/ \x00") {
- if _, ok := hexDigit(c); ok {
- t.Errorf("hexDigit(%q) = ok, want not-ok", c)
- }
- }
-}
diff --git a/object.go b/object.go
deleted file mode 100644
index c2aba1c..0000000
--- a/object.go
+++ /dev/null
@@ -1,335 +0,0 @@
-package pdfdisassembler
-
-import (
- "iter"
- "time"
-)
-
-// Object is the sealed PDF object type. Concrete variants:
-//
-// Name, Integer, Real, Bool, String, *Dict, Array, *Stream, Reference, Null
-//
-// Callers should type-switch or type-assert on the concrete type when they
-// need a specific value. The Resolve* helpers on Reader perform the common
-// dereference-and-assert pattern.
-type Object interface {
- object()
-}
-
-// Name is a PDF name object, e.g. /Length, /Type. The leading slash is not
-// stored.
-type Name string
-
-// Integer is a PDF integer object.
-type Integer int64
-
-// Real is a PDF real-number object.
-type Real float64
-
-// Bool is a PDF boolean object.
-type Bool bool
-
-// String is a PDF string object after parsing. The parser strips the
-// literal-string parentheses or hex-string angle brackets and decodes
-// escape sequences and hexadecimal pairs, but does not re-encode the
-// bytes — they are whatever the producer wrote.
-//
-// For PDF text strings ("Title", "Subject", "Producer", and so on) use
-// Dict.String or DocumentInfo, which apply the text-string decoding rules
-// (PDFDocEncoding, UTF-16BE BOM, UTF-8 BOM).
-type String []byte
-
-// Array is a PDF array of objects.
-type Array []Object
-
-// Reference is an indirect object reference (e.g. "12 0 R").
-type Reference struct {
- Number int
- Generation int
-}
-
-// Null is the PDF null object.
-type Null struct{}
-
-func (Name) object() {}
-func (Integer) object() {}
-func (Real) object() {}
-func (Bool) object() {}
-func (String) object() {}
-func (Array) object() {}
-func (Reference) object() {}
-func (Null) object() {}
-func (*Dict) object() {}
-func (*Stream) object() {}
-
-// Dict is a PDF dictionary that preserves insertion order during iteration.
-type Dict struct {
- keys []string
- values map[string]Object
- // reader is set by the parser so the dereferencing convenience methods
- // (Dict, Array-of-dicts walks) can follow indirect references. nil
- // when the dictionary was synthesised outside of a Reader.
- reader *Reader
-}
-
-// newDict returns an empty dictionary tied to r (may be nil during early
-// parsing of trailer/xref dicts).
-func newDict(r *Reader) *Dict {
- return &Dict{values: map[string]Object{}, reader: r}
-}
-
-// Len returns the number of entries in the dictionary.
-func (d *Dict) Len() int {
- if d == nil {
- return 0
- }
- return len(d.keys)
-}
-
-// Get returns the raw object for key. The returned object may be a
-// Reference; use Dict.Dict / Reader.Resolve if you need the resolved value.
-func (d *Dict) Get(key string) (Object, bool) {
- if d == nil {
- return nil, false
- }
- v, ok := d.values[key]
- return v, ok
-}
-
-// Has reports whether key is present.
-func (d *Dict) Has(key string) bool {
- if d == nil {
- return false
- }
- _, ok := d.values[key]
- return ok
-}
-
-// Keys returns the dictionary keys in insertion order.
-func (d *Dict) Keys() []string {
- if d == nil {
- return nil
- }
- out := make([]string, len(d.keys))
- copy(out, d.keys)
- return out
-}
-
-// Iter returns an iterator over key/value pairs in insertion order.
-func (d *Dict) Iter() iter.Seq2[string, Object] {
- return func(yield func(string, Object) bool) {
- if d == nil {
- return
- }
- for _, k := range d.keys {
- if !yield(k, d.values[k]) {
- return
- }
- }
- }
-}
-
-// set inserts or updates an entry, preserving insertion order on first set.
-func (d *Dict) set(key string, value Object) {
- if _, ok := d.values[key]; !ok {
- d.keys = append(d.keys, key)
- }
- d.values[key] = value
-}
-
-// Name returns the Name value at key. If the value is a Reference, it is
-// resolved first.
-func (d *Dict) Name(key string) (Name, bool) {
- v, ok := d.resolved(key)
- if !ok {
- return "", false
- }
- n, ok := v.(Name)
- return n, ok
-}
-
-// Int returns the Integer value at key. If the value is a Reference, it is
-// resolved first.
-func (d *Dict) Int(key string) (int64, bool) {
- v, ok := d.resolved(key)
- if !ok {
- return 0, false
- }
- n, ok := v.(Integer)
- if !ok {
- return 0, false
- }
- return int64(n), true
-}
-
-// Bool returns the Bool value at key. If the value is a Reference, it is
-// resolved first.
-func (d *Dict) Bool(key string) (bool, bool) {
- v, ok := d.resolved(key)
- if !ok {
- return false, false
- }
- b, ok := v.(Bool)
- return bool(b), ok
-}
-
-// Array returns the Array value at key. If the value is a Reference, it is
-// resolved first.
-func (d *Dict) Array(key string) (Array, bool) {
- v, ok := d.resolved(key)
- if !ok {
- return nil, false
- }
- a, ok := v.(Array)
- return a, ok
-}
-
-// Dict returns the *Dict value at key. If the value is a Reference, it is
-// resolved first.
-func (d *Dict) Dict(key string) (*Dict, bool) {
- v, ok := d.resolved(key)
- if !ok {
- return nil, false
- }
- dd, ok := v.(*Dict)
- return dd, ok
-}
-
-// Stream returns the *Stream value at key. If the value is a Reference, it
-// is resolved first.
-func (d *Dict) Stream(key string) (*Stream, bool) {
- v, ok := d.resolved(key)
- if !ok {
- return nil, false
- }
- s, ok := v.(*Stream)
- return s, ok
-}
-
-// String returns the value at key as a Go string, decoded according to the
-// PDF text-string rules (UTF-16BE BOM, UTF-8 BOM, otherwise PDFDocEncoding).
-// If the value is a Reference, it is resolved first.
-func (d *Dict) String(key string) (string, bool) {
- v, ok := d.resolved(key)
- if !ok {
- return "", false
- }
- s, ok := v.(String)
- if !ok {
- return "", false
- }
- return decodeTextString(s), true
-}
-
-// Bytes returns the raw bytes of a String value at key, without
-// text-string decoding. Useful for byte strings (file identifiers, hashes).
-// If the value is a Reference, it is resolved first.
-func (d *Dict) Bytes(key string) ([]byte, bool) {
- v, ok := d.resolved(key)
- if !ok {
- return nil, false
- }
- s, ok := v.(String)
- if !ok {
- return nil, false
- }
- return []byte(s), true
-}
-
-// resolved returns the value at key, dereferencing once through the
-// reader if the value is a Reference. If resolution fails, ok is false.
-func (d *Dict) resolved(key string) (Object, bool) {
- v, ok := d.Get(key)
- if !ok {
- return nil, false
- }
- if ref, ok := v.(Reference); ok {
- if d.reader == nil {
- return nil, false
- }
- obj, err := d.reader.Resolve(ref)
- if err != nil {
- return nil, false
- }
- return obj, true
- }
- return v, true
-}
-
-// Stream is a stream object. The decoded content is produced by Content,
-// which applies the declared filter chain (FlateDecode, ASCII85, …) and
-// any document-level decryption, and caches the result.
-type Stream struct {
- // Dict is the stream's parameter dictionary, e.g. /Length, /Filter.
- Dict *Dict
- // reader is the document the stream came from.
- reader *Reader
- // rawOffset is the byte offset in the underlying ReadSeeker where the
- // raw stream bytes begin (just after the "stream" keyword and EOL).
- rawOffset int64
- // rawLength is the raw byte count of the stream as declared by /Length.
- rawLength int64
- // objNumber, objGeneration identify the indirect object this stream
- // belongs to. Used for per-object decryption keys.
- objNumber int
- objGeneration int
- // cache holds the decoded content after the first call to Content.
- cache []byte
- cacheErr error
- cached bool
-}
-
-// Content returns the decoded stream bytes. Filters and decryption are
-// applied on the first call and the result is cached for subsequent calls.
-func (s *Stream) Content() ([]byte, error) {
- if s.cached {
- return s.cache, s.cacheErr
- }
- b, err := s.reader.decodeStream(s)
- s.cache = b
- s.cacheErr = err
- s.cached = true
- return b, err
-}
-
-// RawLength returns the declared raw byte length of the stream.
-func (s *Stream) RawLength() int64 {
- return s.rawLength
-}
-
-// RawBytes returns a copy of the stream's raw, undecoded bytes exactly as they
-// appear in the file — before any filter or decryption is applied. For the
-// decoded content, use Content.
-func (s *Stream) RawBytes() ([]byte, error) {
- return s.reader.rawStreamBytes(s)
-}
-
-// ObjectEntry is yielded by Reader.Objects: an in-use indirect object plus
-// its resolved value.
-type ObjectEntry struct {
- Reference Reference
- Object Object
-}
-
-// DocInfo is a value snapshot of the standard /Info dictionary entries.
-// Missing entries are zero values. Custom carries any non-standard keys
-// as raw decoded strings.
-type DocInfo struct {
- Title string
- Author string
- Subject string
- Keywords string
- Creator string
- Producer string
- CreationDate time.Time
- ModDate time.Time
- Custom map[string]string
-}
-
-// EmbeddedFile is one entry from the catalog's EmbeddedFiles name tree (a PDF
-// attachment). Spec is the /Filespec dictionary; its /EF stream holds the
-// bytes.
-type EmbeddedFile struct {
- Name string
- Spec *Dict
-}
diff --git a/object_test.go b/object_test.go
deleted file mode 100644
index 99eb64e..0000000
--- a/object_test.go
+++ /dev/null
@@ -1,179 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "testing"
-)
-
-func TestDictAccessors(t *testing.T) {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R /IntVal 42 /NegInt -7 /BoolVal true " +
- "/StrVal (hi) /NameVal /Foo /ArrVal [ 1 2 3 ] /DictRef 3 0 R >>",
- "<< /Type /Pages /Kids [] /Count 0 >>",
- "<< /Inner (deep) >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- cat, err := r.Catalog()
- if err != nil {
- t.Fatalf("Catalog: %v", err)
- }
-
- if v, ok := cat.Int("IntVal"); !ok || v != 42 {
- t.Errorf("Int(IntVal) = %d, %v", v, ok)
- }
- if v, ok := cat.Int("NegInt"); !ok || v != -7 {
- t.Errorf("Int(NegInt) = %d, %v", v, ok)
- }
- if v, ok := cat.Bool("BoolVal"); !ok || !v {
- t.Errorf("Bool(BoolVal) = %v, %v", v, ok)
- }
- if v, ok := cat.String("StrVal"); !ok || v != "hi" {
- t.Errorf("String(StrVal) = %q, %v", v, ok)
- }
- if v, ok := cat.Bytes("StrVal"); !ok || string(v) != "hi" {
- t.Errorf("Bytes(StrVal) = %q, %v", v, ok)
- }
- if v, ok := cat.Name("NameVal"); !ok || v != "Foo" {
- t.Errorf("Name(NameVal) = %q, %v", v, ok)
- }
- if v, ok := cat.Array("ArrVal"); !ok || len(v) != 3 {
- t.Errorf("Array(ArrVal) len = %d, %v", len(v), ok)
- }
-
- // Dict() follows the indirect reference to object 3.
- inner, ok := cat.Dict("DictRef")
- if !ok {
- t.Fatal("Dict(DictRef) not resolved")
- }
- if s, ok := inner.String("Inner"); !ok || s != "deep" {
- t.Errorf("resolved inner String(Inner) = %q, %v", s, ok)
- }
-
- if !cat.Has("IntVal") || cat.Has("Missing") {
- t.Error("Has wrong")
- }
- if cat.Len() < 8 {
- t.Errorf("Len = %d, want >= 8", cat.Len())
- }
- seen := map[string]bool{}
- for _, k := range cat.Keys() {
- seen[k] = true
- }
- for k := range cat.Iter() {
- if !seen[k] {
- t.Errorf("Iter yielded %q absent from Keys", k)
- }
- }
- if !seen["IntVal"] || !seen["ArrVal"] {
- t.Errorf("Keys missing entries: %v", cat.Keys())
- }
-
- // Type mismatches must report ok=false, not panic or coerce.
- if _, ok := cat.Int("StrVal"); ok {
- t.Error("Int on a string")
- }
- if _, ok := cat.Bool("IntVal"); ok {
- t.Error("Bool on an int")
- }
- if _, ok := cat.Name("IntVal"); ok {
- t.Error("Name on an int")
- }
- if _, ok := cat.Array("IntVal"); ok {
- t.Error("Array on an int")
- }
- if _, ok := cat.Dict("IntVal"); ok {
- t.Error("Dict on an int")
- }
- if _, ok := cat.Stream("IntVal"); ok {
- t.Error("Stream on an int")
- }
- if _, ok := cat.String("IntVal"); ok {
- t.Error("String on an int")
- }
- if _, ok := cat.Bytes("IntVal"); ok {
- t.Error("Bytes on an int")
- }
- if _, ok := cat.Int("Missing"); ok {
- t.Error("Int on a missing key")
- }
-}
-
-// Every accessor must be safe on a nil *Dict (the common "key absent" result).
-func TestDictNilReceiver(t *testing.T) {
- var d *Dict
- if d.Len() != 0 {
- t.Error("Len")
- }
- if d.Has("x") {
- t.Error("Has")
- }
- if d.Keys() != nil {
- t.Error("Keys")
- }
- if _, ok := d.Get("x"); ok {
- t.Error("Get")
- }
- for range d.Iter() {
- t.Error("Iter on nil yielded an entry")
- }
-}
-
-// The typed getters must miss (ok=false) — not coerce or fabricate a zero — on
-// a missing key, a wrong type, or an unresolvable Reference (nil reader).
-func TestDictTypedGetterMisses(t *testing.T) {
- d := newDict(nil)
- d.set("i", Integer(5))
- d.set("b", Bool(true))
- d.set("s", String("hi"))
-
- wrongType := []struct {
- name string
- ok bool
- }{
- {"Bool", func() bool { _, ok := d.Bool("i"); return ok }()},
- {"Int", func() bool { _, ok := d.Int("b"); return ok }()},
- {"Array", func() bool { _, ok := d.Array("i"); return ok }()},
- {"Dict", func() bool { _, ok := d.Dict("i"); return ok }()},
- {"Stream", func() bool { _, ok := d.Stream("i"); return ok }()},
- {"String", func() bool { _, ok := d.String("i"); return ok }()},
- {"Bytes", func() bool { _, ok := d.Bytes("i"); return ok }()},
- }
- for _, tc := range wrongType {
- if tc.ok {
- t.Errorf("%s on a wrong-typed value should miss", tc.name)
- }
- }
-
- missing := []bool{
- func() bool { _, ok := d.Int("x"); return ok }(),
- func() bool { _, ok := d.Bool("x"); return ok }(),
- func() bool { _, ok := d.Array("x"); return ok }(),
- func() bool { _, ok := d.Dict("x"); return ok }(),
- func() bool { _, ok := d.Stream("x"); return ok }(),
- func() bool { _, ok := d.String("x"); return ok }(),
- func() bool { _, ok := d.Bytes("x"); return ok }(),
- }
- for i, ok := range missing {
- if ok {
- t.Errorf("getter %d on a missing key should miss", i)
- }
- }
-
- // A Reference with no backing reader can't be dereferenced.
- d.set("ref", Reference{Number: 9})
- if _, ok := d.Int("ref"); ok {
- t.Error("Reference with nil reader should miss")
- }
-
- // Controls: the right type resolves.
- if v, ok := d.Int("i"); !ok || v != 5 {
- t.Errorf("Int(i) = %d, %v; want 5, true", v, ok)
- }
- if s, ok := d.Bytes("s"); !ok || string(s) != "hi" {
- t.Errorf("Bytes(s) = %q, %v; want hi, true", s, ok)
- }
-}
diff --git a/page.go b/page.go
deleted file mode 100644
index cb47a7c..0000000
--- a/page.go
+++ /dev/null
@@ -1,360 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "errors"
- "fmt"
-)
-
-// maxPageTreeDepth bounds both the page-tree (/Kids) descent and the
-// inheritance (/Parent) walk so a hostile or cyclic structure can't loop
-// forever or overflow the stack.
-const maxPageTreeDepth = 1000
-
-// BoxName identifies one of the page boundary boxes (PDF 32000-1:2008
-// §14.11.2).
-type BoxName string
-
-// The five page boundary boxes, in the spec's containment order (MediaBox is
-// the largest, ArtBox the smallest).
-const (
- MediaBox BoxName = "MediaBox"
- CropBox BoxName = "CropBox"
- BleedBox BoxName = "BleedBox"
- TrimBox BoxName = "TrimBox"
- ArtBox BoxName = "ArtBox"
-)
-
-// boxNames is the canonical box list iterated by Page.Boxes.
-var boxNames = []BoxName{MediaBox, CropBox, BleedBox, TrimBox, ArtBox}
-
-// Rect is a PDF rectangle in default user-space units (points), normalised so
-// LLX <= URX and LLY <= URY regardless of the corner order written in the file.
-type Rect struct {
- LLX, LLY, URX, URY float64
-}
-
-// Width returns the rectangle's horizontal extent.
-func (r Rect) Width() float64 { return r.URX - r.LLX }
-
-// Height returns the rectangle's vertical extent.
-func (r Rect) Height() float64 { return r.URY - r.LLY }
-
-// Page is a handle to a single leaf page (/Type /Page) of the page tree. Its
-// accessors resolve the inheritable attributes — boxes, /Rotate, /Resources —
-// by walking the /Parent chain per PDF 32000-1:2008 §7.7.3.4. Obtain one via
-// Reader.Page or Reader.Pages.
-type Page struct {
- reader *Reader
- dict *Dict
- index int
-}
-
-// Index returns the page's 0-based position in display order.
-func (p *Page) Index() int { return p.index }
-
-// Dict returns the page's own dictionary, without inherited attributes
-// flattened in. Use the Box, Rotation, and Resources accessors for values that
-// may be inherited from an ancestor /Pages node.
-func (p *Page) Dict() *Dict { return p.dict }
-
-// PageCount returns the number of leaf pages in the document.
-func (r *Reader) PageCount() (int, error) {
- pages, err := r.loadPages()
- if err != nil {
- return 0, err
- }
- return len(pages), nil
-}
-
-// Pages returns every leaf page in display (reading) order.
-func (r *Reader) Pages() ([]*Page, error) {
- pages, err := r.loadPages()
- if err != nil {
- return nil, err
- }
- out := make([]*Page, len(pages))
- copy(out, pages)
- return out, nil
-}
-
-// Page returns the leaf page at the given 0-based index in display order.
-func (r *Reader) Page(index int) (*Page, error) {
- pages, err := r.loadPages()
- if err != nil {
- return nil, err
- }
- if index < 0 || index >= len(pages) {
- return nil, fmt.Errorf("pdfdisassembler: page index %d out of range (%d pages)", index, len(pages))
- }
- return pages[index], nil
-}
-
-// loadPages walks the page tree once and caches the flat leaf-page list (and
-// any error) for subsequent calls.
-func (r *Reader) loadPages() ([]*Page, error) {
- if r.pagesLoaded {
- return r.pages, r.pagesErr
- }
- r.pagesLoaded = true
- r.pages, r.pagesErr = r.buildPages()
- return r.pages, r.pagesErr
-}
-
-func (r *Reader) buildPages() ([]*Page, error) {
- cat, err := r.Catalog()
- if err != nil {
- return nil, err
- }
- rootRef, ok := cat.Get("Pages")
- if !ok {
- return nil, errors.New("pdfdisassembler: catalog has no /Pages")
- }
- root, err := r.ResolveDict(rootRef)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: resolve /Pages: %w", err)
- }
- seen := map[Reference]struct{}{}
- if ref, ok := rootRef.(Reference); ok {
- seen[ref] = struct{}{}
- }
- var out []*Page
- r.collectPages(root, seen, 0, &out)
- return out, nil
-}
-
-// collectPages descends the page tree depth-first, appending each leaf page to
-// out in display order. A node is treated as an intermediate /Pages node when
-// it carries /Kids, otherwise as a leaf /Page — /Type is only a hint, since
-// some producers omit it. seen guards against cyclic /Kids references and depth
-// bounds the descent.
-func (r *Reader) collectPages(node *Dict, seen map[Reference]struct{}, depth int, out *[]*Page) {
- if node == nil || depth > maxPageTreeDepth {
- return
- }
- kids, ok := node.Array("Kids")
- if !ok {
- // No resolvable /Kids array. A node that still looks like an
- // intermediate /Pages node — it declares /Type /Pages, /Count, or a
- // /Kids key that failed to resolve — is a broken or empty branch, not a
- // leaf page; emitting it would invent a phantom page.
- if node.Has("Kids") || node.Has("Count") {
- return
- }
- if t, ok := node.Name("Type"); ok && t == "Pages" {
- return
- }
- *out = append(*out, &Page{reader: r, dict: node, index: len(*out)})
- return
- }
- for _, kid := range kids {
- if ref, ok := kid.(Reference); ok {
- if _, dup := seen[ref]; dup {
- continue
- }
- seen[ref] = struct{}{}
- }
- child, err := r.ResolveDict(kid)
- if err != nil {
- continue
- }
- r.collectPages(child, seen, depth+1, out)
- }
-}
-
-// inherited walks the /Parent chain starting at the page dictionary and returns
-// the resolved value of the first ancestor that carries key. ok is false when
-// no ancestor defines it. Resolution is cached, so the same /Parent reference
-// yields the same *Dict pointer — pointer identity (plus the depth bound)
-// terminates a cyclic chain.
-func (p *Page) inherited(key string) (Object, bool) {
- seen := map[*Dict]struct{}{}
- node := p.dict
- for depth := 0; node != nil && depth <= maxPageTreeDepth; depth++ {
- if _, dup := seen[node]; dup {
- return nil, false
- }
- seen[node] = struct{}{}
- if v, ok := node.resolved(key); ok {
- return v, true
- }
- parent, ok := node.Dict("Parent")
- if !ok {
- return nil, false
- }
- node = parent
- }
- return nil, false
-}
-
-// Box returns the named page boundary box. MediaBox and CropBox are resolved
-// through inheritance along the /Parent chain; BleedBox, TrimBox and ArtBox are
-// read from the page object only, since only MediaBox and CropBox carry the
-// (Inheritable) marker in PDF 32000-1:2008 Table 30 (§14.11.2). ok is false
-// when the box is not defined there, or its value is not a well-formed array of
-// four numbers. No spec-default substitution (e.g. a missing box defaulting to
-// CropBox) is applied.
-func (p *Page) Box(name BoxName) (Rect, bool) {
- var v Object
- var ok bool
- if name == MediaBox || name == CropBox {
- v, ok = p.inherited(string(name))
- } else {
- v, ok = p.dict.resolved(string(name))
- }
- if !ok {
- return Rect{}, false
- }
- arr, ok := v.(Array)
- if !ok {
- return Rect{}, false
- }
- return rectFromArray(p.reader, arr)
-}
-
-// Boxes returns every boundary box defined for the page, keyed by name, with
-// the same per-box inheritance rules as Box (MediaBox and CropBox may be
-// inherited from an ancestor; BleedBox/TrimBox/ArtBox are page-level only).
-// Boxes that are not defined are omitted; no spec-default substitution (e.g. a
-// missing CropBox defaulting to MediaBox) is applied.
-func (p *Page) Boxes() map[BoxName]Rect {
- out := map[BoxName]Rect{}
- for _, name := range boxNames {
- if rect, ok := p.Box(name); ok {
- out[name] = rect
- }
- }
- return out
-}
-
-// Rotation returns the page's clockwise display rotation in degrees, resolved
-// through inheritance and normalised to one of 0, 90, 180, 270. A missing,
-// non-integer, or non-multiple-of-90 /Rotate yields 0.
-func (p *Page) Rotation() int {
- v, ok := p.inherited("Rotate")
- if !ok {
- return 0
- }
- n, ok := v.(Integer)
- if !ok {
- return 0
- }
- deg := int(n) % 360
- if deg < 0 {
- deg += 360
- }
- if deg%90 != 0 {
- return 0
- }
- return deg
-}
-
-// Resources returns the page's resource dictionary, resolved through
-// inheritance along the /Parent chain. ok is false when neither the page nor
-// any ancestor defines /Resources.
-func (p *Page) Resources() (*Dict, bool) {
- v, ok := p.inherited("Resources")
- if !ok {
- return nil, false
- }
- d, ok := v.(*Dict)
- return d, ok
-}
-
-// ContentStreams returns the page's content streams in order. /Contents may be
-// a single stream or an array of streams (PDF 32000-1:2008 §7.7.3.3); both are
-// flattened to the underlying Stream objects. It returns nil (no error) when
-// the page has no /Contents. Non-stream entries are skipped defensively.
-func (p *Page) ContentStreams() ([]*Stream, error) {
- v, ok := p.dict.Get("Contents")
- if !ok {
- return nil, nil
- }
- resolved, err := p.reader.Resolve(v)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: resolve /Contents: %w", err)
- }
- var out []*Stream
- switch t := resolved.(type) {
- case *Stream:
- out = append(out, t)
- case Array:
- for _, e := range t {
- s, err := p.reader.Resolve(e)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: resolve /Contents entry: %w", err)
- }
- stm, ok := s.(*Stream)
- if !ok {
- // A missing entry resolves to Null, not an error; dropping it
- // would silently truncate the page's drawing instructions, so
- // fail loudly instead.
- return nil, fmt.Errorf("pdfdisassembler: /Contents array entry resolved to %T, want stream", s)
- }
- out = append(out, stm)
- }
- }
- return out, nil
-}
-
-// Content returns the page's decoded content: every content stream decoded via
-// its filter chain and concatenated with a single newline between streams (a
-// token cannot span a stream boundary, per PDF 32000-1:2008 §7.8.2). The result
-// is the raw drawing-instruction byte stream; this library does not interpret
-// it (see package contentstream for tokenisation).
-func (p *Page) Content() ([]byte, error) {
- streams, err := p.ContentStreams()
- if err != nil {
- return nil, err
- }
- parts := make([][]byte, 0, len(streams))
- for _, s := range streams {
- data, err := s.Content()
- if err != nil {
- return nil, err
- }
- parts = append(parts, data)
- }
- return bytes.Join(parts, []byte{'\n'}), nil
-}
-
-// rectFromArray converts a 4-element PDF array [llx lly urx ury] to a
-// normalised Rect. Entries may be Integer or Real and may be indirect
-// references. ok is false for any other shape.
-func rectFromArray(r *Reader, arr Array) (Rect, bool) {
- if len(arr) != 4 {
- return Rect{}, false
- }
- var v [4]float64
- for i, e := range arr {
- f, ok := numberValue(r, e)
- if !ok {
- return Rect{}, false
- }
- v[i] = f
- }
- rect := Rect{LLX: v[0], LLY: v[1], URX: v[2], URY: v[3]}
- if rect.LLX > rect.URX {
- rect.LLX, rect.URX = rect.URX, rect.LLX
- }
- if rect.LLY > rect.URY {
- rect.LLY, rect.URY = rect.URY, rect.LLY
- }
- return rect, true
-}
-
-// numberValue resolves obj and returns it as a float64 if it is an Integer or
-// Real.
-func numberValue(r *Reader, obj Object) (float64, bool) {
- resolved, err := r.Resolve(obj)
- if err != nil {
- return 0, false
- }
- switch n := resolved.(type) {
- case Integer:
- return float64(n), true
- case Real:
- return float64(n), true
- }
- return 0, false
-}
diff --git a/page_test.go b/page_test.go
deleted file mode 100644
index 4d03986..0000000
--- a/page_test.go
+++ /dev/null
@@ -1,399 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "path/filepath"
- "testing"
-)
-
-func TestPageInheritanceFixture(t *testing.T) {
- r, err := OpenFile(filepath.Join("testdata", "fixtures", "page-inheritance", "input.pdf"))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
-
- n, err := r.PageCount()
- if err != nil {
- t.Fatalf("PageCount: %v", err)
- }
- if n != 2 {
- t.Fatalf("PageCount = %d, want 2", n)
- }
-
- // Page 0 inherits everything: MediaBox/Rotate/Resources two levels up
- // (the /Pages root), CropBox one level up.
- p0, err := r.Page(0)
- if err != nil {
- t.Fatalf("Page(0): %v", err)
- }
- if box, ok := p0.Box(MediaBox); !ok || box != (Rect{0, 0, 612, 792}) {
- t.Errorf("page 0 MediaBox = %+v ok=%v, want {0 0 612 792}", box, ok)
- }
- if box, ok := p0.Box(CropBox); !ok || box != (Rect{10, 10, 602, 782}) {
- t.Errorf("page 0 CropBox = %+v ok=%v, want {10 10 602 782}", box, ok)
- }
- if rot := p0.Rotation(); rot != 90 {
- t.Errorf("page 0 Rotation = %d, want 90", rot)
- }
- res, ok := p0.Resources()
- if !ok {
- t.Fatalf("page 0 Resources not found")
- }
- if _, ok := res.Dict("Font"); !ok {
- t.Errorf("page 0 Resources missing /Font: keys %v", res.Keys())
- }
- // Boxes that no ancestor defines are absent (no spec-default substitution).
- if box, ok := p0.Box(BleedBox); ok {
- t.Errorf("page 0 BleedBox = %+v, want absent", box)
- }
-
- // Page 1 overrides MediaBox and Rotate locally, still inherits CropBox.
- p1, err := r.Page(1)
- if err != nil {
- t.Fatalf("Page(1): %v", err)
- }
- if box, ok := p1.Box(MediaBox); !ok || box != (Rect{0, 0, 200, 200}) {
- t.Errorf("page 1 MediaBox = %+v ok=%v, want {0 0 200 200}", box, ok)
- }
- if box, ok := p1.Box(CropBox); !ok || box != (Rect{10, 10, 602, 782}) {
- t.Errorf("page 1 CropBox = %+v ok=%v, want inherited {10 10 602 782}", box, ok)
- }
- if rot := p1.Rotation(); rot != 0 {
- t.Errorf("page 1 Rotation = %d, want 0 (override)", rot)
- }
-}
-
-func TestPageContentsArrayFixture(t *testing.T) {
- r, err := OpenFile(filepath.Join("testdata", "fixtures", "page-contents-array", "input.pdf"))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
-
- p, err := r.Page(0)
- if err != nil {
- t.Fatalf("Page(0): %v", err)
- }
- streams, err := p.ContentStreams()
- if err != nil {
- t.Fatalf("ContentStreams: %v", err)
- }
- if len(streams) != 2 {
- t.Fatalf("got %d content streams, want 2", len(streams))
- }
- if raw, err := streams[0].RawBytes(); err != nil || string(raw) != "q 1 0 0 1 50 50 cm" {
- t.Errorf("stream 0 RawBytes = %q err=%v", raw, err)
- }
-
- content, err := p.Content()
- if err != nil {
- t.Fatalf("Content: %v", err)
- }
- want := "q 1 0 0 1 50 50 cm\nBT /F1 12 Tf (Hello) Tj ET"
- if string(content) != want {
- t.Errorf("Content = %q, want %q", content, want)
- }
-}
-
-func TestPageContentSingleStream(t *testing.T) {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Page /Parent 2 0 R /Contents 4 0 R >>",
- "<< /Length 5 >>\nstream\nhello\nendstream",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- p, err := r.Page(0)
- if err != nil {
- t.Fatalf("Page(0): %v", err)
- }
- if got, err := p.Content(); err != nil || string(got) != "hello" {
- t.Errorf("Content = %q err=%v, want \"hello\"", got, err)
- }
-}
-
-func TestPageContentNone(t *testing.T) {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Page /Parent 2 0 R >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- p, _ := r.Page(0)
- streams, err := p.ContentStreams()
- if err != nil || streams != nil {
- t.Errorf("ContentStreams = %v err=%v, want nil", streams, err)
- }
- if got, err := p.Content(); err != nil || len(got) != 0 {
- t.Errorf("Content = %q err=%v, want empty", got, err)
- }
-}
-
-func TestPageRotationNormalization(t *testing.T) {
- cases := []struct {
- rotate string
- want int
- }{
- {"450", 90},
- {"-90", 270},
- {"360", 0},
- {"270", 270},
- {"45", 0}, // not a multiple of 90 → defensive 0
- {"(x)", 0}, // wrong type → 0
- }
- for _, tc := range cases {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Page /Parent 2 0 R /Rotate " + tc.rotate + " >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open(%s): %v", tc.rotate, err)
- }
- p, _ := r.Page(0)
- if got := p.Rotation(); got != tc.want {
- t.Errorf("Rotate %s → %d, want %d", tc.rotate, got, tc.want)
- }
- r.Close()
- }
-}
-
-func TestPageBoxNormalization(t *testing.T) {
- // Corners written in the wrong order must come back normalised, and the
- // box may be a Real-valued array.
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Page /Parent 2 0 R /MediaBox [612.0 792.0 0 0] >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- p, _ := r.Page(0)
- box, ok := p.Box(MediaBox)
- if !ok || box != (Rect{0, 0, 612, 792}) {
- t.Fatalf("MediaBox = %+v ok=%v, want {0 0 612 792}", box, ok)
- }
- if box.Width() != 612 || box.Height() != 792 {
- t.Errorf("Width/Height = %v/%v, want 612/792", box.Width(), box.Height())
- }
-}
-
-func TestPageBoxMalformed(t *testing.T) {
- // A box that is not a 4-number array must report absent, not panic.
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612] /CropBox (nope) >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- p, _ := r.Page(0)
- if box, ok := p.Box(MediaBox); ok {
- t.Errorf("short MediaBox = %+v, want absent", box)
- }
- if box, ok := p.Box(CropBox); ok {
- t.Errorf("string CropBox = %+v, want absent", box)
- }
- if boxes := p.Boxes(); len(boxes) != 0 {
- t.Errorf("Boxes = %v, want empty", boxes)
- }
-}
-
-func TestPageNonInheritableBoxes(t *testing.T) {
- // An intermediate /Pages node carrying BleedBox/TrimBox/ArtBox must NOT
- // leak those into descendant leaves — only MediaBox/CropBox are inheritable
- // (PDF 32000-1:2008 Table 30). The leaf's own TrimBox is still reported.
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] /MediaBox [0 0 612 792] /CropBox [1 1 611 791] /BleedBox [2 2 610 790] /TrimBox [3 3 609 789] /ArtBox [4 4 608 788] >>",
- "<< /Type /Page /Parent 2 0 R /TrimBox [9 9 100 100] >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- p, _ := r.Page(0)
-
- // Inheritable: come from the ancestor /Pages node.
- if box, ok := p.Box(MediaBox); !ok || box != (Rect{0, 0, 612, 792}) {
- t.Errorf("MediaBox = %+v ok=%v, want inherited {0 0 612 792}", box, ok)
- }
- if box, ok := p.Box(CropBox); !ok || box != (Rect{1, 1, 611, 791}) {
- t.Errorf("CropBox = %+v ok=%v, want inherited {1 1 611 791}", box, ok)
- }
- // Non-inheritable: ancestor values must not appear.
- if box, ok := p.Box(BleedBox); ok {
- t.Errorf("BleedBox = %+v, want absent (not inheritable)", box)
- }
- if box, ok := p.Box(ArtBox); ok {
- t.Errorf("ArtBox = %+v, want absent (not inheritable)", box)
- }
- // The leaf's own TrimBox is reported, not the ancestor's.
- if box, ok := p.Box(TrimBox); !ok || box != (Rect{9, 9, 100, 100}) {
- t.Errorf("TrimBox = %+v ok=%v, want page-level {9 9 100 100}", box, ok)
- }
- if boxes := p.Boxes(); len(boxes) != 3 {
- t.Errorf("Boxes = %v, want 3 (Media, Crop, Trim)", boxes)
- }
-}
-
-func TestPageContentsArrayBrokenEntryErrors(t *testing.T) {
- // /Contents references object 5, which is absent from the xref and so
- // resolves to null. ContentStreams/Content must fail rather than silently
- // return a truncated stream.
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Page /Parent 2 0 R /Contents [ 4 0 R 5 0 R ] >>",
- "<< /Length 5 >>\nstream\nhello\nendstream",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- p, _ := r.Page(0)
- if streams, err := p.ContentStreams(); err == nil {
- t.Errorf("ContentStreams = %v, want error for missing entry", streams)
- }
- if got, err := p.Content(); err == nil {
- t.Errorf("Content = %q, want error for missing entry", got)
- }
-}
-
-func TestPagesPhantomLeafGuard(t *testing.T) {
- // The root /Pages node declares /Count but its /Kids fails to resolve
- // (object 9 is absent). It must not be emitted as a phantom leaf page.
- for _, root := range []string{
- "<< /Type /Pages /Count 1 /Kids 9 0 R >>", // unresolvable /Kids ref
- "<< /Type /Pages /Count 1 >>", // no /Kids at all
- "<< /Type /Pages >>", // /Type-only intermediate
- } {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- root,
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open(%s): %v", root, err)
- }
- n, err := r.PageCount()
- if err != nil {
- t.Fatalf("PageCount(%s): %v", root, err)
- }
- if n != 0 {
- t.Errorf("root %s: PageCount = %d, want 0 (no phantom leaf)", root, n)
- }
- r.Close()
- }
-}
-
-func TestPageIndexOutOfRange(t *testing.T) {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Page /Parent 2 0 R >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if _, err := r.Page(-1); err == nil {
- t.Error("Page(-1) = nil error, want out-of-range error")
- }
- if _, err := r.Page(1); err == nil {
- t.Error("Page(1) = nil error, want out-of-range error")
- }
-}
-
-func TestPageCyclicKidsTerminates(t *testing.T) {
- // obj 3's /Kids points back to obj 2; the walk must terminate.
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Pages /Kids [ 2 0 R ] >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- n, err := r.PageCount()
- if err != nil {
- t.Fatalf("PageCount: %v", err)
- }
- if n != 0 {
- t.Errorf("PageCount = %d, want 0", n)
- }
-}
-
-func TestPageCyclicParentTerminates(t *testing.T) {
- // The leaf page's /Parent points to itself; inheritance must terminate.
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Page /Parent 3 0 R >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- p, err := r.Page(0)
- if err != nil {
- t.Fatalf("Page(0): %v", err)
- }
- if box, ok := p.Box(MediaBox); ok {
- t.Errorf("MediaBox = %+v, want absent", box)
- }
- if rot := p.Rotation(); rot != 0 {
- t.Errorf("Rotation = %d, want 0", rot)
- }
- if _, ok := p.Resources(); ok {
- t.Errorf("Resources found, want none")
- }
-}
-
-func TestPagesMissingType(t *testing.T) {
- // Neither the intermediate node nor the leaf declares /Type; the walk
- // keys off /Kids presence instead.
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Kids [ 3 0 R ] /Count 1 >>",
- "<< /Parent 2 0 R /MediaBox [0 0 100 100] >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- n, err := r.PageCount()
- if err != nil {
- t.Fatalf("PageCount: %v", err)
- }
- if n != 1 {
- t.Fatalf("PageCount = %d, want 1", n)
- }
- p, _ := r.Page(0)
- if box, ok := p.Box(MediaBox); !ok || box != (Rect{0, 0, 100, 100}) {
- t.Errorf("MediaBox = %+v ok=%v, want {0 0 100 100}", box, ok)
- }
-}
diff --git a/parse.go b/parse.go
deleted file mode 100644
index 50b5ba2..0000000
--- a/parse.go
+++ /dev/null
@@ -1,199 +0,0 @@
-package pdfdisassembler
-
-import (
- "errors"
- "fmt"
- "strconv"
-
- "github.com/speedata/pdfdisassembler/internal/lex"
-)
-
-// maxParseDepth caps array/dict nesting so a hostile PDF can't stack-overflow
-// the recursive parser.
-const maxParseDepth = 1000
-
-// parser is a recursive descent parser over a lex.Lexer that emits direct
-// PDF Objects. It does not chase indirect references — every Reference
-// token becomes a Reference value.
-type parser struct {
- lx *lex.Lexer
- r *Reader
- queue []lex.Token
- depth int
-}
-
-func newParser(lx *lex.Lexer, r *Reader) *parser {
- return &parser{lx: lx, r: r}
-}
-
-func (p *parser) next() (lex.Token, error) {
- if len(p.queue) > 0 {
- t := p.queue[0]
- p.queue = p.queue[1:]
- return t, nil
- }
- return p.lx.Next()
-}
-
-func (p *parser) peek() (lex.Token, error) {
- if len(p.queue) > 0 {
- return p.queue[0], nil
- }
- t, err := p.lx.Next()
- if err != nil {
- return lex.Token{}, err
- }
- p.queue = append(p.queue, t)
- return t, nil
-}
-
-// peekN returns the n-th unread token (1-based). It reads from the lexer
-// to fill the queue as needed.
-func (p *parser) peekN(n int) (lex.Token, error) {
- for len(p.queue) < n {
- t, err := p.lx.Next()
- if err != nil {
- return lex.Token{}, err
- }
- p.queue = append(p.queue, t)
- }
- return p.queue[n-1], nil
-}
-
-// consume removes n tokens from the front of the queue (after a successful
-// peekN). Caller must have ensured the queue has at least n entries.
-func (p *parser) consume(n int) {
- p.queue = p.queue[n:]
-}
-
-// parseObject parses a single direct PDF object. It also recognises the
-// "N G R" indirect-reference triplet, which requires two-token lookahead.
-func (p *parser) parseObject() (Object, error) {
- tok, err := p.next()
- if err != nil {
- return nil, err
- }
- return p.parseObjectFrom(tok)
-}
-
-func (p *parser) parseObjectFrom(tok lex.Token) (Object, error) {
- switch tok.Kind {
- case lex.EOF:
- return nil, errors.New("pdfdisassembler/parse: unexpected EOF")
- case lex.Name:
- return Name(string(tok.Bytes)), nil
- case lex.LitString:
- // Copy: Bytes points into the source buffer of the literal-string
- // reader which is stable for our use, but ownership-wise we'd
- // rather not have callers alias the source.
- b := make(String, len(tok.Bytes))
- copy(b, tok.Bytes)
- return b, nil
- case lex.HexString:
- b := make(String, len(tok.Bytes))
- copy(b, tok.Bytes)
- return b, nil
- case lex.Integer:
- n, err := strconv.ParseInt(string(tok.Bytes), 10, 64)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler/parse: bad integer %q: %w", tok.Bytes, err)
- }
- // Look ahead two tokens for a Reference: "G R".
- t2, err := p.peekN(1)
- if err != nil || t2.Kind != lex.Integer {
- return Integer(n), nil
- }
- t3, err := p.peekN(2)
- if err != nil || t3.Kind != lex.Keyword || string(t3.Bytes) != "R" {
- return Integer(n), nil
- }
- g, err := strconv.Atoi(string(t2.Bytes))
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler/parse: bad generation %q: %w", t2.Bytes, err)
- }
- p.consume(2) // gen and 'R'
- return Reference{Number: int(n), Generation: g}, nil
- case lex.Real:
- f, err := strconv.ParseFloat(string(tok.Bytes), 64)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler/parse: bad real %q: %w", tok.Bytes, err)
- }
- return Real(f), nil
- case lex.Keyword:
- switch string(tok.Bytes) {
- case "true":
- return Bool(true), nil
- case "false":
- return Bool(false), nil
- case "null":
- return Null{}, nil
- }
- return nil, fmt.Errorf("pdfdisassembler/parse: unexpected keyword %q at %d", tok.Bytes, tok.Offset)
- case lex.ArrayStart:
- return p.parseArray()
- case lex.DictStart:
- return p.parseDict()
- case lex.ArrayEnd, lex.DictEnd:
- return nil, fmt.Errorf("pdfdisassembler/parse: stray %s at %d", tok.Kind, tok.Offset)
- }
- return nil, fmt.Errorf("pdfdisassembler/parse: unhandled token %s at %d", tok.Kind, tok.Offset)
-}
-
-func (p *parser) parseArray() (Array, error) {
- p.depth++
- defer func() { p.depth-- }()
- if p.depth > maxParseDepth {
- return nil, fmt.Errorf("pdfdisassembler/parse: nesting too deep (> %d)", maxParseDepth)
- }
- var out Array
- for {
- t, err := p.peek()
- if err != nil {
- return nil, err
- }
- if t.Kind == lex.ArrayEnd {
- p.next()
- return out, nil
- }
- if t.Kind == lex.EOF {
- return nil, errors.New("pdfdisassembler/parse: unterminated array")
- }
- v, err := p.parseObject()
- if err != nil {
- return nil, err
- }
- out = append(out, v)
- }
-}
-
-func (p *parser) parseDict() (*Dict, error) {
- p.depth++
- defer func() { p.depth-- }()
- if p.depth > maxParseDepth {
- return nil, fmt.Errorf("pdfdisassembler/parse: nesting too deep (> %d)", maxParseDepth)
- }
- d := newDict(p.r)
- for {
- t, err := p.peek()
- if err != nil {
- return nil, err
- }
- if t.Kind == lex.DictEnd {
- p.next()
- return d, nil
- }
- if t.Kind == lex.EOF {
- return nil, errors.New("pdfdisassembler/parse: unterminated dictionary")
- }
- if t.Kind != lex.Name {
- return nil, fmt.Errorf("pdfdisassembler/parse: dict key must be a name, got %s at %d", t.Kind, t.Offset)
- }
- p.next()
- key := string(t.Bytes)
- v, err := p.parseObject()
- if err != nil {
- return nil, err
- }
- d.set(key, v)
- }
-}
diff --git a/parse_test.go b/parse_test.go
deleted file mode 100644
index ef7e3af..0000000
--- a/parse_test.go
+++ /dev/null
@@ -1,92 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "fmt"
- "strings"
- "testing"
-
- "github.com/speedata/pdfdisassembler/internal/lex"
-)
-
-// buildPDFWithObjectBody puts body as object 3 in a minimal classical-xref PDF.
-func buildPDFWithObjectBody(t *testing.T, body string) []byte {
- t.Helper()
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
- offsets := make([]int, 4)
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
- offsets[3] = off()
- fmt.Fprintf(&buf, "3 0 obj\n%s\nendobj\n", body)
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 4\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 3; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprint(&buf, "trailer\n<< /Size 4 /Root 1 0 R >>\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-func TestDeeplyNestedRejected(t *testing.T) {
- // Far above the parser's depth cap, but well below a real stack overflow.
- const depth = 2000
- tests := []struct{ name, body string }{
- {"array", strings.Repeat("[", depth) + strings.Repeat("]", depth)},
- {"dict", strings.Repeat("<< /K ", depth) + "0" + strings.Repeat(" >>", depth)},
- }
- for _, tt := range tests {
- t.Run(tt.name, func(t *testing.T) {
- data := buildPDFWithObjectBody(t, tt.body)
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if _, err := r.Resolve(Reference{Number: 3, Generation: 0}); err == nil {
- t.Fatal("expected error for over-deep nesting")
- }
- })
- }
-}
-
-func TestModeratelyNestedArrayResolves(t *testing.T) {
- data := buildPDFWithObjectBody(t, strings.Repeat("[", 100)+strings.Repeat("]", 100))
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- obj, err := r.Resolve(Reference{Number: 3, Generation: 0})
- if err != nil {
- t.Fatalf("Resolve: %v", err)
- }
- if _, ok := obj.(Array); !ok {
- t.Fatalf("got %T, want Array", obj)
- }
-}
-
-// Malformed token streams must error, never panic.
-func TestParseObjectErrors(t *testing.T) {
- for _, src := range []string{
- "", // EOF where an object is expected
- "]", // stray ArrayEnd
- ">>", // stray DictEnd
- "foo", // unexpected keyword
- "[ 1 2", // unterminated array
- "<< /K 1", // unterminated dict
- "<< 1 2 >>", // dict key is not a name
- } {
- t.Run(src, func(t *testing.T) {
- p := newParser(lex.New([]byte(src)), nil)
- if _, err := p.parseObject(); err == nil {
- t.Fatal("expected an error, got nil")
- }
- })
- }
-}
diff --git a/reader.go b/reader.go
deleted file mode 100644
index bf49014..0000000
--- a/reader.go
+++ /dev/null
@@ -1,526 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "errors"
- "fmt"
- "io"
- "iter"
- "os"
- "sort"
- "strconv"
-
- "github.com/speedata/pdfdisassembler/internal/lex"
-)
-
-// DefaultMaxStreamSize is the per-stream decoded-size cap Open uses by default.
-const DefaultMaxStreamSize int64 = 16 << 20
-
-// Option configures a Reader at Open time.
-type Option func(*Reader)
-
-// WithMaxStreamSize sets the per-stream decoded-size cap; n <= 0 disables it.
-// Applied before parsing, so it also bounds streams decoded during Open.
-func WithMaxStreamSize(n int64) Option {
- return func(r *Reader) { r.MaxStreamSize = n }
-}
-
-// Reader is a parsed PDF document. It is not safe for concurrent use.
-type Reader struct {
- src io.ReadSeeker
- closer io.Closer
- buf []byte // entire file contents
- version string
- xref map[Reference]xrefEntry
- trailer *Dict
- catalog *Dict
- info *Dict
- infoLoad bool
-
- // MaxStreamSize caps each stream's decoded size; <= 0 disables it. Setting
- // it after Open misses Open-time (xref/object) streams; use WithMaxStreamSize.
- MaxStreamSize int64
-
- // Encryption.
- encrypt *encryptCtx
-
- // Resolution caches.
- objCache map[Reference]Object
- // resolveStack guards against indirect-reference cycles.
- resolveStack map[Reference]struct{}
-
- // Page-tree cache, populated lazily by loadPages.
- pages []*Page
- pagesLoaded bool
- pagesErr error
-}
-
-// xrefEntry describes a single in-use object.
-type xrefEntry struct {
- // kind is 1 for in-file objects (Offset), 2 for compressed objects
- // (ObjStmNum + Index).
- kind uint8
- offset int64
- objStmNum int
- objStmIdx int
- generation int
-}
-
-// Open parses a PDF from rs. rs must remain valid for the lifetime of the
-// returned Reader.
-func Open(rs io.ReadSeeker, opts ...Option) (*Reader, error) {
- if _, err := rs.Seek(0, io.SeekStart); err != nil {
- return nil, fmt.Errorf("pdfdisassembler: seek: %w", err)
- }
- buf, err := io.ReadAll(rs)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: read: %w", err)
- }
- r := &Reader{
- src: rs,
- buf: buf,
- xref: map[Reference]xrefEntry{},
- objCache: map[Reference]Object{},
- resolveStack: map[Reference]struct{}{},
- MaxStreamSize: DefaultMaxStreamSize,
- }
- for _, opt := range opts {
- opt(r)
- }
- if err := r.parseHeader(); err != nil {
- return nil, err
- }
- if err := r.parseXref(); err != nil {
- return nil, err
- }
- if err := r.initEncrypt(); err != nil {
- return nil, err
- }
- return r, nil
-}
-
-// OpenFile opens path and parses it as a PDF. The file stays open until
-// Reader.Close is called.
-func OpenFile(path string, opts ...Option) (*Reader, error) {
- f, err := os.Open(path)
- if err != nil {
- return nil, err
- }
- r, err := Open(f, opts...)
- if err != nil {
- f.Close()
- return nil, err
- }
- r.closer = f
- return r, nil
-}
-
-// Close releases the underlying resource. For Reader instances created via
-// Open with a non-file ReadSeeker, Close is a no-op.
-func (r *Reader) Close() error {
- if r.closer != nil {
- return r.closer.Close()
- }
- return nil
-}
-
-// Version returns the PDF version declared in the file header (e.g. "1.7"
-// or "2.0"). If the catalog declares a /Version entry that exceeds the
-// header version, the catalog value wins (per spec).
-func (r *Reader) Version() string {
- if r.catalog != nil {
- if n, ok := r.catalog.Name("Version"); ok {
- s := string(n)
- if s > r.version {
- return s
- }
- }
- }
- return r.version
-}
-
-// parseHeader reads the "%PDF-x.y" line. It tolerates up to 1024 leading
-// bytes of garbage (some producers emit MIME prologues).
-func (r *Reader) parseHeader() error {
- scan := r.buf
- limit := len(scan)
- if limit > 1024 {
- limit = 1024
- }
- idx := bytes.Index(scan[:limit], []byte("%PDF-"))
- if idx < 0 {
- return errors.New("pdfdisassembler: not a PDF (missing %PDF- header)")
- }
- rest := scan[idx+5:]
- end := 0
- for end < len(rest) && end < 16 {
- c := rest[end]
- if c == '\r' || c == '\n' || c == ' ' || c == '\t' {
- break
- }
- end++
- }
- r.version = string(rest[:end])
- return nil
-}
-
-// Catalog returns the document catalog dictionary.
-func (r *Reader) Catalog() (*Dict, error) {
- if r.catalog != nil {
- return r.catalog, nil
- }
- if r.trailer == nil {
- return nil, errors.New("pdfdisassembler: no trailer")
- }
- root, ok := r.trailer.Get("Root")
- if !ok {
- return nil, errors.New("pdfdisassembler: trailer has no /Root")
- }
- d, err := r.ResolveDict(root)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: resolve catalog: %w", err)
- }
- r.catalog = d
- return d, nil
-}
-
-// Trailer returns the trailer dictionary.
-func (r *Reader) Trailer() *Dict {
- return r.trailer
-}
-
-// Resolve follows an indirect reference. If obj is not a Reference, returns
-// obj unchanged. Resolution is cached.
-func (r *Reader) Resolve(obj Object) (Object, error) {
- ref, ok := obj.(Reference)
- if !ok {
- return obj, nil
- }
- if cached, ok := r.objCache[ref]; ok {
- return cached, nil
- }
- if _, on := r.resolveStack[ref]; on {
- return Null{}, fmt.Errorf("pdfdisassembler: reference cycle at %d %d R", ref.Number, ref.Generation)
- }
- r.resolveStack[ref] = struct{}{}
- defer delete(r.resolveStack, ref)
-
- entry, ok := r.xref[ref]
- if !ok {
- // Some xref tables omit the requested object. Per spec, missing
- // references resolve to null.
- r.objCache[ref] = Null{}
- return Null{}, nil
- }
- var v Object
- var err error
- switch entry.kind {
- case 1:
- v, err = r.readIndirectAt(entry.offset, ref)
- case 2:
- v, err = r.readCompressedObject(entry.objStmNum, entry.objStmIdx, ref)
- default:
- return nil, fmt.Errorf("pdfdisassembler: unknown xref entry kind for %d %d R", ref.Number, ref.Generation)
- }
- if err != nil {
- return nil, err
- }
- r.objCache[ref] = v
- return v, nil
-}
-
-// ResolveDict resolves obj to a *Dict; errors when obj is missing or is
-// not a dictionary.
-func (r *Reader) ResolveDict(obj Object) (*Dict, error) {
- v, err := r.Resolve(obj)
- if err != nil {
- return nil, err
- }
- if v == nil {
- return nil, errors.New("pdfdisassembler: nil object")
- }
- switch t := v.(type) {
- case *Dict:
- return t, nil
- case *Stream:
- return t.Dict, nil
- case Null:
- return nil, errors.New("pdfdisassembler: dictionary expected, got null")
- }
- return nil, fmt.Errorf("pdfdisassembler: dictionary expected, got %T", v)
-}
-
-// ResolveBool resolves obj to a bool; errors otherwise.
-func (r *Reader) ResolveBool(obj Object) (bool, error) {
- v, err := r.Resolve(obj)
- if err != nil {
- return false, err
- }
- b, ok := v.(Bool)
- if !ok {
- return false, fmt.Errorf("pdfdisassembler: boolean expected, got %T", v)
- }
- return bool(b), nil
-}
-
-// ResolveInt resolves obj to an int64; errors otherwise.
-func (r *Reader) ResolveInt(obj Object) (int64, error) {
- v, err := r.Resolve(obj)
- if err != nil {
- return 0, err
- }
- n, ok := v.(Integer)
- if !ok {
- return 0, fmt.Errorf("pdfdisassembler: integer expected, got %T", v)
- }
- return int64(n), nil
-}
-
-// ResolveArray resolves obj to an Array; errors otherwise.
-func (r *Reader) ResolveArray(obj Object) (Array, error) {
- v, err := r.Resolve(obj)
- if err != nil {
- return nil, err
- }
- a, ok := v.(Array)
- if !ok {
- return nil, fmt.Errorf("pdfdisassembler: array expected, got %T", v)
- }
- return a, nil
-}
-
-// readIndirectAt parses the indirect object starting at offset. The object
-// header (N G obj) is verified against the expected reference.
-func (r *Reader) readIndirectAt(offset int64, expect Reference) (Object, error) {
- if offset < 0 || offset >= int64(len(r.buf)) {
- return nil, fmt.Errorf("pdfdisassembler: xref offset %d out of range", offset)
- }
- lx := lex.New(r.buf)
- lx.SetPos(int(offset))
- p := newParser(lx, r)
-
- // Read "N G obj" header.
- t1, err := p.next()
- if err != nil {
- return nil, err
- }
- t2, err := p.next()
- if err != nil {
- return nil, err
- }
- t3, err := p.next()
- if err != nil {
- return nil, err
- }
- if t1.Kind != lex.Integer || t2.Kind != lex.Integer ||
- t3.Kind != lex.Keyword || string(t3.Bytes) != "obj" {
- return nil, fmt.Errorf("pdfdisassembler: bad indirect header at %d (got %s %s %s)", offset, t1.Kind, t2.Kind, t3.Kind)
- }
- n, _ := strconv.Atoi(string(t1.Bytes))
- g, _ := strconv.Atoi(string(t2.Bytes))
- if n != expect.Number {
- return nil, fmt.Errorf("pdfdisassembler: indirect mismatch at %d: header %d %d, expected %d %d", offset, n, g, expect.Number, expect.Generation)
- }
-
- body, err := p.parseObject()
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: parse body of %d %d R: %w", expect.Number, expect.Generation, err)
- }
-
- // Check for stream.
- t4, err := p.peek()
- if err == nil && t4.Kind == lex.Keyword && string(t4.Bytes) == "stream" {
- p.next()
- d, ok := body.(*Dict)
- if !ok {
- return nil, fmt.Errorf("pdfdisassembler: stream object %d %d R has non-dict body (%T)", expect.Number, expect.Generation, body)
- }
- length, err := r.streamLength(d)
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: /Length for %d %d R: %w", expect.Number, expect.Generation, err)
- }
- raw, err := lx.ReadStreamData(int(length))
- if err != nil {
- return nil, fmt.Errorf("pdfdisassembler: read stream %d %d R: %w", expect.Number, expect.Generation, err)
- }
- // rawOffset is where the raw bytes begin in the file.
- rawStart := lx.Pos() - len(raw)
- return &Stream{
- Dict: d,
- reader: r,
- rawOffset: int64(rawStart),
- rawLength: int64(len(raw)),
- objNumber: expect.Number,
- objGeneration: expect.Generation,
- }, nil
- }
- return body, nil
-}
-
-// streamLength resolves the /Length entry on a stream dict.
-func (r *Reader) streamLength(d *Dict) (int64, error) {
- v, ok := d.Get("Length")
- if !ok {
- return 0, errors.New("missing /Length")
- }
- v, err := r.Resolve(v)
- if err != nil {
- return 0, err
- }
- n, ok := v.(Integer)
- if !ok {
- return 0, fmt.Errorf("/Length is %T, want integer", v)
- }
- if n < 0 {
- return 0, fmt.Errorf("/Length is negative: %d", n)
- }
- return int64(n), nil
-}
-
-// DocumentInfo returns the standard /Info dictionary entries as a value
-// snapshot. Missing entries return zero values.
-func (r *Reader) DocumentInfo() DocInfo {
- if !r.infoLoad {
- r.infoLoad = true
- if r.trailer != nil {
- if obj, ok := r.trailer.Get("Info"); ok {
- if d, err := r.ResolveDict(obj); err == nil {
- r.info = d
- }
- }
- }
- }
- var info DocInfo
- info.Custom = map[string]string{}
- if r.info == nil {
- return info
- }
- for k, v := range r.info.Iter() {
- resolved, err := r.Resolve(v)
- if err != nil {
- continue
- }
- s, ok := resolved.(String)
- if !ok {
- continue
- }
- decoded := decodeTextString(s)
- switch k {
- case "Title":
- info.Title = decoded
- case "Author":
- info.Author = decoded
- case "Subject":
- info.Subject = decoded
- case "Keywords":
- info.Keywords = decoded
- case "Creator":
- info.Creator = decoded
- case "Producer":
- info.Producer = decoded
- case "CreationDate":
- info.CreationDate = parseDate(decoded)
- case "ModDate":
- info.ModDate = parseDate(decoded)
- default:
- info.Custom[k] = decoded
- }
- }
- return info
-}
-
-// Objects iterates every live indirect object in the xref table.
-func (r *Reader) Objects() iter.Seq[ObjectEntry] {
- return func(yield func(ObjectEntry) bool) {
- // Iterate in stable order: by object number.
- refs := make([]Reference, 0, len(r.xref))
- for ref := range r.xref {
- refs = append(refs, ref)
- }
- // The entry count is attacker-controlled, so this must stay O(n log n).
- sort.Slice(refs, func(i, j int) bool {
- if refs[i].Number != refs[j].Number {
- return refs[i].Number < refs[j].Number
- }
- return refs[i].Generation < refs[j].Generation
- })
- for _, ref := range refs {
- obj, err := r.Resolve(ref)
- if err != nil {
- continue
- }
- if !yield(ObjectEntry{Reference: ref, Object: obj}) {
- return
- }
- }
- }
-}
-
-// DecodeStream resolves obj to a stream and returns its decoded content.
-func (r *Reader) DecodeStream(obj Object) ([]byte, error) {
- v, err := r.Resolve(obj)
- if err != nil {
- return nil, err
- }
- s, ok := v.(*Stream)
- if !ok {
- return nil, fmt.Errorf("pdfdisassembler: stream expected, got %T", v)
- }
- return s.Content()
-}
-
-const maxNameTreeDepth = 1000
-
-// EmbeddedFiles returns the document's embedded files (PDF attachments) from
-// the catalog's EmbeddedFiles name tree, in tree order. Returns nil when there
-// are none.
-func (r *Reader) EmbeddedFiles() []EmbeddedFile {
- cat, err := r.Catalog()
- if err != nil {
- return nil
- }
- names, ok := cat.Dict("Names")
- if !ok {
- return nil
- }
- root, ok := names.Dict("EmbeddedFiles")
- if !ok {
- return nil
- }
- var out []EmbeddedFile
- r.walkNameTree(root, map[Reference]struct{}{}, 0, &out)
- return out
-}
-
-// walkNameTree collects (name, /Filespec) pairs from a name-tree node. seen
-// records already-visited /Kids references and depth bounds the descent, so a
-// cyclic or pathologically deep /Kids graph can't loop or overflow the stack.
-func (r *Reader) walkNameTree(node *Dict, seen map[Reference]struct{}, depth int, out *[]EmbeddedFile) {
- if node == nil || depth > maxNameTreeDepth {
- return
- }
- if kids, ok := node.Array("Kids"); ok {
- for _, kid := range kids {
- if ref, ok := kid.(Reference); ok {
- if _, dup := seen[ref]; dup {
- continue
- }
- seen[ref] = struct{}{}
- }
- if child, err := r.ResolveDict(kid); err == nil {
- r.walkNameTree(child, seen, depth+1, out)
- }
- }
- }
- if entries, ok := node.Array("Names"); ok {
- for i := 0; i+1 < len(entries); i += 2 {
- name, ok := entries[i].(String)
- if !ok {
- continue
- }
- if spec, err := r.ResolveDict(entries[i+1]); err == nil {
- *out = append(*out, EmbeddedFile{Name: string(name), Spec: spec})
- }
- }
- }
-}
diff --git a/reader_test.go b/reader_test.go
deleted file mode 100644
index 13985c2..0000000
--- a/reader_test.go
+++ /dev/null
@@ -1,623 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "compress/zlib"
- "fmt"
- "os"
- "path/filepath"
- "strings"
- "testing"
-)
-
-// buildMinimalPDF constructs a tiny valid PDF in memory: a catalog, a
-// pages tree with one empty page, an Info dict, and a classical xref. It
-// returns the raw bytes.
-func buildMinimalPDF(t *testing.T) []byte {
- t.Helper()
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
-
- offsets := make([]int, 5) // index 1..4
-
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
-
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>\nendobj\n")
-
- offsets[3] = off()
- fmt.Fprint(&buf, "3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] >>\nendobj\n")
-
- offsets[4] = off()
- fmt.Fprint(&buf, "4 0 obj\n<< /Title (Hello) /Producer (pdfdisassembler-test) >>\nendobj\n")
-
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 5\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 4; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprintf(&buf, "trailer\n<< /Size 5 /Root 1 0 R /Info 4 0 R >>\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-func TestOpenMinimal(t *testing.T) {
- data := buildMinimalPDF(t)
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
-
- if r.Version() != "1.7" {
- t.Fatalf("version %q", r.Version())
- }
- cat, err := r.Catalog()
- if err != nil {
- t.Fatalf("Catalog: %v", err)
- }
- n, ok := cat.Name("Type")
- if !ok || n != "Catalog" {
- t.Fatalf("/Type %q ok=%v", n, ok)
- }
- pages, ok := cat.Dict("Pages")
- if !ok {
- t.Fatal("Catalog.Dict(Pages)")
- }
- count, ok := pages.Int("Count")
- if !ok || count != 1 {
- t.Fatalf("/Count %d ok=%v", count, ok)
- }
-}
-
-func TestDocumentInfo(t *testing.T) {
- data := buildMinimalPDF(t)
- r, _ := Open(bytes.NewReader(data))
- defer r.Close()
- info := r.DocumentInfo()
- if info.Title != "Hello" {
- t.Fatalf("Title %q", info.Title)
- }
- if !strings.HasPrefix(info.Producer, "pdfdisassembler") {
- t.Fatalf("Producer %q", info.Producer)
- }
-}
-
-func TestObjectsIterator(t *testing.T) {
- data := buildMinimalPDF(t)
- r, _ := Open(bytes.NewReader(data))
- defer r.Close()
- count := 0
- seen := map[int]bool{}
- for entry := range r.Objects() {
- count++
- seen[entry.Reference.Number] = true
- }
- if count != 4 {
- t.Fatalf("count %d", count)
- }
- for i := 1; i <= 4; i++ {
- if !seen[i] {
- t.Fatalf("missing object %d", i)
- }
- }
-}
-
-func TestObjectsIteratorSortedOrder(t *testing.T) {
- r, err := Open(bytes.NewReader(buildMinimalPDF(t)))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- var nums []int
- for entry := range r.Objects() {
- nums = append(nums, entry.Reference.Number)
- }
- if len(nums) == 0 {
- t.Fatal("no objects iterated")
- }
- for i := 1; i < len(nums); i++ {
- if nums[i-1] >= nums[i] {
- t.Fatalf("Objects() not strictly ascending: %v", nums)
- }
- }
-}
-
-// buildPDFWithStream embeds payload as a FlateDecode stream (obj 3) so
-// DecodeStream can be exercised.
-func buildPDFWithStream(t *testing.T, payload []byte) []byte {
- t.Helper()
- var zbuf bytes.Buffer
- zw := zlib.NewWriter(&zbuf)
- zw.Write(payload)
- zw.Close()
- zdata := zbuf.Bytes()
-
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
-
- offsets := make([]int, 4)
-
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
-
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
-
- offsets[3] = off()
- fmt.Fprintf(&buf, "3 0 obj\n<< /Length %d /Filter /FlateDecode >>\nstream\n", len(zdata))
- buf.Write(zdata)
- fmt.Fprint(&buf, "\nendstream\nendobj\n")
-
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 4\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 3; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprintf(&buf, "trailer\n<< /Size 4 /Root 1 0 R >>\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-func TestDecodeStream(t *testing.T) {
- data := buildPDFWithStream(t, []byte("Hello, stream!"))
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
-
- content, err := r.DecodeStream(Reference{Number: 3, Generation: 0})
- if err != nil {
- t.Fatalf("DecodeStream: %v", err)
- }
- if string(content) != "Hello, stream!" {
- t.Fatalf("content %q", content)
- }
-}
-
-func TestStreamSizeLimitEnforced(t *testing.T) {
- // obj 3 decompresses to 2 MiB; the 64 KiB cap must reject it.
- data := buildPDFWithStream(t, make([]byte, 2<<20))
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
-
- r.MaxStreamSize = 64 << 10
- if _, err := r.DecodeStream(Reference{Number: 3, Generation: 0}); err == nil {
- t.Fatal("expected error decoding stream larger than MaxStreamSize, got nil")
- }
-}
-
-func TestWithMaxStreamSizeOption(t *testing.T) {
- data := buildPDFWithStream(t, make([]byte, 2<<20))
- r, err := Open(bytes.NewReader(data), WithMaxStreamSize(64<<10))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if r.MaxStreamSize != 64<<10 {
- t.Fatalf("MaxStreamSize = %d, want %d", r.MaxStreamSize, 64<<10)
- }
- if _, err := r.DecodeStream(Reference{Number: 3, Generation: 0}); err == nil {
- t.Fatal("expected error: stream exceeds the option-set cap")
- }
-}
-
-func TestWithMaxStreamSizeDisable(t *testing.T) {
- data := buildPDFWithStream(t, make([]byte, 2<<20))
- r, err := Open(bytes.NewReader(data), WithMaxStreamSize(0))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- out, err := r.DecodeStream(Reference{Number: 3, Generation: 0})
- if err != nil {
- t.Fatalf("DecodeStream with cap disabled: %v", err)
- }
- if len(out) != 2<<20 {
- t.Fatalf("decoded %d bytes, want %d", len(out), 2<<20)
- }
-}
-
-func TestDefaultMaxStreamSizeSet(t *testing.T) {
- // Open must install a finite default so Open-time decodes are bounded.
- data := buildMinimalPDF(t)
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if r.MaxStreamSize != DefaultMaxStreamSize {
- t.Fatalf("MaxStreamSize = %d, want default %d", r.MaxStreamSize, DefaultMaxStreamSize)
- }
-}
-
-// buildDictPDF puts each body in objs as object i+1 of a classical-xref PDF
-// (obj 1 is the catalog). Bodies are plain objects (no streams).
-func buildDictPDF(t *testing.T, objs []string) []byte {
- t.Helper()
- var buf bytes.Buffer
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
- offsets := make([]int, len(objs)+1)
- for i, body := range objs {
- offsets[i+1] = buf.Len()
- fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", i+1, body)
- }
- xrefOff := buf.Len()
- fmt.Fprintf(&buf, "xref\n0 %d\n%010d %05d f \n", len(objs)+1, 0, 65535)
- for i := 1; i <= len(objs); i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprintf(&buf, "trailer\n<< /Size %d /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n",
- len(objs)+1, xrefOff)
- return buf.Bytes()
-}
-
-func TestEmbeddedFiles(t *testing.T) {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles 3 0 R >> >>",
- "<< /Type /Pages /Kids [] /Count 0 >>",
- "<< /Names [ (a.xml) 4 0 R (b.xml) 5 0 R ] >>",
- "<< /Type /Filespec /F (a.xml) >>",
- "<< /Type /Filespec /F (b.xml) >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- ef := r.EmbeddedFiles()
- if len(ef) != 2 {
- t.Fatalf("got %d files, want 2", len(ef))
- }
- if ef[0].Name != "a.xml" || ef[1].Name != "b.xml" {
- t.Fatalf("names %q, %q", ef[0].Name, ef[1].Name)
- }
- if f, ok := ef[0].Spec.String("F"); !ok || f != "a.xml" {
- t.Fatalf("spec /F %q ok=%v", f, ok)
- }
-}
-
-func TestEmbeddedFilesNestedKids(t *testing.T) {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles 3 0 R >> >>",
- "<< /Type /Pages /Kids [] /Count 0 >>",
- "<< /Kids [ 4 0 R ] >>",
- "<< /Names [ (a.xml) 5 0 R ] >>",
- "<< /Type /Filespec /F (a.xml) >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if ef := r.EmbeddedFiles(); len(ef) != 1 || ef[0].Name != "a.xml" {
- t.Fatalf("got %+v, want one a.xml", ef)
- }
-}
-
-func TestEmbeddedFilesCyclicKidsTerminates(t *testing.T) {
- // obj 3's /Kids references itself; the walk must terminate, not overflow.
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles 3 0 R >> >>",
- "<< /Type /Pages /Kids [] /Count 0 >>",
- "<< /Kids [ 3 0 R ] >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if ef := r.EmbeddedFiles(); len(ef) != 0 {
- t.Fatalf("got %d files, want 0", len(ef))
- }
-}
-
-func TestEmbeddedFilesNone(t *testing.T) {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Kids [] /Count 0 >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if ef := r.EmbeddedFiles(); ef != nil {
- t.Fatalf("got %+v, want nil", ef)
- }
-}
-
-// FuzzOpen asserts the read pipeline never panics on arbitrary input: Open and
-// every accessor may return an error, but must not crash the process.
-func FuzzOpen(f *testing.F) {
- seeds, _ := filepath.Glob("testdata/fixtures/*/input.pdf")
- for _, p := range seeds {
- if b, err := os.ReadFile(p); err == nil {
- f.Add(b)
- }
- }
- f.Add([]byte("%PDF-1.7\n"))
- f.Fuzz(func(t *testing.T, data []byte) {
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- return
- }
- defer r.Close()
- _, _ = r.Catalog()
- _ = r.DocumentInfo()
- _ = r.EmbeddedFiles()
- _ = r.Version()
- for entry := range r.Objects() {
- if s, ok := entry.Object.(*Stream); ok {
- _, _ = s.Content()
- }
- }
- })
-}
-
-func TestResolveHelpers(t *testing.T) {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R /IntRef 3 0 R /BoolRef 4 0 R /ArrRef 5 0 R >>",
- "<< /Type /Pages /Kids [] /Count 0 >>",
- "42",
- "true",
- "[ 1 2 3 ]",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- cat, err := r.Catalog()
- if err != nil {
- t.Fatalf("Catalog: %v", err)
- }
- intRef, _ := cat.Get("IntRef")
- boolRef, _ := cat.Get("BoolRef")
- arrRef, _ := cat.Get("ArrRef")
-
- if v, err := r.ResolveInt(intRef); err != nil || v != 42 {
- t.Errorf("ResolveInt = %d, %v", v, err)
- }
- if v, err := r.ResolveBool(boolRef); err != nil || !v {
- t.Errorf("ResolveBool = %v, %v", v, err)
- }
- if a, err := r.ResolveArray(arrRef); err != nil || len(a) != 3 {
- t.Errorf("ResolveArray len = %d, %v", len(a), err)
- }
- // Type mismatches must error.
- if _, err := r.ResolveInt(boolRef); err == nil {
- t.Error("ResolveInt on a bool")
- }
- if _, err := r.ResolveBool(arrRef); err == nil {
- t.Error("ResolveBool on an array")
- }
- if _, err := r.ResolveArray(intRef); err == nil {
- t.Error("ResolveArray on an int")
- }
-
- if r.Trailer() == nil {
- t.Fatal("nil trailer")
- }
- if _, ok := r.Trailer().Get("Root"); !ok {
- t.Error("trailer missing /Root")
- }
-}
-
-// A catalog /Version higher than the header version wins (PDF 32000-1 §7.5.5).
-func TestVersionCatalogOverride(t *testing.T) {
- data := buildDictPDF(t, []string{
- "<< /Type /Catalog /Pages 2 0 R /Version /2.0 >>",
- "<< /Type /Pages /Kids [] /Count 0 >>",
- })
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if _, err := r.Catalog(); err != nil { // Version() reads the cached catalog
- t.Fatalf("Catalog: %v", err)
- }
- if v := r.Version(); v != "2.0" {
- t.Errorf("Version = %q, want 2.0 (catalog override)", v)
- }
-}
-
-// A non-Reference /Length resolves to itself, so a bare Reader (no xref) drives
-// the missing/non-integer/negative guards directly.
-func TestStreamLengthRejectsBadLength(t *testing.T) {
- r := &Reader{}
- bad := []struct {
- name string
- set func(d *Dict)
- }{
- {"missing", func(d *Dict) {}},
- {"non_integer", func(d *Dict) { d.set("Length", String("x")) }},
- {"negative", func(d *Dict) { d.set("Length", Integer(-5)) }},
- }
- for _, tc := range bad {
- t.Run(tc.name, func(t *testing.T) {
- d := newDict(nil)
- tc.set(d)
- if _, err := r.streamLength(d); err == nil {
- t.Fatal("expected an error, got nil")
- }
- })
- }
-
- // Control: a valid non-negative /Length must still resolve.
- t.Run("valid", func(t *testing.T) {
- d := newDict(nil)
- d.set("Length", Integer(42))
- if n, err := r.streamLength(d); err != nil || n != 42 {
- t.Fatalf("streamLength = %d, %v; want 42, nil", n, err)
- }
- })
-}
-
-// A bare Reader works because a non-Reference value resolves to itself; each
-// helper must error on the wrong type, not pass back a zero value as success.
-func TestResolveHelperTypeErrors(t *testing.T) {
- r := &Reader{}
- errCases := []struct {
- name string
- call func() error
- }{
- {"dict_from_int", func() error { _, e := r.ResolveDict(Integer(1)); return e }},
- {"dict_from_null", func() error { _, e := r.ResolveDict(Null{}); return e }},
- {"dict_from_nil", func() error { _, e := r.ResolveDict(nil); return e }},
- {"bool_from_int", func() error { _, e := r.ResolveBool(Integer(1)); return e }},
- {"int_from_bool", func() error { _, e := r.ResolveInt(Bool(true)); return e }},
- {"array_from_int", func() error { _, e := r.ResolveArray(Integer(1)); return e }},
- {"stream_from_int", func() error { _, e := r.DecodeStream(Integer(1)); return e }},
- }
- for _, tc := range errCases {
- if tc.call() == nil {
- t.Errorf("%s: expected an error, got nil", tc.name)
- }
- }
- // Controls: the right type resolves cleanly.
- if b, err := r.ResolveBool(Bool(true)); err != nil || !b {
- t.Errorf("ResolveBool(true) = %v, %v", b, err)
- }
- if n, err := r.ResolveInt(Integer(7)); err != nil || n != 7 {
- t.Errorf("ResolveInt(7) = %v, %v", n, err)
- }
- if a, err := r.ResolveArray(Array{Integer(1)}); err != nil || len(a) != 1 {
- t.Errorf("ResolveArray = %v, %v", a, err)
- }
-}
-
-// OpenFile must surface the os.Open error for a missing path, and must not leak
-// the descriptor when the file opens but doesn't parse as a PDF.
-func TestOpenFileErrors(t *testing.T) {
- if _, err := OpenFile(filepath.Join(t.TempDir(), "missing.pdf")); err == nil {
- t.Error("OpenFile(missing) should error")
- }
- bad := filepath.Join(t.TempDir(), "bad.pdf")
- if err := os.WriteFile(bad, []byte("not a pdf"), 0o644); err != nil {
- t.Fatal(err)
- }
- if _, err := OpenFile(bad); err == nil {
- t.Error("OpenFile(garbage) should error")
- }
-}
-
-func TestXrefFormat(t *testing.T) {
- withTrailer := func(set func(d *Dict)) *Reader {
- d := newDict(nil)
- set(d)
- return &Reader{trailer: d}
- }
- if got := (&Reader{}).xrefFormat(); got != "unknown" {
- t.Errorf("nil trailer = %q, want unknown", got)
- }
- if got := withTrailer(func(d *Dict) { d.set("Type", Name("XRef")) }).xrefFormat(); got != "stream" {
- t.Errorf("/Type /XRef = %q, want stream", got)
- }
- if got := withTrailer(func(d *Dict) { d.set("XRefStm", Integer(99)) }).xrefFormat(); got != "hybrid" {
- t.Errorf("/XRefStm = %q, want hybrid", got)
- }
- if got := withTrailer(func(d *Dict) { d.set("Size", Integer(4)) }).xrefFormat(); got != "classical" {
- t.Errorf("plain trailer = %q, want classical", got)
- }
-}
-
-// Non-standard /Info keys land in Custom; a non-string entry is skipped, not
-// rendered.
-func TestDocumentInfoRichFields(t *testing.T) {
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
- offsets := make([]int, 4)
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
- offsets[3] = off()
- fmt.Fprint(&buf, "3 0 obj\n<< /Author (Ada) /Subject (Math) /Keywords (a,b) "+
- "/Creator (X) /Producer (Y) /CreationDate (D:20200102030405Z) "+
- "/ModDate (D:20210102030405Z) /Custom (cval) /NotAString 42 >>\nendobj\n")
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 4\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 3; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprint(&buf, "trailer\n<< /Size 4 /Root 1 0 R /Info 3 0 R >>\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
-
- r, err := Open(bytes.NewReader(buf.Bytes()))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- info := r.DocumentInfo()
- if info.Author != "Ada" || info.Subject != "Math" || info.Keywords != "a,b" ||
- info.Creator != "X" || info.Producer != "Y" {
- t.Errorf("string fields wrong: %+v", info)
- }
- if info.CreationDate.Year() != 2020 || info.ModDate.Year() != 2021 {
- t.Errorf("dates wrong: created %v, mod %v", info.CreationDate, info.ModDate)
- }
- if info.Custom["Custom"] != "cval" {
- t.Errorf("Custom[Custom] = %q, want cval", info.Custom["Custom"])
- }
- if _, ok := info.Custom["NotAString"]; ok {
- t.Error("non-string /NotAString should be skipped, not collected")
- }
-}
-
-func TestObjectsIteration(t *testing.T) {
- r, err := Open(bytes.NewReader(buildMinimalPDF(t)))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- var nums []int
- for e := range r.Objects() {
- nums = append(nums, e.Reference.Number)
- }
- if len(nums) < 4 {
- t.Fatalf("iterated %d objects, want >= 4", len(nums))
- }
- for i := 1; i < len(nums); i++ {
- if nums[i] < nums[i-1] {
- t.Errorf("objects out of order: %d before %d", nums[i-1], nums[i])
- }
- }
- count := 0
- for range r.Objects() {
- count++
- break
- }
- if count != 1 {
- t.Fatalf("early break iterated %d, want 1", count)
- }
-}
-
-// A reference to an object absent from the xref table resolves to null per
-// §7.3.10, not an error.
-func TestResolveDanglingReference(t *testing.T) {
- r, err := Open(bytes.NewReader(buildMinimalPDF(t)))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- v, err := r.Resolve(Reference{Number: 99})
- if err != nil {
- t.Fatalf("dangling ref: %v", err)
- }
- if _, ok := v.(Null); !ok {
- t.Errorf("dangling ref = %T, want Null", v)
- }
-}
diff --git a/testdata/fixtures/brotli-stream/golden.json b/testdata/fixtures/brotli-stream/golden.json
deleted file mode 100644
index c7b0236..0000000
--- a/testdata/fixtures/brotli-stream/golden.json
+++ /dev/null
@@ -1,84 +0,0 @@
-{
- "version": "2.0",
- "xref_format": "classical",
- "encrypted": false,
- "trailer": {
- "dict": {
- "Size": {
- "int": 4
- },
- "Root": {
- "ref": "1 0"
- }
- }
- },
- "objects": {
- "1 0": {
- "dict": {
- "Type": {
- "name": "Catalog"
- },
- "Pages": {
- "ref": "2 0"
- },
- "Extensions": {
- "dict": {
- "PDFa": {
- "dict": {
- "Type": {
- "name": "DeveloperExtensions"
- },
- "BaseVersion": {
- "name": "2.0"
- },
- "ExtensionLevel": {
- "int": 1
- },
- "ExtensionRevision": {
- "text": "2026"
- },
- "URL": {
- "text": "https://pdfa.org/resource/extension-brotli"
- }
- }
- }
- }
- }
- }
- },
- "2 0": {
- "dict": {
- "Type": {
- "name": "Pages"
- },
- "Count": {
- "int": 0
- },
- "Kids": {
- "array": []
- }
- }
- },
- "3 0": {
- "stream": {
- "dict": {
- "Length": {
- "int": 25
- },
- "Filter": {
- "name": "BrotliDecode"
- }
- },
- "raw_length": 25,
- "filters": [
- "BrotliDecode"
- ],
- "decoded": {
- "length": 21,
- "sha256": "278aa4570e132c7f29e5fb147da4b8f6f4405b77ada770d5623e7e52cbf51dde",
- "preview_utf8": "Hello, Brotli stream!"
- }
- }
- }
- }
-}
diff --git a/testdata/fixtures/brotli-stream/input.pdf b/testdata/fixtures/brotli-stream/input.pdf
deleted file mode 100644
index 635e746..0000000
Binary files a/testdata/fixtures/brotli-stream/input.pdf and /dev/null differ
diff --git a/testdata/fixtures/flate-stream/golden.json b/testdata/fixtures/flate-stream/golden.json
deleted file mode 100644
index 489bc10..0000000
--- a/testdata/fixtures/flate-stream/golden.json
+++ /dev/null
@@ -1,61 +0,0 @@
-{
- "version": "1.7",
- "xref_format": "classical",
- "encrypted": false,
- "trailer": {
- "dict": {
- "Size": {
- "int": 4
- },
- "Root": {
- "ref": "1 0"
- }
- }
- },
- "objects": {
- "1 0": {
- "dict": {
- "Type": {
- "name": "Catalog"
- },
- "Pages": {
- "ref": "2 0"
- }
- }
- },
- "2 0": {
- "dict": {
- "Type": {
- "name": "Pages"
- },
- "Count": {
- "int": 0
- },
- "Kids": {
- "array": []
- }
- }
- },
- "3 0": {
- "stream": {
- "dict": {
- "Length": {
- "int": 26
- },
- "Filter": {
- "name": "FlateDecode"
- }
- },
- "raw_length": 26,
- "filters": [
- "FlateDecode"
- ],
- "decoded": {
- "length": 14,
- "sha256": "4ac9a1927af81c4e7cd1b812848d3fdac958b613e4754a88af93b308df4ac7df",
- "preview_utf8": "Hello, stream!"
- }
- }
- }
- }
-}
diff --git a/testdata/fixtures/flate-stream/input.pdf b/testdata/fixtures/flate-stream/input.pdf
deleted file mode 100644
index efa6053..0000000
Binary files a/testdata/fixtures/flate-stream/input.pdf and /dev/null differ
diff --git a/testdata/fixtures/generate.go b/testdata/fixtures/generate.go
deleted file mode 100644
index 075166e..0000000
--- a/testdata/fixtures/generate.go
+++ /dev/null
@@ -1,234 +0,0 @@
-//go:build ignore
-
-// Run from the repo root:
-//
-// go run testdata/fixtures/generate.go
-//
-// (re)creates synthetic input.pdf files for the fixtures that we author
-// in code (rather than dropping in real-world samples). After running,
-// refresh goldens with:
-//
-// go test -update -run TestFixtures
-//
-// Then inspect the resulting golden.json before committing.
-package main
-
-import (
- "bytes"
- "compress/zlib"
- "fmt"
- "os"
- "path/filepath"
-
- "github.com/andybalholm/brotli"
-)
-
-func main() {
- write("minimal", minimalPDF())
- write("xref-stream", xrefStreamPDF())
- write("flate-stream", flateStreamPDF())
- write("brotli-stream", brotliStreamPDF())
- write("page-inheritance", pageInheritancePDF())
- write("page-contents-array", pageContentsArrayPDF())
-}
-
-func write(name string, data []byte) {
- dir := filepath.Join("testdata/fixtures", name)
- if err := os.MkdirAll(dir, 0o755); err != nil {
- panic(err)
- }
- path := filepath.Join(dir, "input.pdf")
- if err := os.WriteFile(path, data, 0o644); err != nil {
- panic(err)
- }
- fmt.Printf("wrote %s (%d bytes)\n", path, len(data))
-}
-
-func minimalPDF() []byte {
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
-
- offsets := make([]int, 5)
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>\nendobj\n")
- offsets[3] = off()
- fmt.Fprint(&buf, "3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] >>\nendobj\n")
- offsets[4] = off()
- fmt.Fprint(&buf, "4 0 obj\n<< /Title (Hello) /Producer (pdfdisassembler-fixture) >>\nendobj\n")
-
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 5\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 4; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprintf(&buf, "trailer\n<< /Size 5 /Root 1 0 R /Info 4 0 R >>\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-func xrefStreamPDF() []byte {
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-2.0\n%\xE2\xE3\xCF\xD3\n")
-
- offsets := make([]int, 3)
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
-
- rows := []byte{}
- add := func(typ, f1, f2 uint64) {
- rows = append(rows, byte(typ))
- rows = append(rows, byte(f1>>16), byte(f1>>8), byte(f1))
- rows = append(rows, byte(f2))
- }
- add(0, 0, 0xFFFF)
- add(1, uint64(offsets[1]), 0)
- add(1, uint64(offsets[2]), 0)
-
- var zbuf bytes.Buffer
- zw := zlib.NewWriter(&zbuf)
- zw.Write(rows)
- zw.Close()
- compressed := zbuf.Bytes()
-
- xrefOff := off()
- fmt.Fprintf(&buf,
- "3 0 obj\n<< /Type /XRef /Size 3 /W [1 3 1] /Root 1 0 R /Filter /FlateDecode /Length %d >>\nstream\n",
- len(compressed))
- buf.Write(compressed)
- fmt.Fprint(&buf, "\nendstream\nendobj\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-// classicalPDF assembles objs (1-based bodies) into a PDF with a classical
-// xref table and the given trailer dictionary body (without the surrounding
-// << >>).
-func classicalPDF(version string, objs []string, trailer string) []byte {
- var buf bytes.Buffer
- fmt.Fprintf(&buf, "%%PDF-%s\n%%\xE2\xE3\xCF\xD3\n", version)
- offsets := make([]int, len(objs)+1)
- for i, body := range objs {
- offsets[i+1] = buf.Len()
- fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", i+1, body)
- }
- xrefOff := buf.Len()
- fmt.Fprintf(&buf, "xref\n0 %d\n%010d %05d f \n", len(objs)+1, 0, 65535)
- for i := 1; i <= len(objs); i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprintf(&buf, "trailer\n<< /Size %d %s >>\nstartxref\n%d\n%%%%EOF\n",
- len(objs)+1, trailer, xrefOff)
- return buf.Bytes()
-}
-
-// pageInheritancePDF builds a three-level page tree where leaf pages inherit
-// /MediaBox and /Rotate two levels up (from the /Pages root), /CropBox and
-// /Resources one level up, and the second leaf overrides /MediaBox and /Rotate
-// locally. It exercises attribute inheritance per PDF 32000-1:2008 §7.7.3.4.
-func pageInheritancePDF() []byte {
- return classicalPDF("1.7", []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 2 /Kids [ 3 0 R ] /MediaBox [0 0 612 792] /Resources << /Font << /F1 6 0 R >> >> /Rotate 90 >>",
- "<< /Type /Pages /Count 2 /Parent 2 0 R /Kids [ 4 0 R 5 0 R ] /CropBox [10 10 602 782] >>",
- "<< /Type /Page /Parent 3 0 R >>",
- "<< /Type /Page /Parent 3 0 R /MediaBox [0 0 200 200] /Rotate 0 >>",
- "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>",
- }, "/Root 1 0 R")
-}
-
-// pageContentsArrayPDF builds a single page whose /Contents is an array of two
-// uncompressed content streams that must be concatenated.
-func pageContentsArrayPDF() []byte {
- const c1 = "q 1 0 0 1 50 50 cm"
- const c2 = "BT /F1 12 Tf (Hello) Tj ET"
- return classicalPDF("1.7", []string{
- "<< /Type /Catalog /Pages 2 0 R >>",
- "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>",
- "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Contents [ 4 0 R 5 0 R ] >>",
- fmt.Sprintf("<< /Length %d >>\nstream\n%s\nendstream", len(c1), c1),
- fmt.Sprintf("<< /Length %d >>\nstream\n%s\nendstream", len(c2), c2),
- }, "/Root 1 0 R")
-}
-
-// brotliStreamPDF mirrors flateStreamPDF with a BrotliDecode stream (PDF
-// Association extension EXTN-BROTLI-1). Per that spec the catalog SHOULD
-// declare the extension in an /Extensions dictionary under the PDFa prefix.
-func brotliStreamPDF() []byte {
- const payload = "Hello, Brotli stream!"
- var bbuf bytes.Buffer
- bw := brotli.NewWriter(&bbuf)
- if _, err := bw.Write([]byte(payload)); err != nil {
- panic(err)
- }
- if err := bw.Close(); err != nil {
- panic(err)
- }
- bdata := bbuf.Bytes()
-
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-2.0\n%\xE2\xE3\xCF\xD3\n")
-
- offsets := make([]int, 4)
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R "+
- "/Extensions << /PDFa << /Type /DeveloperExtensions /BaseVersion /2.0 "+
- "/ExtensionLevel 1 /ExtensionRevision (2026) "+
- "/URL (https://pdfa.org/resource/extension-brotli) >> >> >>\nendobj\n")
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
- offsets[3] = off()
- fmt.Fprintf(&buf, "3 0 obj\n<< /Length %d /Filter /BrotliDecode >>\nstream\n", len(bdata))
- buf.Write(bdata)
- fmt.Fprint(&buf, "\nendstream\nendobj\n")
-
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 4\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 3; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprintf(&buf, "trailer\n<< /Size 4 /Root 1 0 R >>\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-func flateStreamPDF() []byte {
- const payload = "Hello, stream!"
- var zbuf bytes.Buffer
- zw := zlib.NewWriter(&zbuf)
- zw.Write([]byte(payload))
- zw.Close()
- zdata := zbuf.Bytes()
-
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n")
-
- offsets := make([]int, 4)
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
- offsets[3] = off()
- fmt.Fprintf(&buf, "3 0 obj\n<< /Length %d /Filter /FlateDecode >>\nstream\n", len(zdata))
- buf.Write(zdata)
- fmt.Fprint(&buf, "\nendstream\nendobj\n")
-
- xrefOff := off()
- fmt.Fprint(&buf, "xref\n0 4\n")
- fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535)
- for i := 1; i <= 3; i++ {
- fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0)
- }
- fmt.Fprintf(&buf, "trailer\n<< /Size 4 /Root 1 0 R >>\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
diff --git a/testdata/fixtures/minimal/golden.json b/testdata/fixtures/minimal/golden.json
deleted file mode 100644
index 4bce377..0000000
--- a/testdata/fixtures/minimal/golden.json
+++ /dev/null
@@ -1,83 +0,0 @@
-{
- "version": "1.7",
- "xref_format": "classical",
- "encrypted": false,
- "trailer": {
- "dict": {
- "Size": {
- "int": 5
- },
- "Root": {
- "ref": "1 0"
- },
- "Info": {
- "ref": "4 0"
- }
- }
- },
- "objects": {
- "1 0": {
- "dict": {
- "Type": {
- "name": "Catalog"
- },
- "Pages": {
- "ref": "2 0"
- }
- }
- },
- "2 0": {
- "dict": {
- "Type": {
- "name": "Pages"
- },
- "Count": {
- "int": 1
- },
- "Kids": {
- "array": [
- {
- "ref": "3 0"
- }
- ]
- }
- }
- },
- "3 0": {
- "dict": {
- "Type": {
- "name": "Page"
- },
- "Parent": {
- "ref": "2 0"
- },
- "MediaBox": {
- "array": [
- {
- "int": 0
- },
- {
- "int": 0
- },
- {
- "int": 612
- },
- {
- "int": 792
- }
- ]
- }
- }
- },
- "4 0": {
- "dict": {
- "Title": {
- "text": "Hello"
- },
- "Producer": {
- "text": "pdfdisassembler-fixture"
- }
- }
- }
- }
-}
diff --git a/testdata/fixtures/minimal/input.pdf b/testdata/fixtures/minimal/input.pdf
deleted file mode 100644
index 9a46979..0000000
Binary files a/testdata/fixtures/minimal/input.pdf and /dev/null differ
diff --git a/testdata/fixtures/page-contents-array/golden.json b/testdata/fixtures/page-contents-array/golden.json
deleted file mode 100644
index 3028856..0000000
--- a/testdata/fixtures/page-contents-array/golden.json
+++ /dev/null
@@ -1,112 +0,0 @@
-{
- "version": "1.7",
- "xref_format": "classical",
- "encrypted": false,
- "trailer": {
- "dict": {
- "Size": {
- "int": 6
- },
- "Root": {
- "ref": "1 0"
- }
- }
- },
- "objects": {
- "1 0": {
- "dict": {
- "Type": {
- "name": "Catalog"
- },
- "Pages": {
- "ref": "2 0"
- }
- }
- },
- "2 0": {
- "dict": {
- "Type": {
- "name": "Pages"
- },
- "Count": {
- "int": 1
- },
- "Kids": {
- "array": [
- {
- "ref": "3 0"
- }
- ]
- }
- }
- },
- "3 0": {
- "dict": {
- "Type": {
- "name": "Page"
- },
- "Parent": {
- "ref": "2 0"
- },
- "MediaBox": {
- "array": [
- {
- "int": 0
- },
- {
- "int": 0
- },
- {
- "int": 200
- },
- {
- "int": 200
- }
- ]
- },
- "Contents": {
- "array": [
- {
- "ref": "4 0"
- },
- {
- "ref": "5 0"
- }
- ]
- }
- }
- },
- "4 0": {
- "stream": {
- "dict": {
- "Length": {
- "int": 18
- }
- },
- "raw_length": 18,
- "filters": [],
- "decoded": {
- "length": 18,
- "sha256": "ac399976dc867824f50e13f5f86d9f5b9be76d49ae538a01d7567552914e9f80",
- "preview_utf8": "q 1 0 0 1 50 50 cm"
- }
- }
- },
- "5 0": {
- "stream": {
- "dict": {
- "Length": {
- "int": 26
- }
- },
- "raw_length": 26,
- "filters": [],
- "decoded": {
- "length": 26,
- "sha256": "9dd85240377330fc34370ec0a297930e7cba5a30efbaab71f21337dbfc1d864e",
- "preview_utf8": "BT /F1 12 Tf (Hello) Tj ET"
- }
- }
- }
- }
-}
diff --git a/testdata/fixtures/page-contents-array/input.pdf b/testdata/fixtures/page-contents-array/input.pdf
deleted file mode 100644
index 76a6970..0000000
Binary files a/testdata/fixtures/page-contents-array/input.pdf and /dev/null differ
diff --git a/testdata/fixtures/page-inheritance/golden.json b/testdata/fixtures/page-inheritance/golden.json
deleted file mode 100644
index 0922e2e..0000000
--- a/testdata/fixtures/page-inheritance/golden.json
+++ /dev/null
@@ -1,165 +0,0 @@
-{
- "version": "1.7",
- "xref_format": "classical",
- "encrypted": false,
- "trailer": {
- "dict": {
- "Size": {
- "int": 7
- },
- "Root": {
- "ref": "1 0"
- }
- }
- },
- "objects": {
- "1 0": {
- "dict": {
- "Type": {
- "name": "Catalog"
- },
- "Pages": {
- "ref": "2 0"
- }
- }
- },
- "2 0": {
- "dict": {
- "Type": {
- "name": "Pages"
- },
- "Count": {
- "int": 2
- },
- "Kids": {
- "array": [
- {
- "ref": "3 0"
- }
- ]
- },
- "MediaBox": {
- "array": [
- {
- "int": 0
- },
- {
- "int": 0
- },
- {
- "int": 612
- },
- {
- "int": 792
- }
- ]
- },
- "Resources": {
- "dict": {
- "Font": {
- "dict": {
- "F1": {
- "ref": "6 0"
- }
- }
- }
- }
- },
- "Rotate": {
- "int": 90
- }
- }
- },
- "3 0": {
- "dict": {
- "Type": {
- "name": "Pages"
- },
- "Count": {
- "int": 2
- },
- "Parent": {
- "ref": "2 0"
- },
- "Kids": {
- "array": [
- {
- "ref": "4 0"
- },
- {
- "ref": "5 0"
- }
- ]
- },
- "CropBox": {
- "array": [
- {
- "int": 10
- },
- {
- "int": 10
- },
- {
- "int": 602
- },
- {
- "int": 782
- }
- ]
- }
- }
- },
- "4 0": {
- "dict": {
- "Type": {
- "name": "Page"
- },
- "Parent": {
- "ref": "3 0"
- }
- }
- },
- "5 0": {
- "dict": {
- "Type": {
- "name": "Page"
- },
- "Parent": {
- "ref": "3 0"
- },
- "MediaBox": {
- "array": [
- {
- "int": 0
- },
- {
- "int": 0
- },
- {
- "int": 200
- },
- {
- "int": 200
- }
- ]
- },
- "Rotate": {
- "int": 0
- }
- }
- },
- "6 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "Subtype": {
- "name": "Type1"
- },
- "BaseFont": {
- "name": "Helvetica"
- }
- }
- }
- }
-}
diff --git a/testdata/fixtures/page-inheritance/input.pdf b/testdata/fixtures/page-inheritance/input.pdf
deleted file mode 100644
index e293dec..0000000
Binary files a/testdata/fixtures/page-inheritance/input.pdf and /dev/null differ
diff --git a/testdata/fixtures/pdfua-demo/README.md b/testdata/fixtures/pdfua-demo/README.md
deleted file mode 100644
index 5729ce6..0000000
--- a/testdata/fixtures/pdfua-demo/README.md
+++ /dev/null
@@ -1,22 +0,0 @@
-# pdfua-demo
-
-A PDF/UA document produced by [glu](https://boxesandglue.dev) (the
-boxesandglue typesetter), originally part of the speedata Marketing
-material `glu-strategie/pdfua-demo.pdf`. Single-page, tagged via
-Markdown front matter.
-
-Covers in one fixture:
-
-- Classical xref
-- `/StructTreeRoot` with H1/H2/P/L/LI/LBody/BlockQuote/Code/Figure tags
-- `/RoleMap` (none — the doc uses standard structure types)
-- `/MarkInfo`, `/Lang`, `/ViewerPreferences`
-- `/Metadata` XMP stream
-- `/ParentTree` with mixed `Nums` array (int + ref + int + array of refs)
-- Type0 / CIDFont / ToUnicode CMap fonts
-- FlateDecode content stream with PDF/UA marked-content operators
- (`/H1 <> BDC` etc.)
-- `ActualText` entries in PDFDocEncoding with the en-dash (`0x85`)
-- A binary file ID (`/ID` array)
-
-This is what we want our parser to be able to read losslessly.
diff --git a/testdata/fixtures/pdfua-demo/golden.json b/testdata/fixtures/pdfua-demo/golden.json
deleted file mode 100644
index ef80cf7..0000000
--- a/testdata/fixtures/pdfua-demo/golden.json
+++ /dev/null
@@ -1,3204 +0,0 @@
-{
- "version": "1.7",
- "xref_format": "classical",
- "encrypted": false,
- "trailer": {
- "dict": {
- "ID": {
- "array": [
- {
- "hex": "62a379b5a12ce31bef329a1f37876600"
- },
- {
- "hex": "62a379b5a12ce31bef329a1f37876600"
- }
- ]
- },
- "Info": {
- "ref": "59 0"
- },
- "Root": {
- "ref": "38 0"
- },
- "Size": {
- "int": 60
- }
- }
- },
- "objects": {
- "1 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "BaseFont": {
- "name": "BZDLFS+CrimsonPro-Regular"
- },
- "DescendantFonts": {
- "array": [
- {
- "ref": "42 0"
- }
- ]
- },
- "Encoding": {
- "name": "Identity-H"
- },
- "Subtype": {
- "name": "Type0"
- },
- "ToUnicode": {
- "ref": "41 0"
- }
- }
- },
- "2 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "BaseFont": {
- "name": "LNQEXG+CrimsonPro-Bold"
- },
- "DescendantFonts": {
- "array": [
- {
- "ref": "46 0"
- }
- ]
- },
- "Encoding": {
- "name": "Identity-H"
- },
- "Subtype": {
- "name": "Type0"
- },
- "ToUnicode": {
- "ref": "45 0"
- }
- }
- },
- "3 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "BaseFont": {
- "name": "AVREHH+CrimsonPro-Italic"
- },
- "DescendantFonts": {
- "array": [
- {
- "ref": "50 0"
- }
- ]
- },
- "Encoding": {
- "name": "Identity-H"
- },
- "Subtype": {
- "name": "Type0"
- },
- "ToUnicode": {
- "ref": "49 0"
- }
- }
- },
- "4 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "BaseFont": {
- "name": "JYZTRN+CamingoCode-Bold"
- },
- "DescendantFonts": {
- "array": [
- {
- "ref": "54 0"
- }
- ]
- },
- "Encoding": {
- "name": "Identity-H"
- },
- "Subtype": {
- "name": "Type0"
- },
- "ToUnicode": {
- "ref": "53 0"
- }
- }
- },
- "5 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "BaseFont": {
- "name": "YCXYQB+CamingoCode-Regular"
- },
- "DescendantFonts": {
- "array": [
- {
- "ref": "58 0"
- }
- ]
- },
- "Encoding": {
- "name": "Identity-H"
- },
- "Subtype": {
- "name": "Type0"
- },
- "ToUnicode": {
- "ref": "57 0"
- }
- }
- },
- "6 0": {
- "dict": {
- "Type": {
- "name": "Page"
- },
- "Contents": {
- "ref": "7 0"
- },
- "Parent": {
- "ref": "32 0"
- },
- "Resources": {
- "dict": {
- "Font": {
- "dict": {
- "F1": {
- "ref": "1 0"
- },
- "F3": {
- "ref": "2 0"
- },
- "F4": {
- "ref": "3 0"
- },
- "F5": {
- "ref": "4 0"
- },
- "F6": {
- "ref": "5 0"
- }
- }
- },
- "XObject": {
- "dict": {
- "ImgBag2": {
- "ref": "8 0"
- }
- }
- }
- }
- },
- "StructParents": {
- "int": 1
- },
- "TrimBox": {
- "array": [
- {
- "int": 0
- },
- {
- "int": 0
- },
- {
- "real": 595.28
- },
- {
- "real": 841.89
- }
- ]
- }
- }
- },
- "7 0": {
- "stream": {
- "dict": {
- "Filter": {
- "name": "FlateDecode"
- },
- "Length": {
- "int": 2737
- },
- "Length1": {
- "int": 14503
- }
- },
- "raw_length": 2737,
- "filters": [
- "FlateDecode"
- ],
- "decoded": {
- "length": 14503,
- "sha256": "5cf965cecc849327e8aff22af072173c9566363887f142c8626d60c8a237e45a",
- "preview_utf8": "/Artifact BMC\n\nEMC\n/H1 <> BDC\n0.05 0.23 0.4 rg BT 100 Tz 0 Ts \n/F3 22 T…"
- }
- }
- },
- "8 0": {
- "stream": {
- "dict": {
- "Filter": {
- "name": "FlateDecode"
- },
- "Type": {
- "name": "XObject"
- },
- "Subtype": {
- "name": "Form"
- },
- "FormType": {
- "int": 1
- },
- "BBox": {
- "array": [
- {
- "real": 0
- },
- {
- "real": 0
- },
- {
- "real": 841.89
- },
- {
- "real": 595.28
- }
- ]
- },
- "StructParent": {
- "int": 0
- },
- "Resources": {
- "dict": {
- "ExtGState": {
- "dict": {
- "a0": {
- "dict": {
- "CA": {
- "int": 1
- },
- "ca": {
- "int": 1
- }
- }
- }
- }
- }
- }
- },
- "Length": {
- "int": 68069
- }
- },
- "raw_length": 68069,
- "filters": [
- "FlateDecode"
- ],
- "decoded": {
- "length": 171668,
- "sha256": "f313e213218ea6cc57612fe8ad542b0d00918f3ad2c76bab1aba18f0fe142049",
- "preview_utf8": "q\n0 0 0 rg /a0 gs\n462.723 162.502 m 447.234 162.502 434.492 150.619 433.027 135.…"
- }
- }
- },
- "9 0": {
- "dict": {
- "Type": {
- "name": "StructTreeRoot"
- },
- "K": {
- "ref": "10 0"
- },
- "ParentTree": {
- "dict": {
- "Nums": {
- "array": [
- {
- "int": 0
- },
- {
- "ref": "30 0"
- },
- {
- "int": 1
- },
- {
- "array": [
- {
- "ref": "11 0"
- },
- {
- "ref": "12 0"
- },
- {
- "ref": "13 0"
- },
- {
- "ref": "14 0"
- },
- {
- "ref": "17 0"
- },
- {
- "ref": "19 0"
- },
- {
- "ref": "21 0"
- },
- {
- "ref": "23 0"
- },
- {
- "ref": "25 0"
- },
- {
- "ref": "26 0"
- },
- {
- "ref": "28 0"
- },
- {
- "ref": "29 0"
- }
- ]
- }
- ]
- }
- }
- }
- }
- },
- "10 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "K": {
- "array": [
- {
- "ref": "11 0"
- },
- {
- "ref": "12 0"
- },
- {
- "ref": "13 0"
- },
- {
- "ref": "14 0"
- },
- {
- "ref": "15 0"
- },
- {
- "ref": "24 0"
- },
- {
- "ref": "26 0"
- },
- {
- "ref": "27 0"
- },
- {
- "ref": "29 0"
- },
- {
- "ref": "30 0"
- }
- ]
- },
- "P": {
- "ref": "9 0"
- },
- "S": {
- "name": "Document"
- }
- }
- },
- "11 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "Markdown to PDF/UA"
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 0
- }
- }
- },
- "P": {
- "ref": "10 0"
- },
- "S": {
- "name": "H1"
- }
- }
- },
- "12 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "This page was typeset from Markdown – and is fully tagged."
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 1
- }
- }
- },
- "P": {
- "ref": "10 0"
- },
- "S": {
- "name": "P"
- }
- }
- },
- "13 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "Rendered with glu (boxes and glue). The YAML front matter declares format: PDF/UA, lang: en and a title: – that is all it takes. From those three lines glu automatically writes StructTreeRoot, MarkInfo, /DisplayDocTitle, /Lang and the XMP tag pdfuaid:part 1."
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 2
- }
- }
- },
- "P": {
- "ref": "10 0"
- },
- "S": {
- "name": "P"
- }
- }
- },
- "14 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "What is structurally tagged here"
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 3
- }
- }
- },
- "P": {
- "ref": "10 0"
- },
- "S": {
- "name": "H2"
- }
- }
- },
- "15 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "K": {
- "array": [
- {
- "ref": "16 0"
- },
- {
- "ref": "18 0"
- },
- {
- "ref": "20 0"
- },
- {
- "ref": "22 0"
- }
- ]
- },
- "P": {
- "ref": "10 0"
- },
- "S": {
- "name": "L"
- }
- }
- },
- "16 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "K": {
- "ref": "17 0"
- },
- "P": {
- "ref": "15 0"
- },
- "S": {
- "name": "LI"
- }
- }
- },
- "17 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "Headings build the outline tree (H1, H2)."
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 4
- }
- }
- },
- "P": {
- "ref": "16 0"
- },
- "S": {
- "name": "LBody"
- }
- }
- },
- "18 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "K": {
- "ref": "19 0"
- },
- "P": {
- "ref": "15 0"
- },
- "S": {
- "name": "LI"
- }
- }
- },
- "19 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "Paragraphs become P, lists become L / LI."
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 5
- }
- }
- },
- "P": {
- "ref": "18 0"
- },
- "S": {
- "name": "LBody"
- }
- }
- },
- "20 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "K": {
- "ref": "21 0"
- },
- "P": {
- "ref": "15 0"
- },
- "S": {
- "name": "LI"
- }
- }
- },
- "21 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "The figure below is a Figure element with /Alt text."
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 6
- }
- }
- },
- "P": {
- "ref": "20 0"
- },
- "S": {
- "name": "LBody"
- }
- }
- },
- "22 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "K": {
- "ref": "23 0"
- },
- "P": {
- "ref": "15 0"
- },
- "S": {
- "name": "LI"
- }
- }
- },
- "23 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "Code blocks are emitted as Code."
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 7
- }
- }
- },
- "P": {
- "ref": "22 0"
- },
- "S": {
- "name": "LBody"
- }
- }
- },
- "24 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "K": {
- "ref": "25 0"
- },
- "P": {
- "ref": "10 0"
- },
- "S": {
- "name": "BlockQuote"
- }
- }
- },
- "25 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "Acrobat, NVDA and PAC all see this as a properly structured document – no artifacts, no untagged content."
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 8
- }
- }
- },
- "P": {
- "ref": "24 0"
- },
- "S": {
- "name": "P"
- }
- }
- },
- "26 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "Front matter"
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 9
- }
- }
- },
- "P": {
- "ref": "10 0"
- },
- "S": {
- "name": "H2"
- }
- }
- },
- "27 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "K": {
- "ref": "28 0"
- },
- "P": {
- "ref": "10 0"
- },
- "S": {
- "name": "P"
- }
- }
- },
- "28 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "---\ntitle: Markdown to PDF/UA with glu\nlang: en\nformat: PDF/UA\n---"
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 10
- }
- }
- },
- "P": {
- "ref": "27 0"
- },
- "S": {
- "name": "Code"
- }
- }
- },
- "29 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "ActualText": {
- "text": "An embedded figure"
- },
- "K": {
- "dict": {
- "Type": {
- "name": "MCR"
- },
- "Pg": {
- "ref": "6 0"
- },
- "MCID": {
- "int": 11
- }
- }
- },
- "P": {
- "ref": "10 0"
- },
- "S": {
- "name": "H2"
- }
- }
- },
- "30 0": {
- "dict": {
- "Type": {
- "name": "StructElem"
- },
- "A": {
- "dict": {
- "O": {
- "name": "Layout"
- },
- "Placement": {
- "name": "Block"
- },
- "BBox": {
- "array": [
- {
- "real": 68.03
- },
- {
- "real": 488.22
- },
- {
- "real": 527.24
- },
- {
- "real": 628.52
- }
- ]
- }
- }
- },
- "Alt": {
- "text": "Photograph of the sea, embedded as a Form XObject"
- },
- "K": {
- "dict": {
- "Type": {
- "name": "OBJR"
- },
- "Obj": {
- "ref": "8 0"
- },
- "Pg": {
- "ref": "6 0"
- }
- }
- },
- "P": {
- "ref": "10 0"
- },
- "S": {
- "name": "Figure"
- }
- }
- },
- "31 0": {
- "stream": {
- "dict": {
- "Type": {
- "name": "Metadata"
- },
- "Length": {
- "int": 1229
- },
- "Subtype": {
- "name": "XML"
- }
- },
- "raw_length": 1229,
- "filters": [],
- "decoded": {
- "length": 1229,
- "sha256": "8f2ced3e481012ce7c7c6b2ac15d96d81161c60ab2ee94d6f60caf5566cb8d53"
- }
- }
- },
- "32 0": {
- "dict": {
- "Type": {
- "name": "Pages"
- },
- "Count": {
- "int": 1
- },
- "Kids": {
- "array": [
- {
- "ref": "6 0"
- }
- ]
- },
- "MediaBox": {
- "array": [
- {
- "int": 0
- },
- {
- "int": 0
- },
- {
- "real": 595.28
- },
- {
- "real": 841.89
- }
- ]
- }
- }
- },
- "33 0": {
- "dict": {
- "Type": {
- "name": "Outlines"
- },
- "Count": {
- "int": 4
- },
- "First": {
- "ref": "34 0"
- },
- "Last": {
- "ref": "37 0"
- }
- }
- },
- "34 0": {
- "dict": {
- "Dest": {
- "array": [
- {
- "ref": "6 0"
- },
- {
- "name": "Fit"
- }
- ]
- },
- "Next": {
- "ref": "35 0"
- },
- "Parent": {
- "ref": "33 0"
- },
- "Title": {
- "text": "Markdown to PDF/UA"
- }
- }
- },
- "35 0": {
- "dict": {
- "Dest": {
- "array": [
- {
- "ref": "6 0"
- },
- {
- "name": "Fit"
- }
- ]
- },
- "Next": {
- "ref": "36 0"
- },
- "Parent": {
- "ref": "33 0"
- },
- "Prev": {
- "ref": "34 0"
- },
- "Title": {
- "text": "What is structurally tagged here"
- }
- }
- },
- "36 0": {
- "dict": {
- "Dest": {
- "array": [
- {
- "ref": "6 0"
- },
- {
- "name": "Fit"
- }
- ]
- },
- "Next": {
- "ref": "37 0"
- },
- "Parent": {
- "ref": "33 0"
- },
- "Prev": {
- "ref": "35 0"
- },
- "Title": {
- "text": "Front matter"
- }
- }
- },
- "37 0": {
- "dict": {
- "Dest": {
- "array": [
- {
- "ref": "6 0"
- },
- {
- "name": "Fit"
- }
- ]
- },
- "Parent": {
- "ref": "33 0"
- },
- "Prev": {
- "ref": "36 0"
- },
- "Title": {
- "text": "An embedded figure"
- }
- }
- },
- "38 0": {
- "dict": {
- "Type": {
- "name": "Catalog"
- },
- "Outlines": {
- "ref": "33 0"
- },
- "Lang": {
- "text": "en"
- },
- "MarkInfo": {
- "dict": {
- "Marked": {
- "bool": true
- },
- "Suspects": {
- "bool": false
- }
- }
- },
- "Metadata": {
- "ref": "31 0"
- },
- "Pages": {
- "ref": "32 0"
- },
- "StructTreeRoot": {
- "ref": "9 0"
- },
- "ViewerPreferences": {
- "dict": {
- "DisplayDocTitle": {
- "bool": true
- }
- }
- }
- }
- },
- "39 0": {
- "stream": {
- "dict": {
- "Filter": {
- "name": "FlateDecode"
- },
- "Length": {
- "int": 4252
- },
- "Length1": {
- "int": 11864
- }
- },
- "raw_length": 4252,
- "filters": [
- "FlateDecode"
- ],
- "decoded": {
- "length": 11864,
- "sha256": "1db52ea7e54237618253599e8809208e196c393a66451435e032da108e451107"
- }
- }
- },
- "40 0": {
- "dict": {
- "Type": {
- "name": "FontDescriptor"
- },
- "Ascent": {
- "int": 918
- },
- "CapHeight": {
- "int": 587
- },
- "Descent": {
- "int": -220
- },
- "Flags": {
- "int": 32
- },
- "FontBBox": {
- "array": [
- {
- "int": -107
- },
- {
- "int": -283
- },
- {
- "int": 1159
- },
- {
- "int": 984
- }
- ]
- },
- "FontFile2": {
- "ref": "39 0"
- },
- "FontName": {
- "name": "BZDLFS+CrimsonPro-Regular"
- },
- "ItalicAngle": {
- "int": 0
- },
- "StemV": {
- "int": 80
- },
- "XHeight": {
- "int": 425
- }
- }
- },
- "41 0": {
- "stream": {
- "dict": {
- "Length": {
- "int": 879
- }
- },
- "raw_length": 879,
- "filters": [],
- "decoded": {
- "length": 879,
- "sha256": "8a595153c92f4d9492523ca3633beb65c34503f63db4c5bcd9e62363c566b99c",
- "preview_utf8": "/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n/CIDSystemInfo << /…"
- }
- }
- },
- "42 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "BaseFont": {
- "name": "BZDLFS+CrimsonPro-Regular"
- },
- "CIDSystemInfo": {
- "dict": {
- "Ordering": {
- "text": "Identity"
- },
- "Registry": {
- "text": "Adobe"
- },
- "Supplement": {
- "int": 0
- }
- }
- },
- "CIDToGIDMap": {
- "name": "Identity"
- },
- "FontDescriptor": {
- "ref": "40 0"
- },
- "Subtype": {
- "name": "CIDFontType2"
- },
- "W": {
- "array": [
- {
- "int": 0
- },
- {
- "array": [
- {
- "int": 500
- },
- {
- "real": 568.359375
- }
- ]
- },
- {
- "int": 30
- },
- {
- "array": [
- {
- "real": 589.84375
- }
- ]
- },
- {
- "int": 68
- },
- {
- "array": [
- {
- "real": 494.140625
- }
- ]
- },
- {
- "int": 76
- },
- {
- "array": [
- {
- "real": 656.25
- }
- ]
- },
- {
- "int": 102
- },
- {
- "array": [
- {
- "real": 499.0234375
- }
- ]
- },
- {
- "int": 112
- },
- {
- "array": [
- {
- "real": 830.078125
- }
- ]
- },
- {
- "int": 160
- },
- {
- "array": [
- {
- "real": 501.953125
- }
- ]
- },
- {
- "int": 163
- },
- {
- "array": [
- {
- "real": 553.7109375
- }
- ]
- },
- {
- "int": 184
- },
- {
- "array": [
- {
- "real": 543.9453125
- }
- ]
- },
- {
- "int": 220
- },
- {
- "array": [
- {
- "real": 568.359375
- },
- {
- "real": 528.3203125
- }
- ]
- },
- {
- "int": 237
- },
- {
- "array": [
- {
- "real": 462.890625
- }
- ]
- },
- {
- "int": 265
- },
- {
- "array": [
- {
- "real": 513.671875
- },
- {
- "real": 415.0390625
- }
- ]
- },
- {
- "int": 273
- },
- {
- "array": [
- {
- "real": 526.3671875
- }
- ]
- },
- {
- "int": 280
- },
- {
- "array": [
- {
- "real": 439.453125
- }
- ]
- },
- {
- "int": 304
- },
- {
- "array": [
- {
- "real": 295.8984375
- },
- {
- "real": 483.3984375
- }
- ]
- },
- {
- "int": 312
- },
- {
- "array": [
- {
- "real": 537.109375
- }
- ]
- },
- {
- "int": 317
- },
- {
- "array": [
- {
- "real": 261.71875
- }
- ]
- },
- {
- "int": 337
- },
- {
- "array": [
- {
- "real": 477.5390625
- }
- ]
- },
- {
- "int": 340
- },
- {
- "array": [
- {
- "real": 262.6953125
- }
- ]
- },
- {
- "int": 349
- },
- {
- "array": [
- {
- "real": 803.7109375
- }
- ]
- },
- {
- "int": 351
- },
- {
- "array": [
- {
- "real": 537.109375
- }
- ]
- },
- {
- "int": 362
- },
- {
- "array": [
- {
- "real": 496.09375
- }
- ]
- },
- {
- "int": 397
- },
- {
- "array": [
- {
- "real": 524.4140625
- }
- ]
- },
- {
- "int": 400
- },
- {
- "array": [
- {
- "real": 355.46875
- }
- ]
- },
- {
- "int": 408
- },
- {
- "array": [
- {
- "real": 392.578125
- }
- ]
- },
- {
- "int": 420
- },
- {
- "array": [
- {
- "real": 331.0546875
- }
- ]
- },
- {
- "int": 428
- },
- {
- "array": [
- {
- "real": 530.2734375
- }
- ]
- },
- {
- "int": 452
- },
- {
- "array": [
- {
- "real": 735.3515625
- }
- ]
- },
- {
- "int": 457
- },
- {
- "array": [
- {
- "real": 469.7265625
- },
- {
- "real": 460.9375
- }
- ]
- },
- {
- "int": 478
- },
- {
- "array": [
- {
- "real": 534.1796875
- }
- ]
- },
- {
- "int": 577
- },
- {
- "array": [
- {
- "real": 228.515625
- },
- {
- "real": 228.515625
- }
- ]
- },
- {
- "int": 587
- },
- {
- "array": [
- {
- "real": 410.15625
- }
- ]
- },
- {
- "int": 590
- },
- {
- "array": [
- {
- "real": 357.421875
- }
- ]
- },
- {
- "int": 594
- },
- {
- "array": [
- {
- "real": 351.5625
- },
- {
- "real": 351.5625
- }
- ]
- },
- {
- "int": 602
- },
- {
- "array": [
- {
- "real": 488.28125
- }
- ]
- },
- {
- "int": 625
- },
- {
- "array": [
- {
- "real": 187.5
- }
- ]
- }
- ]
- }
- }
- },
- "43 0": {
- "stream": {
- "dict": {
- "Filter": {
- "name": "FlateDecode"
- },
- "Length": {
- "int": 3310
- },
- "Length1": {
- "int": 10584
- }
- },
- "raw_length": 3310,
- "filters": [
- "FlateDecode"
- ],
- "decoded": {
- "length": 10584,
- "sha256": "574f1438b46e26875646796fe4745383608ba239d8169eee37ad0f21f543411c"
- }
- }
- },
- "44 0": {
- "dict": {
- "Type": {
- "name": "FontDescriptor"
- },
- "Ascent": {
- "int": 918
- },
- "CapHeight": {
- "int": 587
- },
- "Descent": {
- "int": -220
- },
- "Flags": {
- "int": 32
- },
- "FontBBox": {
- "array": [
- {
- "int": -107
- },
- {
- "int": -283
- },
- {
- "int": 1159
- },
- {
- "int": 984
- }
- ]
- },
- "FontFile2": {
- "ref": "43 0"
- },
- "FontName": {
- "name": "LNQEXG+CrimsonPro-Bold"
- },
- "ItalicAngle": {
- "int": 0
- },
- "StemV": {
- "int": 140
- },
- "XHeight": {
- "int": 425
- }
- }
- },
- "45 0": {
- "stream": {
- "dict": {
- "Length": {
- "int": 710
- }
- },
- "raw_length": 710,
- "filters": [],
- "decoded": {
- "length": 710,
- "sha256": "6849b17e78a731b32e149b4f6f8d4cf7831549c1d38e8b503ef9bead6ffaa236",
- "preview_utf8": "/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n/CIDSystemInfo << /…"
- }
- }
- },
- "46 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "BaseFont": {
- "name": "LNQEXG+CrimsonPro-Bold"
- },
- "CIDSystemInfo": {
- "dict": {
- "Ordering": {
- "text": "Identity"
- },
- "Registry": {
- "text": "Adobe"
- },
- "Supplement": {
- "int": 0
- }
- }
- },
- "CIDToGIDMap": {
- "name": "Identity"
- },
- "FontDescriptor": {
- "ref": "44 0"
- },
- "Subtype": {
- "name": "CIDFontType2"
- },
- "W": {
- "array": [
- {
- "int": 0
- },
- {
- "array": [
- {
- "int": 500
- },
- {
- "real": 594.7265625
- }
- ]
- },
- {
- "int": 37
- },
- {
- "array": [
- {
- "real": 690.4296875
- }
- ]
- },
- {
- "int": 68
- },
- {
- "array": [
- {
- "real": 526.3671875
- }
- ]
- },
- {
- "int": 112
- },
- {
- "array": [
- {
- "real": 870.1171875
- }
- ]
- },
- {
- "int": 160
- },
- {
- "array": [
- {
- "real": 555.6640625
- }
- ]
- },
- {
- "int": 191
- },
- {
- "array": [
- {
- "real": 685.546875
- }
- ]
- },
- {
- "int": 215
- },
- {
- "array": [
- {
- "real": 955.078125
- }
- ]
- },
- {
- "int": 237
- },
- {
- "array": [
- {
- "real": 488.28125
- }
- ]
- },
- {
- "int": 265
- },
- {
- "array": [
- {
- "real": 544.921875
- },
- {
- "real": 449.21875
- }
- ]
- },
- {
- "int": 273
- },
- {
- "array": [
- {
- "real": 547.8515625
- }
- ]
- },
- {
- "int": 280
- },
- {
- "array": [
- {
- "real": 465.8203125
- }
- ]
- },
- {
- "int": 305
- },
- {
- "array": [
- {
- "real": 506.8359375
- }
- ]
- },
- {
- "int": 312
- },
- {
- "array": [
- {
- "real": 574.21875
- }
- ]
- },
- {
- "int": 317
- },
- {
- "array": [
- {
- "real": 291.015625
- }
- ]
- },
- {
- "int": 337
- },
- {
- "array": [
- {
- "real": 543.9453125
- }
- ]
- },
- {
- "int": 340
- },
- {
- "array": [
- {
- "real": 291.015625
- }
- ]
- },
- {
- "int": 349
- },
- {
- "array": [
- {
- "real": 846.6796875
- }
- ]
- },
- {
- "int": 351
- },
- {
- "array": [
- {
- "real": 574.21875
- }
- ]
- },
- {
- "int": 362
- },
- {
- "array": [
- {
- "real": 520.5078125
- }
- ]
- },
- {
- "int": 400
- },
- {
- "array": [
- {
- "real": 401.3671875
- }
- ]
- },
- {
- "int": 408
- },
- {
- "array": [
- {
- "real": 408.203125
- }
- ]
- },
- {
- "int": 420
- },
- {
- "array": [
- {
- "real": 360.3515625
- }
- ]
- },
- {
- "int": 428
- },
- {
- "array": [
- {
- "real": 568.359375
- }
- ]
- },
- {
- "int": 452
- },
- {
- "array": [
- {
- "real": 795.8984375
- }
- ]
- },
- {
- "int": 458
- },
- {
- "array": [
- {
- "real": 486.328125
- }
- ]
- },
- {
- "int": 478
- },
- {
- "array": [
- {
- "real": 584.9609375
- }
- ]
- },
- {
- "int": 590
- },
- {
- "array": [
- {
- "real": 406.25
- }
- ]
- },
- {
- "int": 625
- },
- {
- "array": [
- {
- "real": 192.3828125
- }
- ]
- }
- ]
- }
- }
- },
- "47 0": {
- "stream": {
- "dict": {
- "Filter": {
- "name": "FlateDecode"
- },
- "Length": {
- "int": 3679
- },
- "Length1": {
- "int": 11144
- }
- },
- "raw_length": 3679,
- "filters": [
- "FlateDecode"
- ],
- "decoded": {
- "length": 11144,
- "sha256": "3f54c3113c615cf618c245df334fa62a49c591b372659a292a97c995536f0c5a"
- }
- }
- },
- "48 0": {
- "dict": {
- "Type": {
- "name": "FontDescriptor"
- },
- "Ascent": {
- "int": 918
- },
- "CapHeight": {
- "int": 587
- },
- "Descent": {
- "int": -220
- },
- "Flags": {
- "int": 96
- },
- "FontBBox": {
- "array": [
- {
- "int": -155
- },
- {
- "int": -285
- },
- {
- "int": 1212
- },
- {
- "int": 985
- }
- ]
- },
- "FontFile2": {
- "ref": "47 0"
- },
- "FontName": {
- "name": "AVREHH+CrimsonPro-Italic"
- },
- "ItalicAngle": {
- "int": -12
- },
- "StemV": {
- "int": 80
- },
- "XHeight": {
- "int": 425
- }
- }
- },
- "49 0": {
- "stream": {
- "dict": {
- "Length": {
- "int": 762
- }
- },
- "raw_length": 762,
- "filters": [],
- "decoded": {
- "length": 762,
- "sha256": "54ef34fc7df73d04c468c08e310571b015ba6f8d61f95ab050564f20ac3a61cc",
- "preview_utf8": "/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n/CIDSystemInfo << /…"
- }
- }
- },
- "50 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "BaseFont": {
- "name": "AVREHH+CrimsonPro-Italic"
- },
- "CIDSystemInfo": {
- "dict": {
- "Ordering": {
- "text": "Identity"
- },
- "Registry": {
- "text": "Adobe"
- },
- "Supplement": {
- "int": 0
- }
- }
- },
- "CIDToGIDMap": {
- "name": "Identity"
- },
- "FontDescriptor": {
- "ref": "48 0"
- },
- "Subtype": {
- "name": "CIDFontType2"
- },
- "W": {
- "array": [
- {
- "int": 0
- },
- {
- "array": [
- {
- "int": 500
- },
- {
- "real": 568.359375
- }
- ]
- },
- {
- "int": 30
- },
- {
- "array": [
- {
- "real": 589.84375
- }
- ]
- },
- {
- "int": 37
- },
- {
- "array": [
- {
- "real": 666.015625
- }
- ]
- },
- {
- "int": 112
- },
- {
- "array": [
- {
- "real": 830.078125
- }
- ]
- },
- {
- "int": 114
- },
- {
- "array": [
- {
- "real": 658.203125
- }
- ]
- },
- {
- "int": 160
- },
- {
- "array": [
- {
- "real": 501.953125
- }
- ]
- },
- {
- "int": 184
- },
- {
- "array": [
- {
- "real": 543.9453125
- }
- ]
- },
- {
- "int": 214
- },
- {
- "array": [
- {
- "real": 571.2890625
- }
- ]
- },
- {
- "int": 237
- },
- {
- "array": [
- {
- "real": 470.703125
- }
- ]
- },
- {
- "int": 265
- },
- {
- "array": [
- {
- "real": 444.3359375
- },
- {
- "real": 351.5625
- }
- ]
- },
- {
- "int": 273
- },
- {
- "array": [
- {
- "real": 470.703125
- }
- ]
- },
- {
- "int": 280
- },
- {
- "array": [
- {
- "real": 378.90625
- }
- ]
- },
- {
- "int": 304
- },
- {
- "array": [
- {
- "real": 257.8125
- },
- {
- "real": 445.3125
- }
- ]
- },
- {
- "int": 312
- },
- {
- "array": [
- {
- "real": 481.4453125
- }
- ]
- },
- {
- "int": 317
- },
- {
- "array": [
- {
- "real": 270.5078125
- }
- ]
- },
- {
- "int": 337
- },
- {
- "array": [
- {
- "real": 436.5234375
- }
- ]
- },
- {
- "int": 340
- },
- {
- "array": [
- {
- "real": 244.140625
- }
- ]
- },
- {
- "int": 349
- },
- {
- "array": [
- {
- "real": 731.4453125
- }
- ]
- },
- {
- "int": 351
- },
- {
- "array": [
- {
- "real": 505.859375
- }
- ]
- },
- {
- "int": 362
- },
- {
- "array": [
- {
- "real": 426.7578125
- }
- ]
- },
- {
- "int": 397
- },
- {
- "array": [
- {
- "real": 475.5859375
- }
- ]
- },
- {
- "int": 400
- },
- {
- "array": [
- {
- "real": 348.6328125
- }
- ]
- },
- {
- "int": 408
- },
- {
- "array": [
- {
- "real": 308.59375
- }
- ]
- },
- {
- "int": 420
- },
- {
- "array": [
- {
- "real": 294.921875
- }
- ]
- },
- {
- "int": 428
- },
- {
- "array": [
- {
- "int": 500
- }
- ]
- },
- {
- "int": 452
- },
- {
- "array": [
- {
- "real": 638.671875
- }
- ]
- },
- {
- "int": 458
- },
- {
- "array": [
- {
- "real": 417.96875
- }
- ]
- },
- {
- "int": 576
- },
- {
- "array": [
- {
- "real": 226.5625
- },
- {
- "real": 226.5625
- }
- ]
- },
- {
- "int": 601
- },
- {
- "array": [
- {
- "real": 488.28125
- }
- ]
- },
- {
- "int": 624
- },
- {
- "array": [
- {
- "real": 187.5
- }
- ]
- }
- ]
- }
- }
- },
- "51 0": {
- "stream": {
- "dict": {
- "Filter": {
- "name": "FlateDecode"
- },
- "Length": {
- "int": 1677
- },
- "Length1": {
- "int": 6072
- }
- },
- "raw_length": 1677,
- "filters": [
- "FlateDecode"
- ],
- "decoded": {
- "length": 6072,
- "sha256": "ae26f0eb05cde76698aedc3612f333257d4cc1bfbdd15d27b4ad7fbed784892d"
- }
- }
- },
- "52 0": {
- "dict": {
- "Type": {
- "name": "FontDescriptor"
- },
- "Ascent": {
- "int": 1050
- },
- "CapHeight": {
- "int": 695
- },
- "Descent": {
- "int": -250
- },
- "Flags": {
- "int": 33
- },
- "FontBBox": {
- "array": [
- {
- "int": -9
- },
- {
- "int": -250
- },
- {
- "int": 578
- },
- {
- "int": 1050
- }
- ]
- },
- "FontFile2": {
- "ref": "51 0"
- },
- "FontName": {
- "name": "JYZTRN+CamingoCode-Bold"
- },
- "ItalicAngle": {
- "int": 0
- },
- "StemV": {
- "int": 140
- },
- "XHeight": {
- "int": 495
- }
- }
- },
- "53 0": {
- "stream": {
- "dict": {
- "Length": {
- "int": 371
- }
- },
- "raw_length": 371,
- "filters": [],
- "decoded": {
- "length": 371,
- "sha256": "30d050b1587e0324b8a9bea13f37895c9aa7ef6dfb90a3b6a8303554bb5af055",
- "preview_utf8": "/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n/CIDSystemInfo << /…"
- }
- }
- },
- "54 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "BaseFont": {
- "name": "JYZTRN+CamingoCode-Bold"
- },
- "CIDSystemInfo": {
- "dict": {
- "Ordering": {
- "text": "Identity"
- },
- "Registry": {
- "text": "Adobe"
- },
- "Supplement": {
- "int": 0
- }
- }
- },
- "CIDToGIDMap": {
- "name": "Identity"
- },
- "FontDescriptor": {
- "ref": "52 0"
- },
- "Subtype": {
- "name": "CIDFontType2"
- },
- "W": {
- "array": [
- {
- "int": 0
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 42
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 73
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 121
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- }
- ]
- }
- }
- },
- "55 0": {
- "stream": {
- "dict": {
- "Filter": {
- "name": "FlateDecode"
- },
- "Length": {
- "int": 7654
- },
- "Length1": {
- "int": 17468
- }
- },
- "raw_length": 7654,
- "filters": [
- "FlateDecode"
- ],
- "decoded": {
- "length": 17468,
- "sha256": "3c1742e2fad46b9b3a5df1ea5b25db87d6dd07d5d39cb59c07febd7320f9be9f"
- }
- }
- },
- "56 0": {
- "dict": {
- "Type": {
- "name": "FontDescriptor"
- },
- "Ascent": {
- "int": 1050
- },
- "CapHeight": {
- "int": 695
- },
- "Descent": {
- "int": -250
- },
- "Flags": {
- "int": 33
- },
- "FontBBox": {
- "array": [
- {
- "int": -2
- },
- {
- "int": -250
- },
- {
- "int": 578
- },
- {
- "int": 1050
- }
- ]
- },
- "FontFile2": {
- "ref": "55 0"
- },
- "FontName": {
- "name": "YCXYQB+CamingoCode-Regular"
- },
- "ItalicAngle": {
- "int": 0
- },
- "StemV": {
- "int": 80
- },
- "XHeight": {
- "int": 490
- }
- }
- },
- "57 0": {
- "stream": {
- "dict": {
- "Length": {
- "int": 840
- }
- },
- "raw_length": 840,
- "filters": [],
- "decoded": {
- "length": 840,
- "sha256": "7690a81413d655194a93216ada14d5ccef27866806a9d24cdca14a068ee72b99",
- "preview_utf8": "/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n/CIDSystemInfo << /…"
- }
- }
- },
- "58 0": {
- "dict": {
- "Type": {
- "name": "Font"
- },
- "BaseFont": {
- "name": "YCXYQB+CamingoCode-Regular"
- },
- "CIDSystemInfo": {
- "dict": {
- "Ordering": {
- "text": "Identity"
- },
- "Registry": {
- "text": "Adobe"
- },
- "Supplement": {
- "int": 0
- }
- }
- },
- "CIDToGIDMap": {
- "name": "Identity"
- },
- "FontDescriptor": {
- "ref": "56 0"
- },
- "Subtype": {
- "name": "CIDFontType2"
- },
- "W": {
- "array": [
- {
- "int": 0
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 3
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 5
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 21
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 27
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 31
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 41
- },
- {
- "array": [
- {
- "int": 550
- },
- {
- "int": 550
- }
- ]
- },
- {
- "int": 49
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 53
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 69
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 73
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 79
- },
- {
- "array": [
- {
- "int": 550
- },
- {
- "int": 550
- }
- ]
- },
- {
- "int": 88
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 102
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 105
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 109
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 116
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 121
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 134
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 137
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 146
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 162
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 168
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 182
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 190
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 194
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 211
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 217
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 239
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 242
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 246
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 253
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 258
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 315
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 320
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 348
- },
- {
- "array": [
- {
- "int": 550
- }
- ]
- },
- {
- "int": 369
- },
- {
- "array": [
- {
- "int": 550
- },
- {
- "int": 550
- }
- ]
- }
- ]
- }
- }
- },
- "59 0": {
- "dict": {
- "Author": {
- "text": "glu demo"
- },
- "CreationDate": {
- "text": "D:20260511113125+02'00'"
- },
- "Creator": {
- "text": "boxesandglue.dev"
- },
- "Producer": {
- "text": "boxesandglue.dev"
- },
- "Title": {
- "text": "Markdown to PDF/UA with glu"
- }
- }
- }
- }
-}
diff --git a/testdata/fixtures/pdfua-demo/input.pdf b/testdata/fixtures/pdfua-demo/input.pdf
deleted file mode 100644
index 4776784..0000000
Binary files a/testdata/fixtures/pdfua-demo/input.pdf and /dev/null differ
diff --git a/testdata/fixtures/xref-stream/golden.json b/testdata/fixtures/xref-stream/golden.json
deleted file mode 100644
index 85e986b..0000000
--- a/testdata/fixtures/xref-stream/golden.json
+++ /dev/null
@@ -1,62 +0,0 @@
-{
- "version": "2.0",
- "xref_format": "stream",
- "encrypted": false,
- "trailer": {
- "dict": {
- "Type": {
- "name": "XRef"
- },
- "Size": {
- "int": 3
- },
- "W": {
- "array": [
- {
- "int": 1
- },
- {
- "int": 3
- },
- {
- "int": 1
- }
- ]
- },
- "Root": {
- "ref": "1 0"
- },
- "Filter": {
- "name": "FlateDecode"
- },
- "Length": {
- "int": 27
- }
- }
- },
- "objects": {
- "1 0": {
- "dict": {
- "Type": {
- "name": "Catalog"
- },
- "Pages": {
- "ref": "2 0"
- }
- }
- },
- "2 0": {
- "dict": {
- "Type": {
- "name": "Pages"
- },
- "Count": {
- "int": 0
- },
- "Kids": {
- "array": []
- }
- }
- }
- }
-}
diff --git a/testdata/fixtures/xref-stream/input.pdf b/testdata/fixtures/xref-stream/input.pdf
deleted file mode 100644
index a663acf..0000000
Binary files a/testdata/fixtures/xref-stream/input.pdf and /dev/null differ
diff --git a/text.go b/text.go
deleted file mode 100644
index 17f8c8d..0000000
--- a/text.go
+++ /dev/null
@@ -1,195 +0,0 @@
-package pdfdisassembler
-
-import (
- "encoding/binary"
- "strings"
- "time"
- "unicode/utf16"
-)
-
-// decodeTextString decodes b according to the PDF text-string convention
-// (PDF 32000-1:2008 §7.9.2.2): UTF-16BE with BOM, UTF-8 with BOM (PDF 2.0),
-// otherwise PDFDocEncoding.
-func decodeTextString(b []byte) string {
- switch {
- case len(b) >= 2 && b[0] == 0xFE && b[1] == 0xFF:
- return decodeUTF16BE(b[2:])
- case len(b) >= 2 && b[0] == 0xFF && b[1] == 0xFE:
- // UTF-16LE: not spec'd for PDF text strings but observed in the
- // wild from misbehaving producers; decode rather than mojibake.
- return decodeUTF16LE(b[2:])
- case len(b) >= 3 && b[0] == 0xEF && b[1] == 0xBB && b[2] == 0xBF:
- return string(b[3:])
- default:
- return decodePDFDocEncoding(b)
- }
-}
-
-func decodeUTF16BE(b []byte) string {
- if len(b)%2 != 0 {
- b = b[:len(b)-1]
- }
- u16 := make([]uint16, len(b)/2)
- for i := range u16 {
- u16[i] = binary.BigEndian.Uint16(b[2*i:])
- }
- return string(utf16.Decode(u16))
-}
-
-func decodeUTF16LE(b []byte) string {
- if len(b)%2 != 0 {
- b = b[:len(b)-1]
- }
- u16 := make([]uint16, len(b)/2)
- for i := range u16 {
- u16[i] = binary.LittleEndian.Uint16(b[2*i:])
- }
- return string(utf16.Decode(u16))
-}
-
-// decodePDFDocEncoding maps each byte through the PDFDocEncoding table
-// from PDF 32000-1:2008 Annex D.2.
-func decodePDFDocEncoding(b []byte) string {
- var sb strings.Builder
- sb.Grow(len(b))
- for _, c := range b {
- r := pdfDocEncoding[c]
- if r == 0xFFFD {
- // Undefined slot — emit replacement character.
- sb.WriteRune('�')
- } else {
- sb.WriteRune(r)
- }
- }
- return sb.String()
-}
-
-// pdfDocEncoding is the PDFDocEncoding to Unicode mapping (256 entries).
-// Slots without a Unicode mapping are 0xFFFD.
-var pdfDocEncoding = [256]rune{
- // 0x00–0x17 (control characters mostly unused in PDFDocEncoding)
- 0x0000, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD,
- 0x0008, 0x0009, 0x000A, 0xFFFD, 0x000C, 0x000D, 0xFFFD, 0xFFFD,
- 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD,
- // 0x18–0x1F
- 0x02D8, 0x02C7, 0x02C6, 0x02D9, 0x02DD, 0x02DB, 0x02DA, 0x02DC,
- // 0x20–0x7E identical to ASCII
- 0x0020, 0x0021, 0x0022, 0x0023, 0x0024, 0x0025, 0x0026, 0x0027,
- 0x0028, 0x0029, 0x002A, 0x002B, 0x002C, 0x002D, 0x002E, 0x002F,
- 0x0030, 0x0031, 0x0032, 0x0033, 0x0034, 0x0035, 0x0036, 0x0037,
- 0x0038, 0x0039, 0x003A, 0x003B, 0x003C, 0x003D, 0x003E, 0x003F,
- 0x0040, 0x0041, 0x0042, 0x0043, 0x0044, 0x0045, 0x0046, 0x0047,
- 0x0048, 0x0049, 0x004A, 0x004B, 0x004C, 0x004D, 0x004E, 0x004F,
- 0x0050, 0x0051, 0x0052, 0x0053, 0x0054, 0x0055, 0x0056, 0x0057,
- 0x0058, 0x0059, 0x005A, 0x005B, 0x005C, 0x005D, 0x005E, 0x005F,
- 0x0060, 0x0061, 0x0062, 0x0063, 0x0064, 0x0065, 0x0066, 0x0067,
- 0x0068, 0x0069, 0x006A, 0x006B, 0x006C, 0x006D, 0x006E, 0x006F,
- 0x0070, 0x0071, 0x0072, 0x0073, 0x0074, 0x0075, 0x0076, 0x0077,
- 0x0078, 0x0079, 0x007A, 0x007B, 0x007C, 0x007D, 0x007E, 0xFFFD,
- // 0x80–0x9F: punctuation and symbol additions per PDFDocEncoding
- 0x2022, 0x2020, 0x2021, 0x2026, 0x2014, 0x2013, 0x0192, 0x2044,
- 0x2039, 0x203A, 0x2212, 0x2030, 0x201E, 0x201C, 0x201D, 0x2018,
- 0x2019, 0x201A, 0x2122, 0xFB01, 0xFB02, 0x0141, 0x0152, 0x0160,
- 0x0178, 0x017D, 0x0131, 0x0142, 0x0153, 0x0161, 0x017E, 0xFFFD,
- // 0xA0
- 0x20AC,
- // 0xA1–0xFF: same as ISO Latin-1 / Unicode 0x00A1–0x00FF, except a
- // few slots marked undefined by the spec.
- 0x00A1, 0x00A2, 0x00A3, 0x00A4, 0x00A5, 0x00A6, 0x00A7,
- 0x00A8, 0x00A9, 0x00AA, 0x00AB, 0x00AC, 0xFFFD, 0x00AE, 0x00AF,
- 0x00B0, 0x00B1, 0x00B2, 0x00B3, 0x00B4, 0x00B5, 0x00B6, 0x00B7,
- 0x00B8, 0x00B9, 0x00BA, 0x00BB, 0x00BC, 0x00BD, 0x00BE, 0x00BF,
- 0x00C0, 0x00C1, 0x00C2, 0x00C3, 0x00C4, 0x00C5, 0x00C6, 0x00C7,
- 0x00C8, 0x00C9, 0x00CA, 0x00CB, 0x00CC, 0x00CD, 0x00CE, 0x00CF,
- 0x00D0, 0x00D1, 0x00D2, 0x00D3, 0x00D4, 0x00D5, 0x00D6, 0x00D7,
- 0x00D8, 0x00D9, 0x00DA, 0x00DB, 0x00DC, 0x00DD, 0x00DE, 0x00DF,
- 0x00E0, 0x00E1, 0x00E2, 0x00E3, 0x00E4, 0x00E5, 0x00E6, 0x00E7,
- 0x00E8, 0x00E9, 0x00EA, 0x00EB, 0x00EC, 0x00ED, 0x00EE, 0x00EF,
- 0x00F0, 0x00F1, 0x00F2, 0x00F3, 0x00F4, 0x00F5, 0x00F6, 0x00F7,
- 0x00F8, 0x00F9, 0x00FA, 0x00FB, 0x00FC, 0x00FD, 0x00FE, 0x00FF,
-}
-
-// parseDate parses a PDF date string of the form
-// "D:YYYYMMDDHHmmSSOHH'mm'" or any shorter prefix. Returns the zero time
-// if the input is empty or unparseable.
-func parseDate(s string) time.Time {
- s = strings.TrimSpace(s)
- if s == "" {
- return time.Time{}
- }
- s = strings.TrimPrefix(s, "D:")
- // Defaults per spec: month/day = 01, time = 00, offset = UTC.
- year, month, day := 0, 1, 1
- hour, minute, second := 0, 0, 0
- tzSign := byte('Z')
- tzHour, tzMinute := 0, 0
-
- read := func(n int) (int, bool) {
- if len(s) < n {
- return 0, false
- }
- v := 0
- for i := 0; i < n; i++ {
- c := s[i]
- if c < '0' || c > '9' {
- return 0, false
- }
- v = v*10 + int(c-'0')
- }
- s = s[n:]
- return v, true
- }
-
- if v, ok := read(4); ok {
- year = v
- } else {
- return time.Time{}
- }
- if v, ok := read(2); ok {
- month = v
- }
- if v, ok := read(2); ok {
- day = v
- }
- if v, ok := read(2); ok {
- hour = v
- }
- if v, ok := read(2); ok {
- minute = v
- }
- if v, ok := read(2); ok {
- second = v
- }
-
- if len(s) > 0 {
- switch s[0] {
- case '+', '-', 'Z':
- tzSign = s[0]
- s = s[1:]
- }
- }
- if tzSign != 'Z' {
- if v, ok := read(2); ok {
- tzHour = v
- }
- // Optional apostrophe between hour and minute.
- s = strings.TrimPrefix(s, "'")
- if v, ok := read(2); ok {
- tzMinute = v
- }
- }
-
- loc := time.UTC
- if tzSign == '+' || tzSign == '-' {
- off := tzHour*3600 + tzMinute*60
- if tzSign == '-' {
- off = -off
- }
- loc = time.FixedZone("", off)
- }
-
- if month < 1 || month > 12 || day < 1 || day > 31 {
- return time.Time{}
- }
- return time.Date(year, time.Month(month), day, hour, minute, second, 0, loc)
-}
diff --git a/text_test.go b/text_test.go
deleted file mode 100644
index ae50d5a..0000000
--- a/text_test.go
+++ /dev/null
@@ -1,125 +0,0 @@
-package pdfdisassembler
-
-import (
- "testing"
- "time"
- "unicode/utf16"
-)
-
-// Round-trip via utf16.Encode (the independent inverse): the emoji forces a
-// surrogate pair, and both byte orders dispatch off their BOM.
-func TestDecodeTextStringUTF16RoundTrip(t *testing.T) {
- const s = "Hello, 世界 \U0001F600 é"
- u16 := utf16.Encode([]rune(s))
- be := []byte{0xFE, 0xFF}
- le := []byte{0xFF, 0xFE}
- for _, v := range u16 {
- be = append(be, byte(v>>8), byte(v))
- le = append(le, byte(v), byte(v>>8))
- }
- if got := decodeTextString(be); got != s {
- t.Errorf("UTF-16BE: got %q want %q", got, s)
- }
- if got := decodeTextString(le); got != s {
- t.Errorf("UTF-16LE: got %q want %q", got, s)
- }
-}
-
-// A UTF-16 payload with an odd byte count must drop the dangling byte, not
-// read past the end — for both byte orders.
-func TestDecodeUTF16OddLengthNoPanic(t *testing.T) {
- if got := decodeTextString([]byte{0xFE, 0xFF, 0x00, 0x41, 0x00}); got != "A" {
- t.Errorf("UTF-16BE: got %q, want A", got)
- }
- if got := decodeTextString([]byte{0xFF, 0xFE, 0x41, 0x00, 0x00}); got != "A" {
- t.Errorf("UTF-16LE: got %q, want A", got)
- }
-}
-
-func TestDecodeTextStringDispatch(t *testing.T) {
- // UTF-8 BOM (PDF 2.0): bytes after the BOM are returned verbatim.
- if got := decodeTextString([]byte{0xEF, 0xBB, 0xBF, 'h', 'i'}); got != "hi" {
- t.Errorf("UTF-8 BOM: got %q want hi", got)
- }
- // No BOM: PDFDocEncoding, which is ASCII over 0x20-0x7E.
- if got := decodeTextString([]byte("ASCII")); got != "ASCII" {
- t.Errorf("PDFDocEncoding ASCII: got %q", got)
- }
-}
-
-// Spot-check the PDFDocEncoding table (PDF 32000-1:2008 Annex D.2), including
-// the high-range remaps and an undefined slot that must become U+FFFD.
-func TestDecodePDFDocEncoding(t *testing.T) {
- cases := map[byte]rune{
- 0x41: 'A',
- 0x18: '˘', // breve
- 0x80: '•', // bullet
- 0xA0: '€', // euro sign
- 0xE9: 'é', // Latin-1 range, identity-mapped
- 0x7F: '�', // undefined
- 0x9F: '�', // undefined
- }
- for b, want := range cases {
- got := []rune(decodeTextString([]byte{b}))
- if len(got) != 1 || got[0] != want {
- t.Errorf("byte 0x%02X decoded to %q, want %q", b, string(got), string(want))
- }
- }
-}
-
-func TestParseDate(t *testing.T) {
- utc := func(y int, mo time.Month, d, h, mi, s int) time.Time {
- return time.Date(y, mo, d, h, mi, s, 0, time.UTC)
- }
- tests := []struct {
- in string
- want time.Time
- }{
- {"D:20201231235959Z", utc(2020, 12, 31, 23, 59, 59)},
- {"D:20200229", utc(2020, 2, 29, 0, 0, 0)}, // leap day
- {"D:2020", utc(2020, 1, 1, 0, 0, 0)}, // year only, defaults fill in
- {"20200115", utc(2020, 1, 15, 0, 0, 0)}, // optional D: prefix omitted
- {"", time.Time{}},
- {"garbage", time.Time{}},
- {"D:20201340", time.Time{}}, // month 13 rejected
- }
- for _, tt := range tests {
- if got := parseDate(tt.in); !got.Equal(tt.want) {
- t.Errorf("parseDate(%q) = %v, want %v", tt.in, got, tt.want)
- }
- }
- // Signed timezone offset, with the apostrophe separator.
- got := parseDate("D:20200101120000+05'30'")
- want := time.Date(2020, 1, 1, 12, 0, 0, 0, time.FixedZone("", 5*3600+30*60))
- if !got.Equal(want) {
- t.Errorf("tz parse = %v, want %v", got, want)
- }
-}
-
-// The dump heuristic for rendering a string inline vs hex-escaped. The
-// high-byte cases are the subtle ones: 0x80 is a clean PDFDocEncoding glyph
-// (bullet), 0x9F an undefined slot.
-func TestLooksLikeText(t *testing.T) {
- cases := []struct {
- name string
- in String
- want bool
- }{
- {"utf16be_bom", String{0xFE, 0xFF, 0, 'A'}, true},
- {"utf16le_bom", String{0xFF, 0xFE, 'A', 0}, true},
- {"utf8_bom", String{0xEF, 0xBB, 0xBF, 'h', 'i'}, true},
- {"ascii", String("Hello, World!"), true},
- {"ascii_with_whitespace", String("a\tb\r\nc"), true},
- {"control_byte", String{'a', 0x01, 'b'}, false},
- {"del_byte", String{0x7F}, false},
- {"pdfdoc_high_clean", String{0x80}, true},
- {"pdfdoc_high_undefined", String{0x9F}, false},
- }
- for _, c := range cases {
- t.Run(c.name, func(t *testing.T) {
- if got := looksLikeText(c.in); got != c.want {
- t.Errorf("looksLikeText(% x) = %v, want %v", c.in, got, c.want)
- }
- })
- }
-}
diff --git a/xref.go b/xref.go
deleted file mode 100644
index c9ce9da..0000000
--- a/xref.go
+++ /dev/null
@@ -1,609 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "errors"
- "fmt"
- "strconv"
-
- "github.com/speedata/pdfdisassembler/internal/lex"
-)
-
-func trimLeftSpace(s string) string {
- for i := 0; i < len(s); i++ {
- c := s[i]
- if c != ' ' && c != '\t' {
- return s[i:]
- }
- }
- return ""
-}
-
-// parseXref locates the last startxref offset, then parses the cross-
-// reference table (classical, xref-stream, or hybrid) and walks /Prev
-// chains. If the declared xref location is broken, parseXref falls back
-// to xref recovery (scanning for "obj" markers).
-func (r *Reader) parseXref() error {
- off, err := r.findStartXref()
- if err != nil {
- if recErr := r.recoverXref(); recErr != nil {
- return fmt.Errorf("pdfdisassembler: startxref missing and recovery failed: %v (recovery: %w)", err, recErr)
- }
- return nil
- }
-
- visited := map[int64]bool{}
- cur := off
- for {
- if visited[cur] {
- return fmt.Errorf("pdfdisassembler: xref loop at offset %d", cur)
- }
- visited[cur] = true
-
- prev, err := r.readXrefAt(cur)
- if err != nil {
- // Recover if the first attempt was wrong; xref chains
- // otherwise abort here.
- if len(visited) == 1 {
- if recErr := r.recoverXref(); recErr != nil {
- return fmt.Errorf("pdfdisassembler: xref at %d failed: %v (recovery: %w)", cur, err, recErr)
- }
- return nil
- }
- return err
- }
- if prev == 0 {
- break
- }
- cur = prev
- }
-
- if r.trailer == nil {
- return errors.New("pdfdisassembler: no trailer found")
- }
- return nil
-}
-
-// findStartXref scans the last 1024 bytes of the file for the "startxref"
-// marker and returns the offset that follows it.
-func (r *Reader) findStartXref() (int64, error) {
- const tail = 1024
- start := len(r.buf) - tail
- if start < 0 {
- start = 0
- }
- idx := bytes.LastIndex(r.buf[start:], []byte("startxref"))
- if idx < 0 {
- return 0, errors.New("startxref not found")
- }
- idx += start + len("startxref")
- // Skip whitespace, then read decimal.
- for idx < len(r.buf) && (r.buf[idx] == ' ' || r.buf[idx] == '\t' ||
- r.buf[idx] == '\r' || r.buf[idx] == '\n') {
- idx++
- }
- end := idx
- for end < len(r.buf) && r.buf[end] >= '0' && r.buf[end] <= '9' {
- end++
- }
- if end == idx {
- return 0, errors.New("startxref offset missing")
- }
- off, err := strconv.ParseInt(string(r.buf[idx:end]), 10, 64)
- if err != nil {
- return 0, err
- }
- return off, nil
-}
-
-// readXrefAt parses an xref section starting at offset. Returns the /Prev
-// offset (0 if no previous section).
-func (r *Reader) readXrefAt(offset int64) (int64, error) {
- if offset < 0 || offset >= int64(len(r.buf)) {
- return 0, fmt.Errorf("xref offset %d out of range", offset)
- }
- // Classical sections start with the keyword "xref".
- rest := r.buf[offset:]
- // Skip whitespace.
- i := 0
- for i < len(rest) && lex.IsWhitespace(rest[i]) {
- i++
- }
- if i+4 <= len(rest) && string(rest[i:i+4]) == "xref" {
- return r.readClassicalXrefAt(offset + int64(i))
- }
- return r.readXrefStreamAt(offset)
-}
-
-// readClassicalXrefAt parses a "xref" subsection table and the following
-// trailer dictionary. Returns the /Prev offset (0 if none).
-func (r *Reader) readClassicalXrefAt(offset int64) (int64, error) {
- pos := int(offset)
- // Skip the "xref" keyword and following EOL.
- pos += 4
- pos = skipEOL(r.buf, pos)
-
- for {
- // Each subsection: "first count" then count entries of 20 bytes
- // each. The subsection list ends at "trailer".
- lineEnd := indexEOL(r.buf, pos)
- if lineEnd < 0 {
- return 0, errors.New("classical xref: unterminated subsection header")
- }
- line := string(bytes.TrimSpace(r.buf[pos:lineEnd]))
- if line == "trailer" {
- // trailer follows.
- pos = skipEOL(r.buf, pos+len("trailer"))
- break
- }
- // Some producers put trailer on its own line further down.
- if line == "" {
- pos = skipEOL(r.buf, lineEnd)
- continue
- }
- parts := bytes.Fields(r.buf[pos:lineEnd])
- if len(parts) != 2 {
- return 0, fmt.Errorf("classical xref: bad subsection header %q", line)
- }
- first, err1 := strconv.Atoi(string(parts[0]))
- count, err2 := strconv.Atoi(string(parts[1]))
- if err1 != nil || err2 != nil {
- return 0, fmt.Errorf("classical xref: bad subsection header %q", line)
- }
- pos = skipEOL(r.buf, lineEnd)
-
- for i := 0; i < count; i++ {
- // pos walks the raw buffer by 20 per entry over an attacker-set
- // count; bound it without pos+20, which can overflow int on 32-bit.
- if pos < 0 || pos > len(r.buf)-20 {
- return 0, errors.New("classical xref: truncated entry")
- }
- entry := r.buf[pos : pos+20]
- pos += 20
- // Format: nnnnnnnnnn ggggg t EOL
- if len(entry) < 18 {
- return 0, errors.New("classical xref: short entry")
- }
- offStr := string(entry[0:10])
- genStr := string(entry[11:16])
- flag := entry[17]
- off, _ := strconv.ParseInt(trimLeftSpace(offStr), 10, 64)
- gen, _ := strconv.Atoi(trimLeftSpace(genStr))
- ref := Reference{Number: first + i, Generation: gen}
- if flag == 'n' {
- if _, exists := r.xref[ref]; !exists {
- r.xref[ref] = xrefEntry{kind: 1, offset: off, generation: gen}
- }
- }
- // 'f' entries are free; ignore.
- }
- }
-
- // Parse trailer dictionary.
- lx := lex.New(r.buf)
- lx.SetPos(pos)
- p := newParser(lx, r)
- tok, err := p.next()
- if err != nil {
- return 0, err
- }
- if tok.Kind != lex.DictStart {
- return 0, fmt.Errorf("classical xref: trailer dict missing, got %s", tok.Kind)
- }
- trailer, err := p.parseDict()
- if err != nil {
- return 0, fmt.Errorf("classical xref: trailer parse: %w", err)
- }
- if r.trailer == nil {
- r.trailer = trailer
- } else {
- // Older trailers fill in missing keys only.
- for k, v := range trailer.Iter() {
- if !r.trailer.Has(k) {
- r.trailer.set(k, v)
- }
- }
- }
-
- // Hybrid: trailer may reference an XRefStm.
- if v, ok := trailer.Get("XRefStm"); ok {
- if off, ok := v.(Integer); ok {
- if _, err := r.readXrefStreamAt(int64(off)); err != nil {
- // non-fatal: log via error wrap
- return 0, fmt.Errorf("hybrid XRefStm: %w", err)
- }
- }
- }
-
- if v, ok := trailer.Get("Prev"); ok {
- if n, ok := v.(Integer); ok {
- return int64(n), nil
- }
- }
- return 0, nil
-}
-
-// readXrefStreamAt parses an xref stream at the given offset. Returns the
-// /Prev offset (0 if none).
-func (r *Reader) readXrefStreamAt(offset int64) (int64, error) {
- if offset < 0 || offset >= int64(len(r.buf)) {
- return 0, fmt.Errorf("xref stream offset %d out of range", offset)
- }
- lx := lex.New(r.buf)
- lx.SetPos(int(offset))
- p := newParser(lx, r)
-
- // "N G obj"
- t1, err := p.next()
- if err != nil {
- return 0, err
- }
- t2, err := p.next()
- if err != nil {
- return 0, err
- }
- t3, err := p.next()
- if err != nil {
- return 0, err
- }
- if t1.Kind != lex.Integer || t2.Kind != lex.Integer ||
- t3.Kind != lex.Keyword || string(t3.Bytes) != "obj" {
- return 0, fmt.Errorf("xref stream: bad indirect header at %d", offset)
- }
- objNum, _ := strconv.Atoi(string(t1.Bytes))
- objGen, _ := strconv.Atoi(string(t2.Bytes))
-
- body, err := p.parseObject()
- if err != nil {
- return 0, fmt.Errorf("xref stream: dict parse: %w", err)
- }
- d, ok := body.(*Dict)
- if !ok {
- return 0, fmt.Errorf("xref stream: body is %T, want dict", body)
- }
- tok, err := p.peek()
- if err != nil {
- return 0, err
- }
- if tok.Kind != lex.Keyword || string(tok.Bytes) != "stream" {
- return 0, fmt.Errorf("xref stream: missing stream keyword")
- }
- p.next()
- length, err := r.streamLength(d)
- if err != nil {
- return 0, err
- }
- raw, err := lx.ReadStreamData(int(length))
- if err != nil {
- return 0, err
- }
-
- // The xref stream itself is unencrypted per spec (encrypt context not
- // yet initialised at this point either way).
- stream := &Stream{
- Dict: d,
- reader: r,
- rawOffset: int64(lx.Pos() - len(raw)),
- rawLength: int64(len(raw)),
- objNumber: objNum,
- objGeneration: objGen,
- }
- decoded, err := r.applyFilters(stream, raw, false)
- if err != nil {
- return 0, fmt.Errorf("xref stream: decode: %w", err)
- }
-
- // Store the trailer (xref-stream dict doubles as trailer).
- if r.trailer == nil {
- r.trailer = d
- } else {
- for k, v := range d.Iter() {
- if !r.trailer.Has(k) {
- r.trailer.set(k, v)
- }
- }
- }
-
- // Read W field widths.
- wArr, ok := d.Array("W")
- if !ok || len(wArr) < 3 {
- return 0, errors.New("xref stream: /W missing or too short")
- }
- w := make([]int, len(wArr))
- for i, v := range wArr {
- n, ok := v.(Integer)
- if !ok {
- return 0, fmt.Errorf("xref stream: /W[%d] not integer", i)
- }
- if n < 0 {
- return 0, fmt.Errorf("xref stream: negative /W[%d] = %d", i, n)
- }
- w[i] = int(n)
- }
- rowSize := 0
- for _, n := range w {
- rowSize += n
- }
- if rowSize <= 0 {
- return 0, errors.New("xref stream: zero row size")
- }
-
- // /Index is [first count first count …]; defaults to [0 Size].
- var index []int
- if arr, ok := d.Array("Index"); ok {
- for _, v := range arr {
- n, ok := v.(Integer)
- if !ok {
- return 0, errors.New("xref stream: /Index entry not integer")
- }
- index = append(index, int(n))
- }
- } else {
- size, ok := d.Int("Size")
- if !ok {
- return 0, errors.New("xref stream: /Size missing")
- }
- index = []int{0, int(size)}
- }
-
- rowIdx := 0
- for i := 0; i+1 < len(index); i += 2 {
- first := index[i]
- count := index[i+1]
- for j := 0; j < count; j++ {
- start := rowIdx * rowSize
- if start+rowSize > len(decoded) {
- return 0, errors.New("xref stream: data truncated")
- }
- row := decoded[start : start+rowSize]
- rowIdx++
-
- // Default type when W[0]==0 is 1.
- var t uint64 = 1
- off := 0
- if w[0] > 0 {
- t = readBigEndian(row[0:w[0]])
- off = w[0]
- }
- f1 := readBigEndian(row[off : off+w[1]])
- off += w[1]
- f2 := readBigEndian(row[off : off+w[2]])
-
- ref := Reference{Number: first + j, Generation: int(f2)}
- switch t {
- case 0:
- // free entry; ignore
- case 1:
- ref.Generation = int(f2)
- if _, exists := r.xref[ref]; !exists {
- r.xref[ref] = xrefEntry{kind: 1, offset: int64(f1), generation: int(f2)}
- }
- case 2:
- ref.Generation = 0
- if _, exists := r.xref[ref]; !exists {
- r.xref[ref] = xrefEntry{
- kind: 2,
- objStmNum: int(f1),
- objStmIdx: int(f2),
- }
- }
- default:
- // Unknown type per spec: skip.
- }
- }
- }
-
- if v, ok := d.Get("Prev"); ok {
- if n, ok := v.(Integer); ok {
- return int64(n), nil
- }
- }
- return 0, nil
-}
-
-// readCompressedObject extracts an object from an object stream (ObjStm).
-func (r *Reader) readCompressedObject(objStmNum, idx int, expect Reference) (Object, error) {
- streamRef := Reference{Number: objStmNum, Generation: 0}
- v, err := r.Resolve(streamRef)
- if err != nil {
- return nil, fmt.Errorf("ObjStm %d %d R: %w", objStmNum, 0, err)
- }
- stm, ok := v.(*Stream)
- if !ok {
- return nil, fmt.Errorf("ObjStm %d resolved to %T, want stream", objStmNum, v)
- }
- d := stm.Dict
- if t, ok := d.Name("Type"); !ok || t != "ObjStm" {
- return nil, fmt.Errorf("ObjStm %d not of /Type ObjStm", objStmNum)
- }
- n, ok := d.Int("N")
- if !ok {
- return nil, fmt.Errorf("ObjStm %d missing /N", objStmNum)
- }
- first, ok := d.Int("First")
- if !ok {
- return nil, fmt.Errorf("ObjStm %d missing /First", objStmNum)
- }
- content, err := stm.Content()
- if err != nil {
- return nil, err
- }
- // /First and /N are attacker-controlled; bound them against the decoded
- // length — no ObjStm holds more entries (or a longer header) than bytes.
- if first < 0 || first > int64(len(content)) {
- return nil, fmt.Errorf("ObjStm %d: /First %d out of range for %d-byte stream", objStmNum, first, len(content))
- }
- if n < 0 || n > int64(len(content)) {
- return nil, fmt.Errorf("ObjStm %d: implausible /N %d for %d-byte stream", objStmNum, n, len(content))
- }
-
- // Read N pairs of (objNum, offset) from the header section.
- header := content[:first]
- lx := lex.New(header)
- type pair struct {
- num int
- offset int
- }
- // /N is attacker-controlled and only loosely bounded by len(content); a huge
- // value must not preallocate — append grows pairs to the real entry count.
- const maxObjStmPrealloc = 1 << 12
- pairs := make([]pair, 0, min(int(n), maxObjStmPrealloc))
- for i := int64(0); i < n; i++ {
- t1, err := lx.Next()
- if err != nil || t1.Kind != lex.Integer {
- return nil, fmt.Errorf("ObjStm %d header: bad object number", objStmNum)
- }
- t2, err := lx.Next()
- if err != nil || t2.Kind != lex.Integer {
- return nil, fmt.Errorf("ObjStm %d header: bad offset", objStmNum)
- }
- num, _ := strconv.Atoi(string(t1.Bytes))
- off, _ := strconv.Atoi(string(t2.Bytes))
- pairs = append(pairs, pair{num: num, offset: off})
- }
- if idx < 0 || idx >= len(pairs) {
- return nil, fmt.Errorf("ObjStm %d: index %d out of range (%d entries)", objStmNum, idx, len(pairs))
- }
- if pairs[idx].num != expect.Number {
- return nil, fmt.Errorf("ObjStm %d: index %d declares object %d, expected %d",
- objStmNum, idx, pairs[idx].num, expect.Number)
- }
- objStart := int(first) + pairs[idx].offset
- if objStart < 0 || objStart >= len(content) {
- return nil, fmt.Errorf("ObjStm %d: object %d offset %d out of range", objStmNum, expect.Number, objStart)
- }
- bodyLex := lex.New(content)
- bodyLex.SetPos(objStart)
- bp := newParser(bodyLex, r)
- return bp.parseObject()
-}
-
-// recoverXref scans the file for "obj" tokens and rebuilds the xref table.
-// Called when the declared xref location is broken.
-func (r *Reader) recoverXref() error {
- // Find every "N G obj" occurrence in the file.
- for i := 0; i < len(r.buf); i++ {
- if i+3 > len(r.buf) || string(r.buf[i:i+3]) != "obj" {
- continue
- }
- // Must be preceded by whitespace, two integers, whitespace.
- j := i - 1
- for j >= 0 && lex.IsWhitespace(r.buf[j]) {
- j--
- }
- genEnd := j + 1
- for j >= 0 && r.buf[j] >= '0' && r.buf[j] <= '9' {
- j--
- }
- genStart := j + 1
- if genStart == genEnd {
- continue
- }
- for j >= 0 && lex.IsWhitespace(r.buf[j]) {
- j--
- }
- numEnd := j + 1
- for j >= 0 && r.buf[j] >= '0' && r.buf[j] <= '9' {
- j--
- }
- numStart := j + 1
- if numStart == numEnd {
- continue
- }
- num, err1 := strconv.Atoi(string(r.buf[numStart:numEnd]))
- gen, err2 := strconv.Atoi(string(r.buf[genStart:genEnd]))
- if err1 != nil || err2 != nil {
- continue
- }
- // "obj" must be followed by whitespace.
- if i+3 < len(r.buf) && !lex.IsWhitespace(r.buf[i+3]) {
- continue
- }
- ref := Reference{Number: num, Generation: gen}
- if _, exists := r.xref[ref]; !exists {
- r.xref[ref] = xrefEntry{kind: 1, offset: int64(numStart), generation: gen}
- }
- i += 3
- }
-
- // Also try to find the trailer dict.
- if r.trailer == nil {
- idx := bytes.LastIndex(r.buf, []byte("trailer"))
- if idx >= 0 {
- lx := lex.New(r.buf)
- lx.SetPos(idx + len("trailer"))
- p := newParser(lx, r)
- tok, err := p.next()
- if err == nil && tok.Kind == lex.DictStart {
- if d, err := p.parseDict(); err == nil {
- r.trailer = d
- }
- }
- }
- }
- // If still no trailer but we have a /Root somewhere, try harder by
- // finding a dict with /Root in the recovered objects.
- if r.trailer == nil {
- for ref := range r.xref {
- obj, err := r.Resolve(ref)
- if err != nil {
- continue
- }
- d, ok := obj.(*Dict)
- if !ok {
- continue
- }
- if t, ok := d.Name("Type"); ok && t == "Catalog" {
- t := newDict(r)
- t.set("Root", ref)
- r.trailer = t
- break
- }
- }
- }
- if r.trailer == nil {
- return errors.New("xref recovery: no trailer found")
- }
- return nil
-}
-
-// readBigEndian reads a big-endian unsigned integer of the given byte
-// width.
-func readBigEndian(b []byte) uint64 {
- var v uint64
- for _, c := range b {
- v = (v << 8) | uint64(c)
- }
- return v
-}
-
-func skipEOL(buf []byte, pos int) int {
- for pos < len(buf) {
- c := buf[pos]
- if c == '\r' {
- pos++
- if pos < len(buf) && buf[pos] == '\n' {
- pos++
- }
- return pos
- }
- if c == '\n' {
- return pos + 1
- }
- if c == ' ' || c == '\t' {
- pos++
- continue
- }
- return pos
- }
- return pos
-}
-
-func indexEOL(buf []byte, pos int) int {
- for i := pos; i < len(buf); i++ {
- if buf[i] == '\r' || buf[i] == '\n' {
- return i
- }
- }
- return -1
-}
diff --git a/xref_test.go b/xref_test.go
deleted file mode 100644
index 435d536..0000000
--- a/xref_test.go
+++ /dev/null
@@ -1,461 +0,0 @@
-package pdfdisassembler
-
-import (
- "bytes"
- "compress/zlib"
- "fmt"
- "runtime"
- "testing"
-)
-
-// buildXrefStreamPDF synthesises a PDF that uses an xref stream rather
-// than a classical xref table.
-func buildXrefStreamPDF(t *testing.T) []byte {
- t.Helper()
- var buf bytes.Buffer
- off := func() int { return buf.Len() }
- fmt.Fprint(&buf, "%PDF-2.0\n%\xE2\xE3\xCF\xD3\n")
-
- offsets := make([]int, 4)
- offsets[1] = off()
- fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- offsets[2] = off()
- fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
-
- // Build the xref stream data: 3 entries (objects 0,1,2).
- // Each entry: [type(1), offset(3), gen(1)] big-endian.
- rowSize := 5
- rows := []byte{}
- add := func(typ, f1, f2 uint64) {
- rows = append(rows, byte(typ))
- rows = append(rows, byte(f1>>16), byte(f1>>8), byte(f1))
- rows = append(rows, byte(f2))
- }
- add(0, 0, 0xFFFF) // free
- add(1, uint64(offsets[1]), 0)
- add(1, uint64(offsets[2]), 0)
- _ = rowSize
-
- var zbuf bytes.Buffer
- zw := zlib.NewWriter(&zbuf)
- zw.Write(rows)
- zw.Close()
- compressed := zbuf.Bytes()
-
- xrefOff := off()
- fmt.Fprintf(&buf,
- "3 0 obj\n<< /Type /XRef /Size 3 /W [1 3 1] /Root 1 0 R /Filter /FlateDecode /Length %d >>\nstream\n",
- len(compressed))
- buf.Write(compressed)
- fmt.Fprint(&buf, "\nendstream\nendobj\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff)
- return buf.Bytes()
-}
-
-func TestXrefStream(t *testing.T) {
- data := buildXrefStreamPDF(t)
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- cat, err := r.Catalog()
- if err != nil {
- t.Fatalf("Catalog: %v", err)
- }
- if n, ok := cat.Name("Type"); !ok || n != "Catalog" {
- t.Fatalf("/Type %q ok=%v", n, ok)
- }
- if r.Version() != "2.0" {
- t.Fatalf("version %q", r.Version())
- }
-}
-
-func TestXrefRecovery(t *testing.T) {
- // Build a PDF, then point startxref to garbage.
- data := buildMinimalPDF(t)
- // Replace startxref offset to an invalid number.
- idx := bytes.Index(data, []byte("startxref"))
- if idx < 0 {
- t.Fatal("startxref not found in test data")
- }
- // Overwrite the offset following the "startxref\n" with 999999.
- off := idx + len("startxref\n")
- for i := off; i < len(data); i++ {
- if data[i] == '\n' {
- // Overwrite the digits between off and i with 999999.
- width := i - off
- junk := "9999999"[:width]
- copy(data[off:i], junk)
- break
- }
- }
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open after recovery: %v", err)
- }
- defer r.Close()
- cat, err := r.Catalog()
- if err != nil {
- t.Fatalf("Catalog after recovery: %v", err)
- }
- if n, ok := cat.Name("Type"); !ok || n != "Catalog" {
- t.Fatalf("/Type %q ok=%v", n, ok)
- }
-}
-
-// buildXrefStreamPDFWithW builds a PDF whose cross-reference stream (obj 3)
-// declares the given /W array and carries content as its (unfiltered) row data.
-func buildXrefStreamPDFWithW(t *testing.T, wArray, content string) []byte {
- t.Helper()
- var buf bytes.Buffer
- off := map[int]int{}
- w := func(n int, body string) {
- off[n] = buf.Len()
- fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", n, body)
- }
- buf.WriteString("%PDF-1.5\n%\xe2\xe3\xcf\xd3\n")
- w(1, "<< /Type /Catalog /Pages 2 0 R >>")
- w(2, "<< /Type /Pages /Kids [] /Count 0 >>")
- off[3] = buf.Len()
- fmt.Fprintf(&buf, "3 0 obj\n<< /Type /XRef /W %s /Size 1 /Root 1 0 R /Length %d >>\nstream\n",
- wArray, len(content))
- buf.WriteString(content)
- buf.WriteString("\nendstream\nendobj\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", off[3])
- return buf.Bytes()
-}
-
-// buildObjStmPDF builds a PDF whose catalog (obj 1) lives inside an object
-// stream (obj 3), reached via a type-2 entry in the xref stream (obj 2). The
-// ObjStm dict declares declaredN / declaredFirst, which the caller can set to
-// hostile values; the actual stream is always "1 0 " + catalogBody.
-func buildObjStmPDF(t *testing.T, declaredN, declaredFirst int64, catalogBody string) []byte {
- t.Helper()
- objstm := "1 0 " + catalogBody // header "1 0 " (4 bytes), catalog at offset 4
-
- var buf bytes.Buffer
- buf.WriteString("%PDF-1.5\n%\xe2\xe3\xcf\xd3\n")
-
- off3 := buf.Len()
- fmt.Fprintf(&buf, "3 0 obj\n<< /Type /ObjStm /N %d /First %d /Length %d >>\nstream\n%s\nendstream\nendobj\n",
- declaredN, declaredFirst, len(objstm), objstm)
-
- off2 := buf.Len()
- rows := []byte{
- 0x00, 0x00, 0x00, 0x00, // obj 0: free
- 0x02, 0x00, 0x03, 0x00, // obj 1: type 2 -> ObjStm 3, index 0
- 0x01, byte(off2 >> 8), byte(off2), 0x00, // obj 2: type 1 @ off2
- 0x01, byte(off3 >> 8), byte(off3), 0x00, // obj 3: type 1 @ off3
- }
- fmt.Fprintf(&buf, "2 0 obj\n<< /Type /XRef /W [ 1 2 1 ] /Index [ 0 4 ] /Size 4 /Root 1 0 R /Length %d >>\nstream\n",
- len(rows))
- buf.Write(rows)
- buf.WriteString("\nendstream\nendobj\n")
-
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", off2)
- return buf.Bytes()
-}
-
-func TestXrefStreamNegativeWidthRecovers(t *testing.T) {
- // /W [1 -2 10]: the negative width makes the row decode slice row[1:-1].
- // The parser must recover (not panic), leaving the catalog reachable.
- data := buildXrefStreamPDFWithW(t, "[ 1 -2 10 ]", "123456789")
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if _, err := r.Catalog(); err != nil {
- t.Fatalf("Catalog after recovery: %v", err)
- }
-}
-
-// Control: a well-formed ObjStm catalog must resolve, proving the harness and
-// the type-2 path work (so the hostile cases below aren't false positives).
-func TestObjStmCatalogBaseline(t *testing.T) {
- data := buildObjStmPDF(t, 1, 4, "<< /Type /Catalog >>")
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- cat, err := r.Catalog()
- if err != nil {
- t.Fatalf("Catalog: %v", err)
- }
- if n, ok := cat.Name("Type"); !ok || n != "Catalog" {
- t.Fatalf("/Type %q ok=%v", n, ok)
- }
-}
-
-func TestObjStmRejectsHostileHeader(t *testing.T) {
- // Absurd attacker-controlled /First and /N must surface as errors, not
- // slice/make panics.
- tests := []struct {
- name string
- declaredN int64
- declaredFirst int64
- }{
- {"first beyond stream", 1, 1 << 60},
- {"absurd N", 1 << 60, 4},
- }
- for _, tt := range tests {
- t.Run(tt.name, func(t *testing.T) {
- data := buildObjStmPDF(t, tt.declaredN, tt.declaredFirst, "<< /Type /Catalog >>")
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if _, err := r.Catalog(); err == nil {
- t.Fatal("expected an error, got nil")
- }
- })
- }
-}
-
-// buildObjStmHugeN builds a PDF reachable via an xref stream where object 4
-// is a type-2 entry inside ObjStm 3. The ObjStm decodes to contentLen zero
-// bytes (cheap: FlateDecode compresses them to a few KB) but declares
-// /N == contentLen. A reader that trusts /N as a slice capacity balloons the
-// few-KB file into contentLen*sizeof(pair) bytes before parsing a single entry.
-func buildObjStmHugeN(t *testing.T, contentLen int) []byte {
- t.Helper()
- var zbuf bytes.Buffer
- zw := zlib.NewWriter(&zbuf)
- if _, err := zw.Write(make([]byte, contentLen)); err != nil {
- t.Fatal(err)
- }
- zw.Close()
- compressed := zbuf.Bytes()
-
- var buf bytes.Buffer
- buf.WriteString("%PDF-1.5\n%\xe2\xe3\xcf\xd3\n")
-
- off3 := buf.Len()
- fmt.Fprintf(&buf, "3 0 obj\n<< /Type /ObjStm /N %d /First 4 /Filter /FlateDecode /Length %d >>\nstream\n",
- contentLen, len(compressed))
- buf.Write(compressed)
- buf.WriteString("\nendstream\nendobj\n")
-
- off2 := buf.Len()
- rows := []byte{
- 0x01, byte(off3 >> 8), byte(off3), 0x00, // obj 3: type 1 @ off3
- 0x02, 0x00, 0x03, 0x00, // obj 4: type 2 -> ObjStm 3, index 0
- }
- fmt.Fprintf(&buf, "2 0 obj\n<< /Type /XRef /W [ 1 2 1 ] /Index [ 3 2 ] /Size 5 /Root 1 0 R /Length %d >>\nstream\n",
- len(rows))
- buf.Write(rows)
- buf.WriteString("\nendstream\nendobj\n")
-
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", off2)
- return buf.Bytes()
-}
-
-func TestObjStmHugeNDoesNotAmplifyAllocation(t *testing.T) {
- const contentLen = 16 << 20 // == DefaultMaxStreamSize, the largest /N can be
- data := buildObjStmHugeN(t, contentLen)
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
-
- var before, after runtime.MemStats
- runtime.GC()
- runtime.ReadMemStats(&before)
- _, err = r.Resolve(Reference{Number: 4, Generation: 0})
- runtime.ReadMemStats(&after)
- if err == nil {
- t.Fatal("expected an error resolving the hostile ObjStm entry, got nil")
- }
- // Decoding contentLen zero bytes legitimately costs ~2*contentLen; the
- // 256 MiB the /N prealloc would add is well past this ceiling.
- const limit = 128 << 20
- if used := after.TotalAlloc - before.TotalAlloc; used > limit {
- t.Fatalf("resolving a %d-byte ObjStm allocated %d bytes (> %d limit); /N is amplifying allocation",
- contentLen, used, limit)
- }
-}
-
-// Capping the prealloc must not truncate a legitimate ObjStm whose entry count
-// exceeds the cap: object 100+i carries the value 100+i, and resolving one past
-// the cap must still return its exact value (proving append grew the slice).
-func TestObjStmManyObjectsResolvePastPrealloc(t *testing.T) {
- const m = 5000 // > the internal maxObjStmPrealloc (4096)
-
- var body bytes.Buffer
- offsets := make([]int, m)
- for i := 0; i < m; i++ {
- offsets[i] = body.Len()
- fmt.Fprintf(&body, "%d ", 100+i)
- }
- var head bytes.Buffer
- for i := 0; i < m; i++ {
- fmt.Fprintf(&head, "%d %d ", 100+i, offsets[i])
- }
- content := head.String() + body.String()
-
- var buf bytes.Buffer
- buf.WriteString("%PDF-1.5\n%\xe2\xe3\xcf\xd3\n")
- off1 := buf.Len()
- fmt.Fprintf(&buf, "1 0 obj\n<< /Type /ObjStm /N %d /First %d /Length %d >>\nstream\n",
- m, head.Len(), len(content))
- buf.WriteString(content)
- buf.WriteString("\nendstream\nendobj\n")
-
- last := 100 + m - 1
- off2 := buf.Len()
- rows := []byte{
- 0x01, byte(off1 >> 8), byte(off1), 0x00, 0x00, // obj 1: type 1 @ off1
- 0x02, 0x00, 0x01, byte((m - 1) >> 8), byte((m - 1) & 0xff), // last obj: type 2 -> ObjStm 1, idx m-1
- }
- fmt.Fprintf(&buf, "2 0 obj\n<< /Type /XRef /W [ 1 2 2 ] /Index [ 1 1 %d 1 ] /Size %d /Root 1 0 R /Length %d >>\nstream\n",
- last, last+1, len(rows))
- buf.Write(rows)
- buf.WriteString("\nendstream\nendobj\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", off2)
-
- r, err := Open(bytes.NewReader(buf.Bytes()))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- v, err := r.Resolve(Reference{Number: last, Generation: 0})
- if err != nil {
- t.Fatalf("Resolve object %d: %v", last, err)
- }
- n, ok := v.(Integer)
- if !ok || int(n) != last {
- t.Fatalf("object %d resolved to %v (%T), want Integer %d", last, v, v, last)
- }
-}
-
-// With no startxref and no "trailer" keyword, recovery must scan the rebuilt
-// objects for a /Type /Catalog and synthesise a trailer pointing at it.
-func TestRecoverXrefViaCatalogScan(t *testing.T) {
- var buf bytes.Buffer
- buf.WriteString("%PDF-1.7\n%\xe2\xe3\xcf\xd3\n")
- buf.WriteString("1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- buf.WriteString("2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
- buf.WriteString("%%EOF\n")
-
- r, err := Open(bytes.NewReader(buf.Bytes()))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- cat, err := r.Catalog()
- if err != nil {
- t.Fatalf("Catalog: %v", err)
- }
- if n, ok := cat.Name("Type"); !ok || n != "Catalog" {
- t.Fatalf("/Type %q ok=%v, want Catalog", n, ok)
- }
-}
-
-// An incremental update appends a second xref section whose /Prev points back
-// at the first. The newer section must win: object 1 resolves to its updated
-// body, and trailer keys present only in the older section still resolve.
-func TestPrevChainNewestSectionWins(t *testing.T) {
- var buf bytes.Buffer
- w := func(s string) int { off := buf.Len(); buf.WriteString(s); return off }
- w("%PDF-1.7\n%\xe2\xe3\xcf\xd3\n")
- off1v1 := w("1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n")
- off2 := w("2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n")
-
- xref1 := buf.Len()
- fmt.Fprintf(&buf, "xref\n0 3\n%010d %05d f \n%010d %05d n \n%010d %05d n \n",
- 0, 65535, off1v1, 0, off2, 0)
- buf.WriteString("trailer\n<< /Size 3 /Root 1 0 R /Info 2 0 R >>\n")
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xref1)
-
- off1v2 := buf.Len()
- buf.WriteString("1 0 obj\n<< /Type /Catalog /Pages 2 0 R /Lang (en-US) >>\nendobj\n")
- xref2 := buf.Len()
- fmt.Fprintf(&buf, "xref\n1 1\n%010d %05d n \n", off1v2, 0)
- fmt.Fprintf(&buf, "trailer\n<< /Size 3 /Root 1 0 R /Prev %d >>\n", xref1)
- fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xref2)
-
- r, err := Open(bytes.NewReader(buf.Bytes()))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
-
- cat, err := r.Catalog()
- if err != nil {
- t.Fatalf("Catalog: %v", err)
- }
- if lang, ok := cat.String("Lang"); !ok || lang != "en-US" {
- t.Errorf("/Lang = %q ok=%v, want en-US (updated object 1 not used)", lang, ok)
- }
- // /Info lives only in the older trailer; the merge must preserve it.
- if _, ok := r.Trailer().Get("Info"); !ok {
- t.Error("older trailer's /Info lost after /Prev merge")
- }
-}
-
-// A classical xref subsection declaring far more entries than the file holds
-// must be handled gracefully (recover), not over-read the buffer. The bound is
-// also overflow-safe on 32-bit by construction (untestable on a 64-bit run).
-func TestClassicalXrefHugeCountRecovers(t *testing.T) {
- data := buildMinimalPDF(t)
- data = bytes.Replace(data, []byte("xref\n0 5\n"), []byte("xref\n0 999999999\n"), 1)
- r, err := Open(bytes.NewReader(data))
- if err != nil {
- t.Fatalf("Open: %v", err)
- }
- defer r.Close()
- if _, err := r.Catalog(); err != nil {
- t.Fatalf("Catalog: %v", err)
- }
-}
-
-// CR/LF/CRLF line-ending handling for the classical xref reader (§7.5.4).
-func TestSkipEOL(t *testing.T) {
- cases := []struct {
- name string
- buf string
- pos int
- want int
- }{
- {"crlf", "\r\nX", 0, 2},
- {"lf", "\nX", 0, 1},
- {"lone_cr", "\rX", 0, 1},
- {"cr_at_eof", "\r", 0, 1},
- {"leading_spaces_then_lf", " \nX", 0, 3},
- {"tab_then_crlf", "\t\r\n", 0, 3},
- {"non_whitespace_stays_put", "Xyz", 0, 0},
- {"spaces_then_eof", " ", 0, 2},
- {"empty", "", 0, 0},
- {"pos_already_past_end", "ab", 2, 2},
- }
- for _, tc := range cases {
- t.Run(tc.name, func(t *testing.T) {
- if got := skipEOL([]byte(tc.buf), tc.pos); got != tc.want {
- t.Errorf("skipEOL(%q, %d) = %d, want %d", tc.buf, tc.pos, got, tc.want)
- }
- })
- }
-}
-
-func TestIndexEOL(t *testing.T) {
- cases := []struct {
- buf string
- pos int
- want int
- }{
- {"ab\ncd", 0, 2},
- {"ab\rcd", 0, 2},
- {"abcd", 0, -1},
- {"a\nb", 2, -1},
- {"", 0, -1},
- }
- for _, tc := range cases {
- if got := indexEOL([]byte(tc.buf), tc.pos); got != tc.want {
- t.Errorf("indexEOL(%q, %d) = %d, want %d", tc.buf, tc.pos, got, tc.want)
- }
- }
-}