diff --git a/.badges/main/coverage.svg b/.badges/main/coverage.svg new file mode 100644 index 0000000..5b27c36 --- /dev/null +++ b/.badges/main/coverage.svg @@ -0,0 +1 @@ +coveragecoverage87%87% \ No newline at end of file diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml deleted file mode 100644 index b65afa1..0000000 --- a/.github/workflows/test.yml +++ /dev/null @@ -1,39 +0,0 @@ -name: Test - -on: - push: - branches: [main] - pull_request: - workflow_dispatch: - -permissions: - contents: write # push the coverage badge to the `badges` branch (main only) - -concurrency: - group: ${{ github.workflow }}-${{ github.ref }} - cancel-in-progress: ${{ github.event_name == 'pull_request' }} - -jobs: - test: - name: Test - runs-on: ubuntu-latest - steps: - - name: Checkout - uses: actions/checkout@v6 - - - name: Set up Go - uses: actions/setup-go@v6 - with: - go-version-file: go.mod - - - name: Run tests - run: go test -race -coverprofile=cover.out ./... - - # Root (Docker) action, not /action/source: the source variant compiles the - # tool, which needs a newer Go than go.mod pins — and fails the build. - - name: Coverage badge - uses: vladopajic/go-test-coverage@v2 - with: - profile: cover.out - git-token: ${{ github.ref_name == 'main' && secrets.GITHUB_TOKEN || '' }} - git-branch: badges diff --git a/.gitignore b/.gitignore deleted file mode 100644 index e4b2eef..0000000 --- a/.gitignore +++ /dev/null @@ -1,5 +0,0 @@ -*.test -*.out -coverage.txt -.idea/ -.vscode/ diff --git a/LICENSE b/LICENSE deleted file mode 100644 index 1249904..0000000 --- a/LICENSE +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2026 Patrick Gundlach - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in -all copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN -THE SOFTWARE. diff --git a/README.md b/README.md deleted file mode 100644 index f36baa3..0000000 --- a/README.md +++ /dev/null @@ -1,124 +0,0 @@ -# pdfdisassembler - -[![Test](https://github.com/speedata/pdfdisassembler/actions/workflows/test.yml/badge.svg)](https://github.com/speedata/pdfdisassembler/actions/workflows/test.yml) -[![Coverage](https://github.com/speedata/pdfdisassembler/raw/badges/.badges/main/coverage.svg)](https://github.com/speedata/pdfdisassembler/actions/workflows/test.yml) -[![Go Reference](https://pkg.go.dev/badge/github.com/speedata/pdfdisassembler.svg)](https://pkg.go.dev/github.com/speedata/pdfdisassembler) - - -A focused, read-only PDF parser for Go. Built for tooling that **inspects** -PDFs — accessibility checkers, validators, debuggers — without dragging in -the writing, optimisation, signing and image-rendering machinery that -general-purpose PDF libraries carry. - -Full API documentation: - -## Status - -Pre-1.0. The API may break between minor releases. - -## Why - -The Go PDF ecosystem has a real gap for read-only structural inspection. -Existing libraries are either too large (pdfcpu: ~50 kLOC, multi-MB WASM -overhead), licensed restrictively (unipdf: AGPL/commercial), CGo (go-fitz), -or too thin (rsc/pdf, ledongthuc/pdf). pdfdisassembler targets PDF 1.x and -2.0 reading in pure Go, WASM-friendly by construction. The only external -dependency is `github.com/andybalholm/brotli` (pure Go, MIT), which backs -the BrotliDecode filter; only its decoder side gets linked into consumers. - -## Scope - -In scope: PDF 1.x and 2.0 reading, classical xref and xref streams, -indirect-object resolution, stream filters (FlateDecode, ASCII85, ASCIIHex, -LZW, RunLength, and BrotliDecode — a PDF Association extension to PDF 2.0, -pending ISO 32000 inclusion), text-string decoding (PDFDocEncoding, -UTF-16BE BOM, UTF-8 BOM), catalog + page-tree navigation (page boxes, -rotation, resources and -content streams, with inherited attributes resolved along the `/Parent` -chain), DocumentInfo, XMP metadata access, structure tree traversal, -`/Standard` security handler (V2, V4, V5), defensive parsing. - -Out of scope: writing PDFs, image filters (DCTDecode/JBIG2/JPX/CCITTFax), -image rendering, font internals, XFA, public-key encryption, signature -verification, content-stream graphics-state interpretation, LTV. - -## Usage - -Open a file and read top-level metadata: - -```go -import "github.com/speedata/pdfdisassembler" - -r, err := pdfdisassembler.OpenFile("doc.pdf") -if err != nil { - return err -} -defer r.Close() - -fmt.Println("PDF version:", r.Version()) -info := r.DocumentInfo() -fmt.Println("Title:", info.Title) -``` - -Walk every live indirect object and decode any streams that carry one of -the supported filters: - -```go -r, err := pdfdisassembler.OpenFile("doc.pdf") -if err != nil { - return err -} -defer r.Close() - -for entry := range r.Objects() { - s, ok := entry.Object.(*pdfdisassembler.Stream) - if !ok { - continue - } - ref := entry.Reference - data, err := r.DecodeStream(ref) - if err != nil { - fmt.Printf("%d %d R: %v\n", ref.Number, ref.Generation, err) - continue - } - fmt.Printf("%d %d R: %d bytes raw, %d bytes decoded\n", - ref.Number, ref.Generation, s.RawLength(), len(data)) -} -``` - -More complete examples live under [`examples/`](examples): `inspect` -prints a summary of a PDF, `structtree` walks the `/StructTreeRoot` as a -starting point for accessibility tooling, and `pageinfo` reports per-page -boxes, rotation, resources and content size via the page-tree API. - -## Testing - -Snapshot tests live under `testdata/fixtures//`. Each fixture has an -`input.pdf` and a committed `golden.json`. `TestFixtures` opens every -fixture, runs `Dump`, and compares against the golden — a byte-stable -JSON snapshot of the object graph. - -Adding a fixture: - -1. Drop `input.pdf` into `testdata/fixtures//` -2. `go test -update -run TestFixtures/` — generates `golden.json` -3. **Inspect the golden manually**: does it match what the PDF spec says - should happen? The golden is *the expected behaviour*, not "what the - parser currently does" -4. Commit the PDF, the golden, and an optional `README.md` explaining - what the fixture proves - -For synthetic fixtures, see `testdata/fixtures/generate.go`. Run it from -the repo root to (re)create the in-code fixture PDFs. - -The same dump format is exposed as a CLI: - -``` -go install github.com/speedata/pdfdisassembler/cmd/pdfdump@latest -pdfdump doc.pdf > doc.json -diff <(pdfdump a.pdf) <(pdfdump b.pdf) -``` - -## License - -MIT. See [LICENSE](LICENSE). diff --git a/cmd/pdfdump/main.go b/cmd/pdfdump/main.go deleted file mode 100644 index b9e9152..0000000 --- a/cmd/pdfdump/main.go +++ /dev/null @@ -1,71 +0,0 @@ -// Command pdfdump emits a JSON snapshot of a PDF in the same format used -// by pdfdisassembler's snapshot-test harness. -// -// Usage: -// -// pdfdump [-stream-content] [-no-preview] -// -// Diff two PDFs structurally: -// -// diff <(pdfdump a.pdf) <(pdfdump b.pdf) -// -// Reproduce a parser bug: -// -// pdfdump broken.pdf > broken.json -// # attach broken.pdf and broken.json to the issue -package main - -import ( - "errors" - "flag" - "fmt" - "io" - "log" - "os" - - "github.com/speedata/pdfdisassembler" -) - -func main() { - if err := run(os.Args[1:], os.Stdout); err != nil { - if errors.Is(err, flag.ErrHelp) { - return - } - log.Fatal(err) - } -} - -func run(args []string, stdout io.Writer) error { - fs := flag.NewFlagSet("pdfdump", flag.ContinueOnError) - inlineContent := fs.Bool("stream-content", false, "embed decoded stream bytes as hex under decoded.hex") - noPreview := fs.Bool("no-preview", false, "omit the preview_utf8 field on streams") - fs.Usage = func() { - fmt.Fprintln(os.Stderr, "usage: pdfdump [flags] ") - fs.PrintDefaults() - } - if err := fs.Parse(args); err != nil { - return err - } - if fs.NArg() != 1 { - fs.Usage() - return errors.New("pdfdump: exactly one input file is required") - } - r, err := pdfdisassembler.OpenFile(fs.Arg(0)) - if err != nil { - return err - } - defer r.Close() - - opts := pdfdisassembler.DumpOptions{ - InlineStreamContent: *inlineContent, - } - if *noPreview { - opts.PreviewMaxBytes = -1 - } - data, err := pdfdisassembler.Dump(r, opts) - if err != nil { - return err - } - _, err = stdout.Write(data) - return err -} diff --git a/cmd/pdfdump/main_test.go b/cmd/pdfdump/main_test.go deleted file mode 100644 index c172573..0000000 --- a/cmd/pdfdump/main_test.go +++ /dev/null @@ -1,64 +0,0 @@ -package main - -import ( - "bytes" - "io" - "os" - "path/filepath" - "testing" -) - -// pdfdump is run on untrusted files, so hostile input must yield a graceful -// error or valid JSON — never a panic (which the test framework would surface). -func TestRunHostileInputsNoPanic(t *testing.T) { - cases := map[string][]byte{ - "empty": {}, - "garbage": []byte("not a pdf at all"), - "header only": []byte("%PDF-1.7\n"), - "broken xref": []byte("%PDF-1.7\n1 0 obj\n<< /Type /Catalog >>\nendobj\nstartxref\n999999\n%%EOF"), - "nul bytes": bytes.Repeat([]byte{0}, 256), - "unclosed dict": []byte("%PDF-1.7\n1 0 obj\n<< /Type /Catalog\nendobj\ntrailer\n<< /Root 1 0 R >>\nstartxref\n9\n%%EOF"), - } - dir := t.TempDir() - for name, body := range cases { - t.Run(name, func(t *testing.T) { - p := filepath.Join(dir, "in.pdf") - if err := os.WriteFile(p, body, 0o600); err != nil { - t.Fatal(err) - } - var out bytes.Buffer - if err := run([]string{p}, &out); err == nil { - if out.Len() == 0 || out.Bytes()[0] != '{' { - t.Fatalf("succeeded but produced non-JSON output (%d bytes)", out.Len()) - } - } - }) - } -} - -func TestRunValidFixtures(t *testing.T) { - fixtures, err := filepath.Glob("../../testdata/fixtures/*/input.pdf") - if err != nil || len(fixtures) == 0 { - t.Fatalf("no fixtures found: %v", err) - } - for _, fx := range fixtures { - t.Run(filepath.Base(filepath.Dir(fx)), func(t *testing.T) { - var out bytes.Buffer - if err := run([]string{"-stream-content", fx}, &out); err != nil { - t.Fatalf("run: %v", err) - } - if out.Len() == 0 || out.Bytes()[0] != '{' { - t.Fatalf("expected JSON output, got %d bytes", out.Len()) - } - }) - } -} - -func TestRunArgErrors(t *testing.T) { - if err := run(nil, io.Discard); err == nil { - t.Error("expected an error with no file argument") - } - if err := run([]string{"/nonexistent/nope.pdf"}, io.Discard); err == nil { - t.Error("expected an error for a missing file") - } -} diff --git a/contentstream/doc.go b/contentstream/doc.go deleted file mode 100644 index 0ca5f0f..0000000 --- a/contentstream/doc.go +++ /dev/null @@ -1,36 +0,0 @@ -// Package contentstream tokenises PDF content streams into a sequence -// of operations. Content streams are the postfix-notation graphics -// instructions that paint each page (text-showing operators, path -// operators, graphics-state ops, marked-content tags, …). -// -// The scanner is operand-aware: operands are collected up to each -// operator keyword and surfaced together as one Op. Inline images -// (BI/ID/EI) are folded into a single synthetic EI op so the binary -// image bytes between ID and EI do not derail tokenisation. -// -// The scanner does NOT interpret the operations: it does not track -// graphics state, does not render glyphs, does not resolve XObjects. -// Higher-level consumers (e.g. tagged-PDF validators) layer that logic -// on top. -// -// # Usage -// -// for op, err := range contentstream.New(decoded).All() { -// if err != nil { ... } -// switch op.Operator { -// case "Tf": -// // op.Operands[0].Name is the font resource key -// case "BDC": -// // op.Operands[0].Name is the structure tag -// // op.Operands[1] is either a Name (ref into /Properties) -// // or a Dict (inline properties) -// } -// } -// -// # Scope -// -// The scanner accepts the subset of PDF object syntax that can appear -// in content streams: numbers, names, strings (literal and hex), -// arrays, dictionaries, and operator keywords. Indirect references and -// stream objects do not occur in content streams and are not handled. -package contentstream diff --git a/contentstream/operand.go b/contentstream/operand.go deleted file mode 100644 index 88d6a49..0000000 --- a/contentstream/operand.go +++ /dev/null @@ -1,81 +0,0 @@ -package contentstream - -import "strconv" - -// Kind identifies the type of an operand value. -type Kind int - -const ( - // KindUnknown is the zero value; not produced by the scanner. - KindUnknown Kind = iota - // KindNumber covers both PDF integers and reals. Use Operand.Int() - // to recover an int64 when the producer wrote an integer literal. - KindNumber - // KindName is a PDF name without the leading slash. - KindName - // KindString is a literal or hex string. The raw decoded bytes are - // in Operand.Bytes; the scanner does not apply text-string decoding - // (UTF-16BE BOM, PDFDocEncoding, …) because content-stream strings - // are text shown to the reader and their semantic encoding depends - // on the active font, not on the PDF text-string convention. - KindString - // KindArray holds operands of a PDF array, in source order. The - // most common occurrence is the operand of TJ: a mix of strings - // and number adjustments. - KindArray - // KindDict holds the entries of an inline dictionary. The most - // common occurrence is the property dictionary that follows BDC. - KindDict - // KindBool is rare in content streams but appears in BDC property - // dictionaries occasionally. - KindBool - // KindNull is rare in content streams but appears in BDC property - // dictionaries occasionally. - KindNull -) - -// Operand is a single value pushed onto the operand stack before an -// operator keyword. It is a tagged union: which field is meaningful -// depends on Kind. -type Operand struct { - Kind Kind - // Number carries the parsed numeric value when Kind == KindNumber. - // numStr preserves the original literal so Int() can decide whether - // the producer wrote an integer. - Number float64 - numStr string - // Name is the name body (no leading slash) when Kind == KindName. - Name string - // Bytes is the decoded string payload when Kind == KindString. - Bytes []byte - // Array is the element list when Kind == KindArray. - Array []Operand - // Dict is the entry map when Kind == KindDict. Iteration order is - // not preserved; use Dict.Keys / parse separately if order matters. - Dict Dict - // Bool is the boolean value when Kind == KindBool. - Bool bool -} - -// Int reports the operand as an int64 if the producer wrote an integer -// literal (no decimal point, no exponent). The ok flag is false for -// real-number literals and for non-number operands. -func (o Operand) Int() (int64, bool) { - if o.Kind != KindNumber { - return 0, false - } - for _, c := range o.numStr { - if c == '.' || c == 'e' || c == 'E' { - return 0, false - } - } - v, err := strconv.ParseInt(o.numStr, 10, 64) - if err != nil { - return 0, false - } - return v, true -} - -// Dict is a small key→Operand map for inline content-stream -// dictionaries. Nested dictionaries are supported. -type Dict map[string]Operand diff --git a/contentstream/scanner.go b/contentstream/scanner.go deleted file mode 100644 index bb2cc9a..0000000 --- a/contentstream/scanner.go +++ /dev/null @@ -1,366 +0,0 @@ -package contentstream - -import ( - "errors" - "fmt" - "io" - "iter" - "strconv" - - "github.com/speedata/pdfdisassembler/internal/lex" -) - -// Op is one content-stream operation: zero or more operands followed -// by an operator keyword (e.g. "Tf", "Tj", "BDC"). -// -// For inline-image runs, Operator is "EI" and Image carries the raw -// bytes between ID and EI; the BI dictionary is in Operands[0] as a -// KindDict (or empty if BI carried no entries). -type Op struct { - Operator string - Operands []Operand - Image []byte - // Offset is the byte position of the operator keyword in the - // source slice. Useful for error messages and source ranges. - Offset int64 -} - -// Scanner walks a decoded content stream and yields one Op at a time. -// It is not safe for concurrent use. -type Scanner struct { - lx *lex.Lexer - stack []Operand - depth int - done bool -} - -// New returns a Scanner over the decoded content-stream bytes src. -// src is not copied. For pages whose /Contents is an array of streams, -// concatenate the decoded payloads with a single whitespace byte (per -// PDF 32000-1:2008 §7.8.2) before passing them in. -func New(src []byte) *Scanner { - return &Scanner{lx: lex.New(src)} -} - -// ErrUnexpectedEOF indicates that the scanner ran out of bytes mid- -// operation (e.g. inside a dictionary, or while looking for EI). -var ErrUnexpectedEOF = errors.New("pdfdisassembler/contentstream: unexpected EOF") - -// maxNestDepth bounds array/dict nesting so a hostile content stream can't -// recurse the scanner into a stack overflow. -const maxNestDepth = 1000 - -// maxOperands caps operands accumulated per operation (and per array/dict) so -// a flood of operands can't pin large amounts of memory. -const maxOperands = 100000 - -// Next returns the next operation. At end of stream it returns io.EOF. -// Any other error indicates malformed input; the scanner is not safe -// to keep using after an error. -func (s *Scanner) Next() (Op, error) { - if s.done { - return Op{}, io.EOF - } - for { - if len(s.stack) > maxOperands { - return Op{}, fmt.Errorf("pdfdisassembler/contentstream: too many operands (> %d)", maxOperands) - } - tok, err := s.nextToken() - if err != nil { - return Op{}, err - } - switch tok.Kind { - case lex.EOF: - s.done = true - if len(s.stack) != 0 { - // Trailing operands without an operator — common in - // the wild. Drop them silently. - s.stack = s.stack[:0] - } - return Op{}, io.EOF - case lex.Integer, lex.Real: - n, _ := strconv.ParseFloat(string(tok.Bytes), 64) - s.stack = append(s.stack, Operand{ - Kind: KindNumber, - Number: n, - numStr: string(tok.Bytes), - }) - case lex.Name: - s.stack = append(s.stack, Operand{Kind: KindName, Name: string(tok.Bytes)}) - case lex.LitString, lex.HexString: - s.stack = append(s.stack, Operand{Kind: KindString, Bytes: append([]byte(nil), tok.Bytes...)}) - case lex.ArrayStart: - arr, err := s.readArray() - if err != nil { - return Op{}, err - } - s.stack = append(s.stack, Operand{Kind: KindArray, Array: arr}) - case lex.DictStart: - d, err := s.readDict() - if err != nil { - return Op{}, err - } - s.stack = append(s.stack, Operand{Kind: KindDict, Dict: d}) - case lex.Keyword: - kw := string(tok.Bytes) - switch kw { - case "true": - s.stack = append(s.stack, Operand{Kind: KindBool, Bool: true}) - continue - case "false": - s.stack = append(s.stack, Operand{Kind: KindBool, Bool: false}) - continue - case "null": - s.stack = append(s.stack, Operand{Kind: KindNull}) - continue - case "BI": - img, err := s.readInlineImage() - if err != nil { - return Op{}, err - } - op := Op{Operator: "EI", Operands: s.takeStack(), Image: img, Offset: tok.Offset} - return op, nil - } - op := Op{Operator: kw, Operands: s.takeStack(), Offset: tok.Offset} - return op, nil - default: - return Op{}, fmt.Errorf("pdfdisassembler/contentstream: unexpected token %s at %d", tok.Kind, tok.Offset) - } - } -} - -// All returns a range-over-func iterator that yields each Op until EOF -// or the first error. The error is delivered through the second loop -// variable as on the final iteration. -func (s *Scanner) All() iter.Seq2[Op, error] { - return func(yield func(Op, error) bool) { - for { - op, err := s.Next() - if err == io.EOF { - return - } - if !yield(op, err) { - return - } - if err != nil { - return - } - } - } -} - -func (s *Scanner) takeStack() []Operand { - out := s.stack - s.stack = nil - return out -} - -func (s *Scanner) nextToken() (lex.Token, error) { - return s.lx.Next() -} - -func (s *Scanner) readArray() ([]Operand, error) { - s.depth++ - defer func() { s.depth-- }() - if s.depth > maxNestDepth { - return nil, fmt.Errorf("pdfdisassembler/contentstream: nesting too deep (> %d)", maxNestDepth) - } - var out []Operand - for { - if len(out) > maxOperands { - return nil, fmt.Errorf("pdfdisassembler/contentstream: array too large (> %d)", maxOperands) - } - tok, err := s.nextToken() - if err != nil { - return nil, err - } - switch tok.Kind { - case lex.ArrayEnd: - return out, nil - case lex.EOF: - return nil, ErrUnexpectedEOF - case lex.Integer, lex.Real: - n, _ := strconv.ParseFloat(string(tok.Bytes), 64) - out = append(out, Operand{Kind: KindNumber, Number: n, numStr: string(tok.Bytes)}) - case lex.Name: - out = append(out, Operand{Kind: KindName, Name: string(tok.Bytes)}) - case lex.LitString, lex.HexString: - out = append(out, Operand{Kind: KindString, Bytes: append([]byte(nil), tok.Bytes...)}) - case lex.ArrayStart: - nested, err := s.readArray() - if err != nil { - return nil, err - } - out = append(out, Operand{Kind: KindArray, Array: nested}) - case lex.DictStart: - d, err := s.readDict() - if err != nil { - return nil, err - } - out = append(out, Operand{Kind: KindDict, Dict: d}) - case lex.Keyword: - switch string(tok.Bytes) { - case "true": - out = append(out, Operand{Kind: KindBool, Bool: true}) - case "false": - out = append(out, Operand{Kind: KindBool, Bool: false}) - case "null": - out = append(out, Operand{Kind: KindNull}) - default: - return nil, fmt.Errorf("pdfdisassembler/contentstream: unexpected keyword %q inside array at %d", tok.Bytes, tok.Offset) - } - default: - return nil, fmt.Errorf("pdfdisassembler/contentstream: unexpected token %s inside array at %d", tok.Kind, tok.Offset) - } - } -} - -func (s *Scanner) readDict() (Dict, error) { - s.depth++ - defer func() { s.depth-- }() - if s.depth > maxNestDepth { - return nil, fmt.Errorf("pdfdisassembler/contentstream: nesting too deep (> %d)", maxNestDepth) - } - out := Dict{} - for { - if len(out) > maxOperands { - return nil, fmt.Errorf("pdfdisassembler/contentstream: dict too large (> %d)", maxOperands) - } - tok, err := s.nextToken() - if err != nil { - return nil, err - } - if tok.Kind == lex.DictEnd { - return out, nil - } - if tok.Kind == lex.EOF { - return nil, ErrUnexpectedEOF - } - if tok.Kind != lex.Name { - return nil, fmt.Errorf("pdfdisassembler/contentstream: expected name as dict key at %d, got %s", tok.Offset, tok.Kind) - } - key := string(tok.Bytes) - val, err := s.readValue() - if err != nil { - return nil, err - } - out[key] = val - } -} - -func (s *Scanner) readValue() (Operand, error) { - tok, err := s.nextToken() - if err != nil { - return Operand{}, err - } - switch tok.Kind { - case lex.Integer, lex.Real: - n, _ := strconv.ParseFloat(string(tok.Bytes), 64) - return Operand{Kind: KindNumber, Number: n, numStr: string(tok.Bytes)}, nil - case lex.Name: - return Operand{Kind: KindName, Name: string(tok.Bytes)}, nil - case lex.LitString, lex.HexString: - return Operand{Kind: KindString, Bytes: append([]byte(nil), tok.Bytes...)}, nil - case lex.ArrayStart: - arr, err := s.readArray() - if err != nil { - return Operand{}, err - } - return Operand{Kind: KindArray, Array: arr}, nil - case lex.DictStart: - d, err := s.readDict() - if err != nil { - return Operand{}, err - } - return Operand{Kind: KindDict, Dict: d}, nil - case lex.Keyword: - switch string(tok.Bytes) { - case "true": - return Operand{Kind: KindBool, Bool: true}, nil - case "false": - return Operand{Kind: KindBool, Bool: false}, nil - case "null": - return Operand{Kind: KindNull}, nil - } - return Operand{}, fmt.Errorf("pdfdisassembler/contentstream: unexpected keyword %q where value expected at %d", tok.Bytes, tok.Offset) - case lex.EOF: - return Operand{}, ErrUnexpectedEOF - default: - return Operand{}, fmt.Errorf("pdfdisassembler/contentstream: unexpected token %s where value expected at %d", tok.Kind, tok.Offset) - } -} - -// readInlineImage handles BI…ID…EI. The BI token has already been -// consumed. We read a dictionary body until we see the "ID" keyword, -// then scan the raw source for the "EI" terminator and return the -// bytes in between as the image payload. -func (s *Scanner) readInlineImage() ([]byte, error) { - // BI has no '<<' — entries follow directly until ID. - dict := Dict{} - for { - tok, err := s.nextToken() - if err != nil { - return nil, err - } - if tok.Kind == lex.Keyword && string(tok.Bytes) == "ID" { - break - } - if tok.Kind == lex.EOF { - return nil, ErrUnexpectedEOF - } - if tok.Kind != lex.Name { - return nil, fmt.Errorf("pdfdisassembler/contentstream: expected name inside BI block at %d, got %s", tok.Offset, tok.Kind) - } - key := string(tok.Bytes) - val, err := s.readValue() - if err != nil { - return nil, err - } - dict[key] = val - } - // We stashed the dict in stack so it surfaces as Operand of EI. - s.stack = append(s.stack, Operand{Kind: KindDict, Dict: dict}) - - // Per PDF 32000-1:2008 §8.9.7, ID is followed by exactly one - // whitespace byte, then the raw image data, then EI preceded by - // whitespace. We approximate "preceded by whitespace" rather than - // strictly enforcing exactly-one — producers vary. - src := s.lx.Source() - pos := s.lx.Pos() - if pos < len(src) && (src[pos] == ' ' || src[pos] == '\t' || src[pos] == '\n' || src[pos] == '\r') { - pos++ - } - imgStart := pos - // Scan for "EI" preceded by whitespace and followed by whitespace/EOF. - for pos < len(src) { - // Look for 'E' first. - if src[pos] != 'E' { - pos++ - continue - } - if pos+1 >= len(src) || src[pos+1] != 'I' { - pos++ - continue - } - // Check leading boundary. - if pos == 0 { - pos++ - continue - } - if !lex.IsWhitespace(src[pos-1]) { - pos++ - continue - } - // Check trailing boundary. - if pos+2 == len(src) || lex.IsWhitespace(src[pos+2]) || lex.IsDelimiter(src[pos+2]) { - imgEnd := pos - 1 // strip the whitespace separator - if imgEnd < imgStart { - imgEnd = imgStart // empty image: no data between ID and EI - } - s.lx.SetPos(pos + 2) - return append([]byte(nil), src[imgStart:imgEnd]...), nil - } - pos++ - } - return nil, ErrUnexpectedEOF -} diff --git a/contentstream/scanner_test.go b/contentstream/scanner_test.go deleted file mode 100644 index a598588..0000000 --- a/contentstream/scanner_test.go +++ /dev/null @@ -1,466 +0,0 @@ -package contentstream_test - -import ( - "errors" - "io" - "reflect" - "strings" - "testing" - - "github.com/speedata/pdfdisassembler/contentstream" -) - -func collect(t *testing.T, src string) []contentstream.Op { - t.Helper() - var out []contentstream.Op - sc := contentstream.New([]byte(src)) - for { - op, err := sc.Next() - if errors.Is(err, io.EOF) { - return out - } - if err != nil { - t.Fatalf("scan error: %v", err) - } - out = append(out, op) - } -} - -func TestEmpty(t *testing.T) { - if got := collect(t, ""); len(got) != 0 { - t.Fatalf("want 0 ops, got %d", len(got)) - } - if got := collect(t, " \n\t "); len(got) != 0 { - t.Fatalf("want 0 ops on whitespace-only, got %d", len(got)) - } -} - -func TestSimpleOperators(t *testing.T) { - src := "q 1 0 0 1 100 200 cm Q" - ops := collect(t, src) - if len(ops) != 3 { - t.Fatalf("want 3 ops, got %d (%+v)", len(ops), ops) - } - if ops[0].Operator != "q" || len(ops[0].Operands) != 0 { - t.Errorf("op[0] = %+v, want q with no operands", ops[0]) - } - if ops[1].Operator != "cm" || len(ops[1].Operands) != 6 { - t.Errorf("op[1] = %+v, want cm with 6 operands", ops[1]) - } - if ops[2].Operator != "Q" { - t.Errorf("op[2].Operator = %q, want Q", ops[2].Operator) - } -} - -func TestTfOperator(t *testing.T) { - src := "BT /F1 12 Tf (Hello) Tj ET" - ops := collect(t, src) - if len(ops) != 4 { - t.Fatalf("want 4 ops, got %d", len(ops)) - } - if ops[0].Operator != "BT" || ops[3].Operator != "ET" { - t.Errorf("BT/ET framing missing: %+v", ops) - } - if ops[1].Operator != "Tf" { - t.Fatalf("op[1].Operator = %q, want Tf", ops[1].Operator) - } - if ops[1].Operands[0].Kind != contentstream.KindName || ops[1].Operands[0].Name != "F1" { - t.Errorf("Tf font operand = %+v, want name F1", ops[1].Operands[0]) - } - if ops[1].Operands[1].Kind != contentstream.KindNumber || ops[1].Operands[1].Number != 12 { - t.Errorf("Tf size operand = %+v, want number 12", ops[1].Operands[1]) - } - if ops[2].Operator != "Tj" { - t.Fatalf("op[2].Operator = %q, want Tj", ops[2].Operator) - } - if string(ops[2].Operands[0].Bytes) != "Hello" { - t.Errorf("Tj string = %q, want Hello", ops[2].Operands[0].Bytes) - } -} - -func TestTJArray(t *testing.T) { - src := "[(He) -10 (l) -5 (lo)] TJ" - ops := collect(t, src) - if len(ops) != 1 { - t.Fatalf("want 1 op, got %d", len(ops)) - } - if ops[0].Operator != "TJ" { - t.Fatalf("op.Operator = %q, want TJ", ops[0].Operator) - } - arr := ops[0].Operands[0] - if arr.Kind != contentstream.KindArray { - t.Fatalf("TJ operand kind = %v, want Array", arr.Kind) - } - if len(arr.Array) != 5 { - t.Fatalf("TJ array len = %d, want 5", len(arr.Array)) - } - if string(arr.Array[0].Bytes) != "He" || arr.Array[1].Number != -10 { - t.Errorf("TJ contents off: %+v", arr.Array) - } -} - -func TestBDCInlineDict(t *testing.T) { - src := "/Span << /MCID 7 /Lang (en-US) >> BDC (text) Tj EMC" - ops := collect(t, src) - if len(ops) != 3 { - t.Fatalf("want 3 ops, got %d", len(ops)) - } - if ops[0].Operator != "BDC" { - t.Fatalf("op[0].Operator = %q, want BDC", ops[0].Operator) - } - if ops[0].Operands[0].Kind != contentstream.KindName || ops[0].Operands[0].Name != "Span" { - t.Errorf("BDC tag = %+v, want name Span", ops[0].Operands[0]) - } - props := ops[0].Operands[1] - if props.Kind != contentstream.KindDict { - t.Fatalf("BDC props kind = %v, want Dict", props.Kind) - } - mcid, ok := props.Dict["MCID"] - if !ok { - t.Fatalf("MCID missing from %+v", props.Dict) - } - if n, ok := mcid.Int(); !ok || n != 7 { - t.Errorf("MCID = %v (intOk=%v), want 7", n, ok) - } - if ops[2].Operator != "EMC" { - t.Errorf("op[2].Operator = %q, want EMC", ops[2].Operator) - } -} - -func TestBDCPropertyNameRef(t *testing.T) { - src := "/Artifact /P1 BDC (x) Tj EMC" - ops := collect(t, src) - if len(ops) != 3 { - t.Fatalf("want 3 ops, got %d", len(ops)) - } - if ops[0].Operator != "BDC" { - t.Fatalf("op[0].Operator = %q, want BDC", ops[0].Operator) - } - if ops[0].Operands[0].Name != "Artifact" { - t.Errorf("tag = %q, want Artifact", ops[0].Operands[0].Name) - } - if ops[0].Operands[1].Kind != contentstream.KindName || ops[0].Operands[1].Name != "P1" { - t.Errorf("properties ref = %+v, want name P1", ops[0].Operands[1]) - } -} - -func TestHexString(t *testing.T) { - src := "<48656C6C6F> Tj" - ops := collect(t, src) - if len(ops) != 1 { - t.Fatalf("want 1 op, got %d", len(ops)) - } - if string(ops[0].Operands[0].Bytes) != "Hello" { - t.Errorf("hex Tj = %q, want Hello", ops[0].Operands[0].Bytes) - } -} - -func TestInlineImage(t *testing.T) { - // BI /W 2 /H 2 /CS /G /BPC 8 ID - // 4 raw bytes (\x00\x01\x02\x03) then EI - src := "BI /W 2 /H 2 /CS /G /BPC 8 ID \x00\x01\x02\x03\nEI Q" - ops := collect(t, src) - if len(ops) != 2 { - t.Fatalf("want 2 ops, got %d (%+v)", len(ops), ops) - } - if ops[0].Operator != "EI" { - t.Fatalf("op[0].Operator = %q, want EI", ops[0].Operator) - } - if !reflect.DeepEqual(ops[0].Image, []byte{0, 1, 2, 3}) { - t.Errorf("inline image bytes = % x, want 00 01 02 03", ops[0].Image) - } - if ops[1].Operator != "Q" { - t.Errorf("op[1].Operator = %q, want Q", ops[1].Operator) - } -} - -func TestNumberInt(t *testing.T) { - src := "42 3.14 0 Tr" - ops := collect(t, src) - if len(ops) != 1 { - t.Fatalf("want 1 op, got %d", len(ops)) - } - if n, ok := ops[0].Operands[0].Int(); !ok || n != 42 { - t.Errorf("int %v ok=%v, want 42", n, ok) - } - if _, ok := ops[0].Operands[1].Int(); ok { - t.Errorf("real should not yield Int()") - } -} - -func TestAllIteratorStopsOnError(t *testing.T) { - src := "<>", 5000) + " BDC" - sc := contentstream.New([]byte(src)) - if _, err := sc.Next(); err == nil { - t.Fatal("expected a nesting-depth error, got nil") - } -} - -// Control: moderate nesting must still resolve, proving the limit doesn't -// reject legitimate content. -func TestModeratelyNestedArrayResolves(t *testing.T) { - const depth = 100 - src := strings.Repeat("[", depth) + strings.Repeat("]", depth) + " n" - sc := contentstream.New([]byte(src)) - op, err := sc.Next() - if err != nil { - t.Fatalf("unexpected error at depth %d: %v", depth, err) - } - if op.Operator != "n" || len(op.Operands) != 1 || op.Operands[0].Kind != contentstream.KindArray { - t.Fatalf("want n op with one array operand, got %+v", op) - } -} - -// A flood of operands before an operator — directly or inside one array — -// must be rejected rather than accumulated unboundedly. -func TestScannerOperandFloodRejected(t *testing.T) { - for name, src := range map[string]string{ - "bare": strings.Repeat("1 ", 200000) + "n", - "array": "[" + strings.Repeat("1 ", 200000) + "] n", - } { - t.Run(name, func(t *testing.T) { - if _, err := contentstream.New([]byte(src)).Next(); err == nil { - t.Fatal("expected an operand-flood error, got nil") - } - }) - } -} - -// FuzzScanner asserts the content-stream scanner never panics: Next may error, -// but must not crash on arbitrary input. -func FuzzScanner(f *testing.F) { - f.Add([]byte("q 1 0 0 1 0 0 cm /F1 12 Tf (hi) Tj [(a) -5 (b)] TJ BI /W 1 ID xx EI Q")) - f.Add([]byte("/Span << /MCID 7 >> BDC EMC")) - f.Fuzz(func(t *testing.T, data []byte) { - sc := contentstream.New(data) - for i := 0; i <= len(data); i++ { - if _, err := sc.Next(); err != nil { - break - } - } - }) -} - -// One BDC property dict, deliberately built to hit every readValue value kind. -func TestBDCPropertyValueKinds(t *testing.T) { - src := `/P << /B true /F false /N null /Num 1 /Nm /X /S (s) /A [1 2] /D << /Inner 9 >> >> BDC` - ops := collect(t, src) - if len(ops) != 1 || ops[0].Operator != "BDC" { - t.Fatalf("want one BDC op, got %+v", ops) - } - d := ops[0].Operands[1].Dict - wantKind := map[string]contentstream.Kind{ - "B": contentstream.KindBool, - "F": contentstream.KindBool, - "N": contentstream.KindNull, - "Num": contentstream.KindNumber, - "Nm": contentstream.KindName, - "S": contentstream.KindString, - "A": contentstream.KindArray, - "D": contentstream.KindDict, - } - for k, want := range wantKind { - if d[k].Kind != want { - t.Errorf("%s.Kind = %v, want %v", k, d[k].Kind, want) - } - } - if !d["B"].Bool || d["F"].Bool { - t.Errorf("bool values wrong: B=%v F=%v, want true/false", d["B"].Bool, d["F"].Bool) - } - if string(d["S"].Bytes) != "s" { - t.Errorf("string value = %q, want s", d["S"].Bytes) - } - if len(d["A"].Array) != 2 { - t.Errorf("array value len = %d, want 2", len(d["A"].Array)) - } - if d["D"].Dict["Inner"].Kind != contentstream.KindNumber { - t.Errorf("nested dict /Inner kind = %v, want Number", d["D"].Dict["Inner"].Kind) - } -} - -func TestReadDictValueErrors(t *testing.T) { - for _, src := range []string{ - "/P << /K", // EOF mid-value - "/P << /K ] >> BDC", // delimiter where a value is expected - "/P << /K foo >> BDC", // unexpected keyword where a value is expected - } { - t.Run(src, func(t *testing.T) { - if _, err := contentstream.New([]byte(src)).Next(); err == nil { - t.Fatal("expected an error, got nil") - } - }) - } -} - -// Every readArray element kind, including the ones TJ rarely uses (bool, null, -// name, nested array/dict). -func TestArrayMixedElementKinds(t *testing.T) { - ops := collect(t, "[42 3.14 /Nm (s) <41> true false null [1 2] << /K 9 >>] TJ") - if len(ops) != 1 || ops[0].Operator != "TJ" { - t.Fatalf("want one TJ op, got %+v", ops) - } - got := ops[0].Operands[0].Array - wantKinds := []contentstream.Kind{ - contentstream.KindNumber, contentstream.KindNumber, contentstream.KindName, - contentstream.KindString, contentstream.KindString, contentstream.KindBool, - contentstream.KindBool, contentstream.KindNull, contentstream.KindArray, - contentstream.KindDict, - } - if len(got) != len(wantKinds) { - t.Fatalf("array len = %d, want %d", len(got), len(wantKinds)) - } - for i, want := range wantKinds { - if got[i].Kind != want { - t.Errorf("element %d Kind = %v, want %v", i, got[i].Kind, want) - } - } -} - -func TestArrayElementErrors(t *testing.T) { - for _, src := range []string{"[ foo ] n", "[1 2"} { // bad keyword in array; unterminated - t.Run(src, func(t *testing.T) { - if _, err := contentstream.New([]byte(src)).Next(); err == nil { - t.Fatal("expected an error, got nil") - } - }) - } -} - -// Int must report ok=false on int64 overflow (not silently wrap) and for non-numbers. -func TestOperandIntEdgeCases(t *testing.T) { - if _, ok := collect(t, "/Name n")[0].Operands[0].Int(); ok { - t.Error("name operand yielded an int") - } - if _, ok := collect(t, "99999999999999999999999999 n")[0].Operands[0].Int(); ok { - t.Error("overflowing integer literal yielded an int") - } -} - -// readInlineImage must error, not panic, on a malformed BI block. -func TestInlineImageErrors(t *testing.T) { - for _, src := range []string{ - "BI /W 1", // EOF before ID - "BI 5 ID x EI", // non-name key before ID - "BI /W ] ID x EI", // bad entry value - "BI /W 1 ID abcdefg", // no EI terminator before EOF - } { - t.Run(src, func(t *testing.T) { - if _, err := contentstream.New([]byte(src)).Next(); err == nil { - t.Fatal("expected an error, got nil") - } - }) - } -} - -func TestNextStrayDelimiter(t *testing.T) { - for _, src := range []string{"]", ">>"} { - if _, err := contentstream.New([]byte(src)).Next(); err == nil { - t.Errorf("%q: expected an error, got nil", src) - } - } -} - -// Breaking out of the All range mid-iteration must stop cleanly (the -// yield-returned-false path). -func TestAllEarlyBreak(t *testing.T) { - sc := contentstream.New([]byte("q Q q")) - n := 0 - for op, err := range sc.All() { - if err != nil { - t.Fatalf("unexpected error: %v", err) - } - _ = op - n++ - break - } - if n != 1 { - t.Fatalf("iterated %d ops, want 1 before break", n) - } -} - -// Operands left on the stack at EOF with no trailing operator are dropped, not -// emitted as a bogus op. -func TestTrailingOperandsDropped(t *testing.T) { - if ops := collect(t, "1 2 3"); len(ops) != 0 { - t.Fatalf("got %d ops, want 0 (operands without an operator are dropped)", len(ops)) - } -} - -func TestReadDictStructuralErrors(t *testing.T) { - for _, src := range []string{ - "/X << 1 2 >> BDC", // key is not a name - "/X << /K 1", // EOF before '>>' - } { - t.Run(src, func(t *testing.T) { - if _, err := contentstream.New([]byte(src)).Next(); err == nil { - t.Fatal("expected an error, got nil") - } - }) - } -} - -// An "EI" that appears in the image data without a whitespace boundary must be -// skipped; scanning continues to the real, whitespace-delimited terminator. -func TestInlineImageFakeEIInData(t *testing.T) { - ops := collect(t, "BI ID aEIb EI Q") - if len(ops) < 1 || ops[0].Operator != "EI" { - t.Fatalf("want EI op first, got %+v", ops) - } - if string(ops[0].Image) != "aEIb" { - t.Errorf("image = %q, want aEIb", ops[0].Image) - } -} diff --git a/crypt.go b/crypt.go deleted file mode 100644 index 5624c4c..0000000 --- a/crypt.go +++ /dev/null @@ -1,121 +0,0 @@ -package pdfdisassembler - -import ( - "fmt" - - "github.com/speedata/pdfdisassembler/internal/crypt" -) - -// encryptCtx wraps the document-level encryption state. nil if the PDF is -// unencrypted. -type encryptCtx struct { - handler *crypt.Handler -} - -// initEncrypt reads the trailer /Encrypt entry, builds the crypt.Handler -// using the empty user password. Documents secured with a non-empty user -// password fail to open via the public API; callers will need an explicit -// password hook (TODO: expose). -func (r *Reader) initEncrypt() error { - if r.trailer == nil { - return nil - } - encObj, ok := r.trailer.Get("Encrypt") - if !ok { - return nil - } - encDict, err := r.ResolveDict(encObj) - if err != nil { - return fmt.Errorf("pdfdisassembler: /Encrypt: %w", err) - } - - filter, _ := encDict.Name("Filter") - if filter != "Standard" { - return fmt.Errorf("pdfdisassembler: encryption filter %q not supported", filter) - } - - params, err := encryptParamsFromDict(r, encDict) - if err != nil { - return err - } - - h, err := crypt.New(params, nil) - if err != nil { - return fmt.Errorf("pdfdisassembler: encryption: %w", err) - } - r.encrypt = &encryptCtx{handler: h} - return nil -} - -func encryptParamsFromDict(r *Reader, d *Dict) (crypt.Params, error) { - var p crypt.Params - if v, ok := d.Int("V"); ok { - p.V = int(v) - } - if v, ok := d.Int("R"); ok { - p.R = int(v) - } - if v, ok := d.Int("Length"); ok { - p.Length = int(v) - } else { - p.Length = 40 // V1 default - } - if v, ok := d.Int("P"); ok { - p.P = int32(v) - } - if o, ok := d.Bytes("O"); ok { - p.OwnerEntry = o - } - if u, ok := d.Bytes("U"); ok { - p.UserEntry = u - } - if oe, ok := d.Bytes("OE"); ok { - p.OE = oe - } - if ue, ok := d.Bytes("UE"); ok { - p.UE = ue - } - if perms, ok := d.Bytes("Perms"); ok { - p.Perms = perms - } - p.EncryptMeta = true - if v, ok := d.Bool("EncryptMetadata"); ok { - p.EncryptMeta = v - } - if n, ok := d.Name("StmF"); ok { - p.StmF = string(n) - } - if n, ok := d.Name("StrF"); ok { - p.StrF = string(n) - } - if n, ok := d.Name("EFF"); ok { - p.EFF = string(n) - } - // File ID first element comes from trailer. - if id, ok := r.trailer.Array("ID"); ok && len(id) >= 1 { - if s, ok := id[0].(String); ok { - p.ID0 = []byte(s) - } - } - - // /CF dictionary: name → CFM string. - p.CryptFilters = map[string]string{} - if cf, ok := d.Dict("CF"); ok { - for name, v := range cf.Iter() { - cd, ok := v.(*Dict) - if !ok { - continue - } - if cfm, ok := cd.Name("CFM"); ok { - p.CryptFilters[name] = string(cfm) - } - } - } - return p, nil -} - -func (e *encryptCtx) decryptStream(data []byte, objNum, objGen int) ([]byte, error) { - // V4 streams may carry an inline /Crypt filter overriding the cipher; it - // is not yet honored — the default stream cipher is always used. - return e.handler.DecryptStream(data, objNum, objGen, "") -} diff --git a/crypt_test.go b/crypt_test.go deleted file mode 100644 index ba3aea2..0000000 --- a/crypt_test.go +++ /dev/null @@ -1,287 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "crypto/md5" - "crypto/rc4" - "encoding/hex" - "fmt" - "strings" - "testing" -) - -// buildEncryptedPDF constructs a PDF secured with the /Standard handler -// (V2/R3 RC4) whose /Encrypt dict declares the given /Length in bits. /O and -// /U are 32-byte placeholders; the empty-password key derivation runs during -// Open regardless of whether they validate. -func buildEncryptedPDF(t *testing.T, length int) []byte { - t.Helper() - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - - offsets := make([]int, 4) // index 1..3 - - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - - o := strings.Repeat("ab", 32) // 32 bytes, hex-encoded - u := strings.Repeat("cd", 32) - offsets[3] = off() - fmt.Fprintf(&buf, - "3 0 obj\n<< /Filter /Standard /V 2 /R 3 /Length %d /O <%s> /U <%s> /P -44 >>\nendobj\n", - length, o, u) - - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 4\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 3; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - id := "<00112233445566778899aabbccddeeff>" - fmt.Fprintf(&buf, - "trailer\n<< /Size 4 /Root 1 0 R /Encrypt 3 0 R /ID [%s %s] >>\n", id, id) - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -// A malicious /Encrypt dict can declare a /Length whose key size exceeds the -// 16-byte MD5 digest (or is negative). Open must surface an error, not panic. -func TestEncryptHostileKeyLengthNoPanic(t *testing.T) { - for _, length := range []int{256, 4096, -8} { - t.Run(fmt.Sprintf("length_%d", length), func(t *testing.T) { - data := buildEncryptedPDF(t, length) - if _, err := Open(bytes.NewReader(data)); err == nil { - t.Fatal("expected an error for hostile /Length, got nil") - } - }) - } -} - -// stdPassPad is the 32-byte padding string from PDF 32000-1:2008 algorithm 2, -// used to build an empty-password V2/R3 fixture. -var stdPassPad = []byte{ - 0x28, 0xbf, 0x4e, 0x5e, 0x4e, 0x75, 0x8a, 0x41, - 0x64, 0x00, 0x4e, 0x56, 0xff, 0xfa, 0x01, 0x08, - 0x2e, 0x2e, 0x00, 0xb6, 0xd0, 0x68, 0x3e, 0x80, - 0x2f, 0x0c, 0xa9, 0xfe, 0x64, 0x53, 0x69, 0x7a, -} - -// emptyPwRC4Key derives the V2/R3 file key for the empty user password. -func emptyPwRC4Key(owner, id0 []byte, p int32, bits int) []byte { - h := md5.New() - h.Write(stdPassPad) - h.Write(owner) - h.Write([]byte{byte(uint32(p)), byte(uint32(p) >> 8), byte(uint32(p) >> 16), byte(uint32(p) >> 24)}) - h.Write(id0) - sum := h.Sum(nil) - keyLen := bits / 8 - for i := 0; i < 50; i++ { - s := md5.Sum(sum[:keyLen]) - sum = s[:] - } - key := make([]byte, keyLen) - copy(key, sum[:keyLen]) - return key -} - -// emptyPwU computes the /U value (algorithm 5, R>=3) for the empty password, -// so Open's password validation accepts the fixture. -func emptyPwU(key, id0 []byte) []byte { - h := md5.New() - h.Write(stdPassPad) - h.Write(id0) - digest := h.Sum(nil) - out := make([]byte, 16) - c, _ := rc4.NewCipher(key) - c.XORKeyStream(out, digest) - for i := 1; i <= 19; i++ { - tweaked := make([]byte, len(key)) - for j, b := range key { - tweaked[j] = b ^ byte(i) - } - c2, _ := rc4.NewCipher(tweaked) - c2.XORKeyStream(out, out) - } - u := make([]byte, 32) - copy(u, out) - return u -} - -// objKeyRC4 derives the per-object RC4 key (algorithm 1). -func objKeyRC4(fileKey []byte, num, gen int) []byte { - buf := append([]byte{}, fileKey...) - buf = append(buf, byte(num), byte(num>>8), byte(num>>16), byte(gen), byte(gen>>8)) - sum := md5.Sum(buf) - n := len(fileKey) + 5 - if n > 16 { - n = 16 - } - return sum[:n] -} - -func rc4Crypt(key, data []byte) []byte { - out := make([]byte, len(data)) - c, _ := rc4.NewCipher(key) - c.XORKeyStream(out, data) - return out -} - -// assembleEncryptedPDF builds a classical-xref PDF — catalog (1), pages (2), -// the given /Encrypt dict body (3), and an already-encrypted stream (4) — with -// the trailer wired to /Encrypt 3 0 R and /ID [id0 id0]. -func assembleEncryptedPDF(encryptBody string, encStream, id0 []byte) []byte { - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - offsets := make([]int, 5) // 1..4 - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - offsets[3] = off() - fmt.Fprintf(&buf, "3 0 obj\n%s\nendobj\n", encryptBody) - offsets[4] = off() - fmt.Fprintf(&buf, "4 0 obj\n<< /Length %d >>\nstream\n", len(encStream)) - buf.Write(encStream) - fmt.Fprint(&buf, "\nendstream\nendobj\n") - - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 5\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 4; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - id := hex.EncodeToString(id0) - fmt.Fprintf(&buf, - "trailer\n<< /Size 5 /Root 1 0 R /Encrypt 3 0 R /ID [<%s> <%s>] >>\n", id, id) - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -// buildRC4EncryptedStreamPDF builds a V2/R3 RC4-encrypted PDF (empty password) -// whose object 4 is a stream carrying RC4-encrypted plaintext. -func buildRC4EncryptedStreamPDF(t *testing.T, plaintext []byte) []byte { - t.Helper() - owner := bytes.Repeat([]byte{0x5a}, 32) - id0 := bytes.Repeat([]byte{0x7c}, 16) - const bits = 128 - var p int32 = -44 - fileKey := emptyPwRC4Key(owner, id0, p, bits) - u := emptyPwU(fileKey, id0) - enc := rc4Crypt(objKeyRC4(fileKey, 4, 0), plaintext) - body := fmt.Sprintf("<< /Filter /Standard /V 2 /R 3 /Length %d /O <%s> /U <%s> /P %d >>", - bits, hex.EncodeToString(owner), hex.EncodeToString(u), p) - return assembleEncryptedPDF(body, enc, id0) -} - -// buildV4RC4EncryptedStreamPDF builds a V4/R4 PDF whose StdCF crypt filter uses -// CFM /V2 (RC4) — same empty-password key derivation as V2/R3, reached through -// the V4 /CF + /StmF parsing path. -func buildV4RC4EncryptedStreamPDF(t *testing.T, plaintext []byte) []byte { - t.Helper() - owner := bytes.Repeat([]byte{0x5a}, 32) - id0 := bytes.Repeat([]byte{0x7c}, 16) - const bits = 128 - var p int32 = -44 - fileKey := emptyPwRC4Key(owner, id0, p, bits) - u := emptyPwU(fileKey, id0) - enc := rc4Crypt(objKeyRC4(fileKey, 4, 0), plaintext) - body := fmt.Sprintf("<< /Filter /Standard /V 4 /R 4 /Length %d /O <%s> /U <%s> /P %d "+ - "/CF << /StdCF << /CFM /V2 /Length 16 >> >> /StmF /StdCF /StrF /StdCF /EncryptMetadata true >>", - bits, hex.EncodeToString(owner), hex.EncodeToString(u), p) - return assembleEncryptedPDF(body, enc, id0) -} - -// Open must accept an RC4-encrypted PDF secured with the empty user password -// and decrypt its stream content end-to-end. -func TestOpenDecryptsRC4Stream(t *testing.T) { - plaintext := []byte("BT (top secret invoice) Tj ET") - data := buildRC4EncryptedStreamPDF(t, plaintext) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - got, err := r.DecodeStream(Reference{Number: 4, Generation: 0}) - if err != nil { - t.Fatalf("DecodeStream: %v", err) - } - if !bytes.Equal(got, plaintext) { - t.Fatalf("decrypted stream mismatch:\n got %q\nwant %q", got, plaintext) - } -} - -// The V4 path resolves the stream cipher through /CF + /StmF rather than /V -// directly, so it must be exercised end-to-end too. -func TestOpenDecryptsV4RC4Stream(t *testing.T) { - plaintext := []byte("BT (V4 crypt filter) Tj ET") - data := buildV4RC4EncryptedStreamPDF(t, plaintext) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - got, err := r.DecodeStream(Reference{Number: 4, Generation: 0}) - if err != nil { - t.Fatalf("DecodeStream: %v", err) - } - if !bytes.Equal(got, plaintext) { - t.Fatalf("V4 decrypted stream mismatch:\n got %q\nwant %q", got, plaintext) - } -} - -// buildPDFWithEncryptObj wraps an arbitrary /Encrypt dict body as object 3 of a -// classical-xref PDF, with the trailer pointing /Encrypt at it. -func buildPDFWithEncryptObj(t *testing.T, encryptBody string) []byte { - t.Helper() - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - offsets := make([]int, 4) - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - offsets[3] = off() - fmt.Fprintf(&buf, "3 0 obj\n%s\nendobj\n", encryptBody) - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 4\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 3; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - id := "<00112233445566778899aabbccddeeff>" - fmt.Fprintf(&buf, "trailer\n<< /Size 4 /Root 1 0 R /Encrypt 3 0 R /ID [%s %s] >>\n", id, id) - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -// Only the /Standard security handler is supported; any other /Filter must be -// rejected rather than silently treated as unencrypted. -func TestEncryptNonStandardFilterRejected(t *testing.T) { - data := buildPDFWithEncryptObj(t, "<< /Filter /FooSecurity /V 2 /R 3 /Length 128 >>") - if _, err := Open(bytes.NewReader(data)); err == nil { - t.Fatal("expected an error for a non-Standard /Filter, got nil") - } -} - -// Malformed /Encrypt dictionaries (wrong type, missing fields, short V5 entries) -// must surface as errors during Open, never panics. -func TestEncryptMalformedNoPanic(t *testing.T) { - cases := map[string]string{ - "encrypt not a dict": "42", - "missing V and R": "<< /Filter /Standard >>", - "short O and U": "<< /Filter /Standard /V 2 /R 3 /Length 128 /O <00> /U <00> /P 0 >>", - "v5 short entries": "<< /Filter /Standard /V 5 /R 6 /Length 256 /O <00> /U <00> /OE <00> /UE <00> /Perms <00> /P 0 >>", - } - for name, body := range cases { - t.Run(name, func(t *testing.T) { - if _, err := Open(bytes.NewReader(buildPDFWithEncryptObj(t, body))); err == nil { - t.Fatal("expected an error, got nil") - } - }) - } -} diff --git a/doc.go b/doc.go deleted file mode 100644 index d1e3432..0000000 --- a/doc.go +++ /dev/null @@ -1,46 +0,0 @@ -// Package pdfdisassembler is a focused, read-only PDF parser for Go. -// -// It targets tooling that *inspects* PDFs — accessibility checkers, -// validators, debuggers — without dragging in the writing, optimisation, -// signing and image-rendering machinery that general-purpose PDF libraries -// carry. -// -// # Scope -// -// In scope: PDF 1.x and 2.0 reading, classical xref and xref streams, -// indirect-object resolution, stream filters (FlateDecode, ASCII85, -// ASCIIHex, LZW, RunLength, and BrotliDecode — a PDF Association -// extension to PDF 2.0, pending ISO 32000 inclusion), text-string -// decoding (PDFDocEncoding, UTF-16BE BOM, UTF-8 BOM), catalog + -// page-tree navigation (page boxes, -// rotation, resources and content streams, with inherited attributes -// resolved along the /Parent chain), DocumentInfo, XMP metadata access, -// structure-tree traversal, the /Standard security handler (V2, V4, V5), -// defensive xref recovery. -// -// Out of scope: writing PDFs, image filters (DCTDecode/JBIG2/JPX/CCITTFax), -// image rendering, font internals, XFA, public-key encryption, signature -// verification, content-stream graphics-state interpretation, LTV. -// -// # Usage -// -// r, err := pdfdisassembler.OpenFile("doc.pdf") -// if err != nil { -// return err -// } -// defer r.Close() -// -// fmt.Println("PDF version:", r.Version()) -// info := r.DocumentInfo() -// fmt.Println("Title:", info.Title) -// -// for entry := range r.Objects() { -// // inspect every live indirect object -// _ = entry -// } -// -// # API stability -// -// Pre-1.0. The API may break between minor releases but never within a -// patch release. -package pdfdisassembler diff --git a/dump.go b/dump.go deleted file mode 100644 index 4e2b59a..0000000 --- a/dump.go +++ /dev/null @@ -1,297 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "crypto/sha256" - "encoding/hex" - "encoding/json" - "fmt" -) - -// DumpOptions controls Dump's behaviour. -type DumpOptions struct { - // PreviewMaxBytes is the maximum number of decoded stream bytes shown - // as the preview_utf8 field. Default 80. Set to -1 to disable. - PreviewMaxBytes int - // InlineStreamContent embeds the decoded stream as hex under the - // "decoded.hex" field. Off by default — real PDFs produce huge - // fixtures otherwise. - InlineStreamContent bool -} - -// Dump returns a deterministic JSON snapshot of r suitable for golden-file -// snapshot tests and human inspection. -// -// The output is tagged so every PDF value kind is unambiguous (Name vs. -// Text-String vs. Byte-String, Integer vs. Real, etc.). Indirect references -// are preserved as references, so the object graph is acyclic and diffs -// stay reviewable in PRs. Streams contribute metadata (raw_length, filter -// chain, decoded length, SHA-256, optional preview) but never their full -// content — see DumpOptions.InlineStreamContent if you need it. -// -// The output is intended to be byte-stable across runs; dictionary keys -// are emitted in PDF insertion order, objects in (Number, Generation) -// order. -func Dump(r *Reader, opts DumpOptions) ([]byte, error) { - if opts.PreviewMaxBytes == 0 { - opts.PreviewMaxBytes = 80 - } - - top := orderedMap{} - top = append(top, orderedKV{"version", r.Version()}) - top = append(top, orderedKV{"xref_format", r.xrefFormat()}) - top = append(top, orderedKV{"encrypted", r.encrypt != nil}) - - if r.trailer != nil { - top = append(top, orderedKV{"trailer", dumpDictTagged(r.trailer, opts)}) - } - - objs := orderedMap{} - for entry := range r.Objects() { - key := fmt.Sprintf("%d %d", entry.Reference.Number, entry.Reference.Generation) - objs = append(objs, orderedKV{key, dumpValue(entry.Object, opts)}) - } - top = append(top, orderedKV{"objects", objs}) - - raw, err := json.Marshal(top) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler/dump: marshal: %w", err) - } - var pretty bytes.Buffer - if err := json.Indent(&pretty, raw, "", " "); err != nil { - return nil, fmt.Errorf("pdfdisassembler/dump: indent: %w", err) - } - pretty.WriteByte('\n') - // Unescape the HTML-safe sequences that json.Marshal emits by default. - // PDFs are full of '<<' and '>>' (dict delimiters, hex strings), and - // '&' shows up in metadata XML — readable diffs win over byte-pedantic - // HTML-safety, which is irrelevant here. The transform is safe: these - // escape sequences only appear inside JSON string literals, where the - // unescaped form is equivalent. - return unescapeHTMLSafe(pretty.Bytes()), nil -} - -// unescapeHTMLSafe rewrites the six-byte sequences <, >, & -// back into their single-character forms <, >, &. These escapes only -// appear inside JSON string literals (no backslashes outside strings), -// so substitution is byte-safe and JSON remains valid. -func unescapeHTMLSafe(b []byte) []byte { - b = bytes.ReplaceAll(b, []byte{'\\', 'u', '0', '0', '3', 'c'}, []byte{'<'}) - b = bytes.ReplaceAll(b, []byte{'\\', 'u', '0', '0', '3', 'C'}, []byte{'<'}) - b = bytes.ReplaceAll(b, []byte{'\\', 'u', '0', '0', '3', 'e'}, []byte{'>'}) - b = bytes.ReplaceAll(b, []byte{'\\', 'u', '0', '0', '3', 'E'}, []byte{'>'}) - b = bytes.ReplaceAll(b, []byte{'\\', 'u', '0', '0', '2', '6'}, []byte{'&'}) - return b -} - -// xrefFormat reports how the cross-reference table was stored. -func (r *Reader) xrefFormat() string { - if r.trailer == nil { - return "unknown" - } - if t, ok := r.trailer.Name("Type"); ok && t == "XRef" { - return "stream" - } - if _, ok := r.trailer.Get("XRefStm"); ok { - return "hybrid" - } - return "classical" -} - -// orderedKV is one entry of an orderedMap. -type orderedKV struct { - Key string - Val any -} - -// orderedMap is a JSON object that marshals in insertion order. Used for -// both the top-level dump and every PDF dictionary, so that key order in -// goldens matches the producer's writing order. -type orderedMap []orderedKV - -func (m orderedMap) MarshalJSON() ([]byte, error) { - var buf bytes.Buffer - buf.WriteByte('{') - for i, kv := range m { - if i > 0 { - buf.WriteByte(',') - } - kb, err := json.Marshal(kv.Key) - if err != nil { - return nil, err - } - buf.Write(kb) - buf.WriteByte(':') - vb, err := json.Marshal(kv.Val) - if err != nil { - return nil, err - } - buf.Write(vb) - } - buf.WriteByte('}') - return buf.Bytes(), nil -} - -// dumpValue produces the tagged JSON form of a PDF object. -func dumpValue(v Object, opts DumpOptions) orderedMap { - switch t := v.(type) { - case Name: - return orderedMap{{"name", string(t)}} - case Integer: - return orderedMap{{"int", int64(t)}} - case Real: - return orderedMap{{"real", float64(t)}} - case Bool: - return orderedMap{{"bool", bool(t)}} - case Null: - return orderedMap{{"null", nil}} - case String: - if looksLikeText(t) { - return orderedMap{{"text", decodeTextString(t)}} - } - return orderedMap{{"hex", hex.EncodeToString(t)}} - case Reference: - return orderedMap{{"ref", fmt.Sprintf("%d %d", t.Number, t.Generation)}} - case Array: - out := make([]orderedMap, len(t)) - for i, e := range t { - out[i] = dumpValue(e, opts) - } - return orderedMap{{"array", out}} - case *Dict: - return orderedMap{{"dict", dumpDictContents(t, opts)}} - case *Stream: - return orderedMap{{"stream", dumpStream(t, opts)}} - case nil: - return orderedMap{{"null", nil}} - } - return orderedMap{{"unknown", fmt.Sprintf("%T", v)}} -} - -func dumpDictContents(d *Dict, opts DumpOptions) orderedMap { - if d == nil { - return orderedMap{} - } - out := orderedMap{} - for k, v := range d.Iter() { - out = append(out, orderedKV{k, dumpValue(v, opts)}) - } - return out -} - -func dumpDictTagged(d *Dict, opts DumpOptions) orderedMap { - return orderedMap{{"dict", dumpDictContents(d, opts)}} -} - -func dumpStream(s *Stream, opts DumpOptions) orderedMap { - out := orderedMap{} - out = append(out, orderedKV{"dict", dumpDictContents(s.Dict, opts)}) - out = append(out, orderedKV{"raw_length", s.rawLength}) - - names, _, ferr := s.reader.streamFilterChain(s.Dict) - if ferr == nil { - if names == nil { - names = []string{} - } - out = append(out, orderedKV{"filters", names}) - } else { - out = append(out, orderedKV{"filters_error", ferr.Error()}) - } - - dec := orderedMap{} - decoded, derr := s.Content() - if derr != nil { - dec = append(dec, orderedKV{"error", derr.Error()}) - } else { - sum := sha256.Sum256(decoded) - dec = append(dec, orderedKV{"length", int64(len(decoded))}) - dec = append(dec, orderedKV{"sha256", hex.EncodeToString(sum[:])}) - if opts.PreviewMaxBytes > 0 { - if p := preview(decoded, opts.PreviewMaxBytes); p != "" { - dec = append(dec, orderedKV{"preview_utf8", p}) - } - } - if opts.InlineStreamContent { - dec = append(dec, orderedKV{"hex", hex.EncodeToString(decoded)}) - } - } - out = append(out, orderedKV{"decoded", dec}) - return out -} - -// looksLikeText reports whether s is likely a PDF text string. -// -// The rule: -// -// 1. BOM-prefixed strings (UTF-16BE/LE, UTF-8) → text -// 2. Strings with any C0 control byte (except \t, \r, \n) or DEL → hex -// This catches file identifiers, hashes and encryption blobs, where -// random bytes almost always include something in 0x00–0x1F. -// 3. ASCII-only strings → text -// 4. Strings with high bytes (0x80–0xFF) → decode via PDFDocEncoding; -// if the decoded form contains no U+FFFD (undefined-slot marker) -// and no control runes, treat as text. This catches PDFDocEncoded -// content like ActualText with bullets, en-dashes, etc. -func looksLikeText(s String) bool { - if len(s) >= 2 { - if s[0] == 0xFE && s[1] == 0xFF { - return true - } - if s[0] == 0xFF && s[1] == 0xFE { - return true - } - } - if len(s) >= 3 && s[0] == 0xEF && s[1] == 0xBB && s[2] == 0xBF { - return true - } - hasHigh := false - for _, c := range s { - if c == '\t' || c == '\r' || c == '\n' { - continue - } - if c < 0x20 || c == 0x7F { - return false - } - if c >= 0x80 { - hasHigh = true - } - } - if !hasHigh { - return true - } - // Verify the PDFDocEncoded form is clean. - for _, r := range decodePDFDocEncoding(s) { - if r == 0xFFFD { - return false - } - if r < 0x20 && r != '\t' && r != '\r' && r != '\n' { - return false - } - } - return true -} - -// preview returns up to max bytes of b as a string, or "" if any byte is -// non-printable. Truncated previews are suffixed with an ellipsis. -func preview(b []byte, max int) string { - n := len(b) - truncated := false - if n > max { - n = max - truncated = true - } - head := b[:n] - for _, c := range head { - if c == '\t' || c == '\r' || c == '\n' { - continue - } - if c < 0x20 || c > 0x7E { - return "" - } - } - s := string(head) - if truncated { - s += "…" - } - return s -} diff --git a/examples/inspect/main.go b/examples/inspect/main.go deleted file mode 100644 index 605af7e..0000000 --- a/examples/inspect/main.go +++ /dev/null @@ -1,60 +0,0 @@ -// Command inspect prints a summary of a PDF: version, document info, -// catalog top-level keys, page count. -package main - -import ( - "fmt" - "log" - "os" - - "github.com/speedata/pdfdisassembler" -) - -func main() { - if len(os.Args) < 2 { - fmt.Fprintln(os.Stderr, "usage: inspect ") - os.Exit(2) - } - r, err := pdfdisassembler.OpenFile(os.Args[1]) - if err != nil { - log.Fatal(err) - } - defer r.Close() - - fmt.Printf("PDF version: %s\n", r.Version()) - - info := r.DocumentInfo() - if info.Title != "" { - fmt.Printf("Title: %s\n", info.Title) - } - if info.Author != "" { - fmt.Printf("Author: %s\n", info.Author) - } - if info.Producer != "" { - fmt.Printf("Producer: %s\n", info.Producer) - } - if !info.CreationDate.IsZero() { - fmt.Printf("Created: %s\n", info.CreationDate.Format("2006-01-02 15:04:05")) - } - - cat, err := r.Catalog() - if err != nil { - log.Fatalf("catalog: %v", err) - } - fmt.Println("Catalog keys:") - for k := range cat.Iter() { - fmt.Printf(" /%s\n", k) - } - - if pages, ok := cat.Dict("Pages"); ok { - if n, ok := pages.Int("Count"); ok { - fmt.Printf("Pages: %d\n", n) - } - } - - count := 0 - for range r.Objects() { - count++ - } - fmt.Printf("Live indirect objects: %d\n", count) -} diff --git a/examples/pageinfo/main.go b/examples/pageinfo/main.go deleted file mode 100644 index 1db2ff9..0000000 --- a/examples/pageinfo/main.go +++ /dev/null @@ -1,71 +0,0 @@ -// Command pageinfo prints per-page geometry and content info for a PDF: -// page count and, for each page, its boxes, rotation, resource categories, -// and decoded content size. Intended as a starting point for page-importer -// and layout tooling. -package main - -import ( - "fmt" - "io" - "log" - "os" - "sort" - - "github.com/speedata/pdfdisassembler" -) - -func main() { - if len(os.Args) < 2 { - fmt.Fprintln(os.Stderr, "usage: pageinfo ") - os.Exit(2) - } - r, err := pdfdisassembler.OpenFile(os.Args[1]) - if err != nil { - log.Fatal(err) - } - defer r.Close() - - if err := report(os.Stdout, r); err != nil { - log.Fatal(err) - } -} - -// report writes a human-readable page summary for r to w. The page tree is -// walked once via Reader.Pages; inherited attributes (boxes, rotation, -// resources) are resolved by the Page accessors. -func report(w io.Writer, r *pdfdisassembler.Reader) error { - pages, err := r.Pages() - if err != nil { - return err - } - fmt.Fprintf(w, "Pages: %d\n", len(pages)) - - for _, p := range pages { - // Display pages 1-based, matching how readers number them; the API - // itself is 0-based (Page.Index). - fmt.Fprintf(w, "Page %d:\n", p.Index()+1) - - if box, ok := p.Box(pdfdisassembler.MediaBox); ok { - fmt.Fprintf(w, " MediaBox: %g x %g pt\n", box.Width(), box.Height()) - } - if box, ok := p.Box(pdfdisassembler.CropBox); ok { - fmt.Fprintf(w, " CropBox: [%g %g %g %g]\n", box.LLX, box.LLY, box.URX, box.URY) - } - if rot := p.Rotation(); rot != 0 { - fmt.Fprintf(w, " Rotation: %d\n", rot) - } - if res, ok := p.Resources(); ok { - keys := res.Keys() - sort.Strings(keys) - fmt.Fprintf(w, " Resources: %v\n", keys) - } - - content, err := p.Content() - if err != nil { - fmt.Fprintf(w, " Content: (error: %v)\n", err) - continue - } - fmt.Fprintf(w, " Content: %d bytes decoded\n", len(content)) - } - return nil -} diff --git a/examples/pageinfo/main_test.go b/examples/pageinfo/main_test.go deleted file mode 100644 index 59bd043..0000000 --- a/examples/pageinfo/main_test.go +++ /dev/null @@ -1,108 +0,0 @@ -package main - -import ( - "bytes" - "fmt" - "strings" - "testing" - - "github.com/speedata/pdfdisassembler" -) - -// buildObjPDF assembles 1-based object bodies into a PDF with a classical xref -// and the given trailer dictionary body (without the surrounding << >>). -func buildObjPDF(t *testing.T, objs []string, trailer string) []byte { - t.Helper() - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - offsets := make([]int, len(objs)+1) - for i, body := range objs { - offsets[i+1] = off() - fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", i+1, body) - } - xrefOff := off() - fmt.Fprintf(&buf, "xref\n0 %d\n%010d %05d f \n", len(objs)+1, 0, 65535) - for i := 1; i <= len(objs); i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprintf(&buf, "trailer\n<< /Size %d %s >>\nstartxref\n%d\n%%%%EOF\n", - len(objs)+1, trailer, xrefOff) - return buf.Bytes() -} - -// buildPagesPDF builds a two-page document: page 1 inherits its MediaBox and -// Resources from the /Pages root and carries a content stream; page 2 overrides -// MediaBox, adds a CropBox and /Rotate, and has no content. -func buildPagesPDF(t *testing.T) []byte { - const content = "BT (Hi) Tj ET" // 13 bytes - return buildObjPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 2 /Kids [ 3 0 R 4 0 R ] /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> >>", - "<< /Type /Page /Parent 2 0 R /Contents 6 0 R >>", - "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /CropBox [5 5 195 195] /Rotate 90 >>", - "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>", - fmt.Sprintf("<< /Length %d >>\nstream\n%s\nendstream", len(content), content), - }, "/Root 1 0 R") -} - -func TestReport(t *testing.T) { - r, err := pdfdisassembler.Open(bytes.NewReader(buildPagesPDF(t))) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - - var out bytes.Buffer - if err := report(&out, r); err != nil { - t.Fatalf("report: %v", err) - } - got := out.String() - - wants := []string{ - "Pages: 2", - "Page 1:", - "MediaBox: 612 x 792 pt", // inherited from /Pages root - "Resources: [Font]", // inherited - "Content: 13 bytes decoded", - "Page 2:", - "MediaBox: 200 x 200 pt", // overridden locally - "CropBox: [5 5 195 195]", - "Rotation: 90", - "Content: 0 bytes decoded", // no /Contents - } - for _, w := range wants { - if !strings.Contains(got, w) { - t.Errorf("output missing %q; got:\n%s", w, got) - } - } - - // Page 1 has no rotation, so no Rotation line should appear before "Page 2". - page1 := got[strings.Index(got, "Page 1:"):strings.Index(got, "Page 2:")] - if strings.Contains(page1, "Rotation:") { - t.Errorf("page 1 should not print a Rotation line; got:\n%s", page1) - } -} - -// TestReportCyclicKidsTerminates feeds a page tree whose /Kids cycles back on -// itself; report must return rather than recurse until the stack overflows. -func TestReportCyclicKidsTerminates(t *testing.T) { - data := buildObjPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Pages /Kids [ 2 0 R ] >>", // cycle back to obj 2 - }, "/Root 1 0 R") - r, err := pdfdisassembler.Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - - var out bytes.Buffer - if err := report(&out, r); err != nil { - t.Fatalf("report: %v", err) - } - if got := out.String(); !strings.Contains(got, "Pages: 0") { - t.Errorf("want \"Pages: 0\", got:\n%s", got) - } -} diff --git a/examples/structtree/main.go b/examples/structtree/main.go deleted file mode 100644 index fd2a6cc..0000000 --- a/examples/structtree/main.go +++ /dev/null @@ -1,100 +0,0 @@ -// Command structtree dumps the /StructTreeRoot of a PDF in indented form. -// Intended as a starting point for accessibility-checker tooling. -package main - -import ( - "fmt" - "log" - "os" - "strings" - - "github.com/speedata/pdfdisassembler" -) - -func main() { - if len(os.Args) < 2 { - fmt.Fprintln(os.Stderr, "usage: structtree ") - os.Exit(2) - } - r, err := pdfdisassembler.OpenFile(os.Args[1]) - if err != nil { - log.Fatal(err) - } - defer r.Close() - - cat, err := r.Catalog() - if err != nil { - log.Fatal(err) - } - root, ok := cat.Dict("StructTreeRoot") - if !ok { - fmt.Println("(no StructTreeRoot)") - return - } - - roleMap := map[string]string{} - if rm, ok := root.Dict("RoleMap"); ok { - for k, v := range rm.Iter() { - if n, ok := v.(pdfdisassembler.Name); ok { - roleMap[k] = string(n) - } - } - } - - walk(r, root, roleMap, 0, map[pdfdisassembler.Reference]struct{}{}) -} - -// maxStructDepth bounds the recursion so a deeply nested or cyclic /K tree in a -// hostile PDF can't overflow the stack; seen breaks reference cycles earlier. -const maxStructDepth = 1000 - -// visit reports whether ref is newly seen (false if already visited). -func visit(seen map[pdfdisassembler.Reference]struct{}, ref pdfdisassembler.Reference) bool { - if _, ok := seen[ref]; ok { - return false - } - seen[ref] = struct{}{} - return true -} - -func walk(r *pdfdisassembler.Reader, node *pdfdisassembler.Dict, roleMap map[string]string, depth int, seen map[pdfdisassembler.Reference]struct{}) { - if node == nil || depth > maxStructDepth { - return - } - indent := strings.Repeat(" ", depth) - typeName, _ := node.Name("S") - if typeName == "" { - typeName, _ = node.Name("Type") - } - role := string(typeName) - if mapped, ok := roleMap[role]; ok { - role = role + " -> " + mapped - } - fmt.Printf("%s%s\n", indent, role) - - k, ok := node.Get("K") - if !ok { - return - } - if ref, ok := k.(pdfdisassembler.Reference); ok { - if !visit(seen, ref) { - return - } - if v, err := r.Resolve(ref); err == nil { - k = v - } - } - switch t := k.(type) { - case pdfdisassembler.Array: - for _, child := range t { - if ref, ok := child.(pdfdisassembler.Reference); ok && !visit(seen, ref) { - continue - } - if d, err := r.ResolveDict(child); err == nil { - walk(r, d, roleMap, depth+1, seen) - } - } - case *pdfdisassembler.Dict: - walk(r, t, roleMap, depth+1, seen) - } -} diff --git a/examples/structtree/main_test.go b/examples/structtree/main_test.go deleted file mode 100644 index 08a3eeb..0000000 --- a/examples/structtree/main_test.go +++ /dev/null @@ -1,86 +0,0 @@ -package main - -import ( - "bytes" - "fmt" - "io" - "os" - "strings" - "testing" - - "github.com/speedata/pdfdisassembler" -) - -// buildCyclicStructTreePDF builds a PDF whose /StructTreeRoot /K chain cycles -// (obj 5's /K points back to obj 4). -func buildCyclicStructTreePDF(t *testing.T) []byte { - t.Helper() - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - offsets := make([]int, 6) // 1..5 - obj := func(n int, body string) { - offsets[n] = off() - fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", n, body) - } - obj(1, "<< /Type /Catalog /Pages 2 0 R /StructTreeRoot 3 0 R >>") - obj(2, "<< /Type /Pages /Count 0 /Kids [] >>") - obj(3, "<< /Type /StructTreeRoot /K 4 0 R >>") - obj(4, "<< /Type /StructElem /S /Document /K 5 0 R >>") - obj(5, "<< /Type /StructElem /S /P /K 4 0 R >>") // cycle back to obj 4 - - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 6\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 5; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprintf(&buf, "trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -func captureStdout(t *testing.T, fn func()) string { - t.Helper() - old := os.Stdout - rp, wp, err := os.Pipe() - if err != nil { - t.Fatalf("pipe: %v", err) - } - os.Stdout = wp - done := make(chan string, 1) - go func() { - var b bytes.Buffer - io.Copy(&b, rp) - done <- b.String() - }() - fn() - wp.Close() - os.Stdout = old - return <-done -} - -func TestWalkCyclicStructTreeTerminates(t *testing.T) { - r, err := pdfdisassembler.Open(bytes.NewReader(buildCyclicStructTreePDF(t))) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - cat, err := r.Catalog() - if err != nil { - t.Fatalf("Catalog: %v", err) - } - root, ok := cat.Dict("StructTreeRoot") - if !ok { - t.Fatal("no StructTreeRoot") - } - // Must return (the /K cycle would otherwise recurse until stack overflow); - // the dump must show it descended the chain, proving the cycle is exercised. - out := captureStdout(t, func() { - walk(r, root, map[string]string{}, 0, map[pdfdisassembler.Reference]struct{}{}) - }) - for _, want := range []string{"Document", "P"} { - if !strings.Contains(out, want) { - t.Errorf("dump missing %q; got:\n%s", want, out) - } - } -} diff --git a/filter.go b/filter.go deleted file mode 100644 index 14767de..0000000 --- a/filter.go +++ /dev/null @@ -1,161 +0,0 @@ -package pdfdisassembler - -import ( - "fmt" - - "github.com/speedata/pdfdisassembler/internal/filter" -) - -// decodeStream is the Stream.Content backend. It reads the raw bytes from -// the file, applies decryption if enabled, then runs the declared filter -// chain. -func (r *Reader) decodeStream(s *Stream) ([]byte, error) { - raw, err := r.rawStreamSlice(s) - if err != nil { - return nil, err - } - return r.applyFilters(s, raw, true) -} - -// rawStreamSlice returns the stream's raw bytes as a sub-slice of the file -// buffer (no copy), after bounds-checking the declared extent. -func (r *Reader) rawStreamSlice(s *Stream) ([]byte, error) { - if s.rawOffset < 0 || s.rawOffset+s.rawLength > int64(len(r.buf)) { - return nil, fmt.Errorf("pdfdisassembler: stream %d %d R: bytes out of range", s.objNumber, s.objGeneration) - } - return r.buf[s.rawOffset : s.rawOffset+s.rawLength], nil -} - -// rawStreamBytes returns a copy of the stream's raw bytes (see Stream.RawBytes). -func (r *Reader) rawStreamBytes(s *Stream) ([]byte, error) { - raw, err := r.rawStreamSlice(s) - if err != nil { - return nil, err - } - out := make([]byte, len(raw)) - copy(out, raw) - return out, nil -} - -// applyFilters decrypts (if encrypted is true and an encryption context -// exists) and runs the filter chain declared on the stream dict. -func (r *Reader) applyFilters(s *Stream, raw []byte, encrypted bool) ([]byte, error) { - data := raw - if encrypted && r.encrypt != nil { - // Cross-reference streams are themselves unencrypted; callers - // must pass encrypted=false for those. - dec, err := r.encrypt.decryptStream(data, s.objNumber, s.objGeneration) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: decrypt stream %d %d R: %w", s.objNumber, s.objGeneration, err) - } - data = dec - } - - filters, params, err := r.streamFilterChain(s.Dict) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: stream %d %d R filter chain: %w", s.objNumber, s.objGeneration, err) - } - for i, name := range filters { - // Skip image-only filters: return what we have and report. - if filter.IsImageFilter(name) { - return nil, fmt.Errorf("pdfdisassembler: stream %d %d R uses image-only filter %q (not decoded)", s.objNumber, s.objGeneration, name) - } - p := params[i] - p.MaxOutput = r.MaxStreamSize - out, err := filter.Decode(name, data, p) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: stream %d %d R filter %q: %w", s.objNumber, s.objGeneration, name, err) - } - data = out - } - return data, nil -} - -// streamFilterChain returns the ordered filter names and per-filter params -// for a stream dict. Both /Filter and /F (abbreviation) are accepted. -func (r *Reader) streamFilterChain(d *Dict) ([]string, []filter.Params, error) { - v, ok := d.Get("Filter") - if !ok { - v, ok = d.Get("F") - } - if !ok { - return nil, nil, nil - } - v, err := r.Resolve(v) - if err != nil { - return nil, nil, err - } - var names []string - switch t := v.(type) { - case Name: - names = []string{string(t)} - case Array: - for _, e := range t { - e, err := r.Resolve(e) - if err != nil { - return nil, nil, err - } - n, ok := e.(Name) - if !ok { - return nil, nil, fmt.Errorf("/Filter entry is %T, want Name", e) - } - names = append(names, string(n)) - } - default: - return nil, nil, fmt.Errorf("/Filter is %T", v) - } - - pv, _ := d.Get("DecodeParms") - if pv == nil { - pv, _ = d.Get("DP") - } - if pv != nil { - pv, err = r.Resolve(pv) - if err != nil { - return nil, nil, err - } - } - params := make([]filter.Params, len(names)) - switch t := pv.(type) { - case nil, Null: - // nothing - case *Dict: - if len(names) >= 1 { - params[0] = paramsFromDict(t) - } - case Array: - for i, e := range t { - if i >= len(params) { - break - } - e, err := r.Resolve(e) - if err != nil { - return nil, nil, err - } - if d, ok := e.(*Dict); ok { - params[i] = paramsFromDict(d) - } - } - } - return names, params, nil -} - -func paramsFromDict(d *Dict) filter.Params { - var p filter.Params - if n, ok := d.Int("Predictor"); ok { - p.Predictor = int(n) - } - if n, ok := d.Int("Columns"); ok { - p.Columns = int(n) - } - if n, ok := d.Int("Colors"); ok { - p.Colors = int(n) - } - if n, ok := d.Int("BitsPerComponent"); ok { - p.BitsPerComponent = int(n) - } - if n, ok := d.Int("EarlyChange"); ok && n == 0 { - p.NoEarlyChange = true - } - return p -} diff --git a/filter_test.go b/filter_test.go deleted file mode 100644 index 352a264..0000000 --- a/filter_test.go +++ /dev/null @@ -1,179 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "compress/lzw" - "compress/zlib" - "encoding/ascii85" - "fmt" - "testing" -) - -// buildStreamObjectPDF wraps stream as indirect object 3 with the given stream -// dict entries (e.g. "/Filter /LZWDecode ..."), reachable via a classical xref. -func buildStreamObjectPDF(t *testing.T, dictEntries string, stream []byte) []byte { - t.Helper() - var buf bytes.Buffer - off := func() int { return buf.Len() } - buf.WriteString("%PDF-1.7\n%\xe2\xe3\xcf\xd3\n") - offsets := make([]int, 4) - - offsets[1] = off() - buf.WriteString("1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - offsets[2] = off() - buf.WriteString("2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - offsets[3] = off() - fmt.Fprintf(&buf, "3 0 obj\n<< %s /Length %d >>\nstream\n", dictEntries, len(stream)) - buf.Write(stream) - buf.WriteString("\nendstream\nendobj\n") - - xrefOff := off() - buf.WriteString("xref\n0 4\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 3; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - buf.WriteString("trailer\n<< /Size 4 /Root 1 0 R >>\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -func streamObject3(t *testing.T, data []byte) []byte { - t.Helper() - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - t.Cleanup(func() { r.Close() }) - v, err := r.Resolve(Reference{Number: 3, Generation: 0}) - if err != nil { - t.Fatalf("Resolve: %v", err) - } - stm, ok := v.(*Stream) - if !ok { - t.Fatalf("object 3 is %T, want *Stream", v) - } - got, err := stm.Content() - if err != nil { - t.Fatalf("Content: %v", err) - } - return got -} - -// A stream declaring /DecodeParms << /EarlyChange 0 >> must decode with early -// change off. The stdlib LZW writer emits the non-early convention, so honouring -// the parameter reproduces the input; ignoring it garbles a stream past the -// first code-width boundary. -func TestLZWStreamEarlyChangeZero(t *testing.T) { - orig := make([]byte, 4096) - x := uint32(99) - for i := range orig { - x = x*1664525 + 1013904223 - orig[i] = byte(x >> 24) - } - var enc bytes.Buffer - w := lzw.NewWriter(&enc, lzw.MSB, 8) - w.Write(orig) - w.Close() - - got := streamObject3(t, buildStreamObjectPDF(t, - "/Filter /LZWDecode /DecodeParms << /EarlyChange 0 >>", enc.Bytes())) - if !bytes.Equal(got, orig) { - t.Fatal("LZW stream with /EarlyChange 0 decoded incorrectly") - } -} - -// A /Filter array applies filters in order: the raw bytes are ASCII85 wrapping a -// FlateDecode stream, so the chain must un-ASCII85 then inflate. -func TestStreamFilterChainArray(t *testing.T) { - orig := []byte("chained filters: ASCII85 over Flate over the original bytes") - var fl bytes.Buffer - zw := zlib.NewWriter(&fl) - zw.Write(orig) - zw.Close() - a85 := make([]byte, ascii85.MaxEncodedLen(fl.Len())) - n := ascii85.Encode(a85, fl.Bytes()) - stream := append(a85[:n:n], '~', '>') - - got := streamObject3(t, buildStreamObjectPDF(t, - "/Filter [ /ASCII85Decode /FlateDecode ]", stream)) - if !bytes.Equal(got, orig) { - t.Fatalf("chained decode = %q, want %q", got, orig) - } -} - -func TestParamsFromDict(t *testing.T) { - d := newDict(nil) - d.set("Predictor", Integer(12)) - d.set("Columns", Integer(5)) - d.set("Colors", Integer(3)) - d.set("BitsPerComponent", Integer(16)) - d.set("EarlyChange", Integer(0)) - p := paramsFromDict(d) - if p.Predictor != 12 || p.Columns != 5 || p.Colors != 3 || p.BitsPerComponent != 16 { - t.Errorf("predictor params = %+v, want Predictor=12 Columns=5 Colors=3 BitsPerComponent=16", p) - } - if !p.NoEarlyChange { - t.Error("/EarlyChange 0 must set NoEarlyChange") - } - - // 1 is the LZW default, so it must NOT set NoEarlyChange. - d1 := newDict(nil) - d1.set("EarlyChange", Integer(1)) - if paramsFromDict(d1).NoEarlyChange { - t.Error("/EarlyChange 1 must not set NoEarlyChange") - } - - empty := paramsFromDict(newDict(nil)) - if empty.Predictor != 0 || empty.Columns != 0 || empty.Colors != 0 || - empty.BitsPerComponent != 0 || empty.NoEarlyChange { - t.Errorf("empty dict = %+v, want zero Params", empty) - } -} - -// /F and /DP are the /Filter and /DecodeParms abbreviations; a null /DP entry -// leaves that filter with default params. -func TestStreamFilterChainAbbreviations(t *testing.T) { - orig := []byte("abbreviated filter keys decode the same") - var fl bytes.Buffer - zw := zlib.NewWriter(&fl) - zw.Write(orig) - zw.Close() - a85 := make([]byte, ascii85.MaxEncodedLen(fl.Len())) - n := ascii85.Encode(a85, fl.Bytes()) - stream := append(a85[:n:n], '~', '>') - - got := streamObject3(t, buildStreamObjectPDF(t, - "/F [ /ASCII85Decode /FlateDecode ] /DP [ null null ]", stream)) - if !bytes.Equal(got, orig) { - t.Fatalf("abbreviated /F+/DP decode = %q, want %q", got, orig) - } -} - -// A malformed or unsupported filter chain must surface an error from Content(), -// not panic. -func TestStreamFilterChainErrors(t *testing.T) { - cases := map[string]string{ - "filter_wrong_type": "/Filter 42", - "filter_array_bad_entry": "/Filter [ /FlateDecode 42 ]", - "image_only_filter": "/Filter /DCTDecode", - "undecodable_data": "/Filter /FlateDecode", - } - for name, dictEntries := range cases { - t.Run(name, func(t *testing.T) { - data := buildStreamObjectPDF(t, dictEntries, []byte("not valid filtered data")) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - v, err := r.Resolve(Reference{Number: 3, Generation: 0}) - if err != nil { - t.Fatalf("Resolve: %v", err) - } - if _, err := v.(*Stream).Content(); err == nil { - t.Fatal("expected a Content() error, got nil") - } - }) - } -} diff --git a/fixtures_test.go b/fixtures_test.go deleted file mode 100644 index 197f98c..0000000 --- a/fixtures_test.go +++ /dev/null @@ -1,109 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "flag" - "fmt" - "os" - "path/filepath" - "strings" - "testing" -) - -// updateGoldens regenerates fixture golden.json files instead of comparing. -// Run as: go test -update -run TestFixtures. -var updateGoldens = flag.Bool("update", false, "regenerate fixture golden.json files") - -// TestFixtures iterates every directory under testdata/fixtures, opens -// its input.pdf, and compares Dump output against the committed -// golden.json. A test failure prints the first few differing lines and -// the command to regenerate the golden. -// -// Adding a fixture: -// -// 1. Create testdata/fixtures//input.pdf (real-world or generated) -// 2. go test -update -run TestFixtures/ -// 3. Inspect golden.json — does it match what the spec says should happen? -// 4. Commit input.pdf, golden.json, and (optionally) README.md describing -// what the fixture proves. -func TestFixtures(t *testing.T) { - root := "testdata/fixtures" - entries, err := os.ReadDir(root) - if err != nil { - t.Skipf("no testdata/fixtures: %v", err) - return - } - for _, e := range entries { - if !e.IsDir() { - continue - } - name := e.Name() - t.Run(name, func(t *testing.T) { - runFixture(t, filepath.Join(root, name)) - }) - } -} - -func runFixture(t *testing.T, dir string) { - t.Helper() - inputPath := filepath.Join(dir, "input.pdf") - goldenPath := filepath.Join(dir, "golden.json") - - r, err := OpenFile(inputPath) - if err != nil { - t.Fatalf("Open %s: %v", inputPath, err) - } - defer r.Close() - - got, err := Dump(r, DumpOptions{}) - if err != nil { - t.Fatalf("Dump: %v", err) - } - - if *updateGoldens { - if err := os.WriteFile(goldenPath, got, 0o644); err != nil { - t.Fatalf("write golden: %v", err) - } - t.Logf("updated %s (%d bytes)", goldenPath, len(got)) - return - } - - want, err := os.ReadFile(goldenPath) - if err != nil { - t.Fatalf("read golden %s: %v\n\tregenerate with: go test -update -run %s", - goldenPath, err, t.Name()) - } - if bytes.Equal(got, want) { - return - } - t.Errorf("dump mismatch\n regenerate: go test -update -run %s\n first diffs:\n%s", - t.Name(), firstDiffLines(want, got, 10)) -} - -// firstDiffLines returns at most maxLines of unified-style "-want / +got" -// hints around line-level differences. -func firstDiffLines(want, got []byte, maxLines int) string { - wl := strings.Split(string(want), "\n") - gl := strings.Split(string(got), "\n") - max := len(wl) - if len(gl) > max { - max = len(gl) - } - var out strings.Builder - shown := 0 - for i := 0; i < max && shown < maxLines; i++ { - var w, g string - if i < len(wl) { - w = wl[i] - } - if i < len(gl) { - g = gl[i] - } - if w == g { - continue - } - fmt.Fprintf(&out, " line %d:\n - %s\n + %s\n", i+1, w, g) - shown++ - } - return out.String() -} diff --git a/go.mod b/go.mod deleted file mode 100644 index 5271bc6..0000000 --- a/go.mod +++ /dev/null @@ -1,5 +0,0 @@ -module github.com/speedata/pdfdisassembler - -go 1.23 - -require github.com/andybalholm/brotli v1.2.2 diff --git a/go.sum b/go.sum deleted file mode 100644 index 80d4b3a..0000000 --- a/go.sum +++ /dev/null @@ -1,4 +0,0 @@ -github.com/andybalholm/brotli v1.2.2 h1:HzTuoo2ErYQqf5qvcJInB8uvqSVxRttzkFexPWtnceM= -github.com/andybalholm/brotli v1.2.2/go.mod h1:rzTDkvFWvIrjDXZHkuS16NPggd91W3kUSvPlQ1pLaKY= -github.com/xyproto/randomstring v1.0.5 h1:YtlWPoRdgMu3NZtP45drfy1GKoojuR7hmRcnhZqKjWU= -github.com/xyproto/randomstring v1.0.5/go.mod h1:rgmS5DeNXLivK7YprL0pY+lTuhNQW3iGxZ18UQApw/E= diff --git a/internal/crypt/crypt.go b/internal/crypt/crypt.go deleted file mode 100644 index 595815a..0000000 --- a/internal/crypt/crypt.go +++ /dev/null @@ -1,470 +0,0 @@ -// Package crypt implements the PDF /Standard security handler for -// versions V2 (RC4), V4 (RC4 or AES-128) and V5 (AES-256, PDF 1.7 -// Extension 3 and PDF 2.0). -// -// Only password-based access (the user password "empty string" path -// included) is supported. Public-key encryption (/Adobe.PubSec) is -// out of scope. -package crypt - -import ( - "bytes" - "crypto/aes" - "crypto/cipher" - "crypto/md5" - "crypto/rc4" - "crypto/sha256" - "crypto/sha512" - "errors" - "fmt" -) - -// Algorithm identifies a stream/string cipher. -type Algorithm int - -const ( - AlgRC4 Algorithm = iota + 1 - AlgAES128 - AlgAES256 - AlgIdentity -) - -// Handler holds the state needed to decrypt strings and streams in a PDF -// once a password has been validated. -type Handler struct { - V int // /V version - R int // /R revision - Length int // key length in bits (V2/V4) - FileKey []byte // file encryption key (computed from password) - StringAlg Algorithm // algorithm for strings - StreamAlg Algorithm // algorithm for streams - EmbedAlg Algorithm // algorithm for embedded files - CryptFilts map[string]filterDef - StmF string // default stream filter (V4) - StrF string // default string filter (V4) - EFF string // embedded-file filter (V4) -} - -type filterDef struct { - CFM Algorithm -} - -// Params is the inputs needed to instantiate a Handler from the PDF -// /Encrypt dict. -type Params struct { - V int - R int - Length int - P int32 - OwnerEntry []byte // /O - UserEntry []byte // /U - OE []byte // /OE (V5) - UE []byte // /UE (V5) - Perms []byte // /Perms (V5) - ID0 []byte // first element of /ID - EncryptMeta bool - StmF string - StrF string - EFF string - CryptFilters map[string]string // name → CFM -} - -// New tries to instantiate a Handler given the encryption parameters and -// the user password (empty string is the most common case). -func New(p Params, password []byte) (*Handler, error) { - h := &Handler{ - V: p.V, - R: p.R, - Length: p.Length, - StmF: p.StmF, - StrF: p.StrF, - EFF: p.EFF, - CryptFilts: map[string]filterDef{}, - } - for name, cfm := range p.CryptFilters { - alg, err := algFromCFM(cfm) - if err != nil { - return nil, err - } - h.CryptFilts[name] = filterDef{CFM: alg} - } - - switch p.V { - case 1, 2: - h.StringAlg = AlgRC4 - h.StreamAlg = AlgRC4 - h.EmbedAlg = AlgRC4 - key, err := computeRC4Key(p, password) - if err != nil { - return nil, err - } - h.FileKey = key - case 4: - // V4 introduces /CF, /StmF, /StrF for per-stream/string cipher. - h.StringAlg = h.algFor(p.StrF) - h.StreamAlg = h.algFor(p.StmF) - h.EmbedAlg = h.algFor(p.EFF) - key, err := computeRC4Key(p, password) - if err != nil { - return nil, err - } - h.FileKey = key - case 5: - h.StringAlg = AlgAES256 - h.StreamAlg = AlgAES256 - h.EmbedAlg = AlgAES256 - key, err := computeV5Key(p, password) - if err != nil { - return nil, err - } - h.FileKey = key - default: - return nil, fmt.Errorf("crypt: unsupported /V %d", p.V) - } - return h, nil -} - -func (h *Handler) algFor(filterName string) Algorithm { - if filterName == "" || filterName == "Identity" { - return AlgIdentity - } - if def, ok := h.CryptFilts[filterName]; ok { - return def.CFM - } - return AlgIdentity -} - -func algFromCFM(cfm string) (Algorithm, error) { - switch cfm { - case "V2": - return AlgRC4, nil - case "AESV2": - return AlgAES128, nil - case "AESV3": - return AlgAES256, nil - case "None": - return AlgIdentity, nil - } - return 0, fmt.Errorf("crypt: unknown /CFM %q", cfm) -} - -// DecryptString decrypts a string using the configured string algorithm -// and object identity. -func (h *Handler) DecryptString(data []byte, objNum, objGen int) ([]byte, error) { - return h.decrypt(data, objNum, objGen, h.StringAlg) -} - -// DecryptStream decrypts a stream. cryptFilterName, if non-empty, overrides -// the default stream algorithm (V4 streams can carry an inline /Filter -// chain containing /Crypt with parameters). -func (h *Handler) DecryptStream(data []byte, objNum, objGen int, cryptFilterName string) ([]byte, error) { - alg := h.StreamAlg - if cryptFilterName != "" { - alg = h.algFor(cryptFilterName) - } - return h.decrypt(data, objNum, objGen, alg) -} - -func (h *Handler) decrypt(data []byte, objNum, objGen int, alg Algorithm) ([]byte, error) { - switch alg { - case AlgIdentity: - return data, nil - case AlgRC4: - key := h.objKeyRC4orAES(objNum, objGen, false) - out := make([]byte, len(data)) - c, _ := rc4.NewCipher(key) - c.XORKeyStream(out, data) - return out, nil - case AlgAES128: - key := h.objKeyRC4orAES(objNum, objGen, true) - return aesCBCDecrypt(key, data) - case AlgAES256: - return aesCBCDecrypt(h.FileKey, data) - } - return nil, fmt.Errorf("crypt: unknown algorithm %d", alg) -} - -// objKeyRC4orAES derives the per-object encryption key for V2/V4. -// -// PDF 32000-1:2008 §7.6.2: object key = MD5(fileKey || lo3(objNum) || -// lo2(objGen) || (for AES) "sAlT"). Truncate to min(len(fileKey)+5, 16). -func (h *Handler) objKeyRC4orAES(objNum, objGen int, aes bool) []byte { - buf := make([]byte, 0, len(h.FileKey)+9) - buf = append(buf, h.FileKey...) - buf = append(buf, - byte(objNum), - byte(objNum>>8), - byte(objNum>>16), - byte(objGen), - byte(objGen>>8), - ) - if aes { - buf = append(buf, 's', 'A', 'l', 'T') - } - sum := md5.Sum(buf) - n := len(h.FileKey) + 5 - if n > 16 { - n = 16 - } - return sum[:n] -} - -// aesCBCDecrypt unwraps AES/CBC/PKCS#7 with a 16-byte IV prepended. -func aesCBCDecrypt(key, data []byte) ([]byte, error) { - if len(data) < aes.BlockSize { - return nil, errors.New("crypt: AES data shorter than IV") - } - iv := data[:aes.BlockSize] - body := data[aes.BlockSize:] - if len(body)%aes.BlockSize != 0 { - return nil, errors.New("crypt: AES body not block-aligned") - } - block, err := aes.NewCipher(key) - if err != nil { - return nil, err - } - mode := cipher.NewCBCDecrypter(block, iv) - out := make([]byte, len(body)) - mode.CryptBlocks(out, body) - // Strip PKCS#7 padding. - if len(out) == 0 { - return out, nil - } - pad := int(out[len(out)-1]) - if pad < 1 || pad > aes.BlockSize { - return out, nil // tolerate broken padding - } - if pad > len(out) { - return out, nil - } - return out[:len(out)-pad], nil -} - -// computeRC4Key implements PDF 32000-1:2008 algorithm 2 for the file key -// (V1/V2/V4 with RC4 or AESV2). The user password is the input; the empty -// string is the default. -func computeRC4Key(p Params, password []byte) ([]byte, error) { - pad := padPassword(password) - h := md5.New() - h.Write(pad) - h.Write(p.OwnerEntry) - pBytes := []byte{ - byte(uint32(p.P)), - byte(uint32(p.P) >> 8), - byte(uint32(p.P) >> 16), - byte(uint32(p.P) >> 24), - } - h.Write(pBytes) - h.Write(p.ID0) - if p.R >= 4 && !p.EncryptMeta { - h.Write([]byte{0xff, 0xff, 0xff, 0xff}) - } - sum := h.Sum(nil) - keyLen := p.Length / 8 - if keyLen == 0 { - keyLen = 5 // V1 default - } - // /Length is attacker-controlled; the key is sliced from a 16-byte MD5 - // digest, so anything outside [1, md5.Size] would slice/make out of range. - if keyLen < 1 || keyLen > md5.Size { - return nil, fmt.Errorf("crypt: invalid key length %d bits", p.Length) - } - if p.R >= 3 { - for i := 0; i < 50; i++ { - s := md5.Sum(sum[:keyLen]) - sum = s[:] - } - } - key := make([]byte, keyLen) - copy(key, sum[:keyLen]) - - // Validate password by computing U and comparing. - uExpected, err := computeU(p, key) - if err != nil { - return nil, err - } - if !validU(uExpected, p.UserEntry, p.R) { - return nil, errors.New("crypt: password incorrect (V2/V4)") - } - return key, nil -} - -var passPad = []byte{ - 0x28, 0xbf, 0x4e, 0x5e, 0x4e, 0x75, 0x8a, 0x41, - 0x64, 0x00, 0x4e, 0x56, 0xff, 0xfa, 0x01, 0x08, - 0x2e, 0x2e, 0x00, 0xb6, 0xd0, 0x68, 0x3e, 0x80, - 0x2f, 0x0c, 0xa9, 0xfe, 0x64, 0x53, 0x69, 0x7a, -} - -func padPassword(p []byte) []byte { - if len(p) >= 32 { - return p[:32] - } - out := make([]byte, 32) - copy(out, p) - copy(out[len(p):], passPad) - return out -} - -func computeU(p Params, key []byte) ([]byte, error) { - if p.R == 2 { - out := make([]byte, 32) - c, _ := rc4.NewCipher(key) - c.XORKeyStream(out, passPad) - return out, nil - } - // R >= 3. - h := md5.New() - h.Write(passPad) - h.Write(p.ID0) - digest := h.Sum(nil) - out := make([]byte, 16) - c, _ := rc4.NewCipher(key) - c.XORKeyStream(out, digest) - for i := 1; i <= 19; i++ { - tweaked := make([]byte, len(key)) - for j, b := range key { - tweaked[j] = b ^ byte(i) - } - c2, _ := rc4.NewCipher(tweaked) - c2.XORKeyStream(out, out) - } - final := make([]byte, 32) - copy(final, out) - // Trailing bytes are arbitrary per spec; pad with zeros. - return final, nil -} - -func validU(expected, actual []byte, r int) bool { - if r == 2 { - return bytes.Equal(expected, actual) - } - if len(actual) < 16 { - return false - } - return bytes.Equal(expected[:16], actual[:16]) -} - -// computeV5Key implements PDF 32000-2:2020 §7.6.4 / PDF 1.7 Extension 3 -// §3.5.2 — the AES-256 key derivation. -func computeV5Key(p Params, password []byte) ([]byte, error) { - if len(p.UserEntry) < 48 || len(p.OwnerEntry) < 48 { - return nil, errors.New("crypt: V5 entries too short") - } - if len(p.UE) < 32 || len(p.OE) < 32 { - return nil, errors.New("crypt: V5 /UE or /OE missing") - } - // Limit password length to 127 bytes per spec. - if len(password) > 127 { - password = password[:127] - } - uValHash := p.UserEntry[:32] - uVS := p.UserEntry[32:40] - uKS := p.UserEntry[40:48] - oValHash := p.OwnerEntry[:32] - oVS := p.OwnerEntry[32:40] - oKS := p.OwnerEntry[40:48] - _ = uValHash - _ = oValHash - - // Try user password first. - if hash, err := v5Hash(password, uVS, nil, p.R); err == nil && bytes.Equal(hash, uValHash) { - kHash, err := v5Hash(password, uKS, nil, p.R) - if err != nil { - return nil, err - } - return v5DecryptKey(kHash, p.UE) - } - // Try owner password. - if hash, err := v5Hash(password, oVS, p.UserEntry[:48], p.R); err == nil && bytes.Equal(hash, oValHash) { - kHash, err := v5Hash(password, oKS, p.UserEntry[:48], p.R) - if err != nil { - return nil, err - } - return v5DecryptKey(kHash, p.OE) - } - return nil, errors.New("crypt: password incorrect (V5)") -} - -func v5DecryptKey(kHash, encryptedKey []byte) ([]byte, error) { - if len(kHash) != 32 || len(encryptedKey) != 32 { - return nil, errors.New("crypt: V5 key derivation: wrong sizes") - } - block, err := aes.NewCipher(kHash) - if err != nil { - return nil, err - } - iv := make([]byte, aes.BlockSize) - mode := cipher.NewCBCDecrypter(block, iv) - out := make([]byte, 32) - mode.CryptBlocks(out, encryptedKey) - return out, nil -} - -// v5Hash implements the PDF 2.0 / R=6 password hashing function. For R=5 -// (PDF 1.7 Ext.3) it's just SHA-256. -func v5Hash(password, salt, userKey []byte, R int) ([]byte, error) { - switch R { - case 5: - h := sha256.New() - h.Write(password) - h.Write(salt) - h.Write(userKey) - return h.Sum(nil), nil - case 6: - return r6Hash(password, salt, userKey) - } - return nil, fmt.Errorf("crypt: unsupported R=%d", R) -} - -// r6Hash is the iterated AES-128 hash from PDF 2.0 §7.6.4.3.4. -func r6Hash(password, salt, userKey []byte) ([]byte, error) { - h := sha256.New() - h.Write(password) - h.Write(salt) - h.Write(userKey) - K := h.Sum(nil) - round := 0 - for { - K1 := make([]byte, 0, 64*(len(password)+len(K)+len(userKey))) - for i := 0; i < 64; i++ { - K1 = append(K1, password...) - K1 = append(K1, K...) - K1 = append(K1, userKey...) - } - if len(K) < 32 { - return nil, errors.New("r6Hash: short K") - } - block, err := aes.NewCipher(K[:16]) - if err != nil { - return nil, err - } - mode := cipher.NewCBCEncrypter(block, K[16:32]) - E := make([]byte, len(K1)) - mode.CryptBlocks(E, K1) - - // Treat first 16 bytes as big-endian int and take mod 3. - sum := 0 - for i := 0; i < 16; i++ { - sum = (sum*256 + int(E[i])) % 3 - } - switch sum { - case 0: - s := sha256.Sum256(E) - K = s[:] - case 1: - s := sha512.Sum384(E) - K = s[:] - case 2: - s := sha512.Sum512(E) - K = s[:] - } - round++ - if round >= 64 && int(E[len(E)-1]) <= round-32 { - return K[:32], nil - } - if round > 1000 { - return nil, errors.New("r6Hash: too many rounds") - } - } -} diff --git a/internal/crypt/crypt_test.go b/internal/crypt/crypt_test.go deleted file mode 100644 index bbcfdcd..0000000 --- a/internal/crypt/crypt_test.go +++ /dev/null @@ -1,433 +0,0 @@ -package crypt - -import ( - "bytes" - "crypto/aes" - "crypto/cipher" - "crypto/md5" - "testing" -) - -// New must reject an /Encrypt /Length whose derived key size (Length/8) falls -// outside [1, 16] — the RC4/AESV2 file key is sliced from a 16-byte MD5 digest, -// so a hostile large or negative /Length would slice out of range and panic. -func TestNewRejectsHostileKeyLength(t *testing.T) { - for _, length := range []int{136, 256, 4096, -8} { - base := Params{ - V: 2, - R: 3, - Length: length, - OwnerEntry: make([]byte, 32), - UserEntry: make([]byte, 32), - ID0: make([]byte, 16), - } - if _, err := New(base, nil); err == nil { - t.Fatalf("Length=%d: expected error, got nil", length) - } - } -} - -// fixedIV is a deterministic 16-byte IV for reproducible AES test vectors. -var fixedIV = []byte{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15} - -// aesCBCEncryptStream is the inverse of aesCBCDecrypt: PKCS#7-pad, CBC-encrypt, -// and prepend the IV, producing a blob the handler should decrypt back. -func aesCBCEncryptStream(t *testing.T, key, iv, plaintext []byte) []byte { - t.Helper() - padLen := aes.BlockSize - len(plaintext)%aes.BlockSize - padded := append(append([]byte{}, plaintext...), bytes.Repeat([]byte{byte(padLen)}, padLen)...) - block, err := aes.NewCipher(key) - if err != nil { - t.Fatalf("aes.NewCipher: %v", err) - } - body := make([]byte, len(padded)) - cipher.NewCBCEncrypter(block, iv).CryptBlocks(body, padded) - return append(append([]byte{}, iv...), body...) -} - -func TestDecryptRC4RoundTrip(t *testing.T) { - h := &Handler{FileKey: bytes.Repeat([]byte{0x33}, 16), StreamAlg: AlgRC4, StringAlg: AlgRC4} - plaintext := []byte("the quick brown fox / RC4") - // RC4 is symmetric: decrypting plaintext yields ciphertext. - ct, err := h.DecryptStream(plaintext, 12, 0, "") - if err != nil { - t.Fatalf("encrypt: %v", err) - } - if bytes.Equal(ct, plaintext) { - t.Fatal("ciphertext equals plaintext") - } - got, err := h.DecryptStream(ct, 12, 0, "") - if err != nil { - t.Fatalf("decrypt: %v", err) - } - if !bytes.Equal(got, plaintext) { - t.Fatalf("round-trip mismatch: %q", got) - } - // Per-object keying: the same bytes under a different object number must - // not decrypt to the plaintext. - if other, _ := h.DecryptStream(ct, 99, 0, ""); bytes.Equal(other, plaintext) { - t.Fatal("ciphertext decrypted under wrong object number") - } -} - -func TestDecryptAES128RoundTrip(t *testing.T) { - h := &Handler{FileKey: bytes.Repeat([]byte{0x11}, 16), StreamAlg: AlgAES128, StringAlg: AlgAES128} - plaintext := []byte("attachment bytes under AESV2") - key := h.objKeyRC4orAES(7, 0, true) - ct := aesCBCEncryptStream(t, key, fixedIV, plaintext) - got, err := h.DecryptStream(ct, 7, 0, "") - if err != nil { - t.Fatalf("decrypt: %v", err) - } - if !bytes.Equal(got, plaintext) { - t.Fatalf("round-trip mismatch: %q", got) - } -} - -func TestDecryptAES256RoundTrip(t *testing.T) { - // V5/AESV3 keys streams directly with the file key (no per-object key). - h := &Handler{FileKey: bytes.Repeat([]byte{0x22}, 32), StreamAlg: AlgAES256, StringAlg: AlgAES256} - plaintext := []byte("AES-256 stream content for V5") - ct := aesCBCEncryptStream(t, h.FileKey, fixedIV, plaintext) - got, err := h.DecryptString(ct, 5, 0) - if err != nil { - t.Fatalf("decrypt: %v", err) - } - if !bytes.Equal(got, plaintext) { - t.Fatalf("round-trip mismatch: %q", got) - } -} - -// Attacker-supplied AES blobs (too short for the IV, not block-aligned, empty) -// must surface an error or empty output — never panic. -func TestDecryptAESMalformedNoPanic(t *testing.T) { - h := &Handler{FileKey: bytes.Repeat([]byte{0x11}, 16), StreamAlg: AlgAES128} - cases := map[string][]byte{ - "empty": {}, - "shorter_than_iv": make([]byte, aes.BlockSize-1), - "iv_only": make([]byte, aes.BlockSize), - "unaligned_body": make([]byte, aes.BlockSize+aes.BlockSize-1), - "one_byte": {0x00}, - } - for name, data := range cases { - t.Run(name, func(t *testing.T) { - // Must not panic; result is ignored, the point is robustness. - _, _ = h.DecryptStream(data, 1, 0, "") - }) - } -} - -func TestDecryptIdentityPassthrough(t *testing.T) { - h := &Handler{StreamAlg: AlgIdentity, StringAlg: AlgIdentity} - data := []byte{0xde, 0xad, 0xbe, 0xef} - got, err := h.DecryptStream(data, 1, 0, "") - if err != nil { - t.Fatalf("identity: %v", err) - } - if !bytes.Equal(got, data) { - t.Fatal("identity altered data") - } -} - -// deriveRC4Key mirrors computeRC4Key's derivation (without the /U validation), -// so a test can compute the matching /U for an empty-password fixture. -func deriveRC4Key(p Params, password []byte) []byte { - pad := padPassword(password) - h := md5.New() - h.Write(pad) - h.Write(p.OwnerEntry) - h.Write([]byte{ - byte(uint32(p.P)), byte(uint32(p.P) >> 8), - byte(uint32(p.P) >> 16), byte(uint32(p.P) >> 24), - }) - h.Write(p.ID0) - if p.R >= 4 && !p.EncryptMeta { - h.Write([]byte{0xff, 0xff, 0xff, 0xff}) - } - sum := h.Sum(nil) - keyLen := p.Length / 8 - if keyLen == 0 { - keyLen = 5 - } - if p.R >= 3 { - for i := 0; i < 50; i++ { - s := md5.Sum(sum[:keyLen]) - sum = s[:] - } - } - key := make([]byte, keyLen) - copy(key, sum[:keyLen]) - return key -} - -// New must reconstruct the V2/V4 file key from a correct empty-password /U. -func TestNewV2V4KeyDerivationRoundTrip(t *testing.T) { - cases := []struct { - name string - V, R, bits int - }{ - {"V2R2", 2, 2, 40}, - {"V2R3", 2, 3, 128}, - {"V4R4", 4, 4, 128}, - } - for _, tc := range cases { - t.Run(tc.name, func(t *testing.T) { - password := []byte{} - p := Params{ - V: tc.V, R: tc.R, Length: tc.bits, - OwnerEntry: bytes.Repeat([]byte{0x5a}, 32), - ID0: bytes.Repeat([]byte{0x7c}, 16), - P: -3904, - EncryptMeta: true, - StmF: "StdCF", StrF: "StdCF", - CryptFilters: map[string]string{"StdCF": "V2"}, - } - key := deriveRC4Key(p, password) - u, err := computeU(p, key) - if err != nil { - t.Fatalf("computeU: %v", err) - } - p.UserEntry = u - h, err := New(p, password) - if err != nil { - t.Fatalf("New: %v", err) - } - if !bytes.Equal(h.FileKey, key) { - t.Fatalf("file key mismatch:\n got %x\nwant %x", h.FileKey, key) - } - }) - } -} - -// A wrong /U must be rejected, not accepted with a garbage key. -func TestNewV2RejectsWrongUserEntry(t *testing.T) { - p := Params{ - V: 2, R: 3, Length: 128, - OwnerEntry: bytes.Repeat([]byte{0x5a}, 32), - ID0: bytes.Repeat([]byte{0x7c}, 16), - UserEntry: bytes.Repeat([]byte{0x00}, 32), // not the real /U - } - if _, err := New(p, []byte{}); err == nil { - t.Fatal("expected password-incorrect error, got nil") - } -} - -// aesCBCEncryptRaw is the inverse of v5DecryptKey: CBC-encrypt block-aligned -// data with a zero-prepend-free layout. -func aesCBCEncryptRaw(t *testing.T, key, iv, plaintext []byte) []byte { - t.Helper() - block, err := aes.NewCipher(key) - if err != nil { - t.Fatalf("aes.NewCipher: %v", err) - } - out := make([]byte, len(plaintext)) - cipher.NewCBCEncrypter(block, iv).CryptBlocks(out, plaintext) - return out -} - -// computeV5Key must recover the AES-256 file key from a correct empty-password -// /U, /UE for both R=5 (SHA-256) and R=6 (the iterated r6Hash). -func TestComputeV5KeyRoundTrip(t *testing.T) { - for _, R := range []int{5, 6} { - t.Run(map[int]string{5: "R5", 6: "R6"}[R], func(t *testing.T) { - password := []byte("user-pw") - fileKey := bytes.Repeat([]byte{0x42}, 32) - uVS := bytes.Repeat([]byte{0x01}, 8) - uKS := bytes.Repeat([]byte{0x02}, 8) - - uValHash, err := v5Hash(password, uVS, nil, R) - if err != nil { - t.Fatalf("v5Hash(validation): %v", err) - } - kHash, err := v5Hash(password, uKS, nil, R) - if err != nil { - t.Fatalf("v5Hash(key): %v", err) - } - ue := aesCBCEncryptRaw(t, kHash, make([]byte, aes.BlockSize), fileKey) - - userEntry := append(append(append([]byte{}, uValHash...), uVS...), uKS...) - p := Params{ - V: 5, R: R, - UserEntry: userEntry, // 48 bytes - OwnerEntry: make([]byte, 48), // present but unused (user path matches first) - UE: ue, // 32 bytes - OE: make([]byte, 32), - } - key, err := computeV5Key(p, password) - if err != nil { - t.Fatalf("computeV5Key: %v", err) - } - if !bytes.Equal(key, fileKey) { - t.Fatalf("V5 key mismatch:\n got %x\nwant %x", key, fileKey) - } - }) - } -} - -// New must map each V4 /CF crypt-filter method to a cipher and reject unknown -// ones, for a valid empty-password setup. -func TestNewV4CryptFilterMethods(t *testing.T) { - cases := []struct { - cfm string - wantErr bool - }{ - {"V2", false}, {"AESV2", false}, {"AESV3", false}, {"None", false}, {"Bogus", true}, - } - for _, tc := range cases { - t.Run(tc.cfm, func(t *testing.T) { - p := Params{ - V: 4, R: 4, Length: 128, - OwnerEntry: bytes.Repeat([]byte{0x5a}, 32), - ID0: bytes.Repeat([]byte{0x7c}, 16), - P: -3904, - EncryptMeta: true, - StmF: "StdCF", StrF: "StdCF", - CryptFilters: map[string]string{"StdCF": tc.cfm}, - } - key := deriveRC4Key(p, nil) - u, err := computeU(p, key) - if err != nil { - t.Fatalf("computeU: %v", err) - } - p.UserEntry = u - _, err = New(p, nil) - if tc.wantErr != (err != nil) { - t.Fatalf("CFM %q: wantErr=%v, got err=%v", tc.cfm, tc.wantErr, err) - } - }) - } -} - -// computeV5Key must reject short /U, /O, /UE, /OE entries rather than slicing -// out of range. -func TestComputeV5KeyRejectsShortEntries(t *testing.T) { - cases := map[string]Params{ - "short_user": {V: 5, R: 6, UserEntry: make([]byte, 47), OwnerEntry: make([]byte, 48), UE: make([]byte, 32), OE: make([]byte, 32)}, - "short_owner": {V: 5, R: 6, UserEntry: make([]byte, 48), OwnerEntry: make([]byte, 47), UE: make([]byte, 32), OE: make([]byte, 32)}, - "short_ue": {V: 5, R: 6, UserEntry: make([]byte, 48), OwnerEntry: make([]byte, 48), UE: make([]byte, 31), OE: make([]byte, 32)}, - "short_oe": {V: 5, R: 6, UserEntry: make([]byte, 48), OwnerEntry: make([]byte, 48), UE: make([]byte, 32), OE: make([]byte, 31)}, - } - for name, p := range cases { - t.Run(name, func(t *testing.T) { - if _, err := computeV5Key(p, []byte{}); err == nil { - t.Fatal("expected error for short entry, got nil") - } - }) - } -} - -// Owner-password path. Unlike the user path, the owner hash mixes in the -// 48-byte /U entry (§7.6.4.4.10) — so userEntry here must be well-formed. -func TestComputeV5KeyOwnerPath(t *testing.T) { - const R = 6 - password := []byte("owner-pw") - fileKey := bytes.Repeat([]byte{0x37}, 32) - - // All-zero validation hash (first 32 bytes) can never equal v5Hash output, - // forcing the user path to miss so the owner path is taken. - uVS := bytes.Repeat([]byte{0x11}, 8) - uKS := bytes.Repeat([]byte{0x22}, 8) - userEntry := append(append(make([]byte, 32), uVS...), uKS...) // 48 bytes - - oVS := bytes.Repeat([]byte{0x33}, 8) - oKS := bytes.Repeat([]byte{0x44}, 8) - oValHash, err := v5Hash(password, oVS, userEntry, R) - if err != nil { - t.Fatalf("v5Hash(owner validation): %v", err) - } - oKeyHash, err := v5Hash(password, oKS, userEntry, R) - if err != nil { - t.Fatalf("v5Hash(owner key): %v", err) - } - oe := aesCBCEncryptRaw(t, oKeyHash, make([]byte, aes.BlockSize), fileKey) - ownerEntry := append(append(append([]byte{}, oValHash...), oVS...), oKS...) - - p := Params{ - V: 5, R: R, - UserEntry: userEntry, - OwnerEntry: ownerEntry, - UE: make([]byte, 32), // present but never reached - OE: oe, - } - key, err := computeV5Key(p, password) - if err != nil { - t.Fatalf("computeV5Key: %v", err) - } - if !bytes.Equal(key, fileKey) { - t.Fatalf("owner-path key mismatch:\n got %x\nwant %x", key, fileKey) - } -} - -func TestComputeV5KeyWrongPassword(t *testing.T) { - p := Params{ - V: 5, R: 6, - UserEntry: bytes.Repeat([]byte{0x01}, 48), - OwnerEntry: bytes.Repeat([]byte{0x02}, 48), - UE: bytes.Repeat([]byte{0x03}, 32), - OE: bytes.Repeat([]byte{0x04}, 32), - } - overlong := bytes.Repeat([]byte{'z'}, 200) // > 127: exercises the spec truncation - if _, err := computeV5Key(p, overlong); err == nil { - t.Fatal("expected an error for a non-matching password, got nil") - } -} - -// New must reject an unsupported /V rather than returning a zero handler. -func TestNewUnsupportedVersion(t *testing.T) { - for _, v := range []int{0, 3, 99} { - if _, err := New(Params{V: v}, nil); err == nil { - t.Errorf("New(/V %d) should error", v) - } - } -} - -// New drives the /V 5 branch end to end (AES-256 key derivation + algorithm -// selection), not just computeV5Key in isolation. -func TestNewV5(t *testing.T) { - const R = 6 - password := []byte("v5-user") - fileKey := bytes.Repeat([]byte{0x42}, 32) - uVS := bytes.Repeat([]byte{0x01}, 8) - uKS := bytes.Repeat([]byte{0x02}, 8) - uValHash, err := v5Hash(password, uVS, nil, R) - if err != nil { - t.Fatalf("v5Hash(validation): %v", err) - } - kHash, err := v5Hash(password, uKS, nil, R) - if err != nil { - t.Fatalf("v5Hash(key): %v", err) - } - ue := aesCBCEncryptRaw(t, kHash, make([]byte, aes.BlockSize), fileKey) - userEntry := append(append(append([]byte{}, uValHash...), uVS...), uKS...) - - h, err := New(Params{ - V: 5, R: R, - UserEntry: userEntry, - OwnerEntry: make([]byte, 48), - UE: ue, - OE: make([]byte, 32), - }, password) - if err != nil { - t.Fatalf("New(V5): %v", err) - } - if !bytes.Equal(h.FileKey, fileKey) { - t.Fatal("V5 file key mismatch") - } - if h.StreamAlg != AlgAES256 { - t.Errorf("StreamAlg = %v, want AlgAES256", h.StreamAlg) - } -} - -// A per-stream Identity crypt filter overrides the default algorithm and passes -// the bytes through unchanged. -func TestDecryptStreamIdentityOverride(t *testing.T) { - h := &Handler{StreamAlg: AlgRC4, FileKey: bytes.Repeat([]byte{1}, 16)} - data := []byte("plaintext") - out, err := h.DecryptStream(data, 1, 0, "Identity") - if err != nil { - t.Fatalf("DecryptStream: %v", err) - } - if !bytes.Equal(out, data) { - t.Errorf("Identity override = %q, want %q", out, data) - } -} diff --git a/internal/filter/filter.go b/internal/filter/filter.go deleted file mode 100644 index 757c6e0..0000000 --- a/internal/filter/filter.go +++ /dev/null @@ -1,377 +0,0 @@ -// Package filter implements the PDF stream filters needed for read-only -// inspection of document structure: FlateDecode (with predictors), LZW, -// ASCII85, ASCIIHex, RunLength, and BrotliDecode (a PDF Association -// extension to PDF 2.0, pending ISO 32000 inclusion). -// -// Image-only filters (DCTDecode, JBIG2Decode, JPXDecode, CCITTFaxDecode) -// are intentionally not implemented — pdfdisassembler does not decode -// image streams. -package filter - -import ( - "bytes" - "compress/zlib" - "errors" - "fmt" - "io" - - "github.com/andybalholm/brotli" -) - -// ErrUnsupported is returned for filters this package does not implement. -type ErrUnsupported struct{ Name string } - -func (e ErrUnsupported) Error() string { - return fmt.Sprintf("pdfdisassembler/filter: unsupported filter %q", e.Name) -} - -// Params describes the decode-time parameters for a single filter. -type Params struct { - // FlateDecode/LZWDecode predictor parameters. - Predictor int - Columns int - Colors int - BitsPerComponent int - // NoEarlyChange honours the rare /EarlyChange 0; the zero value keeps - // LZWDecode's default early code-width change. - NoEarlyChange bool - // MaxOutput caps decoded bytes per filter; <= 0 means unlimited. - MaxOutput int64 -} - -// Decode applies the named filter to in. -func Decode(name string, in []byte, p Params) ([]byte, error) { - switch name { - case "FlateDecode", "Fl": - return decodeFlate(in, p) - case "LZWDecode", "LZW": - return decodeLZW(in, p) - case "ASCII85Decode", "A85": - return decodeASCII85(in) - case "ASCIIHexDecode", "AHx": - return decodeASCIIHex(in) - case "RunLengthDecode", "RL": - return decodeRunLength(in, p.MaxOutput) - case "BrotliDecode": - // No abbreviation: the spec keeps BrotliDecode out of the - // inline-image abbreviation table (it is banned there). - return decodeBrotli(in, p) - } - return nil, ErrUnsupported{Name: name} -} - -// IsImageFilter reports whether name designates one of the image-only -// filters that pdfdisassembler intentionally skips. -func IsImageFilter(name string) bool { - switch name { - case "DCTDecode", "DCT", - "JBIG2Decode", - "JPXDecode", - "CCITTFaxDecode", "CCF", - "Crypt": - return true - } - return false -} - -func decodeFlate(in []byte, p Params) ([]byte, error) { - zr, err := zlib.NewReader(bytes.NewReader(in)) - if err != nil { - return nil, fmt.Errorf("FlateDecode: %w", err) - } - defer zr.Close() - dec, err := readAllLimited(zr, p.MaxOutput) - if err != nil { - return nil, fmt.Errorf("FlateDecode: %w", err) - } - if p.Predictor > 1 { - return applyPredictor(dec, p) - } - return dec, nil -} - -// decodeBrotli handles BrotliDecode ("Brotli compression in PDF 2.0", -// PDF Association EXTN-BROTLI-1): stream data compressed per RFC 7932 with -// no additional headers. The spec extends the FlateDecode/LZWDecode -// DecodeParms table to BrotliDecode, so predictors apply exactly as in -// decodeFlate. Large-window Brotli (RFC 9841) is required by the spec but -// not reachable through the underlying decoder's API; such streams (rare, -// encoder opt-in only) fail with a window-bits format error rather than -// decoding incorrectly. -func decodeBrotli(in []byte, p Params) ([]byte, error) { - // brotli.NewReader cannot fail; errors surface on Read. - dec, err := readAllLimited(brotli.NewReader(bytes.NewReader(in)), p.MaxOutput) - if err != nil { - return nil, fmt.Errorf("BrotliDecode: %w", err) - } - if p.Predictor > 1 { - return applyPredictor(dec, p) - } - return dec, nil -} - -// readAllLimited is io.ReadAll that errors once r yields more than max bytes. -// max <= 0 disables the limit. -func readAllLimited(r io.Reader, max int64) ([]byte, error) { - if max <= 0 { - return io.ReadAll(r) - } - out, err := io.ReadAll(io.LimitReader(r, max+1)) // +1 to tell "== max" from "> max" - if err != nil { - return nil, err - } - if int64(len(out)) > max { - return nil, fmt.Errorf("decoded output exceeds %d-byte limit (possible decompression bomb)", max) - } - return out, nil -} - -func decodeASCII85(in []byte) ([]byte, error) { - // Trim "<~" prefix and "~>" suffix if present. - if len(in) >= 2 && in[0] == '<' && in[1] == '~' { - in = in[2:] - } - end := bytes.Index(in, []byte("~>")) - if end >= 0 { - in = in[:end] - } - - var out []byte - var group uint32 - n := 0 - for _, c := range in { - switch { - case c == 'z': - if n != 0 { - return nil, errors.New("ASCII85Decode: 'z' inside group") - } - out = append(out, 0, 0, 0, 0) - continue - case c >= '!' && c <= 'u': - group = group*85 + uint32(c-'!') - n++ - case c == ' ' || c == '\t' || c == '\r' || c == '\n' || c == '\f': - continue - default: - return nil, fmt.Errorf("ASCII85Decode: invalid byte 0x%02x", c) - } - if n == 5 { - out = append(out, - byte(group>>24), - byte(group>>16), - byte(group>>8), - byte(group), - ) - group = 0 - n = 0 - } - } - if n > 0 { - for i := n; i < 5; i++ { - group = group*85 + 84 - } - buf := []byte{ - byte(group >> 24), - byte(group >> 16), - byte(group >> 8), - byte(group), - } - out = append(out, buf[:n-1]...) - } - return out, nil -} - -func decodeASCIIHex(in []byte) ([]byte, error) { - var out []byte - var hi int - have := false - for _, c := range in { - if c == '>' { - break - } - if c == ' ' || c == '\t' || c == '\r' || c == '\n' || c == '\f' { - continue - } - d, ok := hexDigit(c) - if !ok { - return nil, fmt.Errorf("ASCIIHexDecode: invalid hex byte 0x%02x", c) - } - if have { - out = append(out, byte(hi<<4|d)) - have = false - } else { - hi = d - have = true - } - } - if have { - out = append(out, byte(hi<<4)) - } - return out, nil -} - -func decodeRunLength(in []byte, maxOut int64) ([]byte, error) { - var out []byte - for i := 0; i < len(in); { - b := in[i] - i++ - switch { - case b < 128: - n := int(b) + 1 - if i+n > len(in) { - return nil, errors.New("RunLengthDecode: truncated literal") - } - out = append(out, in[i:i+n]...) - i += n - case b > 128: - n := 257 - int(b) - if i >= len(in) { - return nil, errors.New("RunLengthDecode: truncated run") - } - for k := 0; k < n; k++ { - out = append(out, in[i]) - } - i++ - default: // b == 128: EOD - return out, nil - } - if maxOut > 0 && int64(len(out)) > maxOut { - return nil, fmt.Errorf("RunLengthDecode: decoded output exceeds limit of %d bytes", maxOut) - } - } - return out, nil -} - -func hexDigit(c byte) (int, bool) { - switch { - case c >= '0' && c <= '9': - return int(c - '0'), true - case c >= 'a' && c <= 'f': - return int(c-'a') + 10, true - case c >= 'A' && c <= 'F': - return int(c-'A') + 10, true - } - return 0, false -} - -// applyPredictor reverses the PNG / TIFF predictor wrapping applied to -// LZW/Flate data. See PDF 32000-1:2008 §7.4.4.4. -func applyPredictor(in []byte, p Params) ([]byte, error) { - if p.Predictor <= 1 { - return in, nil - } - colors := p.Colors - if colors == 0 { - colors = 1 - } - bpc := p.BitsPerComponent - if bpc == 0 { - bpc = 8 - } - columns := p.Columns - if columns == 0 { - columns = 1 - } - // /Colors, /BitsPerComponent, /Columns are attacker-controlled. Negative - // values (or an overflowing product) drive rowBytes to zero or negative, - // which would divide-by-zero or make a negative-length slice below. - if colors < 1 || bpc < 1 || columns < 1 { - return nil, fmt.Errorf("predictor: /Colors, /BitsPerComponent, /Columns must be positive") - } - bytesPerPixel := (colors*bpc + 7) / 8 - rowBytes := (columns*colors*bpc + 7) / 8 - if bytesPerPixel < 1 || rowBytes < 1 { - return nil, fmt.Errorf("predictor: invalid row geometry (bytesPerPixel=%d, rowBytes=%d)", bytesPerPixel, rowBytes) - } - - if p.Predictor == 2 { - // TIFF predictor 2: per-row horizontal differences. Not commonly - // used in our domain but supported for completeness. - if len(in)%rowBytes != 0 { - return nil, fmt.Errorf("predictor 2: %d bytes not divisible by row %d", len(in), rowBytes) - } - out := make([]byte, len(in)) - for r := 0; r < len(in); r += rowBytes { - row := in[r : r+rowBytes] - dst := out[r : r+rowBytes] - copy(dst, row) - for c := bytesPerPixel; c < rowBytes; c++ { - dst[c] = byte(int(row[c]) + int(dst[c-bytesPerPixel])) - } - } - return out, nil - } - - // PNG predictors: rowBytes data preceded by a 1-byte filter tag. - stride := rowBytes + 1 - if len(in)%stride != 0 { - return nil, fmt.Errorf("predictor PNG: %d bytes not divisible by row %d", len(in), stride) - } - rows := len(in) / stride - out := make([]byte, rows*rowBytes) - prev := make([]byte, rowBytes) - cur := make([]byte, rowBytes) - for r := 0; r < rows; r++ { - tag := in[r*stride] - row := in[r*stride+1 : (r+1)*stride] - switch tag { - case 0: // None - copy(cur, row) - case 1: // Sub - for c := 0; c < rowBytes; c++ { - var left byte - if c >= bytesPerPixel { - left = cur[c-bytesPerPixel] - } - cur[c] = row[c] + left - } - case 2: // Up - for c := 0; c < rowBytes; c++ { - cur[c] = row[c] + prev[c] - } - case 3: // Average - for c := 0; c < rowBytes; c++ { - var left byte - if c >= bytesPerPixel { - left = cur[c-bytesPerPixel] - } - cur[c] = row[c] + byte((int(left)+int(prev[c]))/2) - } - case 4: // Paeth - for c := 0; c < rowBytes; c++ { - var left, upLeft byte - if c >= bytesPerPixel { - left = cur[c-bytesPerPixel] - upLeft = prev[c-bytesPerPixel] - } - cur[c] = row[c] + paeth(left, prev[c], upLeft) - } - default: - return nil, fmt.Errorf("predictor PNG: unknown tag %d", tag) - } - copy(out[r*rowBytes:], cur) - copy(prev, cur) - } - return out, nil -} - -func paeth(a, b, c byte) byte { - p := int(a) + int(b) - int(c) - pa := abs(p - int(a)) - pb := abs(p - int(b)) - pc := abs(p - int(c)) - switch { - case pa <= pb && pa <= pc: - return a - case pb <= pc: - return b - } - return c -} - -func abs(x int) int { - if x < 0 { - return -x - } - return x -} diff --git a/internal/filter/filter_test.go b/internal/filter/filter_test.go deleted file mode 100644 index b10594b..0000000 --- a/internal/filter/filter_test.go +++ /dev/null @@ -1,529 +0,0 @@ -package filter - -import ( - "bytes" - "compress/lzw" - "compress/zlib" - "encoding/ascii85" - "errors" - "fmt" - "strings" - "testing" - - "github.com/andybalholm/brotli" -) - -// lcg fills b with a deterministic pseudo-random byte stream — enough entropy -// to grow the LZW dictionary through every code width and trigger a reset. -func lcg(b []byte, seed uint32) { - x := seed - for i := range b { - x = x*1664525 + 1013904223 - b[i] = byte(x >> 24) - } -} - -// TestLZWRoundTripStdlib decodes streams produced by the standard library's -// MSB LZW writer. That writer uses the non-early code-width change, so decode -// with NoEarlyChange; the varied inputs exercise dictionary reuse, the KwKwK -// case, 9->12-bit width growth, and the dictionary-full reset. -func TestLZWRoundTripStdlib(t *testing.T) { - big := make([]byte, 64<<10) - lcg(big, 1) - for _, orig := range [][]byte{ - []byte("ABABABABABABABABABAB"), - []byte("TOBEORNOTTOBEORTOBEORNOT"), - bytes.Repeat([]byte("xyz "), 4096), - big, - {}, - {42}, - } { - var buf bytes.Buffer - w := lzw.NewWriter(&buf, lzw.MSB, 8) - if _, err := w.Write(orig); err != nil { - t.Fatal(err) - } - w.Close() - got, err := Decode("LZWDecode", buf.Bytes(), Params{NoEarlyChange: true}) - if err != nil { - t.Fatalf("decode (len %d): %v", len(orig), err) - } - if !bytes.Equal(got, orig) { - t.Fatalf("round-trip mismatch for len %d input", len(orig)) - } - } -} - -// TestLZWEarlyChangeHonored proves NoEarlyChange actually selects the decode -// convention: the stdlib stream (non-early) round-trips only with NoEarlyChange -// set; under the early-change default the same bytes must not reproduce the input. -func TestLZWEarlyChangeHonored(t *testing.T) { - orig := make([]byte, 4096) // long enough to cross the first width boundary - lcg(orig, 7) - var buf bytes.Buffer - w := lzw.NewWriter(&buf, lzw.MSB, 8) - w.Write(orig) - w.Close() - enc := buf.Bytes() - - got, err := Decode("LZWDecode", enc, Params{NoEarlyChange: true}) - if err != nil || !bytes.Equal(got, orig) { - t.Fatalf("NoEarlyChange decode of a non-early stream: err=%v equal=%v", err, bytes.Equal(got, orig)) - } - if d, err := Decode("LZWDecode", enc, Params{}); err == nil && bytes.Equal(d, orig) { - t.Fatal("early-change default reproduced a non-early stream: the flag is ignored") - } -} - -// A stream truncated mid-code must not panic: readBits pads the final partial -// word with zeros and decoding stops cleanly (partial output or error). -func TestLZWTruncatedNoPanic(t *testing.T) { - orig := make([]byte, 600) // long enough to reach 10-bit codes - lcg(orig, 3) - var buf bytes.Buffer - w := lzw.NewWriter(&buf, lzw.MSB, 8) - w.Write(orig) - w.Close() - enc := buf.Bytes() - for cut := 1; cut <= 4 && cut < len(enc); cut++ { - _, _ = Decode("LZWDecode", enc[:len(enc)-cut], Params{NoEarlyChange: true}) - } -} - -func TestASCII85RoundTripStdlib(t *testing.T) { - for _, orig := range [][]byte{ - []byte("Hello World!"), - {0, 0, 0, 0, 1, 2, 3}, // includes an all-zero group - {1}, // 1-byte partial group - {1, 2}, // 2-byte partial group - {1, 2, 3}, // 3-byte partial group - bytes.Repeat([]byte{255}, 17), - {}, - } { - enc := make([]byte, ascii85.MaxEncodedLen(len(orig))) - n := ascii85.Encode(enc, orig) - in := append(enc[:n:n], '~', '>') - out, err := Decode("ASCII85Decode", in, Params{}) - if err != nil { - t.Fatalf("decode %v: %v", orig, err) - } - if !bytes.Equal(out, orig) { - t.Fatalf("round-trip mismatch: in=%v out=%v", orig, out) - } - } -} - -func TestASCII85EdgeCases(t *testing.T) { - // 'z' is shorthand for a full zero group; <~ is an optional opening marker; - // whitespace between digits is ignored. - out, err := Decode("ASCII85Decode", []byte("<~z 8 7 c U R D ] i , \" E b o 8 0 ~>"), Params{}) - if err != nil { - t.Fatalf("decode: %v", err) - } - if want := append([]byte{0, 0, 0, 0}, "Hello World!"...); !bytes.Equal(out, want) { - t.Fatalf("got %q, want %q", out, want) - } - - if _, err := Decode("ASCII85Decode", []byte("abc!\x01def~>"), Params{}); err == nil { - t.Fatal("expected an error on a byte outside the ASCII85 alphabet") - } - // 'z' may not appear mid-group (after a partial digit). - if _, err := Decode("ASCII85Decode", []byte("87z~>"), Params{}); err == nil { - t.Fatal("expected an error for 'z' inside a group") - } -} - -func TestASCIIHex(t *testing.T) { - cases := map[string]string{ - "48656C6C6F>": "Hello", - "4 86 56C 6C6F": "Hello", - "48656c6c6f": "Hello", - } - for in, want := range cases { - out, err := Decode("ASCIIHexDecode", []byte(in), Params{}) - if err != nil { - t.Fatalf("%q: %v", in, err) - } - if string(out) != want { - t.Fatalf("%q: got %q want %q", in, out, want) - } - } -} - -func TestASCII85(t *testing.T) { - in := []byte("87cURD]i,\"Ebo80~>") - out, err := Decode("ASCII85Decode", in, Params{}) - if err != nil { - t.Fatal(err) - } - if string(out) != "Hello World!" { - t.Fatalf("got %q", out) - } -} - -func TestRunLength(t *testing.T) { - // 3 literal "ABC" (length-1=2), then 3 copies of 'X' (257-3=254), EOD. - in := []byte{2, 'A', 'B', 'C', 254, 'X', 128} - out, err := Decode("RunLengthDecode", in, Params{}) - if err != nil { - t.Fatal(err) - } - if string(out) != "ABCXXX" { - t.Fatalf("got %q", out) - } -} - -func TestFlate(t *testing.T) { - var buf bytes.Buffer - zw := zlib.NewWriter(&buf) - zw.Write([]byte("hello flate")) - zw.Close() - out, err := Decode("FlateDecode", buf.Bytes(), Params{}) - if err != nil { - t.Fatal(err) - } - if string(out) != "hello flate" { - t.Fatalf("got %q", out) - } -} - -func TestFlatePNGPredictor(t *testing.T) { - // 2 rows, 4 bytes each. Predictor tag 0 = None. - row1 := []byte{0, 1, 2, 3, 4} - row2 := []byte{0, 5, 6, 7, 8} - raw := append(append([]byte{}, row1...), row2...) - var buf bytes.Buffer - zw := zlib.NewWriter(&buf) - zw.Write(raw) - zw.Close() - out, err := Decode("FlateDecode", buf.Bytes(), Params{ - Predictor: 12, - Columns: 4, - Colors: 1, - BitsPerComponent: 8, - }) - if err != nil { - t.Fatal(err) - } - want := []byte{1, 2, 3, 4, 5, 6, 7, 8} - if !bytes.Equal(out, want) { - t.Fatalf("got % x want % x", out, want) - } -} - -func TestPredictorHostileParamsNoPanic(t *testing.T) { - var buf bytes.Buffer - zw := zlib.NewWriter(&buf) - zw.Write([]byte("ABCDEFGH")) - zw.Close() - flate := buf.Bytes() - - cases := []struct { - name string - p Params - }{ - {"tiff_rowbytes_zero", Params{Predictor: 2, Colors: -7, BitsPerComponent: 1, Columns: 1}}, - {"png_stride_zero", Params{Predictor: 12, Colors: -15, BitsPerComponent: 1, Columns: 1}}, - {"png_negative_make", Params{Predictor: 12, Colors: -23, BitsPerComponent: 1, Columns: 1}}, - {"negative_columns", Params{Predictor: 12, Colors: 1, BitsPerComponent: 8, Columns: -4}}, - } - for _, tc := range cases { - t.Run(tc.name, func(t *testing.T) { - if _, err := Decode("FlateDecode", flate, tc.p); err == nil { - t.Fatal("expected an error for hostile predictor params, got nil") - } - }) - } -} - -// pngForwardFilter is the inverse of applyPredictor's PNG path: it applies row -// filter tag to rows, so decode(encode(rows)) must recover rows exactly. -func pngForwardFilter(tag byte, rows [][]byte, bpp int) []byte { - var out []byte - prev := make([]byte, len(rows[0])) - for _, raw := range rows { - out = append(out, tag) - filt := make([]byte, len(raw)) - for c := range raw { - var left, upLeft byte - up := prev[c] - if c >= bpp { - left = raw[c-bpp] - upLeft = prev[c-bpp] - } - switch tag { - case 0: - filt[c] = raw[c] - case 1: - filt[c] = raw[c] - left - case 2: - filt[c] = raw[c] - up - case 3: - filt[c] = raw[c] - byte((int(left)+int(up))/2) - case 4: - filt[c] = raw[c] - paeth(left, up, upLeft) - } - } - out = append(out, filt...) - prev = raw - } - return out -} - -func flate(t *testing.T, b []byte) []byte { - t.Helper() - var buf bytes.Buffer - zw := zlib.NewWriter(&buf) - zw.Write(b) - zw.Close() - return buf.Bytes() -} - -func TestPredictorPNGRoundTrip(t *testing.T) { - rows := [][]byte{ - {10, 20, 30, 40}, - {15, 25, 35, 45}, - {200, 100, 50, 25}, - } - var want []byte - for _, r := range rows { - want = append(want, r...) - } - for tag := byte(0); tag <= 4; tag++ { - t.Run(fmt.Sprintf("tag%d", tag), func(t *testing.T) { - filtered := pngForwardFilter(tag, rows, 1) - out, err := Decode("FlateDecode", flate(t, filtered), Params{ - Predictor: 12, Columns: 4, Colors: 1, BitsPerComponent: 8, - }) - if err != nil { - t.Fatalf("decode: %v", err) - } - if !bytes.Equal(out, want) { - t.Fatalf("tag %d round-trip: got % x want % x", tag, out, want) - } - }) - } -} - -func TestPredictorTIFFRoundTrip(t *testing.T) { - raw := []byte{10, 5, 250, 3} - filt := make([]byte, len(raw)) - filt[0] = raw[0] - for c := 1; c < len(raw); c++ { - filt[c] = raw[c] - raw[c-1] - } - out, err := Decode("FlateDecode", flate(t, filt), Params{ - Predictor: 2, Columns: 4, Colors: 1, BitsPerComponent: 8, - }) - if err != nil { - t.Fatalf("decode: %v", err) - } - if !bytes.Equal(out, raw) { - t.Fatalf("TIFF round-trip: got % x want % x", out, raw) - } -} - -func TestFlateBombRejected(t *testing.T) { - // 1 MiB of zeros (compresses to ~1 KB) against a 4 KB cap. - var buf bytes.Buffer - zw := zlib.NewWriter(&buf) - zw.Write(make([]byte, 1<<20)) - zw.Close() - if _, err := Decode("FlateDecode", buf.Bytes(), Params{MaxOutput: 4096}); err == nil { - t.Fatal("expected error for output exceeding MaxOutput, got nil") - } -} - -func TestFlateUnderLimit(t *testing.T) { - var buf bytes.Buffer - zw := zlib.NewWriter(&buf) - zw.Write([]byte("hello flate")) - zw.Close() - out, err := Decode("FlateDecode", buf.Bytes(), Params{MaxOutput: 1 << 20}) - if err != nil { - t.Fatal(err) - } - if string(out) != "hello flate" { - t.Fatalf("got %q", out) - } -} - -// brotliCompress is the test-side encoder for the read-only BrotliDecode path. -func brotliCompress(t *testing.T, b []byte) []byte { - t.Helper() - var buf bytes.Buffer - bw := brotli.NewWriter(&buf) - if _, err := bw.Write(b); err != nil { - t.Fatal(err) - } - if err := bw.Close(); err != nil { - t.Fatal(err) - } - return buf.Bytes() -} - -func TestBrotliRoundTrip(t *testing.T) { - big := make([]byte, 64<<10) - lcg(big, 5) - for _, orig := range [][]byte{ - []byte("hello brotli"), - bytes.Repeat([]byte("xyz "), 4096), - big, - {}, - {42}, - } { - got, err := Decode("BrotliDecode", brotliCompress(t, orig), Params{MaxOutput: 1 << 20}) - if err != nil { - t.Fatalf("decode (len %d): %v", len(orig), err) - } - if !bytes.Equal(got, orig) { - t.Fatalf("round-trip mismatch for len %d input", len(orig)) - } - } -} - -func TestBrotliCorruptRejected(t *testing.T) { - enc := brotliCompress(t, bytes.Repeat([]byte("pdfdisassembler "), 512)) - cases := map[string][]byte{ - "empty": {}, - "garbage": {0xC1, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF}, - "truncated": enc[:len(enc)/2], - } - for name, in := range cases { - t.Run(name, func(t *testing.T) { - _, err := Decode("BrotliDecode", in, Params{}) - if err == nil { - t.Fatal("expected an error, got nil") - } - if !strings.Contains(err.Error(), "BrotliDecode") { - t.Fatalf("error %q does not name BrotliDecode", err) - } - }) - } -} - -func TestBrotliBombRejected(t *testing.T) { - // 1 MiB of zeros (compresses to a few bytes) against a 4 KB cap. - if _, err := Decode("BrotliDecode", brotliCompress(t, make([]byte, 1<<20)), Params{MaxOutput: 4096}); err == nil { - t.Fatal("expected error for output exceeding MaxOutput, got nil") - } -} - -// The Brotli extension spec extends the Flate/LZW predictor parameters to -// BrotliDecode, so the PNG-predictor path must work identically. -func TestBrotliPNGPredictor(t *testing.T) { - // 2 rows, 4 bytes each. Predictor tag 0 = None. - row1 := []byte{0, 1, 2, 3, 4} - row2 := []byte{0, 5, 6, 7, 8} - raw := append(append([]byte{}, row1...), row2...) - out, err := Decode("BrotliDecode", brotliCompress(t, raw), Params{ - Predictor: 12, - Columns: 4, - Colors: 1, - BitsPerComponent: 8, - }) - if err != nil { - t.Fatal(err) - } - want := []byte{1, 2, 3, 4, 5, 6, 7, 8} - if !bytes.Equal(out, want) { - t.Fatalf("got % x want % x", out, want) - } -} - -// The spec defines no abbreviation for BrotliDecode (it is excluded from the -// inline-image abbreviation table), so "Br" must stay unsupported. -func TestBrotliNoAbbreviation(t *testing.T) { - var unsup ErrUnsupported - if _, err := Decode("Br", []byte{0x3B}, Params{}); !errors.As(err, &unsup) { - t.Fatalf("Decode(Br) error = %v, want ErrUnsupported", err) - } -} - -func TestRunLengthBombRejected(t *testing.T) { - // {129,'X'} expands to 257-129 = 128 copies of 'X', past the 64-byte cap. - if _, err := Decode("RunLengthDecode", []byte{129, 'X'}, Params{MaxOutput: 64}); err == nil { - t.Fatal("expected error for output exceeding MaxOutput, got nil") - } -} - -func TestLZWDecodeAndBombRejected(t *testing.T) { - // PDF-LZW (9-bit, MSB-first) codes 65,66,257 -> "AB" then EOD. - in := []byte{0x20, 0x90, 0xA0, 0x20} - out, err := Decode("LZWDecode", in, Params{}) - if err != nil || string(out) != "AB" { - t.Fatalf("baseline decode: out=%q err=%v", out, err) - } - if _, err := Decode("LZWDecode", in, Params{MaxOutput: 1}); err == nil { - t.Fatal("expected error for output exceeding MaxOutput, got nil") - } -} - -func TestImageFilterRejected(t *testing.T) { - if !IsImageFilter("DCTDecode") { - t.Fatal("DCTDecode should be image filter") - } - if !IsImageFilter("JPXDecode") { - t.Fatal("JPXDecode should be image filter") - } - if IsImageFilter("FlateDecode") { - t.Fatal("FlateDecode should not be image filter") - } -} - -// FuzzDecode asserts every filter and the predictor never panic on arbitrary -// input or attacker-controlled predictor parameters. -func FuzzDecode(f *testing.F) { - var seed bytes.Buffer - zw := zlib.NewWriter(&seed) - zw.Write([]byte("seed data")) - zw.Close() - f.Add(seed.Bytes(), 12, 4, 1, 8) - f.Fuzz(func(t *testing.T, data []byte, predictor, columns, colors, bpc int) { - p := Params{ - Predictor: predictor, - Columns: columns, - Colors: colors, - BitsPerComponent: bpc, - MaxOutput: 1 << 20, - } - for _, name := range []string{"FlateDecode", "LZWDecode", "ASCII85Decode", "ASCIIHexDecode", "RunLengthDecode", "BrotliDecode"} { - _, _ = Decode(name, data, p) - } - }) -} - -// Expected values hand-computed from PNG §6.6 (not copied from output). -// paeth(0,0,10) makes p=a+b-c negative — the one vector here that reaches -// abs's negative branch; don't drop it. -func TestPaeth(t *testing.T) { - cases := []struct { - a, b, c, want byte - }{ - {0, 0, 0, 0}, - {1, 2, 3, 1}, // p=0: pa=1 pb=2 pc=3 -> a - {255, 0, 0, 255}, // p=255: pa=0 -> a - {0, 0, 10, 0}, // p=-10: pa=10 pb=10 pc=20 -> a (negative estimate) - {10, 20, 11, 20}, // p=19: pa=9 pb=1 pc=8 -> b - {0, 10, 6, 6}, // p=4: pa=4 pb=6 pc=2 -> c - } - for _, tc := range cases { - if got := paeth(tc.a, tc.b, tc.c); got != tc.want { - t.Errorf("paeth(%d,%d,%d) = %d, want %d", tc.a, tc.b, tc.c, got, tc.want) - } - } -} - -func TestErrUnsupported(t *testing.T) { - _, err := Decode("DCTDecode", []byte("x"), Params{}) - var unsup ErrUnsupported - if !errors.As(err, &unsup) { - t.Fatalf("Decode(DCTDecode) error = %v, want ErrUnsupported", err) - } - if unsup.Name != "DCTDecode" { - t.Errorf("ErrUnsupported.Name = %q, want DCTDecode", unsup.Name) - } - if want := `pdfdisassembler/filter: unsupported filter "DCTDecode"`; unsup.Error() != want { - t.Errorf("Error() = %q, want %q", unsup.Error(), want) - } -} diff --git a/internal/filter/lzw.go b/internal/filter/lzw.go deleted file mode 100644 index 53dad19..0000000 --- a/internal/filter/lzw.go +++ /dev/null @@ -1,111 +0,0 @@ -package filter - -import "fmt" - -// decodeLZW decodes a PDF LZW stream. Code widths grow from 9 to 12 bits; -// the early-change flag (default 1) shrinks the threshold at which each -// width step happens. -func decodeLZW(in []byte, p Params) ([]byte, error) { - early := 1 - if p.NoEarlyChange { - early = 0 - } - const ( - clearCode = 256 - eodCode = 257 - ) - br := bitReader{src: in} - codeWidth := 9 - dict := make([][]byte, 258, 4096) - for i := 0; i < 256; i++ { - dict[i] = []byte{byte(i)} - } - - var out []byte - prev := -1 - - resize := func() { - codeWidth = 9 - dict = dict[:258] - } - - for { - code, ok := br.readBits(codeWidth) - if !ok { - break - } - switch { - case code == clearCode: - resize() - prev = -1 - continue - case code == eodCode: - return out, nil - } - - var entry []byte - switch { - case int(code) < len(dict): - entry = dict[code] - case int(code) == len(dict) && prev >= 0: - pe := dict[prev] - entry = make([]byte, len(pe)+1) - copy(entry, pe) - entry[len(pe)] = pe[0] - default: - return nil, fmt.Errorf("LZWDecode: invalid code %d at width %d", code, codeWidth) - } - out = append(out, entry...) - if p.MaxOutput > 0 && int64(len(out)) > p.MaxOutput { - return nil, fmt.Errorf("LZWDecode: decoded output exceeds limit of %d bytes (possible decompression bomb)", p.MaxOutput) - } - if prev >= 0 && len(dict) < 4096 { - pe := dict[prev] - ne := make([]byte, len(pe)+1) - copy(ne, pe) - ne[len(pe)] = entry[0] - dict = append(dict, ne) - } - prev = int(code) - - // Grow width: the new code's index will be len(dict). We need to - // switch when the next code may not fit. - threshold := (1 << uint(codeWidth)) - early - if len(dict) >= threshold && codeWidth < 12 { - codeWidth++ - } - } - return out, nil -} - -type bitReader struct { - src []byte - bytePos int - bitPos uint // 0 = MSB unread - buf uint64 - have uint // number of bits buffered -} - -func (b *bitReader) readBits(n int) (uint32, bool) { - for b.have < uint(n) { - if b.bytePos >= len(b.src) { - if b.have == 0 { - return 0, false - } - // Pad with zeros to flush trailing partial word. - b.buf <<= 8 - b.have += 8 - b.bytePos++ - continue - } - b.buf = (b.buf << 8) | uint64(b.src[b.bytePos]) - b.have += 8 - b.bytePos++ - } - shift := b.have - uint(n) - mask := (uint64(1) << uint(n)) - 1 - v := uint32((b.buf >> shift) & mask) - b.have -= uint(n) - b.buf &= (uint64(1) << b.have) - 1 - return v, true -} diff --git a/internal/lex/lex.go b/internal/lex/lex.go deleted file mode 100644 index 7f46b77..0000000 --- a/internal/lex/lex.go +++ /dev/null @@ -1,430 +0,0 @@ -// Package lex tokenises PDF input. It deals with the lexical layer of -// PDF objects — whitespace, comments, names, numbers, strings, arrays, -// dictionaries, the stream/endstream/obj/endobj/R/null/true/false keywords — -// but does not assemble higher-level structures. The parser layered above -// it turns token streams into Object trees. -package lex - -import ( - "errors" - "fmt" -) - -// Kind identifies a token's lexical category. -type Kind int - -const ( - // EOF marks end of input. - EOF Kind = iota - // Name is a PDF name without the leading slash. - Name - // Integer is a literal integer (no decimal point, optional sign). - Integer - // Real is a literal real number (has a decimal point or 'e' exponent — - // PDF does not actually allow exponents but we accept them). - Real - // LitString is a parenthesised literal string with escapes already - // resolved. - LitString - // HexString is an angle-bracketed hex string with hex pairs already - // decoded to bytes. - HexString - // ArrayStart is the '[' token. - ArrayStart - // ArrayEnd is the ']' token. - ArrayEnd - // DictStart is the '<<' token. - DictStart - // DictEnd is the '>>' token. - DictEnd - // Keyword is any unquoted identifier: true, false, null, obj, endobj, - // stream, endstream, R, xref, trailer, startxref, n, f. - Keyword -) - -func (k Kind) String() string { - switch k { - case EOF: - return "EOF" - case Name: - return "Name" - case Integer: - return "Integer" - case Real: - return "Real" - case LitString: - return "LitString" - case HexString: - return "HexString" - case ArrayStart: - return "[" - case ArrayEnd: - return "]" - case DictStart: - return "<<" - case DictEnd: - return ">>" - case Keyword: - return "Keyword" - } - return fmt.Sprintf("Kind(%d)", int(k)) -} - -// Token is a single lexical unit. Bytes carries the token payload; its -// meaning depends on Kind: -// - Name, Keyword: ASCII name body, no leading slash -// - Integer, Real: literal digits -// - LitString, HexString: decoded bytes -// - ArrayStart, ArrayEnd, DictStart, DictEnd, EOF: empty -type Token struct { - Kind Kind - Bytes []byte - Offset int64 // byte offset in the input where this token started -} - -// Lexer converts a byte slice into a stream of Tokens. It is not safe for -// concurrent use. -type Lexer struct { - src []byte - pos int -} - -// New creates a Lexer over src. The src slice is not copied. -func New(src []byte) *Lexer { - return &Lexer{src: src} -} - -// Pos returns the current byte offset. -func (l *Lexer) Pos() int { return l.pos } - -// SetPos rewinds or fast-forwards the lexer. -func (l *Lexer) SetPos(p int) { l.pos = p } - -// Remaining returns the unread portion of the source. -func (l *Lexer) Remaining() []byte { return l.src[l.pos:] } - -// Source returns the underlying source slice. -func (l *Lexer) Source() []byte { return l.src } - -// ErrUnexpectedEOF indicates that the lexer ran out of bytes mid-token. -var ErrUnexpectedEOF = errors.New("pdfdisassembler/lex: unexpected EOF") - -// IsWhitespace reports whether c is a PDF whitespace character (§7.2.2). -func IsWhitespace(c byte) bool { - switch c { - case 0, '\t', '\n', '\f', '\r', ' ': - return true - } - return false -} - -// IsDelimiter reports whether c is a PDF delimiter character (§7.2.2). -func IsDelimiter(c byte) bool { - switch c { - case '(', ')', '<', '>', '[', ']', '{', '}', '/', '%': - return true - } - return false -} - -// IsRegular reports whether c is a regular character (neither whitespace -// nor delimiter). -func IsRegular(c byte) bool { return !IsWhitespace(c) && !IsDelimiter(c) } - -// SkipWhitespace advances over PDF whitespace and comments. -func (l *Lexer) SkipWhitespace() { - for l.pos < len(l.src) { - c := l.src[l.pos] - if IsWhitespace(c) { - l.pos++ - continue - } - if c == '%' { - // Comment to end of line. - for l.pos < len(l.src) && l.src[l.pos] != '\n' && l.src[l.pos] != '\r' { - l.pos++ - } - continue - } - return - } -} - -// Next returns the next token. At EOF it returns a Token with Kind=EOF. -func (l *Lexer) Next() (Token, error) { - l.SkipWhitespace() - if l.pos >= len(l.src) { - return Token{Kind: EOF, Offset: int64(l.pos)}, nil - } - start := l.pos - c := l.src[l.pos] - - switch { - case c == '/': - return l.readName(start) - case c == '(': - return l.readLiteralString(start) - case c == '<': - if l.pos+1 < len(l.src) && l.src[l.pos+1] == '<' { - l.pos += 2 - return Token{Kind: DictStart, Offset: int64(start)}, nil - } - return l.readHexString(start) - case c == '>': - if l.pos+1 < len(l.src) && l.src[l.pos+1] == '>' { - l.pos += 2 - return Token{Kind: DictEnd, Offset: int64(start)}, nil - } - return Token{}, fmt.Errorf("pdfdisassembler/lex: unexpected '>' at %d", l.pos) - case c == '[': - l.pos++ - return Token{Kind: ArrayStart, Offset: int64(start)}, nil - case c == ']': - l.pos++ - return Token{Kind: ArrayEnd, Offset: int64(start)}, nil - case c == '+' || c == '-' || c == '.' || (c >= '0' && c <= '9'): - return l.readNumber(start) - default: - return l.readKeyword(start) - } -} - -func (l *Lexer) readName(start int) (Token, error) { - l.pos++ // skip '/' - nameStart := l.pos - var buf []byte - for l.pos < len(l.src) { - c := l.src[l.pos] - if IsWhitespace(c) || IsDelimiter(c) { - break - } - if c == '#' { - if buf == nil { - buf = append(buf, l.src[nameStart:l.pos]...) - } - if l.pos+2 >= len(l.src) { - return Token{}, ErrUnexpectedEOF - } - hi, ok1 := hexDigit(l.src[l.pos+1]) - lo, ok2 := hexDigit(l.src[l.pos+2]) - if !ok1 || !ok2 { - return Token{}, fmt.Errorf("pdfdisassembler/lex: invalid #XX escape in name at %d", l.pos) - } - buf = append(buf, byte(hi<<4|lo)) - l.pos += 3 - continue - } - if buf != nil { - buf = append(buf, c) - } - l.pos++ - } - if buf == nil { - buf = l.src[nameStart:l.pos] - } - return Token{Kind: Name, Bytes: buf, Offset: int64(start)}, nil -} - -func (l *Lexer) readLiteralString(start int) (Token, error) { - l.pos++ // skip '(' - depth := 1 - var buf []byte - for l.pos < len(l.src) { - c := l.src[l.pos] - switch c { - case '(': - depth++ - buf = append(buf, c) - l.pos++ - case ')': - depth-- - if depth == 0 { - l.pos++ - return Token{Kind: LitString, Bytes: buf, Offset: int64(start)}, nil - } - buf = append(buf, c) - l.pos++ - case '\\': - if l.pos+1 >= len(l.src) { - return Token{}, ErrUnexpectedEOF - } - next := l.src[l.pos+1] - switch next { - case 'n': - buf = append(buf, '\n') - l.pos += 2 - case 'r': - buf = append(buf, '\r') - l.pos += 2 - case 't': - buf = append(buf, '\t') - l.pos += 2 - case 'b': - buf = append(buf, '\b') - l.pos += 2 - case 'f': - buf = append(buf, '\f') - l.pos += 2 - case '(': - buf = append(buf, '(') - l.pos += 2 - case ')': - buf = append(buf, ')') - l.pos += 2 - case '\\': - buf = append(buf, '\\') - l.pos += 2 - case '\n': - // line continuation - l.pos += 2 - case '\r': - l.pos += 2 - if l.pos < len(l.src) && l.src[l.pos] == '\n' { - l.pos++ - } - case '0', '1', '2', '3', '4', '5', '6', '7': - // Octal: up to 3 digits. - v := 0 - n := 0 - p := l.pos + 1 - for n < 3 && p < len(l.src) { - d := l.src[p] - if d < '0' || d > '7' { - break - } - v = v*8 + int(d-'0') - p++ - n++ - } - buf = append(buf, byte(v&0xFF)) - l.pos = p - default: - // Unknown escape: drop the backslash, keep next byte. - buf = append(buf, next) - l.pos += 2 - } - case '\r': - // CR or CRLF inside literal becomes LF (per spec §7.3.4.2). - buf = append(buf, '\n') - l.pos++ - if l.pos < len(l.src) && l.src[l.pos] == '\n' { - l.pos++ - } - default: - buf = append(buf, c) - l.pos++ - } - } - return Token{}, ErrUnexpectedEOF -} - -func (l *Lexer) readHexString(start int) (Token, error) { - l.pos++ // skip '<' - var buf []byte - var hi int - have := false - for l.pos < len(l.src) { - c := l.src[l.pos] - if c == '>' { - if have { - buf = append(buf, byte(hi<<4)) - } - l.pos++ - return Token{Kind: HexString, Bytes: buf, Offset: int64(start)}, nil - } - if IsWhitespace(c) { - l.pos++ - continue - } - d, ok := hexDigit(c) - if !ok { - return Token{}, fmt.Errorf("pdfdisassembler/lex: invalid hex digit %q at %d", c, l.pos) - } - if have { - buf = append(buf, byte(hi<<4|d)) - have = false - } else { - hi = d - have = true - } - l.pos++ - } - return Token{}, ErrUnexpectedEOF -} - -func (l *Lexer) readNumber(start int) (Token, error) { - isReal := false - p := l.pos - if p < len(l.src) && (l.src[p] == '+' || l.src[p] == '-') { - p++ - } - for p < len(l.src) { - c := l.src[p] - if c == '.' { - isReal = true - p++ - continue - } - if c >= '0' && c <= '9' { - p++ - continue - } - break - } - tok := Token{Bytes: l.src[l.pos:p], Offset: int64(start)} - if isReal { - tok.Kind = Real - } else { - tok.Kind = Integer - } - l.pos = p - return tok, nil -} - -func (l *Lexer) readKeyword(start int) (Token, error) { - p := l.pos - for p < len(l.src) && IsRegular(l.src[p]) { - p++ - } - if p == l.pos { - return Token{}, fmt.Errorf("pdfdisassembler/lex: stuck at byte 0x%02x at %d", l.src[l.pos], l.pos) - } - tok := Token{Kind: Keyword, Bytes: l.src[l.pos:p], Offset: int64(start)} - l.pos = p - return tok, nil -} - -func hexDigit(c byte) (int, bool) { - switch { - case c >= '0' && c <= '9': - return int(c - '0'), true - case c >= 'a' && c <= 'f': - return int(c-'a') + 10, true - case c >= 'A' && c <= 'F': - return int(c-'A') + 10, true - } - return 0, false -} - -// ReadStreamData consumes raw stream bytes of the given length, starting -// at the current position. It honours the spec's EOL handling: a single -// LF or CRLF *immediately* after the "stream" keyword is part of the -// keyword line, not the stream content. Callers should call this after -// the "stream" keyword token has been consumed. -func (l *Lexer) ReadStreamData(length int) ([]byte, error) { - // Skip optional CR LF or single LF following "stream". - if l.pos < len(l.src) && l.src[l.pos] == '\r' { - l.pos++ - } - if l.pos < len(l.src) && l.src[l.pos] == '\n' { - l.pos++ - } - // length is the attacker-controlled /Length; compare against the bytes - // remaining rather than computing l.pos+length, which can overflow. - if length < 0 || length > len(l.src)-l.pos { - return nil, ErrUnexpectedEOF - } - out := l.src[l.pos : l.pos+length] - l.pos += length - return out, nil -} diff --git a/internal/lex/lex_test.go b/internal/lex/lex_test.go deleted file mode 100644 index 0d551ab..0000000 --- a/internal/lex/lex_test.go +++ /dev/null @@ -1,285 +0,0 @@ -package lex - -import ( - "bytes" - "testing" -) - -func TestLexerNamesAndNumbers(t *testing.T) { - src := []byte("/Length 12 -3 +4 5.6 .7 8. true false null") - lx := New(src) - want := []struct { - kind Kind - bytes string - }{ - {Name, "Length"}, - {Integer, "12"}, - {Integer, "-3"}, - {Integer, "+4"}, - {Real, "5.6"}, - {Real, ".7"}, - {Real, "8."}, - {Keyword, "true"}, - {Keyword, "false"}, - {Keyword, "null"}, - {EOF, ""}, - } - for i, w := range want { - tok, err := lx.Next() - if err != nil { - t.Fatalf("tok %d: err %v", i, err) - } - if tok.Kind != w.kind { - t.Fatalf("tok %d: kind %v want %v", i, tok.Kind, w.kind) - } - if string(tok.Bytes) != w.bytes { - t.Fatalf("tok %d: bytes %q want %q", i, tok.Bytes, w.bytes) - } - } -} - -func TestLexerNameHashEscape(t *testing.T) { - src := []byte("/A#20B /ABC") - lx := New(src) - t1, _ := lx.Next() - if string(t1.Bytes) != "A B" { - t.Fatalf("got %q want %q", t1.Bytes, "A B") - } - t2, _ := lx.Next() - if string(t2.Bytes) != "ABC" { - t.Fatalf("got %q want %q", t2.Bytes, "ABC") - } -} - -func TestLexerLiteralString(t *testing.T) { - cases := []struct { - in string - want string - }{ - {"(hello)", "hello"}, - {"(a (nested) b)", "a (nested) b"}, - {"(line\\nbreak)", "line\nbreak"}, - {"(\\053\\053)", "++"}, - {"(\\\\)", "\\"}, - {"(a\\\nb)", "ab"}, - } - for _, c := range cases { - lx := New([]byte(c.in)) - tok, err := lx.Next() - if err != nil { - t.Fatalf("%q: %v", c.in, err) - } - if tok.Kind != LitString { - t.Fatalf("%q: kind %v", c.in, tok.Kind) - } - if string(tok.Bytes) != c.want { - t.Fatalf("%q: got %q want %q", c.in, tok.Bytes, c.want) - } - } -} - -func TestLexerHexString(t *testing.T) { - src := []byte("<48656C6C6F>") - tok, err := New(src).Next() - if err != nil { - t.Fatal(err) - } - if tok.Kind != HexString { - t.Fatalf("kind %v", tok.Kind) - } - if string(tok.Bytes) != "Hello" { - t.Fatalf("got %q", tok.Bytes) - } -} - -func TestLexerHexStringOddNibble(t *testing.T) { - src := []byte("") - tok, err := New(src).Next() - if err != nil { - t.Fatal(err) - } - if len(tok.Bytes) != 1 || tok.Bytes[0] != 0xF0 { - t.Fatalf("got % x", tok.Bytes) - } -} - -func TestLexerDictArrayDelims(t *testing.T) { - src := []byte("<< /A 1 >> [ 1 2 3 ]") - lx := New(src) - kinds := []Kind{DictStart, Name, Integer, DictEnd, ArrayStart, Integer, Integer, Integer, ArrayEnd, EOF} - for i, k := range kinds { - tok, err := lx.Next() - if err != nil { - t.Fatalf("tok %d: %v", i, err) - } - if tok.Kind != k { - t.Fatalf("tok %d: kind %v want %v", i, tok.Kind, k) - } - } -} - -func TestLexerComment(t *testing.T) { - src := []byte("% comment\n1 % trailing\n2") - lx := New(src) - t1, _ := lx.Next() - t2, _ := lx.Next() - t3, _ := lx.Next() - if t1.Kind != Integer || string(t1.Bytes) != "1" { - t.Fatalf("t1: %v %q", t1.Kind, t1.Bytes) - } - if t2.Kind != Integer || string(t2.Bytes) != "2" { - t.Fatalf("t2: %v %q", t2.Kind, t2.Bytes) - } - if t3.Kind != EOF { - t.Fatalf("t3 kind %v", t3.Kind) - } -} - -func TestReadStreamDataHostileLength(t *testing.T) { - const maxInt = int(^uint(0) >> 1) - for _, length := range []int{-1, -1000, maxInt} { - // Leading "\n" makes the EOL skip advance pos, so pos+length overflows. - l := New([]byte("\nstream body bytes")) - if _, err := l.ReadStreamData(length); err == nil { - t.Fatalf("length=%d: expected error, got nil", length) - } - } -} - -func TestReadStreamDataValid(t *testing.T) { - l := New([]byte("\nABCDEF")) - out, err := l.ReadStreamData(6) - if err != nil { - t.Fatalf("ReadStreamData: %v", err) - } - if string(out) != "ABCDEF" { - t.Fatalf("got %q want ABCDEF", out) - } -} - -func TestLexerTruncatedTokensError(t *testing.T) { - cases := map[string]string{ - "name escape at eof": "/AB#", - "name escape one digit": "/AB#F", - "name escape bad hex": "/A#GG", - "string backslash at eof": "(abc\\", - "unterminated string": "(abc", - "unterminated hex": "<48", - "bad hex digit": "<4G>", - } - for name, src := range cases { - t.Run(name, func(t *testing.T) { - if _, err := New([]byte(src)).Next(); err == nil { - t.Fatalf("%q: expected error, got nil", src) - } - }) - } -} - -func TestLexerStringEscapes(t *testing.T) { - lx := New([]byte(`(\101\n\)\(end)`)) // octal 'A', newline, literal ) and ( - tok, err := lx.Next() - if err != nil { - t.Fatalf("Next: %v", err) - } - if want := "A\n)(end"; string(tok.Bytes) != want { - t.Fatalf("got %q want %q", tok.Bytes, want) - } -} - -// FuzzLexer asserts tokenisation never panics on arbitrary input. -func FuzzLexer(f *testing.F) { - f.Add([]byte("/Name 123 -4.5 (str\\n) <48656C> << /A 1 >> [ 1 2 ] true null %c")) - f.Fuzz(func(t *testing.T, data []byte) { - lx := New(data) - for i := 0; i <= len(data); i++ { - tok, err := lx.Next() - if err != nil || tok.Kind == EOF { - break - } - } - }) -} - -// Position accessors that back the xref reader's manual seeking. -func TestLexerAccessors(t *testing.T) { - src := []byte("hello world") - lx := New(src) - if lx.Pos() != 0 { - t.Errorf("initial Pos = %d, want 0", lx.Pos()) - } - if !bytes.Equal(lx.Source(), src) { - t.Errorf("Source = %q, want %q", lx.Source(), src) - } - lx.SetPos(6) - if lx.Pos() != 6 { - t.Errorf("Pos after SetPos(6) = %d, want 6", lx.Pos()) - } - if got := lx.Remaining(); string(got) != "world" { - t.Errorf("Remaining = %q, want world", got) - } -} - -// Literal-string escapes/newlines (§7.3.4.2) not already in TestLexerLiteralString. -func TestLexerLiteralStringEscapes(t *testing.T) { - cases := []struct { - name, in, want string - }{ - {"named_escapes", "(\\t\\b\\f\\r)", "\t\b\f\r"}, - {"cr_line_continuation", "(a\\\rb)", "ab"}, - {"crlf_line_continuation", "(a\\\r\nb)", "ab"}, - {"unknown_escape_keeps_char", "(\\q)", "q"}, - {"bare_cr_to_lf", "(a\rb)", "a\nb"}, - {"bare_crlf_to_lf", "(a\r\nb)", "a\nb"}, - } - for _, c := range cases { - t.Run(c.name, func(t *testing.T) { - tok, err := New([]byte(c.in)).Next() - if err != nil { - t.Fatalf("Next: %v", err) - } - if tok.Kind != LitString { - t.Fatalf("kind = %v, want LitString", tok.Kind) - } - if string(tok.Bytes) != c.want { - t.Errorf("got %q, want %q", tok.Bytes, c.want) - } - }) - } -} - -func TestLexerLiteralStringErrors(t *testing.T) { - for _, in := range []string{"(\\", "(abc"} { // trailing backslash; unterminated - if _, err := New([]byte(in)).Next(); err == nil { - t.Errorf("%q: expected an error, got nil", in) - } - } -} - -// A misclassified delimiter byte (§7.2.2) would break tokenisation, so pin the set. -func TestIsDelimiter(t *testing.T) { - for _, c := range []byte("()<>[]{}/%") { - if !IsDelimiter(c) { - t.Errorf("IsDelimiter(%q) = false, want true", c) - } - } - for _, c := range []byte("aZ0 \t.\\") { - if IsDelimiter(c) { - t.Errorf("IsDelimiter(%q) = true, want false", c) - } - } -} - -func TestHexDigit(t *testing.T) { - valid := map[byte]int{'0': 0, '9': 9, 'a': 10, 'f': 15, 'A': 10, 'F': 15} - for c, want := range valid { - if v, ok := hexDigit(c); !ok || v != want { - t.Errorf("hexDigit(%q) = (%d, %v), want (%d, true)", c, v, ok, want) - } - } - for _, c := range []byte("gG/ \x00") { - if _, ok := hexDigit(c); ok { - t.Errorf("hexDigit(%q) = ok, want not-ok", c) - } - } -} diff --git a/object.go b/object.go deleted file mode 100644 index c2aba1c..0000000 --- a/object.go +++ /dev/null @@ -1,335 +0,0 @@ -package pdfdisassembler - -import ( - "iter" - "time" -) - -// Object is the sealed PDF object type. Concrete variants: -// -// Name, Integer, Real, Bool, String, *Dict, Array, *Stream, Reference, Null -// -// Callers should type-switch or type-assert on the concrete type when they -// need a specific value. The Resolve* helpers on Reader perform the common -// dereference-and-assert pattern. -type Object interface { - object() -} - -// Name is a PDF name object, e.g. /Length, /Type. The leading slash is not -// stored. -type Name string - -// Integer is a PDF integer object. -type Integer int64 - -// Real is a PDF real-number object. -type Real float64 - -// Bool is a PDF boolean object. -type Bool bool - -// String is a PDF string object after parsing. The parser strips the -// literal-string parentheses or hex-string angle brackets and decodes -// escape sequences and hexadecimal pairs, but does not re-encode the -// bytes — they are whatever the producer wrote. -// -// For PDF text strings ("Title", "Subject", "Producer", and so on) use -// Dict.String or DocumentInfo, which apply the text-string decoding rules -// (PDFDocEncoding, UTF-16BE BOM, UTF-8 BOM). -type String []byte - -// Array is a PDF array of objects. -type Array []Object - -// Reference is an indirect object reference (e.g. "12 0 R"). -type Reference struct { - Number int - Generation int -} - -// Null is the PDF null object. -type Null struct{} - -func (Name) object() {} -func (Integer) object() {} -func (Real) object() {} -func (Bool) object() {} -func (String) object() {} -func (Array) object() {} -func (Reference) object() {} -func (Null) object() {} -func (*Dict) object() {} -func (*Stream) object() {} - -// Dict is a PDF dictionary that preserves insertion order during iteration. -type Dict struct { - keys []string - values map[string]Object - // reader is set by the parser so the dereferencing convenience methods - // (Dict, Array-of-dicts walks) can follow indirect references. nil - // when the dictionary was synthesised outside of a Reader. - reader *Reader -} - -// newDict returns an empty dictionary tied to r (may be nil during early -// parsing of trailer/xref dicts). -func newDict(r *Reader) *Dict { - return &Dict{values: map[string]Object{}, reader: r} -} - -// Len returns the number of entries in the dictionary. -func (d *Dict) Len() int { - if d == nil { - return 0 - } - return len(d.keys) -} - -// Get returns the raw object for key. The returned object may be a -// Reference; use Dict.Dict / Reader.Resolve if you need the resolved value. -func (d *Dict) Get(key string) (Object, bool) { - if d == nil { - return nil, false - } - v, ok := d.values[key] - return v, ok -} - -// Has reports whether key is present. -func (d *Dict) Has(key string) bool { - if d == nil { - return false - } - _, ok := d.values[key] - return ok -} - -// Keys returns the dictionary keys in insertion order. -func (d *Dict) Keys() []string { - if d == nil { - return nil - } - out := make([]string, len(d.keys)) - copy(out, d.keys) - return out -} - -// Iter returns an iterator over key/value pairs in insertion order. -func (d *Dict) Iter() iter.Seq2[string, Object] { - return func(yield func(string, Object) bool) { - if d == nil { - return - } - for _, k := range d.keys { - if !yield(k, d.values[k]) { - return - } - } - } -} - -// set inserts or updates an entry, preserving insertion order on first set. -func (d *Dict) set(key string, value Object) { - if _, ok := d.values[key]; !ok { - d.keys = append(d.keys, key) - } - d.values[key] = value -} - -// Name returns the Name value at key. If the value is a Reference, it is -// resolved first. -func (d *Dict) Name(key string) (Name, bool) { - v, ok := d.resolved(key) - if !ok { - return "", false - } - n, ok := v.(Name) - return n, ok -} - -// Int returns the Integer value at key. If the value is a Reference, it is -// resolved first. -func (d *Dict) Int(key string) (int64, bool) { - v, ok := d.resolved(key) - if !ok { - return 0, false - } - n, ok := v.(Integer) - if !ok { - return 0, false - } - return int64(n), true -} - -// Bool returns the Bool value at key. If the value is a Reference, it is -// resolved first. -func (d *Dict) Bool(key string) (bool, bool) { - v, ok := d.resolved(key) - if !ok { - return false, false - } - b, ok := v.(Bool) - return bool(b), ok -} - -// Array returns the Array value at key. If the value is a Reference, it is -// resolved first. -func (d *Dict) Array(key string) (Array, bool) { - v, ok := d.resolved(key) - if !ok { - return nil, false - } - a, ok := v.(Array) - return a, ok -} - -// Dict returns the *Dict value at key. If the value is a Reference, it is -// resolved first. -func (d *Dict) Dict(key string) (*Dict, bool) { - v, ok := d.resolved(key) - if !ok { - return nil, false - } - dd, ok := v.(*Dict) - return dd, ok -} - -// Stream returns the *Stream value at key. If the value is a Reference, it -// is resolved first. -func (d *Dict) Stream(key string) (*Stream, bool) { - v, ok := d.resolved(key) - if !ok { - return nil, false - } - s, ok := v.(*Stream) - return s, ok -} - -// String returns the value at key as a Go string, decoded according to the -// PDF text-string rules (UTF-16BE BOM, UTF-8 BOM, otherwise PDFDocEncoding). -// If the value is a Reference, it is resolved first. -func (d *Dict) String(key string) (string, bool) { - v, ok := d.resolved(key) - if !ok { - return "", false - } - s, ok := v.(String) - if !ok { - return "", false - } - return decodeTextString(s), true -} - -// Bytes returns the raw bytes of a String value at key, without -// text-string decoding. Useful for byte strings (file identifiers, hashes). -// If the value is a Reference, it is resolved first. -func (d *Dict) Bytes(key string) ([]byte, bool) { - v, ok := d.resolved(key) - if !ok { - return nil, false - } - s, ok := v.(String) - if !ok { - return nil, false - } - return []byte(s), true -} - -// resolved returns the value at key, dereferencing once through the -// reader if the value is a Reference. If resolution fails, ok is false. -func (d *Dict) resolved(key string) (Object, bool) { - v, ok := d.Get(key) - if !ok { - return nil, false - } - if ref, ok := v.(Reference); ok { - if d.reader == nil { - return nil, false - } - obj, err := d.reader.Resolve(ref) - if err != nil { - return nil, false - } - return obj, true - } - return v, true -} - -// Stream is a stream object. The decoded content is produced by Content, -// which applies the declared filter chain (FlateDecode, ASCII85, …) and -// any document-level decryption, and caches the result. -type Stream struct { - // Dict is the stream's parameter dictionary, e.g. /Length, /Filter. - Dict *Dict - // reader is the document the stream came from. - reader *Reader - // rawOffset is the byte offset in the underlying ReadSeeker where the - // raw stream bytes begin (just after the "stream" keyword and EOL). - rawOffset int64 - // rawLength is the raw byte count of the stream as declared by /Length. - rawLength int64 - // objNumber, objGeneration identify the indirect object this stream - // belongs to. Used for per-object decryption keys. - objNumber int - objGeneration int - // cache holds the decoded content after the first call to Content. - cache []byte - cacheErr error - cached bool -} - -// Content returns the decoded stream bytes. Filters and decryption are -// applied on the first call and the result is cached for subsequent calls. -func (s *Stream) Content() ([]byte, error) { - if s.cached { - return s.cache, s.cacheErr - } - b, err := s.reader.decodeStream(s) - s.cache = b - s.cacheErr = err - s.cached = true - return b, err -} - -// RawLength returns the declared raw byte length of the stream. -func (s *Stream) RawLength() int64 { - return s.rawLength -} - -// RawBytes returns a copy of the stream's raw, undecoded bytes exactly as they -// appear in the file — before any filter or decryption is applied. For the -// decoded content, use Content. -func (s *Stream) RawBytes() ([]byte, error) { - return s.reader.rawStreamBytes(s) -} - -// ObjectEntry is yielded by Reader.Objects: an in-use indirect object plus -// its resolved value. -type ObjectEntry struct { - Reference Reference - Object Object -} - -// DocInfo is a value snapshot of the standard /Info dictionary entries. -// Missing entries are zero values. Custom carries any non-standard keys -// as raw decoded strings. -type DocInfo struct { - Title string - Author string - Subject string - Keywords string - Creator string - Producer string - CreationDate time.Time - ModDate time.Time - Custom map[string]string -} - -// EmbeddedFile is one entry from the catalog's EmbeddedFiles name tree (a PDF -// attachment). Spec is the /Filespec dictionary; its /EF stream holds the -// bytes. -type EmbeddedFile struct { - Name string - Spec *Dict -} diff --git a/object_test.go b/object_test.go deleted file mode 100644 index 99eb64e..0000000 --- a/object_test.go +++ /dev/null @@ -1,179 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "testing" -) - -func TestDictAccessors(t *testing.T) { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R /IntVal 42 /NegInt -7 /BoolVal true " + - "/StrVal (hi) /NameVal /Foo /ArrVal [ 1 2 3 ] /DictRef 3 0 R >>", - "<< /Type /Pages /Kids [] /Count 0 >>", - "<< /Inner (deep) >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - cat, err := r.Catalog() - if err != nil { - t.Fatalf("Catalog: %v", err) - } - - if v, ok := cat.Int("IntVal"); !ok || v != 42 { - t.Errorf("Int(IntVal) = %d, %v", v, ok) - } - if v, ok := cat.Int("NegInt"); !ok || v != -7 { - t.Errorf("Int(NegInt) = %d, %v", v, ok) - } - if v, ok := cat.Bool("BoolVal"); !ok || !v { - t.Errorf("Bool(BoolVal) = %v, %v", v, ok) - } - if v, ok := cat.String("StrVal"); !ok || v != "hi" { - t.Errorf("String(StrVal) = %q, %v", v, ok) - } - if v, ok := cat.Bytes("StrVal"); !ok || string(v) != "hi" { - t.Errorf("Bytes(StrVal) = %q, %v", v, ok) - } - if v, ok := cat.Name("NameVal"); !ok || v != "Foo" { - t.Errorf("Name(NameVal) = %q, %v", v, ok) - } - if v, ok := cat.Array("ArrVal"); !ok || len(v) != 3 { - t.Errorf("Array(ArrVal) len = %d, %v", len(v), ok) - } - - // Dict() follows the indirect reference to object 3. - inner, ok := cat.Dict("DictRef") - if !ok { - t.Fatal("Dict(DictRef) not resolved") - } - if s, ok := inner.String("Inner"); !ok || s != "deep" { - t.Errorf("resolved inner String(Inner) = %q, %v", s, ok) - } - - if !cat.Has("IntVal") || cat.Has("Missing") { - t.Error("Has wrong") - } - if cat.Len() < 8 { - t.Errorf("Len = %d, want >= 8", cat.Len()) - } - seen := map[string]bool{} - for _, k := range cat.Keys() { - seen[k] = true - } - for k := range cat.Iter() { - if !seen[k] { - t.Errorf("Iter yielded %q absent from Keys", k) - } - } - if !seen["IntVal"] || !seen["ArrVal"] { - t.Errorf("Keys missing entries: %v", cat.Keys()) - } - - // Type mismatches must report ok=false, not panic or coerce. - if _, ok := cat.Int("StrVal"); ok { - t.Error("Int on a string") - } - if _, ok := cat.Bool("IntVal"); ok { - t.Error("Bool on an int") - } - if _, ok := cat.Name("IntVal"); ok { - t.Error("Name on an int") - } - if _, ok := cat.Array("IntVal"); ok { - t.Error("Array on an int") - } - if _, ok := cat.Dict("IntVal"); ok { - t.Error("Dict on an int") - } - if _, ok := cat.Stream("IntVal"); ok { - t.Error("Stream on an int") - } - if _, ok := cat.String("IntVal"); ok { - t.Error("String on an int") - } - if _, ok := cat.Bytes("IntVal"); ok { - t.Error("Bytes on an int") - } - if _, ok := cat.Int("Missing"); ok { - t.Error("Int on a missing key") - } -} - -// Every accessor must be safe on a nil *Dict (the common "key absent" result). -func TestDictNilReceiver(t *testing.T) { - var d *Dict - if d.Len() != 0 { - t.Error("Len") - } - if d.Has("x") { - t.Error("Has") - } - if d.Keys() != nil { - t.Error("Keys") - } - if _, ok := d.Get("x"); ok { - t.Error("Get") - } - for range d.Iter() { - t.Error("Iter on nil yielded an entry") - } -} - -// The typed getters must miss (ok=false) — not coerce or fabricate a zero — on -// a missing key, a wrong type, or an unresolvable Reference (nil reader). -func TestDictTypedGetterMisses(t *testing.T) { - d := newDict(nil) - d.set("i", Integer(5)) - d.set("b", Bool(true)) - d.set("s", String("hi")) - - wrongType := []struct { - name string - ok bool - }{ - {"Bool", func() bool { _, ok := d.Bool("i"); return ok }()}, - {"Int", func() bool { _, ok := d.Int("b"); return ok }()}, - {"Array", func() bool { _, ok := d.Array("i"); return ok }()}, - {"Dict", func() bool { _, ok := d.Dict("i"); return ok }()}, - {"Stream", func() bool { _, ok := d.Stream("i"); return ok }()}, - {"String", func() bool { _, ok := d.String("i"); return ok }()}, - {"Bytes", func() bool { _, ok := d.Bytes("i"); return ok }()}, - } - for _, tc := range wrongType { - if tc.ok { - t.Errorf("%s on a wrong-typed value should miss", tc.name) - } - } - - missing := []bool{ - func() bool { _, ok := d.Int("x"); return ok }(), - func() bool { _, ok := d.Bool("x"); return ok }(), - func() bool { _, ok := d.Array("x"); return ok }(), - func() bool { _, ok := d.Dict("x"); return ok }(), - func() bool { _, ok := d.Stream("x"); return ok }(), - func() bool { _, ok := d.String("x"); return ok }(), - func() bool { _, ok := d.Bytes("x"); return ok }(), - } - for i, ok := range missing { - if ok { - t.Errorf("getter %d on a missing key should miss", i) - } - } - - // A Reference with no backing reader can't be dereferenced. - d.set("ref", Reference{Number: 9}) - if _, ok := d.Int("ref"); ok { - t.Error("Reference with nil reader should miss") - } - - // Controls: the right type resolves. - if v, ok := d.Int("i"); !ok || v != 5 { - t.Errorf("Int(i) = %d, %v; want 5, true", v, ok) - } - if s, ok := d.Bytes("s"); !ok || string(s) != "hi" { - t.Errorf("Bytes(s) = %q, %v; want hi, true", s, ok) - } -} diff --git a/page.go b/page.go deleted file mode 100644 index cb47a7c..0000000 --- a/page.go +++ /dev/null @@ -1,360 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "errors" - "fmt" -) - -// maxPageTreeDepth bounds both the page-tree (/Kids) descent and the -// inheritance (/Parent) walk so a hostile or cyclic structure can't loop -// forever or overflow the stack. -const maxPageTreeDepth = 1000 - -// BoxName identifies one of the page boundary boxes (PDF 32000-1:2008 -// §14.11.2). -type BoxName string - -// The five page boundary boxes, in the spec's containment order (MediaBox is -// the largest, ArtBox the smallest). -const ( - MediaBox BoxName = "MediaBox" - CropBox BoxName = "CropBox" - BleedBox BoxName = "BleedBox" - TrimBox BoxName = "TrimBox" - ArtBox BoxName = "ArtBox" -) - -// boxNames is the canonical box list iterated by Page.Boxes. -var boxNames = []BoxName{MediaBox, CropBox, BleedBox, TrimBox, ArtBox} - -// Rect is a PDF rectangle in default user-space units (points), normalised so -// LLX <= URX and LLY <= URY regardless of the corner order written in the file. -type Rect struct { - LLX, LLY, URX, URY float64 -} - -// Width returns the rectangle's horizontal extent. -func (r Rect) Width() float64 { return r.URX - r.LLX } - -// Height returns the rectangle's vertical extent. -func (r Rect) Height() float64 { return r.URY - r.LLY } - -// Page is a handle to a single leaf page (/Type /Page) of the page tree. Its -// accessors resolve the inheritable attributes — boxes, /Rotate, /Resources — -// by walking the /Parent chain per PDF 32000-1:2008 §7.7.3.4. Obtain one via -// Reader.Page or Reader.Pages. -type Page struct { - reader *Reader - dict *Dict - index int -} - -// Index returns the page's 0-based position in display order. -func (p *Page) Index() int { return p.index } - -// Dict returns the page's own dictionary, without inherited attributes -// flattened in. Use the Box, Rotation, and Resources accessors for values that -// may be inherited from an ancestor /Pages node. -func (p *Page) Dict() *Dict { return p.dict } - -// PageCount returns the number of leaf pages in the document. -func (r *Reader) PageCount() (int, error) { - pages, err := r.loadPages() - if err != nil { - return 0, err - } - return len(pages), nil -} - -// Pages returns every leaf page in display (reading) order. -func (r *Reader) Pages() ([]*Page, error) { - pages, err := r.loadPages() - if err != nil { - return nil, err - } - out := make([]*Page, len(pages)) - copy(out, pages) - return out, nil -} - -// Page returns the leaf page at the given 0-based index in display order. -func (r *Reader) Page(index int) (*Page, error) { - pages, err := r.loadPages() - if err != nil { - return nil, err - } - if index < 0 || index >= len(pages) { - return nil, fmt.Errorf("pdfdisassembler: page index %d out of range (%d pages)", index, len(pages)) - } - return pages[index], nil -} - -// loadPages walks the page tree once and caches the flat leaf-page list (and -// any error) for subsequent calls. -func (r *Reader) loadPages() ([]*Page, error) { - if r.pagesLoaded { - return r.pages, r.pagesErr - } - r.pagesLoaded = true - r.pages, r.pagesErr = r.buildPages() - return r.pages, r.pagesErr -} - -func (r *Reader) buildPages() ([]*Page, error) { - cat, err := r.Catalog() - if err != nil { - return nil, err - } - rootRef, ok := cat.Get("Pages") - if !ok { - return nil, errors.New("pdfdisassembler: catalog has no /Pages") - } - root, err := r.ResolveDict(rootRef) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: resolve /Pages: %w", err) - } - seen := map[Reference]struct{}{} - if ref, ok := rootRef.(Reference); ok { - seen[ref] = struct{}{} - } - var out []*Page - r.collectPages(root, seen, 0, &out) - return out, nil -} - -// collectPages descends the page tree depth-first, appending each leaf page to -// out in display order. A node is treated as an intermediate /Pages node when -// it carries /Kids, otherwise as a leaf /Page — /Type is only a hint, since -// some producers omit it. seen guards against cyclic /Kids references and depth -// bounds the descent. -func (r *Reader) collectPages(node *Dict, seen map[Reference]struct{}, depth int, out *[]*Page) { - if node == nil || depth > maxPageTreeDepth { - return - } - kids, ok := node.Array("Kids") - if !ok { - // No resolvable /Kids array. A node that still looks like an - // intermediate /Pages node — it declares /Type /Pages, /Count, or a - // /Kids key that failed to resolve — is a broken or empty branch, not a - // leaf page; emitting it would invent a phantom page. - if node.Has("Kids") || node.Has("Count") { - return - } - if t, ok := node.Name("Type"); ok && t == "Pages" { - return - } - *out = append(*out, &Page{reader: r, dict: node, index: len(*out)}) - return - } - for _, kid := range kids { - if ref, ok := kid.(Reference); ok { - if _, dup := seen[ref]; dup { - continue - } - seen[ref] = struct{}{} - } - child, err := r.ResolveDict(kid) - if err != nil { - continue - } - r.collectPages(child, seen, depth+1, out) - } -} - -// inherited walks the /Parent chain starting at the page dictionary and returns -// the resolved value of the first ancestor that carries key. ok is false when -// no ancestor defines it. Resolution is cached, so the same /Parent reference -// yields the same *Dict pointer — pointer identity (plus the depth bound) -// terminates a cyclic chain. -func (p *Page) inherited(key string) (Object, bool) { - seen := map[*Dict]struct{}{} - node := p.dict - for depth := 0; node != nil && depth <= maxPageTreeDepth; depth++ { - if _, dup := seen[node]; dup { - return nil, false - } - seen[node] = struct{}{} - if v, ok := node.resolved(key); ok { - return v, true - } - parent, ok := node.Dict("Parent") - if !ok { - return nil, false - } - node = parent - } - return nil, false -} - -// Box returns the named page boundary box. MediaBox and CropBox are resolved -// through inheritance along the /Parent chain; BleedBox, TrimBox and ArtBox are -// read from the page object only, since only MediaBox and CropBox carry the -// (Inheritable) marker in PDF 32000-1:2008 Table 30 (§14.11.2). ok is false -// when the box is not defined there, or its value is not a well-formed array of -// four numbers. No spec-default substitution (e.g. a missing box defaulting to -// CropBox) is applied. -func (p *Page) Box(name BoxName) (Rect, bool) { - var v Object - var ok bool - if name == MediaBox || name == CropBox { - v, ok = p.inherited(string(name)) - } else { - v, ok = p.dict.resolved(string(name)) - } - if !ok { - return Rect{}, false - } - arr, ok := v.(Array) - if !ok { - return Rect{}, false - } - return rectFromArray(p.reader, arr) -} - -// Boxes returns every boundary box defined for the page, keyed by name, with -// the same per-box inheritance rules as Box (MediaBox and CropBox may be -// inherited from an ancestor; BleedBox/TrimBox/ArtBox are page-level only). -// Boxes that are not defined are omitted; no spec-default substitution (e.g. a -// missing CropBox defaulting to MediaBox) is applied. -func (p *Page) Boxes() map[BoxName]Rect { - out := map[BoxName]Rect{} - for _, name := range boxNames { - if rect, ok := p.Box(name); ok { - out[name] = rect - } - } - return out -} - -// Rotation returns the page's clockwise display rotation in degrees, resolved -// through inheritance and normalised to one of 0, 90, 180, 270. A missing, -// non-integer, or non-multiple-of-90 /Rotate yields 0. -func (p *Page) Rotation() int { - v, ok := p.inherited("Rotate") - if !ok { - return 0 - } - n, ok := v.(Integer) - if !ok { - return 0 - } - deg := int(n) % 360 - if deg < 0 { - deg += 360 - } - if deg%90 != 0 { - return 0 - } - return deg -} - -// Resources returns the page's resource dictionary, resolved through -// inheritance along the /Parent chain. ok is false when neither the page nor -// any ancestor defines /Resources. -func (p *Page) Resources() (*Dict, bool) { - v, ok := p.inherited("Resources") - if !ok { - return nil, false - } - d, ok := v.(*Dict) - return d, ok -} - -// ContentStreams returns the page's content streams in order. /Contents may be -// a single stream or an array of streams (PDF 32000-1:2008 §7.7.3.3); both are -// flattened to the underlying Stream objects. It returns nil (no error) when -// the page has no /Contents. Non-stream entries are skipped defensively. -func (p *Page) ContentStreams() ([]*Stream, error) { - v, ok := p.dict.Get("Contents") - if !ok { - return nil, nil - } - resolved, err := p.reader.Resolve(v) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: resolve /Contents: %w", err) - } - var out []*Stream - switch t := resolved.(type) { - case *Stream: - out = append(out, t) - case Array: - for _, e := range t { - s, err := p.reader.Resolve(e) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: resolve /Contents entry: %w", err) - } - stm, ok := s.(*Stream) - if !ok { - // A missing entry resolves to Null, not an error; dropping it - // would silently truncate the page's drawing instructions, so - // fail loudly instead. - return nil, fmt.Errorf("pdfdisassembler: /Contents array entry resolved to %T, want stream", s) - } - out = append(out, stm) - } - } - return out, nil -} - -// Content returns the page's decoded content: every content stream decoded via -// its filter chain and concatenated with a single newline between streams (a -// token cannot span a stream boundary, per PDF 32000-1:2008 §7.8.2). The result -// is the raw drawing-instruction byte stream; this library does not interpret -// it (see package contentstream for tokenisation). -func (p *Page) Content() ([]byte, error) { - streams, err := p.ContentStreams() - if err != nil { - return nil, err - } - parts := make([][]byte, 0, len(streams)) - for _, s := range streams { - data, err := s.Content() - if err != nil { - return nil, err - } - parts = append(parts, data) - } - return bytes.Join(parts, []byte{'\n'}), nil -} - -// rectFromArray converts a 4-element PDF array [llx lly urx ury] to a -// normalised Rect. Entries may be Integer or Real and may be indirect -// references. ok is false for any other shape. -func rectFromArray(r *Reader, arr Array) (Rect, bool) { - if len(arr) != 4 { - return Rect{}, false - } - var v [4]float64 - for i, e := range arr { - f, ok := numberValue(r, e) - if !ok { - return Rect{}, false - } - v[i] = f - } - rect := Rect{LLX: v[0], LLY: v[1], URX: v[2], URY: v[3]} - if rect.LLX > rect.URX { - rect.LLX, rect.URX = rect.URX, rect.LLX - } - if rect.LLY > rect.URY { - rect.LLY, rect.URY = rect.URY, rect.LLY - } - return rect, true -} - -// numberValue resolves obj and returns it as a float64 if it is an Integer or -// Real. -func numberValue(r *Reader, obj Object) (float64, bool) { - resolved, err := r.Resolve(obj) - if err != nil { - return 0, false - } - switch n := resolved.(type) { - case Integer: - return float64(n), true - case Real: - return float64(n), true - } - return 0, false -} diff --git a/page_test.go b/page_test.go deleted file mode 100644 index 4d03986..0000000 --- a/page_test.go +++ /dev/null @@ -1,399 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "path/filepath" - "testing" -) - -func TestPageInheritanceFixture(t *testing.T) { - r, err := OpenFile(filepath.Join("testdata", "fixtures", "page-inheritance", "input.pdf")) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - - n, err := r.PageCount() - if err != nil { - t.Fatalf("PageCount: %v", err) - } - if n != 2 { - t.Fatalf("PageCount = %d, want 2", n) - } - - // Page 0 inherits everything: MediaBox/Rotate/Resources two levels up - // (the /Pages root), CropBox one level up. - p0, err := r.Page(0) - if err != nil { - t.Fatalf("Page(0): %v", err) - } - if box, ok := p0.Box(MediaBox); !ok || box != (Rect{0, 0, 612, 792}) { - t.Errorf("page 0 MediaBox = %+v ok=%v, want {0 0 612 792}", box, ok) - } - if box, ok := p0.Box(CropBox); !ok || box != (Rect{10, 10, 602, 782}) { - t.Errorf("page 0 CropBox = %+v ok=%v, want {10 10 602 782}", box, ok) - } - if rot := p0.Rotation(); rot != 90 { - t.Errorf("page 0 Rotation = %d, want 90", rot) - } - res, ok := p0.Resources() - if !ok { - t.Fatalf("page 0 Resources not found") - } - if _, ok := res.Dict("Font"); !ok { - t.Errorf("page 0 Resources missing /Font: keys %v", res.Keys()) - } - // Boxes that no ancestor defines are absent (no spec-default substitution). - if box, ok := p0.Box(BleedBox); ok { - t.Errorf("page 0 BleedBox = %+v, want absent", box) - } - - // Page 1 overrides MediaBox and Rotate locally, still inherits CropBox. - p1, err := r.Page(1) - if err != nil { - t.Fatalf("Page(1): %v", err) - } - if box, ok := p1.Box(MediaBox); !ok || box != (Rect{0, 0, 200, 200}) { - t.Errorf("page 1 MediaBox = %+v ok=%v, want {0 0 200 200}", box, ok) - } - if box, ok := p1.Box(CropBox); !ok || box != (Rect{10, 10, 602, 782}) { - t.Errorf("page 1 CropBox = %+v ok=%v, want inherited {10 10 602 782}", box, ok) - } - if rot := p1.Rotation(); rot != 0 { - t.Errorf("page 1 Rotation = %d, want 0 (override)", rot) - } -} - -func TestPageContentsArrayFixture(t *testing.T) { - r, err := OpenFile(filepath.Join("testdata", "fixtures", "page-contents-array", "input.pdf")) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - - p, err := r.Page(0) - if err != nil { - t.Fatalf("Page(0): %v", err) - } - streams, err := p.ContentStreams() - if err != nil { - t.Fatalf("ContentStreams: %v", err) - } - if len(streams) != 2 { - t.Fatalf("got %d content streams, want 2", len(streams)) - } - if raw, err := streams[0].RawBytes(); err != nil || string(raw) != "q 1 0 0 1 50 50 cm" { - t.Errorf("stream 0 RawBytes = %q err=%v", raw, err) - } - - content, err := p.Content() - if err != nil { - t.Fatalf("Content: %v", err) - } - want := "q 1 0 0 1 50 50 cm\nBT /F1 12 Tf (Hello) Tj ET" - if string(content) != want { - t.Errorf("Content = %q, want %q", content, want) - } -} - -func TestPageContentSingleStream(t *testing.T) { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Page /Parent 2 0 R /Contents 4 0 R >>", - "<< /Length 5 >>\nstream\nhello\nendstream", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - p, err := r.Page(0) - if err != nil { - t.Fatalf("Page(0): %v", err) - } - if got, err := p.Content(); err != nil || string(got) != "hello" { - t.Errorf("Content = %q err=%v, want \"hello\"", got, err) - } -} - -func TestPageContentNone(t *testing.T) { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Page /Parent 2 0 R >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - p, _ := r.Page(0) - streams, err := p.ContentStreams() - if err != nil || streams != nil { - t.Errorf("ContentStreams = %v err=%v, want nil", streams, err) - } - if got, err := p.Content(); err != nil || len(got) != 0 { - t.Errorf("Content = %q err=%v, want empty", got, err) - } -} - -func TestPageRotationNormalization(t *testing.T) { - cases := []struct { - rotate string - want int - }{ - {"450", 90}, - {"-90", 270}, - {"360", 0}, - {"270", 270}, - {"45", 0}, // not a multiple of 90 → defensive 0 - {"(x)", 0}, // wrong type → 0 - } - for _, tc := range cases { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Page /Parent 2 0 R /Rotate " + tc.rotate + " >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open(%s): %v", tc.rotate, err) - } - p, _ := r.Page(0) - if got := p.Rotation(); got != tc.want { - t.Errorf("Rotate %s → %d, want %d", tc.rotate, got, tc.want) - } - r.Close() - } -} - -func TestPageBoxNormalization(t *testing.T) { - // Corners written in the wrong order must come back normalised, and the - // box may be a Real-valued array. - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Page /Parent 2 0 R /MediaBox [612.0 792.0 0 0] >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - p, _ := r.Page(0) - box, ok := p.Box(MediaBox) - if !ok || box != (Rect{0, 0, 612, 792}) { - t.Fatalf("MediaBox = %+v ok=%v, want {0 0 612 792}", box, ok) - } - if box.Width() != 612 || box.Height() != 792 { - t.Errorf("Width/Height = %v/%v, want 612/792", box.Width(), box.Height()) - } -} - -func TestPageBoxMalformed(t *testing.T) { - // A box that is not a 4-number array must report absent, not panic. - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612] /CropBox (nope) >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - p, _ := r.Page(0) - if box, ok := p.Box(MediaBox); ok { - t.Errorf("short MediaBox = %+v, want absent", box) - } - if box, ok := p.Box(CropBox); ok { - t.Errorf("string CropBox = %+v, want absent", box) - } - if boxes := p.Boxes(); len(boxes) != 0 { - t.Errorf("Boxes = %v, want empty", boxes) - } -} - -func TestPageNonInheritableBoxes(t *testing.T) { - // An intermediate /Pages node carrying BleedBox/TrimBox/ArtBox must NOT - // leak those into descendant leaves — only MediaBox/CropBox are inheritable - // (PDF 32000-1:2008 Table 30). The leaf's own TrimBox is still reported. - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] /MediaBox [0 0 612 792] /CropBox [1 1 611 791] /BleedBox [2 2 610 790] /TrimBox [3 3 609 789] /ArtBox [4 4 608 788] >>", - "<< /Type /Page /Parent 2 0 R /TrimBox [9 9 100 100] >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - p, _ := r.Page(0) - - // Inheritable: come from the ancestor /Pages node. - if box, ok := p.Box(MediaBox); !ok || box != (Rect{0, 0, 612, 792}) { - t.Errorf("MediaBox = %+v ok=%v, want inherited {0 0 612 792}", box, ok) - } - if box, ok := p.Box(CropBox); !ok || box != (Rect{1, 1, 611, 791}) { - t.Errorf("CropBox = %+v ok=%v, want inherited {1 1 611 791}", box, ok) - } - // Non-inheritable: ancestor values must not appear. - if box, ok := p.Box(BleedBox); ok { - t.Errorf("BleedBox = %+v, want absent (not inheritable)", box) - } - if box, ok := p.Box(ArtBox); ok { - t.Errorf("ArtBox = %+v, want absent (not inheritable)", box) - } - // The leaf's own TrimBox is reported, not the ancestor's. - if box, ok := p.Box(TrimBox); !ok || box != (Rect{9, 9, 100, 100}) { - t.Errorf("TrimBox = %+v ok=%v, want page-level {9 9 100 100}", box, ok) - } - if boxes := p.Boxes(); len(boxes) != 3 { - t.Errorf("Boxes = %v, want 3 (Media, Crop, Trim)", boxes) - } -} - -func TestPageContentsArrayBrokenEntryErrors(t *testing.T) { - // /Contents references object 5, which is absent from the xref and so - // resolves to null. ContentStreams/Content must fail rather than silently - // return a truncated stream. - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Page /Parent 2 0 R /Contents [ 4 0 R 5 0 R ] >>", - "<< /Length 5 >>\nstream\nhello\nendstream", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - p, _ := r.Page(0) - if streams, err := p.ContentStreams(); err == nil { - t.Errorf("ContentStreams = %v, want error for missing entry", streams) - } - if got, err := p.Content(); err == nil { - t.Errorf("Content = %q, want error for missing entry", got) - } -} - -func TestPagesPhantomLeafGuard(t *testing.T) { - // The root /Pages node declares /Count but its /Kids fails to resolve - // (object 9 is absent). It must not be emitted as a phantom leaf page. - for _, root := range []string{ - "<< /Type /Pages /Count 1 /Kids 9 0 R >>", // unresolvable /Kids ref - "<< /Type /Pages /Count 1 >>", // no /Kids at all - "<< /Type /Pages >>", // /Type-only intermediate - } { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - root, - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open(%s): %v", root, err) - } - n, err := r.PageCount() - if err != nil { - t.Fatalf("PageCount(%s): %v", root, err) - } - if n != 0 { - t.Errorf("root %s: PageCount = %d, want 0 (no phantom leaf)", root, n) - } - r.Close() - } -} - -func TestPageIndexOutOfRange(t *testing.T) { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Page /Parent 2 0 R >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if _, err := r.Page(-1); err == nil { - t.Error("Page(-1) = nil error, want out-of-range error") - } - if _, err := r.Page(1); err == nil { - t.Error("Page(1) = nil error, want out-of-range error") - } -} - -func TestPageCyclicKidsTerminates(t *testing.T) { - // obj 3's /Kids points back to obj 2; the walk must terminate. - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Pages /Kids [ 2 0 R ] >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - n, err := r.PageCount() - if err != nil { - t.Fatalf("PageCount: %v", err) - } - if n != 0 { - t.Errorf("PageCount = %d, want 0", n) - } -} - -func TestPageCyclicParentTerminates(t *testing.T) { - // The leaf page's /Parent points to itself; inheritance must terminate. - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Page /Parent 3 0 R >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - p, err := r.Page(0) - if err != nil { - t.Fatalf("Page(0): %v", err) - } - if box, ok := p.Box(MediaBox); ok { - t.Errorf("MediaBox = %+v, want absent", box) - } - if rot := p.Rotation(); rot != 0 { - t.Errorf("Rotation = %d, want 0", rot) - } - if _, ok := p.Resources(); ok { - t.Errorf("Resources found, want none") - } -} - -func TestPagesMissingType(t *testing.T) { - // Neither the intermediate node nor the leaf declares /Type; the walk - // keys off /Kids presence instead. - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Kids [ 3 0 R ] /Count 1 >>", - "<< /Parent 2 0 R /MediaBox [0 0 100 100] >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - n, err := r.PageCount() - if err != nil { - t.Fatalf("PageCount: %v", err) - } - if n != 1 { - t.Fatalf("PageCount = %d, want 1", n) - } - p, _ := r.Page(0) - if box, ok := p.Box(MediaBox); !ok || box != (Rect{0, 0, 100, 100}) { - t.Errorf("MediaBox = %+v ok=%v, want {0 0 100 100}", box, ok) - } -} diff --git a/parse.go b/parse.go deleted file mode 100644 index 50b5ba2..0000000 --- a/parse.go +++ /dev/null @@ -1,199 +0,0 @@ -package pdfdisassembler - -import ( - "errors" - "fmt" - "strconv" - - "github.com/speedata/pdfdisassembler/internal/lex" -) - -// maxParseDepth caps array/dict nesting so a hostile PDF can't stack-overflow -// the recursive parser. -const maxParseDepth = 1000 - -// parser is a recursive descent parser over a lex.Lexer that emits direct -// PDF Objects. It does not chase indirect references — every Reference -// token becomes a Reference value. -type parser struct { - lx *lex.Lexer - r *Reader - queue []lex.Token - depth int -} - -func newParser(lx *lex.Lexer, r *Reader) *parser { - return &parser{lx: lx, r: r} -} - -func (p *parser) next() (lex.Token, error) { - if len(p.queue) > 0 { - t := p.queue[0] - p.queue = p.queue[1:] - return t, nil - } - return p.lx.Next() -} - -func (p *parser) peek() (lex.Token, error) { - if len(p.queue) > 0 { - return p.queue[0], nil - } - t, err := p.lx.Next() - if err != nil { - return lex.Token{}, err - } - p.queue = append(p.queue, t) - return t, nil -} - -// peekN returns the n-th unread token (1-based). It reads from the lexer -// to fill the queue as needed. -func (p *parser) peekN(n int) (lex.Token, error) { - for len(p.queue) < n { - t, err := p.lx.Next() - if err != nil { - return lex.Token{}, err - } - p.queue = append(p.queue, t) - } - return p.queue[n-1], nil -} - -// consume removes n tokens from the front of the queue (after a successful -// peekN). Caller must have ensured the queue has at least n entries. -func (p *parser) consume(n int) { - p.queue = p.queue[n:] -} - -// parseObject parses a single direct PDF object. It also recognises the -// "N G R" indirect-reference triplet, which requires two-token lookahead. -func (p *parser) parseObject() (Object, error) { - tok, err := p.next() - if err != nil { - return nil, err - } - return p.parseObjectFrom(tok) -} - -func (p *parser) parseObjectFrom(tok lex.Token) (Object, error) { - switch tok.Kind { - case lex.EOF: - return nil, errors.New("pdfdisassembler/parse: unexpected EOF") - case lex.Name: - return Name(string(tok.Bytes)), nil - case lex.LitString: - // Copy: Bytes points into the source buffer of the literal-string - // reader which is stable for our use, but ownership-wise we'd - // rather not have callers alias the source. - b := make(String, len(tok.Bytes)) - copy(b, tok.Bytes) - return b, nil - case lex.HexString: - b := make(String, len(tok.Bytes)) - copy(b, tok.Bytes) - return b, nil - case lex.Integer: - n, err := strconv.ParseInt(string(tok.Bytes), 10, 64) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler/parse: bad integer %q: %w", tok.Bytes, err) - } - // Look ahead two tokens for a Reference: "G R". - t2, err := p.peekN(1) - if err != nil || t2.Kind != lex.Integer { - return Integer(n), nil - } - t3, err := p.peekN(2) - if err != nil || t3.Kind != lex.Keyword || string(t3.Bytes) != "R" { - return Integer(n), nil - } - g, err := strconv.Atoi(string(t2.Bytes)) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler/parse: bad generation %q: %w", t2.Bytes, err) - } - p.consume(2) // gen and 'R' - return Reference{Number: int(n), Generation: g}, nil - case lex.Real: - f, err := strconv.ParseFloat(string(tok.Bytes), 64) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler/parse: bad real %q: %w", tok.Bytes, err) - } - return Real(f), nil - case lex.Keyword: - switch string(tok.Bytes) { - case "true": - return Bool(true), nil - case "false": - return Bool(false), nil - case "null": - return Null{}, nil - } - return nil, fmt.Errorf("pdfdisassembler/parse: unexpected keyword %q at %d", tok.Bytes, tok.Offset) - case lex.ArrayStart: - return p.parseArray() - case lex.DictStart: - return p.parseDict() - case lex.ArrayEnd, lex.DictEnd: - return nil, fmt.Errorf("pdfdisassembler/parse: stray %s at %d", tok.Kind, tok.Offset) - } - return nil, fmt.Errorf("pdfdisassembler/parse: unhandled token %s at %d", tok.Kind, tok.Offset) -} - -func (p *parser) parseArray() (Array, error) { - p.depth++ - defer func() { p.depth-- }() - if p.depth > maxParseDepth { - return nil, fmt.Errorf("pdfdisassembler/parse: nesting too deep (> %d)", maxParseDepth) - } - var out Array - for { - t, err := p.peek() - if err != nil { - return nil, err - } - if t.Kind == lex.ArrayEnd { - p.next() - return out, nil - } - if t.Kind == lex.EOF { - return nil, errors.New("pdfdisassembler/parse: unterminated array") - } - v, err := p.parseObject() - if err != nil { - return nil, err - } - out = append(out, v) - } -} - -func (p *parser) parseDict() (*Dict, error) { - p.depth++ - defer func() { p.depth-- }() - if p.depth > maxParseDepth { - return nil, fmt.Errorf("pdfdisassembler/parse: nesting too deep (> %d)", maxParseDepth) - } - d := newDict(p.r) - for { - t, err := p.peek() - if err != nil { - return nil, err - } - if t.Kind == lex.DictEnd { - p.next() - return d, nil - } - if t.Kind == lex.EOF { - return nil, errors.New("pdfdisassembler/parse: unterminated dictionary") - } - if t.Kind != lex.Name { - return nil, fmt.Errorf("pdfdisassembler/parse: dict key must be a name, got %s at %d", t.Kind, t.Offset) - } - p.next() - key := string(t.Bytes) - v, err := p.parseObject() - if err != nil { - return nil, err - } - d.set(key, v) - } -} diff --git a/parse_test.go b/parse_test.go deleted file mode 100644 index ef7e3af..0000000 --- a/parse_test.go +++ /dev/null @@ -1,92 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "fmt" - "strings" - "testing" - - "github.com/speedata/pdfdisassembler/internal/lex" -) - -// buildPDFWithObjectBody puts body as object 3 in a minimal classical-xref PDF. -func buildPDFWithObjectBody(t *testing.T, body string) []byte { - t.Helper() - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - offsets := make([]int, 4) - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - offsets[3] = off() - fmt.Fprintf(&buf, "3 0 obj\n%s\nendobj\n", body) - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 4\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 3; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprint(&buf, "trailer\n<< /Size 4 /Root 1 0 R >>\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -func TestDeeplyNestedRejected(t *testing.T) { - // Far above the parser's depth cap, but well below a real stack overflow. - const depth = 2000 - tests := []struct{ name, body string }{ - {"array", strings.Repeat("[", depth) + strings.Repeat("]", depth)}, - {"dict", strings.Repeat("<< /K ", depth) + "0" + strings.Repeat(" >>", depth)}, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - data := buildPDFWithObjectBody(t, tt.body) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if _, err := r.Resolve(Reference{Number: 3, Generation: 0}); err == nil { - t.Fatal("expected error for over-deep nesting") - } - }) - } -} - -func TestModeratelyNestedArrayResolves(t *testing.T) { - data := buildPDFWithObjectBody(t, strings.Repeat("[", 100)+strings.Repeat("]", 100)) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - obj, err := r.Resolve(Reference{Number: 3, Generation: 0}) - if err != nil { - t.Fatalf("Resolve: %v", err) - } - if _, ok := obj.(Array); !ok { - t.Fatalf("got %T, want Array", obj) - } -} - -// Malformed token streams must error, never panic. -func TestParseObjectErrors(t *testing.T) { - for _, src := range []string{ - "", // EOF where an object is expected - "]", // stray ArrayEnd - ">>", // stray DictEnd - "foo", // unexpected keyword - "[ 1 2", // unterminated array - "<< /K 1", // unterminated dict - "<< 1 2 >>", // dict key is not a name - } { - t.Run(src, func(t *testing.T) { - p := newParser(lex.New([]byte(src)), nil) - if _, err := p.parseObject(); err == nil { - t.Fatal("expected an error, got nil") - } - }) - } -} diff --git a/reader.go b/reader.go deleted file mode 100644 index bf49014..0000000 --- a/reader.go +++ /dev/null @@ -1,526 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "errors" - "fmt" - "io" - "iter" - "os" - "sort" - "strconv" - - "github.com/speedata/pdfdisassembler/internal/lex" -) - -// DefaultMaxStreamSize is the per-stream decoded-size cap Open uses by default. -const DefaultMaxStreamSize int64 = 16 << 20 - -// Option configures a Reader at Open time. -type Option func(*Reader) - -// WithMaxStreamSize sets the per-stream decoded-size cap; n <= 0 disables it. -// Applied before parsing, so it also bounds streams decoded during Open. -func WithMaxStreamSize(n int64) Option { - return func(r *Reader) { r.MaxStreamSize = n } -} - -// Reader is a parsed PDF document. It is not safe for concurrent use. -type Reader struct { - src io.ReadSeeker - closer io.Closer - buf []byte // entire file contents - version string - xref map[Reference]xrefEntry - trailer *Dict - catalog *Dict - info *Dict - infoLoad bool - - // MaxStreamSize caps each stream's decoded size; <= 0 disables it. Setting - // it after Open misses Open-time (xref/object) streams; use WithMaxStreamSize. - MaxStreamSize int64 - - // Encryption. - encrypt *encryptCtx - - // Resolution caches. - objCache map[Reference]Object - // resolveStack guards against indirect-reference cycles. - resolveStack map[Reference]struct{} - - // Page-tree cache, populated lazily by loadPages. - pages []*Page - pagesLoaded bool - pagesErr error -} - -// xrefEntry describes a single in-use object. -type xrefEntry struct { - // kind is 1 for in-file objects (Offset), 2 for compressed objects - // (ObjStmNum + Index). - kind uint8 - offset int64 - objStmNum int - objStmIdx int - generation int -} - -// Open parses a PDF from rs. rs must remain valid for the lifetime of the -// returned Reader. -func Open(rs io.ReadSeeker, opts ...Option) (*Reader, error) { - if _, err := rs.Seek(0, io.SeekStart); err != nil { - return nil, fmt.Errorf("pdfdisassembler: seek: %w", err) - } - buf, err := io.ReadAll(rs) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: read: %w", err) - } - r := &Reader{ - src: rs, - buf: buf, - xref: map[Reference]xrefEntry{}, - objCache: map[Reference]Object{}, - resolveStack: map[Reference]struct{}{}, - MaxStreamSize: DefaultMaxStreamSize, - } - for _, opt := range opts { - opt(r) - } - if err := r.parseHeader(); err != nil { - return nil, err - } - if err := r.parseXref(); err != nil { - return nil, err - } - if err := r.initEncrypt(); err != nil { - return nil, err - } - return r, nil -} - -// OpenFile opens path and parses it as a PDF. The file stays open until -// Reader.Close is called. -func OpenFile(path string, opts ...Option) (*Reader, error) { - f, err := os.Open(path) - if err != nil { - return nil, err - } - r, err := Open(f, opts...) - if err != nil { - f.Close() - return nil, err - } - r.closer = f - return r, nil -} - -// Close releases the underlying resource. For Reader instances created via -// Open with a non-file ReadSeeker, Close is a no-op. -func (r *Reader) Close() error { - if r.closer != nil { - return r.closer.Close() - } - return nil -} - -// Version returns the PDF version declared in the file header (e.g. "1.7" -// or "2.0"). If the catalog declares a /Version entry that exceeds the -// header version, the catalog value wins (per spec). -func (r *Reader) Version() string { - if r.catalog != nil { - if n, ok := r.catalog.Name("Version"); ok { - s := string(n) - if s > r.version { - return s - } - } - } - return r.version -} - -// parseHeader reads the "%PDF-x.y" line. It tolerates up to 1024 leading -// bytes of garbage (some producers emit MIME prologues). -func (r *Reader) parseHeader() error { - scan := r.buf - limit := len(scan) - if limit > 1024 { - limit = 1024 - } - idx := bytes.Index(scan[:limit], []byte("%PDF-")) - if idx < 0 { - return errors.New("pdfdisassembler: not a PDF (missing %PDF- header)") - } - rest := scan[idx+5:] - end := 0 - for end < len(rest) && end < 16 { - c := rest[end] - if c == '\r' || c == '\n' || c == ' ' || c == '\t' { - break - } - end++ - } - r.version = string(rest[:end]) - return nil -} - -// Catalog returns the document catalog dictionary. -func (r *Reader) Catalog() (*Dict, error) { - if r.catalog != nil { - return r.catalog, nil - } - if r.trailer == nil { - return nil, errors.New("pdfdisassembler: no trailer") - } - root, ok := r.trailer.Get("Root") - if !ok { - return nil, errors.New("pdfdisassembler: trailer has no /Root") - } - d, err := r.ResolveDict(root) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: resolve catalog: %w", err) - } - r.catalog = d - return d, nil -} - -// Trailer returns the trailer dictionary. -func (r *Reader) Trailer() *Dict { - return r.trailer -} - -// Resolve follows an indirect reference. If obj is not a Reference, returns -// obj unchanged. Resolution is cached. -func (r *Reader) Resolve(obj Object) (Object, error) { - ref, ok := obj.(Reference) - if !ok { - return obj, nil - } - if cached, ok := r.objCache[ref]; ok { - return cached, nil - } - if _, on := r.resolveStack[ref]; on { - return Null{}, fmt.Errorf("pdfdisassembler: reference cycle at %d %d R", ref.Number, ref.Generation) - } - r.resolveStack[ref] = struct{}{} - defer delete(r.resolveStack, ref) - - entry, ok := r.xref[ref] - if !ok { - // Some xref tables omit the requested object. Per spec, missing - // references resolve to null. - r.objCache[ref] = Null{} - return Null{}, nil - } - var v Object - var err error - switch entry.kind { - case 1: - v, err = r.readIndirectAt(entry.offset, ref) - case 2: - v, err = r.readCompressedObject(entry.objStmNum, entry.objStmIdx, ref) - default: - return nil, fmt.Errorf("pdfdisassembler: unknown xref entry kind for %d %d R", ref.Number, ref.Generation) - } - if err != nil { - return nil, err - } - r.objCache[ref] = v - return v, nil -} - -// ResolveDict resolves obj to a *Dict; errors when obj is missing or is -// not a dictionary. -func (r *Reader) ResolveDict(obj Object) (*Dict, error) { - v, err := r.Resolve(obj) - if err != nil { - return nil, err - } - if v == nil { - return nil, errors.New("pdfdisassembler: nil object") - } - switch t := v.(type) { - case *Dict: - return t, nil - case *Stream: - return t.Dict, nil - case Null: - return nil, errors.New("pdfdisassembler: dictionary expected, got null") - } - return nil, fmt.Errorf("pdfdisassembler: dictionary expected, got %T", v) -} - -// ResolveBool resolves obj to a bool; errors otherwise. -func (r *Reader) ResolveBool(obj Object) (bool, error) { - v, err := r.Resolve(obj) - if err != nil { - return false, err - } - b, ok := v.(Bool) - if !ok { - return false, fmt.Errorf("pdfdisassembler: boolean expected, got %T", v) - } - return bool(b), nil -} - -// ResolveInt resolves obj to an int64; errors otherwise. -func (r *Reader) ResolveInt(obj Object) (int64, error) { - v, err := r.Resolve(obj) - if err != nil { - return 0, err - } - n, ok := v.(Integer) - if !ok { - return 0, fmt.Errorf("pdfdisassembler: integer expected, got %T", v) - } - return int64(n), nil -} - -// ResolveArray resolves obj to an Array; errors otherwise. -func (r *Reader) ResolveArray(obj Object) (Array, error) { - v, err := r.Resolve(obj) - if err != nil { - return nil, err - } - a, ok := v.(Array) - if !ok { - return nil, fmt.Errorf("pdfdisassembler: array expected, got %T", v) - } - return a, nil -} - -// readIndirectAt parses the indirect object starting at offset. The object -// header (N G obj) is verified against the expected reference. -func (r *Reader) readIndirectAt(offset int64, expect Reference) (Object, error) { - if offset < 0 || offset >= int64(len(r.buf)) { - return nil, fmt.Errorf("pdfdisassembler: xref offset %d out of range", offset) - } - lx := lex.New(r.buf) - lx.SetPos(int(offset)) - p := newParser(lx, r) - - // Read "N G obj" header. - t1, err := p.next() - if err != nil { - return nil, err - } - t2, err := p.next() - if err != nil { - return nil, err - } - t3, err := p.next() - if err != nil { - return nil, err - } - if t1.Kind != lex.Integer || t2.Kind != lex.Integer || - t3.Kind != lex.Keyword || string(t3.Bytes) != "obj" { - return nil, fmt.Errorf("pdfdisassembler: bad indirect header at %d (got %s %s %s)", offset, t1.Kind, t2.Kind, t3.Kind) - } - n, _ := strconv.Atoi(string(t1.Bytes)) - g, _ := strconv.Atoi(string(t2.Bytes)) - if n != expect.Number { - return nil, fmt.Errorf("pdfdisassembler: indirect mismatch at %d: header %d %d, expected %d %d", offset, n, g, expect.Number, expect.Generation) - } - - body, err := p.parseObject() - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: parse body of %d %d R: %w", expect.Number, expect.Generation, err) - } - - // Check for stream. - t4, err := p.peek() - if err == nil && t4.Kind == lex.Keyword && string(t4.Bytes) == "stream" { - p.next() - d, ok := body.(*Dict) - if !ok { - return nil, fmt.Errorf("pdfdisassembler: stream object %d %d R has non-dict body (%T)", expect.Number, expect.Generation, body) - } - length, err := r.streamLength(d) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: /Length for %d %d R: %w", expect.Number, expect.Generation, err) - } - raw, err := lx.ReadStreamData(int(length)) - if err != nil { - return nil, fmt.Errorf("pdfdisassembler: read stream %d %d R: %w", expect.Number, expect.Generation, err) - } - // rawOffset is where the raw bytes begin in the file. - rawStart := lx.Pos() - len(raw) - return &Stream{ - Dict: d, - reader: r, - rawOffset: int64(rawStart), - rawLength: int64(len(raw)), - objNumber: expect.Number, - objGeneration: expect.Generation, - }, nil - } - return body, nil -} - -// streamLength resolves the /Length entry on a stream dict. -func (r *Reader) streamLength(d *Dict) (int64, error) { - v, ok := d.Get("Length") - if !ok { - return 0, errors.New("missing /Length") - } - v, err := r.Resolve(v) - if err != nil { - return 0, err - } - n, ok := v.(Integer) - if !ok { - return 0, fmt.Errorf("/Length is %T, want integer", v) - } - if n < 0 { - return 0, fmt.Errorf("/Length is negative: %d", n) - } - return int64(n), nil -} - -// DocumentInfo returns the standard /Info dictionary entries as a value -// snapshot. Missing entries return zero values. -func (r *Reader) DocumentInfo() DocInfo { - if !r.infoLoad { - r.infoLoad = true - if r.trailer != nil { - if obj, ok := r.trailer.Get("Info"); ok { - if d, err := r.ResolveDict(obj); err == nil { - r.info = d - } - } - } - } - var info DocInfo - info.Custom = map[string]string{} - if r.info == nil { - return info - } - for k, v := range r.info.Iter() { - resolved, err := r.Resolve(v) - if err != nil { - continue - } - s, ok := resolved.(String) - if !ok { - continue - } - decoded := decodeTextString(s) - switch k { - case "Title": - info.Title = decoded - case "Author": - info.Author = decoded - case "Subject": - info.Subject = decoded - case "Keywords": - info.Keywords = decoded - case "Creator": - info.Creator = decoded - case "Producer": - info.Producer = decoded - case "CreationDate": - info.CreationDate = parseDate(decoded) - case "ModDate": - info.ModDate = parseDate(decoded) - default: - info.Custom[k] = decoded - } - } - return info -} - -// Objects iterates every live indirect object in the xref table. -func (r *Reader) Objects() iter.Seq[ObjectEntry] { - return func(yield func(ObjectEntry) bool) { - // Iterate in stable order: by object number. - refs := make([]Reference, 0, len(r.xref)) - for ref := range r.xref { - refs = append(refs, ref) - } - // The entry count is attacker-controlled, so this must stay O(n log n). - sort.Slice(refs, func(i, j int) bool { - if refs[i].Number != refs[j].Number { - return refs[i].Number < refs[j].Number - } - return refs[i].Generation < refs[j].Generation - }) - for _, ref := range refs { - obj, err := r.Resolve(ref) - if err != nil { - continue - } - if !yield(ObjectEntry{Reference: ref, Object: obj}) { - return - } - } - } -} - -// DecodeStream resolves obj to a stream and returns its decoded content. -func (r *Reader) DecodeStream(obj Object) ([]byte, error) { - v, err := r.Resolve(obj) - if err != nil { - return nil, err - } - s, ok := v.(*Stream) - if !ok { - return nil, fmt.Errorf("pdfdisassembler: stream expected, got %T", v) - } - return s.Content() -} - -const maxNameTreeDepth = 1000 - -// EmbeddedFiles returns the document's embedded files (PDF attachments) from -// the catalog's EmbeddedFiles name tree, in tree order. Returns nil when there -// are none. -func (r *Reader) EmbeddedFiles() []EmbeddedFile { - cat, err := r.Catalog() - if err != nil { - return nil - } - names, ok := cat.Dict("Names") - if !ok { - return nil - } - root, ok := names.Dict("EmbeddedFiles") - if !ok { - return nil - } - var out []EmbeddedFile - r.walkNameTree(root, map[Reference]struct{}{}, 0, &out) - return out -} - -// walkNameTree collects (name, /Filespec) pairs from a name-tree node. seen -// records already-visited /Kids references and depth bounds the descent, so a -// cyclic or pathologically deep /Kids graph can't loop or overflow the stack. -func (r *Reader) walkNameTree(node *Dict, seen map[Reference]struct{}, depth int, out *[]EmbeddedFile) { - if node == nil || depth > maxNameTreeDepth { - return - } - if kids, ok := node.Array("Kids"); ok { - for _, kid := range kids { - if ref, ok := kid.(Reference); ok { - if _, dup := seen[ref]; dup { - continue - } - seen[ref] = struct{}{} - } - if child, err := r.ResolveDict(kid); err == nil { - r.walkNameTree(child, seen, depth+1, out) - } - } - } - if entries, ok := node.Array("Names"); ok { - for i := 0; i+1 < len(entries); i += 2 { - name, ok := entries[i].(String) - if !ok { - continue - } - if spec, err := r.ResolveDict(entries[i+1]); err == nil { - *out = append(*out, EmbeddedFile{Name: string(name), Spec: spec}) - } - } - } -} diff --git a/reader_test.go b/reader_test.go deleted file mode 100644 index 13985c2..0000000 --- a/reader_test.go +++ /dev/null @@ -1,623 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "compress/zlib" - "fmt" - "os" - "path/filepath" - "strings" - "testing" -) - -// buildMinimalPDF constructs a tiny valid PDF in memory: a catalog, a -// pages tree with one empty page, an Info dict, and a classical xref. It -// returns the raw bytes. -func buildMinimalPDF(t *testing.T) []byte { - t.Helper() - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - - offsets := make([]int, 5) // index 1..4 - - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>\nendobj\n") - - offsets[3] = off() - fmt.Fprint(&buf, "3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] >>\nendobj\n") - - offsets[4] = off() - fmt.Fprint(&buf, "4 0 obj\n<< /Title (Hello) /Producer (pdfdisassembler-test) >>\nendobj\n") - - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 5\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 4; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprintf(&buf, "trailer\n<< /Size 5 /Root 1 0 R /Info 4 0 R >>\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -func TestOpenMinimal(t *testing.T) { - data := buildMinimalPDF(t) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - - if r.Version() != "1.7" { - t.Fatalf("version %q", r.Version()) - } - cat, err := r.Catalog() - if err != nil { - t.Fatalf("Catalog: %v", err) - } - n, ok := cat.Name("Type") - if !ok || n != "Catalog" { - t.Fatalf("/Type %q ok=%v", n, ok) - } - pages, ok := cat.Dict("Pages") - if !ok { - t.Fatal("Catalog.Dict(Pages)") - } - count, ok := pages.Int("Count") - if !ok || count != 1 { - t.Fatalf("/Count %d ok=%v", count, ok) - } -} - -func TestDocumentInfo(t *testing.T) { - data := buildMinimalPDF(t) - r, _ := Open(bytes.NewReader(data)) - defer r.Close() - info := r.DocumentInfo() - if info.Title != "Hello" { - t.Fatalf("Title %q", info.Title) - } - if !strings.HasPrefix(info.Producer, "pdfdisassembler") { - t.Fatalf("Producer %q", info.Producer) - } -} - -func TestObjectsIterator(t *testing.T) { - data := buildMinimalPDF(t) - r, _ := Open(bytes.NewReader(data)) - defer r.Close() - count := 0 - seen := map[int]bool{} - for entry := range r.Objects() { - count++ - seen[entry.Reference.Number] = true - } - if count != 4 { - t.Fatalf("count %d", count) - } - for i := 1; i <= 4; i++ { - if !seen[i] { - t.Fatalf("missing object %d", i) - } - } -} - -func TestObjectsIteratorSortedOrder(t *testing.T) { - r, err := Open(bytes.NewReader(buildMinimalPDF(t))) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - var nums []int - for entry := range r.Objects() { - nums = append(nums, entry.Reference.Number) - } - if len(nums) == 0 { - t.Fatal("no objects iterated") - } - for i := 1; i < len(nums); i++ { - if nums[i-1] >= nums[i] { - t.Fatalf("Objects() not strictly ascending: %v", nums) - } - } -} - -// buildPDFWithStream embeds payload as a FlateDecode stream (obj 3) so -// DecodeStream can be exercised. -func buildPDFWithStream(t *testing.T, payload []byte) []byte { - t.Helper() - var zbuf bytes.Buffer - zw := zlib.NewWriter(&zbuf) - zw.Write(payload) - zw.Close() - zdata := zbuf.Bytes() - - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - - offsets := make([]int, 4) - - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - - offsets[3] = off() - fmt.Fprintf(&buf, "3 0 obj\n<< /Length %d /Filter /FlateDecode >>\nstream\n", len(zdata)) - buf.Write(zdata) - fmt.Fprint(&buf, "\nendstream\nendobj\n") - - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 4\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 3; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprintf(&buf, "trailer\n<< /Size 4 /Root 1 0 R >>\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -func TestDecodeStream(t *testing.T) { - data := buildPDFWithStream(t, []byte("Hello, stream!")) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - - content, err := r.DecodeStream(Reference{Number: 3, Generation: 0}) - if err != nil { - t.Fatalf("DecodeStream: %v", err) - } - if string(content) != "Hello, stream!" { - t.Fatalf("content %q", content) - } -} - -func TestStreamSizeLimitEnforced(t *testing.T) { - // obj 3 decompresses to 2 MiB; the 64 KiB cap must reject it. - data := buildPDFWithStream(t, make([]byte, 2<<20)) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - - r.MaxStreamSize = 64 << 10 - if _, err := r.DecodeStream(Reference{Number: 3, Generation: 0}); err == nil { - t.Fatal("expected error decoding stream larger than MaxStreamSize, got nil") - } -} - -func TestWithMaxStreamSizeOption(t *testing.T) { - data := buildPDFWithStream(t, make([]byte, 2<<20)) - r, err := Open(bytes.NewReader(data), WithMaxStreamSize(64<<10)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if r.MaxStreamSize != 64<<10 { - t.Fatalf("MaxStreamSize = %d, want %d", r.MaxStreamSize, 64<<10) - } - if _, err := r.DecodeStream(Reference{Number: 3, Generation: 0}); err == nil { - t.Fatal("expected error: stream exceeds the option-set cap") - } -} - -func TestWithMaxStreamSizeDisable(t *testing.T) { - data := buildPDFWithStream(t, make([]byte, 2<<20)) - r, err := Open(bytes.NewReader(data), WithMaxStreamSize(0)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - out, err := r.DecodeStream(Reference{Number: 3, Generation: 0}) - if err != nil { - t.Fatalf("DecodeStream with cap disabled: %v", err) - } - if len(out) != 2<<20 { - t.Fatalf("decoded %d bytes, want %d", len(out), 2<<20) - } -} - -func TestDefaultMaxStreamSizeSet(t *testing.T) { - // Open must install a finite default so Open-time decodes are bounded. - data := buildMinimalPDF(t) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if r.MaxStreamSize != DefaultMaxStreamSize { - t.Fatalf("MaxStreamSize = %d, want default %d", r.MaxStreamSize, DefaultMaxStreamSize) - } -} - -// buildDictPDF puts each body in objs as object i+1 of a classical-xref PDF -// (obj 1 is the catalog). Bodies are plain objects (no streams). -func buildDictPDF(t *testing.T, objs []string) []byte { - t.Helper() - var buf bytes.Buffer - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - offsets := make([]int, len(objs)+1) - for i, body := range objs { - offsets[i+1] = buf.Len() - fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", i+1, body) - } - xrefOff := buf.Len() - fmt.Fprintf(&buf, "xref\n0 %d\n%010d %05d f \n", len(objs)+1, 0, 65535) - for i := 1; i <= len(objs); i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprintf(&buf, "trailer\n<< /Size %d /Root 1 0 R >>\nstartxref\n%d\n%%%%EOF\n", - len(objs)+1, xrefOff) - return buf.Bytes() -} - -func TestEmbeddedFiles(t *testing.T) { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles 3 0 R >> >>", - "<< /Type /Pages /Kids [] /Count 0 >>", - "<< /Names [ (a.xml) 4 0 R (b.xml) 5 0 R ] >>", - "<< /Type /Filespec /F (a.xml) >>", - "<< /Type /Filespec /F (b.xml) >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - ef := r.EmbeddedFiles() - if len(ef) != 2 { - t.Fatalf("got %d files, want 2", len(ef)) - } - if ef[0].Name != "a.xml" || ef[1].Name != "b.xml" { - t.Fatalf("names %q, %q", ef[0].Name, ef[1].Name) - } - if f, ok := ef[0].Spec.String("F"); !ok || f != "a.xml" { - t.Fatalf("spec /F %q ok=%v", f, ok) - } -} - -func TestEmbeddedFilesNestedKids(t *testing.T) { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles 3 0 R >> >>", - "<< /Type /Pages /Kids [] /Count 0 >>", - "<< /Kids [ 4 0 R ] >>", - "<< /Names [ (a.xml) 5 0 R ] >>", - "<< /Type /Filespec /F (a.xml) >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if ef := r.EmbeddedFiles(); len(ef) != 1 || ef[0].Name != "a.xml" { - t.Fatalf("got %+v, want one a.xml", ef) - } -} - -func TestEmbeddedFilesCyclicKidsTerminates(t *testing.T) { - // obj 3's /Kids references itself; the walk must terminate, not overflow. - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R /Names << /EmbeddedFiles 3 0 R >> >>", - "<< /Type /Pages /Kids [] /Count 0 >>", - "<< /Kids [ 3 0 R ] >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if ef := r.EmbeddedFiles(); len(ef) != 0 { - t.Fatalf("got %d files, want 0", len(ef)) - } -} - -func TestEmbeddedFilesNone(t *testing.T) { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Kids [] /Count 0 >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if ef := r.EmbeddedFiles(); ef != nil { - t.Fatalf("got %+v, want nil", ef) - } -} - -// FuzzOpen asserts the read pipeline never panics on arbitrary input: Open and -// every accessor may return an error, but must not crash the process. -func FuzzOpen(f *testing.F) { - seeds, _ := filepath.Glob("testdata/fixtures/*/input.pdf") - for _, p := range seeds { - if b, err := os.ReadFile(p); err == nil { - f.Add(b) - } - } - f.Add([]byte("%PDF-1.7\n")) - f.Fuzz(func(t *testing.T, data []byte) { - r, err := Open(bytes.NewReader(data)) - if err != nil { - return - } - defer r.Close() - _, _ = r.Catalog() - _ = r.DocumentInfo() - _ = r.EmbeddedFiles() - _ = r.Version() - for entry := range r.Objects() { - if s, ok := entry.Object.(*Stream); ok { - _, _ = s.Content() - } - } - }) -} - -func TestResolveHelpers(t *testing.T) { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R /IntRef 3 0 R /BoolRef 4 0 R /ArrRef 5 0 R >>", - "<< /Type /Pages /Kids [] /Count 0 >>", - "42", - "true", - "[ 1 2 3 ]", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - cat, err := r.Catalog() - if err != nil { - t.Fatalf("Catalog: %v", err) - } - intRef, _ := cat.Get("IntRef") - boolRef, _ := cat.Get("BoolRef") - arrRef, _ := cat.Get("ArrRef") - - if v, err := r.ResolveInt(intRef); err != nil || v != 42 { - t.Errorf("ResolveInt = %d, %v", v, err) - } - if v, err := r.ResolveBool(boolRef); err != nil || !v { - t.Errorf("ResolveBool = %v, %v", v, err) - } - if a, err := r.ResolveArray(arrRef); err != nil || len(a) != 3 { - t.Errorf("ResolveArray len = %d, %v", len(a), err) - } - // Type mismatches must error. - if _, err := r.ResolveInt(boolRef); err == nil { - t.Error("ResolveInt on a bool") - } - if _, err := r.ResolveBool(arrRef); err == nil { - t.Error("ResolveBool on an array") - } - if _, err := r.ResolveArray(intRef); err == nil { - t.Error("ResolveArray on an int") - } - - if r.Trailer() == nil { - t.Fatal("nil trailer") - } - if _, ok := r.Trailer().Get("Root"); !ok { - t.Error("trailer missing /Root") - } -} - -// A catalog /Version higher than the header version wins (PDF 32000-1 §7.5.5). -func TestVersionCatalogOverride(t *testing.T) { - data := buildDictPDF(t, []string{ - "<< /Type /Catalog /Pages 2 0 R /Version /2.0 >>", - "<< /Type /Pages /Kids [] /Count 0 >>", - }) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if _, err := r.Catalog(); err != nil { // Version() reads the cached catalog - t.Fatalf("Catalog: %v", err) - } - if v := r.Version(); v != "2.0" { - t.Errorf("Version = %q, want 2.0 (catalog override)", v) - } -} - -// A non-Reference /Length resolves to itself, so a bare Reader (no xref) drives -// the missing/non-integer/negative guards directly. -func TestStreamLengthRejectsBadLength(t *testing.T) { - r := &Reader{} - bad := []struct { - name string - set func(d *Dict) - }{ - {"missing", func(d *Dict) {}}, - {"non_integer", func(d *Dict) { d.set("Length", String("x")) }}, - {"negative", func(d *Dict) { d.set("Length", Integer(-5)) }}, - } - for _, tc := range bad { - t.Run(tc.name, func(t *testing.T) { - d := newDict(nil) - tc.set(d) - if _, err := r.streamLength(d); err == nil { - t.Fatal("expected an error, got nil") - } - }) - } - - // Control: a valid non-negative /Length must still resolve. - t.Run("valid", func(t *testing.T) { - d := newDict(nil) - d.set("Length", Integer(42)) - if n, err := r.streamLength(d); err != nil || n != 42 { - t.Fatalf("streamLength = %d, %v; want 42, nil", n, err) - } - }) -} - -// A bare Reader works because a non-Reference value resolves to itself; each -// helper must error on the wrong type, not pass back a zero value as success. -func TestResolveHelperTypeErrors(t *testing.T) { - r := &Reader{} - errCases := []struct { - name string - call func() error - }{ - {"dict_from_int", func() error { _, e := r.ResolveDict(Integer(1)); return e }}, - {"dict_from_null", func() error { _, e := r.ResolveDict(Null{}); return e }}, - {"dict_from_nil", func() error { _, e := r.ResolveDict(nil); return e }}, - {"bool_from_int", func() error { _, e := r.ResolveBool(Integer(1)); return e }}, - {"int_from_bool", func() error { _, e := r.ResolveInt(Bool(true)); return e }}, - {"array_from_int", func() error { _, e := r.ResolveArray(Integer(1)); return e }}, - {"stream_from_int", func() error { _, e := r.DecodeStream(Integer(1)); return e }}, - } - for _, tc := range errCases { - if tc.call() == nil { - t.Errorf("%s: expected an error, got nil", tc.name) - } - } - // Controls: the right type resolves cleanly. - if b, err := r.ResolveBool(Bool(true)); err != nil || !b { - t.Errorf("ResolveBool(true) = %v, %v", b, err) - } - if n, err := r.ResolveInt(Integer(7)); err != nil || n != 7 { - t.Errorf("ResolveInt(7) = %v, %v", n, err) - } - if a, err := r.ResolveArray(Array{Integer(1)}); err != nil || len(a) != 1 { - t.Errorf("ResolveArray = %v, %v", a, err) - } -} - -// OpenFile must surface the os.Open error for a missing path, and must not leak -// the descriptor when the file opens but doesn't parse as a PDF. -func TestOpenFileErrors(t *testing.T) { - if _, err := OpenFile(filepath.Join(t.TempDir(), "missing.pdf")); err == nil { - t.Error("OpenFile(missing) should error") - } - bad := filepath.Join(t.TempDir(), "bad.pdf") - if err := os.WriteFile(bad, []byte("not a pdf"), 0o644); err != nil { - t.Fatal(err) - } - if _, err := OpenFile(bad); err == nil { - t.Error("OpenFile(garbage) should error") - } -} - -func TestXrefFormat(t *testing.T) { - withTrailer := func(set func(d *Dict)) *Reader { - d := newDict(nil) - set(d) - return &Reader{trailer: d} - } - if got := (&Reader{}).xrefFormat(); got != "unknown" { - t.Errorf("nil trailer = %q, want unknown", got) - } - if got := withTrailer(func(d *Dict) { d.set("Type", Name("XRef")) }).xrefFormat(); got != "stream" { - t.Errorf("/Type /XRef = %q, want stream", got) - } - if got := withTrailer(func(d *Dict) { d.set("XRefStm", Integer(99)) }).xrefFormat(); got != "hybrid" { - t.Errorf("/XRefStm = %q, want hybrid", got) - } - if got := withTrailer(func(d *Dict) { d.set("Size", Integer(4)) }).xrefFormat(); got != "classical" { - t.Errorf("plain trailer = %q, want classical", got) - } -} - -// Non-standard /Info keys land in Custom; a non-string entry is skipped, not -// rendered. -func TestDocumentInfoRichFields(t *testing.T) { - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - offsets := make([]int, 4) - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - offsets[3] = off() - fmt.Fprint(&buf, "3 0 obj\n<< /Author (Ada) /Subject (Math) /Keywords (a,b) "+ - "/Creator (X) /Producer (Y) /CreationDate (D:20200102030405Z) "+ - "/ModDate (D:20210102030405Z) /Custom (cval) /NotAString 42 >>\nendobj\n") - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 4\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 3; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprint(&buf, "trailer\n<< /Size 4 /Root 1 0 R /Info 3 0 R >>\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - - r, err := Open(bytes.NewReader(buf.Bytes())) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - info := r.DocumentInfo() - if info.Author != "Ada" || info.Subject != "Math" || info.Keywords != "a,b" || - info.Creator != "X" || info.Producer != "Y" { - t.Errorf("string fields wrong: %+v", info) - } - if info.CreationDate.Year() != 2020 || info.ModDate.Year() != 2021 { - t.Errorf("dates wrong: created %v, mod %v", info.CreationDate, info.ModDate) - } - if info.Custom["Custom"] != "cval" { - t.Errorf("Custom[Custom] = %q, want cval", info.Custom["Custom"]) - } - if _, ok := info.Custom["NotAString"]; ok { - t.Error("non-string /NotAString should be skipped, not collected") - } -} - -func TestObjectsIteration(t *testing.T) { - r, err := Open(bytes.NewReader(buildMinimalPDF(t))) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - var nums []int - for e := range r.Objects() { - nums = append(nums, e.Reference.Number) - } - if len(nums) < 4 { - t.Fatalf("iterated %d objects, want >= 4", len(nums)) - } - for i := 1; i < len(nums); i++ { - if nums[i] < nums[i-1] { - t.Errorf("objects out of order: %d before %d", nums[i-1], nums[i]) - } - } - count := 0 - for range r.Objects() { - count++ - break - } - if count != 1 { - t.Fatalf("early break iterated %d, want 1", count) - } -} - -// A reference to an object absent from the xref table resolves to null per -// §7.3.10, not an error. -func TestResolveDanglingReference(t *testing.T) { - r, err := Open(bytes.NewReader(buildMinimalPDF(t))) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - v, err := r.Resolve(Reference{Number: 99}) - if err != nil { - t.Fatalf("dangling ref: %v", err) - } - if _, ok := v.(Null); !ok { - t.Errorf("dangling ref = %T, want Null", v) - } -} diff --git a/testdata/fixtures/brotli-stream/golden.json b/testdata/fixtures/brotli-stream/golden.json deleted file mode 100644 index c7b0236..0000000 --- a/testdata/fixtures/brotli-stream/golden.json +++ /dev/null @@ -1,84 +0,0 @@ -{ - "version": "2.0", - "xref_format": "classical", - "encrypted": false, - "trailer": { - "dict": { - "Size": { - "int": 4 - }, - "Root": { - "ref": "1 0" - } - } - }, - "objects": { - "1 0": { - "dict": { - "Type": { - "name": "Catalog" - }, - "Pages": { - "ref": "2 0" - }, - "Extensions": { - "dict": { - "PDFa": { - "dict": { - "Type": { - "name": "DeveloperExtensions" - }, - "BaseVersion": { - "name": "2.0" - }, - "ExtensionLevel": { - "int": 1 - }, - "ExtensionRevision": { - "text": "2026" - }, - "URL": { - "text": "https://pdfa.org/resource/extension-brotli" - } - } - } - } - } - } - }, - "2 0": { - "dict": { - "Type": { - "name": "Pages" - }, - "Count": { - "int": 0 - }, - "Kids": { - "array": [] - } - } - }, - "3 0": { - "stream": { - "dict": { - "Length": { - "int": 25 - }, - "Filter": { - "name": "BrotliDecode" - } - }, - "raw_length": 25, - "filters": [ - "BrotliDecode" - ], - "decoded": { - "length": 21, - "sha256": "278aa4570e132c7f29e5fb147da4b8f6f4405b77ada770d5623e7e52cbf51dde", - "preview_utf8": "Hello, Brotli stream!" - } - } - } - } -} diff --git a/testdata/fixtures/brotli-stream/input.pdf b/testdata/fixtures/brotli-stream/input.pdf deleted file mode 100644 index 635e746..0000000 Binary files a/testdata/fixtures/brotli-stream/input.pdf and /dev/null differ diff --git a/testdata/fixtures/flate-stream/golden.json b/testdata/fixtures/flate-stream/golden.json deleted file mode 100644 index 489bc10..0000000 --- a/testdata/fixtures/flate-stream/golden.json +++ /dev/null @@ -1,61 +0,0 @@ -{ - "version": "1.7", - "xref_format": "classical", - "encrypted": false, - "trailer": { - "dict": { - "Size": { - "int": 4 - }, - "Root": { - "ref": "1 0" - } - } - }, - "objects": { - "1 0": { - "dict": { - "Type": { - "name": "Catalog" - }, - "Pages": { - "ref": "2 0" - } - } - }, - "2 0": { - "dict": { - "Type": { - "name": "Pages" - }, - "Count": { - "int": 0 - }, - "Kids": { - "array": [] - } - } - }, - "3 0": { - "stream": { - "dict": { - "Length": { - "int": 26 - }, - "Filter": { - "name": "FlateDecode" - } - }, - "raw_length": 26, - "filters": [ - "FlateDecode" - ], - "decoded": { - "length": 14, - "sha256": "4ac9a1927af81c4e7cd1b812848d3fdac958b613e4754a88af93b308df4ac7df", - "preview_utf8": "Hello, stream!" - } - } - } - } -} diff --git a/testdata/fixtures/flate-stream/input.pdf b/testdata/fixtures/flate-stream/input.pdf deleted file mode 100644 index efa6053..0000000 Binary files a/testdata/fixtures/flate-stream/input.pdf and /dev/null differ diff --git a/testdata/fixtures/generate.go b/testdata/fixtures/generate.go deleted file mode 100644 index 075166e..0000000 --- a/testdata/fixtures/generate.go +++ /dev/null @@ -1,234 +0,0 @@ -//go:build ignore - -// Run from the repo root: -// -// go run testdata/fixtures/generate.go -// -// (re)creates synthetic input.pdf files for the fixtures that we author -// in code (rather than dropping in real-world samples). After running, -// refresh goldens with: -// -// go test -update -run TestFixtures -// -// Then inspect the resulting golden.json before committing. -package main - -import ( - "bytes" - "compress/zlib" - "fmt" - "os" - "path/filepath" - - "github.com/andybalholm/brotli" -) - -func main() { - write("minimal", minimalPDF()) - write("xref-stream", xrefStreamPDF()) - write("flate-stream", flateStreamPDF()) - write("brotli-stream", brotliStreamPDF()) - write("page-inheritance", pageInheritancePDF()) - write("page-contents-array", pageContentsArrayPDF()) -} - -func write(name string, data []byte) { - dir := filepath.Join("testdata/fixtures", name) - if err := os.MkdirAll(dir, 0o755); err != nil { - panic(err) - } - path := filepath.Join(dir, "input.pdf") - if err := os.WriteFile(path, data, 0o644); err != nil { - panic(err) - } - fmt.Printf("wrote %s (%d bytes)\n", path, len(data)) -} - -func minimalPDF() []byte { - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - - offsets := make([]int, 5) - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>\nendobj\n") - offsets[3] = off() - fmt.Fprint(&buf, "3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] >>\nendobj\n") - offsets[4] = off() - fmt.Fprint(&buf, "4 0 obj\n<< /Title (Hello) /Producer (pdfdisassembler-fixture) >>\nendobj\n") - - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 5\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 4; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprintf(&buf, "trailer\n<< /Size 5 /Root 1 0 R /Info 4 0 R >>\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -func xrefStreamPDF() []byte { - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-2.0\n%\xE2\xE3\xCF\xD3\n") - - offsets := make([]int, 3) - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - - rows := []byte{} - add := func(typ, f1, f2 uint64) { - rows = append(rows, byte(typ)) - rows = append(rows, byte(f1>>16), byte(f1>>8), byte(f1)) - rows = append(rows, byte(f2)) - } - add(0, 0, 0xFFFF) - add(1, uint64(offsets[1]), 0) - add(1, uint64(offsets[2]), 0) - - var zbuf bytes.Buffer - zw := zlib.NewWriter(&zbuf) - zw.Write(rows) - zw.Close() - compressed := zbuf.Bytes() - - xrefOff := off() - fmt.Fprintf(&buf, - "3 0 obj\n<< /Type /XRef /Size 3 /W [1 3 1] /Root 1 0 R /Filter /FlateDecode /Length %d >>\nstream\n", - len(compressed)) - buf.Write(compressed) - fmt.Fprint(&buf, "\nendstream\nendobj\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -// classicalPDF assembles objs (1-based bodies) into a PDF with a classical -// xref table and the given trailer dictionary body (without the surrounding -// << >>). -func classicalPDF(version string, objs []string, trailer string) []byte { - var buf bytes.Buffer - fmt.Fprintf(&buf, "%%PDF-%s\n%%\xE2\xE3\xCF\xD3\n", version) - offsets := make([]int, len(objs)+1) - for i, body := range objs { - offsets[i+1] = buf.Len() - fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", i+1, body) - } - xrefOff := buf.Len() - fmt.Fprintf(&buf, "xref\n0 %d\n%010d %05d f \n", len(objs)+1, 0, 65535) - for i := 1; i <= len(objs); i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprintf(&buf, "trailer\n<< /Size %d %s >>\nstartxref\n%d\n%%%%EOF\n", - len(objs)+1, trailer, xrefOff) - return buf.Bytes() -} - -// pageInheritancePDF builds a three-level page tree where leaf pages inherit -// /MediaBox and /Rotate two levels up (from the /Pages root), /CropBox and -// /Resources one level up, and the second leaf overrides /MediaBox and /Rotate -// locally. It exercises attribute inheritance per PDF 32000-1:2008 §7.7.3.4. -func pageInheritancePDF() []byte { - return classicalPDF("1.7", []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 2 /Kids [ 3 0 R ] /MediaBox [0 0 612 792] /Resources << /Font << /F1 6 0 R >> >> /Rotate 90 >>", - "<< /Type /Pages /Count 2 /Parent 2 0 R /Kids [ 4 0 R 5 0 R ] /CropBox [10 10 602 782] >>", - "<< /Type /Page /Parent 3 0 R >>", - "<< /Type /Page /Parent 3 0 R /MediaBox [0 0 200 200] /Rotate 0 >>", - "<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>", - }, "/Root 1 0 R") -} - -// pageContentsArrayPDF builds a single page whose /Contents is an array of two -// uncompressed content streams that must be concatenated. -func pageContentsArrayPDF() []byte { - const c1 = "q 1 0 0 1 50 50 cm" - const c2 = "BT /F1 12 Tf (Hello) Tj ET" - return classicalPDF("1.7", []string{ - "<< /Type /Catalog /Pages 2 0 R >>", - "<< /Type /Pages /Count 1 /Kids [ 3 0 R ] >>", - "<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] /Contents [ 4 0 R 5 0 R ] >>", - fmt.Sprintf("<< /Length %d >>\nstream\n%s\nendstream", len(c1), c1), - fmt.Sprintf("<< /Length %d >>\nstream\n%s\nendstream", len(c2), c2), - }, "/Root 1 0 R") -} - -// brotliStreamPDF mirrors flateStreamPDF with a BrotliDecode stream (PDF -// Association extension EXTN-BROTLI-1). Per that spec the catalog SHOULD -// declare the extension in an /Extensions dictionary under the PDFa prefix. -func brotliStreamPDF() []byte { - const payload = "Hello, Brotli stream!" - var bbuf bytes.Buffer - bw := brotli.NewWriter(&bbuf) - if _, err := bw.Write([]byte(payload)); err != nil { - panic(err) - } - if err := bw.Close(); err != nil { - panic(err) - } - bdata := bbuf.Bytes() - - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-2.0\n%\xE2\xE3\xCF\xD3\n") - - offsets := make([]int, 4) - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R "+ - "/Extensions << /PDFa << /Type /DeveloperExtensions /BaseVersion /2.0 "+ - "/ExtensionLevel 1 /ExtensionRevision (2026) "+ - "/URL (https://pdfa.org/resource/extension-brotli) >> >> >>\nendobj\n") - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - offsets[3] = off() - fmt.Fprintf(&buf, "3 0 obj\n<< /Length %d /Filter /BrotliDecode >>\nstream\n", len(bdata)) - buf.Write(bdata) - fmt.Fprint(&buf, "\nendstream\nendobj\n") - - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 4\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 3; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprintf(&buf, "trailer\n<< /Size 4 /Root 1 0 R >>\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -func flateStreamPDF() []byte { - const payload = "Hello, stream!" - var zbuf bytes.Buffer - zw := zlib.NewWriter(&zbuf) - zw.Write([]byte(payload)) - zw.Close() - zdata := zbuf.Bytes() - - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-1.7\n%\xE2\xE3\xCF\xD3\n") - - offsets := make([]int, 4) - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - offsets[3] = off() - fmt.Fprintf(&buf, "3 0 obj\n<< /Length %d /Filter /FlateDecode >>\nstream\n", len(zdata)) - buf.Write(zdata) - fmt.Fprint(&buf, "\nendstream\nendobj\n") - - xrefOff := off() - fmt.Fprint(&buf, "xref\n0 4\n") - fmt.Fprintf(&buf, "%010d %05d f \n", 0, 65535) - for i := 1; i <= 3; i++ { - fmt.Fprintf(&buf, "%010d %05d n \n", offsets[i], 0) - } - fmt.Fprintf(&buf, "trailer\n<< /Size 4 /Root 1 0 R >>\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} diff --git a/testdata/fixtures/minimal/golden.json b/testdata/fixtures/minimal/golden.json deleted file mode 100644 index 4bce377..0000000 --- a/testdata/fixtures/minimal/golden.json +++ /dev/null @@ -1,83 +0,0 @@ -{ - "version": "1.7", - "xref_format": "classical", - "encrypted": false, - "trailer": { - "dict": { - "Size": { - "int": 5 - }, - "Root": { - "ref": "1 0" - }, - "Info": { - "ref": "4 0" - } - } - }, - "objects": { - "1 0": { - "dict": { - "Type": { - "name": "Catalog" - }, - "Pages": { - "ref": "2 0" - } - } - }, - "2 0": { - "dict": { - "Type": { - "name": "Pages" - }, - "Count": { - "int": 1 - }, - "Kids": { - "array": [ - { - "ref": "3 0" - } - ] - } - } - }, - "3 0": { - "dict": { - "Type": { - "name": "Page" - }, - "Parent": { - "ref": "2 0" - }, - "MediaBox": { - "array": [ - { - "int": 0 - }, - { - "int": 0 - }, - { - "int": 612 - }, - { - "int": 792 - } - ] - } - } - }, - "4 0": { - "dict": { - "Title": { - "text": "Hello" - }, - "Producer": { - "text": "pdfdisassembler-fixture" - } - } - } - } -} diff --git a/testdata/fixtures/minimal/input.pdf b/testdata/fixtures/minimal/input.pdf deleted file mode 100644 index 9a46979..0000000 Binary files a/testdata/fixtures/minimal/input.pdf and /dev/null differ diff --git a/testdata/fixtures/page-contents-array/golden.json b/testdata/fixtures/page-contents-array/golden.json deleted file mode 100644 index 3028856..0000000 --- a/testdata/fixtures/page-contents-array/golden.json +++ /dev/null @@ -1,112 +0,0 @@ -{ - "version": "1.7", - "xref_format": "classical", - "encrypted": false, - "trailer": { - "dict": { - "Size": { - "int": 6 - }, - "Root": { - "ref": "1 0" - } - } - }, - "objects": { - "1 0": { - "dict": { - "Type": { - "name": "Catalog" - }, - "Pages": { - "ref": "2 0" - } - } - }, - "2 0": { - "dict": { - "Type": { - "name": "Pages" - }, - "Count": { - "int": 1 - }, - "Kids": { - "array": [ - { - "ref": "3 0" - } - ] - } - } - }, - "3 0": { - "dict": { - "Type": { - "name": "Page" - }, - "Parent": { - "ref": "2 0" - }, - "MediaBox": { - "array": [ - { - "int": 0 - }, - { - "int": 0 - }, - { - "int": 200 - }, - { - "int": 200 - } - ] - }, - "Contents": { - "array": [ - { - "ref": "4 0" - }, - { - "ref": "5 0" - } - ] - } - } - }, - "4 0": { - "stream": { - "dict": { - "Length": { - "int": 18 - } - }, - "raw_length": 18, - "filters": [], - "decoded": { - "length": 18, - "sha256": "ac399976dc867824f50e13f5f86d9f5b9be76d49ae538a01d7567552914e9f80", - "preview_utf8": "q 1 0 0 1 50 50 cm" - } - } - }, - "5 0": { - "stream": { - "dict": { - "Length": { - "int": 26 - } - }, - "raw_length": 26, - "filters": [], - "decoded": { - "length": 26, - "sha256": "9dd85240377330fc34370ec0a297930e7cba5a30efbaab71f21337dbfc1d864e", - "preview_utf8": "BT /F1 12 Tf (Hello) Tj ET" - } - } - } - } -} diff --git a/testdata/fixtures/page-contents-array/input.pdf b/testdata/fixtures/page-contents-array/input.pdf deleted file mode 100644 index 76a6970..0000000 Binary files a/testdata/fixtures/page-contents-array/input.pdf and /dev/null differ diff --git a/testdata/fixtures/page-inheritance/golden.json b/testdata/fixtures/page-inheritance/golden.json deleted file mode 100644 index 0922e2e..0000000 --- a/testdata/fixtures/page-inheritance/golden.json +++ /dev/null @@ -1,165 +0,0 @@ -{ - "version": "1.7", - "xref_format": "classical", - "encrypted": false, - "trailer": { - "dict": { - "Size": { - "int": 7 - }, - "Root": { - "ref": "1 0" - } - } - }, - "objects": { - "1 0": { - "dict": { - "Type": { - "name": "Catalog" - }, - "Pages": { - "ref": "2 0" - } - } - }, - "2 0": { - "dict": { - "Type": { - "name": "Pages" - }, - "Count": { - "int": 2 - }, - "Kids": { - "array": [ - { - "ref": "3 0" - } - ] - }, - "MediaBox": { - "array": [ - { - "int": 0 - }, - { - "int": 0 - }, - { - "int": 612 - }, - { - "int": 792 - } - ] - }, - "Resources": { - "dict": { - "Font": { - "dict": { - "F1": { - "ref": "6 0" - } - } - } - } - }, - "Rotate": { - "int": 90 - } - } - }, - "3 0": { - "dict": { - "Type": { - "name": "Pages" - }, - "Count": { - "int": 2 - }, - "Parent": { - "ref": "2 0" - }, - "Kids": { - "array": [ - { - "ref": "4 0" - }, - { - "ref": "5 0" - } - ] - }, - "CropBox": { - "array": [ - { - "int": 10 - }, - { - "int": 10 - }, - { - "int": 602 - }, - { - "int": 782 - } - ] - } - } - }, - "4 0": { - "dict": { - "Type": { - "name": "Page" - }, - "Parent": { - "ref": "3 0" - } - } - }, - "5 0": { - "dict": { - "Type": { - "name": "Page" - }, - "Parent": { - "ref": "3 0" - }, - "MediaBox": { - "array": [ - { - "int": 0 - }, - { - "int": 0 - }, - { - "int": 200 - }, - { - "int": 200 - } - ] - }, - "Rotate": { - "int": 0 - } - } - }, - "6 0": { - "dict": { - "Type": { - "name": "Font" - }, - "Subtype": { - "name": "Type1" - }, - "BaseFont": { - "name": "Helvetica" - } - } - } - } -} diff --git a/testdata/fixtures/page-inheritance/input.pdf b/testdata/fixtures/page-inheritance/input.pdf deleted file mode 100644 index e293dec..0000000 Binary files a/testdata/fixtures/page-inheritance/input.pdf and /dev/null differ diff --git a/testdata/fixtures/pdfua-demo/README.md b/testdata/fixtures/pdfua-demo/README.md deleted file mode 100644 index 5729ce6..0000000 --- a/testdata/fixtures/pdfua-demo/README.md +++ /dev/null @@ -1,22 +0,0 @@ -# pdfua-demo - -A PDF/UA document produced by [glu](https://boxesandglue.dev) (the -boxesandglue typesetter), originally part of the speedata Marketing -material `glu-strategie/pdfua-demo.pdf`. Single-page, tagged via -Markdown front matter. - -Covers in one fixture: - -- Classical xref -- `/StructTreeRoot` with H1/H2/P/L/LI/LBody/BlockQuote/Code/Figure tags -- `/RoleMap` (none — the doc uses standard structure types) -- `/MarkInfo`, `/Lang`, `/ViewerPreferences` -- `/Metadata` XMP stream -- `/ParentTree` with mixed `Nums` array (int + ref + int + array of refs) -- Type0 / CIDFont / ToUnicode CMap fonts -- FlateDecode content stream with PDF/UA marked-content operators - (`/H1 <> BDC` etc.) -- `ActualText` entries in PDFDocEncoding with the en-dash (`0x85`) -- A binary file ID (`/ID` array) - -This is what we want our parser to be able to read losslessly. diff --git a/testdata/fixtures/pdfua-demo/golden.json b/testdata/fixtures/pdfua-demo/golden.json deleted file mode 100644 index ef80cf7..0000000 --- a/testdata/fixtures/pdfua-demo/golden.json +++ /dev/null @@ -1,3204 +0,0 @@ -{ - "version": "1.7", - "xref_format": "classical", - "encrypted": false, - "trailer": { - "dict": { - "ID": { - "array": [ - { - "hex": "62a379b5a12ce31bef329a1f37876600" - }, - { - "hex": "62a379b5a12ce31bef329a1f37876600" - } - ] - }, - "Info": { - "ref": "59 0" - }, - "Root": { - "ref": "38 0" - }, - "Size": { - "int": 60 - } - } - }, - "objects": { - "1 0": { - "dict": { - "Type": { - "name": "Font" - }, - "BaseFont": { - "name": "BZDLFS+CrimsonPro-Regular" - }, - "DescendantFonts": { - "array": [ - { - "ref": "42 0" - } - ] - }, - "Encoding": { - "name": "Identity-H" - }, - "Subtype": { - "name": "Type0" - }, - "ToUnicode": { - "ref": "41 0" - } - } - }, - "2 0": { - "dict": { - "Type": { - "name": "Font" - }, - "BaseFont": { - "name": "LNQEXG+CrimsonPro-Bold" - }, - "DescendantFonts": { - "array": [ - { - "ref": "46 0" - } - ] - }, - "Encoding": { - "name": "Identity-H" - }, - "Subtype": { - "name": "Type0" - }, - "ToUnicode": { - "ref": "45 0" - } - } - }, - "3 0": { - "dict": { - "Type": { - "name": "Font" - }, - "BaseFont": { - "name": "AVREHH+CrimsonPro-Italic" - }, - "DescendantFonts": { - "array": [ - { - "ref": "50 0" - } - ] - }, - "Encoding": { - "name": "Identity-H" - }, - "Subtype": { - "name": "Type0" - }, - "ToUnicode": { - "ref": "49 0" - } - } - }, - "4 0": { - "dict": { - "Type": { - "name": "Font" - }, - "BaseFont": { - "name": "JYZTRN+CamingoCode-Bold" - }, - "DescendantFonts": { - "array": [ - { - "ref": "54 0" - } - ] - }, - "Encoding": { - "name": "Identity-H" - }, - "Subtype": { - "name": "Type0" - }, - "ToUnicode": { - "ref": "53 0" - } - } - }, - "5 0": { - "dict": { - "Type": { - "name": "Font" - }, - "BaseFont": { - "name": "YCXYQB+CamingoCode-Regular" - }, - "DescendantFonts": { - "array": [ - { - "ref": "58 0" - } - ] - }, - "Encoding": { - "name": "Identity-H" - }, - "Subtype": { - "name": "Type0" - }, - "ToUnicode": { - "ref": "57 0" - } - } - }, - "6 0": { - "dict": { - "Type": { - "name": "Page" - }, - "Contents": { - "ref": "7 0" - }, - "Parent": { - "ref": "32 0" - }, - "Resources": { - "dict": { - "Font": { - "dict": { - "F1": { - "ref": "1 0" - }, - "F3": { - "ref": "2 0" - }, - "F4": { - "ref": "3 0" - }, - "F5": { - "ref": "4 0" - }, - "F6": { - "ref": "5 0" - } - } - }, - "XObject": { - "dict": { - "ImgBag2": { - "ref": "8 0" - } - } - } - } - }, - "StructParents": { - "int": 1 - }, - "TrimBox": { - "array": [ - { - "int": 0 - }, - { - "int": 0 - }, - { - "real": 595.28 - }, - { - "real": 841.89 - } - ] - } - } - }, - "7 0": { - "stream": { - "dict": { - "Filter": { - "name": "FlateDecode" - }, - "Length": { - "int": 2737 - }, - "Length1": { - "int": 14503 - } - }, - "raw_length": 2737, - "filters": [ - "FlateDecode" - ], - "decoded": { - "length": 14503, - "sha256": "5cf965cecc849327e8aff22af072173c9566363887f142c8626d60c8a237e45a", - "preview_utf8": "/Artifact BMC\n\nEMC\n/H1 <> BDC\n0.05 0.23 0.4 rg BT 100 Tz 0 Ts \n/F3 22 T…" - } - } - }, - "8 0": { - "stream": { - "dict": { - "Filter": { - "name": "FlateDecode" - }, - "Type": { - "name": "XObject" - }, - "Subtype": { - "name": "Form" - }, - "FormType": { - "int": 1 - }, - "BBox": { - "array": [ - { - "real": 0 - }, - { - "real": 0 - }, - { - "real": 841.89 - }, - { - "real": 595.28 - } - ] - }, - "StructParent": { - "int": 0 - }, - "Resources": { - "dict": { - "ExtGState": { - "dict": { - "a0": { - "dict": { - "CA": { - "int": 1 - }, - "ca": { - "int": 1 - } - } - } - } - } - } - }, - "Length": { - "int": 68069 - } - }, - "raw_length": 68069, - "filters": [ - "FlateDecode" - ], - "decoded": { - "length": 171668, - "sha256": "f313e213218ea6cc57612fe8ad542b0d00918f3ad2c76bab1aba18f0fe142049", - "preview_utf8": "q\n0 0 0 rg /a0 gs\n462.723 162.502 m 447.234 162.502 434.492 150.619 433.027 135.…" - } - } - }, - "9 0": { - "dict": { - "Type": { - "name": "StructTreeRoot" - }, - "K": { - "ref": "10 0" - }, - "ParentTree": { - "dict": { - "Nums": { - "array": [ - { - "int": 0 - }, - { - "ref": "30 0" - }, - { - "int": 1 - }, - { - "array": [ - { - "ref": "11 0" - }, - { - "ref": "12 0" - }, - { - "ref": "13 0" - }, - { - "ref": "14 0" - }, - { - "ref": "17 0" - }, - { - "ref": "19 0" - }, - { - "ref": "21 0" - }, - { - "ref": "23 0" - }, - { - "ref": "25 0" - }, - { - "ref": "26 0" - }, - { - "ref": "28 0" - }, - { - "ref": "29 0" - } - ] - } - ] - } - } - } - } - }, - "10 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "K": { - "array": [ - { - "ref": "11 0" - }, - { - "ref": "12 0" - }, - { - "ref": "13 0" - }, - { - "ref": "14 0" - }, - { - "ref": "15 0" - }, - { - "ref": "24 0" - }, - { - "ref": "26 0" - }, - { - "ref": "27 0" - }, - { - "ref": "29 0" - }, - { - "ref": "30 0" - } - ] - }, - "P": { - "ref": "9 0" - }, - "S": { - "name": "Document" - } - } - }, - "11 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "Markdown to PDF/UA" - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 0 - } - } - }, - "P": { - "ref": "10 0" - }, - "S": { - "name": "H1" - } - } - }, - "12 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "This page was typeset from Markdown – and is fully tagged." - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 1 - } - } - }, - "P": { - "ref": "10 0" - }, - "S": { - "name": "P" - } - } - }, - "13 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "Rendered with glu (boxes and glue). The YAML front matter declares format: PDF/UA, lang: en and a title: – that is all it takes. From those three lines glu automatically writes StructTreeRoot, MarkInfo, /DisplayDocTitle, /Lang and the XMP tag pdfuaid:part 1." - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 2 - } - } - }, - "P": { - "ref": "10 0" - }, - "S": { - "name": "P" - } - } - }, - "14 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "What is structurally tagged here" - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 3 - } - } - }, - "P": { - "ref": "10 0" - }, - "S": { - "name": "H2" - } - } - }, - "15 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "K": { - "array": [ - { - "ref": "16 0" - }, - { - "ref": "18 0" - }, - { - "ref": "20 0" - }, - { - "ref": "22 0" - } - ] - }, - "P": { - "ref": "10 0" - }, - "S": { - "name": "L" - } - } - }, - "16 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "K": { - "ref": "17 0" - }, - "P": { - "ref": "15 0" - }, - "S": { - "name": "LI" - } - } - }, - "17 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "Headings build the outline tree (H1, H2)." - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 4 - } - } - }, - "P": { - "ref": "16 0" - }, - "S": { - "name": "LBody" - } - } - }, - "18 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "K": { - "ref": "19 0" - }, - "P": { - "ref": "15 0" - }, - "S": { - "name": "LI" - } - } - }, - "19 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "Paragraphs become P, lists become L / LI." - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 5 - } - } - }, - "P": { - "ref": "18 0" - }, - "S": { - "name": "LBody" - } - } - }, - "20 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "K": { - "ref": "21 0" - }, - "P": { - "ref": "15 0" - }, - "S": { - "name": "LI" - } - } - }, - "21 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "The figure below is a Figure element with /Alt text." - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 6 - } - } - }, - "P": { - "ref": "20 0" - }, - "S": { - "name": "LBody" - } - } - }, - "22 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "K": { - "ref": "23 0" - }, - "P": { - "ref": "15 0" - }, - "S": { - "name": "LI" - } - } - }, - "23 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "Code blocks are emitted as Code." - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 7 - } - } - }, - "P": { - "ref": "22 0" - }, - "S": { - "name": "LBody" - } - } - }, - "24 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "K": { - "ref": "25 0" - }, - "P": { - "ref": "10 0" - }, - "S": { - "name": "BlockQuote" - } - } - }, - "25 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "Acrobat, NVDA and PAC all see this as a properly structured document – no artifacts, no untagged content." - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 8 - } - } - }, - "P": { - "ref": "24 0" - }, - "S": { - "name": "P" - } - } - }, - "26 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "Front matter" - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 9 - } - } - }, - "P": { - "ref": "10 0" - }, - "S": { - "name": "H2" - } - } - }, - "27 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "K": { - "ref": "28 0" - }, - "P": { - "ref": "10 0" - }, - "S": { - "name": "P" - } - } - }, - "28 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "---\ntitle: Markdown to PDF/UA with glu\nlang: en\nformat: PDF/UA\n---" - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 10 - } - } - }, - "P": { - "ref": "27 0" - }, - "S": { - "name": "Code" - } - } - }, - "29 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "ActualText": { - "text": "An embedded figure" - }, - "K": { - "dict": { - "Type": { - "name": "MCR" - }, - "Pg": { - "ref": "6 0" - }, - "MCID": { - "int": 11 - } - } - }, - "P": { - "ref": "10 0" - }, - "S": { - "name": "H2" - } - } - }, - "30 0": { - "dict": { - "Type": { - "name": "StructElem" - }, - "A": { - "dict": { - "O": { - "name": "Layout" - }, - "Placement": { - "name": "Block" - }, - "BBox": { - "array": [ - { - "real": 68.03 - }, - { - "real": 488.22 - }, - { - "real": 527.24 - }, - { - "real": 628.52 - } - ] - } - } - }, - "Alt": { - "text": "Photograph of the sea, embedded as a Form XObject" - }, - "K": { - "dict": { - "Type": { - "name": "OBJR" - }, - "Obj": { - "ref": "8 0" - }, - "Pg": { - "ref": "6 0" - } - } - }, - "P": { - "ref": "10 0" - }, - "S": { - "name": "Figure" - } - } - }, - "31 0": { - "stream": { - "dict": { - "Type": { - "name": "Metadata" - }, - "Length": { - "int": 1229 - }, - "Subtype": { - "name": "XML" - } - }, - "raw_length": 1229, - "filters": [], - "decoded": { - "length": 1229, - "sha256": "8f2ced3e481012ce7c7c6b2ac15d96d81161c60ab2ee94d6f60caf5566cb8d53" - } - } - }, - "32 0": { - "dict": { - "Type": { - "name": "Pages" - }, - "Count": { - "int": 1 - }, - "Kids": { - "array": [ - { - "ref": "6 0" - } - ] - }, - "MediaBox": { - "array": [ - { - "int": 0 - }, - { - "int": 0 - }, - { - "real": 595.28 - }, - { - "real": 841.89 - } - ] - } - } - }, - "33 0": { - "dict": { - "Type": { - "name": "Outlines" - }, - "Count": { - "int": 4 - }, - "First": { - "ref": "34 0" - }, - "Last": { - "ref": "37 0" - } - } - }, - "34 0": { - "dict": { - "Dest": { - "array": [ - { - "ref": "6 0" - }, - { - "name": "Fit" - } - ] - }, - "Next": { - "ref": "35 0" - }, - "Parent": { - "ref": "33 0" - }, - "Title": { - "text": "Markdown to PDF/UA" - } - } - }, - "35 0": { - "dict": { - "Dest": { - "array": [ - { - "ref": "6 0" - }, - { - "name": "Fit" - } - ] - }, - "Next": { - "ref": "36 0" - }, - "Parent": { - "ref": "33 0" - }, - "Prev": { - "ref": "34 0" - }, - "Title": { - "text": "What is structurally tagged here" - } - } - }, - "36 0": { - "dict": { - "Dest": { - "array": [ - { - "ref": "6 0" - }, - { - "name": "Fit" - } - ] - }, - "Next": { - "ref": "37 0" - }, - "Parent": { - "ref": "33 0" - }, - "Prev": { - "ref": "35 0" - }, - "Title": { - "text": "Front matter" - } - } - }, - "37 0": { - "dict": { - "Dest": { - "array": [ - { - "ref": "6 0" - }, - { - "name": "Fit" - } - ] - }, - "Parent": { - "ref": "33 0" - }, - "Prev": { - "ref": "36 0" - }, - "Title": { - "text": "An embedded figure" - } - } - }, - "38 0": { - "dict": { - "Type": { - "name": "Catalog" - }, - "Outlines": { - "ref": "33 0" - }, - "Lang": { - "text": "en" - }, - "MarkInfo": { - "dict": { - "Marked": { - "bool": true - }, - "Suspects": { - "bool": false - } - } - }, - "Metadata": { - "ref": "31 0" - }, - "Pages": { - "ref": "32 0" - }, - "StructTreeRoot": { - "ref": "9 0" - }, - "ViewerPreferences": { - "dict": { - "DisplayDocTitle": { - "bool": true - } - } - } - } - }, - "39 0": { - "stream": { - "dict": { - "Filter": { - "name": "FlateDecode" - }, - "Length": { - "int": 4252 - }, - "Length1": { - "int": 11864 - } - }, - "raw_length": 4252, - "filters": [ - "FlateDecode" - ], - "decoded": { - "length": 11864, - "sha256": "1db52ea7e54237618253599e8809208e196c393a66451435e032da108e451107" - } - } - }, - "40 0": { - "dict": { - "Type": { - "name": "FontDescriptor" - }, - "Ascent": { - "int": 918 - }, - "CapHeight": { - "int": 587 - }, - "Descent": { - "int": -220 - }, - "Flags": { - "int": 32 - }, - "FontBBox": { - "array": [ - { - "int": -107 - }, - { - "int": -283 - }, - { - "int": 1159 - }, - { - "int": 984 - } - ] - }, - "FontFile2": { - "ref": "39 0" - }, - "FontName": { - "name": "BZDLFS+CrimsonPro-Regular" - }, - "ItalicAngle": { - "int": 0 - }, - "StemV": { - "int": 80 - }, - "XHeight": { - "int": 425 - } - } - }, - "41 0": { - "stream": { - "dict": { - "Length": { - "int": 879 - } - }, - "raw_length": 879, - "filters": [], - "decoded": { - "length": 879, - "sha256": "8a595153c92f4d9492523ca3633beb65c34503f63db4c5bcd9e62363c566b99c", - "preview_utf8": "/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n/CIDSystemInfo << /…" - } - } - }, - "42 0": { - "dict": { - "Type": { - "name": "Font" - }, - "BaseFont": { - "name": "BZDLFS+CrimsonPro-Regular" - }, - "CIDSystemInfo": { - "dict": { - "Ordering": { - "text": "Identity" - }, - "Registry": { - "text": "Adobe" - }, - "Supplement": { - "int": 0 - } - } - }, - "CIDToGIDMap": { - "name": "Identity" - }, - "FontDescriptor": { - "ref": "40 0" - }, - "Subtype": { - "name": "CIDFontType2" - }, - "W": { - "array": [ - { - "int": 0 - }, - { - "array": [ - { - "int": 500 - }, - { - "real": 568.359375 - } - ] - }, - { - "int": 30 - }, - { - "array": [ - { - "real": 589.84375 - } - ] - }, - { - "int": 68 - }, - { - "array": [ - { - "real": 494.140625 - } - ] - }, - { - "int": 76 - }, - { - "array": [ - { - "real": 656.25 - } - ] - }, - { - "int": 102 - }, - { - "array": [ - { - "real": 499.0234375 - } - ] - }, - { - "int": 112 - }, - { - "array": [ - { - "real": 830.078125 - } - ] - }, - { - "int": 160 - }, - { - "array": [ - { - "real": 501.953125 - } - ] - }, - { - "int": 163 - }, - { - "array": [ - { - "real": 553.7109375 - } - ] - }, - { - "int": 184 - }, - { - "array": [ - { - "real": 543.9453125 - } - ] - }, - { - "int": 220 - }, - { - "array": [ - { - "real": 568.359375 - }, - { - "real": 528.3203125 - } - ] - }, - { - "int": 237 - }, - { - "array": [ - { - "real": 462.890625 - } - ] - }, - { - "int": 265 - }, - { - "array": [ - { - "real": 513.671875 - }, - { - "real": 415.0390625 - } - ] - }, - { - "int": 273 - }, - { - "array": [ - { - "real": 526.3671875 - } - ] - }, - { - "int": 280 - }, - { - "array": [ - { - "real": 439.453125 - } - ] - }, - { - "int": 304 - }, - { - "array": [ - { - "real": 295.8984375 - }, - { - "real": 483.3984375 - } - ] - }, - { - "int": 312 - }, - { - "array": [ - { - "real": 537.109375 - } - ] - }, - { - "int": 317 - }, - { - "array": [ - { - "real": 261.71875 - } - ] - }, - { - "int": 337 - }, - { - "array": [ - { - "real": 477.5390625 - } - ] - }, - { - "int": 340 - }, - { - "array": [ - { - "real": 262.6953125 - } - ] - }, - { - "int": 349 - }, - { - "array": [ - { - "real": 803.7109375 - } - ] - }, - { - "int": 351 - }, - { - "array": [ - { - "real": 537.109375 - } - ] - }, - { - "int": 362 - }, - { - "array": [ - { - "real": 496.09375 - } - ] - }, - { - "int": 397 - }, - { - "array": [ - { - "real": 524.4140625 - } - ] - }, - { - "int": 400 - }, - { - "array": [ - { - "real": 355.46875 - } - ] - }, - { - "int": 408 - }, - { - "array": [ - { - "real": 392.578125 - } - ] - }, - { - "int": 420 - }, - { - "array": [ - { - "real": 331.0546875 - } - ] - }, - { - "int": 428 - }, - { - "array": [ - { - "real": 530.2734375 - } - ] - }, - { - "int": 452 - }, - { - "array": [ - { - "real": 735.3515625 - } - ] - }, - { - "int": 457 - }, - { - "array": [ - { - "real": 469.7265625 - }, - { - "real": 460.9375 - } - ] - }, - { - "int": 478 - }, - { - "array": [ - { - "real": 534.1796875 - } - ] - }, - { - "int": 577 - }, - { - "array": [ - { - "real": 228.515625 - }, - { - "real": 228.515625 - } - ] - }, - { - "int": 587 - }, - { - "array": [ - { - "real": 410.15625 - } - ] - }, - { - "int": 590 - }, - { - "array": [ - { - "real": 357.421875 - } - ] - }, - { - "int": 594 - }, - { - "array": [ - { - "real": 351.5625 - }, - { - "real": 351.5625 - } - ] - }, - { - "int": 602 - }, - { - "array": [ - { - "real": 488.28125 - } - ] - }, - { - "int": 625 - }, - { - "array": [ - { - "real": 187.5 - } - ] - } - ] - } - } - }, - "43 0": { - "stream": { - "dict": { - "Filter": { - "name": "FlateDecode" - }, - "Length": { - "int": 3310 - }, - "Length1": { - "int": 10584 - } - }, - "raw_length": 3310, - "filters": [ - "FlateDecode" - ], - "decoded": { - "length": 10584, - "sha256": "574f1438b46e26875646796fe4745383608ba239d8169eee37ad0f21f543411c" - } - } - }, - "44 0": { - "dict": { - "Type": { - "name": "FontDescriptor" - }, - "Ascent": { - "int": 918 - }, - "CapHeight": { - "int": 587 - }, - "Descent": { - "int": -220 - }, - "Flags": { - "int": 32 - }, - "FontBBox": { - "array": [ - { - "int": -107 - }, - { - "int": -283 - }, - { - "int": 1159 - }, - { - "int": 984 - } - ] - }, - "FontFile2": { - "ref": "43 0" - }, - "FontName": { - "name": "LNQEXG+CrimsonPro-Bold" - }, - "ItalicAngle": { - "int": 0 - }, - "StemV": { - "int": 140 - }, - "XHeight": { - "int": 425 - } - } - }, - "45 0": { - "stream": { - "dict": { - "Length": { - "int": 710 - } - }, - "raw_length": 710, - "filters": [], - "decoded": { - "length": 710, - "sha256": "6849b17e78a731b32e149b4f6f8d4cf7831549c1d38e8b503ef9bead6ffaa236", - "preview_utf8": "/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n/CIDSystemInfo << /…" - } - } - }, - "46 0": { - "dict": { - "Type": { - "name": "Font" - }, - "BaseFont": { - "name": "LNQEXG+CrimsonPro-Bold" - }, - "CIDSystemInfo": { - "dict": { - "Ordering": { - "text": "Identity" - }, - "Registry": { - "text": "Adobe" - }, - "Supplement": { - "int": 0 - } - } - }, - "CIDToGIDMap": { - "name": "Identity" - }, - "FontDescriptor": { - "ref": "44 0" - }, - "Subtype": { - "name": "CIDFontType2" - }, - "W": { - "array": [ - { - "int": 0 - }, - { - "array": [ - { - "int": 500 - }, - { - "real": 594.7265625 - } - ] - }, - { - "int": 37 - }, - { - "array": [ - { - "real": 690.4296875 - } - ] - }, - { - "int": 68 - }, - { - "array": [ - { - "real": 526.3671875 - } - ] - }, - { - "int": 112 - }, - { - "array": [ - { - "real": 870.1171875 - } - ] - }, - { - "int": 160 - }, - { - "array": [ - { - "real": 555.6640625 - } - ] - }, - { - "int": 191 - }, - { - "array": [ - { - "real": 685.546875 - } - ] - }, - { - "int": 215 - }, - { - "array": [ - { - "real": 955.078125 - } - ] - }, - { - "int": 237 - }, - { - "array": [ - { - "real": 488.28125 - } - ] - }, - { - "int": 265 - }, - { - "array": [ - { - "real": 544.921875 - }, - { - "real": 449.21875 - } - ] - }, - { - "int": 273 - }, - { - "array": [ - { - "real": 547.8515625 - } - ] - }, - { - "int": 280 - }, - { - "array": [ - { - "real": 465.8203125 - } - ] - }, - { - "int": 305 - }, - { - "array": [ - { - "real": 506.8359375 - } - ] - }, - { - "int": 312 - }, - { - "array": [ - { - "real": 574.21875 - } - ] - }, - { - "int": 317 - }, - { - "array": [ - { - "real": 291.015625 - } - ] - }, - { - "int": 337 - }, - { - "array": [ - { - "real": 543.9453125 - } - ] - }, - { - "int": 340 - }, - { - "array": [ - { - "real": 291.015625 - } - ] - }, - { - "int": 349 - }, - { - "array": [ - { - "real": 846.6796875 - } - ] - }, - { - "int": 351 - }, - { - "array": [ - { - "real": 574.21875 - } - ] - }, - { - "int": 362 - }, - { - "array": [ - { - "real": 520.5078125 - } - ] - }, - { - "int": 400 - }, - { - "array": [ - { - "real": 401.3671875 - } - ] - }, - { - "int": 408 - }, - { - "array": [ - { - "real": 408.203125 - } - ] - }, - { - "int": 420 - }, - { - "array": [ - { - "real": 360.3515625 - } - ] - }, - { - "int": 428 - }, - { - "array": [ - { - "real": 568.359375 - } - ] - }, - { - "int": 452 - }, - { - "array": [ - { - "real": 795.8984375 - } - ] - }, - { - "int": 458 - }, - { - "array": [ - { - "real": 486.328125 - } - ] - }, - { - "int": 478 - }, - { - "array": [ - { - "real": 584.9609375 - } - ] - }, - { - "int": 590 - }, - { - "array": [ - { - "real": 406.25 - } - ] - }, - { - "int": 625 - }, - { - "array": [ - { - "real": 192.3828125 - } - ] - } - ] - } - } - }, - "47 0": { - "stream": { - "dict": { - "Filter": { - "name": "FlateDecode" - }, - "Length": { - "int": 3679 - }, - "Length1": { - "int": 11144 - } - }, - "raw_length": 3679, - "filters": [ - "FlateDecode" - ], - "decoded": { - "length": 11144, - "sha256": "3f54c3113c615cf618c245df334fa62a49c591b372659a292a97c995536f0c5a" - } - } - }, - "48 0": { - "dict": { - "Type": { - "name": "FontDescriptor" - }, - "Ascent": { - "int": 918 - }, - "CapHeight": { - "int": 587 - }, - "Descent": { - "int": -220 - }, - "Flags": { - "int": 96 - }, - "FontBBox": { - "array": [ - { - "int": -155 - }, - { - "int": -285 - }, - { - "int": 1212 - }, - { - "int": 985 - } - ] - }, - "FontFile2": { - "ref": "47 0" - }, - "FontName": { - "name": "AVREHH+CrimsonPro-Italic" - }, - "ItalicAngle": { - "int": -12 - }, - "StemV": { - "int": 80 - }, - "XHeight": { - "int": 425 - } - } - }, - "49 0": { - "stream": { - "dict": { - "Length": { - "int": 762 - } - }, - "raw_length": 762, - "filters": [], - "decoded": { - "length": 762, - "sha256": "54ef34fc7df73d04c468c08e310571b015ba6f8d61f95ab050564f20ac3a61cc", - "preview_utf8": "/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n/CIDSystemInfo << /…" - } - } - }, - "50 0": { - "dict": { - "Type": { - "name": "Font" - }, - "BaseFont": { - "name": "AVREHH+CrimsonPro-Italic" - }, - "CIDSystemInfo": { - "dict": { - "Ordering": { - "text": "Identity" - }, - "Registry": { - "text": "Adobe" - }, - "Supplement": { - "int": 0 - } - } - }, - "CIDToGIDMap": { - "name": "Identity" - }, - "FontDescriptor": { - "ref": "48 0" - }, - "Subtype": { - "name": "CIDFontType2" - }, - "W": { - "array": [ - { - "int": 0 - }, - { - "array": [ - { - "int": 500 - }, - { - "real": 568.359375 - } - ] - }, - { - "int": 30 - }, - { - "array": [ - { - "real": 589.84375 - } - ] - }, - { - "int": 37 - }, - { - "array": [ - { - "real": 666.015625 - } - ] - }, - { - "int": 112 - }, - { - "array": [ - { - "real": 830.078125 - } - ] - }, - { - "int": 114 - }, - { - "array": [ - { - "real": 658.203125 - } - ] - }, - { - "int": 160 - }, - { - "array": [ - { - "real": 501.953125 - } - ] - }, - { - "int": 184 - }, - { - "array": [ - { - "real": 543.9453125 - } - ] - }, - { - "int": 214 - }, - { - "array": [ - { - "real": 571.2890625 - } - ] - }, - { - "int": 237 - }, - { - "array": [ - { - "real": 470.703125 - } - ] - }, - { - "int": 265 - }, - { - "array": [ - { - "real": 444.3359375 - }, - { - "real": 351.5625 - } - ] - }, - { - "int": 273 - }, - { - "array": [ - { - "real": 470.703125 - } - ] - }, - { - "int": 280 - }, - { - "array": [ - { - "real": 378.90625 - } - ] - }, - { - "int": 304 - }, - { - "array": [ - { - "real": 257.8125 - }, - { - "real": 445.3125 - } - ] - }, - { - "int": 312 - }, - { - "array": [ - { - "real": 481.4453125 - } - ] - }, - { - "int": 317 - }, - { - "array": [ - { - "real": 270.5078125 - } - ] - }, - { - "int": 337 - }, - { - "array": [ - { - "real": 436.5234375 - } - ] - }, - { - "int": 340 - }, - { - "array": [ - { - "real": 244.140625 - } - ] - }, - { - "int": 349 - }, - { - "array": [ - { - "real": 731.4453125 - } - ] - }, - { - "int": 351 - }, - { - "array": [ - { - "real": 505.859375 - } - ] - }, - { - "int": 362 - }, - { - "array": [ - { - "real": 426.7578125 - } - ] - }, - { - "int": 397 - }, - { - "array": [ - { - "real": 475.5859375 - } - ] - }, - { - "int": 400 - }, - { - "array": [ - { - "real": 348.6328125 - } - ] - }, - { - "int": 408 - }, - { - "array": [ - { - "real": 308.59375 - } - ] - }, - { - "int": 420 - }, - { - "array": [ - { - "real": 294.921875 - } - ] - }, - { - "int": 428 - }, - { - "array": [ - { - "int": 500 - } - ] - }, - { - "int": 452 - }, - { - "array": [ - { - "real": 638.671875 - } - ] - }, - { - "int": 458 - }, - { - "array": [ - { - "real": 417.96875 - } - ] - }, - { - "int": 576 - }, - { - "array": [ - { - "real": 226.5625 - }, - { - "real": 226.5625 - } - ] - }, - { - "int": 601 - }, - { - "array": [ - { - "real": 488.28125 - } - ] - }, - { - "int": 624 - }, - { - "array": [ - { - "real": 187.5 - } - ] - } - ] - } - } - }, - "51 0": { - "stream": { - "dict": { - "Filter": { - "name": "FlateDecode" - }, - "Length": { - "int": 1677 - }, - "Length1": { - "int": 6072 - } - }, - "raw_length": 1677, - "filters": [ - "FlateDecode" - ], - "decoded": { - "length": 6072, - "sha256": "ae26f0eb05cde76698aedc3612f333257d4cc1bfbdd15d27b4ad7fbed784892d" - } - } - }, - "52 0": { - "dict": { - "Type": { - "name": "FontDescriptor" - }, - "Ascent": { - "int": 1050 - }, - "CapHeight": { - "int": 695 - }, - "Descent": { - "int": -250 - }, - "Flags": { - "int": 33 - }, - "FontBBox": { - "array": [ - { - "int": -9 - }, - { - "int": -250 - }, - { - "int": 578 - }, - { - "int": 1050 - } - ] - }, - "FontFile2": { - "ref": "51 0" - }, - "FontName": { - "name": "JYZTRN+CamingoCode-Bold" - }, - "ItalicAngle": { - "int": 0 - }, - "StemV": { - "int": 140 - }, - "XHeight": { - "int": 495 - } - } - }, - "53 0": { - "stream": { - "dict": { - "Length": { - "int": 371 - } - }, - "raw_length": 371, - "filters": [], - "decoded": { - "length": 371, - "sha256": "30d050b1587e0324b8a9bea13f37895c9aa7ef6dfb90a3b6a8303554bb5af055", - "preview_utf8": "/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n/CIDSystemInfo << /…" - } - } - }, - "54 0": { - "dict": { - "Type": { - "name": "Font" - }, - "BaseFont": { - "name": "JYZTRN+CamingoCode-Bold" - }, - "CIDSystemInfo": { - "dict": { - "Ordering": { - "text": "Identity" - }, - "Registry": { - "text": "Adobe" - }, - "Supplement": { - "int": 0 - } - } - }, - "CIDToGIDMap": { - "name": "Identity" - }, - "FontDescriptor": { - "ref": "52 0" - }, - "Subtype": { - "name": "CIDFontType2" - }, - "W": { - "array": [ - { - "int": 0 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 42 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 73 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 121 - }, - { - "array": [ - { - "int": 550 - } - ] - } - ] - } - } - }, - "55 0": { - "stream": { - "dict": { - "Filter": { - "name": "FlateDecode" - }, - "Length": { - "int": 7654 - }, - "Length1": { - "int": 17468 - } - }, - "raw_length": 7654, - "filters": [ - "FlateDecode" - ], - "decoded": { - "length": 17468, - "sha256": "3c1742e2fad46b9b3a5df1ea5b25db87d6dd07d5d39cb59c07febd7320f9be9f" - } - } - }, - "56 0": { - "dict": { - "Type": { - "name": "FontDescriptor" - }, - "Ascent": { - "int": 1050 - }, - "CapHeight": { - "int": 695 - }, - "Descent": { - "int": -250 - }, - "Flags": { - "int": 33 - }, - "FontBBox": { - "array": [ - { - "int": -2 - }, - { - "int": -250 - }, - { - "int": 578 - }, - { - "int": 1050 - } - ] - }, - "FontFile2": { - "ref": "55 0" - }, - "FontName": { - "name": "YCXYQB+CamingoCode-Regular" - }, - "ItalicAngle": { - "int": 0 - }, - "StemV": { - "int": 80 - }, - "XHeight": { - "int": 490 - } - } - }, - "57 0": { - "stream": { - "dict": { - "Length": { - "int": 840 - } - }, - "raw_length": 840, - "filters": [], - "decoded": { - "length": 840, - "sha256": "7690a81413d655194a93216ada14d5ccef27866806a9d24cdca14a068ee72b99", - "preview_utf8": "/CIDInit /ProcSet findresource begin\n12 dict begin\nbegincmap\n/CIDSystemInfo << /…" - } - } - }, - "58 0": { - "dict": { - "Type": { - "name": "Font" - }, - "BaseFont": { - "name": "YCXYQB+CamingoCode-Regular" - }, - "CIDSystemInfo": { - "dict": { - "Ordering": { - "text": "Identity" - }, - "Registry": { - "text": "Adobe" - }, - "Supplement": { - "int": 0 - } - } - }, - "CIDToGIDMap": { - "name": "Identity" - }, - "FontDescriptor": { - "ref": "56 0" - }, - "Subtype": { - "name": "CIDFontType2" - }, - "W": { - "array": [ - { - "int": 0 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 3 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 5 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 21 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 27 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 31 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 41 - }, - { - "array": [ - { - "int": 550 - }, - { - "int": 550 - } - ] - }, - { - "int": 49 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 53 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 69 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 73 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 79 - }, - { - "array": [ - { - "int": 550 - }, - { - "int": 550 - } - ] - }, - { - "int": 88 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 102 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 105 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 109 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 116 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 121 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 134 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 137 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 146 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 162 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 168 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 182 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 190 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 194 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 211 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 217 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 239 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 242 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 246 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 253 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 258 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 315 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 320 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 348 - }, - { - "array": [ - { - "int": 550 - } - ] - }, - { - "int": 369 - }, - { - "array": [ - { - "int": 550 - }, - { - "int": 550 - } - ] - } - ] - } - } - }, - "59 0": { - "dict": { - "Author": { - "text": "glu demo" - }, - "CreationDate": { - "text": "D:20260511113125+02'00'" - }, - "Creator": { - "text": "boxesandglue.dev" - }, - "Producer": { - "text": "boxesandglue.dev" - }, - "Title": { - "text": "Markdown to PDF/UA with glu" - } - } - } - } -} diff --git a/testdata/fixtures/pdfua-demo/input.pdf b/testdata/fixtures/pdfua-demo/input.pdf deleted file mode 100644 index 4776784..0000000 Binary files a/testdata/fixtures/pdfua-demo/input.pdf and /dev/null differ diff --git a/testdata/fixtures/xref-stream/golden.json b/testdata/fixtures/xref-stream/golden.json deleted file mode 100644 index 85e986b..0000000 --- a/testdata/fixtures/xref-stream/golden.json +++ /dev/null @@ -1,62 +0,0 @@ -{ - "version": "2.0", - "xref_format": "stream", - "encrypted": false, - "trailer": { - "dict": { - "Type": { - "name": "XRef" - }, - "Size": { - "int": 3 - }, - "W": { - "array": [ - { - "int": 1 - }, - { - "int": 3 - }, - { - "int": 1 - } - ] - }, - "Root": { - "ref": "1 0" - }, - "Filter": { - "name": "FlateDecode" - }, - "Length": { - "int": 27 - } - } - }, - "objects": { - "1 0": { - "dict": { - "Type": { - "name": "Catalog" - }, - "Pages": { - "ref": "2 0" - } - } - }, - "2 0": { - "dict": { - "Type": { - "name": "Pages" - }, - "Count": { - "int": 0 - }, - "Kids": { - "array": [] - } - } - } - } -} diff --git a/testdata/fixtures/xref-stream/input.pdf b/testdata/fixtures/xref-stream/input.pdf deleted file mode 100644 index a663acf..0000000 Binary files a/testdata/fixtures/xref-stream/input.pdf and /dev/null differ diff --git a/text.go b/text.go deleted file mode 100644 index 17f8c8d..0000000 --- a/text.go +++ /dev/null @@ -1,195 +0,0 @@ -package pdfdisassembler - -import ( - "encoding/binary" - "strings" - "time" - "unicode/utf16" -) - -// decodeTextString decodes b according to the PDF text-string convention -// (PDF 32000-1:2008 §7.9.2.2): UTF-16BE with BOM, UTF-8 with BOM (PDF 2.0), -// otherwise PDFDocEncoding. -func decodeTextString(b []byte) string { - switch { - case len(b) >= 2 && b[0] == 0xFE && b[1] == 0xFF: - return decodeUTF16BE(b[2:]) - case len(b) >= 2 && b[0] == 0xFF && b[1] == 0xFE: - // UTF-16LE: not spec'd for PDF text strings but observed in the - // wild from misbehaving producers; decode rather than mojibake. - return decodeUTF16LE(b[2:]) - case len(b) >= 3 && b[0] == 0xEF && b[1] == 0xBB && b[2] == 0xBF: - return string(b[3:]) - default: - return decodePDFDocEncoding(b) - } -} - -func decodeUTF16BE(b []byte) string { - if len(b)%2 != 0 { - b = b[:len(b)-1] - } - u16 := make([]uint16, len(b)/2) - for i := range u16 { - u16[i] = binary.BigEndian.Uint16(b[2*i:]) - } - return string(utf16.Decode(u16)) -} - -func decodeUTF16LE(b []byte) string { - if len(b)%2 != 0 { - b = b[:len(b)-1] - } - u16 := make([]uint16, len(b)/2) - for i := range u16 { - u16[i] = binary.LittleEndian.Uint16(b[2*i:]) - } - return string(utf16.Decode(u16)) -} - -// decodePDFDocEncoding maps each byte through the PDFDocEncoding table -// from PDF 32000-1:2008 Annex D.2. -func decodePDFDocEncoding(b []byte) string { - var sb strings.Builder - sb.Grow(len(b)) - for _, c := range b { - r := pdfDocEncoding[c] - if r == 0xFFFD { - // Undefined slot — emit replacement character. - sb.WriteRune('�') - } else { - sb.WriteRune(r) - } - } - return sb.String() -} - -// pdfDocEncoding is the PDFDocEncoding to Unicode mapping (256 entries). -// Slots without a Unicode mapping are 0xFFFD. -var pdfDocEncoding = [256]rune{ - // 0x00–0x17 (control characters mostly unused in PDFDocEncoding) - 0x0000, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, - 0x0008, 0x0009, 0x000A, 0xFFFD, 0x000C, 0x000D, 0xFFFD, 0xFFFD, - 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, 0xFFFD, - // 0x18–0x1F - 0x02D8, 0x02C7, 0x02C6, 0x02D9, 0x02DD, 0x02DB, 0x02DA, 0x02DC, - // 0x20–0x7E identical to ASCII - 0x0020, 0x0021, 0x0022, 0x0023, 0x0024, 0x0025, 0x0026, 0x0027, - 0x0028, 0x0029, 0x002A, 0x002B, 0x002C, 0x002D, 0x002E, 0x002F, - 0x0030, 0x0031, 0x0032, 0x0033, 0x0034, 0x0035, 0x0036, 0x0037, - 0x0038, 0x0039, 0x003A, 0x003B, 0x003C, 0x003D, 0x003E, 0x003F, - 0x0040, 0x0041, 0x0042, 0x0043, 0x0044, 0x0045, 0x0046, 0x0047, - 0x0048, 0x0049, 0x004A, 0x004B, 0x004C, 0x004D, 0x004E, 0x004F, - 0x0050, 0x0051, 0x0052, 0x0053, 0x0054, 0x0055, 0x0056, 0x0057, - 0x0058, 0x0059, 0x005A, 0x005B, 0x005C, 0x005D, 0x005E, 0x005F, - 0x0060, 0x0061, 0x0062, 0x0063, 0x0064, 0x0065, 0x0066, 0x0067, - 0x0068, 0x0069, 0x006A, 0x006B, 0x006C, 0x006D, 0x006E, 0x006F, - 0x0070, 0x0071, 0x0072, 0x0073, 0x0074, 0x0075, 0x0076, 0x0077, - 0x0078, 0x0079, 0x007A, 0x007B, 0x007C, 0x007D, 0x007E, 0xFFFD, - // 0x80–0x9F: punctuation and symbol additions per PDFDocEncoding - 0x2022, 0x2020, 0x2021, 0x2026, 0x2014, 0x2013, 0x0192, 0x2044, - 0x2039, 0x203A, 0x2212, 0x2030, 0x201E, 0x201C, 0x201D, 0x2018, - 0x2019, 0x201A, 0x2122, 0xFB01, 0xFB02, 0x0141, 0x0152, 0x0160, - 0x0178, 0x017D, 0x0131, 0x0142, 0x0153, 0x0161, 0x017E, 0xFFFD, - // 0xA0 - 0x20AC, - // 0xA1–0xFF: same as ISO Latin-1 / Unicode 0x00A1–0x00FF, except a - // few slots marked undefined by the spec. - 0x00A1, 0x00A2, 0x00A3, 0x00A4, 0x00A5, 0x00A6, 0x00A7, - 0x00A8, 0x00A9, 0x00AA, 0x00AB, 0x00AC, 0xFFFD, 0x00AE, 0x00AF, - 0x00B0, 0x00B1, 0x00B2, 0x00B3, 0x00B4, 0x00B5, 0x00B6, 0x00B7, - 0x00B8, 0x00B9, 0x00BA, 0x00BB, 0x00BC, 0x00BD, 0x00BE, 0x00BF, - 0x00C0, 0x00C1, 0x00C2, 0x00C3, 0x00C4, 0x00C5, 0x00C6, 0x00C7, - 0x00C8, 0x00C9, 0x00CA, 0x00CB, 0x00CC, 0x00CD, 0x00CE, 0x00CF, - 0x00D0, 0x00D1, 0x00D2, 0x00D3, 0x00D4, 0x00D5, 0x00D6, 0x00D7, - 0x00D8, 0x00D9, 0x00DA, 0x00DB, 0x00DC, 0x00DD, 0x00DE, 0x00DF, - 0x00E0, 0x00E1, 0x00E2, 0x00E3, 0x00E4, 0x00E5, 0x00E6, 0x00E7, - 0x00E8, 0x00E9, 0x00EA, 0x00EB, 0x00EC, 0x00ED, 0x00EE, 0x00EF, - 0x00F0, 0x00F1, 0x00F2, 0x00F3, 0x00F4, 0x00F5, 0x00F6, 0x00F7, - 0x00F8, 0x00F9, 0x00FA, 0x00FB, 0x00FC, 0x00FD, 0x00FE, 0x00FF, -} - -// parseDate parses a PDF date string of the form -// "D:YYYYMMDDHHmmSSOHH'mm'" or any shorter prefix. Returns the zero time -// if the input is empty or unparseable. -func parseDate(s string) time.Time { - s = strings.TrimSpace(s) - if s == "" { - return time.Time{} - } - s = strings.TrimPrefix(s, "D:") - // Defaults per spec: month/day = 01, time = 00, offset = UTC. - year, month, day := 0, 1, 1 - hour, minute, second := 0, 0, 0 - tzSign := byte('Z') - tzHour, tzMinute := 0, 0 - - read := func(n int) (int, bool) { - if len(s) < n { - return 0, false - } - v := 0 - for i := 0; i < n; i++ { - c := s[i] - if c < '0' || c > '9' { - return 0, false - } - v = v*10 + int(c-'0') - } - s = s[n:] - return v, true - } - - if v, ok := read(4); ok { - year = v - } else { - return time.Time{} - } - if v, ok := read(2); ok { - month = v - } - if v, ok := read(2); ok { - day = v - } - if v, ok := read(2); ok { - hour = v - } - if v, ok := read(2); ok { - minute = v - } - if v, ok := read(2); ok { - second = v - } - - if len(s) > 0 { - switch s[0] { - case '+', '-', 'Z': - tzSign = s[0] - s = s[1:] - } - } - if tzSign != 'Z' { - if v, ok := read(2); ok { - tzHour = v - } - // Optional apostrophe between hour and minute. - s = strings.TrimPrefix(s, "'") - if v, ok := read(2); ok { - tzMinute = v - } - } - - loc := time.UTC - if tzSign == '+' || tzSign == '-' { - off := tzHour*3600 + tzMinute*60 - if tzSign == '-' { - off = -off - } - loc = time.FixedZone("", off) - } - - if month < 1 || month > 12 || day < 1 || day > 31 { - return time.Time{} - } - return time.Date(year, time.Month(month), day, hour, minute, second, 0, loc) -} diff --git a/text_test.go b/text_test.go deleted file mode 100644 index ae50d5a..0000000 --- a/text_test.go +++ /dev/null @@ -1,125 +0,0 @@ -package pdfdisassembler - -import ( - "testing" - "time" - "unicode/utf16" -) - -// Round-trip via utf16.Encode (the independent inverse): the emoji forces a -// surrogate pair, and both byte orders dispatch off their BOM. -func TestDecodeTextStringUTF16RoundTrip(t *testing.T) { - const s = "Hello, 世界 \U0001F600 é" - u16 := utf16.Encode([]rune(s)) - be := []byte{0xFE, 0xFF} - le := []byte{0xFF, 0xFE} - for _, v := range u16 { - be = append(be, byte(v>>8), byte(v)) - le = append(le, byte(v), byte(v>>8)) - } - if got := decodeTextString(be); got != s { - t.Errorf("UTF-16BE: got %q want %q", got, s) - } - if got := decodeTextString(le); got != s { - t.Errorf("UTF-16LE: got %q want %q", got, s) - } -} - -// A UTF-16 payload with an odd byte count must drop the dangling byte, not -// read past the end — for both byte orders. -func TestDecodeUTF16OddLengthNoPanic(t *testing.T) { - if got := decodeTextString([]byte{0xFE, 0xFF, 0x00, 0x41, 0x00}); got != "A" { - t.Errorf("UTF-16BE: got %q, want A", got) - } - if got := decodeTextString([]byte{0xFF, 0xFE, 0x41, 0x00, 0x00}); got != "A" { - t.Errorf("UTF-16LE: got %q, want A", got) - } -} - -func TestDecodeTextStringDispatch(t *testing.T) { - // UTF-8 BOM (PDF 2.0): bytes after the BOM are returned verbatim. - if got := decodeTextString([]byte{0xEF, 0xBB, 0xBF, 'h', 'i'}); got != "hi" { - t.Errorf("UTF-8 BOM: got %q want hi", got) - } - // No BOM: PDFDocEncoding, which is ASCII over 0x20-0x7E. - if got := decodeTextString([]byte("ASCII")); got != "ASCII" { - t.Errorf("PDFDocEncoding ASCII: got %q", got) - } -} - -// Spot-check the PDFDocEncoding table (PDF 32000-1:2008 Annex D.2), including -// the high-range remaps and an undefined slot that must become U+FFFD. -func TestDecodePDFDocEncoding(t *testing.T) { - cases := map[byte]rune{ - 0x41: 'A', - 0x18: '˘', // breve - 0x80: '•', // bullet - 0xA0: '€', // euro sign - 0xE9: 'é', // Latin-1 range, identity-mapped - 0x7F: '�', // undefined - 0x9F: '�', // undefined - } - for b, want := range cases { - got := []rune(decodeTextString([]byte{b})) - if len(got) != 1 || got[0] != want { - t.Errorf("byte 0x%02X decoded to %q, want %q", b, string(got), string(want)) - } - } -} - -func TestParseDate(t *testing.T) { - utc := func(y int, mo time.Month, d, h, mi, s int) time.Time { - return time.Date(y, mo, d, h, mi, s, 0, time.UTC) - } - tests := []struct { - in string - want time.Time - }{ - {"D:20201231235959Z", utc(2020, 12, 31, 23, 59, 59)}, - {"D:20200229", utc(2020, 2, 29, 0, 0, 0)}, // leap day - {"D:2020", utc(2020, 1, 1, 0, 0, 0)}, // year only, defaults fill in - {"20200115", utc(2020, 1, 15, 0, 0, 0)}, // optional D: prefix omitted - {"", time.Time{}}, - {"garbage", time.Time{}}, - {"D:20201340", time.Time{}}, // month 13 rejected - } - for _, tt := range tests { - if got := parseDate(tt.in); !got.Equal(tt.want) { - t.Errorf("parseDate(%q) = %v, want %v", tt.in, got, tt.want) - } - } - // Signed timezone offset, with the apostrophe separator. - got := parseDate("D:20200101120000+05'30'") - want := time.Date(2020, 1, 1, 12, 0, 0, 0, time.FixedZone("", 5*3600+30*60)) - if !got.Equal(want) { - t.Errorf("tz parse = %v, want %v", got, want) - } -} - -// The dump heuristic for rendering a string inline vs hex-escaped. The -// high-byte cases are the subtle ones: 0x80 is a clean PDFDocEncoding glyph -// (bullet), 0x9F an undefined slot. -func TestLooksLikeText(t *testing.T) { - cases := []struct { - name string - in String - want bool - }{ - {"utf16be_bom", String{0xFE, 0xFF, 0, 'A'}, true}, - {"utf16le_bom", String{0xFF, 0xFE, 'A', 0}, true}, - {"utf8_bom", String{0xEF, 0xBB, 0xBF, 'h', 'i'}, true}, - {"ascii", String("Hello, World!"), true}, - {"ascii_with_whitespace", String("a\tb\r\nc"), true}, - {"control_byte", String{'a', 0x01, 'b'}, false}, - {"del_byte", String{0x7F}, false}, - {"pdfdoc_high_clean", String{0x80}, true}, - {"pdfdoc_high_undefined", String{0x9F}, false}, - } - for _, c := range cases { - t.Run(c.name, func(t *testing.T) { - if got := looksLikeText(c.in); got != c.want { - t.Errorf("looksLikeText(% x) = %v, want %v", c.in, got, c.want) - } - }) - } -} diff --git a/xref.go b/xref.go deleted file mode 100644 index c9ce9da..0000000 --- a/xref.go +++ /dev/null @@ -1,609 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "errors" - "fmt" - "strconv" - - "github.com/speedata/pdfdisassembler/internal/lex" -) - -func trimLeftSpace(s string) string { - for i := 0; i < len(s); i++ { - c := s[i] - if c != ' ' && c != '\t' { - return s[i:] - } - } - return "" -} - -// parseXref locates the last startxref offset, then parses the cross- -// reference table (classical, xref-stream, or hybrid) and walks /Prev -// chains. If the declared xref location is broken, parseXref falls back -// to xref recovery (scanning for "obj" markers). -func (r *Reader) parseXref() error { - off, err := r.findStartXref() - if err != nil { - if recErr := r.recoverXref(); recErr != nil { - return fmt.Errorf("pdfdisassembler: startxref missing and recovery failed: %v (recovery: %w)", err, recErr) - } - return nil - } - - visited := map[int64]bool{} - cur := off - for { - if visited[cur] { - return fmt.Errorf("pdfdisassembler: xref loop at offset %d", cur) - } - visited[cur] = true - - prev, err := r.readXrefAt(cur) - if err != nil { - // Recover if the first attempt was wrong; xref chains - // otherwise abort here. - if len(visited) == 1 { - if recErr := r.recoverXref(); recErr != nil { - return fmt.Errorf("pdfdisassembler: xref at %d failed: %v (recovery: %w)", cur, err, recErr) - } - return nil - } - return err - } - if prev == 0 { - break - } - cur = prev - } - - if r.trailer == nil { - return errors.New("pdfdisassembler: no trailer found") - } - return nil -} - -// findStartXref scans the last 1024 bytes of the file for the "startxref" -// marker and returns the offset that follows it. -func (r *Reader) findStartXref() (int64, error) { - const tail = 1024 - start := len(r.buf) - tail - if start < 0 { - start = 0 - } - idx := bytes.LastIndex(r.buf[start:], []byte("startxref")) - if idx < 0 { - return 0, errors.New("startxref not found") - } - idx += start + len("startxref") - // Skip whitespace, then read decimal. - for idx < len(r.buf) && (r.buf[idx] == ' ' || r.buf[idx] == '\t' || - r.buf[idx] == '\r' || r.buf[idx] == '\n') { - idx++ - } - end := idx - for end < len(r.buf) && r.buf[end] >= '0' && r.buf[end] <= '9' { - end++ - } - if end == idx { - return 0, errors.New("startxref offset missing") - } - off, err := strconv.ParseInt(string(r.buf[idx:end]), 10, 64) - if err != nil { - return 0, err - } - return off, nil -} - -// readXrefAt parses an xref section starting at offset. Returns the /Prev -// offset (0 if no previous section). -func (r *Reader) readXrefAt(offset int64) (int64, error) { - if offset < 0 || offset >= int64(len(r.buf)) { - return 0, fmt.Errorf("xref offset %d out of range", offset) - } - // Classical sections start with the keyword "xref". - rest := r.buf[offset:] - // Skip whitespace. - i := 0 - for i < len(rest) && lex.IsWhitespace(rest[i]) { - i++ - } - if i+4 <= len(rest) && string(rest[i:i+4]) == "xref" { - return r.readClassicalXrefAt(offset + int64(i)) - } - return r.readXrefStreamAt(offset) -} - -// readClassicalXrefAt parses a "xref" subsection table and the following -// trailer dictionary. Returns the /Prev offset (0 if none). -func (r *Reader) readClassicalXrefAt(offset int64) (int64, error) { - pos := int(offset) - // Skip the "xref" keyword and following EOL. - pos += 4 - pos = skipEOL(r.buf, pos) - - for { - // Each subsection: "first count" then count entries of 20 bytes - // each. The subsection list ends at "trailer". - lineEnd := indexEOL(r.buf, pos) - if lineEnd < 0 { - return 0, errors.New("classical xref: unterminated subsection header") - } - line := string(bytes.TrimSpace(r.buf[pos:lineEnd])) - if line == "trailer" { - // trailer follows. - pos = skipEOL(r.buf, pos+len("trailer")) - break - } - // Some producers put trailer on its own line further down. - if line == "" { - pos = skipEOL(r.buf, lineEnd) - continue - } - parts := bytes.Fields(r.buf[pos:lineEnd]) - if len(parts) != 2 { - return 0, fmt.Errorf("classical xref: bad subsection header %q", line) - } - first, err1 := strconv.Atoi(string(parts[0])) - count, err2 := strconv.Atoi(string(parts[1])) - if err1 != nil || err2 != nil { - return 0, fmt.Errorf("classical xref: bad subsection header %q", line) - } - pos = skipEOL(r.buf, lineEnd) - - for i := 0; i < count; i++ { - // pos walks the raw buffer by 20 per entry over an attacker-set - // count; bound it without pos+20, which can overflow int on 32-bit. - if pos < 0 || pos > len(r.buf)-20 { - return 0, errors.New("classical xref: truncated entry") - } - entry := r.buf[pos : pos+20] - pos += 20 - // Format: nnnnnnnnnn ggggg t EOL - if len(entry) < 18 { - return 0, errors.New("classical xref: short entry") - } - offStr := string(entry[0:10]) - genStr := string(entry[11:16]) - flag := entry[17] - off, _ := strconv.ParseInt(trimLeftSpace(offStr), 10, 64) - gen, _ := strconv.Atoi(trimLeftSpace(genStr)) - ref := Reference{Number: first + i, Generation: gen} - if flag == 'n' { - if _, exists := r.xref[ref]; !exists { - r.xref[ref] = xrefEntry{kind: 1, offset: off, generation: gen} - } - } - // 'f' entries are free; ignore. - } - } - - // Parse trailer dictionary. - lx := lex.New(r.buf) - lx.SetPos(pos) - p := newParser(lx, r) - tok, err := p.next() - if err != nil { - return 0, err - } - if tok.Kind != lex.DictStart { - return 0, fmt.Errorf("classical xref: trailer dict missing, got %s", tok.Kind) - } - trailer, err := p.parseDict() - if err != nil { - return 0, fmt.Errorf("classical xref: trailer parse: %w", err) - } - if r.trailer == nil { - r.trailer = trailer - } else { - // Older trailers fill in missing keys only. - for k, v := range trailer.Iter() { - if !r.trailer.Has(k) { - r.trailer.set(k, v) - } - } - } - - // Hybrid: trailer may reference an XRefStm. - if v, ok := trailer.Get("XRefStm"); ok { - if off, ok := v.(Integer); ok { - if _, err := r.readXrefStreamAt(int64(off)); err != nil { - // non-fatal: log via error wrap - return 0, fmt.Errorf("hybrid XRefStm: %w", err) - } - } - } - - if v, ok := trailer.Get("Prev"); ok { - if n, ok := v.(Integer); ok { - return int64(n), nil - } - } - return 0, nil -} - -// readXrefStreamAt parses an xref stream at the given offset. Returns the -// /Prev offset (0 if none). -func (r *Reader) readXrefStreamAt(offset int64) (int64, error) { - if offset < 0 || offset >= int64(len(r.buf)) { - return 0, fmt.Errorf("xref stream offset %d out of range", offset) - } - lx := lex.New(r.buf) - lx.SetPos(int(offset)) - p := newParser(lx, r) - - // "N G obj" - t1, err := p.next() - if err != nil { - return 0, err - } - t2, err := p.next() - if err != nil { - return 0, err - } - t3, err := p.next() - if err != nil { - return 0, err - } - if t1.Kind != lex.Integer || t2.Kind != lex.Integer || - t3.Kind != lex.Keyword || string(t3.Bytes) != "obj" { - return 0, fmt.Errorf("xref stream: bad indirect header at %d", offset) - } - objNum, _ := strconv.Atoi(string(t1.Bytes)) - objGen, _ := strconv.Atoi(string(t2.Bytes)) - - body, err := p.parseObject() - if err != nil { - return 0, fmt.Errorf("xref stream: dict parse: %w", err) - } - d, ok := body.(*Dict) - if !ok { - return 0, fmt.Errorf("xref stream: body is %T, want dict", body) - } - tok, err := p.peek() - if err != nil { - return 0, err - } - if tok.Kind != lex.Keyword || string(tok.Bytes) != "stream" { - return 0, fmt.Errorf("xref stream: missing stream keyword") - } - p.next() - length, err := r.streamLength(d) - if err != nil { - return 0, err - } - raw, err := lx.ReadStreamData(int(length)) - if err != nil { - return 0, err - } - - // The xref stream itself is unencrypted per spec (encrypt context not - // yet initialised at this point either way). - stream := &Stream{ - Dict: d, - reader: r, - rawOffset: int64(lx.Pos() - len(raw)), - rawLength: int64(len(raw)), - objNumber: objNum, - objGeneration: objGen, - } - decoded, err := r.applyFilters(stream, raw, false) - if err != nil { - return 0, fmt.Errorf("xref stream: decode: %w", err) - } - - // Store the trailer (xref-stream dict doubles as trailer). - if r.trailer == nil { - r.trailer = d - } else { - for k, v := range d.Iter() { - if !r.trailer.Has(k) { - r.trailer.set(k, v) - } - } - } - - // Read W field widths. - wArr, ok := d.Array("W") - if !ok || len(wArr) < 3 { - return 0, errors.New("xref stream: /W missing or too short") - } - w := make([]int, len(wArr)) - for i, v := range wArr { - n, ok := v.(Integer) - if !ok { - return 0, fmt.Errorf("xref stream: /W[%d] not integer", i) - } - if n < 0 { - return 0, fmt.Errorf("xref stream: negative /W[%d] = %d", i, n) - } - w[i] = int(n) - } - rowSize := 0 - for _, n := range w { - rowSize += n - } - if rowSize <= 0 { - return 0, errors.New("xref stream: zero row size") - } - - // /Index is [first count first count …]; defaults to [0 Size]. - var index []int - if arr, ok := d.Array("Index"); ok { - for _, v := range arr { - n, ok := v.(Integer) - if !ok { - return 0, errors.New("xref stream: /Index entry not integer") - } - index = append(index, int(n)) - } - } else { - size, ok := d.Int("Size") - if !ok { - return 0, errors.New("xref stream: /Size missing") - } - index = []int{0, int(size)} - } - - rowIdx := 0 - for i := 0; i+1 < len(index); i += 2 { - first := index[i] - count := index[i+1] - for j := 0; j < count; j++ { - start := rowIdx * rowSize - if start+rowSize > len(decoded) { - return 0, errors.New("xref stream: data truncated") - } - row := decoded[start : start+rowSize] - rowIdx++ - - // Default type when W[0]==0 is 1. - var t uint64 = 1 - off := 0 - if w[0] > 0 { - t = readBigEndian(row[0:w[0]]) - off = w[0] - } - f1 := readBigEndian(row[off : off+w[1]]) - off += w[1] - f2 := readBigEndian(row[off : off+w[2]]) - - ref := Reference{Number: first + j, Generation: int(f2)} - switch t { - case 0: - // free entry; ignore - case 1: - ref.Generation = int(f2) - if _, exists := r.xref[ref]; !exists { - r.xref[ref] = xrefEntry{kind: 1, offset: int64(f1), generation: int(f2)} - } - case 2: - ref.Generation = 0 - if _, exists := r.xref[ref]; !exists { - r.xref[ref] = xrefEntry{ - kind: 2, - objStmNum: int(f1), - objStmIdx: int(f2), - } - } - default: - // Unknown type per spec: skip. - } - } - } - - if v, ok := d.Get("Prev"); ok { - if n, ok := v.(Integer); ok { - return int64(n), nil - } - } - return 0, nil -} - -// readCompressedObject extracts an object from an object stream (ObjStm). -func (r *Reader) readCompressedObject(objStmNum, idx int, expect Reference) (Object, error) { - streamRef := Reference{Number: objStmNum, Generation: 0} - v, err := r.Resolve(streamRef) - if err != nil { - return nil, fmt.Errorf("ObjStm %d %d R: %w", objStmNum, 0, err) - } - stm, ok := v.(*Stream) - if !ok { - return nil, fmt.Errorf("ObjStm %d resolved to %T, want stream", objStmNum, v) - } - d := stm.Dict - if t, ok := d.Name("Type"); !ok || t != "ObjStm" { - return nil, fmt.Errorf("ObjStm %d not of /Type ObjStm", objStmNum) - } - n, ok := d.Int("N") - if !ok { - return nil, fmt.Errorf("ObjStm %d missing /N", objStmNum) - } - first, ok := d.Int("First") - if !ok { - return nil, fmt.Errorf("ObjStm %d missing /First", objStmNum) - } - content, err := stm.Content() - if err != nil { - return nil, err - } - // /First and /N are attacker-controlled; bound them against the decoded - // length — no ObjStm holds more entries (or a longer header) than bytes. - if first < 0 || first > int64(len(content)) { - return nil, fmt.Errorf("ObjStm %d: /First %d out of range for %d-byte stream", objStmNum, first, len(content)) - } - if n < 0 || n > int64(len(content)) { - return nil, fmt.Errorf("ObjStm %d: implausible /N %d for %d-byte stream", objStmNum, n, len(content)) - } - - // Read N pairs of (objNum, offset) from the header section. - header := content[:first] - lx := lex.New(header) - type pair struct { - num int - offset int - } - // /N is attacker-controlled and only loosely bounded by len(content); a huge - // value must not preallocate — append grows pairs to the real entry count. - const maxObjStmPrealloc = 1 << 12 - pairs := make([]pair, 0, min(int(n), maxObjStmPrealloc)) - for i := int64(0); i < n; i++ { - t1, err := lx.Next() - if err != nil || t1.Kind != lex.Integer { - return nil, fmt.Errorf("ObjStm %d header: bad object number", objStmNum) - } - t2, err := lx.Next() - if err != nil || t2.Kind != lex.Integer { - return nil, fmt.Errorf("ObjStm %d header: bad offset", objStmNum) - } - num, _ := strconv.Atoi(string(t1.Bytes)) - off, _ := strconv.Atoi(string(t2.Bytes)) - pairs = append(pairs, pair{num: num, offset: off}) - } - if idx < 0 || idx >= len(pairs) { - return nil, fmt.Errorf("ObjStm %d: index %d out of range (%d entries)", objStmNum, idx, len(pairs)) - } - if pairs[idx].num != expect.Number { - return nil, fmt.Errorf("ObjStm %d: index %d declares object %d, expected %d", - objStmNum, idx, pairs[idx].num, expect.Number) - } - objStart := int(first) + pairs[idx].offset - if objStart < 0 || objStart >= len(content) { - return nil, fmt.Errorf("ObjStm %d: object %d offset %d out of range", objStmNum, expect.Number, objStart) - } - bodyLex := lex.New(content) - bodyLex.SetPos(objStart) - bp := newParser(bodyLex, r) - return bp.parseObject() -} - -// recoverXref scans the file for "obj" tokens and rebuilds the xref table. -// Called when the declared xref location is broken. -func (r *Reader) recoverXref() error { - // Find every "N G obj" occurrence in the file. - for i := 0; i < len(r.buf); i++ { - if i+3 > len(r.buf) || string(r.buf[i:i+3]) != "obj" { - continue - } - // Must be preceded by whitespace, two integers, whitespace. - j := i - 1 - for j >= 0 && lex.IsWhitespace(r.buf[j]) { - j-- - } - genEnd := j + 1 - for j >= 0 && r.buf[j] >= '0' && r.buf[j] <= '9' { - j-- - } - genStart := j + 1 - if genStart == genEnd { - continue - } - for j >= 0 && lex.IsWhitespace(r.buf[j]) { - j-- - } - numEnd := j + 1 - for j >= 0 && r.buf[j] >= '0' && r.buf[j] <= '9' { - j-- - } - numStart := j + 1 - if numStart == numEnd { - continue - } - num, err1 := strconv.Atoi(string(r.buf[numStart:numEnd])) - gen, err2 := strconv.Atoi(string(r.buf[genStart:genEnd])) - if err1 != nil || err2 != nil { - continue - } - // "obj" must be followed by whitespace. - if i+3 < len(r.buf) && !lex.IsWhitespace(r.buf[i+3]) { - continue - } - ref := Reference{Number: num, Generation: gen} - if _, exists := r.xref[ref]; !exists { - r.xref[ref] = xrefEntry{kind: 1, offset: int64(numStart), generation: gen} - } - i += 3 - } - - // Also try to find the trailer dict. - if r.trailer == nil { - idx := bytes.LastIndex(r.buf, []byte("trailer")) - if idx >= 0 { - lx := lex.New(r.buf) - lx.SetPos(idx + len("trailer")) - p := newParser(lx, r) - tok, err := p.next() - if err == nil && tok.Kind == lex.DictStart { - if d, err := p.parseDict(); err == nil { - r.trailer = d - } - } - } - } - // If still no trailer but we have a /Root somewhere, try harder by - // finding a dict with /Root in the recovered objects. - if r.trailer == nil { - for ref := range r.xref { - obj, err := r.Resolve(ref) - if err != nil { - continue - } - d, ok := obj.(*Dict) - if !ok { - continue - } - if t, ok := d.Name("Type"); ok && t == "Catalog" { - t := newDict(r) - t.set("Root", ref) - r.trailer = t - break - } - } - } - if r.trailer == nil { - return errors.New("xref recovery: no trailer found") - } - return nil -} - -// readBigEndian reads a big-endian unsigned integer of the given byte -// width. -func readBigEndian(b []byte) uint64 { - var v uint64 - for _, c := range b { - v = (v << 8) | uint64(c) - } - return v -} - -func skipEOL(buf []byte, pos int) int { - for pos < len(buf) { - c := buf[pos] - if c == '\r' { - pos++ - if pos < len(buf) && buf[pos] == '\n' { - pos++ - } - return pos - } - if c == '\n' { - return pos + 1 - } - if c == ' ' || c == '\t' { - pos++ - continue - } - return pos - } - return pos -} - -func indexEOL(buf []byte, pos int) int { - for i := pos; i < len(buf); i++ { - if buf[i] == '\r' || buf[i] == '\n' { - return i - } - } - return -1 -} diff --git a/xref_test.go b/xref_test.go deleted file mode 100644 index 435d536..0000000 --- a/xref_test.go +++ /dev/null @@ -1,461 +0,0 @@ -package pdfdisassembler - -import ( - "bytes" - "compress/zlib" - "fmt" - "runtime" - "testing" -) - -// buildXrefStreamPDF synthesises a PDF that uses an xref stream rather -// than a classical xref table. -func buildXrefStreamPDF(t *testing.T) []byte { - t.Helper() - var buf bytes.Buffer - off := func() int { return buf.Len() } - fmt.Fprint(&buf, "%PDF-2.0\n%\xE2\xE3\xCF\xD3\n") - - offsets := make([]int, 4) - offsets[1] = off() - fmt.Fprint(&buf, "1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - offsets[2] = off() - fmt.Fprint(&buf, "2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - - // Build the xref stream data: 3 entries (objects 0,1,2). - // Each entry: [type(1), offset(3), gen(1)] big-endian. - rowSize := 5 - rows := []byte{} - add := func(typ, f1, f2 uint64) { - rows = append(rows, byte(typ)) - rows = append(rows, byte(f1>>16), byte(f1>>8), byte(f1)) - rows = append(rows, byte(f2)) - } - add(0, 0, 0xFFFF) // free - add(1, uint64(offsets[1]), 0) - add(1, uint64(offsets[2]), 0) - _ = rowSize - - var zbuf bytes.Buffer - zw := zlib.NewWriter(&zbuf) - zw.Write(rows) - zw.Close() - compressed := zbuf.Bytes() - - xrefOff := off() - fmt.Fprintf(&buf, - "3 0 obj\n<< /Type /XRef /Size 3 /W [1 3 1] /Root 1 0 R /Filter /FlateDecode /Length %d >>\nstream\n", - len(compressed)) - buf.Write(compressed) - fmt.Fprint(&buf, "\nendstream\nendobj\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xrefOff) - return buf.Bytes() -} - -func TestXrefStream(t *testing.T) { - data := buildXrefStreamPDF(t) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - cat, err := r.Catalog() - if err != nil { - t.Fatalf("Catalog: %v", err) - } - if n, ok := cat.Name("Type"); !ok || n != "Catalog" { - t.Fatalf("/Type %q ok=%v", n, ok) - } - if r.Version() != "2.0" { - t.Fatalf("version %q", r.Version()) - } -} - -func TestXrefRecovery(t *testing.T) { - // Build a PDF, then point startxref to garbage. - data := buildMinimalPDF(t) - // Replace startxref offset to an invalid number. - idx := bytes.Index(data, []byte("startxref")) - if idx < 0 { - t.Fatal("startxref not found in test data") - } - // Overwrite the offset following the "startxref\n" with 999999. - off := idx + len("startxref\n") - for i := off; i < len(data); i++ { - if data[i] == '\n' { - // Overwrite the digits between off and i with 999999. - width := i - off - junk := "9999999"[:width] - copy(data[off:i], junk) - break - } - } - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open after recovery: %v", err) - } - defer r.Close() - cat, err := r.Catalog() - if err != nil { - t.Fatalf("Catalog after recovery: %v", err) - } - if n, ok := cat.Name("Type"); !ok || n != "Catalog" { - t.Fatalf("/Type %q ok=%v", n, ok) - } -} - -// buildXrefStreamPDFWithW builds a PDF whose cross-reference stream (obj 3) -// declares the given /W array and carries content as its (unfiltered) row data. -func buildXrefStreamPDFWithW(t *testing.T, wArray, content string) []byte { - t.Helper() - var buf bytes.Buffer - off := map[int]int{} - w := func(n int, body string) { - off[n] = buf.Len() - fmt.Fprintf(&buf, "%d 0 obj\n%s\nendobj\n", n, body) - } - buf.WriteString("%PDF-1.5\n%\xe2\xe3\xcf\xd3\n") - w(1, "<< /Type /Catalog /Pages 2 0 R >>") - w(2, "<< /Type /Pages /Kids [] /Count 0 >>") - off[3] = buf.Len() - fmt.Fprintf(&buf, "3 0 obj\n<< /Type /XRef /W %s /Size 1 /Root 1 0 R /Length %d >>\nstream\n", - wArray, len(content)) - buf.WriteString(content) - buf.WriteString("\nendstream\nendobj\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", off[3]) - return buf.Bytes() -} - -// buildObjStmPDF builds a PDF whose catalog (obj 1) lives inside an object -// stream (obj 3), reached via a type-2 entry in the xref stream (obj 2). The -// ObjStm dict declares declaredN / declaredFirst, which the caller can set to -// hostile values; the actual stream is always "1 0 " + catalogBody. -func buildObjStmPDF(t *testing.T, declaredN, declaredFirst int64, catalogBody string) []byte { - t.Helper() - objstm := "1 0 " + catalogBody // header "1 0 " (4 bytes), catalog at offset 4 - - var buf bytes.Buffer - buf.WriteString("%PDF-1.5\n%\xe2\xe3\xcf\xd3\n") - - off3 := buf.Len() - fmt.Fprintf(&buf, "3 0 obj\n<< /Type /ObjStm /N %d /First %d /Length %d >>\nstream\n%s\nendstream\nendobj\n", - declaredN, declaredFirst, len(objstm), objstm) - - off2 := buf.Len() - rows := []byte{ - 0x00, 0x00, 0x00, 0x00, // obj 0: free - 0x02, 0x00, 0x03, 0x00, // obj 1: type 2 -> ObjStm 3, index 0 - 0x01, byte(off2 >> 8), byte(off2), 0x00, // obj 2: type 1 @ off2 - 0x01, byte(off3 >> 8), byte(off3), 0x00, // obj 3: type 1 @ off3 - } - fmt.Fprintf(&buf, "2 0 obj\n<< /Type /XRef /W [ 1 2 1 ] /Index [ 0 4 ] /Size 4 /Root 1 0 R /Length %d >>\nstream\n", - len(rows)) - buf.Write(rows) - buf.WriteString("\nendstream\nendobj\n") - - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", off2) - return buf.Bytes() -} - -func TestXrefStreamNegativeWidthRecovers(t *testing.T) { - // /W [1 -2 10]: the negative width makes the row decode slice row[1:-1]. - // The parser must recover (not panic), leaving the catalog reachable. - data := buildXrefStreamPDFWithW(t, "[ 1 -2 10 ]", "123456789") - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if _, err := r.Catalog(); err != nil { - t.Fatalf("Catalog after recovery: %v", err) - } -} - -// Control: a well-formed ObjStm catalog must resolve, proving the harness and -// the type-2 path work (so the hostile cases below aren't false positives). -func TestObjStmCatalogBaseline(t *testing.T) { - data := buildObjStmPDF(t, 1, 4, "<< /Type /Catalog >>") - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - cat, err := r.Catalog() - if err != nil { - t.Fatalf("Catalog: %v", err) - } - if n, ok := cat.Name("Type"); !ok || n != "Catalog" { - t.Fatalf("/Type %q ok=%v", n, ok) - } -} - -func TestObjStmRejectsHostileHeader(t *testing.T) { - // Absurd attacker-controlled /First and /N must surface as errors, not - // slice/make panics. - tests := []struct { - name string - declaredN int64 - declaredFirst int64 - }{ - {"first beyond stream", 1, 1 << 60}, - {"absurd N", 1 << 60, 4}, - } - for _, tt := range tests { - t.Run(tt.name, func(t *testing.T) { - data := buildObjStmPDF(t, tt.declaredN, tt.declaredFirst, "<< /Type /Catalog >>") - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if _, err := r.Catalog(); err == nil { - t.Fatal("expected an error, got nil") - } - }) - } -} - -// buildObjStmHugeN builds a PDF reachable via an xref stream where object 4 -// is a type-2 entry inside ObjStm 3. The ObjStm decodes to contentLen zero -// bytes (cheap: FlateDecode compresses them to a few KB) but declares -// /N == contentLen. A reader that trusts /N as a slice capacity balloons the -// few-KB file into contentLen*sizeof(pair) bytes before parsing a single entry. -func buildObjStmHugeN(t *testing.T, contentLen int) []byte { - t.Helper() - var zbuf bytes.Buffer - zw := zlib.NewWriter(&zbuf) - if _, err := zw.Write(make([]byte, contentLen)); err != nil { - t.Fatal(err) - } - zw.Close() - compressed := zbuf.Bytes() - - var buf bytes.Buffer - buf.WriteString("%PDF-1.5\n%\xe2\xe3\xcf\xd3\n") - - off3 := buf.Len() - fmt.Fprintf(&buf, "3 0 obj\n<< /Type /ObjStm /N %d /First 4 /Filter /FlateDecode /Length %d >>\nstream\n", - contentLen, len(compressed)) - buf.Write(compressed) - buf.WriteString("\nendstream\nendobj\n") - - off2 := buf.Len() - rows := []byte{ - 0x01, byte(off3 >> 8), byte(off3), 0x00, // obj 3: type 1 @ off3 - 0x02, 0x00, 0x03, 0x00, // obj 4: type 2 -> ObjStm 3, index 0 - } - fmt.Fprintf(&buf, "2 0 obj\n<< /Type /XRef /W [ 1 2 1 ] /Index [ 3 2 ] /Size 5 /Root 1 0 R /Length %d >>\nstream\n", - len(rows)) - buf.Write(rows) - buf.WriteString("\nendstream\nendobj\n") - - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", off2) - return buf.Bytes() -} - -func TestObjStmHugeNDoesNotAmplifyAllocation(t *testing.T) { - const contentLen = 16 << 20 // == DefaultMaxStreamSize, the largest /N can be - data := buildObjStmHugeN(t, contentLen) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - - var before, after runtime.MemStats - runtime.GC() - runtime.ReadMemStats(&before) - _, err = r.Resolve(Reference{Number: 4, Generation: 0}) - runtime.ReadMemStats(&after) - if err == nil { - t.Fatal("expected an error resolving the hostile ObjStm entry, got nil") - } - // Decoding contentLen zero bytes legitimately costs ~2*contentLen; the - // 256 MiB the /N prealloc would add is well past this ceiling. - const limit = 128 << 20 - if used := after.TotalAlloc - before.TotalAlloc; used > limit { - t.Fatalf("resolving a %d-byte ObjStm allocated %d bytes (> %d limit); /N is amplifying allocation", - contentLen, used, limit) - } -} - -// Capping the prealloc must not truncate a legitimate ObjStm whose entry count -// exceeds the cap: object 100+i carries the value 100+i, and resolving one past -// the cap must still return its exact value (proving append grew the slice). -func TestObjStmManyObjectsResolvePastPrealloc(t *testing.T) { - const m = 5000 // > the internal maxObjStmPrealloc (4096) - - var body bytes.Buffer - offsets := make([]int, m) - for i := 0; i < m; i++ { - offsets[i] = body.Len() - fmt.Fprintf(&body, "%d ", 100+i) - } - var head bytes.Buffer - for i := 0; i < m; i++ { - fmt.Fprintf(&head, "%d %d ", 100+i, offsets[i]) - } - content := head.String() + body.String() - - var buf bytes.Buffer - buf.WriteString("%PDF-1.5\n%\xe2\xe3\xcf\xd3\n") - off1 := buf.Len() - fmt.Fprintf(&buf, "1 0 obj\n<< /Type /ObjStm /N %d /First %d /Length %d >>\nstream\n", - m, head.Len(), len(content)) - buf.WriteString(content) - buf.WriteString("\nendstream\nendobj\n") - - last := 100 + m - 1 - off2 := buf.Len() - rows := []byte{ - 0x01, byte(off1 >> 8), byte(off1), 0x00, 0x00, // obj 1: type 1 @ off1 - 0x02, 0x00, 0x01, byte((m - 1) >> 8), byte((m - 1) & 0xff), // last obj: type 2 -> ObjStm 1, idx m-1 - } - fmt.Fprintf(&buf, "2 0 obj\n<< /Type /XRef /W [ 1 2 2 ] /Index [ 1 1 %d 1 ] /Size %d /Root 1 0 R /Length %d >>\nstream\n", - last, last+1, len(rows)) - buf.Write(rows) - buf.WriteString("\nendstream\nendobj\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", off2) - - r, err := Open(bytes.NewReader(buf.Bytes())) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - v, err := r.Resolve(Reference{Number: last, Generation: 0}) - if err != nil { - t.Fatalf("Resolve object %d: %v", last, err) - } - n, ok := v.(Integer) - if !ok || int(n) != last { - t.Fatalf("object %d resolved to %v (%T), want Integer %d", last, v, v, last) - } -} - -// With no startxref and no "trailer" keyword, recovery must scan the rebuilt -// objects for a /Type /Catalog and synthesise a trailer pointing at it. -func TestRecoverXrefViaCatalogScan(t *testing.T) { - var buf bytes.Buffer - buf.WriteString("%PDF-1.7\n%\xe2\xe3\xcf\xd3\n") - buf.WriteString("1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - buf.WriteString("2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - buf.WriteString("%%EOF\n") - - r, err := Open(bytes.NewReader(buf.Bytes())) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - cat, err := r.Catalog() - if err != nil { - t.Fatalf("Catalog: %v", err) - } - if n, ok := cat.Name("Type"); !ok || n != "Catalog" { - t.Fatalf("/Type %q ok=%v, want Catalog", n, ok) - } -} - -// An incremental update appends a second xref section whose /Prev points back -// at the first. The newer section must win: object 1 resolves to its updated -// body, and trailer keys present only in the older section still resolve. -func TestPrevChainNewestSectionWins(t *testing.T) { - var buf bytes.Buffer - w := func(s string) int { off := buf.Len(); buf.WriteString(s); return off } - w("%PDF-1.7\n%\xe2\xe3\xcf\xd3\n") - off1v1 := w("1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n") - off2 := w("2 0 obj\n<< /Type /Pages /Count 0 /Kids [] >>\nendobj\n") - - xref1 := buf.Len() - fmt.Fprintf(&buf, "xref\n0 3\n%010d %05d f \n%010d %05d n \n%010d %05d n \n", - 0, 65535, off1v1, 0, off2, 0) - buf.WriteString("trailer\n<< /Size 3 /Root 1 0 R /Info 2 0 R >>\n") - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xref1) - - off1v2 := buf.Len() - buf.WriteString("1 0 obj\n<< /Type /Catalog /Pages 2 0 R /Lang (en-US) >>\nendobj\n") - xref2 := buf.Len() - fmt.Fprintf(&buf, "xref\n1 1\n%010d %05d n \n", off1v2, 0) - fmt.Fprintf(&buf, "trailer\n<< /Size 3 /Root 1 0 R /Prev %d >>\n", xref1) - fmt.Fprintf(&buf, "startxref\n%d\n%%%%EOF\n", xref2) - - r, err := Open(bytes.NewReader(buf.Bytes())) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - - cat, err := r.Catalog() - if err != nil { - t.Fatalf("Catalog: %v", err) - } - if lang, ok := cat.String("Lang"); !ok || lang != "en-US" { - t.Errorf("/Lang = %q ok=%v, want en-US (updated object 1 not used)", lang, ok) - } - // /Info lives only in the older trailer; the merge must preserve it. - if _, ok := r.Trailer().Get("Info"); !ok { - t.Error("older trailer's /Info lost after /Prev merge") - } -} - -// A classical xref subsection declaring far more entries than the file holds -// must be handled gracefully (recover), not over-read the buffer. The bound is -// also overflow-safe on 32-bit by construction (untestable on a 64-bit run). -func TestClassicalXrefHugeCountRecovers(t *testing.T) { - data := buildMinimalPDF(t) - data = bytes.Replace(data, []byte("xref\n0 5\n"), []byte("xref\n0 999999999\n"), 1) - r, err := Open(bytes.NewReader(data)) - if err != nil { - t.Fatalf("Open: %v", err) - } - defer r.Close() - if _, err := r.Catalog(); err != nil { - t.Fatalf("Catalog: %v", err) - } -} - -// CR/LF/CRLF line-ending handling for the classical xref reader (§7.5.4). -func TestSkipEOL(t *testing.T) { - cases := []struct { - name string - buf string - pos int - want int - }{ - {"crlf", "\r\nX", 0, 2}, - {"lf", "\nX", 0, 1}, - {"lone_cr", "\rX", 0, 1}, - {"cr_at_eof", "\r", 0, 1}, - {"leading_spaces_then_lf", " \nX", 0, 3}, - {"tab_then_crlf", "\t\r\n", 0, 3}, - {"non_whitespace_stays_put", "Xyz", 0, 0}, - {"spaces_then_eof", " ", 0, 2}, - {"empty", "", 0, 0}, - {"pos_already_past_end", "ab", 2, 2}, - } - for _, tc := range cases { - t.Run(tc.name, func(t *testing.T) { - if got := skipEOL([]byte(tc.buf), tc.pos); got != tc.want { - t.Errorf("skipEOL(%q, %d) = %d, want %d", tc.buf, tc.pos, got, tc.want) - } - }) - } -} - -func TestIndexEOL(t *testing.T) { - cases := []struct { - buf string - pos int - want int - }{ - {"ab\ncd", 0, 2}, - {"ab\rcd", 0, 2}, - {"abcd", 0, -1}, - {"a\nb", 2, -1}, - {"", 0, -1}, - } - for _, tc := range cases { - if got := indexEOL([]byte(tc.buf), tc.pos); got != tc.want { - t.Errorf("indexEOL(%q, %d) = %d, want %d", tc.buf, tc.pos, got, tc.want) - } - } -}