Skip to content

Commit 3467005

Browse files
phrrngtnclaude
andcommitted
examples/pdf_overlay: SHA-preserving PDF annotation overlay from profiling
Closes the extract->profile->annotate loop: bb_pdf extracts the box table, DuckDB SQL imputes a domain per box, emitted as a standalone XFDF sidecar that references the PDF via <f href> so the source SHA256 is invariant (no baked copy). Coordinate flip bb_pdf top-left -> PDF bottom-left; validated visually on IRS f1040sc (1682 boxes, pixel-accurate, source byte-identical). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
1 parent e0c0884 commit 3467005

4 files changed

Lines changed: 112 additions & 0 deletions

File tree

‎examples/pdf_overlay/.gitignore‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
1+
# sample data + generated artifacts
2+
*.pdf
3+
*.xfdf
4+
*.boxes.json
5+
*.png

‎examples/pdf_overlay/README.md‎

Lines changed: 22 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,22 @@
1+
# pdf_overlay — context-sensitive annotations from profiling, SHA-preserving
2+
3+
Closes the extract → profile → annotate loop on a real PDF: `bb_pdf` extracts the
4+
box table, DuckDB SQL imputes a domain per box, and the result is emitted as a
5+
standalone **XFDF sidecar** — the source PDF's bytes are never touched, so its
6+
**SHA256 is invariant** (baking annotations into a copy would mint a new SHA; we
7+
don't). Acrobat/Reader render the `.xfdf` over the PDF via its `<f href>`; viewers
8+
without XFDF support get the same findings as a draw-on-render overlay layer.
9+
10+
```bash
11+
./fetch_sample.sh # public-domain IRS Schedule C
12+
BBOXES_DUCKDB_EXT=/path/to/bboxes.duckdb_extension \
13+
python overlay.py f1040sc.pdf # -> f1040sc.xfdf (source untouched)
14+
```
15+
16+
Notes:
17+
- **Coordinates:** `bb_pdf` is top-left-origin (y down); XFDF is PDF bottom-left,
18+
so rects flip against page height (`lly=H-(y+h)`, `ury=H-y`). Validated visually.
19+
- **Imputation:** the SQL `CASE` is a stand-in — replace it with a join against
20+
blobfilters domain fingerprints (catalog/authority) for real domain matching,
21+
and colour out-of-domain values (residue) as violators.
22+
- Deps: `pypdfium2`, `pillow`, and the `duckdb` CLI with the `bboxes` extension.
Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
1+
#!/usr/bin/env bash
2+
# Public-domain sample: IRS Schedule C (Form 1040). Born-digital, complex layout.
3+
set -euo pipefail
4+
curl -sL --max-time 40 "https://www.irs.gov/pub/irs-pdf/f1040sc.pdf" -o f1040sc.pdf
5+
echo "fetched f1040sc.pdf ($(wc -c < f1040sc.pdf) bytes)"

‎examples/pdf_overlay/overlay.py‎

Lines changed: 80 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,80 @@
1+
#!/usr/bin/env python3
2+
"""Context-sensitive annotation overlay for a PDF, driven by blobboxes bb_pdf
3+
extraction + SQL domain imputation, emitted as a standalone XFDF *sidecar*.
4+
5+
The source PDF is NEVER modified: annotations live in a separate .xfdf that
6+
references the PDF via <f href>, so the document's SHA256 is invariant. (Baking
7+
annotations into a copy would mint a new SHA — deliberately avoided.)
8+
9+
Coordinate note: bb_pdf returns TOP-LEFT-origin boxes (x,y,w,h, y down).
10+
XFDF rects are PDF user space (bottom-left), so we flip against page height:
11+
llx=x, urx=x+w, lly=H-(y+h), ury=H-y.
12+
"""
13+
import argparse, json, hashlib, html, os, subprocess, sys
14+
from collections import Counter
15+
import pypdfium2 as pdfium
16+
from PIL import ImageDraw
17+
18+
COLOR = {"year":"#2563eb","ein":"#dc2626","currency":"#16a34a","number":"#0d9488",
19+
"lineref":"#ea580c","integer":"#7c3aed","text":"#94a3b8"}
20+
21+
# bb_pdf extract + SQL domain imputation (a stand-in for catalog/authority
22+
# matching via blobfilters; swap the CASE for a join against domain fingerprints).
23+
SQL = r"""
24+
LOAD '__EXT__';
25+
COPY (
26+
SELECT page_id, x, y, w, h, text,
27+
CASE
28+
WHEN regexp_matches(text, '^\d{4}$') THEN 'year'
29+
WHEN regexp_matches(text, '^\d{2}-?\d{7}$') THEN 'ein'
30+
WHEN regexp_matches(text, '^\$[\d,]+(\.\d{2})?$') THEN 'currency'
31+
WHEN regexp_matches(text, '^[\d,]+\.\d+$') THEN 'number'
32+
WHEN regexp_matches(text, '^\d{1,2}[a-z]?$') THEN 'lineref'
33+
WHEN regexp_matches(text, '^\d+$') THEN 'integer'
34+
ELSE 'text'
35+
END AS domain
36+
FROM bb_pdf('__PDF__')
37+
WHERE text IS NOT NULL AND trim(text) <> ''
38+
) TO '__OUT__' (FORMAT json, ARRAY true);
39+
"""
40+
41+
def main():
42+
ap = argparse.ArgumentParser()
43+
ap.add_argument("pdf")
44+
ap.add_argument("--ext", default=os.environ.get("BBOXES_DUCKDB_EXT"),
45+
help="path to bboxes.duckdb_extension (or set BBOXES_DUCKDB_EXT)")
46+
ap.add_argument("--out", default=None, help="output .xfdf path")
47+
a = ap.parse_args()
48+
if not a.ext:
49+
sys.exit("need --ext or BBOXES_DUCKDB_EXT (path to bboxes.duckdb_extension)")
50+
out = a.out or os.path.splitext(a.pdf)[0] + ".xfdf"
51+
boxes_json = os.path.splitext(a.pdf)[0] + ".boxes.json"
52+
53+
sha_before = hashlib.sha256(open(a.pdf, "rb").read()).hexdigest()
54+
55+
sql = SQL.replace("__EXT__", a.ext).replace("__PDF__", a.pdf).replace("__OUT__", boxes_json)
56+
subprocess.run(["duckdb", "-unsigned", "-c", sql], check=True)
57+
boxes = json.load(open(boxes_json))
58+
59+
pdf = pdfium.PdfDocument(a.pdf)
60+
H = [pdf[i].get_size()[1] for i in range(len(pdf))]
61+
ann = []
62+
for b in boxes:
63+
p, x, y, w, h, dom = b["page_id"], b["x"], b["y"], b["w"], b["h"], b["domain"]
64+
c = COLOR.get(dom, "#94a3b8")
65+
llx, urx, lly, ury = x, x + w, H[p] - (y + h), H[p] - y
66+
ann.append(f'<square page="{p}" rect="{llx:.2f},{lly:.2f},{urx:.2f},{ury:.2f}" '
67+
f'color="{c}" interior-color="{c}" opacity="0.30" width="0.75" title="{dom}">'
68+
f'<contents>{html.escape(dom)}: {html.escape(b["text"][:40])}</contents></square>')
69+
open(out, "w").write('<?xml version="1.0" encoding="UTF-8"?>\n'
70+
'<xfdf xmlns="http://ns.adobe.com/xfdf/" xml:space="preserve">\n'
71+
f'<f href="{os.path.basename(a.pdf)}"/>\n<annots>\n' + "\n".join(ann) + "\n</annots>\n</xfdf>\n")
72+
73+
sha_after = hashlib.sha256(open(a.pdf, "rb").read()).hexdigest()
74+
assert sha_before == sha_after, "source PDF changed!"
75+
print(f"{len(boxes)} annotations -> {out}")
76+
print("domains:", dict(Counter(b["domain"] for b in boxes)))
77+
print(f"source SHA256 {sha_before[:16]}… PRESERVED (byte-identical)")
78+
79+
if __name__ == "__main__":
80+
main()

0 commit comments

Comments
 (0)