|
| 1 | +#!/usr/bin/env python3 |
| 2 | +"""Context-sensitive annotation overlay for a PDF, driven by blobboxes bb_pdf |
| 3 | +extraction + SQL domain imputation, emitted as a standalone XFDF *sidecar*. |
| 4 | +
|
| 5 | +The source PDF is NEVER modified: annotations live in a separate .xfdf that |
| 6 | +references the PDF via <f href>, so the document's SHA256 is invariant. (Baking |
| 7 | +annotations into a copy would mint a new SHA — deliberately avoided.) |
| 8 | +
|
| 9 | +Coordinate note: bb_pdf returns TOP-LEFT-origin boxes (x,y,w,h, y down). |
| 10 | +XFDF rects are PDF user space (bottom-left), so we flip against page height: |
| 11 | + llx=x, urx=x+w, lly=H-(y+h), ury=H-y. |
| 12 | +""" |
| 13 | +import argparse, json, hashlib, html, os, subprocess, sys |
| 14 | +from collections import Counter |
| 15 | +import pypdfium2 as pdfium |
| 16 | +from PIL import ImageDraw |
| 17 | + |
| 18 | +COLOR = {"year":"#2563eb","ein":"#dc2626","currency":"#16a34a","number":"#0d9488", |
| 19 | + "lineref":"#ea580c","integer":"#7c3aed","text":"#94a3b8"} |
| 20 | + |
| 21 | +# bb_pdf extract + SQL domain imputation (a stand-in for catalog/authority |
| 22 | +# matching via blobfilters; swap the CASE for a join against domain fingerprints). |
| 23 | +SQL = r""" |
| 24 | +LOAD '__EXT__'; |
| 25 | +COPY ( |
| 26 | + SELECT page_id, x, y, w, h, text, |
| 27 | + CASE |
| 28 | + WHEN regexp_matches(text, '^\d{4}$') THEN 'year' |
| 29 | + WHEN regexp_matches(text, '^\d{2}-?\d{7}$') THEN 'ein' |
| 30 | + WHEN regexp_matches(text, '^\$[\d,]+(\.\d{2})?$') THEN 'currency' |
| 31 | + WHEN regexp_matches(text, '^[\d,]+\.\d+$') THEN 'number' |
| 32 | + WHEN regexp_matches(text, '^\d{1,2}[a-z]?$') THEN 'lineref' |
| 33 | + WHEN regexp_matches(text, '^\d+$') THEN 'integer' |
| 34 | + ELSE 'text' |
| 35 | + END AS domain |
| 36 | + FROM bb_pdf('__PDF__') |
| 37 | + WHERE text IS NOT NULL AND trim(text) <> '' |
| 38 | +) TO '__OUT__' (FORMAT json, ARRAY true); |
| 39 | +""" |
| 40 | + |
| 41 | +def main(): |
| 42 | + ap = argparse.ArgumentParser() |
| 43 | + ap.add_argument("pdf") |
| 44 | + ap.add_argument("--ext", default=os.environ.get("BBOXES_DUCKDB_EXT"), |
| 45 | + help="path to bboxes.duckdb_extension (or set BBOXES_DUCKDB_EXT)") |
| 46 | + ap.add_argument("--out", default=None, help="output .xfdf path") |
| 47 | + a = ap.parse_args() |
| 48 | + if not a.ext: |
| 49 | + sys.exit("need --ext or BBOXES_DUCKDB_EXT (path to bboxes.duckdb_extension)") |
| 50 | + out = a.out or os.path.splitext(a.pdf)[0] + ".xfdf" |
| 51 | + boxes_json = os.path.splitext(a.pdf)[0] + ".boxes.json" |
| 52 | + |
| 53 | + sha_before = hashlib.sha256(open(a.pdf, "rb").read()).hexdigest() |
| 54 | + |
| 55 | + sql = SQL.replace("__EXT__", a.ext).replace("__PDF__", a.pdf).replace("__OUT__", boxes_json) |
| 56 | + subprocess.run(["duckdb", "-unsigned", "-c", sql], check=True) |
| 57 | + boxes = json.load(open(boxes_json)) |
| 58 | + |
| 59 | + pdf = pdfium.PdfDocument(a.pdf) |
| 60 | + H = [pdf[i].get_size()[1] for i in range(len(pdf))] |
| 61 | + ann = [] |
| 62 | + for b in boxes: |
| 63 | + p, x, y, w, h, dom = b["page_id"], b["x"], b["y"], b["w"], b["h"], b["domain"] |
| 64 | + c = COLOR.get(dom, "#94a3b8") |
| 65 | + llx, urx, lly, ury = x, x + w, H[p] - (y + h), H[p] - y |
| 66 | + ann.append(f'<square page="{p}" rect="{llx:.2f},{lly:.2f},{urx:.2f},{ury:.2f}" ' |
| 67 | + f'color="{c}" interior-color="{c}" opacity="0.30" width="0.75" title="{dom}">' |
| 68 | + f'<contents>{html.escape(dom)}: {html.escape(b["text"][:40])}</contents></square>') |
| 69 | + open(out, "w").write('<?xml version="1.0" encoding="UTF-8"?>\n' |
| 70 | + '<xfdf xmlns="http://ns.adobe.com/xfdf/" xml:space="preserve">\n' |
| 71 | + f'<f href="{os.path.basename(a.pdf)}"/>\n<annots>\n' + "\n".join(ann) + "\n</annots>\n</xfdf>\n") |
| 72 | + |
| 73 | + sha_after = hashlib.sha256(open(a.pdf, "rb").read()).hexdigest() |
| 74 | + assert sha_before == sha_after, "source PDF changed!" |
| 75 | + print(f"{len(boxes)} annotations -> {out}") |
| 76 | + print("domains:", dict(Counter(b["domain"] for b in boxes))) |
| 77 | + print(f"source SHA256 {sha_before[:16]}… PRESERVED (byte-identical)") |
| 78 | + |
| 79 | +if __name__ == "__main__": |
| 80 | + main() |
0 commit comments