From 8ab647a27c18018479268c476f8805bc8f2b7d51 Mon Sep 17 00:00:00 2001 From: dev-belly Date: Wed, 30 Sep 2026 16:53:07 +0200 Subject: [PATCH] Protect high-risk audit CSV from spreadsheet formulas --- README.md | 18 +++++++++++------- src/reporting.py | 3 ++- src/review_plan.py | 10 ++-------- src/utils.py | 7 +++++++ tests/test_reporting.py | 26 ++++++++++++++++++++++++++ 5 files changed, 48 insertions(+), 16 deletions(-) diff --git a/README.md b/README.md index 14afb77..06daa5b 100644 --- a/README.md +++ b/README.md @@ -3,7 +3,7 @@ [![CI](https://github.com/dev-belly/AuditLens/actions/workflows/ci.yml/badge.svg)](https://github.com/dev-belly/AuditLens/actions/workflows/ci.yml) [![Python 3.11+](https://img.shields.io/badge/python-3.11%2B-blue.svg)](https://www.python.org/downloads/) [![License: MIT](https://img.shields.io/badge/license-MIT-green.svg)](LICENSE) -[![Tests: 462](https://img.shields.io/badge/tests-462%20passing-brightgreen.svg)](#testing) +[![Tests: 463](https://img.shields.io/badge/tests-463%20passing-brightgreen.svg)](#testing) **Financial anomaly detection and audit analytics over a 30,000-voucher general ledger.** @@ -58,7 +58,7 @@ Three design commitments drive everything else: | Anomalies caught by neither | **15** | | Isolation Forest | ROC-AUC **0.816**, precision 0.291, recall 0.286 | | Benford first-digit MAD | **0.00218** — close conformity | -| Test suite | **462 tests**, including dashboard render, warehouse and review-plan checks | +| Test suite | **463 tests**, including dashboard render, warehouse and review-plan checks | ### The most important number here is 98.2%, and it is a warning @@ -395,6 +395,10 @@ features that drove it, because "the model said so" is not an audit explanation. reason — a score with no explanation is treated as a bug, and currently zero of the 248 High/Critical vouchers violate it. +The circulated high-risk voucher CSV prefixes formula-like text with an +apostrophe for spreadsheet viewing; scores, source fields and ordering remain +unchanged in the research data. + --- ## Audit review workpaper @@ -559,7 +563,7 @@ AuditLens/ │ ├── components.py # KPI cards, risk badges, charts, tables │ └── views/ # the six pages ├── sql/audit_queries.sql # 15 named business queries -├── tests/ # 462 tests, incl. cross-process reproducibility +├── tests/ # 463 tests, incl. cross-process reproducibility ├── docs/ │ ├── architecture.md # design decisions, data contracts, what is deliberately excluded │ ├── methodology.md # every threshold, every weight, every mistake @@ -573,7 +577,7 @@ AuditLens/ │ ├── notebook_content.py # what the three notebooks contain, kept apart from the builder │ ├── capture_screenshots.py # headless captures of all six dashboard pages, via DevTools Protocol │ └── ci_summary.py # headline figures for the CI job summary -├── .github/workflows/ci.yml # pipeline + 462 tests + a reproducibility check, on 3.11 and 3.12 +├── .github/workflows/ci.yml # pipeline + 463 tests + a reproducibility check, on 3.11 and 3.12 └── Makefile ``` @@ -744,7 +748,7 @@ builder, so two consecutive builds are byte-identical. ```bash pip install -r requirements.txt python src/run_pipeline.py # ~30 seconds, deterministic -python -m pytest tests/ -q # 462 tests +python -m pytest tests/ -q # 463 tests ``` Two consecutive runs produce byte-identical reports under `outputs/reports/`. This is @@ -752,7 +756,7 @@ enforced by `tests/test_reproducibility.py`, not assumed. ## Testing -462 tests. `make test` runs the lot; `make test-fast` skips the dashboard render suite. +463 tests. `make test` runs the lot; `make test-fast` skips the dashboard render suite. | File | Tests | What it pins down | |---|---|---| @@ -764,7 +768,7 @@ enforced by `tests/test_reproducibility.py`, not assumed. | `test_feature_engineering.py` | 9 | The model's feature matrix: no ground-truth column can reach it, the matrix and its descriptions cover exactly the same set, and the committed `model_metrics.json` records the features the code actually built | | `test_utils.py` | 19 | `as_flag_series` across every dtype the label takes, including the two string traps | | `test_ci_summary.py` | 9 | The CI job summary renders, carries its benchmark caveat, and degrades to a dash on schema drift rather than breaking the build | -| `test_reporting.py` | 7 | The flag-vs-anomaly distinction in `detector_overlap` and stable ordering for equal risk scores in the review extract | +| `test_reporting.py` | 8 | The flag-vs-anomaly distinction in `detector_overlap`, stable ordering for equal risk scores, and formula-safe text in the review extract | | `test_reproducibility.py` | 4 | Runs the generator in two subprocesses with different `PYTHONHASHSEED` values and compares hashes | | `test_notebooks.py` | 30 | Every cell that prints or plots carries output, every notebook embeds a chart, and no random Styler id or logged timestamp survives into a committed notebook | | `test_sql_queries.py` | 20 | Splits the `-- name:` query library, and executes all 15 queries against the warehouse — the guard against `run_sql_file` turning a broken query into an empty frame | diff --git a/src/reporting.py b/src/reporting.py index 31a12ca..64f26dc 100644 --- a/src/reporting.py +++ b/src/reporting.py @@ -55,6 +55,7 @@ load_dataframe, load_json, save_json, + spreadsheet_safe_cell, ) LOGGER = get_logger(__name__) @@ -430,7 +431,7 @@ def write_high_risk_extract(df: pd.DataFrame) -> Path: extract = df.loc[df["risk_level"].isin(HIGH_RISK_LEVELS), columns].sort_values( ["audit_risk_score", "transaction_id"], ascending=[False, True] ) - extract.to_csv(HIGH_RISK_CSV, index=False, encoding="utf-8-sig") + extract.map(spreadsheet_safe_cell).to_csv(HIGH_RISK_CSV, index=False, encoding="utf-8-sig") LOGGER.info("Wrote %s (%s vouchers)", HIGH_RISK_CSV.name, len(extract)) return HIGH_RISK_CSV diff --git a/src/review_plan.py b/src/review_plan.py index 9d13990..934ca46 100644 --- a/src/review_plan.py +++ b/src/review_plan.py @@ -37,6 +37,7 @@ get_logger, load_dataframe, save_json, + spreadsheet_safe_cell, ) LOGGER = get_logger(__name__) @@ -240,13 +241,6 @@ def build_review_plan( return ReviewPlan(queue=queue, summary=summary) -def _spreadsheet_safe(value: Any) -> Any: - """Keep text cells from being interpreted as formulas when opened as CSV.""" - if isinstance(value, str) and value.lstrip().startswith(("=", "+", "-", "@")): - return "'" + value - return value - - def write_review_plan( plan: ReviewPlan, queue_path: Path = REVIEW_PLAN_CSV, @@ -255,7 +249,7 @@ def write_review_plan( """Write the auditor queue and a checksum-bound selection record.""" queue_path.parent.mkdir(parents=True, exist_ok=True) summary_path.parent.mkdir(parents=True, exist_ok=True) - export_queue = plan.queue.map(_spreadsheet_safe) + export_queue = plan.queue.map(spreadsheet_safe_cell) contents = export_queue.to_csv(index=False, lineterminator="\n", float_format="%.10g") summary = {**plan.summary, "queue_sha256": hashlib.sha256(contents.encode()).hexdigest()} queue_path.write_bytes(contents.encode("utf-8")) diff --git a/src/utils.py b/src/utils.py index fdeb724..92cf8db 100644 --- a/src/utils.py +++ b/src/utils.py @@ -421,6 +421,13 @@ def load_json(path: Path) -> dict[str, Any]: return json.load(handle) +def spreadsheet_safe_cell(value: Any) -> Any: + """Prefix formula-like text in auditor-facing CSV exports.""" + if isinstance(value, str) and value.lstrip().startswith(("=", "+", "-", "@")): + return "'" + value + return value + + def save_dataframe(df: pd.DataFrame, path: Path) -> Path: """Persist a dataframe to parquet (or CSV when the suffix is ``.csv``).""" path.parent.mkdir(parents=True, exist_ok=True) diff --git a/tests/test_reporting.py b/tests/test_reporting.py index addd3d7..c48ca6a 100644 --- a/tests/test_reporting.py +++ b/tests/test_reporting.py @@ -46,6 +46,32 @@ def test_high_risk_extract_orders_equal_scores_consistently( assert pd.read_csv(output)["transaction_id"].tolist() == ["TX002", "TX001", "TX003"] +def test_high_risk_extract_escapes_formula_like_text_without_changing_source( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + output = tmp_path / "high_risk.csv" + monkeypatch.setattr(reporting, "HIGH_RISK_CSV", output) + voucher_id = '=HYPERLINK("https://example.test")' + vouchers = pd.DataFrame([{ + "transaction_id": voucher_id, + "audit_risk_score": 82.0, + "risk_level": "High", + "vendor_name": " +SUM(1,2)", + "description": "-1+2", + "risk_reason_text": "@SUM(1,2)", + }]) + + reporting.write_high_risk_extract(vouchers) + row = pd.read_csv(output, dtype=str, keep_default_na=False).iloc[0] + assert row["transaction_id"] == "'" + voucher_id + assert row["vendor_name"] == "' +SUM(1,2)" + assert row["description"] == "'-1+2" + assert row["risk_reason_text"] == "'@SUM(1,2)" + assert row["audit_risk_score"] == "82.0" + assert vouchers.loc[0, "transaction_id"] == voucher_id + assert vouchers.loc[0, "vendor_name"] == " +SUM(1,2)" + + def _frame(rows: list[dict]) -> pd.DataFrame: """Build a minimal scored frame from (rule, model, anomaly) triples.""" return pd.DataFrame(