From 1044b93b15cce7421d32b876b4a9495036655abc Mon Sep 17 00:00:00 2001 From: Ramesh Padmanabhaiah <22363102+codeforester@users.noreply.github.com> Date: Wed, 30 Sep 2026 20:00:51 +0530 Subject: [PATCH 1/3] security: guard delimited spreadsheet formulas --- docs/api-reference.md | 4 ++-- docs/output-contracts.md | 4 ++++ lib/python/base_cli/output.py | 12 +++++++++--- tests/test_output.py | 25 +++++++++++++++++++++++++ 4 files changed, 40 insertions(+), 5 deletions(-) diff --git a/docs/api-reference.md b/docs/api-reference.md index 56abbd1..94e0384 100644 --- a/docs/api-reference.md +++ b/docs/api-reference.md @@ -1318,7 +1318,7 @@ base_cli.option(...) ### `render_document` **Kind:** function -**Signature:** `render_document(document: 'Mapping[str, Any]', *, requested_format: 'str | None', records_key: 'str | None' = None, columns: 'Sequence[tuple[str, str]] | None' = None, stream: 'TextIO | None' = None) -> 'str'` +**Signature:** `render_document(document: 'Mapping[str, Any]', *, requested_format: 'str | None', records_key: 'str | None' = None, columns: 'Sequence[tuple[str, str]] | None' = None, stream: 'TextIO | None' = None, formula_guard: 'bool' = True) -> 'str'` **Behavior:** Render a structured report or leave terminal text to its existing renderer. @@ -1334,7 +1334,7 @@ base_cli.render_document(...) ### `render_records` **Kind:** function -**Signature:** `render_records(records: 'Iterable[Mapping[str, Any]]', *, requested_format: 'str | None', columns: 'Sequence[tuple[str, str]]', stream: 'TextIO | None' = None, footer: 'str | None' = None, minimum_widths: 'Sequence[int] | None' = None, terminal_width: 'int | None' = None, max_cell_width: 'int | None' = 80, rich: 'bool' = False) -> 'str'` +**Signature:** `render_records(records: 'Iterable[Mapping[str, Any]]', *, requested_format: 'str | None', columns: 'Sequence[tuple[str, str]]', stream: 'TextIO | None' = None, footer: 'str | None' = None, minimum_widths: 'Sequence[int] | None' = None, terminal_width: 'int | None' = None, max_cell_width: 'int | None' = 80, rich: 'bool' = False, formula_guard: 'bool' = True) -> 'str'` **Behavior:** Render records according to the shared public output contract. diff --git a/docs/output-contracts.md b/docs/output-contracts.md index 5fe5fd2..ce1b903 100644 --- a/docs/output-contracts.md +++ b/docs/output-contracts.md @@ -16,6 +16,10 @@ Delimited output is intentionally automation-friendly: - no column header or footer is emitted; - values use the standard `csv` quoting rules, while ANSI escape sequences and other control characters are replaced with spaces. +- cells beginning with `=`, `+`, `-`, or `@` receive a leading apostrophe by + default so spreadsheet programs treat them as text rather than formulas; + pass `formula_guard=False` only when a downstream consumer explicitly needs + the original leading character. `ndjson` is the bounded machine-output format for large or long-running results. It consumes the input iterable once and writes one flushed JSON object diff --git a/lib/python/base_cli/output.py b/lib/python/base_cli/output.py index 4258c41..2e3232e 100644 --- a/lib/python/base_cli/output.py +++ b/lib/python/base_cli/output.py @@ -121,6 +121,7 @@ def render_records( terminal_width: int | None = None, max_cell_width: int | None = _DEFAULT_MAX_CELL_WIDTH, rich: bool = False, + formula_guard: bool = True, ) -> str: """Render records according to the shared public output contract. @@ -144,7 +145,7 @@ def render_records( delimiter = "," if resolved == "csv" else "\t" writer = csv.writer(target, delimiter=delimiter, lineterminator="\n") for row in record_list: - writer.writerow([_delimited_value(row.get(key)) for _header, key in columns]) + writer.writerow([_delimited_value(row.get(key), formula_guard=formula_guard) for _header, key in columns]) return resolved if resolved == "ndjson": @@ -187,6 +188,7 @@ def render_document( records_key: str | None = None, columns: Sequence[tuple[str, str]] | None = None, stream: TextIO | None = None, + formula_guard: bool = True, ) -> str: """Render a structured report or leave terminal text to its existing renderer. @@ -233,6 +235,7 @@ def render_document( requested_format=resolved, columns=selected_columns, stream=target, + formula_guard=formula_guard, ) return resolved @@ -272,7 +275,7 @@ def _validate_delimited_records( dumps_strict_json(value, separators=(",", ":")) -def _delimited_value(value: Any) -> str: +def _delimited_value(value: Any, *, formula_guard: bool = True) -> str: """Return a safe scalar for redirected CSV/TSV output. Delimited output is commonly piped into another process. Keep the normal @@ -281,7 +284,10 @@ def _delimited_value(value: Any) -> str: record across physical lines. """ - return _table_cell(_cell_value(value)) + cell = _table_cell(_cell_value(value)) + if formula_guard and cell[:1] in {"=", "+", "-", "@"}: + return f"'{cell}" + return cell def _write_table( diff --git a/tests/test_output.py b/tests/test_output.py index 15bf6a3..d64595d 100644 --- a/tests/test_output.py +++ b/tests/test_output.py @@ -91,6 +91,31 @@ def test_delimited_emitters_validate_nested_values_before_writing(self) -> None: ) self.assertEqual(stream.getvalue(), "") + def test_delimited_emitters_guard_spreadsheet_formulas_by_default(self) -> None: + records = ({"name": "=SUM(A1:A2)", "path": "+cmd"}, {"name": "-10", "path": "@user"}) + + for requested_format, expected in ( + ("csv", "'=SUM(A1:A2),'+cmd\n'-10,'@user\n"), + ("tsv", "'=SUM(A1:A2)\t'+cmd\n'-10\t'@user\n"), + ): + with self.subTest(format=requested_format): + stream = io.StringIO() + render_records(records, requested_format=requested_format, columns=COLUMNS, stream=stream) + self.assertEqual(stream.getvalue(), expected) + + def test_delimited_formula_guard_can_be_disabled_explicitly(self) -> None: + stream = io.StringIO() + + render_records( + ({"name": "=SUM(A1:A2)", "path": "@user"},), + requested_format="csv", + columns=COLUMNS, + stream=stream, + formula_guard=False, + ) + + self.assertEqual(stream.getvalue(), "=SUM(A1:A2),@user\n") + def test_tsv_consumes_one_pass_iterable_without_materializing(self) -> None: consumed = False From 57f20e90d87f733aa388b7e362e05d09007fb909 Mon Sep 17 00:00:00 2001 From: Ramesh Padmanabhaiah <22363102+codeforester@users.noreply.github.com> Date: Thu, 1 Oct 2026 00:19:50 +0530 Subject: [PATCH 2/3] fix: cover all delimited formula triggers --- CHANGELOG.md | 3 +++ docs/output-contracts.md | 5 +++-- lib/python/base_cli/output.py | 17 ++++++++++++----- tests/test_output.py | 18 ++++++++++++++++++ 4 files changed, 36 insertions(+), 7 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 59d0ef2..2ef6838 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,6 +20,9 @@ and versions are tracked in the repo-root `VERSION` file. - Align the Typer support floor with the tested matrix and cover representative minimum/maximum Typer and Click version pairings. +- Prefix formula-leading CSV/TSV cells with an apostrophe by default to protect + spreadsheet consumers; pass `formula_guard=False` only for an audited raw + value contract. ### Fixed diff --git a/docs/output-contracts.md b/docs/output-contracts.md index ce1b903..fe6423e 100644 --- a/docs/output-contracts.md +++ b/docs/output-contracts.md @@ -16,8 +16,9 @@ Delimited output is intentionally automation-friendly: - no column header or footer is emitted; - values use the standard `csv` quoting rules, while ANSI escape sequences and other control characters are replaced with spaces. -- cells beginning with `=`, `+`, `-`, or `@` receive a leading apostrophe by - default so spreadsheet programs treat them as text rather than formulas; +- cells beginning with `=`, `+`, `-`, `@`, tab, or carriage return receive a + leading apostrophe by default so spreadsheet programs treat them as text + rather than formulas; pass `formula_guard=False` only when a downstream consumer explicitly needs the original leading character. diff --git a/lib/python/base_cli/output.py b/lib/python/base_cli/output.py index 2e3232e..a8ad397 100644 --- a/lib/python/base_cli/output.py +++ b/lib/python/base_cli/output.py @@ -23,6 +23,7 @@ _ANSI_ESCAPE_RE = re.compile(r"\x1b(?:\[[0-?]*[ -/]*[@-~]|\][^\x07]*(?:\x07|\x1b\\))") _DEFAULT_TERMINAL_WIDTH = 120 _DEFAULT_MAX_CELL_WIDTH = 80 +_FORMULA_TRIGGER_CHARS = frozenset("=+-@\t\r") class OutputFormatError(ValueError): @@ -133,7 +134,10 @@ def render_records( columns. Terminal cells use Unicode display-cell widths and are bounded by ``terminal_width`` and ``max_cell_width`` with deterministic ellipsis truncation. ``rich=True`` opts terminal text into the optional Rich - renderer and otherwise falls back to the built-in table. + renderer and otherwise falls back to the built-in table. ``formula_guard`` + controls the default apostrophe prefix for formula-leading CSV/TSV cells; + disabling it is an explicit security decision for consumers that need raw + values. """ target = stream if stream is not None else sys.stdout @@ -194,8 +198,10 @@ def render_document( Structured formats preserve the complete document. Delimited output uses the selected record list (or the document itself) and never emits report - prose, headers, or footers. A terminal ``text`` request returns ``text`` - without writing so the caller can keep its established human report. + prose, headers, or footers. ``formula_guard`` has the same CSV/TSV security + behavior as ``render_records``. A terminal ``text`` request returns + ``text`` without writing so the caller can keep its established human + report. """ target = stream if stream is not None else sys.stdout @@ -284,8 +290,9 @@ def _delimited_value(value: Any, *, formula_guard: bool = True) -> str: record across physical lines. """ - cell = _table_cell(_cell_value(value)) - if formula_guard and cell[:1] in {"=", "+", "-", "@"}: + raw_cell = _cell_value(value) + cell = _table_cell(raw_cell) + if formula_guard and raw_cell[:1] in _FORMULA_TRIGGER_CHARS: return f"'{cell}" return cell diff --git a/tests/test_output.py b/tests/test_output.py index d64595d..0f39f9c 100644 --- a/tests/test_output.py +++ b/tests/test_output.py @@ -12,6 +12,7 @@ NDJSON_SCHEMA_VERSION, NdjsonWriter, OutputFormatError, + _delimited_value, render_document, render_records, resolve_output_format, @@ -116,6 +117,23 @@ def test_delimited_formula_guard_can_be_disabled_explicitly(self) -> None: self.assertEqual(stream.getvalue(), "=SUM(A1:A2),@user\n") + def test_delimited_formula_guard_covers_tab_and_carriage_return_before_sanitizing(self) -> None: + values = ("\t=SUM(A1:A2)", "\r@user") + + with mock.patch("base_cli.output._table_cell", side_effect=lambda value: value): + guarded = [_delimited_value(value) for value in values] + + self.assertEqual(guarded, ["'\t=SUM(A1:A2)", "'\r@user"]) + + stream = io.StringIO() + render_records( + ({"name": values[0], "path": values[1]},), + requested_format="csv", + columns=COLUMNS, + stream=stream, + ) + self.assertEqual(next(csv.reader(io.StringIO(stream.getvalue()))), ["' =SUM(A1:A2)", "' @user"]) + def test_tsv_consumes_one_pass_iterable_without_materializing(self) -> None: consumed = False From def8f89dba63734291e8f88b64bdf0954576ea29 Mon Sep 17 00:00:00 2001 From: Ramesh Padmanabhaiah <22363102+codeforester@users.noreply.github.com> Date: Sat, 3 Oct 2026 00:54:28 +0530 Subject: [PATCH 3/3] ci: separate sustained persistence cost from hosted filesystem tails --- docs/performance.md | 12 +++++++++++- scripts/benchmark_runtime.py | 9 +++++++-- tests/test_benchmark_runtime.py | 12 ++++++++++++ 3 files changed, 30 insertions(+), 3 deletions(-) diff --git a/docs/performance.md b/docs/performance.md index 8195656..96e0797 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -63,7 +63,7 @@ scheduler outlier block a change. | Cold no-op invocation, including startup and dispatch | 2,000 ms | 2,000 ms | 4,000 ms | 4,000 ms | | Base-cli lifecycle increment over Click warm dispatch | 5 ms | 5 ms | 15 ms | 15 ms | | Warm invocation and non-persistence feature scenarios | 50 ms | 50 ms | 100 ms | 100 ms | -| File-persistence-enabled scenario | 50 ms | 50 ms | 250 ms | 50 ms | +| File-persistence-enabled scenario | 125 ms | 125 ms | 250 ms | 50 ms | An initial 31-sample local calibration on macOS (Python 3.14.6, Apple Silicon) measured approximately 101 ms for base-cli cold import, 0.56 ms for warm @@ -76,6 +76,16 @@ budget instead of weakening other warm-scenario gates. These measurements are CI calibration evidence, not adoption claims or release comparisons; review subsequent retained artifacts before tightening platform budgets. +October 2026 hosted recalibration separates sustained persistence cost from +filesystem tails on Unix/macOS: median must remain at most **50 ms** and p95 +at most **125 ms**. The previous 50 ms p95 cap repeatedly rejected otherwise +unchanged runtime code, including the validation-only PR. Observed pairs were +14.66/118.04 ms (Unix median/p95) and 24.93/61.93 and 26.12/87.37 ms (macOS). +Evidence: [Unix run](https://github.com/basefoundry/base-cli/actions/runs/37048785893) +and [macOS validation-only run](https://github.com/basefoundry/base-cli/actions/runs/37052368353). +A sustained slowdown over 50 ms still fails; p95 over 125 ms also fails. +Windows, WSL, parser, import, and non-persistence limits are unchanged. + Each report is versioned as `base-cli.benchmark` schema version 1 and contains the package version, source revision, UTC timestamp, platform profile, Python version/ABI, OS release, architecture, CPU count, sample count, medians, p95, diff --git a/scripts/benchmark_runtime.py b/scripts/benchmark_runtime.py index 24f3e78..f5ade65 100755 --- a/scripts/benchmark_runtime.py +++ b/scripts/benchmark_runtime.py @@ -48,8 +48,8 @@ "wsl": 100.0, } PERSISTENCE_ENABLED_P95_BUDGETS_MS = { - "unix": 50.0, - "macos": 50.0, + "unix": 125.0, + "macos": 125.0, "windows": 250.0, "wsl": 50.0, } @@ -331,6 +331,11 @@ def _check_results(results: dict[str, FrameworkMetrics]) -> list[str]: feature_budget = _feature_budget_for_platform(name, BENCHMARK_PLATFORM) if p95 is not None and p95 > feature_budget: failures.append(f"base-cli {name} p95 exceeded {feature_budget:.0f} ms") + if BENCHMARK_PLATFORM in {"unix", "macos"} and isinstance(features, dict): + persistence = features.get("persistence_enabled_ms", {}) + median = persistence.get("median") if isinstance(persistence, dict) else None + if not isinstance(median, (int, float)) or not 0 <= median <= 50.0: + failures.append("base-cli persistence_enabled_ms median is missing, invalid, or exceeded 50 ms") return failures diff --git a/tests/test_benchmark_runtime.py b/tests/test_benchmark_runtime.py index 8528e5e..7518e06 100644 --- a/tests/test_benchmark_runtime.py +++ b/tests/test_benchmark_runtime.py @@ -125,6 +125,18 @@ def test_windows_persistence_budget_rejects_material_regressions(self) -> None: self.assertTrue(any("persistence_enabled_ms p95 exceeded 250 ms" in failure for failure in failures)) + def test_persistence_budget_separates_sustained_cost_from_filesystem_tails(self) -> None: + for profile in ("unix", "macos"): + for median, p95, fails in ((26.0, 118.0, False), (51.0, 60.0, True), (26.0, 126.0, True)): + with self.subTest(profile=profile, median=median, p95=p95): + metrics = self._complete_results() + sample = self._summary(p95) + sample["median"] = median + metrics["base-cli"]["features"]["persistence_enabled_ms"] = sample + with mock.patch.object(benchmark_runtime, "BENCHMARK_PLATFORM", profile): + failures = benchmark_runtime._check_results(metrics) + self.assertEqual(any("persistence_enabled_ms" in failure for failure in failures), fails) + def test_github_summary_separates_lifecycle_overhead_from_parser(self) -> None: metrics = self._complete_results(lifecycle_p95=4.0, click_p95=1.5) report = {