diff --git a/docs/performance.md b/docs/performance.md index 8195656..96e0797 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -63,7 +63,7 @@ scheduler outlier block a change. | Cold no-op invocation, including startup and dispatch | 2,000 ms | 2,000 ms | 4,000 ms | 4,000 ms | | Base-cli lifecycle increment over Click warm dispatch | 5 ms | 5 ms | 15 ms | 15 ms | | Warm invocation and non-persistence feature scenarios | 50 ms | 50 ms | 100 ms | 100 ms | -| File-persistence-enabled scenario | 50 ms | 50 ms | 250 ms | 50 ms | +| File-persistence-enabled scenario | 125 ms | 125 ms | 250 ms | 50 ms | An initial 31-sample local calibration on macOS (Python 3.14.6, Apple Silicon) measured approximately 101 ms for base-cli cold import, 0.56 ms for warm @@ -76,6 +76,16 @@ budget instead of weakening other warm-scenario gates. These measurements are CI calibration evidence, not adoption claims or release comparisons; review subsequent retained artifacts before tightening platform budgets. +October 2026 hosted recalibration separates sustained persistence cost from +filesystem tails on Unix/macOS: median must remain at most **50 ms** and p95 +at most **125 ms**. The previous 50 ms p95 cap repeatedly rejected otherwise +unchanged runtime code, including the validation-only PR. Observed pairs were +14.66/118.04 ms (Unix median/p95) and 24.93/61.93 and 26.12/87.37 ms (macOS). +Evidence: [Unix run](https://github.com/basefoundry/base-cli/actions/runs/37048785893) +and [macOS validation-only run](https://github.com/basefoundry/base-cli/actions/runs/37052368353). +A sustained slowdown over 50 ms still fails; p95 over 125 ms also fails. +Windows, WSL, parser, import, and non-persistence limits are unchanged. + Each report is versioned as `base-cli.benchmark` schema version 1 and contains the package version, source revision, UTC timestamp, platform profile, Python version/ABI, OS release, architecture, CPU count, sample count, medians, p95, diff --git a/lib/python/base_cli/output.py b/lib/python/base_cli/output.py index 4258c41..287382d 100644 --- a/lib/python/base_cli/output.py +++ b/lib/python/base_cli/output.py @@ -311,12 +311,14 @@ def _write_table( table_rows = [[_table_cell(_cell_value(record.get(key))) for _header, key in columns] for record in records] headers = [_table_cell(header) for header, _key in columns] + header_widths = [_display_width(header) for header in headers] + row_widths = [[_display_width(value) for value in row] for row in table_rows] widths = [ - max(_display_width(header), selected_minimums[index] if index < len(selected_minimums) else 0) - for index, header in enumerate(headers) + max(header_widths[index], selected_minimums[index] if index < len(selected_minimums) else 0) + for index, _header in enumerate(headers) ] - for row in table_rows: - widths = [max(width, _display_width(value)) for width, value in zip(widths, row, strict=False)] + for measured_row in row_widths: + widths = [max(width, measured) for width, measured in zip(widths, measured_row, strict=False)] if max_cell_width is not None: if max_cell_width < 1: @@ -327,8 +329,17 @@ def _write_table( raise ValueError("terminal_width must be greater than 0 when set") available_width = terminal_width if terminal_width is not None else _terminal_width(stream) widths = _fit_table_width(widths, available_width) - bounded_headers = [_truncate(header, width) for header, width in zip(headers, widths, strict=False)] - bounded_rows = [[_truncate(value, width) for value, width in zip(row, widths, strict=False)] for row in table_rows] + bounded_headers = [ + _truncate(header, width, measured_width=header_widths[index]) + for index, (header, width) in enumerate(zip(headers, widths, strict=False)) + ] + bounded_rows = [ + [ + _truncate(value, width, measured_width=row_widths[row_index][column_index]) + for column_index, (value, width) in enumerate(zip(row, widths, strict=False)) + ] + for row_index, row in enumerate(table_rows) + ] if rich and try_render_rich_table( stream, @@ -360,18 +371,59 @@ def _terminal_width(stream: TextIO) -> int: return _DEFAULT_TERMINAL_WIDTH -def _fit_table_width(widths: list[int], terminal_width: int) -> list[int]: +def _fit_table_width( + widths: list[int], + terminal_width: int, + *, + _work_counter: list[int] | None = None, +) -> list[int]: if not widths: return widths available = max(1, terminal_width - 2 * (len(widths) - 1)) if sum(widths) <= available: return widths result = list(widths) - while sum(result) > available: - index = max(range(len(result)), key=result.__getitem__) - if result[index] <= 1: + remaining = sum(result) - available + + def record_work(units: int) -> None: + if _work_counter is not None: + _work_counter[0] += units + + # ``order`` is a stable snapshot of the columns sorted by current width. + # ``position`` marks the first column not in the active width level, while + # ``active_count`` tracks how many columns share that level. Mutating only + # the active prefix keeps ties deterministic as widths are reduced. + order = sorted(range(len(result)), key=lambda index: (-result[index], index)) + position = 0 + level = result[order[0]] + active_count = 0 + while position < len(order) and result[order[position]] == level: + active_count += 1 + position += 1 + while remaining > 0 and active_count: + # When every remaining column is active, floor the next level at one; + # there is no legal width below one even if the arithmetic target is 0. + next_level = result[order[position]] if position < len(order) else 1 + next_level = max(1, next_level) + capacity = (level - next_level) * active_count + if capacity <= 0: + break + if remaining <= capacity: + quotient, remainder = divmod(remaining, active_count) + for rank in range(active_count): + index = order[rank] + result[index] -= quotient + (rank < remainder) + record_work(active_count) + remaining = 0 break - result[index] -= 1 + for rank in range(active_count): + result[order[rank]] = next_level + record_work(active_count) + remaining -= capacity + level = next_level + while position < len(order) and result[order[position]] == level: + active_count += 1 + position += 1 return result @@ -394,8 +446,8 @@ def _display_width(value: str) -> int: return width -def _truncate(value: str, width: int) -> str: - if _display_width(value) <= width: +def _truncate(value: str, width: int, *, measured_width: int | None = None) -> str: + if (measured_width if measured_width is not None else _display_width(value)) <= width: return value if width <= 1: return "…"[:width] diff --git a/scripts/benchmark_runtime.py b/scripts/benchmark_runtime.py index 24f3e78..f5ade65 100755 --- a/scripts/benchmark_runtime.py +++ b/scripts/benchmark_runtime.py @@ -48,8 +48,8 @@ "wsl": 100.0, } PERSISTENCE_ENABLED_P95_BUDGETS_MS = { - "unix": 50.0, - "macos": 50.0, + "unix": 125.0, + "macos": 125.0, "windows": 250.0, "wsl": 50.0, } @@ -331,6 +331,11 @@ def _check_results(results: dict[str, FrameworkMetrics]) -> list[str]: feature_budget = _feature_budget_for_platform(name, BENCHMARK_PLATFORM) if p95 is not None and p95 > feature_budget: failures.append(f"base-cli {name} p95 exceeded {feature_budget:.0f} ms") + if BENCHMARK_PLATFORM in {"unix", "macos"} and isinstance(features, dict): + persistence = features.get("persistence_enabled_ms", {}) + median = persistence.get("median") if isinstance(persistence, dict) else None + if not isinstance(median, (int, float)) or not 0 <= median <= 50.0: + failures.append("base-cli persistence_enabled_ms median is missing, invalid, or exceeded 50 ms") return failures diff --git a/tests/test_benchmark_runtime.py b/tests/test_benchmark_runtime.py index 8528e5e..7518e06 100644 --- a/tests/test_benchmark_runtime.py +++ b/tests/test_benchmark_runtime.py @@ -125,6 +125,18 @@ def test_windows_persistence_budget_rejects_material_regressions(self) -> None: self.assertTrue(any("persistence_enabled_ms p95 exceeded 250 ms" in failure for failure in failures)) + def test_persistence_budget_separates_sustained_cost_from_filesystem_tails(self) -> None: + for profile in ("unix", "macos"): + for median, p95, fails in ((26.0, 118.0, False), (51.0, 60.0, True), (26.0, 126.0, True)): + with self.subTest(profile=profile, median=median, p95=p95): + metrics = self._complete_results() + sample = self._summary(p95) + sample["median"] = median + metrics["base-cli"]["features"]["persistence_enabled_ms"] = sample + with mock.patch.object(benchmark_runtime, "BENCHMARK_PLATFORM", profile): + failures = benchmark_runtime._check_results(metrics) + self.assertEqual(any("persistence_enabled_ms" in failure for failure in failures), fails) + def test_github_summary_separates_lifecycle_overhead_from_parser(self) -> None: metrics = self._complete_results(lifecycle_p95=4.0, click_p95=1.5) report = { diff --git a/tests/test_output.py b/tests/test_output.py index 15bf6a3..a20ae53 100644 --- a/tests/test_output.py +++ b/tests/test_output.py @@ -12,6 +12,7 @@ NDJSON_SCHEMA_VERSION, NdjsonWriter, OutputFormatError, + _fit_table_width, render_document, render_records, resolve_output_format, @@ -140,6 +141,17 @@ def test_terminal_table_uses_display_width_and_deterministic_truncation(self) -> self.assertIn("…", output) self.assertLessEqual(max(len(line) for line in output.splitlines()), 20) + def test_table_width_fitting_handles_large_widths_without_per_unit_loop(self) -> None: + widths = [10_000_000, 9_000_000, 8_000_000, 7_000_000] + work = [0] + + fitted = _fit_table_width(widths, terminal_width=120, _work_counter=work) + + self.assertEqual(sum(fitted), 114) + self.assertGreaterEqual(min(fitted), 1) + self.assertEqual(fitted, [28, 28, 29, 29]) + self.assertLessEqual(work[0], len(widths) * 3) + def test_terminal_width_and_cell_width_validate_inputs(self) -> None: with self.assertRaisesRegex(ValueError, "terminal_width"): render_records(