Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 11 additions & 1 deletion docs/performance.md
Original file line number Diff line number Diff line change
Expand Up @@ -63,7 +63,7 @@ scheduler outlier block a change.
| Cold no-op invocation, including startup and dispatch | 2,000 ms | 2,000 ms | 4,000 ms | 4,000 ms |
| Base-cli lifecycle increment over Click warm dispatch | 5 ms | 5 ms | 15 ms | 15 ms |
| Warm invocation and non-persistence feature scenarios | 50 ms | 50 ms | 100 ms | 100 ms |
| File-persistence-enabled scenario | 50 ms | 50 ms | 250 ms | 50 ms |
| File-persistence-enabled scenario | 125 ms | 125 ms | 250 ms | 50 ms |

An initial 31-sample local calibration on macOS (Python 3.14.6, Apple Silicon)
measured approximately 101 ms for base-cli cold import, 0.56 ms for warm
Expand All @@ -76,6 +76,16 @@ budget instead of weakening other warm-scenario gates. These measurements are
CI calibration evidence, not adoption claims or release comparisons; review
subsequent retained artifacts before tightening platform budgets.

October 2026 hosted recalibration separates sustained persistence cost from
filesystem tails on Unix/macOS: median must remain at most **50 ms** and p95
at most **125 ms**. The previous 50 ms p95 cap repeatedly rejected otherwise
unchanged runtime code, including the validation-only PR. Observed pairs were
14.66/118.04 ms (Unix median/p95) and 24.93/61.93 and 26.12/87.37 ms (macOS).
Evidence: [Unix run](https://github.com/basefoundry/base-cli/actions/runs/37048785893)
and [macOS validation-only run](https://github.com/basefoundry/base-cli/actions/runs/37052368353).
A sustained slowdown over 50 ms still fails; p95 over 125 ms also fails.
Windows, WSL, parser, import, and non-persistence limits are unchanged.

Each report is versioned as `base-cli.benchmark` schema version 1 and contains
the package version, source revision, UTC timestamp, platform profile, Python
version/ABI, OS release, architecture, CPU count, sample count, medians, p95,
Expand Down
78 changes: 65 additions & 13 deletions lib/python/base_cli/output.py
Original file line number Diff line number Diff line change
Expand Up @@ -311,12 +311,14 @@ def _write_table(

table_rows = [[_table_cell(_cell_value(record.get(key))) for _header, key in columns] for record in records]
headers = [_table_cell(header) for header, _key in columns]
header_widths = [_display_width(header) for header in headers]
row_widths = [[_display_width(value) for value in row] for row in table_rows]
widths = [
max(_display_width(header), selected_minimums[index] if index < len(selected_minimums) else 0)
for index, header in enumerate(headers)
max(header_widths[index], selected_minimums[index] if index < len(selected_minimums) else 0)
for index, _header in enumerate(headers)
]
for row in table_rows:
widths = [max(width, _display_width(value)) for width, value in zip(widths, row, strict=False)]
for measured_row in row_widths:
widths = [max(width, measured) for width, measured in zip(widths, measured_row, strict=False)]

if max_cell_width is not None:
if max_cell_width < 1:
Expand All @@ -327,8 +329,17 @@ def _write_table(
raise ValueError("terminal_width must be greater than 0 when set")
available_width = terminal_width if terminal_width is not None else _terminal_width(stream)
widths = _fit_table_width(widths, available_width)
bounded_headers = [_truncate(header, width) for header, width in zip(headers, widths, strict=False)]
bounded_rows = [[_truncate(value, width) for value, width in zip(row, widths, strict=False)] for row in table_rows]
bounded_headers = [
_truncate(header, width, measured_width=header_widths[index])
for index, (header, width) in enumerate(zip(headers, widths, strict=False))
]
bounded_rows = [
[
_truncate(value, width, measured_width=row_widths[row_index][column_index])
for column_index, (value, width) in enumerate(zip(row, widths, strict=False))
]
for row_index, row in enumerate(table_rows)
]

if rich and try_render_rich_table(
stream,
Expand Down Expand Up @@ -360,18 +371,59 @@ def _terminal_width(stream: TextIO) -> int:
return _DEFAULT_TERMINAL_WIDTH


def _fit_table_width(widths: list[int], terminal_width: int) -> list[int]:
def _fit_table_width(
widths: list[int],
terminal_width: int,
*,
_work_counter: list[int] | None = None,
) -> list[int]:
if not widths:
return widths
available = max(1, terminal_width - 2 * (len(widths) - 1))
if sum(widths) <= available:
return widths
result = list(widths)
while sum(result) > available:
index = max(range(len(result)), key=result.__getitem__)
if result[index] <= 1:
remaining = sum(result) - available

def record_work(units: int) -> None:
if _work_counter is not None:
_work_counter[0] += units

# ``order`` is a stable snapshot of the columns sorted by current width.
# ``position`` marks the first column not in the active width level, while
# ``active_count`` tracks how many columns share that level. Mutating only
# the active prefix keeps ties deterministic as widths are reduced.
order = sorted(range(len(result)), key=lambda index: (-result[index], index))
position = 0
level = result[order[0]]
active_count = 0
while position < len(order) and result[order[position]] == level:
active_count += 1
position += 1
while remaining > 0 and active_count:
Comment thread
codeforester marked this conversation as resolved.
# When every remaining column is active, floor the next level at one;
# there is no legal width below one even if the arithmetic target is 0.
next_level = result[order[position]] if position < len(order) else 1
next_level = max(1, next_level)
capacity = (level - next_level) * active_count
if capacity <= 0:
break
if remaining <= capacity:
quotient, remainder = divmod(remaining, active_count)
for rank in range(active_count):
index = order[rank]
result[index] -= quotient + (rank < remainder)
record_work(active_count)
remaining = 0
break
result[index] -= 1
for rank in range(active_count):
result[order[rank]] = next_level
record_work(active_count)
remaining -= capacity
level = next_level
while position < len(order) and result[order[position]] == level:
active_count += 1
position += 1
return result


Expand All @@ -394,8 +446,8 @@ def _display_width(value: str) -> int:
return width


def _truncate(value: str, width: int) -> str:
if _display_width(value) <= width:
def _truncate(value: str, width: int, *, measured_width: int | None = None) -> str:
if (measured_width if measured_width is not None else _display_width(value)) <= width:
return value
if width <= 1:
return "…"[:width]
Expand Down
9 changes: 7 additions & 2 deletions scripts/benchmark_runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -48,8 +48,8 @@
"wsl": 100.0,
}
PERSISTENCE_ENABLED_P95_BUDGETS_MS = {
"unix": 50.0,
"macos": 50.0,
"unix": 125.0,
"macos": 125.0,
"windows": 250.0,
"wsl": 50.0,
}
Expand Down Expand Up @@ -331,6 +331,11 @@ def _check_results(results: dict[str, FrameworkMetrics]) -> list[str]:
feature_budget = _feature_budget_for_platform(name, BENCHMARK_PLATFORM)
if p95 is not None and p95 > feature_budget:
failures.append(f"base-cli {name} p95 exceeded {feature_budget:.0f} ms")
if BENCHMARK_PLATFORM in {"unix", "macos"} and isinstance(features, dict):
persistence = features.get("persistence_enabled_ms", {})
median = persistence.get("median") if isinstance(persistence, dict) else None
if not isinstance(median, (int, float)) or not 0 <= median <= 50.0:
failures.append("base-cli persistence_enabled_ms median is missing, invalid, or exceeded 50 ms")
return failures


Expand Down
12 changes: 12 additions & 0 deletions tests/test_benchmark_runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -125,6 +125,18 @@ def test_windows_persistence_budget_rejects_material_regressions(self) -> None:

self.assertTrue(any("persistence_enabled_ms p95 exceeded 250 ms" in failure for failure in failures))

def test_persistence_budget_separates_sustained_cost_from_filesystem_tails(self) -> None:
for profile in ("unix", "macos"):
for median, p95, fails in ((26.0, 118.0, False), (51.0, 60.0, True), (26.0, 126.0, True)):
with self.subTest(profile=profile, median=median, p95=p95):
metrics = self._complete_results()
sample = self._summary(p95)
sample["median"] = median
metrics["base-cli"]["features"]["persistence_enabled_ms"] = sample
with mock.patch.object(benchmark_runtime, "BENCHMARK_PLATFORM", profile):
failures = benchmark_runtime._check_results(metrics)
self.assertEqual(any("persistence_enabled_ms" in failure for failure in failures), fails)

def test_github_summary_separates_lifecycle_overhead_from_parser(self) -> None:
metrics = self._complete_results(lifecycle_p95=4.0, click_p95=1.5)
report = {
Expand Down
12 changes: 12 additions & 0 deletions tests/test_output.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@
NDJSON_SCHEMA_VERSION,
NdjsonWriter,
OutputFormatError,
_fit_table_width,
render_document,
render_records,
resolve_output_format,
Expand Down Expand Up @@ -140,6 +141,17 @@ def test_terminal_table_uses_display_width_and_deterministic_truncation(self) ->
self.assertIn("…", output)
self.assertLessEqual(max(len(line) for line in output.splitlines()), 20)

def test_table_width_fitting_handles_large_widths_without_per_unit_loop(self) -> None:
Comment thread
codeforester marked this conversation as resolved.
widths = [10_000_000, 9_000_000, 8_000_000, 7_000_000]
work = [0]

fitted = _fit_table_width(widths, terminal_width=120, _work_counter=work)

self.assertEqual(sum(fitted), 114)
self.assertGreaterEqual(min(fitted), 1)
self.assertEqual(fitted, [28, 28, 29, 29])
self.assertLessEqual(work[0], len(widths) * 3)

def test_terminal_width_and_cell_width_validate_inputs(self) -> None:
with self.assertRaisesRegex(ValueError, "terminal_width"):
render_records(
Expand Down
Loading