From 1dd55ae7b6b8bdf4688b0b359dc61a489223dc71 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Tue, 14 Jul 2026 12:53:52 -0400 Subject: [PATCH 01/19] style: black-format dedup changes to fix CI MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The alert-dedup merge (#6) landed with two files that failed the `black --check --line-length 110` backend gate. Formatting only — no logic change; 20 core tests still pass. Co-Authored-By: Claude Opus 4.8 --- core/orchestrator.py | 8 +++++--- core/tests/test_orchestrator.py | 3 +-- 2 files changed, 6 insertions(+), 5 deletions(-) diff --git a/core/orchestrator.py b/core/orchestrator.py index e0429cf..5fdec1b 100644 --- a/core/orchestrator.py +++ b/core/orchestrator.py @@ -12,8 +12,8 @@ # Move to Postgres/Redis if you run multiple workers or need durability. _DEDUP_WINDOW_S = float(os.getenv("SENTINEL_DEDUP_WINDOW_S", "300")) _DEDUP_THRESHOLD = int(os.getenv("SENTINEL_DEDUP_THRESHOLD", "1")) # Nth hit in window fires -_recent_alerts: dict[str, list[float]] = defaultdict(list) # signature -> hit times -_open_signatures: dict[str, str] = {} # signature -> live incident_id +_recent_alerts: dict[str, list[float]] = defaultdict(list) # signature -> hit times +_open_signatures: dict[str, str] = {} # signature -> live incident_id def _dedup_gate(signature: str) -> tuple[str, str | None]: @@ -132,7 +132,9 @@ def resolve_incident(incident_id: str) -> dict: ) print(f"[orchestrator] {incident_id} -- postmortem LLM unavailable, degrading: {e}") db.update_postmortem(incident_id, draft) - _clear_signature((incident.get("trigger_data") or {}).get("error_signature", "")) # resolved: reopen to new alerts + _clear_signature( + (incident.get("trigger_data") or {}).get("error_signature", "") + ) # resolved: reopen to new alerts print(f"[orchestrator] {incident_id} -> resolved, postmortem generated") print(f"[metrics] {incident_id} resolve_to_postmortem_s={time.monotonic() - t0:.1f}") return db.get_incident(incident_id) diff --git a/core/tests/test_orchestrator.py b/core/tests/test_orchestrator.py index 646fb07..2f16d97 100644 --- a/core/tests/test_orchestrator.py +++ b/core/tests/test_orchestrator.py @@ -49,8 +49,7 @@ def test_dedup_folds_repeat_alerts(): def test_dedup_threshold_suppresses_below_burst(): """With a threshold >1, alerts below the burst count are suppressed, not opened.""" _reset_dedup() - with patch.object(orch, "_DEDUP_THRESHOLD", 3), \ - patch("core.orchestrator.db.save_incident") as mock_save: + with patch.object(orch, "_DEDUP_THRESHOLD", 3), patch("core.orchestrator.db.save_incident") as mock_save: r1 = handle_alert(_ALERT) r2 = handle_alert(_ALERT) r3 = handle_alert(_ALERT) From 24c34e0cd18fe86c4329d4a1c12bcd9314633d01 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 13:43:02 -0400 Subject: [PATCH 02/19] feat(core): make Chroma path configurable via SENTINEL_CHROMA_PATH Co-Authored-By: Claude Opus 5.5 --- core/services/vector_store.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/core/services/vector_store.py b/core/services/vector_store.py index b74e819..41b93e0 100644 --- a/core/services/vector_store.py +++ b/core/services/vector_store.py @@ -1,7 +1,8 @@ +import os from pathlib import Path import chromadb -CHROMA_PATH = ".chroma" +CHROMA_PATH = os.getenv("SENTINEL_CHROMA_PATH", ".chroma") COLLECTION = "runbooks" From 1eed3bf279c7aa98039862f6ead25798e40cbba3 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 13:44:57 -0400 Subject: [PATCH 03/19] feat(runbooks): expand corpus to 10 generic runbooks, freeze as v1 Co-Authored-By: Claude Opus 5.5 --- eval/__init__.py | 0 eval/config.py | 2 ++ sandbox/runbooks/auth_failures.md | 29 +++++++++++++++++++++ sandbox/runbooks/config_parse_errors.md | 29 +++++++++++++++++++++ sandbox/runbooks/dependency_regressions.md | 29 +++++++++++++++++++++ sandbox/runbooks/file_handle_exhaustion.md | 29 +++++++++++++++++++++ sandbox/runbooks/null_field_handling.md | 29 +++++++++++++++++++++ sandbox/runbooks/pagination_errors.md | 29 +++++++++++++++++++++ sandbox/runbooks/rate_limiting.md | 29 +++++++++++++++++++++ sandbox/runbooks/request_timeouts.md | 30 ++++++++++++++++++++++ 10 files changed, 235 insertions(+) create mode 100644 eval/__init__.py create mode 100644 eval/config.py create mode 100644 sandbox/runbooks/auth_failures.md create mode 100644 sandbox/runbooks/config_parse_errors.md create mode 100644 sandbox/runbooks/dependency_regressions.md create mode 100644 sandbox/runbooks/file_handle_exhaustion.md create mode 100644 sandbox/runbooks/null_field_handling.md create mode 100644 sandbox/runbooks/pagination_errors.md create mode 100644 sandbox/runbooks/rate_limiting.md create mode 100644 sandbox/runbooks/request_timeouts.md diff --git a/eval/__init__.py b/eval/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/eval/config.py b/eval/config.py new file mode 100644 index 0000000..0f327a0 --- /dev/null +++ b/eval/config.py @@ -0,0 +1,2 @@ +# Frozen eval inputs. Bump a version whenever its inputs change; never edit after a compared run. +RUNBOOK_CORPUS_VERSION = "v1" # sandbox/runbooks/*.md, 10 docs, frozen at Phase 9B diff --git a/sandbox/runbooks/auth_failures.md b/sandbox/runbooks/auth_failures.md new file mode 100644 index 0000000..598d240 --- /dev/null +++ b/sandbox/runbooks/auth_failures.md @@ -0,0 +1,29 @@ +# Runbook: Authentication and Credential Failures + +**ID:** rb_auth_01 +**Severity:** Critical +**Tags:** auth, 401, 403, credentials, tokens + +## Symptoms + +- Spike in HTTP 401 Unauthorized or 403 Forbidden responses +- Log lines containing `InvalidSignatureError`, `ExpiredSignatureError`, `invalid_grant`, `authentication failed`, or `permission denied` +- Calls to a third-party API rejected while the provider's status page is green + +## Root Causes + +1. **Expired or rotated credential**: an API key, certificate, or secret rotated without updating consumers +2. **Clock skew**: token validation failing on `iat`/`exp` checks because of drift between hosts +3. **Wrong key or audience**: signing key, issuer, or audience changed in config or code +4. **Scope or permission change**: a role lost a permission the service relies on + +## Immediate Actions + +1. Determine whether rejections are inbound (our users rejected) or outbound (we are rejected by a dependency) +2. Check credential expiry dates and recent rotations +3. Compare token validation settings (algorithm, issuer, audience, leeway) against the last good deploy +4. Verify host clocks are synchronized + +## Escalation + +Treat any unexplained auth change as a potential security event and notify the security on-call. diff --git a/sandbox/runbooks/config_parse_errors.md b/sandbox/runbooks/config_parse_errors.md new file mode 100644 index 0000000..1941689 --- /dev/null +++ b/sandbox/runbooks/config_parse_errors.md @@ -0,0 +1,29 @@ +# Runbook: Configuration Type and Parse Errors + +**ID:** rb_config_01 +**Severity:** Critical +**Tags:** config, env, parsing, startup, types + +## Symptoms + +- Service fails on startup, or fails on the first request that reads a setting +- Log lines containing `ValueError: invalid literal for int()`, `TypeError: '<' not supported between instances of 'str' and 'int'`, `JSONDecodeError`, or `could not convert string to float` +- Errors appear immediately after a deploy or config change, across all instances at once + +## Root Causes + +1. **Wrong type**: a numeric or boolean setting provided as a string (environment variables are always strings) +2. **Malformed config file**: invalid JSON/YAML, trailing commas, wrong nesting +3. **Renamed or removed key**: code reads a key the config no longer contains, falling back to an unexpected default +4. **Unit mismatch**: seconds vs. milliseconds, bytes vs. megabytes + +## Immediate Actions + +1. Diff the effective configuration against the last known good deploy +2. Validate config files with a parser before redeploying +3. Check that every setting read from the environment is explicitly cast to its expected type +4. Roll back the config change if the correct value is not obvious + +## Escalation + +If secrets or credentials are part of the changed configuration, involve the platform team. diff --git a/sandbox/runbooks/dependency_regressions.md b/sandbox/runbooks/dependency_regressions.md new file mode 100644 index 0000000..c9fd41d --- /dev/null +++ b/sandbox/runbooks/dependency_regressions.md @@ -0,0 +1,29 @@ +# Runbook: Dependency Version Regressions + +**ID:** rb_deps_01 +**Severity:** High +**Tags:** dependencies, upgrade, importerror, compatibility + +## Symptoms + +- Errors immediately after a deploy that only changed lockfiles or requirements +- Log lines containing `ImportError`, `ModuleNotFoundError`, `AttributeError: module ... has no attribute`, or `unexpected keyword argument` +- Behaviour changes (serialization, defaults, timezone handling) without application code changes + +## Root Causes + +1. **Breaking change in a major or minor upgrade**: removed or renamed API +2. **Unpinned transitive dependency**: an indirect package moved to a new version +3. **Changed default behaviour**: a library changed a default value between versions +4. **Environment drift**: build image or runtime version differs from what was tested + +## Immediate Actions + +1. Diff the resolved dependency set against the last good deploy +2. Read the changelog of every package that changed version +3. Pin the previous version and redeploy if the cause is not immediately clear +4. Add the affected call to a test that runs against the pinned versions + +## Escalation + +If a security patch forced the upgrade, coordinate with the security team before rolling back. diff --git a/sandbox/runbooks/file_handle_exhaustion.md b/sandbox/runbooks/file_handle_exhaustion.md new file mode 100644 index 0000000..cf9e1ab --- /dev/null +++ b/sandbox/runbooks/file_handle_exhaustion.md @@ -0,0 +1,29 @@ +# Runbook: Disk Space and File Handle Exhaustion + +**ID:** rb_disk_01 +**Severity:** High +**Tags:** disk, file-descriptors, enospc, emfile + +## Symptoms + +- Log lines containing `OSError: [Errno 24] Too many open files`, `[Errno 28] No space left on device`, `ENOSPC`, or `EMFILE` +- Writes, uploads, or log output failing while reads still work +- Error rate rising steadily with uptime and clearing after a restart + +## Root Causes + +1. **Unclosed handles**: files, sockets, or subprocess pipes opened without a context manager or close +2. **Unbounded temp or log files**: files written per request and never cleaned up +3. **Missing rotation**: log rotation disabled or misconfigured +4. **Low ulimit**: file descriptor limit lowered in the runtime environment + +## Immediate Actions + +1. Check free disk space (`df -h`) and open descriptors (`ls /proc//fd | wc -l`) +2. Identify what is holding handles or space (`lsof -p `, `du -sh` on temp and log directories) +3. Review recent changes that open files, sockets, or temp files +4. Restart the process to release handles, then clean up temp space + +## Escalation + +If the volume is shared with a database, escalate immediately to avoid data corruption. diff --git a/sandbox/runbooks/null_field_handling.md b/sandbox/runbooks/null_field_handling.md new file mode 100644 index 0000000..cf921e5 --- /dev/null +++ b/sandbox/runbooks/null_field_handling.md @@ -0,0 +1,29 @@ +# Runbook: Null or Missing Field Errors + +**ID:** rb_null_01 +**Severity:** High +**Tags:** null, none, keyerror, validation, 500 + +## Symptoms + +- HTTP 500 errors on a subset of requests, often tied to particular records or users +- Log lines containing `'NoneType' object has no attribute`, `KeyError`, `TypeError: 'NoneType' object is not subscriptable`, or `Cannot read properties of undefined` +- Error rate correlates with specific input shapes rather than with overall traffic + +## Root Causes + +1. **Optional field treated as required**: code accesses a field that some records or payloads omit +2. **Schema change without backfill**: a new column or key exists for new rows but is null for older ones +3. **Changed default**: a lookup or helper that used to return an empty value now returns `None` +4. **Upstream contract change**: a dependency stopped sending a field + +## Immediate Actions + +1. Capture a failing payload or record ID from the traceback and inspect which field is missing +2. Check recent changes to data models, serializers, and helper return values +3. Add a guarded default at the access site if a rollback is not possible +4. Query for how many records lack the field to size the impact + +## Escalation + +If corrupted or partially written records are involved, involve the data owner before backfilling. diff --git a/sandbox/runbooks/pagination_errors.md b/sandbox/runbooks/pagination_errors.md new file mode 100644 index 0000000..d56332b --- /dev/null +++ b/sandbox/runbooks/pagination_errors.md @@ -0,0 +1,29 @@ +# Runbook: Pagination and Off-by-One Errors + +**ID:** rb_paging_01 +**Severity:** Medium +**Tags:** pagination, off-by-one, indexerror, data-correctness + +## Symptoms + +- `IndexError: list index out of range` or similar boundary errors on list endpoints +- Clients report duplicated or missing items between pages +- Errors concentrated on the last page, the first page, or empty result sets + +## Root Causes + +1. **Boundary arithmetic**: offset computed from 1-based page numbers as if 0-based, or vice versa +2. **Inclusive vs. exclusive ranges**: slice or range end treated inconsistently +3. **Unstable sort order**: paging over a query without a deterministic ORDER BY +4. **Empty-set handling**: code assumes at least one result + +## Immediate Actions + +1. Reproduce with a page size of 1 and with an empty result set +2. Review recent changes to offset, limit, cursor, and slicing logic +3. Confirm the query has a deterministic ordering key +4. Compare item counts across pages against a single unpaged query + +## Escalation + +If clients have already consumed incorrect data (e.g. exports, billing), notify the owning product team. diff --git a/sandbox/runbooks/rate_limiting.md b/sandbox/runbooks/rate_limiting.md new file mode 100644 index 0000000..0da8a9a --- /dev/null +++ b/sandbox/runbooks/rate_limiting.md @@ -0,0 +1,29 @@ +# Runbook: Rate Limiting and HTTP 429 + +**ID:** rb_ratelimit_01 +**Severity:** Medium +**Tags:** 429, rate-limit, throttling, quota + +## Symptoms + +- HTTP 429 Too Many Requests, either returned to our clients or received from a dependency +- Log lines containing `429`, `Too Many Requests`, `rate limit exceeded`, `quota exceeded`, or `Retry-After` +- Rejections arrive in bursts and recover on a fixed interval + +## Root Causes + +1. **Request volume increase**: a change that calls a dependency more often, e.g. per item inside a loop instead of per batch +2. **Lost caching**: a cache bypassed, disabled, or given a much shorter TTL +3. **Aggressive retries**: retrying 429s immediately instead of honouring `Retry-After` +4. **Lowered limit**: a quota or limiter threshold changed in config + +## Immediate Actions + +1. Determine whether we are being limited or doing the limiting +2. Compare outbound request rate per dependency before and after the start of the incident +3. Check recent changes to caching, batching, retry policies, and limiter settings +4. Back off or reduce concurrency on the affected client + +## Escalation + +If a vendor quota must be raised, contact the account owner for that vendor. diff --git a/sandbox/runbooks/request_timeouts.md b/sandbox/runbooks/request_timeouts.md new file mode 100644 index 0000000..ae67253 --- /dev/null +++ b/sandbox/runbooks/request_timeouts.md @@ -0,0 +1,30 @@ +# Runbook: Request Timeouts and Upstream Latency + +**ID:** rb_timeout_01 +**Severity:** High +**Tags:** latency, timeout, 504, upstream + +## Symptoms + +- HTTP 504 Gateway Timeout or 503 responses from the edge or load balancer +- Log lines containing `ReadTimeout`, `TimeoutError`, `timed out`, or `deadline exceeded` +- p95/p99 latency climbing on one endpoint while others stay normal +- Worker or thread pool saturation as requests wait on slow calls + +## Root Causes + +1. **Timeout budget mismatch**: a client-side timeout shorter than the upstream's normal response time, or an inner timeout longer than the outer one +2. **Slow dependency**: a downstream service or query whose latency has regressed +3. **Retry amplification**: retries without backoff multiplying load on an already slow upstream +4. **Blocking call on a hot path**: synchronous I/O added to a request handler + +## Immediate Actions + +1. Identify which hop is slow: compare latency at the edge, the app, and each dependency +2. Review timeout and retry settings across the call chain; inner timeouts must be shorter than outer ones +3. Check recent changes to timeout constants, retry policies, and client configuration +4. Shed load or raise the outer timeout temporarily if a dependency is healthy but slow + +## Escalation + +If a shared dependency is slow for more than 15 minutes, page the owning team. From 8d593b0764fd5981895a6b06a0dca5000e862f25 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 14:04:21 -0400 Subject: [PATCH 04/19] feat(eval): offline-buildable eval harness, scorer, and leakage tests Co-Authored-By: Claude Opus 5.5 --- .github/workflows/ci.yml | 6 +- eval/config.py | 12 +++ eval/fixture.py | 107 +++++++++++++++++++++++++ eval/report.py | 167 +++++++++++++++++++++++++++++++++++++++ eval/run_eval.py | 142 +++++++++++++++++++++++++++++++++ eval/tests/__init__.py | 0 eval/tests/test_eval.py | 146 ++++++++++++++++++++++++++++++++++ 7 files changed, 577 insertions(+), 3 deletions(-) create mode 100644 eval/fixture.py create mode 100644 eval/report.py create mode 100644 eval/run_eval.py create mode 100644 eval/tests/__init__.py create mode 100644 eval/tests/test_eval.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index eca9d8b..411c4b1 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -18,9 +18,9 @@ jobs: python-version: "3.11" cache: pip - run: pip install -e core/ pytest jsonschema black flake8 - - run: black --check --line-length 110 core sandbox - - run: flake8 core sandbox - - run: pytest core/tests -q + - run: black --check --line-length 110 core sandbox eval + - run: flake8 core sandbox eval + - run: pytest core/tests eval/tests -q dashboard: runs-on: ubuntu-latest diff --git a/eval/config.py b/eval/config.py index 0f327a0..a614063 100644 --- a/eval/config.py +++ b/eval/config.py @@ -1,2 +1,14 @@ # Frozen eval inputs. Bump a version whenever its inputs change; never edit after a compared run. RUNBOOK_CORPUS_VERSION = "v1" # sandbox/runbooks/*.md, 10 docs, frozen at Phase 9B +EVAL_SET_VERSION = "v1" # eval/cases/*.json + eval/bases/, 29 cases, frozen at Phase 9C + +ALERT_TS = "2026-09-01T12:00:00+00:00" # fixed so fixture commit dates are reproducible +DEFAULT_TRIALS = 3 + +# Candidate runbook similarity floors; report.py prints the false-match rate at each. +# The chosen floor is frozen in 9E from 9D's retrieval scores, before any RAG run. +FLOOR_CANDIDATES = [0.15, 0.2, 0.25, 0.3, 0.35, 0.4] + +# Words that would announce the culprit. Checked (word-prefix, case-insensitive) in commit +# messages and in every file the case writes, including the shared base. +BANNED_WORDS = ["bug", "inject", "break", "fault", "fail", "chaos", "oops", "fix", "hotfix", "revert"] diff --git a/eval/fixture.py b/eval/fixture.py new file mode 100644 index 0000000..3a24db5 --- /dev/null +++ b/eval/fixture.py @@ -0,0 +1,107 @@ +"""Builds a throwaway git repo for one eval case. Never touches the Sentinel repo.""" + +import json +import os +import shutil +import stat +import subprocess +import tempfile +from contextlib import contextmanager +from datetime import datetime, timedelta +from pathlib import Path + +from eval.config import ALERT_TS + +BASES_DIR = Path(__file__).parent / "bases" +CASES_DIR = Path(__file__).parent / "cases" + + +def load_case(path) -> dict: + return json.loads(Path(path).read_text(encoding="utf-8")) + + +def all_case_paths(pattern: str = "*.json") -> list[Path]: + return sorted(CASES_DIR.glob(pattern)) + + +def _git(repo: Path, *args, when: datetime | None = None, author: str = "dev@example.com") -> str: + env = {**os.environ, "GIT_AUTHOR_EMAIL": author, "GIT_COMMITTER_EMAIL": author} + env["GIT_AUTHOR_NAME"] = env["GIT_COMMITTER_NAME"] = author.split("@")[0] + if when: + # git log --since/--until filters on committer date, so both must be set + env["GIT_AUTHOR_DATE"] = env["GIT_COMMITTER_DATE"] = when.isoformat() + out = subprocess.run( + ["git", "-c", "core.autocrlf=false", "-c", "commit.gpgsign=false", *args], + cwd=repo, + env=env, + capture_output=True, + text=True, + encoding="utf-8", + check=True, + ) + return out.stdout.strip() + + +def _apply(repo: Path, files: dict): + """Each value is either full file content (str) or a list of [old, new] edits.""" + for rel, change in files.items(): + path = repo / rel + path.parent.mkdir(parents=True, exist_ok=True) + if isinstance(change, str): + path.write_text(change, encoding="utf-8", newline="\n") + continue + text = path.read_text(encoding="utf-8") + for old, new in change: + if text.count(old) != 1: + raise ValueError( + f"{rel}: edit target must occur exactly once, found {text.count(old)}: {old!r}" + ) + text = text.replace(old, new) + path.write_text(text, encoding="utf-8", newline="\n") + + +def _rmtree(path: str): + # git marks object files read-only; Windows refuses to delete those without a chmod + def _chmod_retry(func, p, _exc): + os.chmod(p, stat.S_IWRITE) + func(p) + + shutil.rmtree(path, onerror=_chmod_retry) + + +@contextmanager +def build(case: dict): + """Yields (repo_path, alert, label). label gains culprit_hash for the scorer.""" + alert_dt = datetime.fromisoformat(ALERT_TS) + tmp = tempfile.mkdtemp(prefix="sentinel_eval_") + repo = Path(tmp) + try: + _git(repo, "init", "-q") + base_dir = BASES_DIR / case["base"] + for src in base_dir.rglob("*"): + if src.is_file(): + dest = repo / src.relative_to(base_dir) + dest.parent.mkdir(parents=True, exist_ok=True) + dest.write_bytes(src.read_bytes().replace(b"\r\n", b"\n")) + _apply(repo, case.get("base_files", {})) + _git(repo, "add", "-A") + _git(repo, "commit", "-q", "-m", "initial import", when=alert_dt - timedelta(days=3)) + + hashes = [] + for c in case["commits"]: + _apply(repo, c["files"]) + _git(repo, "add", "-A") + when = alert_dt - timedelta(minutes=c["minutes_before_alert"]) + _git(repo, "commit", "-q", "-m", c["message"], when=when, author=c["author"]) + hashes.append(_git(repo, "rev-parse", "HEAD")[:7]) # same truncation as git_client + + alert = { + "source": "mock_sentry", + "alert_name": case["alert"]["alert_name"], + "timestamp": ALERT_TS, + "error_signature": case["alert"]["error_signature"], + } + label = {**case["label"], "culprit_hash": hashes[case["label"]["culprit_index"]]} + yield str(repo), alert, label + finally: + _rmtree(tmp) diff --git a/eval/report.py b/eval/report.py new file mode 100644 index 0000000..bd50fda --- /dev/null +++ b/eval/report.py @@ -0,0 +1,167 @@ +"""Scores eval results files. One file: summary. Two files (baseline, rag): side by side + paired. + +python -m eval.report eval/results/baseline_.json [eval/results/rag_.json] +""" + +import json +import statistics +import sys +from collections import defaultdict +from pathlib import Path + +from eval.config import FLOOR_CANDIDATES + + +def load(path) -> dict: + return json.loads(Path(path).read_text(encoding="utf-8")) + + +def _pct(k, n) -> str: + return f"{100 * k / n:.1f}% ({k}/{n})" if n else "n/a (0/0)" + + +def summarize(records: list[dict]) -> dict: + """LLM errors have culprit_rank None, so they count as misses in every ranking metric.""" + n = len(records) + ranks = [r["culprit_rank"] for r in records] + right, wrong = [], [] + for r in records: + if r["error"] or not r["ranking"]: + continue + (right if r["culprit_rank"] == 1 else wrong).append(r["ranking"][0]["confidence_score"]) + return { + "n": n, + "top1": sum(k == 1 for k in ranks), + "top3": sum(k is not None and k <= 3 for k in ranks), + "mrr": sum(1 / k for k in ranks if k) / n if n else 0.0, + "errors": sum(bool(r["error"]) for r in records), + "conf_right": statistics.mean(right) if right else None, + "conf_wrong": statistics.mean(wrong) if wrong else None, + "n_right": len(right), + "n_wrong": len(wrong), + "latency_median_s": statistics.median(r["rank_latency_s"] for r in records) if records else None, + } + + +def retrieval(records: list[dict]) -> dict: + """Retrieval is deterministic per case, so score one record per case.""" + per_case = {} + for r in records: + per_case.setdefault(r["case_id"], r) + expected = [r for r in per_case.values() if r["expected_runbook_id"]] + nulls = [r for r in per_case.values() if not r["expected_runbook_id"]] + + def top(r): + return ( + r["retrieved_runbooks"][0] if r["retrieved_runbooks"] else {"id": None, "similarity_score": 0.0} + ) + + def expected_score(r): + return next( + ( + rb["similarity_score"] + for rb in r["retrieved_runbooks"] + if rb["id"] == r["expected_runbook_id"] + ), + None, + ) + + return { + "n_expected": len(expected), + "top1_hits": sum(top(r)["id"] == r["expected_runbook_id"] for r in expected), + "n_null": len(nulls), + "correct_scores": sorted(s for s in map(expected_score, expected) if s is not None), + "null_top_scores": sorted(top(r)["similarity_score"] for r in nulls), + "floors": { + f: { + # null case with anything above floor = wrong context would be sent + "false_match": sum(top(r)["similarity_score"] >= f for r in nulls), + # expected case whose correct runbook sits below floor = right context withheld + "correct_dropped": sum((expected_score(r) or 0.0) < f for r in expected), + } + for f in FLOOR_CANDIDATES + }, + } + + +def paired(a: list[dict], b: list[dict]) -> list[tuple]: + """(case_id, hits_a, trials_a, hits_b, trials_b) per case.""" + hits = defaultdict(lambda: [0, 0, 0, 0]) + for i, recs in enumerate((a, b)): + for r in recs: + h = hits[r["case_id"]] + h[2 * i] += r["culprit_rank"] == 1 + h[2 * i + 1] += 1 + return [(cid, *h) for cid, h in sorted(hits.items())] + + +def _fmt_conf(x): + return f"{x:.2f}" if x is not None else "n/a" + + +def print_summary(name: str, res: dict): + s = summarize(res["records"]) + print( + f"\n== {name}: condition={res['condition']} eval_set={res['eval_set_version']} " + f"corpus={res['runbook_corpus_version']} model={res['model']} dry_run={res['dry_run']}" + ) + print(f" top-1 {_pct(s['top1'], s['n'])}") + print(f" top-3 {_pct(s['top3'], s['n'])}") + print(f" MRR {s['mrr']:.3f} (n={s['n']})") + print(f" LLM errors {s['errors']}/{s['n']} (counted as misses)") + print( + f" conf of #1 right {_fmt_conf(s['conf_right'])} (n={s['n_right']}), " + f"wrong {_fmt_conf(s['conf_wrong'])} (n={s['n_wrong']})" + ) + print(f" rank latency median {s['latency_median_s']}s (n={s['n']})") + truncated = sorted({r["case_id"] for r in res["records"] if r["truncated_diffs"]}) + if truncated: + print(f" WARNING diffs hit the 3000-char truncation in: {', '.join(truncated)}") + + +def print_retrieval(res: dict): + rt = retrieval(res["records"]) + print("\n== Retrieval (one record per case)") + print(f" runbook top-1 on expected cases {_pct(rt['top1_hits'], rt['n_expected'])}") + print(f" correct-runbook scores {rt['correct_scores']}") + print(f" null-case top-1 scores {rt['null_top_scores']}") + print(f" {'floor':>6} {'null false-match':>18} {'correct dropped':>17}") + for f, v in rt["floors"].items(): + print( + f" {f:>6} {_pct(v['false_match'], rt['n_null']):>18} " + f"{_pct(v['correct_dropped'], rt['n_expected']):>17}" + ) + + +def print_comparison(a: dict, b: dict): + rows = paired(a["records"], b["records"]) + wins = sum(hb / tb > ha / ta for _, ha, ta, hb, tb in rows) + losses = sum(hb / tb < ha / ta for _, ha, ta, hb, tb in rows) + print(f"\n== Paired per-case top-1 hits ({a['condition']} vs {b['condition']})") + print(f" {'case':<34} {a['condition']:>9} {b['condition']:>9}") + for cid, ha, ta, hb, tb in rows: + mark = "+" if hb / tb > ha / ta else "-" if hb / tb < ha / ta else " " + print(f" {cid:<34} {ha:>5}/{ta:<3} {hb:>5}/{tb:<3} {mark}") + ties = len(rows) - wins - losses + print(f" {b['condition']} wins {wins}, losses {losses}, ties {ties} (n={len(rows)} cases)") + + print("\n== Null-runbook cases only (does irrelevant context hurt?)") + for res in (a, b): + s = summarize([r for r in res["records"] if not r["expected_runbook_id"]]) + print(f" {res['condition']:<9} top-1 {_pct(s['top1'], s['n'])}, MRR {s['mrr']:.3f}") + + +def main(argv=None): + paths = argv if argv is not None else sys.argv[1:] + if not 1 <= len(paths) <= 2: + sys.exit("usage: python -m eval.report RESULTS.json [RESULTS_B.json]") + results = [load(p) for p in paths] + for p, res in zip(paths, results): + print_summary(Path(p).name, res) + print_retrieval(results[0]) + if len(results) == 2: + print_comparison(*results) + + +if __name__ == "__main__": + main() diff --git a/eval/run_eval.py b/eval/run_eval.py new file mode 100644 index 0000000..bb9d9fc --- /dev/null +++ b/eval/run_eval.py @@ -0,0 +1,142 @@ +"""Runs the frozen eval set through the real Sentinel ranking path. + +python -m eval.run_eval --condition baseline --trials 3 --out eval/results/baseline_.json +python -m eval.run_eval --condition baseline --dry-run --out .json # mocked chat(), no API cost +""" + +import argparse +import json +import re +import subprocess +import tempfile +import time +from contextlib import nullcontext +from datetime import datetime, timezone +from pathlib import Path +from unittest.mock import patch + +from core import orchestrator +from core.services import git_client, llm_analyzer, openrouter, vector_store +from eval import config, fixture + +ROOT = Path(__file__).parent.parent + + +def run_trial(case: dict, condition: str) -> dict: + with fixture.build(case) as (repo, alert, label): + diffs = git_client.get_recent_diffs(repo, alert["timestamp"]) + rag = condition == "rag" + runbooks = ( + vector_store.find_matching_runbooks(alert["error_signature"], include_content=True) + if rag + else vector_store.find_matching_runbooks(alert["error_signature"]) + ) + record = { + "case_id": case["id"], + "category": case["category"], + "culprit_hash": label["culprit_hash"], + "expected_runbook_id": label["expected_runbook_id"], + "commits_in_window": len(diffs), + "truncated_diffs": sum(len(d["diff"]) >= 3000 for d in diffs), + "retrieved_runbooks": [ + {"id": r["id"], "similarity_score": r["similarity_score"]} for r in runbooks + ], + "ranking": [], + "culprit_rank": None, + "error": None, + } + t0 = time.monotonic() + try: + # the harness never builds a prompt itself; it calls the production function + ranked = ( + llm_analyzer.rank_suspect_commits(diffs, alert, runbooks=runbooks) + if rag + else llm_analyzer.rank_suspect_commits(diffs, alert) + ) + record["ranking"] = [ + {"commit_hash": r["commit_hash"][:7], "confidence_score": r["confidence_score"]} + for r in ranked + ] + hashes = [r["commit_hash"] for r in record["ranking"]] + if label["culprit_hash"] in hashes: + record["culprit_rank"] = hashes.index(label["culprit_hash"]) + 1 + except llm_analyzer.LLMUnavailable as e: + record["error"] = f"LLMUnavailable: {e}" # counted as a miss by report.py, never dropped + record["rank_latency_s"] = round(time.monotonic() - t0, 2) + return record + + +def _fake_chat(messages, **_): + """Dry-run stand-in for openrouter.chat: ranks commits in prompt order.""" + hashes = re.findall(r"^COMMIT (\w{7}) \|", messages[0]["content"], flags=re.M) + ranked = [ + { + "commit_hash": h, + "author": "x", + "timestamp": "x", + "rationale": "dry run", + "confidence_score": 0.9 - i / 10, + } + for i, h in enumerate(hashes) + ] + args = json.dumps({"ranked_commits": ranked}) + return { + "choices": [{"message": {"tool_calls": [{"function": {"name": "rank_commits", "arguments": args}}]}}] + } + + +def _git_sha() -> str: + sha = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=ROOT, capture_output=True, text=True + ).stdout.strip() + dirty = subprocess.run(["git", "status", "--porcelain"], cwd=ROOT, capture_output=True, text=True).stdout + return sha + ("-dirty" if dirty.strip() else "") + + +def main(argv=None): + p = argparse.ArgumentParser() + p.add_argument("--condition", choices=["baseline", "rag"], required=True) + p.add_argument("--trials", type=int, default=config.DEFAULT_TRIALS) + p.add_argument("--cases", default="*.json", help="glob inside eval/cases") + p.add_argument("--out", required=True) + p.add_argument("--dry-run", action="store_true", help="mock chat(); no API calls") + args = p.parse_args(argv) + + # isolated Chroma store holding exactly the frozen corpus, never the dev .chroma + vector_store.CHROMA_PATH = tempfile.mkdtemp(prefix="sentinel_eval_chroma_") + n_runbooks = vector_store.ingest_runbooks(orchestrator.RUNBOOKS_DIR) + + cases = [fixture.load_case(path) for path in fixture.all_case_paths(args.cases)] + out = { + "condition": args.condition, + "dry_run": args.dry_run, + "eval_set_version": config.EVAL_SET_VERSION, + "runbook_corpus_version": config.RUNBOOK_CORPUS_VERSION, + "runbooks_ingested": n_runbooks, + "model": openrouter.MODEL, + "sentinel_git_sha": _git_sha(), + "started_at": datetime.now(timezone.utc).isoformat(), + "trials": args.trials, + "n_cases": len(cases), + "records": [], + } + out_path = Path(args.out) + out_path.parent.mkdir(parents=True, exist_ok=True) + + with patch.object(llm_analyzer, "chat", _fake_chat) if args.dry_run else nullcontext(): + for trial in range(args.trials): + for case in cases: + rec = {"trial": trial, **run_trial(case, args.condition)} + out["records"].append(rec) + # rewrite after every trial so an interrupted run keeps what it measured + out_path.write_text(json.dumps(out, indent=2), encoding="utf-8") + status = rec["error"] or f"culprit rank {rec['culprit_rank']}" + print(f"[eval] trial {trial} {case['id']}: {status} ({rec['rank_latency_s']}s)") + + out["finished_at"] = datetime.now(timezone.utc).isoformat() + out_path.write_text(json.dumps(out, indent=2), encoding="utf-8") + print(f"[eval] wrote {len(out['records'])} records to {out_path}") + + +if __name__ == "__main__": + main() diff --git a/eval/tests/__init__.py b/eval/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/eval/tests/test_eval.py b/eval/tests/test_eval.py new file mode 100644 index 0000000..2c75642 --- /dev/null +++ b/eval/tests/test_eval.py @@ -0,0 +1,146 @@ +"""Offline eval-harness tests: no API calls, no embedding model, vector store mocked.""" + +import inspect +import re +from unittest.mock import patch + +import pytest + +from core.services import git_client, llm_analyzer +from eval import config, fixture, report, run_eval + +CASES = [fixture.load_case(p) for p in fixture.all_case_paths()] +_BANNED = re.compile(r"\b(" + "|".join(config.BANNED_WORDS) + r")", re.IGNORECASE) +# label-independent stand-in, so a label value can only reach the prompt by leaking +_DUMMY_RUNBOOKS = [ + { + "id": "rb_dummy", + "title": "Dummy", + "similarity_score": 0.5, + "primary_action": "n/a", + "content": "## Symptoms\n- something\n", + } +] + + +def _commit_texts(case): + for commit in case["commits"]: + yield commit["message"] + for change in commit["files"].values(): + if isinstance(change, str): + yield change + else: + for _old, new in change: + yield new + + +# ---------- case lint ---------- + + +def test_case_set_shape(): + assert 24 <= len(CASES) <= 30 + assert len({c["category"] for c in CASES}) >= 6 + nulls = sum(c["label"]["expected_runbook_id"] is None for c in CASES) + assert 0.2 <= nulls / len(CASES) <= 0.3 + + +@pytest.mark.parametrize("case", CASES, ids=[c["id"] for c in CASES]) +def test_case_lint(case): + assert set(case) >= {"id", "category", "base", "alert", "commits", "label"} + assert 3 <= len(case["commits"]) <= 6 + assert 0 <= case["label"]["culprit_index"] < len(case["commits"]) + minutes = [c["minutes_before_alert"] for c in case["commits"]] + assert minutes == sorted(minutes, reverse=True), "commits must be listed oldest first" + assert all(0 < m < 60 for m in minutes), "every commit must fall inside the 60-min window" + for text in _commit_texts(case): + assert not _BANNED.search(text), f"banned word in: {text[:80]!r}" + + +def test_base_has_no_banned_words(): + for path in (fixture.BASES_DIR).rglob("*"): + if path.is_file(): + assert not _BANNED.search(path.read_text(encoding="utf-8")), path + + +# ---------- fixture ---------- + + +@pytest.mark.parametrize("case", CASES, ids=[c["id"] for c in CASES]) +def test_fixture_commits_land_in_window_in_order(case): + with fixture.build(case) as (repo, alert, label): + diffs = git_client.get_recent_diffs(repo, alert["timestamp"]) + assert [d["subject"] for d in diffs] == [c["message"] for c in reversed(case["commits"])] + assert label["culprit_hash"] in {d["commit_hash"] for d in diffs} + assert all(len(d["diff"]) < 3000 for d in diffs), "diff hit the truncation limit" + + +# ---------- scorer ---------- + + +def _rec(case_id, rank, conf=0.9, error=None, expected="rb_a"): + ranking = [] if error else [{"commit_hash": "x", "confidence_score": conf}] + return { + "case_id": case_id, + "culprit_rank": rank, + "ranking": ranking, + "error": error, + "rank_latency_s": 1.0, + "expected_runbook_id": expected, + "retrieved_runbooks": [{"id": "rb_a", "similarity_score": 0.4}], + "truncated_diffs": 0, + } + + +def test_summarize_counts_errors_as_misses(): + s = report.summarize([_rec("a", 1), _rec("b", 2, conf=0.6), _rec("c", 4), _rec("d", None, error="x")]) + assert (s["n"], s["top1"], s["top3"], s["errors"]) == (4, 1, 2, 1) + assert s["mrr"] == pytest.approx((1 + 1 / 2 + 1 / 4 + 0) / 4) + assert s["conf_right"] == 0.9 and s["conf_wrong"] == pytest.approx(0.75) + + +def test_retrieval_and_floors(): + recs = [_rec("a", 1), _rec("b", 1, expected=None), _rec("b", 1, expected=None)] + rt = report.retrieval(recs) + assert (rt["n_expected"], rt["top1_hits"], rt["n_null"]) == (1, 1, 1) # one record per case + assert rt["floors"][0.35] == {"false_match": 1, "correct_dropped": 0} + assert rt["floors"][0.4] == {"false_match": 1, "correct_dropped": 0} + + +def test_paired(): + a = [_rec("a", 1), _rec("a", 2), _rec("b", 1)] + b = [_rec("a", 1), _rec("a", 1), _rec("b", 3)] + assert report.paired(a, b) == [("a", 1, 2, 2, 2), ("b", 1, 1, 0, 1)] + + +# ---------- leakage + dry run through the real ranking path ---------- + +_RAG_READY = "runbooks" in inspect.signature(llm_analyzer.rank_suspect_commits).parameters + + +@pytest.mark.parametrize( + "condition", + [ + "baseline", + pytest.param("rag", marks=pytest.mark.skipif(not _RAG_READY, reason="runbooks param lands in 9E")), + ], +) +def test_no_label_leaks_into_prompt(condition): + prompts = [] + + def capture(messages, **kw): + prompts.append(messages[0]["content"]) + return run_eval._fake_chat(messages, **kw) + + with ( + patch.object(llm_analyzer, "chat", capture), + patch.object(run_eval.vector_store, "find_matching_runbooks", return_value=_DUMMY_RUNBOOKS), + ): + for case in CASES: + rec = run_eval.run_trial(case, condition) + assert rec["error"] is None and rec["culprit_rank"] is not None, case["id"] + prompt = prompts[-1] + leaks = [case["id"], case["category"], case["label"]["expected_runbook_id"]] + for value in filter(None, leaks): + assert value not in prompt, f"{case['id']}: {value!r} leaked into prompt" + assert "culprit" not in prompt.lower() + assert len(prompts) == len(CASES) From d6f6bd1175ede7efc58447edb2877b4c739f1b67 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 14:04:21 -0400 Subject: [PATCH 05/19] feat(eval): 29-case fault set on a shared orders-service base, frozen as v1 Co-Authored-By: Claude Opus 5.5 --- eval/bases/orders_service/README.md | 8 ++ eval/bases/orders_service/app/auth.py | 23 ++++++ eval/bases/orders_service/app/cache.py | 23 ++++++ .../orders_service/app/clients/inventory.py | 26 ++++++ .../orders_service/app/clients/payments.py | 22 +++++ eval/bases/orders_service/app/db.py | 35 ++++++++ eval/bases/orders_service/app/exports.py | 16 ++++ eval/bases/orders_service/app/gateway.py | 15 ++++ .../orders_service/app/handlers/orders.py | 30 +++++++ eval/bases/orders_service/app/pagination.py | 18 +++++ eval/bases/orders_service/app/serializers.py | 29 +++++++ eval/bases/orders_service/app/settings.py | 13 +++ eval/bases/orders_service/app/worker.py | 27 +++++++ eval/bases/orders_service/config.json | 10 +++ eval/bases/orders_service/requirements.txt | 4 + eval/cases/auth_algorithm_02.json | 54 +++++++++++++ eval/cases/auth_audience_rename_01.json | 71 ++++++++++++++++ eval/cases/auth_leeway_zero_03.json | 67 ++++++++++++++++ eval/cases/cache_recursion_04.json | 58 ++++++++++++++ eval/cases/cfg_cache_size_string_01.json | 79 ++++++++++++++++++ eval/cases/cfg_env_override_string_03.json | 80 +++++++++++++++++++ eval/cases/cfg_json_trailing_comma_02.json | 71 ++++++++++++++++ eval/cases/cfg_renamed_gateway_key_04.json | 54 +++++++++++++ eval/cases/db_pool_leak_01.json | 67 ++++++++++++++++ eval/cases/deps_psycopg3_01.json | 67 ++++++++++++++++ eval/cases/deps_redis_legacy_02.json | 71 ++++++++++++++++ eval/cases/fd_export_writer_01.json | 54 +++++++++++++ eval/cases/mem_worker_seen_ids_01.json | 79 ++++++++++++++++++ eval/cases/null_guest_email_01.json | 71 ++++++++++++++++ eval/cases/null_since_filter_03.json | 80 +++++++++++++++++++ eval/cases/null_stock_payload_02.json | 66 +++++++++++++++ eval/cases/page_newest_empty_01.json | 54 +++++++++++++ eval/cases/page_size_zero_02.json | 67 ++++++++++++++++ eval/cases/payments_form_body_07.json | 80 +++++++++++++++++++ eval/cases/queue_args_swapped_03.json | 58 ++++++++++++++ eval/cases/rl_inventory_uncached_01.json | 67 ++++++++++++++++ eval/cases/rl_payment_retries_02.json | 68 ++++++++++++++++ eval/cases/ser_decimal_json_02.json | 71 ++++++++++++++++ eval/cases/sku_lowercase_06.json | 62 ++++++++++++++ eval/cases/sql_param_count_05.json | 67 ++++++++++++++++ eval/cases/tmo_gateway_deadline_02.json | 67 ++++++++++++++++ eval/cases/tmo_inventory_units_01.json | 66 +++++++++++++++ eval/cases/tmo_list_live_stock_03.json | 54 +++++++++++++ eval/cases/tz_naive_since_01.json | 54 +++++++++++++ 44 files changed, 2223 insertions(+) create mode 100644 eval/bases/orders_service/README.md create mode 100644 eval/bases/orders_service/app/auth.py create mode 100644 eval/bases/orders_service/app/cache.py create mode 100644 eval/bases/orders_service/app/clients/inventory.py create mode 100644 eval/bases/orders_service/app/clients/payments.py create mode 100644 eval/bases/orders_service/app/db.py create mode 100644 eval/bases/orders_service/app/exports.py create mode 100644 eval/bases/orders_service/app/gateway.py create mode 100644 eval/bases/orders_service/app/handlers/orders.py create mode 100644 eval/bases/orders_service/app/pagination.py create mode 100644 eval/bases/orders_service/app/serializers.py create mode 100644 eval/bases/orders_service/app/settings.py create mode 100644 eval/bases/orders_service/app/worker.py create mode 100644 eval/bases/orders_service/config.json create mode 100644 eval/bases/orders_service/requirements.txt create mode 100644 eval/cases/auth_algorithm_02.json create mode 100644 eval/cases/auth_audience_rename_01.json create mode 100644 eval/cases/auth_leeway_zero_03.json create mode 100644 eval/cases/cache_recursion_04.json create mode 100644 eval/cases/cfg_cache_size_string_01.json create mode 100644 eval/cases/cfg_env_override_string_03.json create mode 100644 eval/cases/cfg_json_trailing_comma_02.json create mode 100644 eval/cases/cfg_renamed_gateway_key_04.json create mode 100644 eval/cases/db_pool_leak_01.json create mode 100644 eval/cases/deps_psycopg3_01.json create mode 100644 eval/cases/deps_redis_legacy_02.json create mode 100644 eval/cases/fd_export_writer_01.json create mode 100644 eval/cases/mem_worker_seen_ids_01.json create mode 100644 eval/cases/null_guest_email_01.json create mode 100644 eval/cases/null_since_filter_03.json create mode 100644 eval/cases/null_stock_payload_02.json create mode 100644 eval/cases/page_newest_empty_01.json create mode 100644 eval/cases/page_size_zero_02.json create mode 100644 eval/cases/payments_form_body_07.json create mode 100644 eval/cases/queue_args_swapped_03.json create mode 100644 eval/cases/rl_inventory_uncached_01.json create mode 100644 eval/cases/rl_payment_retries_02.json create mode 100644 eval/cases/ser_decimal_json_02.json create mode 100644 eval/cases/sku_lowercase_06.json create mode 100644 eval/cases/sql_param_count_05.json create mode 100644 eval/cases/tmo_gateway_deadline_02.json create mode 100644 eval/cases/tmo_inventory_units_01.json create mode 100644 eval/cases/tmo_list_live_stock_03.json create mode 100644 eval/cases/tz_naive_since_01.json diff --git a/eval/bases/orders_service/README.md b/eval/bases/orders_service/README.md new file mode 100644 index 0000000..7f9f051 --- /dev/null +++ b/eval/bases/orders_service/README.md @@ -0,0 +1,8 @@ +# orders-service + +Order placement, listing, and CSV export for the storefront. + +- `app/handlers/` request handlers +- `app/clients/` outbound HTTP clients (payments, inventory) +- `app/worker.py` fulfilment queue consumer +- `config.json` defaults; override with `ORDERS_
_` (JSON-encoded) diff --git a/eval/bases/orders_service/app/auth.py b/eval/bases/orders_service/app/auth.py new file mode 100644 index 0000000..c0c32f5 --- /dev/null +++ b/eval/bases/orders_service/app/auth.py @@ -0,0 +1,23 @@ +import jwt + +from app import settings + +_ALGORITHMS = ["RS256"] + + +def verify_token(token: str, public_key: str) -> dict: + return jwt.decode( + token, + public_key, + algorithms=_ALGORITHMS, + audience=settings.get("auth", "audience"), + issuer=settings.get("auth", "issuer"), + leeway=settings.get("auth", "leeway_s"), + ) + + +def current_user_id(headers: dict, public_key: str) -> str: + scheme, _, token = headers.get("authorization", "").partition(" ") + if scheme.lower() != "bearer" or not token: + raise PermissionError("missing bearer token") + return verify_token(token, public_key)["sub"] diff --git a/eval/bases/orders_service/app/cache.py b/eval/bases/orders_service/app/cache.py new file mode 100644 index 0000000..cbee03a --- /dev/null +++ b/eval/bases/orders_service/app/cache.py @@ -0,0 +1,23 @@ +import time + + +class TTLCache: + def __init__(self, ttl_s: float, max_entries: int): + self.ttl_s = ttl_s + self.max_entries = max_entries + self._data: dict = {} + + def get(self, key): + entry = self._data.get(key) + if entry is None: + return None + value, expires_at = entry + if time.monotonic() > expires_at: + del self._data[key] + return None + return value + + def set(self, key, value): + if len(self._data) >= self.max_entries: + self._data.pop(next(iter(self._data))) + self._data[key] = (value, time.monotonic() + self.ttl_s) diff --git a/eval/bases/orders_service/app/clients/inventory.py b/eval/bases/orders_service/app/clients/inventory.py new file mode 100644 index 0000000..a26db1d --- /dev/null +++ b/eval/bases/orders_service/app/clients/inventory.py @@ -0,0 +1,26 @@ +import httpx + +from app import settings +from app.cache import TTLCache + +_stock_cache = TTLCache( + ttl_s=settings.get("cache", "ttl_s"), max_entries=settings.get("cache", "max_entries") +) + + +def get_stock(sku: str) -> int: + cached = _stock_cache.get(sku) + if cached is not None: + return cached + resp = httpx.get( + f"{settings.get('inventory', 'base_url')}/v2/stock/{sku}", + timeout=settings.get("inventory", "timeout_s"), + ) + resp.raise_for_status() + qty = resp.json()["available"] + _stock_cache.set(sku, qty) + return qty + + +def get_stock_many(skus: list[str]) -> dict[str, int]: + return {sku: get_stock(sku) for sku in skus} diff --git a/eval/bases/orders_service/app/clients/payments.py b/eval/bases/orders_service/app/clients/payments.py new file mode 100644 index 0000000..39388dc --- /dev/null +++ b/eval/bases/orders_service/app/clients/payments.py @@ -0,0 +1,22 @@ +import httpx + +from app import settings + + +def charge(order_id: str, amount_cents: int, card_token: str) -> dict: + resp = httpx.post( + f"{settings.get('payments', 'base_url')}/v1/charges", + json={"order_id": order_id, "amount_cents": amount_cents, "card_token": card_token}, + timeout=settings.get("payments", "timeout_s"), + ) + resp.raise_for_status() + return resp.json() + + +def refund(charge_id: str) -> dict: + resp = httpx.post( + f"{settings.get('payments', 'base_url')}/v1/charges/{charge_id}/refund", + timeout=settings.get("payments", "timeout_s"), + ) + resp.raise_for_status() + return resp.json() diff --git a/eval/bases/orders_service/app/db.py b/eval/bases/orders_service/app/db.py new file mode 100644 index 0000000..e80ea59 --- /dev/null +++ b/eval/bases/orders_service/app/db.py @@ -0,0 +1,35 @@ +import psycopg2.pool + +from app import settings + +_pool = None + + +def get_pool(): + global _pool + if _pool is None: + _pool = psycopg2.pool.ThreadedConnectionPool( + minconn=1, + maxconn=settings.get("db", "pool_size"), + host=settings.get("db", "host"), + port=settings.get("db", "port"), + dbname=settings.get("db", "name"), + connect_timeout=settings.get("db", "connect_timeout_s"), + ) + return _pool + + +def fetch_all(sql: str, params: tuple = ()) -> list: + pool = get_pool() + conn = pool.getconn() + try: + with conn.cursor() as cur: + cur.execute(sql, params) + return cur.fetchall() + finally: + pool.putconn(conn) + + +def fetch_one(sql: str, params: tuple = ()): + rows = fetch_all(sql, params) + return rows[0] if rows else None diff --git a/eval/bases/orders_service/app/exports.py b/eval/bases/orders_service/app/exports.py new file mode 100644 index 0000000..7a6f3da --- /dev/null +++ b/eval/bases/orders_service/app/exports.py @@ -0,0 +1,16 @@ +import csv +import os +from datetime import datetime, timezone + +from app import settings + + +def export_orders_csv(orders: list[dict]) -> str: + stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%S") + path = os.path.join(settings.get("exports", "dir"), f"orders_{stamp}.csv") + with open(path, "w", newline="") as fh: + writer = csv.writer(fh) + writer.writerow(["id", "status", "total", "created_at"]) + for o in orders: + writer.writerow([o["id"], o["status"], o["total"], o["created_at"]]) + return path diff --git a/eval/bases/orders_service/app/gateway.py b/eval/bases/orders_service/app/gateway.py new file mode 100644 index 0000000..40e714a --- /dev/null +++ b/eval/bases/orders_service/app/gateway.py @@ -0,0 +1,15 @@ +"""Edge handler: runs a request handler under the upstream deadline.""" + +import concurrent.futures + +from app import settings + +_executor = concurrent.futures.ThreadPoolExecutor(max_workers=32) + + +def run_with_deadline(handler, *args, **kwargs) -> dict: + future = _executor.submit(handler, *args, **kwargs) + try: + return future.result(timeout=settings.get("gateway", "upstream_timeout_s")) + except concurrent.futures.TimeoutError: + return {"status": 504, "body": {"error": "upstream timeout"}} diff --git a/eval/bases/orders_service/app/handlers/orders.py b/eval/bases/orders_service/app/handlers/orders.py new file mode 100644 index 0000000..225e372 --- /dev/null +++ b/eval/bases/orders_service/app/handlers/orders.py @@ -0,0 +1,30 @@ +from app import db, serializers +from app.clients import inventory, payments +from app.pagination import clamp_page_size, paginate + +_ORDER_SQL = "SELECT * FROM orders_view WHERE id = %s" +_LIST_SQL = "SELECT * FROM orders_view WHERE customer_id = %s ORDER BY created_at DESC, id DESC" + + +def get_order(order_id: str) -> dict: + row = db.fetch_one(_ORDER_SQL, (order_id,)) + if row is None: + return {"status": 404, "body": {"error": "order not found"}} + return {"status": 200, "body": serializers.order_to_dict(row)} + + +def list_orders(customer_id: str, page: int = 1, page_size: int | None = None) -> dict: + rows = db.fetch_all(_LIST_SQL, (customer_id,)) + body = paginate([serializers.order_to_dict(r) for r in rows], page, clamp_page_size(page_size)) + return {"status": 200, "body": body} + + +def create_order(customer_id: str, lines: list[dict], card_token: str) -> dict: + stock = inventory.get_stock_many([line["sku"] for line in lines]) + short = [line["sku"] for line in lines if stock[line["sku"]] < line["qty"]] + if short: + return {"status": 409, "body": {"error": "insufficient stock", "skus": short}} + total = sum(line["qty"] * line["unit_price_cents"] for line in lines) + order_id = db.fetch_one("SELECT create_order(%s, %s)", (customer_id, total))[0] + payments.charge(order_id, total, card_token) + return {"status": 201, "body": {"id": order_id, "total": serializers.money(total)}} diff --git a/eval/bases/orders_service/app/pagination.py b/eval/bases/orders_service/app/pagination.py new file mode 100644 index 0000000..235d288 --- /dev/null +++ b/eval/bases/orders_service/app/pagination.py @@ -0,0 +1,18 @@ +from app import settings + + +def page_bounds(page: int, page_size: int) -> tuple[int, int]: + """page is 1-based; returns a half-open [start, end) slice.""" + start = (page - 1) * page_size + return start, start + page_size + + +def clamp_page_size(requested: int | None) -> int: + if requested is None: + return settings.get("api", "page_size") + return max(1, min(requested, settings.get("api", "max_page_size"))) + + +def paginate(items: list, page: int, page_size: int) -> dict: + start, end = page_bounds(page, page_size) + return {"items": items[start:end], "page": page, "has_more": end < len(items)} diff --git a/eval/bases/orders_service/app/serializers.py b/eval/bases/orders_service/app/serializers.py new file mode 100644 index 0000000..ec8853c --- /dev/null +++ b/eval/bases/orders_service/app/serializers.py @@ -0,0 +1,29 @@ +from datetime import datetime +from decimal import Decimal + + +def money(cents: int) -> str: + return str((Decimal(cents) / 100).quantize(Decimal("0.01"))) + + +def order_to_dict(row: dict) -> dict: + return { + "id": row["id"], + "status": row["status"], + "total": money(row["total_cents"]), + "currency": row["currency"], + "created_at": row["created_at"].isoformat(), + "customer": { + "id": row["customer_id"], + "email": row["customer_email"], + }, + "items": [line_to_dict(line) for line in row["lines"]], + } + + +def line_to_dict(line: dict) -> dict: + return {"sku": line["sku"], "qty": line["qty"], "unit_price": money(line["unit_price_cents"])} + + +def parse_since(value: str | None) -> datetime | None: + return datetime.fromisoformat(value) if value else None diff --git a/eval/bases/orders_service/app/settings.py b/eval/bases/orders_service/app/settings.py new file mode 100644 index 0000000..1013590 --- /dev/null +++ b/eval/bases/orders_service/app/settings.py @@ -0,0 +1,13 @@ +import json +import os +from pathlib import Path + +_CONFIG = json.loads((Path(__file__).parent.parent / "config.json").read_text()) + + +def get(section: str, key: str): + """Config value, overridable by ORDERS_
_ in the environment.""" + env_key = f"ORDERS_{section.upper()}_{key.upper()}" + if env_key in os.environ: + return json.loads(os.environ[env_key]) + return _CONFIG[section][key] diff --git a/eval/bases/orders_service/app/worker.py b/eval/bases/orders_service/app/worker.py new file mode 100644 index 0000000..80d6405 --- /dev/null +++ b/eval/bases/orders_service/app/worker.py @@ -0,0 +1,27 @@ +"""Background worker: drains the fulfilment queue.""" + +import json +import time + +import redis + +_r = redis.Redis(host="10.0.4.20", port=6379) +QUEUE = "fulfilment" +PROCESSING = "fulfilment:processing" + + +def handle(job: dict): + from app.clients import inventory + + inventory.get_stock(job["sku"]) + + +def run_forever(): + while True: + raw = _r.brpoplpush(QUEUE, PROCESSING, timeout=5) + if raw is None: + continue + job = json.loads(raw) + handle(job) + _r.lrem(PROCESSING, 1, raw) + time.sleep(0.01) diff --git a/eval/bases/orders_service/config.json b/eval/bases/orders_service/config.json new file mode 100644 index 0000000..c27440e --- /dev/null +++ b/eval/bases/orders_service/config.json @@ -0,0 +1,10 @@ +{ + "db": {"host": "10.0.4.12", "port": 5432, "name": "orders", "pool_size": 10, "connect_timeout_s": 5}, + "payments": {"base_url": "https://payments.internal", "timeout_s": 8, "retries": 2}, + "inventory": {"base_url": "https://inventory.internal", "timeout_s": 3}, + "gateway": {"upstream_timeout_s": 15}, + "cache": {"ttl_s": 300, "max_entries": 5000}, + "api": {"page_size": 50, "max_page_size": 200}, + "auth": {"issuer": "https://auth.internal", "audience": "orders-api", "leeway_s": 30}, + "exports": {"dir": "/var/lib/orders/exports"} +} diff --git a/eval/bases/orders_service/requirements.txt b/eval/bases/orders_service/requirements.txt new file mode 100644 index 0000000..acab8c1 --- /dev/null +++ b/eval/bases/orders_service/requirements.txt @@ -0,0 +1,4 @@ +httpx==0.27.2 +psycopg2-binary==2.9.9 +PyJWT==2.8.0 +redis==5.0.8 diff --git a/eval/cases/auth_algorithm_02.json b/eval/cases/auth_algorithm_02.json new file mode 100644 index 0000000..b9eed76 --- /dev/null +++ b/eval/cases/auth_algorithm_02.json @@ -0,0 +1,54 @@ +{ + "id": "auth_algorithm_02", + "category": "auth_rejection", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_401_Unauthorized_Spike", + "error_signature": "jwt.exceptions.InvalidAlgorithmError: The specified alg value is not allowed" + }, + "commits": [ + { + "message": "config: widen token leeway to 60s", + "author": "l.becker@example.com", + "minutes_before_alert": 44, + "files": { + "config.json": [ + [ + "\"leeway_s\": 30", + "\"leeway_s\": 60" + ] + ] + } + }, + { + "message": "payments: send idempotency key with charges", + "author": "p.nair@example.com", + "minutes_before_alert": 28, + "files": { + "app/clients/payments.py": [ + [ + " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n", + " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n headers={\"Idempotency-Key\": f\"charge-{order_id}\"},\n" + ] + ] + } + }, + { + "message": "auth: prepare for ES256 signing keys", + "author": "a.chen@example.com", + "minutes_before_alert": 13, + "files": { + "app/auth.py": [ + [ + "_ALGORITHMS = [\"RS256\"]", + "_ALGORITHMS = [\"ES256\"]" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": "auth_failures" + } +} diff --git a/eval/cases/auth_audience_rename_01.json b/eval/cases/auth_audience_rename_01.json new file mode 100644 index 0000000..e203729 --- /dev/null +++ b/eval/cases/auth_audience_rename_01.json @@ -0,0 +1,71 @@ +{ + "id": "auth_audience_rename_01", + "category": "auth_rejection", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_401_Unauthorized_Spike", + "error_signature": "jwt.exceptions.InvalidAudienceError: Audience doesn't match" + }, + "commits": [ + { + "message": "auth: strip whitespace around bearer token", + "author": "j.silva@example.com", + "minutes_before_alert": 53, + "files": { + "app/auth.py": [ + [ + " scheme, _, token = headers.get(\"authorization\", \"\").partition(\" \")\n", + " scheme, _, token = headers.get(\"authorization\", \"\").strip().partition(\" \")\n token = token.strip()\n" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "m.okafor@example.com", + "minutes_before_alert": 37, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + }, + { + "message": "config: align service identifiers with service catalog", + "author": "r.haddad@example.com", + "minutes_before_alert": 21, + "files": { + "config.json": [ + [ + "\"audience\": \"orders-api\"", + "\"audience\": \"orders\"" + ] + ] + } + }, + { + "message": "exports: include currency column", + "author": "a.chen@example.com", + "minutes_before_alert": 5, + "files": { + "app/exports.py": [ + [ + "writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])", + "writer.writerow([\"id\", \"status\", \"total\", \"currency\", \"created_at\"])" + ], + [ + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])", + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"currency\"], o[\"created_at\"]])" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": "auth_failures" + } +} diff --git a/eval/cases/auth_leeway_zero_03.json b/eval/cases/auth_leeway_zero_03.json new file mode 100644 index 0000000..5a417f2 --- /dev/null +++ b/eval/cases/auth_leeway_zero_03.json @@ -0,0 +1,67 @@ +{ + "id": "auth_leeway_zero_03", + "category": "auth_rejection", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_401_Unauthorized_Spike", + "error_signature": "jwt.exceptions.ImmatureSignatureError: The token is not yet valid (iat)" + }, + "commits": [ + { + "message": "security: remove token leeway per audit finding SEC-212", + "author": "r.haddad@example.com", + "minutes_before_alert": 59, + "files": { + "config.json": [ + [ + "\"leeway_s\": 30", + "\"leeway_s\": 0" + ] + ] + } + }, + { + "message": "auth: strip whitespace around bearer token", + "author": "a.chen@example.com", + "minutes_before_alert": 42, + "files": { + "app/auth.py": [ + [ + " scheme, _, token = headers.get(\"authorization\", \"\").partition(\" \")\n", + " scheme, _, token = headers.get(\"authorization\", \"\").strip().partition(\" \")\n token = token.strip()\n" + ] + ] + } + }, + { + "message": "gateway: raise worker pool to 48", + "author": "m.okafor@example.com", + "minutes_before_alert": 26, + "files": { + "app/gateway.py": [ + [ + "max_workers=32", + "max_workers=48" + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "j.silva@example.com", + "minutes_before_alert": 10, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": "auth_failures" + } +} diff --git a/eval/cases/cache_recursion_04.json b/eval/cases/cache_recursion_04.json new file mode 100644 index 0000000..ac3e898 --- /dev/null +++ b/eval/cases/cache_recursion_04.json @@ -0,0 +1,58 @@ +{ + "id": "cache_recursion_04", + "category": "infinite_recursion", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "RecursionError: maximum recursion depth exceeded" + }, + "commits": [ + { + "message": "config: shorten stock cache ttl", + "author": "r.haddad@example.com", + "minutes_before_alert": 47, + "files": { + "config.json": [ + [ + "\"ttl_s\": 300", + "\"ttl_s\": 240" + ] + ] + } + }, + { + "message": "cache: share expiry check between get and membership test", + "author": "m.okafor@example.com", + "minutes_before_alert": 33, + "files": { + "app/cache.py": [ + [ + " def get(self, key):\n entry = self._data.get(key)\n if entry is None:\n return None\n", + " def __contains__(self, key):\n return self.get(key) is not None\n\n def get(self, key):\n if key not in self:\n return None\n entry = self._data.get(key)\n" + ] + ] + } + }, + { + "message": "inventory: send request id header", + "author": "j.silva@example.com", + "minutes_before_alert": 16, + "files": { + "app/clients/inventory.py": [ + [ + "def get_stock(sku: str) -> int:\n", + "def get_stock(sku: str, request_id: str | None = None) -> int:\n" + ], + [ + " timeout=settings.get(\"inventory\", \"timeout_s\"),\n", + " headers={\"X-Request-Id\": request_id} if request_id else None,\n timeout=settings.get(\"inventory\", \"timeout_s\"),\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 1, + "expected_runbook_id": null + } +} diff --git a/eval/cases/cfg_cache_size_string_01.json b/eval/cases/cfg_cache_size_string_01.json new file mode 100644 index 0000000..43d8cca --- /dev/null +++ b/eval/cases/cfg_cache_size_string_01.json @@ -0,0 +1,79 @@ +{ + "id": "cfg_cache_size_string_01", + "category": "config_type_error", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "TypeError: '>=' not supported between instances of 'int' and 'str'" + }, + "commits": [ + { + "message": "cache: track hit and miss counts", + "author": "a.chen@example.com", + "minutes_before_alert": 55, + "files": { + "app/cache.py": [ + [ + " self._data: dict = {}\n", + " self._data: dict = {}\n self.hits = 0\n self.misses = 0\n" + ], + [ + " entry = self._data.get(key)\n if entry is None:\n return None\n", + " entry = self._data.get(key)\n if entry is None:\n self.misses += 1\n return None\n" + ], + [ + " del self._data[key]\n return None\n return value\n", + " del self._data[key]\n self.misses += 1\n return None\n self.hits += 1\n return value\n" + ] + ] + } + }, + { + "message": "config: raise stock cache size for holiday catalog", + "author": "m.okafor@example.com", + "minutes_before_alert": 41, + "files": { + "config.json": [ + [ + "\"cache\": {\"ttl_s\": 300, \"max_entries\": 5000}", + "\"cache\": {\"ttl_s\": 300, \"max_entries\": \"20_000\"}" + ] + ] + } + }, + { + "message": "inventory: send request id header", + "author": "j.silva@example.com", + "minutes_before_alert": 27, + "files": { + "app/clients/inventory.py": [ + [ + "def get_stock(sku: str) -> int:\n", + "def get_stock(sku: str, request_id: str | None = None) -> int:\n" + ], + [ + " timeout=settings.get(\"inventory\", \"timeout_s\"),\n", + " headers={\"X-Request-Id\": request_id} if request_id else None,\n timeout=settings.get(\"inventory\", \"timeout_s\"),\n" + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "p.nair@example.com", + "minutes_before_alert": 12, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 1, + "expected_runbook_id": "config_parse_errors" + } +} diff --git a/eval/cases/cfg_env_override_string_03.json b/eval/cases/cfg_env_override_string_03.json new file mode 100644 index 0000000..f21b317 --- /dev/null +++ b/eval/cases/cfg_env_override_string_03.json @@ -0,0 +1,80 @@ +{ + "id": "cfg_env_override_string_03", + "category": "config_type_error", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "TypeError: '<' not supported between instances of 'str' and 'int'" + }, + "commits": [ + { + "message": "settings: read env overrides as plain strings", + "author": "p.nair@example.com", + "minutes_before_alert": 57, + "files": { + "app/settings.py": [ + [ + " return json.loads(os.environ[env_key])\n", + " return os.environ[env_key]\n" + ] + ] + } + }, + { + "message": "pagination: document page_size bounds", + "author": "j.silva@example.com", + "minutes_before_alert": 46, + "files": { + "app/pagination.py": [ + [ + "def clamp_page_size(requested: int | None) -> int:\n", + "def clamp_page_size(requested: int | None) -> int:\n \"\"\"Requested size clamped to [1, api.max_page_size]; api.page_size when omitted.\"\"\"\n" + ] + ] + } + }, + { + "message": "orders: order history by id only", + "author": "a.chen@example.com", + "minutes_before_alert": 31, + "files": { + "app/handlers/orders.py": [ + [ + "ORDER BY created_at DESC, id DESC\"", + "ORDER BY id DESC\"" + ] + ] + } + }, + { + "message": "worker: shorter idle poll", + "author": "l.becker@example.com", + "minutes_before_alert": 18, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + }, + { + "message": "cache: type hints", + "author": "r.haddad@example.com", + "minutes_before_alert": 6, + "files": { + "app/cache.py": [ + [ + " def get(self, key):\n", + " def get(self, key: str):\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": "config_parse_errors" + } +} diff --git a/eval/cases/cfg_json_trailing_comma_02.json b/eval/cases/cfg_json_trailing_comma_02.json new file mode 100644 index 0000000..faba877 --- /dev/null +++ b/eval/cases/cfg_json_trailing_comma_02.json @@ -0,0 +1,71 @@ +{ + "id": "cfg_json_trailing_comma_02", + "category": "config_type_error", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_503_Service_Unavailable", + "error_signature": "json.decoder.JSONDecodeError: Expecting property name enclosed in double quotes: line 10 column 55 (char 599)" + }, + "commits": [ + { + "message": "config: add feature toggles section", + "author": "l.becker@example.com", + "minutes_before_alert": 52, + "files": { + "config.json": [ + [ + " \"exports\": {\"dir\": \"/var/lib/orders/exports\"}\n", + " \"exports\": {\"dir\": \"/var/lib/orders/exports\"},\n \"features\": {\"live_stock\": true, \"csv_export\": true,}\n" + ] + ] + } + }, + { + "message": "settings: add section() helper", + "author": "a.chen@example.com", + "minutes_before_alert": 38, + "files": { + "app/settings.py": [ + [ + " return _CONFIG[section][key]\n", + " return _CONFIG[section][key]\n\n\ndef section(name: str) -> dict:\n return dict(_CONFIG[name])\n" + ] + ] + } + }, + { + "message": "exports: include currency column", + "author": "r.haddad@example.com", + "minutes_before_alert": 24, + "files": { + "app/exports.py": [ + [ + "writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])", + "writer.writerow([\"id\", \"status\", \"total\", \"currency\", \"created_at\"])" + ], + [ + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])", + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"currency\"], o[\"created_at\"]])" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "m.okafor@example.com", + "minutes_before_alert": 9, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": "config_parse_errors" + } +} diff --git a/eval/cases/cfg_renamed_gateway_key_04.json b/eval/cases/cfg_renamed_gateway_key_04.json new file mode 100644 index 0000000..d36d49b --- /dev/null +++ b/eval/cases/cfg_renamed_gateway_key_04.json @@ -0,0 +1,54 @@ +{ + "id": "cfg_renamed_gateway_key_04", + "category": "config_type_error", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "KeyError: 'upstream_timeout_s'" + }, + "commits": [ + { + "message": "docs: describe exports module", + "author": "m.okafor@example.com", + "minutes_before_alert": 50, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + }, + { + "message": "gateway: raise worker pool to 48", + "author": "p.nair@example.com", + "minutes_before_alert": 35, + "files": { + "app/gateway.py": [ + [ + "max_workers=32", + "max_workers=48" + ] + ] + } + }, + { + "message": "config: drop unit suffixes from gateway keys", + "author": "a.chen@example.com", + "minutes_before_alert": 14, + "files": { + "config.json": [ + [ + "\"gateway\": {\"upstream_timeout_s\": 15}", + "\"gateway\": {\"upstream_timeout\": 15}" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": "config_parse_errors" + } +} diff --git a/eval/cases/db_pool_leak_01.json b/eval/cases/db_pool_leak_01.json new file mode 100644 index 0000000..2271b44 --- /dev/null +++ b/eval/cases/db_pool_leak_01.json @@ -0,0 +1,67 @@ +{ + "id": "db_pool_leak_01", + "category": "db_pool_exhaustion", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "psycopg2.pool.PoolError: connection pool exhausted" + }, + "commits": [ + { + "message": "db: simplify fetch_all", + "author": "p.nair@example.com", + "minutes_before_alert": 58, + "files": { + "app/db.py": [ + [ + " conn = pool.getconn()\n try:\n with conn.cursor() as cur:\n cur.execute(sql, params)\n return cur.fetchall()\n finally:\n pool.putconn(conn)\n", + " conn = pool.getconn()\n with conn.cursor() as cur:\n cur.execute(sql, params)\n rows = cur.fetchall()\n pool.putconn(conn)\n return rows\n" + ] + ] + } + }, + { + "message": "config: lower db pool size to fit pgbouncer limits", + "author": "l.becker@example.com", + "minutes_before_alert": 44, + "files": { + "config.json": [ + [ + "\"pool_size\": 10", + "\"pool_size\": 8" + ] + ] + } + }, + { + "message": "orders: name the stock shortfall list", + "author": "j.silva@example.com", + "minutes_before_alert": 29, + "files": { + "app/handlers/orders.py": [ + [ + " short = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if short:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": short}}\n", + " shortfall = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if shortfall:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": shortfall}}\n" + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "r.haddad@example.com", + "minutes_before_alert": 14, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": "db_failures" + } +} diff --git a/eval/cases/deps_psycopg3_01.json b/eval/cases/deps_psycopg3_01.json new file mode 100644 index 0000000..ae76e3a --- /dev/null +++ b/eval/cases/deps_psycopg3_01.json @@ -0,0 +1,67 @@ +{ + "id": "deps_psycopg3_01", + "category": "dependency_version", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_503_Service_Unavailable", + "error_signature": "ModuleNotFoundError: No module named 'psycopg2'" + }, + "commits": [ + { + "message": "deps: bump PyJWT to 2.9.0", + "author": "p.nair@example.com", + "minutes_before_alert": 55, + "files": { + "requirements.txt": [ + [ + "PyJWT==2.8.0", + "PyJWT==2.9.0" + ] + ] + } + }, + { + "message": "db: tag connections with application_name", + "author": "l.becker@example.com", + "minutes_before_alert": 40, + "files": { + "app/db.py": [ + [ + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n", + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n application_name=\"orders-service\",\n" + ] + ] + } + }, + { + "message": "deps: move to psycopg 3", + "author": "j.silva@example.com", + "minutes_before_alert": 25, + "files": { + "requirements.txt": [ + [ + "psycopg2-binary==2.9.9", + "psycopg[binary]==3.2.1" + ] + ] + } + }, + { + "message": "cache: type hints", + "author": "a.chen@example.com", + "minutes_before_alert": 10, + "files": { + "app/cache.py": [ + [ + " def get(self, key):\n", + " def get(self, key: str):\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": "dependency_regressions" + } +} diff --git a/eval/cases/deps_redis_legacy_02.json b/eval/cases/deps_redis_legacy_02.json new file mode 100644 index 0000000..0f954b3 --- /dev/null +++ b/eval/cases/deps_redis_legacy_02.json @@ -0,0 +1,71 @@ +{ + "id": "deps_redis_legacy_02", + "category": "dependency_version", + "base": "orders_service", + "alert": { + "alert_name": "FulfilmentWorker_Crashloop", + "error_signature": "redis.exceptions.ResponseError: value is not an integer or out of range" + }, + "commits": [ + { + "message": "worker: shorter idle poll", + "author": "m.okafor@example.com", + "minutes_before_alert": 50, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + }, + { + "message": "deps: pin redis to 2.10.6 to match fulfilment host image", + "author": "r.haddad@example.com", + "minutes_before_alert": 36, + "files": { + "requirements.txt": [ + [ + "redis==5.0.8", + "redis==2.10.6" + ] + ] + } + }, + { + "message": "deps: bump PyJWT to 2.9.0", + "author": "p.nair@example.com", + "minutes_before_alert": 22, + "files": { + "requirements.txt": [ + [ + "PyJWT==2.8.0", + "PyJWT==2.9.0" + ] + ] + } + }, + { + "message": "exports: include currency column", + "author": "l.becker@example.com", + "minutes_before_alert": 8, + "files": { + "app/exports.py": [ + [ + "writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])", + "writer.writerow([\"id\", \"status\", \"total\", \"currency\", \"created_at\"])" + ], + [ + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])", + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"currency\"], o[\"created_at\"]])" + ] + ] + } + } + ], + "label": { + "culprit_index": 1, + "expected_runbook_id": "dependency_regressions" + } +} diff --git a/eval/cases/fd_export_writer_01.json b/eval/cases/fd_export_writer_01.json new file mode 100644 index 0000000..8b4986a --- /dev/null +++ b/eval/cases/fd_export_writer_01.json @@ -0,0 +1,54 @@ +{ + "id": "fd_export_writer_01", + "category": "file_handle_leak", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "OSError: [Errno 24] Too many open files: '/mnt/shared/orders/exports/orders_20260901T114702.csv'" + }, + "commits": [ + { + "message": "config: move exports to shared volume", + "author": "r.haddad@example.com", + "minutes_before_alert": 41, + "files": { + "config.json": [ + [ + "\"/var/lib/orders/exports\"", + "\"/mnt/shared/orders/exports\"" + ] + ] + } + }, + { + "message": "worker: shorter idle poll", + "author": "m.okafor@example.com", + "minutes_before_alert": 26, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + }, + { + "message": "exports: stream rows through a reusable writer", + "author": "a.chen@example.com", + "minutes_before_alert": 9, + "files": { + "app/exports.py": [ + [ + " with open(path, \"w\", newline=\"\") as fh:\n writer = csv.writer(fh)\n writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])\n for o in orders:\n writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])\n return path\n", + " writer = _open_writer(path)\n writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])\n for o in orders:\n writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])\n return path\n\n\ndef _open_writer(path: str):\n return csv.writer(open(path, \"w\", newline=\"\"))\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": "file_handle_exhaustion" + } +} diff --git a/eval/cases/mem_worker_seen_ids_01.json b/eval/cases/mem_worker_seen_ids_01.json new file mode 100644 index 0000000..bfc5f3c --- /dev/null +++ b/eval/cases/mem_worker_seen_ids_01.json @@ -0,0 +1,79 @@ +{ + "id": "mem_worker_seen_ids_01", + "category": "memory_growth", + "base": "orders_service", + "alert": { + "alert_name": "FulfilmentWorker_Memory_High", + "error_signature": "fulfilment-worker OOMKilled (exit code 137): rss 3.8GiB of 4GiB limit, rss growing linearly with jobs processed" + }, + "commits": [ + { + "message": "config: raise stock cache size", + "author": "m.okafor@example.com", + "minutes_before_alert": 54, + "files": { + "config.json": [ + [ + "\"max_entries\": 5000", + "\"max_entries\": 8000" + ] + ] + } + }, + { + "message": "worker: skip duplicate fulfilment jobs", + "author": "a.chen@example.com", + "minutes_before_alert": 39, + "files": { + "app/worker.py": [ + [ + "QUEUE = \"fulfilment\"\n", + "_seen_jobs: list[str] = []\nQUEUE = \"fulfilment\"\n" + ], + [ + " job = json.loads(raw)\n handle(job)\n", + " job = json.loads(raw)\n if job[\"id\"] not in _seen_jobs:\n handle(job)\n _seen_jobs.append(job[\"id\"])\n" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "l.becker@example.com", + "minutes_before_alert": 24, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + }, + { + "message": "cache: track hit and miss counts", + "author": "j.silva@example.com", + "minutes_before_alert": 11, + "files": { + "app/cache.py": [ + [ + " self._data: dict = {}\n", + " self._data: dict = {}\n self.hits = 0\n self.misses = 0\n" + ], + [ + " entry = self._data.get(key)\n if entry is None:\n return None\n", + " entry = self._data.get(key)\n if entry is None:\n self.misses += 1\n return None\n" + ], + [ + " del self._data[key]\n return None\n return value\n", + " del self._data[key]\n self.misses += 1\n return None\n self.hits += 1\n return value\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 1, + "expected_runbook_id": "memory_leaks" + } +} diff --git a/eval/cases/null_guest_email_01.json b/eval/cases/null_guest_email_01.json new file mode 100644 index 0000000..a75a86f --- /dev/null +++ b/eval/cases/null_guest_email_01.json @@ -0,0 +1,71 @@ +{ + "id": "null_guest_email_01", + "category": "null_or_missing_field", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "AttributeError: 'NoneType' object has no attribute 'split'" + }, + "commits": [ + { + "message": "serializers: thousands separator in money", + "author": "a.chen@example.com", + "minutes_before_alert": 56, + "files": { + "app/serializers.py": [ + [ + " return str((Decimal(cents) / 100).quantize(Decimal(\"0.01\")))\n", + " return f\"{(Decimal(cents) / 100).quantize(Decimal('0.01')):,}\"\n" + ] + ] + } + }, + { + "message": "orders: expose customer email domain for B2B reporting", + "author": "p.nair@example.com", + "minutes_before_alert": 39, + "files": { + "app/serializers.py": [ + [ + " \"email\": row[\"customer_email\"],\n", + " \"email\": row[\"customer_email\"],\n \"email_domain\": row[\"customer_email\"].split(\"@\")[1].lower(),\n" + ] + ] + } + }, + { + "message": "exports: include currency column", + "author": "l.becker@example.com", + "minutes_before_alert": 23, + "files": { + "app/exports.py": [ + [ + "writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])", + "writer.writerow([\"id\", \"status\", \"total\", \"currency\", \"created_at\"])" + ], + [ + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])", + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"currency\"], o[\"created_at\"]])" + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "r.haddad@example.com", + "minutes_before_alert": 10, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 1, + "expected_runbook_id": "null_field_handling" + } +} diff --git a/eval/cases/null_since_filter_03.json b/eval/cases/null_since_filter_03.json new file mode 100644 index 0000000..c5ca672 --- /dev/null +++ b/eval/cases/null_since_filter_03.json @@ -0,0 +1,80 @@ +{ + "id": "null_since_filter_03", + "category": "null_or_missing_field", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "TypeError: '>=' not supported between instances of 'datetime.datetime' and 'NoneType'" + }, + "commits": [ + { + "message": "docs: describe exports module", + "author": "l.becker@example.com", + "minutes_before_alert": 58, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + }, + { + "message": "serializers: accept Z suffix in since", + "author": "r.haddad@example.com", + "minutes_before_alert": 45, + "files": { + "app/serializers.py": [ + [ + " return datetime.fromisoformat(value) if value else None\n", + " return datetime.fromisoformat(value.replace(\"Z\", \"+00:00\")) if value else None\n" + ] + ] + } + }, + { + "message": "pagination: document page_size bounds", + "author": "a.chen@example.com", + "minutes_before_alert": 32, + "files": { + "app/pagination.py": [ + [ + "def clamp_page_size(requested: int | None) -> int:\n", + "def clamp_page_size(requested: int | None) -> int:\n \"\"\"Requested size clamped to [1, api.max_page_size]; api.page_size when omitted.\"\"\"\n" + ] + ] + } + }, + { + "message": "orders: filter order history by since", + "author": "p.nair@example.com", + "minutes_before_alert": 19, + "files": { + "app/handlers/orders.py": [ + [ + "def list_orders(customer_id: str, page: int = 1, page_size: int | None = None) -> dict:\n rows = db.fetch_all(_LIST_SQL, (customer_id,))\n", + "def list_orders(\n customer_id: str, page: int = 1, page_size: int | None = None, since: str | None = None\n) -> dict:\n rows = db.fetch_all(_LIST_SQL, (customer_id,))\n rows = [r for r in rows if r[\"created_at\"] >= serializers.parse_since(since)]\n" + ] + ] + } + }, + { + "message": "worker: shorter idle poll", + "author": "m.okafor@example.com", + "minutes_before_alert": 7, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + } + ], + "label": { + "culprit_index": 3, + "expected_runbook_id": "null_field_handling" + } +} diff --git a/eval/cases/null_stock_payload_02.json b/eval/cases/null_stock_payload_02.json new file mode 100644 index 0000000..a325fd2 --- /dev/null +++ b/eval/cases/null_stock_payload_02.json @@ -0,0 +1,66 @@ +{ + "id": "null_stock_payload_02", + "category": "null_or_missing_field", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "TypeError: '<' not supported between instances of 'NoneType' and 'int'" + }, + "commits": [ + { + "message": "inventory: parse stock payload with .get", + "author": "j.silva@example.com", + "minutes_before_alert": 49, + "files": { + "app/clients/inventory.py": [ + [ + " qty = resp.json()[\"available\"]\n", + " qty = resp.json().get(\"available\")\n" + ] + ] + } + }, + { + "message": "orders: log order totals", + "author": "a.chen@example.com", + "minutes_before_alert": 34, + "files": { + "app/handlers/orders.py": [ + [ + "from app import db, serializers\n", + "import logging\n\nfrom app import db, serializers\n" + ], + [ + "_ORDER_SQL =", + "log = logging.getLogger(__name__)\n\n_ORDER_SQL =" + ], + [ + " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n", + " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n log.info(\"order total computed\", extra={\"customer_id\": customer_id, \"total_cents\": total})\n" + ] + ] + } + }, + { + "message": "payments: refund takes optional amount", + "author": "m.okafor@example.com", + "minutes_before_alert": 15, + "files": { + "app/clients/payments.py": [ + [ + "def refund(charge_id: str) -> dict:\n", + "def refund(charge_id: str, amount_cents: int | None = None) -> dict:\n" + ], + [ + " f\"{settings.get('payments', 'base_url')}/v1/charges/{charge_id}/refund\",\n", + " f\"{settings.get('payments', 'base_url')}/v1/charges/{charge_id}/refund\",\n json={\"amount_cents\": amount_cents} if amount_cents else None,\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": "null_field_handling" + } +} diff --git a/eval/cases/page_newest_empty_01.json b/eval/cases/page_newest_empty_01.json new file mode 100644 index 0000000..d5e3e28 --- /dev/null +++ b/eval/cases/page_newest_empty_01.json @@ -0,0 +1,54 @@ +{ + "id": "page_newest_empty_01", + "category": "pagination_boundary", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "IndexError: list index out of range" + }, + "commits": [ + { + "message": "pagination: include total count", + "author": "m.okafor@example.com", + "minutes_before_alert": 51, + "files": { + "app/pagination.py": [ + [ + " return {\"items\": items[start:end], \"page\": page, \"has_more\": end < len(items)}\n", + " return {\"items\": items[start:end], \"page\": page, \"has_more\": end < len(items), \"total\": len(items)}\n" + ] + ] + } + }, + { + "message": "serializers: accept Z suffix in since", + "author": "l.becker@example.com", + "minutes_before_alert": 30, + "files": { + "app/serializers.py": [ + [ + " return datetime.fromisoformat(value) if value else None\n", + " return datetime.fromisoformat(value.replace(\"Z\", \"+00:00\")) if value else None\n" + ] + ] + } + }, + { + "message": "orders: include newest order timestamp in list response", + "author": "p.nair@example.com", + "minutes_before_alert": 12, + "files": { + "app/handlers/orders.py": [ + [ + " return {\"status\": 200, \"body\": body}\n\n\ndef create_order", + " body[\"newest_at\"] = rows[0][\"created_at\"].isoformat()\n return {\"status\": 200, \"body\": body}\n\n\ndef create_order" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": "pagination_errors" + } +} diff --git a/eval/cases/page_size_zero_02.json b/eval/cases/page_size_zero_02.json new file mode 100644 index 0000000..60379d3 --- /dev/null +++ b/eval/cases/page_size_zero_02.json @@ -0,0 +1,67 @@ +{ + "id": "page_size_zero_02", + "category": "pagination_boundary", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "ZeroDivisionError: integer division or modulo by zero" + }, + "commits": [ + { + "message": "pagination: let clients request exact page sizes", + "author": "a.chen@example.com", + "minutes_before_alert": 48, + "files": { + "app/pagination.py": [ + [ + " return max(1, min(requested, settings.get(\"api\", \"max_page_size\")))\n", + " return min(requested, settings.get(\"api\", \"max_page_size\"))\n" + ] + ] + } + }, + { + "message": "pagination: return total page count", + "author": "r.haddad@example.com", + "minutes_before_alert": 33, + "files": { + "app/pagination.py": [ + [ + " return {\"items\": items[start:end], \"page\": page, \"has_more\": end < len(items)}\n", + " total_pages = -(-len(items) // page_size)\n return {\"items\": items[start:end], \"page\": page, \"has_more\": end < len(items), \"total_pages\": total_pages}\n" + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "m.okafor@example.com", + "minutes_before_alert": 17, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + }, + { + "message": "worker: shorter idle poll", + "author": "j.silva@example.com", + "minutes_before_alert": 4, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": "pagination_errors" + } +} diff --git a/eval/cases/payments_form_body_07.json b/eval/cases/payments_form_body_07.json new file mode 100644 index 0000000..26b3b7d --- /dev/null +++ b/eval/cases/payments_form_body_07.json @@ -0,0 +1,80 @@ +{ + "id": "payments_form_body_07", + "category": "http_content_type", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "httpx.HTTPStatusError: Client error '415 Unsupported Media Type' for url 'https://payments.internal/v1/charges'" + }, + "commits": [ + { + "message": "config: give payments more headroom", + "author": "j.silva@example.com", + "minutes_before_alert": 57, + "files": { + "config.json": [ + [ + "\"timeout_s\": 8, \"retries\": 2", + "\"timeout_s\": 10, \"retries\": 2" + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "l.becker@example.com", + "minutes_before_alert": 45, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + }, + { + "message": "orders: name the stock shortfall list", + "author": "r.haddad@example.com", + "minutes_before_alert": 32, + "files": { + "app/handlers/orders.py": [ + [ + " short = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if short:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": short}}\n", + " shortfall = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if shortfall:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": shortfall}}\n" + ] + ] + } + }, + { + "message": "payments: send charges as form fields for gateway v1 compatibility", + "author": "p.nair@example.com", + "minutes_before_alert": 18, + "files": { + "app/clients/payments.py": [ + [ + " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n", + " data={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n" + ] + ] + } + }, + { + "message": "cache: type hints", + "author": "a.chen@example.com", + "minutes_before_alert": 5, + "files": { + "app/cache.py": [ + [ + " def get(self, key):\n", + " def get(self, key: str):\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 3, + "expected_runbook_id": null + } +} diff --git a/eval/cases/queue_args_swapped_03.json b/eval/cases/queue_args_swapped_03.json new file mode 100644 index 0000000..5a048df --- /dev/null +++ b/eval/cases/queue_args_swapped_03.json @@ -0,0 +1,58 @@ +{ + "id": "queue_args_swapped_03", + "category": "queue_stall", + "base": "orders_service", + "alert": { + "alert_name": "Queue_Depth_Critical", + "error_signature": "fulfilment queue depth 18234 (threshold 5000), consumer throughput 0 jobs/min, worker pods healthy" + }, + "commits": [ + { + "message": "worker: import inventory at module level", + "author": "p.nair@example.com", + "minutes_before_alert": 49, + "files": { + "app/worker.py": [ + [ + "def handle(job: dict):\n from app.clients import inventory\n\n inventory.get_stock(job[\"sku\"])\n", + "def handle(job: dict):\n inventory.get_stock(job[\"sku\"])\n" + ], + [ + "import redis\n", + "import redis\n\nfrom app.clients import inventory\n" + ] + ] + } + }, + { + "message": "deps: bump PyJWT to 2.9.0", + "author": "a.chen@example.com", + "minutes_before_alert": 31, + "files": { + "requirements.txt": [ + [ + "PyJWT==2.8.0", + "PyJWT==2.9.0" + ] + ] + } + }, + { + "message": "worker: pass queue names by keyword", + "author": "l.becker@example.com", + "minutes_before_alert": 15, + "files": { + "app/worker.py": [ + [ + " raw = _r.brpoplpush(QUEUE, PROCESSING, timeout=5)\n", + " raw = _r.brpoplpush(src=PROCESSING, dst=QUEUE, timeout=5)\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": null + } +} diff --git a/eval/cases/rl_inventory_uncached_01.json b/eval/cases/rl_inventory_uncached_01.json new file mode 100644 index 0000000..1cd94cc --- /dev/null +++ b/eval/cases/rl_inventory_uncached_01.json @@ -0,0 +1,67 @@ +{ + "id": "rl_inventory_uncached_01", + "category": "rate_limit_429", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "httpx.HTTPStatusError: Client error '429 Too Many Requests' for url 'https://inventory.internal/v2/stock/SKU-88412'" + }, + "commits": [ + { + "message": "config: shorten stock cache ttl", + "author": "a.chen@example.com", + "minutes_before_alert": 57, + "files": { + "config.json": [ + [ + "\"ttl_s\": 300", + "\"ttl_s\": 240" + ] + ] + } + }, + { + "message": "pagination: document page_size bounds", + "author": "l.becker@example.com", + "minutes_before_alert": 43, + "files": { + "app/pagination.py": [ + [ + "def clamp_page_size(requested: int | None) -> int:\n", + "def clamp_page_size(requested: int | None) -> int:\n \"\"\"Requested size clamped to [1, api.max_page_size]; api.page_size when omitted.\"\"\"\n" + ] + ] + } + }, + { + "message": "inventory: read fresh stock for multi-item orders", + "author": "p.nair@example.com", + "minutes_before_alert": 28, + "files": { + "app/clients/inventory.py": [ + [ + "def get_stock_many(skus: list[str]) -> dict[str, int]:\n return {sku: get_stock(sku) for sku in skus}\n", + "def get_stock_many(skus: list[str]) -> dict[str, int]:\n out = {}\n for sku in skus:\n resp = httpx.get(\n f\"{settings.get('inventory', 'base_url')}/v2/stock/{sku}\",\n timeout=settings.get(\"inventory\", \"timeout_s\"),\n )\n resp.raise_for_status()\n out[sku] = resp.json()[\"available\"]\n return out\n" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "m.okafor@example.com", + "minutes_before_alert": 13, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": "rate_limiting" + } +} diff --git a/eval/cases/rl_payment_retries_02.json b/eval/cases/rl_payment_retries_02.json new file mode 100644 index 0000000..86ac38e --- /dev/null +++ b/eval/cases/rl_payment_retries_02.json @@ -0,0 +1,68 @@ +{ + "id": "rl_payment_retries_02", + "category": "rate_limit_429", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "httpx.HTTPStatusError: Client error '429 Too Many Requests' for url 'https://payments.internal/v1/charges'" + }, + "commits": [ + { + "message": "payments: retry charges on transient errors", + "author": "j.silva@example.com", + "minutes_before_alert": 46, + "files": { + "config.json": [ + [ + "\"timeout_s\": 8, \"retries\": 2", + "\"timeout_s\": 8, \"retries\": 5" + ] + ], + "app/clients/payments.py": [ + [ + "def charge(order_id: str, amount_cents: int, card_token: str) -> dict:\n resp = httpx.post(\n f\"{settings.get('payments', 'base_url')}/v1/charges\",\n json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n timeout=settings.get(\"payments\", \"timeout_s\"),\n )\n resp.raise_for_status()\n return resp.json()\n", + "def charge(order_id: str, amount_cents: int, card_token: str) -> dict:\n for _ in range(settings.get(\"payments\", \"retries\") + 1):\n resp = httpx.post(\n f\"{settings.get('payments', 'base_url')}/v1/charges\",\n json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n timeout=settings.get(\"payments\", \"timeout_s\"),\n )\n if resp.status_code < 400:\n return resp.json()\n resp.raise_for_status()\n return resp.json()\n" + ] + ] + } + }, + { + "message": "orders: log order totals", + "author": "a.chen@example.com", + "minutes_before_alert": 31, + "files": { + "app/handlers/orders.py": [ + [ + "from app import db, serializers\n", + "import logging\n\nfrom app import db, serializers\n" + ], + [ + "_ORDER_SQL =", + "log = logging.getLogger(__name__)\n\n_ORDER_SQL =" + ], + [ + " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n", + " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n log.info(\"order total computed\", extra={\"customer_id\": customer_id, \"total_cents\": total})\n" + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "r.haddad@example.com", + "minutes_before_alert": 15, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": "rate_limiting" + } +} diff --git a/eval/cases/ser_decimal_json_02.json b/eval/cases/ser_decimal_json_02.json new file mode 100644 index 0000000..fc80b6f --- /dev/null +++ b/eval/cases/ser_decimal_json_02.json @@ -0,0 +1,71 @@ +{ + "id": "ser_decimal_json_02", + "category": "json_serialization", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "TypeError: Object of type Decimal is not JSON serializable" + }, + "commits": [ + { + "message": "payments: send idempotency key with charges", + "author": "a.chen@example.com", + "minutes_before_alert": 52, + "files": { + "app/clients/payments.py": [ + [ + " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n", + " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n headers={\"Idempotency-Key\": f\"charge-{order_id}\"},\n" + ] + ] + } + }, + { + "message": "serializers: thousands separator in money", + "author": "l.becker@example.com", + "minutes_before_alert": 38, + "files": { + "app/serializers.py": [ + [ + " return str((Decimal(cents) / 100).quantize(Decimal(\"0.01\")))\n", + " return f\"{(Decimal(cents) / 100).quantize(Decimal('0.01')):,}\"\n" + ] + ] + } + }, + { + "message": "orders: compute totals in Decimal", + "author": "m.okafor@example.com", + "minutes_before_alert": 23, + "files": { + "app/handlers/orders.py": [ + [ + "from app import db, serializers\n", + "from decimal import Decimal\n\nfrom app import db, serializers\n" + ], + [ + " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n", + " total = sum(Decimal(line[\"qty\"]) * Decimal(line[\"unit_price_cents\"]) for line in lines)\n" + ] + ] + } + }, + { + "message": "worker: shorter idle poll", + "author": "r.haddad@example.com", + "minutes_before_alert": 9, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": null + } +} diff --git a/eval/cases/sku_lowercase_06.json b/eval/cases/sku_lowercase_06.json new file mode 100644 index 0000000..b525a1b --- /dev/null +++ b/eval/cases/sku_lowercase_06.json @@ -0,0 +1,62 @@ +{ + "id": "sku_lowercase_06", + "category": "identifier_format", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "httpx.HTTPStatusError: Client error '404 Not Found' for url 'https://inventory.internal/v2/stock/sku-10293'" + }, + "commits": [ + { + "message": "inventory: normalise SKU keys for cache hits", + "author": "m.okafor@example.com", + "minutes_before_alert": 44, + "files": { + "app/clients/inventory.py": [ + [ + "def get_stock(sku: str) -> int:\n", + "def get_stock(sku: str) -> int:\n sku = sku.strip().lower()\n" + ] + ] + } + }, + { + "message": "cache: track hit and miss counts", + "author": "a.chen@example.com", + "minutes_before_alert": 30, + "files": { + "app/cache.py": [ + [ + " self._data: dict = {}\n", + " self._data: dict = {}\n self.hits = 0\n self.misses = 0\n" + ], + [ + " entry = self._data.get(key)\n if entry is None:\n return None\n", + " entry = self._data.get(key)\n if entry is None:\n self.misses += 1\n return None\n" + ], + [ + " del self._data[key]\n return None\n return value\n", + " del self._data[key]\n self.misses += 1\n return None\n self.hits += 1\n return value\n" + ] + ] + } + }, + { + "message": "config: lower inventory timeout to 2s", + "author": "r.haddad@example.com", + "minutes_before_alert": 14, + "files": { + "config.json": [ + [ + "\"timeout_s\": 3}", + "\"timeout_s\": 2}" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": null + } +} diff --git a/eval/cases/sql_param_count_05.json b/eval/cases/sql_param_count_05.json new file mode 100644 index 0000000..8fd9df5 --- /dev/null +++ b/eval/cases/sql_param_count_05.json @@ -0,0 +1,67 @@ +{ + "id": "sql_param_count_05", + "category": "sql_param_mismatch", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "IndexError: tuple index out of range" + }, + "commits": [ + { + "message": "pagination: include total count", + "author": "l.becker@example.com", + "minutes_before_alert": 53, + "files": { + "app/pagination.py": [ + [ + " return {\"items\": items[start:end], \"page\": page, \"has_more\": end < len(items)}\n", + " return {\"items\": items[start:end], \"page\": page, \"has_more\": end < len(items), \"total\": len(items)}\n" + ] + ] + } + }, + { + "message": "db: tag connections with application_name", + "author": "r.haddad@example.com", + "minutes_before_alert": 37, + "files": { + "app/db.py": [ + [ + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n", + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n application_name=\"orders-service\",\n" + ] + ] + } + }, + { + "message": "serializers: accept Z suffix in since", + "author": "a.chen@example.com", + "minutes_before_alert": 22, + "files": { + "app/serializers.py": [ + [ + " return datetime.fromisoformat(value) if value else None\n", + " return datetime.fromisoformat(value.replace(\"Z\", \"+00:00\")) if value else None\n" + ] + ] + } + }, + { + "message": "orders: hide cancelled orders from history", + "author": "p.nair@example.com", + "minutes_before_alert": 6, + "files": { + "app/handlers/orders.py": [ + [ + "\"SELECT * FROM orders_view WHERE customer_id = %s ORDER BY", + "\"SELECT * FROM orders_view WHERE customer_id = %s AND status <> %s ORDER BY" + ] + ] + } + } + ], + "label": { + "culprit_index": 3, + "expected_runbook_id": null + } +} diff --git a/eval/cases/tmo_gateway_deadline_02.json b/eval/cases/tmo_gateway_deadline_02.json new file mode 100644 index 0000000..4712e65 --- /dev/null +++ b/eval/cases/tmo_gateway_deadline_02.json @@ -0,0 +1,67 @@ +{ + "id": "tmo_gateway_deadline_02", + "category": "latency_timeout", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_504_Gateway_Timeout", + "error_signature": "504 Gateway Timeout: POST /v1/orders exceeded upstream deadline (payments.charge p50 6.2s, unchanged)" + }, + "commits": [ + { + "message": "config: tighten edge deadline to meet p99 target", + "author": "p.nair@example.com", + "minutes_before_alert": 54, + "files": { + "config.json": [ + [ + "\"gateway\": {\"upstream_timeout_s\": 15}", + "\"gateway\": {\"upstream_timeout_s\": 5}" + ] + ] + } + }, + { + "message": "payments: send idempotency key with charges", + "author": "m.okafor@example.com", + "minutes_before_alert": 40, + "files": { + "app/clients/payments.py": [ + [ + " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n", + " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n headers={\"Idempotency-Key\": f\"charge-{order_id}\"},\n" + ] + ] + } + }, + { + "message": "orders: name the stock shortfall list", + "author": "a.chen@example.com", + "minutes_before_alert": 25, + "files": { + "app/handlers/orders.py": [ + [ + " short = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if short:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": short}}\n", + " shortfall = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if shortfall:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": shortfall}}\n" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "l.becker@example.com", + "minutes_before_alert": 11, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": "request_timeouts" + } +} diff --git a/eval/cases/tmo_inventory_units_01.json b/eval/cases/tmo_inventory_units_01.json new file mode 100644 index 0000000..337bd94 --- /dev/null +++ b/eval/cases/tmo_inventory_units_01.json @@ -0,0 +1,66 @@ +{ + "id": "tmo_inventory_units_01", + "category": "latency_timeout", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "httpx.ConnectTimeout: timed out" + }, + "commits": [ + { + "message": "config: give payments more headroom", + "author": "l.becker@example.com", + "minutes_before_alert": 50, + "files": { + "config.json": [ + [ + "\"timeout_s\": 8, \"retries\": 2", + "\"timeout_s\": 10, \"retries\": 2" + ] + ] + } + }, + { + "message": "inventory: hoist timeout lookup to module level", + "author": "j.silva@example.com", + "minutes_before_alert": 33, + "files": { + "app/clients/inventory.py": [ + [ + "_stock_cache = TTLCache(", + "_TIMEOUT = settings.get(\"inventory\", \"timeout_s\") / 1000 # httpx timeouts are in ms\n_stock_cache = TTLCache(" + ], + [ + " timeout=settings.get(\"inventory\", \"timeout_s\"),\n", + " timeout=_TIMEOUT,\n" + ] + ] + } + }, + { + "message": "worker: shorter idle poll", + "author": "r.haddad@example.com", + "minutes_before_alert": 20, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + }, + { + "message": "tests: cover page_bounds", + "author": "a.chen@example.com", + "minutes_before_alert": 6, + "files": { + "tests/test_pagination.py": "from app.pagination import page_bounds\n\n\ndef test_first_page_starts_at_zero():\n assert page_bounds(1, 50) == (0, 50)\n\n\ndef test_second_page():\n assert page_bounds(2, 50) == (50, 100)\n" + } + } + ], + "label": { + "culprit_index": 1, + "expected_runbook_id": "request_timeouts" + } +} diff --git a/eval/cases/tmo_list_live_stock_03.json b/eval/cases/tmo_list_live_stock_03.json new file mode 100644 index 0000000..2760e27 --- /dev/null +++ b/eval/cases/tmo_list_live_stock_03.json @@ -0,0 +1,54 @@ +{ + "id": "tmo_list_live_stock_03", + "category": "latency_timeout", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_504_Gateway_Timeout", + "error_signature": "504 Gateway Timeout: GET /v1/customers/{customer_id}/orders exceeded upstream deadline" + }, + "commits": [ + { + "message": "config: shorten stock cache ttl", + "author": "r.haddad@example.com", + "minutes_before_alert": 47, + "files": { + "config.json": [ + [ + "\"ttl_s\": 300", + "\"ttl_s\": 120" + ] + ] + } + }, + { + "message": "pagination: document page_size bounds", + "author": "j.silva@example.com", + "minutes_before_alert": 29, + "files": { + "app/pagination.py": [ + [ + "def clamp_page_size(requested: int | None) -> int:\n", + "def clamp_page_size(requested: int | None) -> int:\n \"\"\"Requested size clamped to [1, api.max_page_size]; api.page_size when omitted.\"\"\"\n" + ] + ] + } + }, + { + "message": "orders: show live availability in order history", + "author": "m.okafor@example.com", + "minutes_before_alert": 8, + "files": { + "app/handlers/orders.py": [ + [ + " rows = db.fetch_all(_LIST_SQL, (customer_id,))\n body = paginate([serializers.order_to_dict(r) for r in rows], page, clamp_page_size(page_size))\n", + " rows = db.fetch_all(_LIST_SQL, (customer_id,))\n orders = [serializers.order_to_dict(r) for r in rows]\n for order in orders:\n stock = inventory.get_stock_many([item[\"sku\"] for item in order[\"items\"]])\n for item in order[\"items\"]:\n item[\"available\"] = stock[item[\"sku\"]]\n body = paginate(orders, page, clamp_page_size(page_size))\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": "request_timeouts" + } +} diff --git a/eval/cases/tz_naive_since_01.json b/eval/cases/tz_naive_since_01.json new file mode 100644 index 0000000..7ce5902 --- /dev/null +++ b/eval/cases/tz_naive_since_01.json @@ -0,0 +1,54 @@ +{ + "id": "tz_naive_since_01", + "category": "timezone_mismatch", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "TypeError: can't compare offset-naive and offset-aware datetimes" + }, + "commits": [ + { + "message": "orders: filter order history by since", + "author": "j.silva@example.com", + "minutes_before_alert": 45, + "files": { + "app/handlers/orders.py": [ + [ + "def list_orders(customer_id: str, page: int = 1, page_size: int | None = None) -> dict:\n rows = db.fetch_all(_LIST_SQL, (customer_id,))\n", + "def list_orders(\n customer_id: str, page: int = 1, page_size: int | None = None, since: str | None = None\n) -> dict:\n rows = db.fetch_all(_LIST_SQL, (customer_id,))\n if since is not None:\n rows = [r for r in rows if r[\"created_at\"] >= serializers.parse_since(since)]\n" + ] + ] + } + }, + { + "message": "serializers: drop tz info from since for db comparisons", + "author": "r.haddad@example.com", + "minutes_before_alert": 30, + "files": { + "app/serializers.py": [ + [ + " return datetime.fromisoformat(value) if value else None\n", + " return datetime.fromisoformat(value).replace(tzinfo=None) if value else None\n" + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "p.nair@example.com", + "minutes_before_alert": 12, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 1, + "expected_runbook_id": null + } +} From 8f07d049ad6efeff29da17407556ed99a6ba1017 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 14:14:17 -0400 Subject: [PATCH 06/19] fix(eval): ignore untracked files in results git-sha dirty check Co-Authored-By: Claude Opus 5.5 --- eval/run_eval.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/eval/run_eval.py b/eval/run_eval.py index bb9d9fc..b8c05a8 100644 --- a/eval/run_eval.py +++ b/eval/run_eval.py @@ -89,7 +89,7 @@ def _git_sha() -> str: sha = subprocess.run( ["git", "rev-parse", "HEAD"], cwd=ROOT, capture_output=True, text=True ).stdout.strip() - dirty = subprocess.run(["git", "status", "--porcelain"], cwd=ROOT, capture_output=True, text=True).stdout + dirty = subprocess.run(["git", "status", "--porcelain", "--untracked-files=no"], cwd=ROOT, capture_output=True, text=True).stdout return sha + ("-dirty" if dirty.strip() else "") From 41c852719587280aed75ba53ffbf4a737123e3f1 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 14:14:29 -0400 Subject: [PATCH 07/19] style(eval): black-format run_eval Co-Authored-By: Claude Opus 5.5 --- eval/run_eval.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/eval/run_eval.py b/eval/run_eval.py index b8c05a8..c65779e 100644 --- a/eval/run_eval.py +++ b/eval/run_eval.py @@ -89,7 +89,9 @@ def _git_sha() -> str: sha = subprocess.run( ["git", "rev-parse", "HEAD"], cwd=ROOT, capture_output=True, text=True ).stdout.strip() - dirty = subprocess.run(["git", "status", "--porcelain", "--untracked-files=no"], cwd=ROOT, capture_output=True, text=True).stdout + dirty = subprocess.run( + ["git", "status", "--porcelain", "--untracked-files=no"], cwd=ROOT, capture_output=True, text=True + ).stdout return sha + ("-dirty" if dirty.strip() else "") From fd3fabab41c13b2abc301b160b8e4febe4700117 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 14:22:38 -0400 Subject: [PATCH 08/19] feat(eval): abort runs after 5 consecutive LLM errors; archive the 402-failed baseline Co-Authored-By: Claude Opus 5.5 --- eval/report.py | 2 + eval/results/aborted/README.md | 6 + eval/results/baseline_2026-10-04.json | 2441 +++++++++++++++++++++++++ eval/run_eval.py | 30 +- eval/tests/test_eval.py | 18 + 5 files changed, 2488 insertions(+), 9 deletions(-) create mode 100644 eval/results/aborted/README.md create mode 100644 eval/results/baseline_2026-10-04.json diff --git a/eval/report.py b/eval/report.py index bd50fda..425d2cc 100644 --- a/eval/report.py +++ b/eval/report.py @@ -114,6 +114,8 @@ def print_summary(name: str, res: dict): f"wrong {_fmt_conf(s['conf_wrong'])} (n={s['n_wrong']})" ) print(f" rank latency median {s['latency_median_s']}s (n={s['n']})") + if res.get("aborted"): + print(f" WARNING run ABORTED, not a valid measurement: {res['aborted']}") truncated = sorted({r["case_id"] for r in res["records"] if r["truncated_diffs"]}) if truncated: print(f" WARNING diffs hit the 3000-char truncation in: {', '.join(truncated)}") diff --git a/eval/results/aborted/README.md b/eval/results/aborted/README.md new file mode 100644 index 0000000..068a8ed --- /dev/null +++ b/eval/results/aborted/README.md @@ -0,0 +1,6 @@ +# Aborted runs + +Kept unedited for the record; never scored as measurements. + +- `baseline_2026-10-04_402.json`: OpenRouter returned `402 Payment Required` on 82/87 calls (account credit exhausted). + The 5 calls that succeeded all ranked the culprit #1. Predates the harness abort-on-consecutive-errors guard. diff --git a/eval/results/baseline_2026-10-04.json b/eval/results/baseline_2026-10-04.json new file mode 100644 index 0000000..b6b28ee --- /dev/null +++ b/eval/results/baseline_2026-10-04.json @@ -0,0 +1,2441 @@ +{ + "condition": "baseline", + "dry_run": false, + "eval_set_version": "v1", + "runbook_corpus_version": "v1", + "runbooks_ingested": 10, + "model": "anthropic/claude-sonnet-5", + "sentinel_git_sha": "41c852719587280aed75ba53ffbf4a737123e3f1", + "started_at": "2026-10-04T18:14:41.029496+00:00", + "trials": 3, + "n_cases": 29, + "records": [ + { + "trial": 0, + "case_id": "auth_algorithm_02", + "category": "auth_rejection", + "culprit_hash": "144f54b", + "expected_runbook_id": "auth_failures", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.305 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.24 + }, + { + "id": "null_field_handling", + "similarity_score": 0.216 + } + ], + "ranking": [ + { + "commit_hash": "144f54b", + "confidence_score": 0.97 + }, + { + "commit_hash": "86e027a", + "confidence_score": 0.05 + }, + { + "commit_hash": "191abd1", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.42 + }, + { + "trial": 0, + "case_id": "auth_audience_rename_01", + "category": "auth_rejection", + "culprit_hash": "9bec53f", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.264 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.135 + }, + { + "id": "rate_limiting", + "similarity_score": 0.106 + } + ], + "ranking": [ + { + "commit_hash": "9bec53f", + "confidence_score": 0.97 + }, + { + "commit_hash": "212b6f9", + "confidence_score": 0.05 + }, + { + "commit_hash": "60a1b2e", + "confidence_score": 0.01 + }, + { + "commit_hash": "41b03cf", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.95 + }, + { + "trial": 0, + "case_id": "auth_leeway_zero_03", + "category": "auth_rejection", + "culprit_hash": "3cd3de4", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.383 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.209 + }, + { + "id": "null_field_handling", + "similarity_score": 0.147 + } + ], + "ranking": [ + { + "commit_hash": "3cd3de4", + "confidence_score": 0.95 + }, + { + "commit_hash": "1c66960", + "confidence_score": 0.1 + }, + { + "commit_hash": "2f58cd9", + "confidence_score": 0.05 + }, + { + "commit_hash": "fee17ca", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.45 + }, + { + "trial": 0, + "case_id": "cache_recursion_04", + "category": "infinite_recursion", + "culprit_hash": "1b419cd", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.69 + }, + { + "trial": 0, + "case_id": "cfg_cache_size_string_01", + "category": "config_type_error", + "culprit_hash": "5ab5431", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.45 + }, + { + "trial": 0, + "case_id": "cfg_env_override_string_03", + "category": "config_type_error", + "culprit_hash": "2239586", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.234 + }, + { + "id": "pagination_errors", + "similarity_score": 0.187 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.11 + }, + { + "trial": 0, + "case_id": "cfg_json_trailing_comma_02", + "category": "config_type_error", + "culprit_hash": "46e2927", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.262 + }, + { + "id": "null_field_handling", + "similarity_score": 0.113 + }, + { + "id": "auth_failures", + "similarity_score": 0.067 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.66 + }, + { + "trial": 0, + "case_id": "cfg_renamed_gateway_key_04", + "category": "config_type_error", + "culprit_hash": "9f89fe3", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.548 + }, + { + "id": "memory_leaks", + "similarity_score": 0.384 + }, + { + "id": "db_failures", + "similarity_score": 0.347 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.33 + }, + { + "trial": 0, + "case_id": "db_pool_leak_01", + "category": "db_pool_exhaustion", + "culprit_hash": "54c490d", + "expected_runbook_id": "db_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 0.98 + }, + { + "trial": 0, + "case_id": "deps_psycopg3_01", + "category": "dependency_version", + "culprit_hash": "49d9702", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.299 + }, + { + "id": "db_failures", + "similarity_score": 0.288 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.192 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.09 + }, + { + "trial": 0, + "case_id": "deps_redis_legacy_02", + "category": "dependency_version", + "culprit_hash": "c673656", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.52 + }, + { + "trial": 0, + "case_id": "fd_export_writer_01", + "category": "file_handle_leak", + "culprit_hash": "03bf82d", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.456 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.341 + }, + { + "id": "memory_leaks", + "similarity_score": 0.294 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.14 + }, + { + "trial": 0, + "case_id": "mem_worker_seen_ids_01", + "category": "memory_growth", + "culprit_hash": "6064408", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.462 + }, + { + "id": "rate_limiting", + "similarity_score": 0.449 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.433 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.48 + }, + { + "trial": 0, + "case_id": "null_guest_email_01", + "category": "null_or_missing_field", + "culprit_hash": "926570f", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.236 + }, + { + "id": "null_field_handling", + "similarity_score": 0.175 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.25 + }, + { + "trial": 0, + "case_id": "null_since_filter_03", + "category": "null_or_missing_field", + "culprit_hash": "f6961f1", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.232 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.175 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.41 + }, + { + "trial": 0, + "case_id": "null_stock_payload_02", + "category": "null_or_missing_field", + "culprit_hash": "23fce9a", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.271 + }, + { + "id": "pagination_errors", + "similarity_score": 0.226 + }, + { + "id": "null_field_handling", + "similarity_score": 0.21 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.22 + }, + { + "trial": 0, + "case_id": "page_newest_empty_01", + "category": "pagination_boundary", + "culprit_hash": "bc477e7", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.375 + }, + { + "id": "memory_leaks", + "similarity_score": 0.24 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.223 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.67 + }, + { + "trial": 0, + "case_id": "page_size_zero_02", + "category": "pagination_boundary", + "culprit_hash": "ff6ef3c", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.64 + }, + { + "trial": 0, + "case_id": "payments_form_body_07", + "category": "http_content_type", + "culprit_hash": "60716f4", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 2.31 + }, + { + "trial": 0, + "case_id": "queue_args_swapped_03", + "category": "queue_stall", + "culprit_hash": "38cf2c1", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.364 + }, + { + "id": "rate_limiting", + "similarity_score": 0.326 + }, + { + "id": "db_failures", + "similarity_score": 0.282 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.28 + }, + { + "trial": 0, + "case_id": "rl_inventory_uncached_01", + "category": "rate_limit_429", + "culprit_hash": "5acb3ab", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.36 + }, + { + "id": "rate_limiting", + "similarity_score": 0.351 + }, + { + "id": "null_field_handling", + "similarity_score": 0.278 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.31 + }, + { + "trial": 0, + "case_id": "rl_payment_retries_02", + "category": "rate_limit_429", + "culprit_hash": "bfb3bde", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.342 + }, + { + "id": "rate_limiting", + "similarity_score": 0.32 + }, + { + "id": "request_timeouts", + "similarity_score": 0.284 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.5 + }, + { + "trial": 0, + "case_id": "ser_decimal_json_02", + "category": "json_serialization", + "culprit_hash": "5203835", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.28 + }, + { + "trial": 0, + "case_id": "sku_lowercase_06", + "category": "identifier_format", + "culprit_hash": "b8f58f4", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.25 + }, + { + "trial": 0, + "case_id": "sql_param_count_05", + "category": "sql_param_mismatch", + "culprit_hash": "b2d1a3c", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.39 + }, + { + "trial": 0, + "case_id": "tmo_gateway_deadline_02", + "category": "latency_timeout", + "culprit_hash": "5039371", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.507 + }, + { + "id": "rate_limiting", + "similarity_score": 0.363 + }, + { + "id": "db_failures", + "similarity_score": 0.353 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.62 + }, + { + "trial": 0, + "case_id": "tmo_inventory_units_01", + "category": "latency_timeout", + "culprit_hash": "fdf72fb", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.404 + }, + { + "id": "auth_failures", + "similarity_score": 0.334 + }, + { + "id": "rate_limiting", + "similarity_score": 0.256 + } + ], + "ranking": [ + { + "commit_hash": "fdf72fb", + "confidence_score": 0.95 + }, + { + "commit_hash": "dd3b18a", + "confidence_score": 0.05 + }, + { + "commit_hash": "b756e42", + "confidence_score": 0.02 + }, + { + "commit_hash": "5a40192", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.44 + }, + { + "trial": 0, + "case_id": "tmo_list_live_stock_03", + "category": "latency_timeout", + "culprit_hash": "fa0acb3", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.566 + }, + { + "id": "auth_failures", + "similarity_score": 0.41 + }, + { + "id": "rate_limiting", + "similarity_score": 0.399 + } + ], + "ranking": [ + { + "commit_hash": "fa0acb3", + "confidence_score": 0.92 + }, + { + "commit_hash": "b9c96f5", + "confidence_score": 0.35 + }, + { + "commit_hash": "2f789c7", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.22 + }, + { + "trial": 0, + "case_id": "tz_naive_since_01", + "category": "timezone_mismatch", + "culprit_hash": "d010a5b", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.14 + }, + { + "trial": 1, + "case_id": "auth_algorithm_02", + "category": "auth_rejection", + "culprit_hash": "144f54b", + "expected_runbook_id": "auth_failures", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.305 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.24 + }, + { + "id": "null_field_handling", + "similarity_score": 0.216 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.16 + }, + { + "trial": 1, + "case_id": "auth_audience_rename_01", + "category": "auth_rejection", + "culprit_hash": "9bec53f", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.264 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.135 + }, + { + "id": "rate_limiting", + "similarity_score": 0.106 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.23 + }, + { + "trial": 1, + "case_id": "auth_leeway_zero_03", + "category": "auth_rejection", + "culprit_hash": "3cd3de4", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.383 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.209 + }, + { + "id": "null_field_handling", + "similarity_score": 0.147 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.23 + }, + { + "trial": 1, + "case_id": "cache_recursion_04", + "category": "infinite_recursion", + "culprit_hash": "1b419cd", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.2 + }, + { + "trial": 1, + "case_id": "cfg_cache_size_string_01", + "category": "config_type_error", + "culprit_hash": "5ab5431", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.45 + }, + { + "trial": 1, + "case_id": "cfg_env_override_string_03", + "category": "config_type_error", + "culprit_hash": "2239586", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.234 + }, + { + "id": "pagination_errors", + "similarity_score": 0.187 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.19 + }, + { + "trial": 1, + "case_id": "cfg_json_trailing_comma_02", + "category": "config_type_error", + "culprit_hash": "46e2927", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.262 + }, + { + "id": "null_field_handling", + "similarity_score": 0.113 + }, + { + "id": "auth_failures", + "similarity_score": 0.067 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.2 + }, + { + "trial": 1, + "case_id": "cfg_renamed_gateway_key_04", + "category": "config_type_error", + "culprit_hash": "9f89fe3", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.548 + }, + { + "id": "memory_leaks", + "similarity_score": 0.384 + }, + { + "id": "db_failures", + "similarity_score": 0.347 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.28 + }, + { + "trial": 1, + "case_id": "db_pool_leak_01", + "category": "db_pool_exhaustion", + "culprit_hash": "54c490d", + "expected_runbook_id": "db_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.14 + }, + { + "trial": 1, + "case_id": "deps_psycopg3_01", + "category": "dependency_version", + "culprit_hash": "49d9702", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.299 + }, + { + "id": "db_failures", + "similarity_score": 0.288 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.192 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.39 + }, + { + "trial": 1, + "case_id": "deps_redis_legacy_02", + "category": "dependency_version", + "culprit_hash": "c673656", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.17 + }, + { + "trial": 1, + "case_id": "fd_export_writer_01", + "category": "file_handle_leak", + "culprit_hash": "03bf82d", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.456 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.341 + }, + { + "id": "memory_leaks", + "similarity_score": 0.294 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.24 + }, + { + "trial": 1, + "case_id": "mem_worker_seen_ids_01", + "category": "memory_growth", + "culprit_hash": "6064408", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.462 + }, + { + "id": "rate_limiting", + "similarity_score": 0.449 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.433 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.12 + }, + { + "trial": 1, + "case_id": "null_guest_email_01", + "category": "null_or_missing_field", + "culprit_hash": "926570f", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.236 + }, + { + "id": "null_field_handling", + "similarity_score": 0.175 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.16 + }, + { + "trial": 1, + "case_id": "null_since_filter_03", + "category": "null_or_missing_field", + "culprit_hash": "f6961f1", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.232 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.175 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.09 + }, + { + "trial": 1, + "case_id": "null_stock_payload_02", + "category": "null_or_missing_field", + "culprit_hash": "23fce9a", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.271 + }, + { + "id": "pagination_errors", + "similarity_score": 0.226 + }, + { + "id": "null_field_handling", + "similarity_score": 0.21 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.47 + }, + { + "trial": 1, + "case_id": "page_newest_empty_01", + "category": "pagination_boundary", + "culprit_hash": "bc477e7", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.375 + }, + { + "id": "memory_leaks", + "similarity_score": 0.24 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.223 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.45 + }, + { + "trial": 1, + "case_id": "page_size_zero_02", + "category": "pagination_boundary", + "culprit_hash": "ff6ef3c", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.17 + }, + { + "trial": 1, + "case_id": "payments_form_body_07", + "category": "http_content_type", + "culprit_hash": "60716f4", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.42 + }, + { + "trial": 1, + "case_id": "queue_args_swapped_03", + "category": "queue_stall", + "culprit_hash": "38cf2c1", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.364 + }, + { + "id": "rate_limiting", + "similarity_score": 0.326 + }, + { + "id": "db_failures", + "similarity_score": 0.282 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.22 + }, + { + "trial": 1, + "case_id": "rl_inventory_uncached_01", + "category": "rate_limit_429", + "culprit_hash": "5acb3ab", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.36 + }, + { + "id": "rate_limiting", + "similarity_score": 0.351 + }, + { + "id": "null_field_handling", + "similarity_score": 0.278 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.22 + }, + { + "trial": 1, + "case_id": "rl_payment_retries_02", + "category": "rate_limit_429", + "culprit_hash": "bfb3bde", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.342 + }, + { + "id": "rate_limiting", + "similarity_score": 0.32 + }, + { + "id": "request_timeouts", + "similarity_score": 0.284 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.2 + }, + { + "trial": 1, + "case_id": "ser_decimal_json_02", + "category": "json_serialization", + "culprit_hash": "5203835", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.47 + }, + { + "trial": 1, + "case_id": "sku_lowercase_06", + "category": "identifier_format", + "culprit_hash": "b8f58f4", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.11 + }, + { + "trial": 1, + "case_id": "sql_param_count_05", + "category": "sql_param_mismatch", + "culprit_hash": "b2d1a3c", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.3 + }, + { + "trial": 1, + "case_id": "tmo_gateway_deadline_02", + "category": "latency_timeout", + "culprit_hash": "5039371", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.507 + }, + { + "id": "rate_limiting", + "similarity_score": 0.363 + }, + { + "id": "db_failures", + "similarity_score": 0.353 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.42 + }, + { + "trial": 1, + "case_id": "tmo_inventory_units_01", + "category": "latency_timeout", + "culprit_hash": "fdf72fb", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.404 + }, + { + "id": "auth_failures", + "similarity_score": 0.334 + }, + { + "id": "rate_limiting", + "similarity_score": 0.256 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.22 + }, + { + "trial": 1, + "case_id": "tmo_list_live_stock_03", + "category": "latency_timeout", + "culprit_hash": "fa0acb3", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.566 + }, + { + "id": "auth_failures", + "similarity_score": 0.41 + }, + { + "id": "rate_limiting", + "similarity_score": 0.399 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.23 + }, + { + "trial": 1, + "case_id": "tz_naive_since_01", + "category": "timezone_mismatch", + "culprit_hash": "d010a5b", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.22 + }, + { + "trial": 2, + "case_id": "auth_algorithm_02", + "category": "auth_rejection", + "culprit_hash": "144f54b", + "expected_runbook_id": "auth_failures", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.305 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.24 + }, + { + "id": "null_field_handling", + "similarity_score": 0.216 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.31 + }, + { + "trial": 2, + "case_id": "auth_audience_rename_01", + "category": "auth_rejection", + "culprit_hash": "9bec53f", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.264 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.135 + }, + { + "id": "rate_limiting", + "similarity_score": 0.106 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.14 + }, + { + "trial": 2, + "case_id": "auth_leeway_zero_03", + "category": "auth_rejection", + "culprit_hash": "3cd3de4", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.383 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.209 + }, + { + "id": "null_field_handling", + "similarity_score": 0.147 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.24 + }, + { + "trial": 2, + "case_id": "cache_recursion_04", + "category": "infinite_recursion", + "culprit_hash": "1b419cd", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.33 + }, + { + "trial": 2, + "case_id": "cfg_cache_size_string_01", + "category": "config_type_error", + "culprit_hash": "5ab5431", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.17 + }, + { + "trial": 2, + "case_id": "cfg_env_override_string_03", + "category": "config_type_error", + "culprit_hash": "2239586", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.234 + }, + { + "id": "pagination_errors", + "similarity_score": 0.187 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.3 + }, + { + "trial": 2, + "case_id": "cfg_json_trailing_comma_02", + "category": "config_type_error", + "culprit_hash": "46e2927", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.262 + }, + { + "id": "null_field_handling", + "similarity_score": 0.113 + }, + { + "id": "auth_failures", + "similarity_score": 0.067 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.27 + }, + { + "trial": 2, + "case_id": "cfg_renamed_gateway_key_04", + "category": "config_type_error", + "culprit_hash": "9f89fe3", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.548 + }, + { + "id": "memory_leaks", + "similarity_score": 0.384 + }, + { + "id": "db_failures", + "similarity_score": 0.347 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.14 + }, + { + "trial": 2, + "case_id": "db_pool_leak_01", + "category": "db_pool_exhaustion", + "culprit_hash": "54c490d", + "expected_runbook_id": "db_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.16 + }, + { + "trial": 2, + "case_id": "deps_psycopg3_01", + "category": "dependency_version", + "culprit_hash": "49d9702", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.299 + }, + { + "id": "db_failures", + "similarity_score": 0.288 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.192 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.25 + }, + { + "trial": 2, + "case_id": "deps_redis_legacy_02", + "category": "dependency_version", + "culprit_hash": "c673656", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.25 + }, + { + "trial": 2, + "case_id": "fd_export_writer_01", + "category": "file_handle_leak", + "culprit_hash": "03bf82d", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.456 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.341 + }, + { + "id": "memory_leaks", + "similarity_score": 0.294 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.22 + }, + { + "trial": 2, + "case_id": "mem_worker_seen_ids_01", + "category": "memory_growth", + "culprit_hash": "6064408", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.462 + }, + { + "id": "rate_limiting", + "similarity_score": 0.449 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.433 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.59 + }, + { + "trial": 2, + "case_id": "null_guest_email_01", + "category": "null_or_missing_field", + "culprit_hash": "926570f", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.236 + }, + { + "id": "null_field_handling", + "similarity_score": 0.175 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.36 + }, + { + "trial": 2, + "case_id": "null_since_filter_03", + "category": "null_or_missing_field", + "culprit_hash": "f6961f1", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.232 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.175 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.36 + }, + { + "trial": 2, + "case_id": "null_stock_payload_02", + "category": "null_or_missing_field", + "culprit_hash": "23fce9a", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.271 + }, + { + "id": "pagination_errors", + "similarity_score": 0.226 + }, + { + "id": "null_field_handling", + "similarity_score": 0.21 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.31 + }, + { + "trial": 2, + "case_id": "page_newest_empty_01", + "category": "pagination_boundary", + "culprit_hash": "bc477e7", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.375 + }, + { + "id": "memory_leaks", + "similarity_score": 0.24 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.223 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.39 + }, + { + "trial": 2, + "case_id": "page_size_zero_02", + "category": "pagination_boundary", + "culprit_hash": "ff6ef3c", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.23 + }, + { + "trial": 2, + "case_id": "payments_form_body_07", + "category": "http_content_type", + "culprit_hash": "60716f4", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.42 + }, + { + "trial": 2, + "case_id": "queue_args_swapped_03", + "category": "queue_stall", + "culprit_hash": "38cf2c1", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.364 + }, + { + "id": "rate_limiting", + "similarity_score": 0.326 + }, + { + "id": "db_failures", + "similarity_score": 0.282 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.38 + }, + { + "trial": 2, + "case_id": "rl_inventory_uncached_01", + "category": "rate_limit_429", + "culprit_hash": "5acb3ab", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.36 + }, + { + "id": "rate_limiting", + "similarity_score": 0.351 + }, + { + "id": "null_field_handling", + "similarity_score": 0.278 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.55 + }, + { + "trial": 2, + "case_id": "rl_payment_retries_02", + "category": "rate_limit_429", + "culprit_hash": "bfb3bde", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.342 + }, + { + "id": "rate_limiting", + "similarity_score": 0.32 + }, + { + "id": "request_timeouts", + "similarity_score": 0.284 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.33 + }, + { + "trial": 2, + "case_id": "ser_decimal_json_02", + "category": "json_serialization", + "culprit_hash": "5203835", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.5 + }, + { + "trial": 2, + "case_id": "sku_lowercase_06", + "category": "identifier_format", + "culprit_hash": "b8f58f4", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.38 + }, + { + "trial": 2, + "case_id": "sql_param_count_05", + "category": "sql_param_mismatch", + "culprit_hash": "b2d1a3c", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.44 + }, + { + "trial": 2, + "case_id": "tmo_gateway_deadline_02", + "category": "latency_timeout", + "culprit_hash": "5039371", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.507 + }, + { + "id": "rate_limiting", + "similarity_score": 0.363 + }, + { + "id": "db_failures", + "similarity_score": 0.353 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.23 + }, + { + "trial": 2, + "case_id": "tmo_inventory_units_01", + "category": "latency_timeout", + "culprit_hash": "fdf72fb", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.404 + }, + { + "id": "auth_failures", + "similarity_score": 0.334 + }, + { + "id": "rate_limiting", + "similarity_score": 0.256 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.31 + }, + { + "trial": 2, + "case_id": "tmo_list_live_stock_03", + "category": "latency_timeout", + "culprit_hash": "fa0acb3", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.566 + }, + { + "id": "auth_failures", + "similarity_score": 0.41 + }, + { + "id": "rate_limiting", + "similarity_score": 0.399 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.28 + }, + { + "trial": 2, + "case_id": "tz_naive_since_01", + "category": "timezone_mismatch", + "culprit_hash": "d010a5b", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [], + "culprit_rank": null, + "error": "LLMUnavailable: LLM request failed: Client error '402 Payment Required' for url 'https://openrouter.ai/api/v1/chat/completions'\nFor more information check: https://developer.mozilla.org/en-US/docs/Web/HTTP/Status/402", + "rank_latency_s": 1.25 + } + ], + "finished_at": "2026-10-04T18:21:17.109517+00:00" +} \ No newline at end of file diff --git a/eval/run_eval.py b/eval/run_eval.py index c65779e..199ec1e 100644 --- a/eval/run_eval.py +++ b/eval/run_eval.py @@ -20,6 +20,7 @@ from eval import config, fixture ROOT = Path(__file__).parent.parent +MAX_CONSECUTIVE_ERRORS = 5 def run_trial(case: dict, condition: str) -> dict: @@ -125,19 +126,30 @@ def main(argv=None): out_path = Path(args.out) out_path.parent.mkdir(parents=True, exist_ok=True) + streak = 0 with patch.object(llm_analyzer, "chat", _fake_chat) if args.dry_run else nullcontext(): - for trial in range(args.trials): - for case in cases: - rec = {"trial": trial, **run_trial(case, args.condition)} - out["records"].append(rec) - # rewrite after every trial so an interrupted run keeps what it measured - out_path.write_text(json.dumps(out, indent=2), encoding="utf-8") - status = rec["error"] or f"culprit rank {rec['culprit_rank']}" - print(f"[eval] trial {trial} {case['id']}: {status} ({rec['rank_latency_s']}s)") + for trial, case in ((t, c) for t in range(args.trials) for c in cases): + rec = {"trial": trial, **run_trial(case, args.condition)} + out["records"].append(rec) + # rewrite after every trial so an interrupted run keeps what it measured + out_path.write_text(json.dumps(out, indent=2), encoding="utf-8") + status = rec["error"] or f"culprit rank {rec['culprit_rank']}" + print(f"[eval] trial {trial} {case['id']}: {status} ({rec['rank_latency_s']}s)", flush=True) + streak = streak + 1 if rec["error"] else 0 + if streak >= MAX_CONSECUTIVE_ERRORS: + # ponytail: a dead key / empty balance is not a model failure; stop instead of + # burning the remaining cases. The partial file is kept and marked, never scored as a run. + out["aborted"] = f"{streak} consecutive LLM errors; last: {rec['error'][:200]}" + break out["finished_at"] = datetime.now(timezone.utc).isoformat() out_path.write_text(json.dumps(out, indent=2), encoding="utf-8") - print(f"[eval] wrote {len(out['records'])} records to {out_path}") + print( + f"[eval] wrote {len(out['records'])} records to {out_path}" + + (" (ABORTED)" if "aborted" in out else "") + ) + if "aborted" in out: + raise SystemExit(f"[eval] aborted: {out['aborted']}") if __name__ == "__main__": diff --git a/eval/tests/test_eval.py b/eval/tests/test_eval.py index 2c75642..8461c61 100644 --- a/eval/tests/test_eval.py +++ b/eval/tests/test_eval.py @@ -4,6 +4,7 @@ import re from unittest.mock import patch +import httpx import pytest from core.services import git_client, llm_analyzer @@ -144,3 +145,20 @@ def capture(messages, **kw): assert value not in prompt, f"{case['id']}: {value!r} leaked into prompt" assert "culprit" not in prompt.lower() assert len(prompts) == len(CASES) + + +def test_run_aborts_after_consecutive_llm_errors(tmp_path): + out = tmp_path / "r.json" + + def dead_key(*_a, **_kw): + raise httpx.HTTPError("402 Payment Required") + + with ( + patch.object(llm_analyzer, "chat", dead_key), + patch.object(run_eval.vector_store, "ingest_runbooks", return_value=0), + patch.object(run_eval.vector_store, "find_matching_runbooks", return_value=[]), + pytest.raises(SystemExit), + ): + run_eval.main(["--condition", "baseline", "--trials", "3", "--out", str(out)]) + res = report.load(out) + assert len(res["records"]) == run_eval.MAX_CONSECUTIVE_ERRORS and "aborted" in res From c2d93bd64a7ddde5ccc59a808e8307cea4765fd8 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 14:22:52 -0400 Subject: [PATCH 09/19] chore(eval): move 402-failed baseline under results/aborted Co-Authored-By: Claude Opus 5.5 --- .../baseline_2026-10-04_402.json} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename eval/results/{baseline_2026-10-04.json => aborted/baseline_2026-10-04_402.json} (100%) diff --git a/eval/results/baseline_2026-10-04.json b/eval/results/aborted/baseline_2026-10-04_402.json similarity index 100% rename from eval/results/baseline_2026-10-04.json rename to eval/results/aborted/baseline_2026-10-04_402.json From 5f93cb743ecb1f3a163f268ba6da2fa0c081af2d Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 14:30:02 -0400 Subject: [PATCH 10/19] feat(core): call Claude directly via the Anthropic SDK instead of OpenRouter Same model (claude-sonnet-5) and same ranking prompt; forced rank_commits tool call now uses the native tool_use block. max_tokens raised to 16000 because Sonnet 5 runs adaptive thinking by default and thinking counts toward the cap. Co-Authored-By: Claude Opus 5.5 --- .env.example | 2 +- .github/workflows/ci.yml | 2 +- README.md | 4 +-- core/pyproject.toml | 1 + core/services/llm.py | 23 ++++++++++++ core/services/llm_analyzer.py | 67 ++++++++++++++++------------------- core/services/openrouter.py | 27 -------------- core/services/postmortem.py | 17 ++++----- core/tests/test_llm_logic.py | 40 +++++++++------------ core/tests/test_phase3.py | 7 ++-- core/tests/test_resilience.py | 27 +++++++------- docker-compose.yml | 2 +- eval/run_eval.py | 13 ++++--- eval/tests/test_eval.py | 3 +- 14 files changed, 112 insertions(+), 123 deletions(-) create mode 100644 core/services/llm.py delete mode 100644 core/services/openrouter.py diff --git a/.env.example b/.env.example index 0440377..eff2f80 100644 --- a/.env.example +++ b/.env.example @@ -1,5 +1,5 @@ # Required -OPENROUTER_API_KEY=sk-or-... +ANTHROPIC_API_KEY=sk-ant-... # Required for local (non-Docker) runs; docker-compose provides its own DATABASE_URL=postgresql://user:password@localhost:5432/sentinel # Optional — Slack diagnostic cards are skipped if unset diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 411c4b1..654e846 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -10,7 +10,7 @@ jobs: env: # tests mock all external boundaries; these only satisfy import-time env lookups DATABASE_URL: postgresql://ci:ci@localhost:5432/ci - OPENROUTER_API_KEY: ci-dummy + ANTHROPIC_API_KEY: ci-dummy steps: - uses: actions/checkout@v4 - uses: actions/setup-python@v5 diff --git a/README.md b/README.md index 2535344..a9b3a06 100644 --- a/README.md +++ b/README.md @@ -16,7 +16,7 @@ Slack, and drafts the postmortem on resolution. ```bash git clone https://github.com/Lushenwar/Sentinel.git && cd Sentinel -cp .env.example .env # set OPENROUTER_API_KEY (and optionally SLACK_WEBHOOK_URL) +cp .env.example .env # set ANTHROPIC_API_KEY (and optionally SLACK_WEBHOOK_URL) docker-compose up # dashboard :3000, core API :8000, sandbox app :8001, Postgres :5432 # trigger an incident (in another terminal) @@ -30,7 +30,7 @@ docker-compose exec core python -m sandbox.chaos_cli resolve-bug --incident =2.9", "python-dotenv>=1.0", "httpx>=0.27", + "anthropic>=0.116", "chromadb>=0.5", ] diff --git a/core/services/llm.py b/core/services/llm.py new file mode 100644 index 0000000..30579b5 --- /dev/null +++ b/core/services/llm.py @@ -0,0 +1,23 @@ +from functools import cache + +import anthropic +from dotenv import load_dotenv + +load_dotenv() +MODEL = "claude-sonnet-5" + + +@cache +def _client() -> anthropic.Anthropic: + return anthropic.Anthropic() # ANTHROPIC_API_KEY from env / .env + + +def chat( + messages: list[dict], + tools: list[dict] | None = None, + tool_choice: dict | None = None, + max_tokens: int = 16000, +) -> anthropic.types.Message: + """One Messages API call. Raises anthropic.APIError on transport or API failure.""" + extra = {"tools": tools, "tool_choice": tool_choice} if tools else {} + return _client().messages.create(model=MODEL, max_tokens=max_tokens, messages=messages, **extra) diff --git a/core/services/llm_analyzer.py b/core/services/llm_analyzer.py index 8aaa384..63f8ac5 100644 --- a/core/services/llm_analyzer.py +++ b/core/services/llm_analyzer.py @@ -1,6 +1,6 @@ -import json -import httpx -from .openrouter import chat +import anthropic + +from .llm import chat class LLMUnavailable(Exception): @@ -8,29 +8,26 @@ class LLMUnavailable(Exception): _RANK_TOOL = { - "type": "function", - "function": { - "name": "rank_commits", - "description": "Rank commits by likelihood of causing the incident. Include ALL commits, scored 0-1.", - "parameters": { - "type": "object", - "required": ["ranked_commits"], - "properties": { - "ranked_commits": { - "type": "array", - "items": { - "type": "object", - "required": ["commit_hash", "author", "timestamp", "rationale", "confidence_score"], - "properties": { - "commit_hash": {"type": "string"}, - "author": {"type": "string"}, - "timestamp": {"type": "string"}, - "rationale": {"type": "string"}, - "confidence_score": {"type": "number", "minimum": 0, "maximum": 1}, - }, + "name": "rank_commits", + "description": "Rank commits by likelihood of causing the incident. Include ALL commits, scored 0-1.", + "input_schema": { + "type": "object", + "required": ["ranked_commits"], + "properties": { + "ranked_commits": { + "type": "array", + "items": { + "type": "object", + "required": ["commit_hash", "author", "timestamp", "rationale", "confidence_score"], + "properties": { + "commit_hash": {"type": "string"}, + "author": {"type": "string"}, + "timestamp": {"type": "string"}, + "rationale": {"type": "string"}, + "confidence_score": {"type": "number", "minimum": 0, "maximum": 1}, }, - } - }, + }, + } }, }, } @@ -62,24 +59,22 @@ def rank_suspect_commits(diffs: list[dict], alert: dict) -> list[dict]: response = chat( messages=[{"role": "user", "content": prompt}], tools=[_RANK_TOOL], - tool_choice={"type": "function", "function": {"name": "rank_commits"}}, - max_tokens=1024, + tool_choice={"type": "tool", "name": "rank_commits"}, ) - message = response["choices"][0]["message"] - except httpx.HTTPError as e: + except anthropic.APIError as e: raise LLMUnavailable(f"LLM request failed: {e}") from e - except (KeyError, IndexError, TypeError) as e: - raise LLMUnavailable(f"unexpected LLM response shape: {e}") from e - for call in message.get("tool_calls") or []: - if call["function"]["name"] == "rank_commits": + for block in response.content: + if block.type == "tool_use" and block.name == "rank_commits": try: - ranked = json.loads(call["function"]["arguments"])["ranked_commits"] + ranked = block.input["ranked_commits"] for c in ranked: if not 0 <= c["confidence_score"] <= 1: raise ValueError(f"confidence_score out of range: {c['confidence_score']}") return sorted(ranked, key=lambda c: c["confidence_score"], reverse=True) - except (json.JSONDecodeError, KeyError, TypeError, ValueError) as e: + except (KeyError, TypeError, ValueError) as e: raise LLMUnavailable(f"malformed rank_commits output: {e}") from e - raise LLMUnavailable("empty or refused response: no rank_commits tool call returned") + raise LLMUnavailable( + f"empty or refused response: no rank_commits tool call returned (stop_reason={response.stop_reason})" + ) diff --git a/core/services/openrouter.py b/core/services/openrouter.py deleted file mode 100644 index 4638dd9..0000000 --- a/core/services/openrouter.py +++ /dev/null @@ -1,27 +0,0 @@ -import os -import httpx -from dotenv import load_dotenv - -load_dotenv() -API_URL = "https://openrouter.ai/api/v1/chat/completions" -MODEL = "anthropic/claude-sonnet-5" - - -def chat( - messages: list[dict], - tools: list[dict] | None = None, - tool_choice: dict | None = None, - max_tokens: int = 1024, -) -> dict: - payload = {"model": MODEL, "messages": messages, "max_tokens": max_tokens} - if tools: - payload["tools"] = tools - payload["tool_choice"] = tool_choice - response = httpx.post( - API_URL, - headers={"Authorization": f"Bearer {os.environ['OPENROUTER_API_KEY']}"}, - json=payload, - timeout=60.0, - ) - response.raise_for_status() - return response.json() diff --git a/core/services/postmortem.py b/core/services/postmortem.py index 64c1253..7cd0786 100644 --- a/core/services/postmortem.py +++ b/core/services/postmortem.py @@ -1,5 +1,6 @@ -import httpx -from .openrouter import chat +import anthropic + +from .llm import chat from .llm_analyzer import LLMUnavailable @@ -22,12 +23,12 @@ def generate_postmortem(incident: dict) -> str: ) try: - response = chat(messages=[{"role": "user", "content": prompt}], max_tokens=1024) - content = response["choices"][0]["message"]["content"] - except httpx.HTTPError as e: + response = chat(messages=[{"role": "user", "content": prompt}]) + except anthropic.APIError as e: raise LLMUnavailable(f"LLM request failed: {e}") from e - except (KeyError, IndexError, TypeError) as e: - raise LLMUnavailable(f"unexpected LLM response shape: {e}") from e + content = "".join(b.text for b in response.content if b.type == "text") if not content: - raise LLMUnavailable("empty or refused response: no postmortem content returned") + raise LLMUnavailable( + f"empty or refused response: no postmortem content returned (stop_reason={response.stop_reason})" + ) return content diff --git a/core/tests/test_llm_logic.py b/core/tests/test_llm_logic.py index c39cf7c..e764f2c 100644 --- a/core/tests/test_llm_logic.py +++ b/core/tests/test_llm_logic.py @@ -1,4 +1,5 @@ -# ponytail: mocks openrouter.chat so test runs without a live API key +# ponytail: mocks llm.chat so test runs without a live API key +from types import SimpleNamespace as NS from unittest.mock import patch from core.services.llm_analyzer import rank_suspect_commits @@ -22,27 +23,20 @@ def test_rank_returns_empty_for_no_diffs(): assert rank_suspect_commits([], _ALERT) == [] -def test_rank_calls_openrouter_and_returns_sorted(): - fake_response = { - "choices": [ - { - "message": { - "tool_calls": [ - { - "function": { - "name": "rank_commits", - "arguments": ( - '{"ranked_commits": [{"commit_hash": "a1b2c3d", ' - '"author": "dev@co.com", "timestamp": "2026-07-03T11:25:00Z", ' - '"rationale": "Directly set db_failure bug.", "confidence_score": 0.95}]}' - ), - }, - } - ], - }, - } - ], - } +def test_rank_calls_claude_and_returns_sorted(): + ranked = [ + { + "commit_hash": "a1b2c3d", + "author": "dev@co.com", + "timestamp": "2026-07-03T11:25:00Z", + "rationale": "Directly set db_failure bug.", + "confidence_score": 0.95, + } + ] + fake_response = NS( + stop_reason="tool_use", + content=[NS(type="tool_use", name="rank_commits", input={"ranked_commits": ranked})], + ) with patch("core.services.llm_analyzer.chat", return_value=fake_response): result = rank_suspect_commits(_DIFFS, _ALERT) @@ -53,5 +47,5 @@ def test_rank_calls_openrouter_and_returns_sorted(): if __name__ == "__main__": test_rank_returns_empty_for_no_diffs() - test_rank_calls_openrouter_and_returns_sorted() + test_rank_calls_claude_and_returns_sorted() print("✓ llm_logic tests passed") diff --git a/core/tests/test_phase3.py b/core/tests/test_phase3.py index 8845b06..e77ce45 100644 --- a/core/tests/test_phase3.py +++ b/core/tests/test_phase3.py @@ -1,4 +1,5 @@ # ponytail: mocks httpx/anthropic so tests run without live webhook or API key +from types import SimpleNamespace as NS from unittest.mock import patch, MagicMock from core.services.notifier import build_incident_card, post_incident_to_slack from core.services.postmortem import generate_postmortem @@ -46,9 +47,9 @@ def test_post_incident_to_slack_posts_when_configured(): def test_generate_postmortem_returns_llm_text(): - fake_response = { - "choices": [{"message": {"content": "## Summary\nDB pool misconfigured."}}], - } + fake_response = NS( + stop_reason="end_turn", content=[NS(type="text", text="## Summary\nDB pool misconfigured.")] + ) with patch("core.services.postmortem.chat", return_value=fake_response): result = generate_postmortem(_INCIDENT) assert "Summary" in result diff --git a/core/tests/test_resilience.py b/core/tests/test_resilience.py index c122166..f401adf 100644 --- a/core/tests/test_resilience.py +++ b/core/tests/test_resilience.py @@ -1,9 +1,11 @@ -# ponytail: all external boundaries (db, git, chroma, openrouter, slack) mocked — offline + deterministic +# ponytail: all external boundaries (db, git, chroma, claude api, slack) mocked — offline + deterministic import json import subprocess from pathlib import Path +from types import SimpleNamespace as NS from unittest.mock import patch +import anthropic import httpx import jsonschema import pytest @@ -48,45 +50,44 @@ ] -def _tool_response(arguments: str): - return { - "choices": [ - {"message": {"tool_calls": [{"function": {"name": "rank_commits", "arguments": arguments}}]}} - ] - } +def _tool_response(tool_input: dict): + return NS(stop_reason="tool_use", content=[NS(type="tool_use", name="rank_commits", input=tool_input)]) # ---------- LLM failure modes (7A) ---------- def test_llm_timeout_raises_llm_unavailable(): - with patch("core.services.llm_analyzer.chat", side_effect=httpx.TimeoutException("timed out")): + with patch( + "core.services.llm_analyzer.chat", + side_effect=anthropic.APITimeoutError(request=httpx.Request("POST", "https://api.anthropic.com")), + ): with pytest.raises(LLMUnavailable, match="request failed"): rank_suspect_commits(_DIFFS, _ALERT) -def test_llm_malformed_json_raises_llm_unavailable(): - with patch("core.services.llm_analyzer.chat", return_value=_tool_response("{not json")): +def test_llm_missing_field_raises_llm_unavailable(): + with patch("core.services.llm_analyzer.chat", return_value=_tool_response({"ranked": []})): with pytest.raises(LLMUnavailable, match="malformed"): rank_suspect_commits(_DIFFS, _ALERT) def test_llm_non_schema_output_raises_llm_unavailable(): - bad = json.dumps({"ranked_commits": [{"commit_hash": "abc", "confidence_score": 7}]}) + bad = {"ranked_commits": [{"commit_hash": "abc", "confidence_score": 7}]} with patch("core.services.llm_analyzer.chat", return_value=_tool_response(bad)): with pytest.raises(LLMUnavailable, match="malformed"): rank_suspect_commits(_DIFFS, _ALERT) def test_llm_empty_response_raises_llm_unavailable(): - with patch("core.services.llm_analyzer.chat", return_value={"choices": [{"message": {}}]}): + with patch("core.services.llm_analyzer.chat", return_value=NS(stop_reason="refusal", content=[])): with pytest.raises(LLMUnavailable, match="empty or refused"): rank_suspect_commits(_DIFFS, _ALERT) def test_postmortem_empty_content_raises_llm_unavailable(): incident = {"id": "inc_x", "trigger_data": _ALERT, "diagnostics": {}} - with patch("core.services.postmortem.chat", return_value={"choices": [{"message": {"content": ""}}]}): + with patch("core.services.postmortem.chat", return_value=NS(stop_reason="end_turn", content=[])): with pytest.raises(LLMUnavailable): generate_postmortem(incident) diff --git a/docker-compose.yml b/docker-compose.yml index a7d61c6..6ee4d96 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -19,7 +19,7 @@ services: dockerfile: Dockerfile.core environment: DATABASE_URL: postgresql://postgres:${POSTGRES_PASSWORD:-sentinel}@postgres:5432/postgres - OPENROUTER_API_KEY: ${OPENROUTER_API_KEY} + ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY} SLACK_WEBHOOK_URL: ${SLACK_WEBHOOK_URL:-} ports: - "8000:8000" diff --git a/eval/run_eval.py b/eval/run_eval.py index 199ec1e..29a6549 100644 --- a/eval/run_eval.py +++ b/eval/run_eval.py @@ -13,10 +13,11 @@ from contextlib import nullcontext from datetime import datetime, timezone from pathlib import Path +from types import SimpleNamespace as NS from unittest.mock import patch from core import orchestrator -from core.services import git_client, llm_analyzer, openrouter, vector_store +from core.services import git_client, llm, llm_analyzer, vector_store from eval import config, fixture ROOT = Path(__file__).parent.parent @@ -68,7 +69,7 @@ def run_trial(case: dict, condition: str) -> dict: def _fake_chat(messages, **_): - """Dry-run stand-in for openrouter.chat: ranks commits in prompt order.""" + """Dry-run stand-in for llm.chat: ranks commits in prompt order.""" hashes = re.findall(r"^COMMIT (\w{7}) \|", messages[0]["content"], flags=re.M) ranked = [ { @@ -80,10 +81,8 @@ def _fake_chat(messages, **_): } for i, h in enumerate(hashes) ] - args = json.dumps({"ranked_commits": ranked}) - return { - "choices": [{"message": {"tool_calls": [{"function": {"name": "rank_commits", "arguments": args}}]}}] - } + block = NS(type="tool_use", name="rank_commits", input={"ranked_commits": ranked}) + return NS(stop_reason="tool_use", content=[block]) def _git_sha() -> str: @@ -116,7 +115,7 @@ def main(argv=None): "eval_set_version": config.EVAL_SET_VERSION, "runbook_corpus_version": config.RUNBOOK_CORPUS_VERSION, "runbooks_ingested": n_runbooks, - "model": openrouter.MODEL, + "model": llm.MODEL, "sentinel_git_sha": _git_sha(), "started_at": datetime.now(timezone.utc).isoformat(), "trials": args.trials, diff --git a/eval/tests/test_eval.py b/eval/tests/test_eval.py index 8461c61..3aa6f1b 100644 --- a/eval/tests/test_eval.py +++ b/eval/tests/test_eval.py @@ -4,6 +4,7 @@ import re from unittest.mock import patch +import anthropic import httpx import pytest @@ -151,7 +152,7 @@ def test_run_aborts_after_consecutive_llm_errors(tmp_path): out = tmp_path / "r.json" def dead_key(*_a, **_kw): - raise httpx.HTTPError("402 Payment Required") + raise anthropic.APIConnectionError(request=httpx.Request("POST", "https://api.anthropic.com")) with ( patch.object(llm_analyzer, "chat", dead_key), From 7b4dc90672fea30b2d164986681fba151bb9f4d0 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 14:58:14 -0400 Subject: [PATCH 11/19] feat(core): send anthropic-workspace-id header when ANTHROPIC_WORKSPACE_ID is set Co-Authored-By: Claude Opus 5.5 --- .env.example | 1 + core/services/llm.py | 5 ++++- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/.env.example b/.env.example index eff2f80..5305a55 100644 --- a/.env.example +++ b/.env.example @@ -6,3 +6,4 @@ DATABASE_URL=postgresql://user:password@localhost:5432/sentinel SLACK_WEBHOOK_URL=https://hooks.slack.com/services/... # Optional — docker-compose Postgres password (defaults to "sentinel") POSTGRES_PASSWORD=sentinel +ANTHROPIC_WORKSPACE_ID= # only needed if the key is not workspace-scoped diff --git a/core/services/llm.py b/core/services/llm.py index 30579b5..456e4fb 100644 --- a/core/services/llm.py +++ b/core/services/llm.py @@ -1,3 +1,4 @@ +import os from functools import cache import anthropic @@ -9,7 +10,9 @@ @cache def _client() -> anthropic.Anthropic: - return anthropic.Anthropic() # ANTHROPIC_API_KEY from env / .env + # ANTHROPIC_API_KEY from env / .env; org-level keys also need the workspace header + workspace = os.getenv("ANTHROPIC_WORKSPACE_ID") + return anthropic.Anthropic(default_headers={"anthropic-workspace-id": workspace} if workspace else None) def chat( From b677ac1a6f243ec510ca13c40a47e88daa5fe4e8 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 15:16:15 -0400 Subject: [PATCH 12/19] feat(eval): baseline run on eval set v1 (87/87 top-1, ceiling) Co-Authored-By: Claude Opus 5.5 --- eval/results/baseline_2026-10-04.json | 3747 +++++++++++++++++++++++++ 1 file changed, 3747 insertions(+) create mode 100644 eval/results/baseline_2026-10-04.json diff --git a/eval/results/baseline_2026-10-04.json b/eval/results/baseline_2026-10-04.json new file mode 100644 index 0000000..8176a5f --- /dev/null +++ b/eval/results/baseline_2026-10-04.json @@ -0,0 +1,3747 @@ +{ + "condition": "baseline", + "dry_run": false, + "eval_set_version": "v1", + "runbook_corpus_version": "v1", + "runbooks_ingested": 10, + "model": "claude-sonnet-5", + "sentinel_git_sha": "7b4dc90672fea30b2d164986681fba151bb9f4d0", + "started_at": "2026-10-04T19:04:42.109687+00:00", + "trials": 3, + "n_cases": 29, + "records": [ + { + "trial": 0, + "case_id": "auth_algorithm_02", + "category": "auth_rejection", + "culprit_hash": "144f54b", + "expected_runbook_id": "auth_failures", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.305 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.24 + }, + { + "id": "null_field_handling", + "similarity_score": 0.216 + } + ], + "ranking": [ + { + "commit_hash": "144f54b", + "confidence_score": 0.97 + }, + { + "commit_hash": "86e027a", + "confidence_score": 0.05 + }, + { + "commit_hash": "191abd1", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.38 + }, + { + "trial": 0, + "case_id": "auth_audience_rename_01", + "category": "auth_rejection", + "culprit_hash": "9bec53f", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.264 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.135 + }, + { + "id": "rate_limiting", + "similarity_score": 0.106 + } + ], + "ranking": [ + { + "commit_hash": "9bec53f", + "confidence_score": 0.97 + }, + { + "commit_hash": "212b6f9", + "confidence_score": 0.1 + }, + { + "commit_hash": "60a1b2e", + "confidence_score": 0.0 + }, + { + "commit_hash": "41b03cf", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.03 + }, + { + "trial": 0, + "case_id": "auth_leeway_zero_03", + "category": "auth_rejection", + "culprit_hash": "3cd3de4", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.383 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.209 + }, + { + "id": "null_field_handling", + "similarity_score": 0.147 + } + ], + "ranking": [ + { + "commit_hash": "3cd3de4", + "confidence_score": 0.96 + }, + { + "commit_hash": "1c66960", + "confidence_score": 0.08 + }, + { + "commit_hash": "2f58cd9", + "confidence_score": 0.02 + }, + { + "commit_hash": "fee17ca", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.97 + }, + { + "trial": 0, + "case_id": "cache_recursion_04", + "category": "infinite_recursion", + "culprit_hash": "1b419cd", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [ + { + "commit_hash": "1b419cd", + "confidence_score": 0.98 + }, + { + "commit_hash": "a62b02a", + "confidence_score": 0.03 + }, + { + "commit_hash": "dc66d28", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 3.77 + }, + { + "trial": 0, + "case_id": "cfg_cache_size_string_01", + "category": "config_type_error", + "culprit_hash": "5ab5431", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [ + { + "commit_hash": "5ab5431", + "confidence_score": 0.97 + }, + { + "commit_hash": "f21baad", + "confidence_score": 0.25 + }, + { + "commit_hash": "8149df6", + "confidence_score": 0.03 + }, + { + "commit_hash": "607db60", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.12 + }, + { + "trial": 0, + "case_id": "cfg_env_override_string_03", + "category": "config_type_error", + "culprit_hash": "2239586", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.234 + }, + { + "id": "pagination_errors", + "similarity_score": 0.187 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [ + { + "commit_hash": "2239586", + "confidence_score": 0.95 + }, + { + "commit_hash": "10b026c", + "confidence_score": 0.05 + }, + { + "commit_hash": "8505a4f", + "confidence_score": 0.05 + }, + { + "commit_hash": "82cabc5", + "confidence_score": 0.03 + }, + { + "commit_hash": "e9a3a8a", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.17 + }, + { + "trial": 0, + "case_id": "cfg_json_trailing_comma_02", + "category": "config_type_error", + "culprit_hash": "46e2927", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.262 + }, + { + "id": "null_field_handling", + "similarity_score": 0.113 + }, + { + "id": "auth_failures", + "similarity_score": 0.067 + } + ], + "ranking": [ + { + "commit_hash": "46e2927", + "confidence_score": 0.97 + }, + { + "commit_hash": "d36f489", + "confidence_score": 0.05 + }, + { + "commit_hash": "abbe275", + "confidence_score": 0.02 + }, + { + "commit_hash": "c05d363", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 3.61 + }, + { + "trial": 0, + "case_id": "cfg_renamed_gateway_key_04", + "category": "config_type_error", + "culprit_hash": "9f89fe3", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.548 + }, + { + "id": "memory_leaks", + "similarity_score": 0.384 + }, + { + "id": "db_failures", + "similarity_score": 0.347 + } + ], + "ranking": [ + { + "commit_hash": "9f89fe3", + "confidence_score": 0.98 + }, + { + "commit_hash": "e4268c5", + "confidence_score": 0.03 + }, + { + "commit_hash": "f4be19e", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 3.91 + }, + { + "trial": 0, + "case_id": "db_pool_leak_01", + "category": "db_pool_exhaustion", + "culprit_hash": "54c490d", + "expected_runbook_id": "db_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "54c490d", + "confidence_score": 0.9 + }, + { + "commit_hash": "8e31a09", + "confidence_score": 0.4 + }, + { + "commit_hash": "3a50c47", + "confidence_score": 0.05 + }, + { + "commit_hash": "8d87780", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.05 + }, + { + "trial": 0, + "case_id": "deps_psycopg3_01", + "category": "dependency_version", + "culprit_hash": "49d9702", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.299 + }, + { + "id": "db_failures", + "similarity_score": 0.288 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.192 + } + ], + "ranking": [ + { + "commit_hash": "49d9702", + "confidence_score": 0.95 + }, + { + "commit_hash": "a189b64", + "confidence_score": 0.25 + }, + { + "commit_hash": "75bfae2", + "confidence_score": 0.03 + }, + { + "commit_hash": "53df8fc", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.86 + }, + { + "trial": 0, + "case_id": "deps_redis_legacy_02", + "category": "dependency_version", + "culprit_hash": "c673656", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [ + { + "commit_hash": "c673656", + "confidence_score": 0.85 + }, + { + "commit_hash": "0a25099", + "confidence_score": 0.4 + }, + { + "commit_hash": "d050e21", + "confidence_score": 0.05 + }, + { + "commit_hash": "32d1b2f", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.38 + }, + { + "trial": 0, + "case_id": "fd_export_writer_01", + "category": "file_handle_leak", + "culprit_hash": "03bf82d", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.456 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.341 + }, + { + "id": "memory_leaks", + "similarity_score": 0.294 + } + ], + "ranking": [ + { + "commit_hash": "03bf82d", + "confidence_score": 0.95 + }, + { + "commit_hash": "d53a28a", + "confidence_score": 0.15 + }, + { + "commit_hash": "2faab8c", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.33 + }, + { + "trial": 0, + "case_id": "mem_worker_seen_ids_01", + "category": "memory_growth", + "culprit_hash": "6064408", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.462 + }, + { + "id": "rate_limiting", + "similarity_score": 0.449 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.433 + } + ], + "ranking": [ + { + "commit_hash": "6064408", + "confidence_score": 0.95 + }, + { + "commit_hash": "d7b8995", + "confidence_score": 0.15 + }, + { + "commit_hash": "8a394ba", + "confidence_score": 0.03 + }, + { + "commit_hash": "6a16ad9", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.42 + }, + { + "trial": 0, + "case_id": "null_guest_email_01", + "category": "null_or_missing_field", + "culprit_hash": "926570f", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.236 + }, + { + "id": "null_field_handling", + "similarity_score": 0.175 + } + ], + "ranking": [ + { + "commit_hash": "926570f", + "confidence_score": 0.95 + }, + { + "commit_hash": "97c8a04", + "confidence_score": 0.05 + }, + { + "commit_hash": "10d72da", + "confidence_score": 0.02 + }, + { + "commit_hash": "a3fdcce", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.2 + }, + { + "trial": 0, + "case_id": "null_since_filter_03", + "category": "null_or_missing_field", + "culprit_hash": "f6961f1", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.232 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.175 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [ + { + "commit_hash": "f6961f1", + "confidence_score": 0.97 + }, + { + "commit_hash": "b2ade80", + "confidence_score": 0.2 + }, + { + "commit_hash": "d1168fa", + "confidence_score": 0.02 + }, + { + "commit_hash": "4790dd3", + "confidence_score": 0.02 + }, + { + "commit_hash": "1f5dba9", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.92 + }, + { + "trial": 0, + "case_id": "null_stock_payload_02", + "category": "null_or_missing_field", + "culprit_hash": "23fce9a", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.271 + }, + { + "id": "pagination_errors", + "similarity_score": 0.226 + }, + { + "id": "null_field_handling", + "similarity_score": 0.21 + } + ], + "ranking": [ + { + "commit_hash": "23fce9a", + "confidence_score": 0.85 + }, + { + "commit_hash": "36ef579", + "confidence_score": 0.1 + }, + { + "commit_hash": "52766bf", + "confidence_score": 0.05 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.34 + }, + { + "trial": 0, + "case_id": "page_newest_empty_01", + "category": "pagination_boundary", + "culprit_hash": "bc477e7", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.375 + }, + { + "id": "memory_leaks", + "similarity_score": 0.24 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.223 + } + ], + "ranking": [ + { + "commit_hash": "bc477e7", + "confidence_score": 0.95 + }, + { + "commit_hash": "c83e505", + "confidence_score": 0.03 + }, + { + "commit_hash": "c90f462", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 3.34 + }, + { + "trial": 0, + "case_id": "page_size_zero_02", + "category": "pagination_boundary", + "culprit_hash": "ff6ef3c", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [ + { + "commit_hash": "ff6ef3c", + "confidence_score": 0.92 + }, + { + "commit_hash": "2cf1af0", + "confidence_score": 0.85 + }, + { + "commit_hash": "d71c1e8", + "confidence_score": 0.02 + }, + { + "commit_hash": "0bbbc78", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.69 + }, + { + "trial": 0, + "case_id": "payments_form_body_07", + "category": "http_content_type", + "culprit_hash": "60716f4", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [ + { + "commit_hash": "60716f4", + "confidence_score": 0.98 + }, + { + "commit_hash": "6562507", + "confidence_score": 0.03 + }, + { + "commit_hash": "cf0c3a5", + "confidence_score": 0.02 + }, + { + "commit_hash": "72a23df", + "confidence_score": 0.01 + }, + { + "commit_hash": "0c0ff75", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.02 + }, + { + "trial": 0, + "case_id": "queue_args_swapped_03", + "category": "queue_stall", + "culprit_hash": "38cf2c1", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.364 + }, + { + "id": "rate_limiting", + "similarity_score": 0.326 + }, + { + "id": "db_failures", + "similarity_score": 0.282 + } + ], + "ranking": [ + { + "commit_hash": "38cf2c1", + "confidence_score": 0.95 + }, + { + "commit_hash": "478174d", + "confidence_score": 0.1 + }, + { + "commit_hash": "da9706d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.8 + }, + { + "trial": 0, + "case_id": "rl_inventory_uncached_01", + "category": "rate_limit_429", + "culprit_hash": "5acb3ab", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.36 + }, + { + "id": "rate_limiting", + "similarity_score": 0.351 + }, + { + "id": "null_field_handling", + "similarity_score": 0.278 + } + ], + "ranking": [ + { + "commit_hash": "5acb3ab", + "confidence_score": 0.92 + }, + { + "commit_hash": "12e9a97", + "confidence_score": 0.35 + }, + { + "commit_hash": "8559a88", + "confidence_score": 0.02 + }, + { + "commit_hash": "a6681fb", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.95 + }, + { + "trial": 0, + "case_id": "rl_payment_retries_02", + "category": "rate_limit_429", + "culprit_hash": "bfb3bde", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.342 + }, + { + "id": "rate_limiting", + "similarity_score": 0.32 + }, + { + "id": "request_timeouts", + "similarity_score": 0.284 + } + ], + "ranking": [ + { + "commit_hash": "bfb3bde", + "confidence_score": 0.95 + }, + { + "commit_hash": "25f465e", + "confidence_score": 0.05 + }, + { + "commit_hash": "222533c", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.2 + }, + { + "trial": 0, + "case_id": "ser_decimal_json_02", + "category": "json_serialization", + "culprit_hash": "5203835", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [ + { + "commit_hash": "5203835", + "confidence_score": 0.85 + }, + { + "commit_hash": "e33f86a", + "confidence_score": 0.2 + }, + { + "commit_hash": "6e27326", + "confidence_score": 0.02 + }, + { + "commit_hash": "aae3c38", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.64 + }, + { + "trial": 0, + "case_id": "sku_lowercase_06", + "category": "identifier_format", + "culprit_hash": "b8f58f4", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [ + { + "commit_hash": "b8f58f4", + "confidence_score": 0.8 + }, + { + "commit_hash": "579d3c7", + "confidence_score": 0.15 + }, + { + "commit_hash": "7aaedf9", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.47 + }, + { + "trial": 0, + "case_id": "sql_param_count_05", + "category": "sql_param_mismatch", + "culprit_hash": "b2d1a3c", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [ + { + "commit_hash": "b2d1a3c", + "confidence_score": 0.9 + }, + { + "commit_hash": "372baf7", + "confidence_score": 0.05 + }, + { + "commit_hash": "6c918d9", + "confidence_score": 0.03 + }, + { + "commit_hash": "565a0b7", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.14 + }, + { + "trial": 0, + "case_id": "tmo_gateway_deadline_02", + "category": "latency_timeout", + "culprit_hash": "5039371", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.507 + }, + { + "id": "rate_limiting", + "similarity_score": 0.363 + }, + { + "id": "db_failures", + "similarity_score": 0.353 + } + ], + "ranking": [ + { + "commit_hash": "5039371", + "confidence_score": 0.95 + }, + { + "commit_hash": "ef6e54e", + "confidence_score": 0.05 + }, + { + "commit_hash": "6cd56ca", + "confidence_score": 0.02 + }, + { + "commit_hash": "99ad00c", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.72 + }, + { + "trial": 0, + "case_id": "tmo_inventory_units_01", + "category": "latency_timeout", + "culprit_hash": "fdf72fb", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.404 + }, + { + "id": "auth_failures", + "similarity_score": 0.334 + }, + { + "id": "rate_limiting", + "similarity_score": 0.256 + } + ], + "ranking": [ + { + "commit_hash": "fdf72fb", + "confidence_score": 0.95 + }, + { + "commit_hash": "dd3b18a", + "confidence_score": 0.05 + }, + { + "commit_hash": "b756e42", + "confidence_score": 0.02 + }, + { + "commit_hash": "5a40192", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.09 + }, + { + "trial": 0, + "case_id": "tmo_list_live_stock_03", + "category": "latency_timeout", + "culprit_hash": "fa0acb3", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.566 + }, + { + "id": "auth_failures", + "similarity_score": 0.41 + }, + { + "id": "rate_limiting", + "similarity_score": 0.399 + } + ], + "ranking": [ + { + "commit_hash": "fa0acb3", + "confidence_score": 0.95 + }, + { + "commit_hash": "b9c96f5", + "confidence_score": 0.35 + }, + { + "commit_hash": "2f789c7", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.95 + }, + { + "trial": 0, + "case_id": "tz_naive_since_01", + "category": "timezone_mismatch", + "culprit_hash": "d010a5b", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [ + { + "commit_hash": "d010a5b", + "confidence_score": 0.85 + }, + { + "commit_hash": "468d269", + "confidence_score": 0.6 + }, + { + "commit_hash": "7073934", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.3 + }, + { + "trial": 1, + "case_id": "auth_algorithm_02", + "category": "auth_rejection", + "culprit_hash": "144f54b", + "expected_runbook_id": "auth_failures", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.305 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.24 + }, + { + "id": "null_field_handling", + "similarity_score": 0.216 + } + ], + "ranking": [ + { + "commit_hash": "144f54b", + "confidence_score": 0.97 + }, + { + "commit_hash": "86e027a", + "confidence_score": 0.05 + }, + { + "commit_hash": "191abd1", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.3 + }, + { + "trial": 1, + "case_id": "auth_audience_rename_01", + "category": "auth_rejection", + "culprit_hash": "9bec53f", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.264 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.135 + }, + { + "id": "rate_limiting", + "similarity_score": 0.106 + } + ], + "ranking": [ + { + "commit_hash": "9bec53f", + "confidence_score": 0.97 + }, + { + "commit_hash": "212b6f9", + "confidence_score": 0.05 + }, + { + "commit_hash": "41b03cf", + "confidence_score": 0.0 + }, + { + "commit_hash": "60a1b2e", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.91 + }, + { + "trial": 1, + "case_id": "auth_leeway_zero_03", + "category": "auth_rejection", + "culprit_hash": "3cd3de4", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.383 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.209 + }, + { + "id": "null_field_handling", + "similarity_score": 0.147 + } + ], + "ranking": [ + { + "commit_hash": "3cd3de4", + "confidence_score": 0.95 + }, + { + "commit_hash": "1c66960", + "confidence_score": 0.05 + }, + { + "commit_hash": "2f58cd9", + "confidence_score": 0.02 + }, + { + "commit_hash": "fee17ca", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.81 + }, + { + "trial": 1, + "case_id": "cache_recursion_04", + "category": "infinite_recursion", + "culprit_hash": "1b419cd", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [ + { + "commit_hash": "1b419cd", + "confidence_score": 0.98 + }, + { + "commit_hash": "a62b02a", + "confidence_score": 0.03 + }, + { + "commit_hash": "dc66d28", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.16 + }, + { + "trial": 1, + "case_id": "cfg_cache_size_string_01", + "category": "config_type_error", + "culprit_hash": "5ab5431", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [ + { + "commit_hash": "5ab5431", + "confidence_score": 0.97 + }, + { + "commit_hash": "f21baad", + "confidence_score": 0.1 + }, + { + "commit_hash": "8149df6", + "confidence_score": 0.03 + }, + { + "commit_hash": "607db60", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.0 + }, + { + "trial": 1, + "case_id": "cfg_env_override_string_03", + "category": "config_type_error", + "culprit_hash": "2239586", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.234 + }, + { + "id": "pagination_errors", + "similarity_score": 0.187 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [ + { + "commit_hash": "2239586", + "confidence_score": 0.9 + }, + { + "commit_hash": "10b026c", + "confidence_score": 0.05 + }, + { + "commit_hash": "8505a4f", + "confidence_score": 0.05 + }, + { + "commit_hash": "e9a3a8a", + "confidence_score": 0.02 + }, + { + "commit_hash": "82cabc5", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.81 + }, + { + "trial": 1, + "case_id": "cfg_json_trailing_comma_02", + "category": "config_type_error", + "culprit_hash": "46e2927", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.262 + }, + { + "id": "null_field_handling", + "similarity_score": 0.113 + }, + { + "id": "auth_failures", + "similarity_score": 0.067 + } + ], + "ranking": [ + { + "commit_hash": "46e2927", + "confidence_score": 0.97 + }, + { + "commit_hash": "d36f489", + "confidence_score": 0.05 + }, + { + "commit_hash": "abbe275", + "confidence_score": 0.02 + }, + { + "commit_hash": "c05d363", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.44 + }, + { + "trial": 1, + "case_id": "cfg_renamed_gateway_key_04", + "category": "config_type_error", + "culprit_hash": "9f89fe3", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.548 + }, + { + "id": "memory_leaks", + "similarity_score": 0.384 + }, + { + "id": "db_failures", + "similarity_score": 0.347 + } + ], + "ranking": [ + { + "commit_hash": "9f89fe3", + "confidence_score": 0.98 + }, + { + "commit_hash": "e4268c5", + "confidence_score": 0.03 + }, + { + "commit_hash": "f4be19e", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 3.5 + }, + { + "trial": 1, + "case_id": "db_pool_leak_01", + "category": "db_pool_exhaustion", + "culprit_hash": "54c490d", + "expected_runbook_id": "db_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "54c490d", + "confidence_score": 0.9 + }, + { + "commit_hash": "8e31a09", + "confidence_score": 0.55 + }, + { + "commit_hash": "3a50c47", + "confidence_score": 0.05 + }, + { + "commit_hash": "8d87780", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.91 + }, + { + "trial": 1, + "case_id": "deps_psycopg3_01", + "category": "dependency_version", + "culprit_hash": "49d9702", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.299 + }, + { + "id": "db_failures", + "similarity_score": 0.288 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.192 + } + ], + "ranking": [ + { + "commit_hash": "49d9702", + "confidence_score": 0.95 + }, + { + "commit_hash": "a189b64", + "confidence_score": 0.15 + }, + { + "commit_hash": "75bfae2", + "confidence_score": 0.02 + }, + { + "commit_hash": "53df8fc", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.83 + }, + { + "trial": 1, + "case_id": "deps_redis_legacy_02", + "category": "dependency_version", + "culprit_hash": "c673656", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [ + { + "commit_hash": "c673656", + "confidence_score": 0.85 + }, + { + "commit_hash": "0a25099", + "confidence_score": 0.45 + }, + { + "commit_hash": "d050e21", + "confidence_score": 0.03 + }, + { + "commit_hash": "32d1b2f", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.67 + }, + { + "trial": 1, + "case_id": "fd_export_writer_01", + "category": "file_handle_leak", + "culprit_hash": "03bf82d", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.456 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.341 + }, + { + "id": "memory_leaks", + "similarity_score": 0.294 + } + ], + "ranking": [ + { + "commit_hash": "03bf82d", + "confidence_score": 0.95 + }, + { + "commit_hash": "d53a28a", + "confidence_score": 0.15 + }, + { + "commit_hash": "2faab8c", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.66 + }, + { + "trial": 1, + "case_id": "mem_worker_seen_ids_01", + "category": "memory_growth", + "culprit_hash": "6064408", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.462 + }, + { + "id": "rate_limiting", + "similarity_score": 0.449 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.433 + } + ], + "ranking": [ + { + "commit_hash": "6064408", + "confidence_score": 0.95 + }, + { + "commit_hash": "d7b8995", + "confidence_score": 0.15 + }, + { + "commit_hash": "8a394ba", + "confidence_score": 0.05 + }, + { + "commit_hash": "6a16ad9", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.41 + }, + { + "trial": 1, + "case_id": "null_guest_email_01", + "category": "null_or_missing_field", + "culprit_hash": "926570f", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.236 + }, + { + "id": "null_field_handling", + "similarity_score": 0.175 + } + ], + "ranking": [ + { + "commit_hash": "926570f", + "confidence_score": 0.95 + }, + { + "commit_hash": "10d72da", + "confidence_score": 0.02 + }, + { + "commit_hash": "97c8a04", + "confidence_score": 0.02 + }, + { + "commit_hash": "a3fdcce", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.92 + }, + { + "trial": 1, + "case_id": "null_since_filter_03", + "category": "null_or_missing_field", + "culprit_hash": "f6961f1", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.232 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.175 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [ + { + "commit_hash": "f6961f1", + "confidence_score": 0.97 + }, + { + "commit_hash": "b2ade80", + "confidence_score": 0.15 + }, + { + "commit_hash": "d1168fa", + "confidence_score": 0.02 + }, + { + "commit_hash": "4790dd3", + "confidence_score": 0.01 + }, + { + "commit_hash": "1f5dba9", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.75 + }, + { + "trial": 1, + "case_id": "null_stock_payload_02", + "category": "null_or_missing_field", + "culprit_hash": "23fce9a", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.271 + }, + { + "id": "pagination_errors", + "similarity_score": 0.226 + }, + { + "id": "null_field_handling", + "similarity_score": 0.21 + } + ], + "ranking": [ + { + "commit_hash": "23fce9a", + "confidence_score": 0.85 + }, + { + "commit_hash": "36ef579", + "confidence_score": 0.1 + }, + { + "commit_hash": "52766bf", + "confidence_score": 0.05 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.14 + }, + { + "trial": 1, + "case_id": "page_newest_empty_01", + "category": "pagination_boundary", + "culprit_hash": "bc477e7", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.375 + }, + { + "id": "memory_leaks", + "similarity_score": 0.24 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.223 + } + ], + "ranking": [ + { + "commit_hash": "bc477e7", + "confidence_score": 0.95 + }, + { + "commit_hash": "c83e505", + "confidence_score": 0.05 + }, + { + "commit_hash": "c90f462", + "confidence_score": 0.03 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 3.81 + }, + { + "trial": 1, + "case_id": "page_size_zero_02", + "category": "pagination_boundary", + "culprit_hash": "ff6ef3c", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [ + { + "commit_hash": "ff6ef3c", + "confidence_score": 0.9 + }, + { + "commit_hash": "2cf1af0", + "confidence_score": 0.85 + }, + { + "commit_hash": "0bbbc78", + "confidence_score": 0.02 + }, + { + "commit_hash": "d71c1e8", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.34 + }, + { + "trial": 1, + "case_id": "payments_form_body_07", + "category": "http_content_type", + "culprit_hash": "60716f4", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [ + { + "commit_hash": "60716f4", + "confidence_score": 0.97 + }, + { + "commit_hash": "cf0c3a5", + "confidence_score": 0.03 + }, + { + "commit_hash": "6562507", + "confidence_score": 0.01 + }, + { + "commit_hash": "0c0ff75", + "confidence_score": 0.0 + }, + { + "commit_hash": "72a23df", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.97 + }, + { + "trial": 1, + "case_id": "queue_args_swapped_03", + "category": "queue_stall", + "culprit_hash": "38cf2c1", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.364 + }, + { + "id": "rate_limiting", + "similarity_score": 0.326 + }, + { + "id": "db_failures", + "similarity_score": 0.282 + } + ], + "ranking": [ + { + "commit_hash": "38cf2c1", + "confidence_score": 0.95 + }, + { + "commit_hash": "478174d", + "confidence_score": 0.05 + }, + { + "commit_hash": "da9706d", + "confidence_score": 0.03 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 3.8 + }, + { + "trial": 1, + "case_id": "rl_inventory_uncached_01", + "category": "rate_limit_429", + "culprit_hash": "5acb3ab", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.36 + }, + { + "id": "rate_limiting", + "similarity_score": 0.351 + }, + { + "id": "null_field_handling", + "similarity_score": 0.278 + } + ], + "ranking": [ + { + "commit_hash": "5acb3ab", + "confidence_score": 0.92 + }, + { + "commit_hash": "12e9a97", + "confidence_score": 0.35 + }, + { + "commit_hash": "8559a88", + "confidence_score": 0.02 + }, + { + "commit_hash": "a6681fb", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.34 + }, + { + "trial": 1, + "case_id": "rl_payment_retries_02", + "category": "rate_limit_429", + "culprit_hash": "bfb3bde", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.342 + }, + { + "id": "rate_limiting", + "similarity_score": 0.32 + }, + { + "id": "request_timeouts", + "similarity_score": 0.284 + } + ], + "ranking": [ + { + "commit_hash": "bfb3bde", + "confidence_score": 0.92 + }, + { + "commit_hash": "25f465e", + "confidence_score": 0.05 + }, + { + "commit_hash": "222533c", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.98 + }, + { + "trial": 1, + "case_id": "ser_decimal_json_02", + "category": "json_serialization", + "culprit_hash": "5203835", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [ + { + "commit_hash": "5203835", + "confidence_score": 0.85 + }, + { + "commit_hash": "e33f86a", + "confidence_score": 0.2 + }, + { + "commit_hash": "6e27326", + "confidence_score": 0.03 + }, + { + "commit_hash": "aae3c38", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.7 + }, + { + "trial": 1, + "case_id": "sku_lowercase_06", + "category": "identifier_format", + "culprit_hash": "b8f58f4", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [ + { + "commit_hash": "b8f58f4", + "confidence_score": 0.85 + }, + { + "commit_hash": "579d3c7", + "confidence_score": 0.15 + }, + { + "commit_hash": "7aaedf9", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.17 + }, + { + "trial": 1, + "case_id": "sql_param_count_05", + "category": "sql_param_mismatch", + "culprit_hash": "b2d1a3c", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [ + { + "commit_hash": "b2d1a3c", + "confidence_score": 0.85 + }, + { + "commit_hash": "6c918d9", + "confidence_score": 0.15 + }, + { + "commit_hash": "372baf7", + "confidence_score": 0.1 + }, + { + "commit_hash": "565a0b7", + "confidence_score": 0.05 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.58 + }, + { + "trial": 1, + "case_id": "tmo_gateway_deadline_02", + "category": "latency_timeout", + "culprit_hash": "5039371", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.507 + }, + { + "id": "rate_limiting", + "similarity_score": 0.363 + }, + { + "id": "db_failures", + "similarity_score": 0.353 + } + ], + "ranking": [ + { + "commit_hash": "5039371", + "confidence_score": 0.96 + }, + { + "commit_hash": "ef6e54e", + "confidence_score": 0.05 + }, + { + "commit_hash": "6cd56ca", + "confidence_score": 0.02 + }, + { + "commit_hash": "99ad00c", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.64 + }, + { + "trial": 1, + "case_id": "tmo_inventory_units_01", + "category": "latency_timeout", + "culprit_hash": "fdf72fb", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.404 + }, + { + "id": "auth_failures", + "similarity_score": 0.334 + }, + { + "id": "rate_limiting", + "similarity_score": 0.256 + } + ], + "ranking": [ + { + "commit_hash": "fdf72fb", + "confidence_score": 0.95 + }, + { + "commit_hash": "dd3b18a", + "confidence_score": 0.05 + }, + { + "commit_hash": "b756e42", + "confidence_score": 0.05 + }, + { + "commit_hash": "5a40192", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.97 + }, + { + "trial": 1, + "case_id": "tmo_list_live_stock_03", + "category": "latency_timeout", + "culprit_hash": "fa0acb3", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.566 + }, + { + "id": "auth_failures", + "similarity_score": 0.41 + }, + { + "id": "rate_limiting", + "similarity_score": 0.399 + } + ], + "ranking": [ + { + "commit_hash": "fa0acb3", + "confidence_score": 0.92 + }, + { + "commit_hash": "b9c96f5", + "confidence_score": 0.35 + }, + { + "commit_hash": "2f789c7", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.74 + }, + { + "trial": 1, + "case_id": "tz_naive_since_01", + "category": "timezone_mismatch", + "culprit_hash": "d010a5b", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [ + { + "commit_hash": "d010a5b", + "confidence_score": 0.9 + }, + { + "commit_hash": "468d269", + "confidence_score": 0.6 + }, + { + "commit_hash": "7073934", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 3.66 + }, + { + "trial": 2, + "case_id": "auth_algorithm_02", + "category": "auth_rejection", + "culprit_hash": "144f54b", + "expected_runbook_id": "auth_failures", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.305 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.24 + }, + { + "id": "null_field_handling", + "similarity_score": 0.216 + } + ], + "ranking": [ + { + "commit_hash": "144f54b", + "confidence_score": 0.97 + }, + { + "commit_hash": "86e027a", + "confidence_score": 0.05 + }, + { + "commit_hash": "191abd1", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.0 + }, + { + "trial": 2, + "case_id": "auth_audience_rename_01", + "category": "auth_rejection", + "culprit_hash": "9bec53f", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.264 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.135 + }, + { + "id": "rate_limiting", + "similarity_score": 0.106 + } + ], + "ranking": [ + { + "commit_hash": "9bec53f", + "confidence_score": 0.98 + }, + { + "commit_hash": "212b6f9", + "confidence_score": 0.05 + }, + { + "commit_hash": "41b03cf", + "confidence_score": 0.0 + }, + { + "commit_hash": "60a1b2e", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.19 + }, + { + "trial": 2, + "case_id": "auth_leeway_zero_03", + "category": "auth_rejection", + "culprit_hash": "3cd3de4", + "expected_runbook_id": "auth_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.383 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.209 + }, + { + "id": "null_field_handling", + "similarity_score": 0.147 + } + ], + "ranking": [ + { + "commit_hash": "3cd3de4", + "confidence_score": 0.95 + }, + { + "commit_hash": "1c66960", + "confidence_score": 0.05 + }, + { + "commit_hash": "2f58cd9", + "confidence_score": 0.02 + }, + { + "commit_hash": "fee17ca", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.17 + }, + { + "trial": 2, + "case_id": "cache_recursion_04", + "category": "infinite_recursion", + "culprit_hash": "1b419cd", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [ + { + "commit_hash": "1b419cd", + "confidence_score": 0.98 + }, + { + "commit_hash": "a62b02a", + "confidence_score": 0.03 + }, + { + "commit_hash": "dc66d28", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.2 + }, + { + "trial": 2, + "case_id": "cfg_cache_size_string_01", + "category": "config_type_error", + "culprit_hash": "5ab5431", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [ + { + "commit_hash": "5ab5431", + "confidence_score": 0.95 + }, + { + "commit_hash": "f21baad", + "confidence_score": 0.15 + }, + { + "commit_hash": "8149df6", + "confidence_score": 0.05 + }, + { + "commit_hash": "607db60", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.2 + }, + { + "trial": 2, + "case_id": "cfg_env_override_string_03", + "category": "config_type_error", + "culprit_hash": "2239586", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.234 + }, + { + "id": "pagination_errors", + "similarity_score": 0.187 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [ + { + "commit_hash": "2239586", + "confidence_score": 0.9 + }, + { + "commit_hash": "10b026c", + "confidence_score": 0.05 + }, + { + "commit_hash": "8505a4f", + "confidence_score": 0.05 + }, + { + "commit_hash": "e9a3a8a", + "confidence_score": 0.02 + }, + { + "commit_hash": "82cabc5", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.34 + }, + { + "trial": 2, + "case_id": "cfg_json_trailing_comma_02", + "category": "config_type_error", + "culprit_hash": "46e2927", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.262 + }, + { + "id": "null_field_handling", + "similarity_score": 0.113 + }, + { + "id": "auth_failures", + "similarity_score": 0.067 + } + ], + "ranking": [ + { + "commit_hash": "46e2927", + "confidence_score": 0.98 + }, + { + "commit_hash": "d36f489", + "confidence_score": 0.1 + }, + { + "commit_hash": "abbe275", + "confidence_score": 0.02 + }, + { + "commit_hash": "c05d363", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.5 + }, + { + "trial": 2, + "case_id": "cfg_renamed_gateway_key_04", + "category": "config_type_error", + "culprit_hash": "9f89fe3", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.548 + }, + { + "id": "memory_leaks", + "similarity_score": 0.384 + }, + { + "id": "db_failures", + "similarity_score": 0.347 + } + ], + "ranking": [ + { + "commit_hash": "9f89fe3", + "confidence_score": 0.98 + }, + { + "commit_hash": "e4268c5", + "confidence_score": 0.05 + }, + { + "commit_hash": "f4be19e", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 3.77 + }, + { + "trial": 2, + "case_id": "db_pool_leak_01", + "category": "db_pool_exhaustion", + "culprit_hash": "54c490d", + "expected_runbook_id": "db_failures", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "54c490d", + "confidence_score": 0.92 + }, + { + "commit_hash": "8e31a09", + "confidence_score": 0.45 + }, + { + "commit_hash": "3a50c47", + "confidence_score": 0.03 + }, + { + "commit_hash": "8d87780", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.95 + }, + { + "trial": 2, + "case_id": "deps_psycopg3_01", + "category": "dependency_version", + "culprit_hash": "49d9702", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.299 + }, + { + "id": "db_failures", + "similarity_score": 0.288 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.192 + } + ], + "ranking": [ + { + "commit_hash": "49d9702", + "confidence_score": 0.97 + }, + { + "commit_hash": "a189b64", + "confidence_score": 0.1 + }, + { + "commit_hash": "75bfae2", + "confidence_score": 0.02 + }, + { + "commit_hash": "53df8fc", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.78 + }, + { + "trial": 2, + "case_id": "deps_redis_legacy_02", + "category": "dependency_version", + "culprit_hash": "c673656", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [ + { + "commit_hash": "c673656", + "confidence_score": 0.85 + }, + { + "commit_hash": "0a25099", + "confidence_score": 0.45 + }, + { + "commit_hash": "d050e21", + "confidence_score": 0.05 + }, + { + "commit_hash": "32d1b2f", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.56 + }, + { + "trial": 2, + "case_id": "fd_export_writer_01", + "category": "file_handle_leak", + "culprit_hash": "03bf82d", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.456 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.341 + }, + { + "id": "memory_leaks", + "similarity_score": 0.294 + } + ], + "ranking": [ + { + "commit_hash": "03bf82d", + "confidence_score": 0.95 + }, + { + "commit_hash": "d53a28a", + "confidence_score": 0.15 + }, + { + "commit_hash": "2faab8c", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.3 + }, + { + "trial": 2, + "case_id": "mem_worker_seen_ids_01", + "category": "memory_growth", + "culprit_hash": "6064408", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.462 + }, + { + "id": "rate_limiting", + "similarity_score": 0.449 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.433 + } + ], + "ranking": [ + { + "commit_hash": "6064408", + "confidence_score": 0.92 + }, + { + "commit_hash": "d7b8995", + "confidence_score": 0.2 + }, + { + "commit_hash": "8a394ba", + "confidence_score": 0.03 + }, + { + "commit_hash": "6a16ad9", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.45 + }, + { + "trial": 2, + "case_id": "null_guest_email_01", + "category": "null_or_missing_field", + "culprit_hash": "926570f", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.236 + }, + { + "id": "null_field_handling", + "similarity_score": 0.175 + } + ], + "ranking": [ + { + "commit_hash": "926570f", + "confidence_score": 0.95 + }, + { + "commit_hash": "97c8a04", + "confidence_score": 0.05 + }, + { + "commit_hash": "10d72da", + "confidence_score": 0.05 + }, + { + "commit_hash": "a3fdcce", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.88 + }, + { + "trial": 2, + "case_id": "null_since_filter_03", + "category": "null_or_missing_field", + "culprit_hash": "f6961f1", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.232 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.175 + }, + { + "id": "memory_leaks", + "similarity_score": 0.16 + } + ], + "ranking": [ + { + "commit_hash": "f6961f1", + "confidence_score": 0.97 + }, + { + "commit_hash": "b2ade80", + "confidence_score": 0.25 + }, + { + "commit_hash": "d1168fa", + "confidence_score": 0.02 + }, + { + "commit_hash": "4790dd3", + "confidence_score": 0.01 + }, + { + "commit_hash": "1f5dba9", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.92 + }, + { + "trial": 2, + "case_id": "null_stock_payload_02", + "category": "null_or_missing_field", + "culprit_hash": "23fce9a", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.271 + }, + { + "id": "pagination_errors", + "similarity_score": 0.226 + }, + { + "id": "null_field_handling", + "similarity_score": 0.21 + } + ], + "ranking": [ + { + "commit_hash": "23fce9a", + "confidence_score": 0.85 + }, + { + "commit_hash": "52766bf", + "confidence_score": 0.1 + }, + { + "commit_hash": "36ef579", + "confidence_score": 0.05 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.38 + }, + { + "trial": 2, + "case_id": "page_newest_empty_01", + "category": "pagination_boundary", + "culprit_hash": "bc477e7", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.375 + }, + { + "id": "memory_leaks", + "similarity_score": 0.24 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.223 + } + ], + "ranking": [ + { + "commit_hash": "bc477e7", + "confidence_score": 0.95 + }, + { + "commit_hash": "c83e505", + "confidence_score": 0.05 + }, + { + "commit_hash": "c90f462", + "confidence_score": 0.03 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 3.39 + }, + { + "trial": 2, + "case_id": "page_size_zero_02", + "category": "pagination_boundary", + "culprit_hash": "ff6ef3c", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [ + { + "commit_hash": "ff6ef3c", + "confidence_score": 0.85 + }, + { + "commit_hash": "2cf1af0", + "confidence_score": 0.8 + }, + { + "commit_hash": "0bbbc78", + "confidence_score": 0.02 + }, + { + "commit_hash": "d71c1e8", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.31 + }, + { + "trial": 2, + "case_id": "payments_form_body_07", + "category": "http_content_type", + "culprit_hash": "60716f4", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [ + { + "commit_hash": "60716f4", + "confidence_score": 0.97 + }, + { + "commit_hash": "cf0c3a5", + "confidence_score": 0.03 + }, + { + "commit_hash": "6562507", + "confidence_score": 0.02 + }, + { + "commit_hash": "0c0ff75", + "confidence_score": 0.0 + }, + { + "commit_hash": "72a23df", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.34 + }, + { + "trial": 2, + "case_id": "queue_args_swapped_03", + "category": "queue_stall", + "culprit_hash": "38cf2c1", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.364 + }, + { + "id": "rate_limiting", + "similarity_score": 0.326 + }, + { + "id": "db_failures", + "similarity_score": 0.282 + } + ], + "ranking": [ + { + "commit_hash": "38cf2c1", + "confidence_score": 0.96 + }, + { + "commit_hash": "478174d", + "confidence_score": 0.1 + }, + { + "commit_hash": "da9706d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.08 + }, + { + "trial": 2, + "case_id": "rl_inventory_uncached_01", + "category": "rate_limit_429", + "culprit_hash": "5acb3ab", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.36 + }, + { + "id": "rate_limiting", + "similarity_score": 0.351 + }, + { + "id": "null_field_handling", + "similarity_score": 0.278 + } + ], + "ranking": [ + { + "commit_hash": "5acb3ab", + "confidence_score": 0.92 + }, + { + "commit_hash": "12e9a97", + "confidence_score": 0.55 + }, + { + "commit_hash": "8559a88", + "confidence_score": 0.03 + }, + { + "commit_hash": "a6681fb", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.53 + }, + { + "trial": 2, + "case_id": "rl_payment_retries_02", + "category": "rate_limit_429", + "culprit_hash": "bfb3bde", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.342 + }, + { + "id": "rate_limiting", + "similarity_score": 0.32 + }, + { + "id": "request_timeouts", + "similarity_score": 0.284 + } + ], + "ranking": [ + { + "commit_hash": "bfb3bde", + "confidence_score": 0.9 + }, + { + "commit_hash": "25f465e", + "confidence_score": 0.05 + }, + { + "commit_hash": "222533c", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.23 + }, + { + "trial": 2, + "case_id": "ser_decimal_json_02", + "category": "json_serialization", + "culprit_hash": "5203835", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [ + { + "commit_hash": "5203835", + "confidence_score": 0.9 + }, + { + "commit_hash": "e33f86a", + "confidence_score": 0.2 + }, + { + "commit_hash": "6e27326", + "confidence_score": 0.02 + }, + { + "commit_hash": "aae3c38", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.67 + }, + { + "trial": 2, + "case_id": "sku_lowercase_06", + "category": "identifier_format", + "culprit_hash": "b8f58f4", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [ + { + "commit_hash": "b8f58f4", + "confidence_score": 0.85 + }, + { + "commit_hash": "579d3c7", + "confidence_score": 0.15 + }, + { + "commit_hash": "7aaedf9", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.69 + }, + { + "trial": 2, + "case_id": "sql_param_count_05", + "category": "sql_param_mismatch", + "culprit_hash": "b2d1a3c", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [ + { + "commit_hash": "b2d1a3c", + "confidence_score": 0.85 + }, + { + "commit_hash": "372baf7", + "confidence_score": 0.05 + }, + { + "commit_hash": "6c918d9", + "confidence_score": 0.05 + }, + { + "commit_hash": "565a0b7", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.69 + }, + { + "trial": 2, + "case_id": "tmo_gateway_deadline_02", + "category": "latency_timeout", + "culprit_hash": "5039371", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.507 + }, + { + "id": "rate_limiting", + "similarity_score": 0.363 + }, + { + "id": "db_failures", + "similarity_score": 0.353 + } + ], + "ranking": [ + { + "commit_hash": "5039371", + "confidence_score": 0.97 + }, + { + "commit_hash": "ef6e54e", + "confidence_score": 0.05 + }, + { + "commit_hash": "6cd56ca", + "confidence_score": 0.01 + }, + { + "commit_hash": "99ad00c", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.97 + }, + { + "trial": 2, + "case_id": "tmo_inventory_units_01", + "category": "latency_timeout", + "culprit_hash": "fdf72fb", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.404 + }, + { + "id": "auth_failures", + "similarity_score": 0.334 + }, + { + "id": "rate_limiting", + "similarity_score": 0.256 + } + ], + "ranking": [ + { + "commit_hash": "fdf72fb", + "confidence_score": 0.92 + }, + { + "commit_hash": "dd3b18a", + "confidence_score": 0.05 + }, + { + "commit_hash": "b756e42", + "confidence_score": 0.03 + }, + { + "commit_hash": "5a40192", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.06 + }, + { + "trial": 2, + "case_id": "tmo_list_live_stock_03", + "category": "latency_timeout", + "culprit_hash": "fa0acb3", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.566 + }, + { + "id": "auth_failures", + "similarity_score": 0.41 + }, + { + "id": "rate_limiting", + "similarity_score": 0.399 + } + ], + "ranking": [ + { + "commit_hash": "fa0acb3", + "confidence_score": 0.92 + }, + { + "commit_hash": "b9c96f5", + "confidence_score": 0.35 + }, + { + "commit_hash": "2f789c7", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.06 + }, + { + "trial": 2, + "case_id": "tz_naive_since_01", + "category": "timezone_mismatch", + "culprit_hash": "d010a5b", + "expected_runbook_id": null, + "commits_in_window": 3, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [ + { + "commit_hash": "d010a5b", + "confidence_score": 0.9 + }, + { + "commit_hash": "468d269", + "confidence_score": 0.6 + }, + { + "commit_hash": "7073934", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.16 + } + ], + "finished_at": "2026-10-04T19:15:48.298677+00:00" +} \ No newline at end of file From 2e1eb8d6c9243d9cd742b07f819a6a9c17c6a7f3 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 15:37:42 -0400 Subject: [PATCH 13/19] feat(eval): harder eval set v2 (24 cases, decoy per case), replaces v1 after ceiling Every case now carries a decoy that matches the alert as well as the culprit at first glance; label.note records why the label is right (scorer-only). Co-Authored-By: Claude Opus 5.5 --- eval/bases/orders_service/app/cart.py | 20 ++++ eval/cases/auth_algorithm_02.json | 54 --------- eval/cases/auth_issuer_mesh_hosts.json | 89 +++++++++++++++ eval/cases/auth_leeway_zero_03.json | 67 ----------- eval/cases/cache_contains_recursion.json | 89 +++++++++++++++ eval/cases/cache_recursion_04.json | 58 ---------- eval/cases/cart_price_field_v2.json | 85 ++++++++++++++ eval/cases/cart_sku_lowercased.json | 72 ++++++++++++ eval/cases/cfg_env_override_string_03.json | 80 ------------- eval/cases/cfg_json_trailing_comma_02.json | 71 ------------ eval/cases/cfg_renamed_gateway_key_04.json | 54 --------- eval/cases/checkout_deadline_extra_call.json | 85 ++++++++++++++ eval/cases/config_reformat_hidden_string.json | 76 +++++++++++++ eval/cases/db_fetch_one_leak.json | 81 ++++++++++++++ eval/cases/db_pool_leak_01.json | 67 ----------- eval/cases/deps_psycopg3_01.json | 67 ----------- eval/cases/deps_redis_legacy_02.json | 71 ------------ eval/cases/export_tempfile_never_closed.json | 90 +++++++++++++++ eval/cases/fd_export_writer_01.json | 54 --------- ...ail_01.json => guest_email_normalise.json} | 60 ++++++---- eval/cases/httpx_028_drops_proxies.json | 86 ++++++++++++++ .../inventory_cache_key_per_request.json | 105 ++++++++++++++++++ ...ring_01.json => inventory_v3_payload.json} | 64 ++++++----- ...e_01.json => jwt_leeway_milliseconds.json} | 66 +++++++---- ...son_02.json => money_returns_decimal.json} | 46 +++++--- eval/cases/null_since_filter_03.json | 80 ------------- eval/cases/null_stock_payload_02.json | 66 ----------- eval/cases/order_read_sync_audit.json | 86 ++++++++++++++ ...t_05.json => order_sql_missing_param.json} | 48 +++++--- eval/cases/page_newest_empty_01.json | 54 --------- ...o_02.json => page_size_zero_division.json} | 40 ++++--- .../payments_body_without_content_type.json | 93 ++++++++++++++++ eval/cases/payments_form_body_07.json | 80 ------------- ...tries_02.json => payments_retry_loop.json} | 59 ++++++---- eval/cases/queue_args_swapped_03.json | 58 ---------- eval/cases/redis_legacy_lrem_args.json | 81 ++++++++++++++ eval/cases/rl_inventory_uncached_01.json | 67 ----------- eval/cases/settings_env_quotes.json | 81 ++++++++++++++ eval/cases/since_filter_naive_cutoff.json | 89 +++++++++++++++ eval/cases/sku_lowercase_06.json | 62 ----------- eval/cases/tmo_gateway_deadline_02.json | 67 ----------- eval/cases/tmo_inventory_units_01.json | 66 ----------- eval/cases/tmo_list_live_stock_03.json | 54 --------- eval/cases/tz_naive_since_01.json | 54 --------- eval/cases/worker_queue_renamed.json | 81 ++++++++++++++ ....json => worker_retains_job_payloads.json} | 54 +++++---- eval/config.py | 3 +- 47 files changed, 1673 insertions(+), 1507 deletions(-) create mode 100644 eval/bases/orders_service/app/cart.py delete mode 100644 eval/cases/auth_algorithm_02.json create mode 100644 eval/cases/auth_issuer_mesh_hosts.json delete mode 100644 eval/cases/auth_leeway_zero_03.json create mode 100644 eval/cases/cache_contains_recursion.json delete mode 100644 eval/cases/cache_recursion_04.json create mode 100644 eval/cases/cart_price_field_v2.json create mode 100644 eval/cases/cart_sku_lowercased.json delete mode 100644 eval/cases/cfg_env_override_string_03.json delete mode 100644 eval/cases/cfg_json_trailing_comma_02.json delete mode 100644 eval/cases/cfg_renamed_gateway_key_04.json create mode 100644 eval/cases/checkout_deadline_extra_call.json create mode 100644 eval/cases/config_reformat_hidden_string.json create mode 100644 eval/cases/db_fetch_one_leak.json delete mode 100644 eval/cases/db_pool_leak_01.json delete mode 100644 eval/cases/deps_psycopg3_01.json delete mode 100644 eval/cases/deps_redis_legacy_02.json create mode 100644 eval/cases/export_tempfile_never_closed.json delete mode 100644 eval/cases/fd_export_writer_01.json rename eval/cases/{null_guest_email_01.json => guest_email_normalise.json} (51%) create mode 100644 eval/cases/httpx_028_drops_proxies.json create mode 100644 eval/cases/inventory_cache_key_per_request.json rename eval/cases/{cfg_cache_size_string_01.json => inventory_v3_payload.json} (58%) rename eval/cases/{auth_audience_rename_01.json => jwt_leeway_milliseconds.json} (51%) rename eval/cases/{ser_decimal_json_02.json => money_returns_decimal.json} (53%) delete mode 100644 eval/cases/null_since_filter_03.json delete mode 100644 eval/cases/null_stock_payload_02.json create mode 100644 eval/cases/order_read_sync_audit.json rename eval/cases/{sql_param_count_05.json => order_sql_missing_param.json} (57%) delete mode 100644 eval/cases/page_newest_empty_01.json rename eval/cases/{page_size_zero_02.json => page_size_zero_division.json} (58%) create mode 100644 eval/cases/payments_body_without_content_type.json delete mode 100644 eval/cases/payments_form_body_07.json rename eval/cases/{rl_payment_retries_02.json => payments_retry_loop.json} (71%) delete mode 100644 eval/cases/queue_args_swapped_03.json create mode 100644 eval/cases/redis_legacy_lrem_args.json delete mode 100644 eval/cases/rl_inventory_uncached_01.json create mode 100644 eval/cases/settings_env_quotes.json create mode 100644 eval/cases/since_filter_naive_cutoff.json delete mode 100644 eval/cases/sku_lowercase_06.json delete mode 100644 eval/cases/tmo_gateway_deadline_02.json delete mode 100644 eval/cases/tmo_inventory_units_01.json delete mode 100644 eval/cases/tmo_list_live_stock_03.json delete mode 100644 eval/cases/tz_naive_since_01.json create mode 100644 eval/cases/worker_queue_renamed.json rename eval/cases/{mem_worker_seen_ids_01.json => worker_retains_job_payloads.json} (57%) diff --git a/eval/bases/orders_service/app/cart.py b/eval/bases/orders_service/app/cart.py new file mode 100644 index 0000000..4afc74e --- /dev/null +++ b/eval/bases/orders_service/app/cart.py @@ -0,0 +1,20 @@ +from app import db +from app.handlers.orders import create_order + +_CART_SQL = "SELECT payload FROM carts WHERE id = %s" + + +def load_cart(cart_id: str) -> dict: + row = db.fetch_one(_CART_SQL, (cart_id,)) + return row[0] if row else {"items": []} + + +def cart_to_lines(cart: dict) -> list[dict]: + return [ + {"sku": item["sku"], "qty": item["qty"], "unit_price_cents": item["unit_price_cents"]} + for item in cart["items"] + ] + + +def checkout(customer_id: str, cart_id: str, card_token: str) -> dict: + return create_order(customer_id, cart_to_lines(load_cart(cart_id)), card_token) diff --git a/eval/cases/auth_algorithm_02.json b/eval/cases/auth_algorithm_02.json deleted file mode 100644 index b9eed76..0000000 --- a/eval/cases/auth_algorithm_02.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "id": "auth_algorithm_02", - "category": "auth_rejection", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_401_Unauthorized_Spike", - "error_signature": "jwt.exceptions.InvalidAlgorithmError: The specified alg value is not allowed" - }, - "commits": [ - { - "message": "config: widen token leeway to 60s", - "author": "l.becker@example.com", - "minutes_before_alert": 44, - "files": { - "config.json": [ - [ - "\"leeway_s\": 30", - "\"leeway_s\": 60" - ] - ] - } - }, - { - "message": "payments: send idempotency key with charges", - "author": "p.nair@example.com", - "minutes_before_alert": 28, - "files": { - "app/clients/payments.py": [ - [ - " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n", - " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n headers={\"Idempotency-Key\": f\"charge-{order_id}\"},\n" - ] - ] - } - }, - { - "message": "auth: prepare for ES256 signing keys", - "author": "a.chen@example.com", - "minutes_before_alert": 13, - "files": { - "app/auth.py": [ - [ - "_ALGORITHMS = [\"RS256\"]", - "_ALGORITHMS = [\"ES256\"]" - ] - ] - } - } - ], - "label": { - "culprit_index": 2, - "expected_runbook_id": "auth_failures" - } -} diff --git a/eval/cases/auth_issuer_mesh_hosts.json b/eval/cases/auth_issuer_mesh_hosts.json new file mode 100644 index 0000000..691d55d --- /dev/null +++ b/eval/cases/auth_issuer_mesh_hosts.json @@ -0,0 +1,89 @@ +{ + "id": "auth_issuer_mesh_hosts", + "category": "auth_rejection", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_401_Unauthorized_Spike", + "error_signature": "jwt.exceptions.InvalidIssuerError: Invalid issuer" + }, + "commits": [ + { + "message": "auth: strip whitespace around bearer token", + "author": "r.haddad@example.com", + "minutes_before_alert": 57, + "files": { + "app/auth.py": [ + [ + " scheme, _, token = headers.get(\"authorization\", \"\").partition(\" \")\n", + " scheme, _, token = headers.get(\"authorization\", \"\").strip().partition(\" \")\n token = token.strip()\n" + ] + ] + } + }, + { + "message": "auth: accept ES256 alongside RS256", + "author": "a.chen@example.com", + "minutes_before_alert": 44, + "files": { + "app/auth.py": [ + [ + "_ALGORITHMS = [\"RS256\"]", + "_ALGORITHMS = [\"RS256\", \"ES256\"]" + ] + ] + } + }, + { + "message": "config: tighten token leeway", + "author": "m.okafor@example.com", + "minutes_before_alert": 31, + "files": { + "config.json": [ + [ + "\"leeway_s\": 30", + "\"leeway_s\": 10" + ] + ] + } + }, + { + "message": "db: tag connections with application_name", + "author": "j.silva@example.com", + "minutes_before_alert": 18, + "files": { + "app/db.py": [ + [ + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n", + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n application_name=\"orders-service\",\n" + ] + ] + } + }, + { + "message": "config: move internal service URLs to mesh hostnames", + "author": "p.nair@example.com", + "minutes_before_alert": 5, + "files": { + "config.json": [ + [ + "\"https://payments.internal\"", + "\"https://payments.orders.svc.mesh\"" + ], + [ + "\"https://inventory.internal\"", + "\"https://inventory.orders.svc.mesh\"" + ], + [ + "\"issuer\": \"https://auth.internal\"", + "\"issuer\": \"https://auth.orders.svc.mesh\"" + ] + ] + } + } + ], + "label": { + "culprit_index": 4, + "expected_runbook_id": "auth_failures", + "note": "Tokens still carry iss=https://auth.internal; the mesh rename changed the expected issuer, not just a transport URL. Leeway affects exp/iat checks and the algorithm list only widens what is accepted." + } +} diff --git a/eval/cases/auth_leeway_zero_03.json b/eval/cases/auth_leeway_zero_03.json deleted file mode 100644 index 5a417f2..0000000 --- a/eval/cases/auth_leeway_zero_03.json +++ /dev/null @@ -1,67 +0,0 @@ -{ - "id": "auth_leeway_zero_03", - "category": "auth_rejection", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_401_Unauthorized_Spike", - "error_signature": "jwt.exceptions.ImmatureSignatureError: The token is not yet valid (iat)" - }, - "commits": [ - { - "message": "security: remove token leeway per audit finding SEC-212", - "author": "r.haddad@example.com", - "minutes_before_alert": 59, - "files": { - "config.json": [ - [ - "\"leeway_s\": 30", - "\"leeway_s\": 0" - ] - ] - } - }, - { - "message": "auth: strip whitespace around bearer token", - "author": "a.chen@example.com", - "minutes_before_alert": 42, - "files": { - "app/auth.py": [ - [ - " scheme, _, token = headers.get(\"authorization\", \"\").partition(\" \")\n", - " scheme, _, token = headers.get(\"authorization\", \"\").strip().partition(\" \")\n token = token.strip()\n" - ] - ] - } - }, - { - "message": "gateway: raise worker pool to 48", - "author": "m.okafor@example.com", - "minutes_before_alert": 26, - "files": { - "app/gateway.py": [ - [ - "max_workers=32", - "max_workers=48" - ] - ] - } - }, - { - "message": "docs: describe exports module", - "author": "j.silva@example.com", - "minutes_before_alert": 10, - "files": { - "README.md": [ - [ - "- `app/worker.py` fulfilment queue consumer\n", - "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 0, - "expected_runbook_id": "auth_failures" - } -} diff --git a/eval/cases/cache_contains_recursion.json b/eval/cases/cache_contains_recursion.json new file mode 100644 index 0000000..ab6912c --- /dev/null +++ b/eval/cases/cache_contains_recursion.json @@ -0,0 +1,89 @@ +{ + "id": "cache_contains_recursion", + "category": "infinite_recursion", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "RecursionError: maximum recursion depth exceeded" + }, + "commits": [ + { + "message": "cache: membership test respects expiry", + "author": "j.silva@example.com", + "minutes_before_alert": 57, + "files": { + "app/cache.py": [ + [ + " def get(self, key):\n", + " def __contains__(self, key):\n entry = self._data.get(key)\n return entry is not None and time.monotonic() <= entry[1]\n\n def get(self, key):\n" + ] + ] + } + }, + { + "message": "settings: add section() helper", + "author": "p.nair@example.com", + "minutes_before_alert": 44, + "files": { + "app/settings.py": [ + [ + " return _CONFIG[section][key]\n", + " return _CONFIG[section][key]\n\n\ndef section(name: str) -> dict:\n return dict(_CONFIG[name])\n" + ] + ] + } + }, + { + "message": "cache: one expiry check shared by get and membership", + "author": "l.becker@example.com", + "minutes_before_alert": 31, + "files": { + "app/cache.py": [ + [ + " entry = self._data.get(key)\n return entry is not None and time.monotonic() <= entry[1]\n", + " return self.get(key) is not None\n" + ], + [ + " def get(self, key):\n entry = self._data.get(key)\n if entry is None:\n return None\n", + " def get(self, key):\n if key not in self:\n return None\n entry = self._data.get(key)\n" + ] + ] + } + }, + { + "message": "inventory: send request id header", + "author": "r.haddad@example.com", + "minutes_before_alert": 18, + "files": { + "app/clients/inventory.py": [ + [ + "def get_stock(sku: str) -> int:\n", + "def get_stock(sku: str, request_id: str | None = None) -> int:\n" + ], + [ + " timeout=settings.get(\"inventory\", \"timeout_s\"),\n", + " headers={\"X-Request-Id\": request_id} if request_id else None,\n timeout=settings.get(\"inventory\", \"timeout_s\"),\n" + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "a.chen@example.com", + "minutes_before_alert": 5, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": null, + "note": "__contains__ now calls get() and get() starts with 'key not in self', so each calls the other. The earlier __contains__ read _data directly and was fine." + } +} diff --git a/eval/cases/cache_recursion_04.json b/eval/cases/cache_recursion_04.json deleted file mode 100644 index ac3e898..0000000 --- a/eval/cases/cache_recursion_04.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "id": "cache_recursion_04", - "category": "infinite_recursion", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "RecursionError: maximum recursion depth exceeded" - }, - "commits": [ - { - "message": "config: shorten stock cache ttl", - "author": "r.haddad@example.com", - "minutes_before_alert": 47, - "files": { - "config.json": [ - [ - "\"ttl_s\": 300", - "\"ttl_s\": 240" - ] - ] - } - }, - { - "message": "cache: share expiry check between get and membership test", - "author": "m.okafor@example.com", - "minutes_before_alert": 33, - "files": { - "app/cache.py": [ - [ - " def get(self, key):\n entry = self._data.get(key)\n if entry is None:\n return None\n", - " def __contains__(self, key):\n return self.get(key) is not None\n\n def get(self, key):\n if key not in self:\n return None\n entry = self._data.get(key)\n" - ] - ] - } - }, - { - "message": "inventory: send request id header", - "author": "j.silva@example.com", - "minutes_before_alert": 16, - "files": { - "app/clients/inventory.py": [ - [ - "def get_stock(sku: str) -> int:\n", - "def get_stock(sku: str, request_id: str | None = None) -> int:\n" - ], - [ - " timeout=settings.get(\"inventory\", \"timeout_s\"),\n", - " headers={\"X-Request-Id\": request_id} if request_id else None,\n timeout=settings.get(\"inventory\", \"timeout_s\"),\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 1, - "expected_runbook_id": null - } -} diff --git a/eval/cases/cart_price_field_v2.json b/eval/cases/cart_price_field_v2.json new file mode 100644 index 0000000..2de684d --- /dev/null +++ b/eval/cases/cart_price_field_v2.json @@ -0,0 +1,85 @@ +{ + "id": "cart_price_field_v2", + "category": "null_or_missing_field", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "TypeError: unsupported operand type(s) for *: 'int' and 'NoneType'" + }, + "commits": [ + { + "message": "docs: describe exports module", + "author": "p.nair@example.com", + "minutes_before_alert": 57, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + }, + { + "message": "orders: extract line total helper", + "author": "l.becker@example.com", + "minutes_before_alert": 44, + "files": { + "app/handlers/orders.py": [ + [ + " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n", + " total = sum(_line_total(line) for line in lines)\n" + ], + [ + " return {\"status\": 201, \"body\": {\"id\": order_id, \"total\": serializers.money(total)}}\n", + " return {\"status\": 201, \"body\": {\"id\": order_id, \"total\": serializers.money(total)}}\n\n\ndef _line_total(line: dict) -> int:\n return line[\"qty\"] * line[\"unit_price_cents\"]\n" + ] + ] + } + }, + { + "message": "cart: read prices from the v2 cart payload", + "author": "r.haddad@example.com", + "minutes_before_alert": 31, + "files": { + "app/cart.py": [ + [ + "\"unit_price_cents\": item[\"unit_price_cents\"]}", + "\"unit_price_cents\": item.get(\"price_cents\")}" + ] + ] + } + }, + { + "message": "serializers: thousands separator in money", + "author": "a.chen@example.com", + "minutes_before_alert": 18, + "files": { + "app/serializers.py": [ + [ + " return str((Decimal(cents) / 100).quantize(Decimal(\"0.01\")))\n", + " return f\"{(Decimal(cents) / 100).quantize(Decimal('0.01')):,}\"\n" + ] + ] + } + }, + { + "message": "worker: shorter idle poll", + "author": "m.okafor@example.com", + "minutes_before_alert": 5, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": "null_field_handling", + "note": "Carts saved before the v2 payload only carry unit_price_cents, so .get('price_cents') yields None and qty * None raises inside _line_total. The helper extraction is behaviour-preserving." + } +} diff --git a/eval/cases/cart_sku_lowercased.json b/eval/cases/cart_sku_lowercased.json new file mode 100644 index 0000000..7117ee9 --- /dev/null +++ b/eval/cases/cart_sku_lowercased.json @@ -0,0 +1,72 @@ +{ + "id": "cart_sku_lowercased", + "category": "identifier_format", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "httpx.HTTPStatusError: Client error '404 Not Found' for url 'https://inventory.internal/v2/stock/sku-10293'" + }, + "commits": [ + { + "message": "inventory: url-encode SKUs in stock lookups", + "author": "l.becker@example.com", + "minutes_before_alert": 57, + "files": { + "app/clients/inventory.py": [ + [ + "import httpx\n", + "from urllib.parse import quote\n\nimport httpx\n" + ], + [ + "/v2/stock/{sku}\",", + "/v2/stock/{quote(sku)}\"," + ] + ] + } + }, + { + "message": "cache: type hints", + "author": "r.haddad@example.com", + "minutes_before_alert": 40, + "files": { + "app/cache.py": [ + [ + " def get(self, key):\n", + " def get(self, key: str):\n" + ] + ] + } + }, + { + "message": "cart: normalise SKU casing", + "author": "a.chen@example.com", + "minutes_before_alert": 22, + "files": { + "app/cart.py": [ + [ + "{\"sku\": item[\"sku\"], ", + "{\"sku\": item[\"sku\"].strip().lower(), " + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "m.okafor@example.com", + "minutes_before_alert": 5, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": null, + "note": "SKUs are upper-case (SKU-10293); the cart now lower-cases them before the stock lookup. quote() leaves 'SKU-10293' unchanged." + } +} diff --git a/eval/cases/cfg_env_override_string_03.json b/eval/cases/cfg_env_override_string_03.json deleted file mode 100644 index f21b317..0000000 --- a/eval/cases/cfg_env_override_string_03.json +++ /dev/null @@ -1,80 +0,0 @@ -{ - "id": "cfg_env_override_string_03", - "category": "config_type_error", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "TypeError: '<' not supported between instances of 'str' and 'int'" - }, - "commits": [ - { - "message": "settings: read env overrides as plain strings", - "author": "p.nair@example.com", - "minutes_before_alert": 57, - "files": { - "app/settings.py": [ - [ - " return json.loads(os.environ[env_key])\n", - " return os.environ[env_key]\n" - ] - ] - } - }, - { - "message": "pagination: document page_size bounds", - "author": "j.silva@example.com", - "minutes_before_alert": 46, - "files": { - "app/pagination.py": [ - [ - "def clamp_page_size(requested: int | None) -> int:\n", - "def clamp_page_size(requested: int | None) -> int:\n \"\"\"Requested size clamped to [1, api.max_page_size]; api.page_size when omitted.\"\"\"\n" - ] - ] - } - }, - { - "message": "orders: order history by id only", - "author": "a.chen@example.com", - "minutes_before_alert": 31, - "files": { - "app/handlers/orders.py": [ - [ - "ORDER BY created_at DESC, id DESC\"", - "ORDER BY id DESC\"" - ] - ] - } - }, - { - "message": "worker: shorter idle poll", - "author": "l.becker@example.com", - "minutes_before_alert": 18, - "files": { - "app/worker.py": [ - [ - "timeout=5)", - "timeout=2)" - ] - ] - } - }, - { - "message": "cache: type hints", - "author": "r.haddad@example.com", - "minutes_before_alert": 6, - "files": { - "app/cache.py": [ - [ - " def get(self, key):\n", - " def get(self, key: str):\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 0, - "expected_runbook_id": "config_parse_errors" - } -} diff --git a/eval/cases/cfg_json_trailing_comma_02.json b/eval/cases/cfg_json_trailing_comma_02.json deleted file mode 100644 index faba877..0000000 --- a/eval/cases/cfg_json_trailing_comma_02.json +++ /dev/null @@ -1,71 +0,0 @@ -{ - "id": "cfg_json_trailing_comma_02", - "category": "config_type_error", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_503_Service_Unavailable", - "error_signature": "json.decoder.JSONDecodeError: Expecting property name enclosed in double quotes: line 10 column 55 (char 599)" - }, - "commits": [ - { - "message": "config: add feature toggles section", - "author": "l.becker@example.com", - "minutes_before_alert": 52, - "files": { - "config.json": [ - [ - " \"exports\": {\"dir\": \"/var/lib/orders/exports\"}\n", - " \"exports\": {\"dir\": \"/var/lib/orders/exports\"},\n \"features\": {\"live_stock\": true, \"csv_export\": true,}\n" - ] - ] - } - }, - { - "message": "settings: add section() helper", - "author": "a.chen@example.com", - "minutes_before_alert": 38, - "files": { - "app/settings.py": [ - [ - " return _CONFIG[section][key]\n", - " return _CONFIG[section][key]\n\n\ndef section(name: str) -> dict:\n return dict(_CONFIG[name])\n" - ] - ] - } - }, - { - "message": "exports: include currency column", - "author": "r.haddad@example.com", - "minutes_before_alert": 24, - "files": { - "app/exports.py": [ - [ - "writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])", - "writer.writerow([\"id\", \"status\", \"total\", \"currency\", \"created_at\"])" - ], - [ - "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])", - "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"currency\"], o[\"created_at\"]])" - ] - ] - } - }, - { - "message": "docs: local development notes", - "author": "m.okafor@example.com", - "minutes_before_alert": 9, - "files": { - "README.md": [ - [ - "(JSON-encoded)\n", - "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 0, - "expected_runbook_id": "config_parse_errors" - } -} diff --git a/eval/cases/cfg_renamed_gateway_key_04.json b/eval/cases/cfg_renamed_gateway_key_04.json deleted file mode 100644 index d36d49b..0000000 --- a/eval/cases/cfg_renamed_gateway_key_04.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "id": "cfg_renamed_gateway_key_04", - "category": "config_type_error", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "KeyError: 'upstream_timeout_s'" - }, - "commits": [ - { - "message": "docs: describe exports module", - "author": "m.okafor@example.com", - "minutes_before_alert": 50, - "files": { - "README.md": [ - [ - "- `app/worker.py` fulfilment queue consumer\n", - "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" - ] - ] - } - }, - { - "message": "gateway: raise worker pool to 48", - "author": "p.nair@example.com", - "minutes_before_alert": 35, - "files": { - "app/gateway.py": [ - [ - "max_workers=32", - "max_workers=48" - ] - ] - } - }, - { - "message": "config: drop unit suffixes from gateway keys", - "author": "a.chen@example.com", - "minutes_before_alert": 14, - "files": { - "config.json": [ - [ - "\"gateway\": {\"upstream_timeout_s\": 15}", - "\"gateway\": {\"upstream_timeout\": 15}" - ] - ] - } - } - ], - "label": { - "culprit_index": 2, - "expected_runbook_id": "config_parse_errors" - } -} diff --git a/eval/cases/checkout_deadline_extra_call.json b/eval/cases/checkout_deadline_extra_call.json new file mode 100644 index 0000000..72e4b01 --- /dev/null +++ b/eval/cases/checkout_deadline_extra_call.json @@ -0,0 +1,85 @@ +{ + "id": "checkout_deadline_extra_call", + "category": "latency_timeout", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_504_Gateway_Timeout", + "error_signature": "504 Gateway Timeout: POST /v1/orders exceeded upstream deadline; p50 latency 9.6s -> 15.8s, inventory.get_stock p50 unchanged" + }, + "commits": [ + { + "message": "config: give inventory more headroom during reindex", + "author": "j.silva@example.com", + "minutes_before_alert": 57, + "files": { + "config.json": [ + [ + "\"timeout_s\": 3}", + "\"timeout_s\": 5}" + ] + ] + } + }, + { + "message": "config: one more payments retry", + "author": "p.nair@example.com", + "minutes_before_alert": 44, + "files": { + "config.json": [ + [ + "\"retries\": 2}", + "\"retries\": 3}" + ] + ] + } + }, + { + "message": "gateway: raise worker pool to 48", + "author": "l.becker@example.com", + "minutes_before_alert": 31, + "files": { + "app/gateway.py": [ + [ + "max_workers=32", + "max_workers=48" + ] + ] + } + }, + { + "message": "payments: check spend limit before charging", + "author": "r.haddad@example.com", + "minutes_before_alert": 18, + "files": { + "app/clients/payments.py": [ + [ + "def charge(order_id: str, amount_cents: int, card_token: str) -> dict:\n resp = httpx.post(\n", + "def charge(order_id: str, amount_cents: int, card_token: str) -> dict:\n limits = httpx.get(\n f\"{settings.get('payments', 'base_url')}/v1/limits\",\n params={\"order_id\": order_id},\n timeout=settings.get(\"payments\", \"timeout_s\"),\n )\n limits.raise_for_status()\n if amount_cents > limits.json()[\"remaining_cents\"]:\n raise PermissionError(\"spend limit reached\")\n resp = httpx.post(\n" + ] + ] + } + }, + { + "message": "exports: include currency column", + "author": "a.chen@example.com", + "minutes_before_alert": 5, + "files": { + "app/exports.py": [ + [ + "writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])", + "writer.writerow([\"id\", \"status\", \"total\", \"currency\", \"created_at\"])" + ], + [ + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])", + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"currency\"], o[\"created_at\"]])" + ] + ] + } + } + ], + "label": { + "culprit_index": 3, + "expected_runbook_id": "request_timeouts", + "note": "A second sequential payments round trip roughly doubles the payments share of checkout latency, pushing p50 past the 15s gateway deadline. Inventory p50 is unchanged, and payments.retries is never read by any code." + } +} diff --git a/eval/cases/config_reformat_hidden_string.json b/eval/cases/config_reformat_hidden_string.json new file mode 100644 index 0000000..2e67eea --- /dev/null +++ b/eval/cases/config_reformat_hidden_string.json @@ -0,0 +1,76 @@ +{ + "id": "config_reformat_hidden_string", + "category": "config_type_error", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "TypeError: '>=' not supported between instances of 'int' and 'str'" + }, + "commits": [ + { + "message": "settings: add section() helper", + "author": "l.becker@example.com", + "minutes_before_alert": 57, + "files": { + "app/settings.py": [ + [ + " return _CONFIG[section][key]\n", + " return _CONFIG[section][key]\n\n\ndef section(name: str) -> dict:\n return dict(_CONFIG[name])\n" + ] + ] + } + }, + { + "message": "config: one key per line for reviewable diffs", + "author": "r.haddad@example.com", + "minutes_before_alert": 44, + "files": { + "config.json": "{\n \"db\": {\n \"host\": \"10.0.4.12\",\n \"port\": 5432,\n \"name\": \"orders\",\n \"pool_size\": 10,\n \"connect_timeout_s\": 5\n },\n \"payments\": {\n \"base_url\": \"https://payments.internal\",\n \"timeout_s\": 8,\n \"retries\": 2\n },\n \"inventory\": {\n \"base_url\": \"https://inventory.internal\",\n \"timeout_s\": 3\n },\n \"gateway\": {\n \"upstream_timeout_s\": 15\n },\n \"cache\": {\n \"ttl_s\": 300,\n \"max_entries\": \"5000\"\n },\n \"api\": {\n \"page_size\": 50,\n \"max_page_size\": 200\n },\n \"auth\": {\n \"issuer\": \"https://auth.internal\",\n \"audience\": \"orders-api\",\n \"leeway_s\": 30\n },\n \"exports\": {\n \"dir\": \"/var/lib/orders/exports\"\n }\n}\n" + } + }, + { + "message": "config: longer stock cache ttl", + "author": "a.chen@example.com", + "minutes_before_alert": 31, + "files": { + "config.json": [ + [ + "\"ttl_s\": 300", + "\"ttl_s\": 600" + ] + ] + } + }, + { + "message": "cache: type hints", + "author": "m.okafor@example.com", + "minutes_before_alert": 18, + "files": { + "app/cache.py": [ + [ + " def get(self, key):\n", + " def get(self, key: str):\n" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "j.silva@example.com", + "minutes_before_alert": 5, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 1, + "expected_runbook_id": "config_parse_errors", + "note": "The one-key-per-line reformat quietly turned cache.max_entries into the string \"5000\"; TTLCache.set compares len(...) >= max_entries. The TTL change keeps an int." + } +} diff --git a/eval/cases/db_fetch_one_leak.json b/eval/cases/db_fetch_one_leak.json new file mode 100644 index 0000000..60e2cf9 --- /dev/null +++ b/eval/cases/db_fetch_one_leak.json @@ -0,0 +1,81 @@ +{ + "id": "db_fetch_one_leak", + "category": "db_pool_exhaustion", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "psycopg2.pool.PoolError: connection pool exhausted" + }, + "commits": [ + { + "message": "db: tag connections with application_name", + "author": "a.chen@example.com", + "minutes_before_alert": 57, + "files": { + "app/db.py": [ + [ + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n", + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n application_name=\"orders-service\",\n" + ] + ] + } + }, + { + "message": "db: fetch_one without materialising every row", + "author": "m.okafor@example.com", + "minutes_before_alert": 44, + "files": { + "app/db.py": [ + [ + "def fetch_one(sql: str, params: tuple = ()):\n rows = fetch_all(sql, params)\n return rows[0] if rows else None\n", + "def fetch_one(sql: str, params: tuple = ()):\n pool = get_pool()\n conn = pool.getconn()\n with conn.cursor() as cur:\n cur.execute(sql, params)\n row = cur.fetchone()\n if row is None:\n return None\n pool.putconn(conn)\n return row\n" + ] + ] + } + }, + { + "message": "config: shrink db pool to fit pgbouncer per-client limit", + "author": "j.silva@example.com", + "minutes_before_alert": 31, + "files": { + "config.json": [ + [ + "\"pool_size\": 10", + "\"pool_size\": 6" + ] + ] + } + }, + { + "message": "orders: name the stock shortfall list", + "author": "p.nair@example.com", + "minutes_before_alert": 18, + "files": { + "app/handlers/orders.py": [ + [ + " short = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if short:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": short}}\n", + " shortfall = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if shortfall:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": shortfall}}\n" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "l.becker@example.com", + "minutes_before_alert": 5, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 1, + "expected_runbook_id": "db_failures", + "note": "fetch_one returns before putconn on every miss (and on any exception), leaking one connection per lookup that finds nothing; a pool of 6 only lowers the ceiling, it cannot be exhausted without a leak." + } +} diff --git a/eval/cases/db_pool_leak_01.json b/eval/cases/db_pool_leak_01.json deleted file mode 100644 index 2271b44..0000000 --- a/eval/cases/db_pool_leak_01.json +++ /dev/null @@ -1,67 +0,0 @@ -{ - "id": "db_pool_leak_01", - "category": "db_pool_exhaustion", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "psycopg2.pool.PoolError: connection pool exhausted" - }, - "commits": [ - { - "message": "db: simplify fetch_all", - "author": "p.nair@example.com", - "minutes_before_alert": 58, - "files": { - "app/db.py": [ - [ - " conn = pool.getconn()\n try:\n with conn.cursor() as cur:\n cur.execute(sql, params)\n return cur.fetchall()\n finally:\n pool.putconn(conn)\n", - " conn = pool.getconn()\n with conn.cursor() as cur:\n cur.execute(sql, params)\n rows = cur.fetchall()\n pool.putconn(conn)\n return rows\n" - ] - ] - } - }, - { - "message": "config: lower db pool size to fit pgbouncer limits", - "author": "l.becker@example.com", - "minutes_before_alert": 44, - "files": { - "config.json": [ - [ - "\"pool_size\": 10", - "\"pool_size\": 8" - ] - ] - } - }, - { - "message": "orders: name the stock shortfall list", - "author": "j.silva@example.com", - "minutes_before_alert": 29, - "files": { - "app/handlers/orders.py": [ - [ - " short = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if short:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": short}}\n", - " shortfall = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if shortfall:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": shortfall}}\n" - ] - ] - } - }, - { - "message": "docs: describe exports module", - "author": "r.haddad@example.com", - "minutes_before_alert": 14, - "files": { - "README.md": [ - [ - "- `app/worker.py` fulfilment queue consumer\n", - "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 0, - "expected_runbook_id": "db_failures" - } -} diff --git a/eval/cases/deps_psycopg3_01.json b/eval/cases/deps_psycopg3_01.json deleted file mode 100644 index ae76e3a..0000000 --- a/eval/cases/deps_psycopg3_01.json +++ /dev/null @@ -1,67 +0,0 @@ -{ - "id": "deps_psycopg3_01", - "category": "dependency_version", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_503_Service_Unavailable", - "error_signature": "ModuleNotFoundError: No module named 'psycopg2'" - }, - "commits": [ - { - "message": "deps: bump PyJWT to 2.9.0", - "author": "p.nair@example.com", - "minutes_before_alert": 55, - "files": { - "requirements.txt": [ - [ - "PyJWT==2.8.0", - "PyJWT==2.9.0" - ] - ] - } - }, - { - "message": "db: tag connections with application_name", - "author": "l.becker@example.com", - "minutes_before_alert": 40, - "files": { - "app/db.py": [ - [ - " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n", - " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n application_name=\"orders-service\",\n" - ] - ] - } - }, - { - "message": "deps: move to psycopg 3", - "author": "j.silva@example.com", - "minutes_before_alert": 25, - "files": { - "requirements.txt": [ - [ - "psycopg2-binary==2.9.9", - "psycopg[binary]==3.2.1" - ] - ] - } - }, - { - "message": "cache: type hints", - "author": "a.chen@example.com", - "minutes_before_alert": 10, - "files": { - "app/cache.py": [ - [ - " def get(self, key):\n", - " def get(self, key: str):\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 2, - "expected_runbook_id": "dependency_regressions" - } -} diff --git a/eval/cases/deps_redis_legacy_02.json b/eval/cases/deps_redis_legacy_02.json deleted file mode 100644 index 0f954b3..0000000 --- a/eval/cases/deps_redis_legacy_02.json +++ /dev/null @@ -1,71 +0,0 @@ -{ - "id": "deps_redis_legacy_02", - "category": "dependency_version", - "base": "orders_service", - "alert": { - "alert_name": "FulfilmentWorker_Crashloop", - "error_signature": "redis.exceptions.ResponseError: value is not an integer or out of range" - }, - "commits": [ - { - "message": "worker: shorter idle poll", - "author": "m.okafor@example.com", - "minutes_before_alert": 50, - "files": { - "app/worker.py": [ - [ - "timeout=5)", - "timeout=2)" - ] - ] - } - }, - { - "message": "deps: pin redis to 2.10.6 to match fulfilment host image", - "author": "r.haddad@example.com", - "minutes_before_alert": 36, - "files": { - "requirements.txt": [ - [ - "redis==5.0.8", - "redis==2.10.6" - ] - ] - } - }, - { - "message": "deps: bump PyJWT to 2.9.0", - "author": "p.nair@example.com", - "minutes_before_alert": 22, - "files": { - "requirements.txt": [ - [ - "PyJWT==2.8.0", - "PyJWT==2.9.0" - ] - ] - } - }, - { - "message": "exports: include currency column", - "author": "l.becker@example.com", - "minutes_before_alert": 8, - "files": { - "app/exports.py": [ - [ - "writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])", - "writer.writerow([\"id\", \"status\", \"total\", \"currency\", \"created_at\"])" - ], - [ - "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])", - "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"currency\"], o[\"created_at\"]])" - ] - ] - } - } - ], - "label": { - "culprit_index": 1, - "expected_runbook_id": "dependency_regressions" - } -} diff --git a/eval/cases/export_tempfile_never_closed.json b/eval/cases/export_tempfile_never_closed.json new file mode 100644 index 0000000..6668fbd --- /dev/null +++ b/eval/cases/export_tempfile_never_closed.json @@ -0,0 +1,90 @@ +{ + "id": "export_tempfile_never_closed", + "category": "file_handle_leak", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "OSError: [Errno 24] Too many open files" + }, + "commits": [ + { + "message": "audit: append order events to a local audit log", + "author": "j.silva@example.com", + "minutes_before_alert": 57, + "files": { + "app/audit.py": "import json\nimport os\nfrom datetime import datetime, timezone\n\nAUDIT_LOG = os.environ.get(\"ORDERS_AUDIT_LOG\", \"/var/log/orders/audit.jsonl\")\n\n\ndef record(event: str, subject_id: str):\n line = {\"ts\": datetime.now(timezone.utc).isoformat(), \"event\": event, \"subject\": subject_id}\n with open(AUDIT_LOG, \"a\") as fh:\n fh.write(json.dumps(line) + \"\\n\")\n", + "app/handlers/orders.py": [ + [ + "from app import db, serializers\n", + "from app import audit, db, serializers\n" + ], + [ + " payments.charge(order_id, total, card_token)\n", + " payments.charge(order_id, total, card_token)\n audit.record(\"order.created\", order_id)\n" + ] + ] + } + }, + { + "message": "exports: write atomically via temp file and rename", + "author": "p.nair@example.com", + "minutes_before_alert": 44, + "files": { + "app/exports.py": [ + [ + "import os\n", + "import os\nimport tempfile\n" + ], + [ + " with open(path, \"w\", newline=\"\") as fh:\n writer = csv.writer(fh)\n writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])\n for o in orders:\n writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])\n return path\n", + " tmp = tempfile.NamedTemporaryFile(\"w\", dir=os.path.dirname(path), suffix=\".tmp\", delete=False, newline=\"\")\n writer = csv.writer(tmp)\n writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])\n for o in orders:\n writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])\n tmp.flush()\n os.replace(tmp.name, path)\n return path\n" + ] + ] + } + }, + { + "message": "config: move exports to shared volume", + "author": "l.becker@example.com", + "minutes_before_alert": 31, + "files": { + "config.json": [ + [ + "\"/var/lib/orders/exports\"", + "\"/mnt/shared/orders/exports\"" + ] + ] + } + }, + { + "message": "deps: bump PyJWT to 2.9.0", + "author": "r.haddad@example.com", + "minutes_before_alert": 18, + "files": { + "requirements.txt": [ + [ + "PyJWT==2.8.0", + "PyJWT==2.9.0" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "a.chen@example.com", + "minutes_before_alert": 5, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 1, + "expected_runbook_id": "file_handle_exhaustion", + "note": "The temp-file rewrite never closes the NamedTemporaryFile, leaking a descriptor per export. The audit log opens a file per order but closes it via the with block." + } +} diff --git a/eval/cases/fd_export_writer_01.json b/eval/cases/fd_export_writer_01.json deleted file mode 100644 index 8b4986a..0000000 --- a/eval/cases/fd_export_writer_01.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "id": "fd_export_writer_01", - "category": "file_handle_leak", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "OSError: [Errno 24] Too many open files: '/mnt/shared/orders/exports/orders_20260901T114702.csv'" - }, - "commits": [ - { - "message": "config: move exports to shared volume", - "author": "r.haddad@example.com", - "minutes_before_alert": 41, - "files": { - "config.json": [ - [ - "\"/var/lib/orders/exports\"", - "\"/mnt/shared/orders/exports\"" - ] - ] - } - }, - { - "message": "worker: shorter idle poll", - "author": "m.okafor@example.com", - "minutes_before_alert": 26, - "files": { - "app/worker.py": [ - [ - "timeout=5)", - "timeout=2)" - ] - ] - } - }, - { - "message": "exports: stream rows through a reusable writer", - "author": "a.chen@example.com", - "minutes_before_alert": 9, - "files": { - "app/exports.py": [ - [ - " with open(path, \"w\", newline=\"\") as fh:\n writer = csv.writer(fh)\n writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])\n for o in orders:\n writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])\n return path\n", - " writer = _open_writer(path)\n writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])\n for o in orders:\n writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])\n return path\n\n\ndef _open_writer(path: str):\n return csv.writer(open(path, \"w\", newline=\"\"))\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 2, - "expected_runbook_id": "file_handle_exhaustion" - } -} diff --git a/eval/cases/null_guest_email_01.json b/eval/cases/guest_email_normalise.json similarity index 51% rename from eval/cases/null_guest_email_01.json rename to eval/cases/guest_email_normalise.json index a75a86f..7c32011 100644 --- a/eval/cases/null_guest_email_01.json +++ b/eval/cases/guest_email_normalise.json @@ -1,42 +1,42 @@ { - "id": "null_guest_email_01", + "id": "guest_email_normalise", "category": "null_or_missing_field", "base": "orders_service", "alert": { "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "AttributeError: 'NoneType' object has no attribute 'split'" + "error_signature": "AttributeError: 'NoneType' object has no attribute 'strip'" }, "commits": [ { - "message": "serializers: thousands separator in money", - "author": "a.chen@example.com", - "minutes_before_alert": 56, + "message": "orders: expose customer email domain for B2B reporting", + "author": "l.becker@example.com", + "minutes_before_alert": 57, "files": { "app/serializers.py": [ [ - " return str((Decimal(cents) / 100).quantize(Decimal(\"0.01\")))\n", - " return f\"{(Decimal(cents) / 100).quantize(Decimal('0.01')):,}\"\n" + " \"email\": row[\"customer_email\"],\n", + " \"email\": row[\"customer_email\"],\n \"email_domain\": (row[\"customer_email\"] or \"\").partition(\"@\")[2],\n" ] ] } }, { - "message": "orders: expose customer email domain for B2B reporting", - "author": "p.nair@example.com", - "minutes_before_alert": 39, + "message": "pagination: document page_size bounds", + "author": "r.haddad@example.com", + "minutes_before_alert": 44, "files": { - "app/serializers.py": [ + "app/pagination.py": [ [ - " \"email\": row[\"customer_email\"],\n", - " \"email\": row[\"customer_email\"],\n \"email_domain\": row[\"customer_email\"].split(\"@\")[1].lower(),\n" + "def clamp_page_size(requested: int | None) -> int:\n", + "def clamp_page_size(requested: int | None) -> int:\n \"\"\"Requested size clamped to [1, api.max_page_size]; api.page_size when omitted.\"\"\"\n" ] ] } }, { "message": "exports: include currency column", - "author": "l.becker@example.com", - "minutes_before_alert": 23, + "author": "a.chen@example.com", + "minutes_before_alert": 31, "files": { "app/exports.py": [ [ @@ -51,21 +51,35 @@ } }, { - "message": "docs: describe exports module", - "author": "r.haddad@example.com", - "minutes_before_alert": 10, + "message": "cache: type hints", + "author": "m.okafor@example.com", + "minutes_before_alert": 18, + "files": { + "app/cache.py": [ + [ + " def get(self, key):\n", + " def get(self, key: str):\n" + ] + ] + } + }, + { + "message": "serializers: normalise customer emails", + "author": "j.silva@example.com", + "minutes_before_alert": 5, "files": { - "README.md": [ + "app/serializers.py": [ [ - "- `app/worker.py` fulfilment queue consumer\n", - "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + " \"email\": row[\"customer_email\"],\n", + " \"email\": row[\"customer_email\"].strip().lower(),\n" ] ] } } ], "label": { - "culprit_index": 1, - "expected_runbook_id": "null_field_handling" + "culprit_index": 4, + "expected_runbook_id": "null_field_handling", + "note": "Guest orders have customer_email NULL; .strip() on None raises. The email_domain commit guards None." } } diff --git a/eval/cases/httpx_028_drops_proxies.json b/eval/cases/httpx_028_drops_proxies.json new file mode 100644 index 0000000..1f9356d --- /dev/null +++ b/eval/cases/httpx_028_drops_proxies.json @@ -0,0 +1,86 @@ +{ + "id": "httpx_028_drops_proxies", + "category": "dependency_version", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_503_Service_Unavailable", + "error_signature": "TypeError: Client.__init__() got an unexpected keyword argument 'proxies'" + }, + "commits": [ + { + "message": "payments: route charges through the egress proxy", + "author": "p.nair@example.com", + "minutes_before_alert": 57, + "files": { + "config.json": [ + [ + "\"retries\": 2}", + "\"retries\": 2, \"egress_proxy\": \"http://egress.internal:3128\"}" + ] + ], + "app/clients/payments.py": [ + [ + "from app import settings\n", + "from app import settings\n\n_http = httpx.Client(proxies=settings.get(\"payments\", \"egress_proxy\"))\n" + ], + [ + "def charge(order_id: str, amount_cents: int, card_token: str) -> dict:\n resp = httpx.post(\n", + "def charge(order_id: str, amount_cents: int, card_token: str) -> dict:\n resp = _http.post(\n" + ] + ] + } + }, + { + "message": "cache: track hit and miss counts", + "author": "l.becker@example.com", + "minutes_before_alert": 40, + "files": { + "app/cache.py": [ + [ + " self._data: dict = {}\n", + " self._data: dict = {}\n self.hits = 0\n self.misses = 0\n" + ], + [ + " entry = self._data.get(key)\n if entry is None:\n return None\n", + " entry = self._data.get(key)\n if entry is None:\n self.misses += 1\n return None\n" + ], + [ + " del self._data[key]\n return None\n return value\n", + " del self._data[key]\n self.misses += 1\n return None\n self.hits += 1\n return value\n" + ] + ] + } + }, + { + "message": "db: tag connections with application_name", + "author": "r.haddad@example.com", + "minutes_before_alert": 22, + "files": { + "app/db.py": [ + [ + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n", + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n application_name=\"orders-service\",\n" + ] + ] + } + }, + { + "message": "deps: bump httpx to 0.28.1", + "author": "a.chen@example.com", + "minutes_before_alert": 5, + "files": { + "requirements.txt": [ + [ + "httpx==0.27.2", + "httpx==0.28.1" + ] + ] + } + } + ], + "label": { + "culprit_index": 3, + "expected_runbook_id": "dependency_regressions", + "note": "httpx 0.28 removed the deprecated proxies= argument (replaced by proxy=). The proxy commit was valid against the pinned 0.27.2; the version bump is what broke it." + } +} diff --git a/eval/cases/inventory_cache_key_per_request.json b/eval/cases/inventory_cache_key_per_request.json new file mode 100644 index 0000000..7f689e4 --- /dev/null +++ b/eval/cases/inventory_cache_key_per_request.json @@ -0,0 +1,105 @@ +{ + "id": "inventory_cache_key_per_request", + "category": "rate_limit_429", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "httpx.HTTPStatusError: Client error '429 Too Many Requests' for url 'https://inventory.internal/v2/stock/SKU-30117' (inventory request rate 31x 7-day baseline)" + }, + "commits": [ + { + "message": "cache: track hit and miss counts", + "author": "m.okafor@example.com", + "minutes_before_alert": 57, + "files": { + "app/cache.py": [ + [ + " self._data: dict = {}\n", + " self._data: dict = {}\n self.hits = 0\n self.misses = 0\n" + ], + [ + " entry = self._data.get(key)\n if entry is None:\n return None\n", + " entry = self._data.get(key)\n if entry is None:\n self.misses += 1\n return None\n" + ], + [ + " del self._data[key]\n return None\n return value\n", + " del self._data[key]\n self.misses += 1\n return None\n self.hits += 1\n return value\n" + ] + ] + } + }, + { + "message": "config: fresher stock levels for flash sales", + "author": "j.silva@example.com", + "minutes_before_alert": 44, + "files": { + "config.json": [ + [ + "\"ttl_s\": 300", + "\"ttl_s\": 60" + ] + ] + } + }, + { + "message": "deps: bump PyJWT to 2.9.0", + "author": "p.nair@example.com", + "minutes_before_alert": 31, + "files": { + "requirements.txt": [ + [ + "PyJWT==2.8.0", + "PyJWT==2.9.0" + ] + ] + } + }, + { + "message": "docs: describe exports module", + "author": "l.becker@example.com", + "minutes_before_alert": 18, + "files": { + "README.md": [ + [ + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + ] + ] + } + }, + { + "message": "inventory: tag stock lookups with a request id", + "author": "r.haddad@example.com", + "minutes_before_alert": 5, + "files": { + "app/clients/inventory.py": [ + [ + "import httpx\n", + "import uuid\n\nimport httpx\n" + ], + [ + "def get_stock(sku: str) -> int:\n cached = _stock_cache.get(sku)\n", + "def get_stock(sku: str, request_id: str | None = None) -> int:\n cache_key = f\"{sku}:{request_id}\"\n cached = _stock_cache.get(cache_key)\n" + ], + [ + " timeout=settings.get(\"inventory\", \"timeout_s\"),\n", + " headers={\"X-Request-Id\": request_id} if request_id else None,\n timeout=settings.get(\"inventory\", \"timeout_s\"),\n" + ], + [ + " _stock_cache.set(sku, qty)\n", + " _stock_cache.set(cache_key, qty)\n" + ], + [ + "def get_stock_many(skus: list[str]) -> dict[str, int]:\n return {sku: get_stock(sku) for sku in skus}\n", + "def get_stock_many(skus: list[str]) -> dict[str, int]:\n request_id = uuid.uuid4().hex # correlate the batch in inventory logs\n return {sku: get_stock(sku, request_id) for sku in skus}\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 4, + "expected_runbook_id": "rate_limiting", + "note": "The cache key now includes a fresh uuid per batch, so the stock cache never hits; a 60s TTL alone can raise lookups at most ~5x over 300s, not 31x." + } +} diff --git a/eval/cases/cfg_cache_size_string_01.json b/eval/cases/inventory_v3_payload.json similarity index 58% rename from eval/cases/cfg_cache_size_string_01.json rename to eval/cases/inventory_v3_payload.json index 43d8cca..b893e40 100644 --- a/eval/cases/cfg_cache_size_string_01.json +++ b/eval/cases/inventory_v3_payload.json @@ -1,16 +1,33 @@ { - "id": "cfg_cache_size_string_01", - "category": "config_type_error", + "id": "inventory_v3_payload", + "category": "null_or_missing_field", "base": "orders_service", "alert": { "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "TypeError: '>=' not supported between instances of 'int' and 'str'" + "error_signature": "KeyError: 'available'" }, "commits": [ + { + "message": "inventory: send request id header", + "author": "m.okafor@example.com", + "minutes_before_alert": 57, + "files": { + "app/clients/inventory.py": [ + [ + "def get_stock(sku: str) -> int:\n", + "def get_stock(sku: str, request_id: str | None = None) -> int:\n" + ], + [ + " timeout=settings.get(\"inventory\", \"timeout_s\"),\n", + " headers={\"X-Request-Id\": request_id} if request_id else None,\n timeout=settings.get(\"inventory\", \"timeout_s\"),\n" + ] + ] + } + }, { "message": "cache: track hit and miss counts", - "author": "a.chen@example.com", - "minutes_before_alert": 55, + "author": "j.silva@example.com", + "minutes_before_alert": 44, "files": { "app/cache.py": [ [ @@ -29,39 +46,35 @@ } }, { - "message": "config: raise stock cache size for holiday catalog", - "author": "m.okafor@example.com", - "minutes_before_alert": 41, + "message": "inventory: move to the v3 stock API", + "author": "p.nair@example.com", + "minutes_before_alert": 31, "files": { - "config.json": [ + "app/clients/inventory.py": [ [ - "\"cache\": {\"ttl_s\": 300, \"max_entries\": 5000}", - "\"cache\": {\"ttl_s\": 300, \"max_entries\": \"20_000\"}" + "/v2/stock/{sku}\",", + "/v3/stock/{sku}\"," ] ] } }, { - "message": "inventory: send request id header", - "author": "j.silva@example.com", - "minutes_before_alert": 27, + "message": "orders: name the stock shortfall list", + "author": "l.becker@example.com", + "minutes_before_alert": 18, "files": { - "app/clients/inventory.py": [ + "app/handlers/orders.py": [ [ - "def get_stock(sku: str) -> int:\n", - "def get_stock(sku: str, request_id: str | None = None) -> int:\n" - ], - [ - " timeout=settings.get(\"inventory\", \"timeout_s\"),\n", - " headers={\"X-Request-Id\": request_id} if request_id else None,\n timeout=settings.get(\"inventory\", \"timeout_s\"),\n" + " short = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if short:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": short}}\n", + " shortfall = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if shortfall:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": shortfall}}\n" ] ] } }, { "message": "docs: describe exports module", - "author": "p.nair@example.com", - "minutes_before_alert": 12, + "author": "r.haddad@example.com", + "minutes_before_alert": 5, "files": { "README.md": [ [ @@ -73,7 +86,8 @@ } ], "label": { - "culprit_index": 1, - "expected_runbook_id": "config_parse_errors" + "culprit_index": 2, + "expected_runbook_id": "null_field_handling", + "note": "The response parser still reads ['available'] while the client now calls a different API version; the only change to what the response contains. The other inventory and order commits never touch parsing." } } diff --git a/eval/cases/auth_audience_rename_01.json b/eval/cases/jwt_leeway_milliseconds.json similarity index 51% rename from eval/cases/auth_audience_rename_01.json rename to eval/cases/jwt_leeway_milliseconds.json index e203729..cee8aa4 100644 --- a/eval/cases/auth_audience_rename_01.json +++ b/eval/cases/jwt_leeway_milliseconds.json @@ -1,16 +1,16 @@ { - "id": "auth_audience_rename_01", + "id": "jwt_leeway_milliseconds", "category": "auth_rejection", "base": "orders_service", "alert": { "alert_name": "HTTP_401_Unauthorized_Spike", - "error_signature": "jwt.exceptions.InvalidAudienceError: Audience doesn't match" + "error_signature": "jwt.exceptions.ImmatureSignatureError: The token is not yet valid (iat); api hosts ~4s behind auth.internal" }, "commits": [ { "message": "auth: strip whitespace around bearer token", "author": "j.silva@example.com", - "minutes_before_alert": 53, + "minutes_before_alert": 57, "files": { "app/auth.py": [ [ @@ -21,51 +21,69 @@ } }, { - "message": "docs: local development notes", - "author": "m.okafor@example.com", - "minutes_before_alert": 37, + "message": "config: widen token leeway", + "author": "p.nair@example.com", + "minutes_before_alert": 44, "files": { - "README.md": [ + "config.json": [ + [ + "\"leeway_s\": 30", + "\"leeway_s\": 45" + ] + ] + } + }, + { + "message": "exports: include currency column", + "author": "l.becker@example.com", + "minutes_before_alert": 31, + "files": { + "app/exports.py": [ + [ + "writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])", + "writer.writerow([\"id\", \"status\", \"total\", \"currency\", \"created_at\"])" + ], [ - "(JSON-encoded)\n", - "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])", + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"currency\"], o[\"created_at\"]])" ] ] } }, { - "message": "config: align service identifiers with service catalog", + "message": "auth: pass leeway to PyJWT as a timedelta", "author": "r.haddad@example.com", - "minutes_before_alert": 21, + "minutes_before_alert": 18, "files": { - "config.json": [ + "app/auth.py": [ [ - "\"audience\": \"orders-api\"", - "\"audience\": \"orders\"" + "import jwt\n", + "from datetime import timedelta\n\nimport jwt\n" + ], + [ + " leeway=settings.get(\"auth\", \"leeway_s\"),\n", + " leeway=timedelta(milliseconds=settings.get(\"auth\", \"leeway_s\")),\n" ] ] } }, { - "message": "exports: include currency column", + "message": "deps: bump PyJWT to 2.9.0", "author": "a.chen@example.com", "minutes_before_alert": 5, "files": { - "app/exports.py": [ + "requirements.txt": [ [ - "writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])", - "writer.writerow([\"id\", \"status\", \"total\", \"currency\", \"created_at\"])" - ], - [ - "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])", - "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"currency\"], o[\"created_at\"]])" + "PyJWT==2.8.0", + "PyJWT==2.9.0" ] ] } } ], "label": { - "culprit_index": 2, - "expected_runbook_id": "auth_failures" + "culprit_index": 3, + "expected_runbook_id": "auth_failures", + "note": "leeway_s is seconds but is now wrapped as milliseconds (45ms), so ~4s of clock skew fails the iat check. Widening leeway_s to 45 would tolerate it; the PyJWT minor bump does not change leeway handling." } } diff --git a/eval/cases/ser_decimal_json_02.json b/eval/cases/money_returns_decimal.json similarity index 53% rename from eval/cases/ser_decimal_json_02.json rename to eval/cases/money_returns_decimal.json index fc80b6f..cb2f9b6 100644 --- a/eval/cases/ser_decimal_json_02.json +++ b/eval/cases/money_returns_decimal.json @@ -1,5 +1,5 @@ { - "id": "ser_decimal_json_02", + "id": "money_returns_decimal", "category": "json_serialization", "base": "orders_service", "alert": { @@ -10,7 +10,7 @@ { "message": "payments: send idempotency key with charges", "author": "a.chen@example.com", - "minutes_before_alert": 52, + "minutes_before_alert": 57, "files": { "app/clients/payments.py": [ [ @@ -22,8 +22,8 @@ }, { "message": "serializers: thousands separator in money", - "author": "l.becker@example.com", - "minutes_before_alert": 38, + "author": "m.okafor@example.com", + "minutes_before_alert": 44, "files": { "app/serializers.py": [ [ @@ -34,26 +34,26 @@ } }, { - "message": "orders: compute totals in Decimal", - "author": "m.okafor@example.com", - "minutes_before_alert": 23, + "message": "exports: include currency column", + "author": "j.silva@example.com", + "minutes_before_alert": 31, "files": { - "app/handlers/orders.py": [ + "app/exports.py": [ [ - "from app import db, serializers\n", - "from decimal import Decimal\n\nfrom app import db, serializers\n" + "writer.writerow([\"id\", \"status\", \"total\", \"created_at\"])", + "writer.writerow([\"id\", \"status\", \"total\", \"currency\", \"created_at\"])" ], [ - " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n", - " total = sum(Decimal(line[\"qty\"]) * Decimal(line[\"unit_price_cents\"]) for line in lines)\n" + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"created_at\"]])", + "writer.writerow([o[\"id\"], o[\"status\"], o[\"total\"], o[\"currency\"], o[\"created_at\"]])" ] ] } }, { "message": "worker: shorter idle poll", - "author": "r.haddad@example.com", - "minutes_before_alert": 9, + "author": "p.nair@example.com", + "minutes_before_alert": 18, "files": { "app/worker.py": [ [ @@ -62,10 +62,24 @@ ] ] } + }, + { + "message": "serializers: return money as Decimal for callers doing arithmetic", + "author": "l.becker@example.com", + "minutes_before_alert": 5, + "files": { + "app/serializers.py": [ + [ + " return f\"{(Decimal(cents) / 100).quantize(Decimal('0.01')):,}\"\n", + " return (Decimal(cents) / 100).quantize(Decimal(\"0.01\"))\n" + ] + ] + } } ], "label": { - "culprit_index": 2, - "expected_runbook_id": null + "culprit_index": 4, + "expected_runbook_id": null, + "note": "money() now returns a Decimal that lands in every order response body. The thousands-separator change still returned a str." } } diff --git a/eval/cases/null_since_filter_03.json b/eval/cases/null_since_filter_03.json deleted file mode 100644 index c5ca672..0000000 --- a/eval/cases/null_since_filter_03.json +++ /dev/null @@ -1,80 +0,0 @@ -{ - "id": "null_since_filter_03", - "category": "null_or_missing_field", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "TypeError: '>=' not supported between instances of 'datetime.datetime' and 'NoneType'" - }, - "commits": [ - { - "message": "docs: describe exports module", - "author": "l.becker@example.com", - "minutes_before_alert": 58, - "files": { - "README.md": [ - [ - "- `app/worker.py` fulfilment queue consumer\n", - "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" - ] - ] - } - }, - { - "message": "serializers: accept Z suffix in since", - "author": "r.haddad@example.com", - "minutes_before_alert": 45, - "files": { - "app/serializers.py": [ - [ - " return datetime.fromisoformat(value) if value else None\n", - " return datetime.fromisoformat(value.replace(\"Z\", \"+00:00\")) if value else None\n" - ] - ] - } - }, - { - "message": "pagination: document page_size bounds", - "author": "a.chen@example.com", - "minutes_before_alert": 32, - "files": { - "app/pagination.py": [ - [ - "def clamp_page_size(requested: int | None) -> int:\n", - "def clamp_page_size(requested: int | None) -> int:\n \"\"\"Requested size clamped to [1, api.max_page_size]; api.page_size when omitted.\"\"\"\n" - ] - ] - } - }, - { - "message": "orders: filter order history by since", - "author": "p.nair@example.com", - "minutes_before_alert": 19, - "files": { - "app/handlers/orders.py": [ - [ - "def list_orders(customer_id: str, page: int = 1, page_size: int | None = None) -> dict:\n rows = db.fetch_all(_LIST_SQL, (customer_id,))\n", - "def list_orders(\n customer_id: str, page: int = 1, page_size: int | None = None, since: str | None = None\n) -> dict:\n rows = db.fetch_all(_LIST_SQL, (customer_id,))\n rows = [r for r in rows if r[\"created_at\"] >= serializers.parse_since(since)]\n" - ] - ] - } - }, - { - "message": "worker: shorter idle poll", - "author": "m.okafor@example.com", - "minutes_before_alert": 7, - "files": { - "app/worker.py": [ - [ - "timeout=5)", - "timeout=2)" - ] - ] - } - } - ], - "label": { - "culprit_index": 3, - "expected_runbook_id": "null_field_handling" - } -} diff --git a/eval/cases/null_stock_payload_02.json b/eval/cases/null_stock_payload_02.json deleted file mode 100644 index a325fd2..0000000 --- a/eval/cases/null_stock_payload_02.json +++ /dev/null @@ -1,66 +0,0 @@ -{ - "id": "null_stock_payload_02", - "category": "null_or_missing_field", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "TypeError: '<' not supported between instances of 'NoneType' and 'int'" - }, - "commits": [ - { - "message": "inventory: parse stock payload with .get", - "author": "j.silva@example.com", - "minutes_before_alert": 49, - "files": { - "app/clients/inventory.py": [ - [ - " qty = resp.json()[\"available\"]\n", - " qty = resp.json().get(\"available\")\n" - ] - ] - } - }, - { - "message": "orders: log order totals", - "author": "a.chen@example.com", - "minutes_before_alert": 34, - "files": { - "app/handlers/orders.py": [ - [ - "from app import db, serializers\n", - "import logging\n\nfrom app import db, serializers\n" - ], - [ - "_ORDER_SQL =", - "log = logging.getLogger(__name__)\n\n_ORDER_SQL =" - ], - [ - " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n", - " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n log.info(\"order total computed\", extra={\"customer_id\": customer_id, \"total_cents\": total})\n" - ] - ] - } - }, - { - "message": "payments: refund takes optional amount", - "author": "m.okafor@example.com", - "minutes_before_alert": 15, - "files": { - "app/clients/payments.py": [ - [ - "def refund(charge_id: str) -> dict:\n", - "def refund(charge_id: str, amount_cents: int | None = None) -> dict:\n" - ], - [ - " f\"{settings.get('payments', 'base_url')}/v1/charges/{charge_id}/refund\",\n", - " f\"{settings.get('payments', 'base_url')}/v1/charges/{charge_id}/refund\",\n json={\"amount_cents\": amount_cents} if amount_cents else None,\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 0, - "expected_runbook_id": "null_field_handling" - } -} diff --git a/eval/cases/order_read_sync_audit.json b/eval/cases/order_read_sync_audit.json new file mode 100644 index 0000000..9167d9f --- /dev/null +++ b/eval/cases/order_read_sync_audit.json @@ -0,0 +1,86 @@ +{ + "id": "order_read_sync_audit", + "category": "latency_timeout", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_504_Gateway_Timeout", + "error_signature": "504 Gateway Timeout: GET /v1/orders/{id}; p50 latency 40ms -> 10.0s, gateway worker utilisation 22%" + }, + "commits": [ + { + "message": "gateway: reduce worker threads to cut memory", + "author": "a.chen@example.com", + "minutes_before_alert": 57, + "files": { + "app/gateway.py": [ + [ + "max_workers=32", + "max_workers=16" + ] + ] + } + }, + { + "message": "serializers: accept Z suffix in since", + "author": "m.okafor@example.com", + "minutes_before_alert": 44, + "files": { + "app/serializers.py": [ + [ + " return datetime.fromisoformat(value) if value else None\n", + " return datetime.fromisoformat(value.replace(\"Z\", \"+00:00\")) if value else None\n" + ] + ] + } + }, + { + "message": "config: inventory timeout 10s while v2 rolls out", + "author": "j.silva@example.com", + "minutes_before_alert": 31, + "files": { + "config.json": [ + [ + "\"timeout_s\": 3}", + "\"timeout_s\": 10}" + ] + ] + } + }, + { + "message": "orders: audit order reads for compliance", + "author": "p.nair@example.com", + "minutes_before_alert": 18, + "files": { + "app/audit.py": "import httpx\n\nAUDIT_URL = \"https://audit.internal/v1/events\"\n\n\ndef record(event: str, subject_id: str):\n httpx.post(AUDIT_URL, json={\"event\": event, \"subject\": subject_id}, timeout=10.0)\n", + "app/handlers/orders.py": [ + [ + "from app import db, serializers\n", + "from app import audit, db, serializers\n" + ], + [ + "def get_order(order_id: str) -> dict:\n row = db.fetch_one(", + "def get_order(order_id: str) -> dict:\n audit.record(\"order.read\", order_id)\n row = db.fetch_one(" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "l.becker@example.com", + "minutes_before_alert": 5, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 3, + "expected_runbook_id": "request_timeouts", + "note": "get_order now makes a synchronous call with a 10s timeout on every read; p50 pinned at exactly 10.0s is that timeout. get_order never calls inventory, and 22% worker utilisation rules out thread starvation." + } +} diff --git a/eval/cases/sql_param_count_05.json b/eval/cases/order_sql_missing_param.json similarity index 57% rename from eval/cases/sql_param_count_05.json rename to eval/cases/order_sql_missing_param.json index 8fd9df5..c6ebc06 100644 --- a/eval/cases/sql_param_count_05.json +++ b/eval/cases/order_sql_missing_param.json @@ -1,5 +1,5 @@ { - "id": "sql_param_count_05", + "id": "order_sql_missing_param", "category": "sql_param_mismatch", "base": "orders_service", "alert": { @@ -7,10 +7,27 @@ "error_signature": "IndexError: tuple index out of range" }, "commits": [ + { + "message": "orders: hide cancelled orders from history", + "author": "p.nair@example.com", + "minutes_before_alert": 57, + "files": { + "app/handlers/orders.py": [ + [ + "\"SELECT * FROM orders_view WHERE customer_id = %s ORDER BY", + "\"SELECT * FROM orders_view WHERE customer_id = %s AND status <> %s ORDER BY" + ], + [ + " rows = db.fetch_all(_LIST_SQL, (customer_id,))\n", + " rows = db.fetch_all(_LIST_SQL, (customer_id, \"cancelled\"))\n" + ] + ] + } + }, { "message": "pagination: include total count", "author": "l.becker@example.com", - "minutes_before_alert": 53, + "minutes_before_alert": 44, "files": { "app/pagination.py": [ [ @@ -23,7 +40,7 @@ { "message": "db: tag connections with application_name", "author": "r.haddad@example.com", - "minutes_before_alert": 37, + "minutes_before_alert": 31, "files": { "app/db.py": [ [ @@ -34,27 +51,27 @@ } }, { - "message": "serializers: accept Z suffix in since", + "message": "orders: scope single-order reads to tenant", "author": "a.chen@example.com", - "minutes_before_alert": 22, + "minutes_before_alert": 18, "files": { - "app/serializers.py": [ + "app/handlers/orders.py": [ [ - " return datetime.fromisoformat(value) if value else None\n", - " return datetime.fromisoformat(value.replace(\"Z\", \"+00:00\")) if value else None\n" + "_ORDER_SQL = \"SELECT * FROM orders_view WHERE id = %s\"", + "_ORDER_SQL = \"SELECT * FROM orders_view WHERE id = %s AND tenant_id = %s\"" ] ] } }, { - "message": "orders: hide cancelled orders from history", - "author": "p.nair@example.com", - "minutes_before_alert": 6, + "message": "docs: local development notes", + "author": "m.okafor@example.com", + "minutes_before_alert": 5, "files": { - "app/handlers/orders.py": [ + "README.md": [ [ - "\"SELECT * FROM orders_view WHERE customer_id = %s ORDER BY", - "\"SELECT * FROM orders_view WHERE customer_id = %s AND status <> %s ORDER BY" + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" ] ] } @@ -62,6 +79,7 @@ ], "label": { "culprit_index": 3, - "expected_runbook_id": null + "expected_runbook_id": null, + "note": "_ORDER_SQL gained a second placeholder but get_order still passes (order_id,). The list query added a placeholder and its parameter together." } } diff --git a/eval/cases/page_newest_empty_01.json b/eval/cases/page_newest_empty_01.json deleted file mode 100644 index d5e3e28..0000000 --- a/eval/cases/page_newest_empty_01.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "id": "page_newest_empty_01", - "category": "pagination_boundary", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "IndexError: list index out of range" - }, - "commits": [ - { - "message": "pagination: include total count", - "author": "m.okafor@example.com", - "minutes_before_alert": 51, - "files": { - "app/pagination.py": [ - [ - " return {\"items\": items[start:end], \"page\": page, \"has_more\": end < len(items)}\n", - " return {\"items\": items[start:end], \"page\": page, \"has_more\": end < len(items), \"total\": len(items)}\n" - ] - ] - } - }, - { - "message": "serializers: accept Z suffix in since", - "author": "l.becker@example.com", - "minutes_before_alert": 30, - "files": { - "app/serializers.py": [ - [ - " return datetime.fromisoformat(value) if value else None\n", - " return datetime.fromisoformat(value.replace(\"Z\", \"+00:00\")) if value else None\n" - ] - ] - } - }, - { - "message": "orders: include newest order timestamp in list response", - "author": "p.nair@example.com", - "minutes_before_alert": 12, - "files": { - "app/handlers/orders.py": [ - [ - " return {\"status\": 200, \"body\": body}\n\n\ndef create_order", - " body[\"newest_at\"] = rows[0][\"created_at\"].isoformat()\n return {\"status\": 200, \"body\": body}\n\n\ndef create_order" - ] - ] - } - } - ], - "label": { - "culprit_index": 2, - "expected_runbook_id": "pagination_errors" - } -} diff --git a/eval/cases/page_size_zero_02.json b/eval/cases/page_size_zero_division.json similarity index 58% rename from eval/cases/page_size_zero_02.json rename to eval/cases/page_size_zero_division.json index 60379d3..2718c88 100644 --- a/eval/cases/page_size_zero_02.json +++ b/eval/cases/page_size_zero_division.json @@ -1,5 +1,5 @@ { - "id": "page_size_zero_02", + "id": "page_size_zero_division", "category": "pagination_boundary", "base": "orders_service", "alert": { @@ -10,7 +10,7 @@ { "message": "pagination: let clients request exact page sizes", "author": "a.chen@example.com", - "minutes_before_alert": 48, + "minutes_before_alert": 57, "files": { "app/pagination.py": [ [ @@ -22,8 +22,8 @@ }, { "message": "pagination: return total page count", - "author": "r.haddad@example.com", - "minutes_before_alert": 33, + "author": "m.okafor@example.com", + "minutes_before_alert": 44, "files": { "app/pagination.py": [ [ @@ -34,22 +34,35 @@ } }, { - "message": "docs: describe exports module", - "author": "m.okafor@example.com", - "minutes_before_alert": 17, + "message": "config: smaller default page", + "author": "j.silva@example.com", + "minutes_before_alert": 31, + "files": { + "config.json": [ + [ + "\"page_size\": 50,", + "\"page_size\": 25," + ] + ] + } + }, + { + "message": "pagination: document page_size bounds", + "author": "p.nair@example.com", + "minutes_before_alert": 18, "files": { - "README.md": [ + "app/pagination.py": [ [ - "- `app/worker.py` fulfilment queue consumer\n", - "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" + "def clamp_page_size(requested: int | None) -> int:\n", + "def clamp_page_size(requested: int | None) -> int:\n \"\"\"Requested size clamped to [1, api.max_page_size]; api.page_size when omitted.\"\"\"\n" ] ] } }, { "message": "worker: shorter idle poll", - "author": "j.silva@example.com", - "minutes_before_alert": 4, + "author": "l.becker@example.com", + "minutes_before_alert": 5, "files": { "app/worker.py": [ [ @@ -62,6 +75,7 @@ ], "label": { "culprit_index": 0, - "expected_runbook_id": "pagination_errors" + "expected_runbook_id": "pagination_errors", + "note": "Dropping the max(1, ...) clamp lets page_size=0 through; the later total_pages line is where it divides, but that line is safe for any page_size >= 1." } } diff --git a/eval/cases/payments_body_without_content_type.json b/eval/cases/payments_body_without_content_type.json new file mode 100644 index 0000000..7b1d472 --- /dev/null +++ b/eval/cases/payments_body_without_content_type.json @@ -0,0 +1,93 @@ +{ + "id": "payments_body_without_content_type", + "category": "http_content_type", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "httpx.HTTPStatusError: Client error '415 Unsupported Media Type' for url 'https://payments.internal/v1/charges'" + }, + "commits": [ + { + "message": "payments: send idempotency key with charges", + "author": "r.haddad@example.com", + "minutes_before_alert": 57, + "files": { + "app/clients/payments.py": [ + [ + " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n", + " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n headers={\"Idempotency-Key\": f\"charge-{order_id}\"},\n" + ] + ] + } + }, + { + "message": "config: give payments more headroom", + "author": "a.chen@example.com", + "minutes_before_alert": 44, + "files": { + "config.json": [ + [ + "\"timeout_s\": 8,", + "\"timeout_s\": 10," + ] + ] + } + }, + { + "message": "payments: serialise charge body with sorted keys for stable idempotency hashing", + "author": "m.okafor@example.com", + "minutes_before_alert": 31, + "files": { + "app/clients/payments.py": [ + [ + "import httpx\n", + "import json\n\nimport httpx\n" + ], + [ + " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n", + " content=json.dumps(\n {\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token}, sort_keys=True\n ),\n" + ] + ] + } + }, + { + "message": "orders: log order totals", + "author": "j.silva@example.com", + "minutes_before_alert": 18, + "files": { + "app/handlers/orders.py": [ + [ + "from app import db, serializers\n", + "import logging\n\nfrom app import db, serializers\n" + ], + [ + "_ORDER_SQL =", + "log = logging.getLogger(__name__)\n\n_ORDER_SQL =" + ], + [ + " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n", + " total = sum(line[\"qty\"] * line[\"unit_price_cents\"] for line in lines)\n log.info(\"order total computed\", extra={\"customer_id\": customer_id, \"total_cents\": total})\n" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "p.nair@example.com", + "minutes_before_alert": 5, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": null, + "note": "httpx sets Content-Type: application/json only for json=; with content= the body is sent untyped. The idempotency header commit leaves the json= body intact." + } +} diff --git a/eval/cases/payments_form_body_07.json b/eval/cases/payments_form_body_07.json deleted file mode 100644 index 26b3b7d..0000000 --- a/eval/cases/payments_form_body_07.json +++ /dev/null @@ -1,80 +0,0 @@ -{ - "id": "payments_form_body_07", - "category": "http_content_type", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "httpx.HTTPStatusError: Client error '415 Unsupported Media Type' for url 'https://payments.internal/v1/charges'" - }, - "commits": [ - { - "message": "config: give payments more headroom", - "author": "j.silva@example.com", - "minutes_before_alert": 57, - "files": { - "config.json": [ - [ - "\"timeout_s\": 8, \"retries\": 2", - "\"timeout_s\": 10, \"retries\": 2" - ] - ] - } - }, - { - "message": "docs: describe exports module", - "author": "l.becker@example.com", - "minutes_before_alert": 45, - "files": { - "README.md": [ - [ - "- `app/worker.py` fulfilment queue consumer\n", - "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" - ] - ] - } - }, - { - "message": "orders: name the stock shortfall list", - "author": "r.haddad@example.com", - "minutes_before_alert": 32, - "files": { - "app/handlers/orders.py": [ - [ - " short = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if short:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": short}}\n", - " shortfall = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if shortfall:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": shortfall}}\n" - ] - ] - } - }, - { - "message": "payments: send charges as form fields for gateway v1 compatibility", - "author": "p.nair@example.com", - "minutes_before_alert": 18, - "files": { - "app/clients/payments.py": [ - [ - " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n", - " data={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n" - ] - ] - } - }, - { - "message": "cache: type hints", - "author": "a.chen@example.com", - "minutes_before_alert": 5, - "files": { - "app/cache.py": [ - [ - " def get(self, key):\n", - " def get(self, key: str):\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 3, - "expected_runbook_id": null - } -} diff --git a/eval/cases/rl_payment_retries_02.json b/eval/cases/payments_retry_loop.json similarity index 71% rename from eval/cases/rl_payment_retries_02.json rename to eval/cases/payments_retry_loop.json index 86ac38e..eb66ae6 100644 --- a/eval/cases/rl_payment_retries_02.json +++ b/eval/cases/payments_retry_loop.json @@ -1,35 +1,29 @@ { - "id": "rl_payment_retries_02", + "id": "payments_retry_loop", "category": "rate_limit_429", "base": "orders_service", "alert": { "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "httpx.HTTPStatusError: Client error '429 Too Many Requests' for url 'https://payments.internal/v1/charges'" + "error_signature": "httpx.HTTPStatusError: Client error '429 Too Many Requests' for url 'https://payments.internal/v1/charges' (outbound charge requests 4.8x order volume)" }, "commits": [ { - "message": "payments: retry charges on transient errors", - "author": "j.silva@example.com", - "minutes_before_alert": 46, + "message": "config: allow more payment retries during processor maintenance", + "author": "m.okafor@example.com", + "minutes_before_alert": 57, "files": { "config.json": [ [ - "\"timeout_s\": 8, \"retries\": 2", - "\"timeout_s\": 8, \"retries\": 5" - ] - ], - "app/clients/payments.py": [ - [ - "def charge(order_id: str, amount_cents: int, card_token: str) -> dict:\n resp = httpx.post(\n f\"{settings.get('payments', 'base_url')}/v1/charges\",\n json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n timeout=settings.get(\"payments\", \"timeout_s\"),\n )\n resp.raise_for_status()\n return resp.json()\n", - "def charge(order_id: str, amount_cents: int, card_token: str) -> dict:\n for _ in range(settings.get(\"payments\", \"retries\") + 1):\n resp = httpx.post(\n f\"{settings.get('payments', 'base_url')}/v1/charges\",\n json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n timeout=settings.get(\"payments\", \"timeout_s\"),\n )\n if resp.status_code < 400:\n return resp.json()\n resp.raise_for_status()\n return resp.json()\n" + "\"retries\": 2}", + "\"retries\": 4}" ] ] } }, { "message": "orders: log order totals", - "author": "a.chen@example.com", - "minutes_before_alert": 31, + "author": "j.silva@example.com", + "minutes_before_alert": 44, "files": { "app/handlers/orders.py": [ [ @@ -49,8 +43,8 @@ }, { "message": "docs: describe exports module", - "author": "r.haddad@example.com", - "minutes_before_alert": 15, + "author": "p.nair@example.com", + "minutes_before_alert": 31, "files": { "README.md": [ [ @@ -59,10 +53,37 @@ ] ] } + }, + { + "message": "payments: retry charges on transient errors", + "author": "l.becker@example.com", + "minutes_before_alert": 18, + "files": { + "app/clients/payments.py": [ + [ + "def charge(order_id: str, amount_cents: int, card_token: str) -> dict:\n resp = httpx.post(\n f\"{settings.get('payments', 'base_url')}/v1/charges\",\n json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n timeout=settings.get(\"payments\", \"timeout_s\"),\n )\n resp.raise_for_status()\n return resp.json()\n", + "def charge(order_id: str, amount_cents: int, card_token: str) -> dict:\n for _ in range(settings.get(\"payments\", \"retries\") + 1):\n resp = httpx.post(\n f\"{settings.get('payments', 'base_url')}/v1/charges\",\n json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n timeout=settings.get(\"payments\", \"timeout_s\"),\n )\n if resp.status_code < 400:\n return resp.json()\n resp.raise_for_status()\n return resp.json()\n" + ] + ] + } + }, + { + "message": "cache: type hints", + "author": "r.haddad@example.com", + "minutes_before_alert": 5, + "files": { + "app/cache.py": [ + [ + " def get(self, key):\n", + " def get(self, key: str):\n" + ] + ] + } } ], "label": { - "culprit_index": 0, - "expected_runbook_id": "rate_limiting" + "culprit_index": 3, + "expected_runbook_id": "rate_limiting", + "note": "payments.retries was not read by any code until the retry loop; the loop retries 429s immediately with no backoff, giving 1+4=5 attempts per order. The config bump was inert on its own." } } diff --git a/eval/cases/queue_args_swapped_03.json b/eval/cases/queue_args_swapped_03.json deleted file mode 100644 index 5a048df..0000000 --- a/eval/cases/queue_args_swapped_03.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "id": "queue_args_swapped_03", - "category": "queue_stall", - "base": "orders_service", - "alert": { - "alert_name": "Queue_Depth_Critical", - "error_signature": "fulfilment queue depth 18234 (threshold 5000), consumer throughput 0 jobs/min, worker pods healthy" - }, - "commits": [ - { - "message": "worker: import inventory at module level", - "author": "p.nair@example.com", - "minutes_before_alert": 49, - "files": { - "app/worker.py": [ - [ - "def handle(job: dict):\n from app.clients import inventory\n\n inventory.get_stock(job[\"sku\"])\n", - "def handle(job: dict):\n inventory.get_stock(job[\"sku\"])\n" - ], - [ - "import redis\n", - "import redis\n\nfrom app.clients import inventory\n" - ] - ] - } - }, - { - "message": "deps: bump PyJWT to 2.9.0", - "author": "a.chen@example.com", - "minutes_before_alert": 31, - "files": { - "requirements.txt": [ - [ - "PyJWT==2.8.0", - "PyJWT==2.9.0" - ] - ] - } - }, - { - "message": "worker: pass queue names by keyword", - "author": "l.becker@example.com", - "minutes_before_alert": 15, - "files": { - "app/worker.py": [ - [ - " raw = _r.brpoplpush(QUEUE, PROCESSING, timeout=5)\n", - " raw = _r.brpoplpush(src=PROCESSING, dst=QUEUE, timeout=5)\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 2, - "expected_runbook_id": null - } -} diff --git a/eval/cases/redis_legacy_lrem_args.json b/eval/cases/redis_legacy_lrem_args.json new file mode 100644 index 0000000..ba5349d --- /dev/null +++ b/eval/cases/redis_legacy_lrem_args.json @@ -0,0 +1,81 @@ +{ + "id": "redis_legacy_lrem_args", + "category": "dependency_version", + "base": "orders_service", + "alert": { + "alert_name": "FulfilmentWorker_Crashloop", + "error_signature": "redis.exceptions.ResponseError: value is not an integer or out of range" + }, + "commits": [ + { + "message": "deps: pin redis client to 2.10.6 to match fulfilment host image", + "author": "l.becker@example.com", + "minutes_before_alert": 57, + "files": { + "requirements.txt": [ + [ + "redis==5.0.8", + "redis==2.10.6" + ] + ] + } + }, + { + "message": "worker: remove acked job from the tail of the processing list", + "author": "r.haddad@example.com", + "minutes_before_alert": 44, + "files": { + "app/worker.py": [ + [ + "_r.lrem(PROCESSING, 1, raw)", + "_r.lrem(PROCESSING, -1, raw)" + ] + ] + } + }, + { + "message": "worker: shorter idle poll", + "author": "a.chen@example.com", + "minutes_before_alert": 31, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + }, + { + "message": "deps: bump PyJWT to 2.9.0", + "author": "m.okafor@example.com", + "minutes_before_alert": 18, + "files": { + "requirements.txt": [ + [ + "PyJWT==2.8.0", + "PyJWT==2.9.0" + ] + ] + } + }, + { + "message": "cache: type hints", + "author": "j.silva@example.com", + "minutes_before_alert": 5, + "files": { + "app/cache.py": [ + [ + " def get(self, key):\n", + " def get(self, key: str):\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": "dependency_regressions", + "note": "redis-py 2.x uses lrem(name, value, num); 3.0+ uses lrem(name, count, value). Under 2.10.6 the job payload is sent as the count. A count of -1 is valid in the current argument order." + } +} diff --git a/eval/cases/rl_inventory_uncached_01.json b/eval/cases/rl_inventory_uncached_01.json deleted file mode 100644 index 1cd94cc..0000000 --- a/eval/cases/rl_inventory_uncached_01.json +++ /dev/null @@ -1,67 +0,0 @@ -{ - "id": "rl_inventory_uncached_01", - "category": "rate_limit_429", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "httpx.HTTPStatusError: Client error '429 Too Many Requests' for url 'https://inventory.internal/v2/stock/SKU-88412'" - }, - "commits": [ - { - "message": "config: shorten stock cache ttl", - "author": "a.chen@example.com", - "minutes_before_alert": 57, - "files": { - "config.json": [ - [ - "\"ttl_s\": 300", - "\"ttl_s\": 240" - ] - ] - } - }, - { - "message": "pagination: document page_size bounds", - "author": "l.becker@example.com", - "minutes_before_alert": 43, - "files": { - "app/pagination.py": [ - [ - "def clamp_page_size(requested: int | None) -> int:\n", - "def clamp_page_size(requested: int | None) -> int:\n \"\"\"Requested size clamped to [1, api.max_page_size]; api.page_size when omitted.\"\"\"\n" - ] - ] - } - }, - { - "message": "inventory: read fresh stock for multi-item orders", - "author": "p.nair@example.com", - "minutes_before_alert": 28, - "files": { - "app/clients/inventory.py": [ - [ - "def get_stock_many(skus: list[str]) -> dict[str, int]:\n return {sku: get_stock(sku) for sku in skus}\n", - "def get_stock_many(skus: list[str]) -> dict[str, int]:\n out = {}\n for sku in skus:\n resp = httpx.get(\n f\"{settings.get('inventory', 'base_url')}/v2/stock/{sku}\",\n timeout=settings.get(\"inventory\", \"timeout_s\"),\n )\n resp.raise_for_status()\n out[sku] = resp.json()[\"available\"]\n return out\n" - ] - ] - } - }, - { - "message": "docs: local development notes", - "author": "m.okafor@example.com", - "minutes_before_alert": 13, - "files": { - "README.md": [ - [ - "(JSON-encoded)\n", - "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 2, - "expected_runbook_id": "rate_limiting" - } -} diff --git a/eval/cases/settings_env_quotes.json b/eval/cases/settings_env_quotes.json new file mode 100644 index 0000000..20050b1 --- /dev/null +++ b/eval/cases/settings_env_quotes.json @@ -0,0 +1,81 @@ +{ + "id": "settings_env_quotes", + "category": "config_type_error", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_503_Service_Unavailable", + "error_signature": "psycopg2.OperationalError: could not translate host name \"\"10.0.4.12\"\" to address: Name or service not known" + }, + "commits": [ + { + "message": "db: tag connections with application_name", + "author": "p.nair@example.com", + "minutes_before_alert": 57, + "files": { + "app/db.py": [ + [ + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n", + " connect_timeout=settings.get(\"db\", \"connect_timeout_s\"),\n application_name=\"orders-service\",\n" + ] + ] + } + }, + { + "message": "settings: add section() helper", + "author": "l.becker@example.com", + "minutes_before_alert": 44, + "files": { + "app/settings.py": [ + [ + " return _CONFIG[section][key]\n", + " return _CONFIG[section][key]\n\n\ndef section(name: str) -> dict:\n return dict(_CONFIG[name])\n" + ] + ] + } + }, + { + "message": "settings: accept plain env overrides without JSON quoting", + "author": "r.haddad@example.com", + "minutes_before_alert": 31, + "files": { + "app/settings.py": [ + [ + " return json.loads(os.environ[env_key])\n", + " raw = os.environ[env_key]\n return int(raw) if raw.isdigit() else raw\n" + ] + ] + } + }, + { + "message": "config: db pool to 12", + "author": "a.chen@example.com", + "minutes_before_alert": 18, + "files": { + "config.json": [ + [ + "\"pool_size\": 10", + "\"pool_size\": 12" + ] + ] + } + }, + { + "message": "worker: shorter idle poll", + "author": "m.okafor@example.com", + "minutes_before_alert": 5, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": "config_parse_errors", + "note": "Deployments set overrides JSON-encoded (README), e.g. ORDERS_DB_HOST='\"10.0.4.12\"'; without json.loads the quotes stay in the host name. Not a database or network problem." + } +} diff --git a/eval/cases/since_filter_naive_cutoff.json b/eval/cases/since_filter_naive_cutoff.json new file mode 100644 index 0000000..94db78c --- /dev/null +++ b/eval/cases/since_filter_naive_cutoff.json @@ -0,0 +1,89 @@ +{ + "id": "since_filter_naive_cutoff", + "category": "timezone_mismatch", + "base": "orders_service", + "alert": { + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "TypeError: can't compare offset-naive and offset-aware datetimes" + }, + "commits": [ + { + "message": "serializers: accept Z suffix in since", + "author": "r.haddad@example.com", + "minutes_before_alert": 57, + "files": { + "app/serializers.py": [ + [ + " return datetime.fromisoformat(value) if value else None\n", + " return datetime.fromisoformat(value.replace(\"Z\", \"+00:00\")) if value else None\n" + ] + ] + } + }, + { + "message": "exports: local timestamps in file names", + "author": "a.chen@example.com", + "minutes_before_alert": 44, + "files": { + "app/exports.py": [ + [ + " stamp = datetime.now(timezone.utc).strftime(", + " stamp = datetime.now().strftime(" + ] + ] + } + }, + { + "message": "orders: filter order history by since", + "author": "m.okafor@example.com", + "minutes_before_alert": 31, + "files": { + "app/handlers/orders.py": [ + [ + "def list_orders(customer_id: str, page: int = 1, page_size: int | None = None) -> dict:\n rows = db.fetch_all(_LIST_SQL, (customer_id,))\n", + "def list_orders(\n customer_id: str, page: int = 1, page_size: int | None = None, since: str | None = None\n) -> dict:\n rows = db.fetch_all(_LIST_SQL, (customer_id,))\n if since is not None:\n cutoff = serializers.parse_since(since).replace(tzinfo=None)\n rows = [r for r in rows if r[\"created_at\"] >= cutoff]\n" + ] + ] + } + }, + { + "message": "cache: track hit and miss counts", + "author": "j.silva@example.com", + "minutes_before_alert": 18, + "files": { + "app/cache.py": [ + [ + " self._data: dict = {}\n", + " self._data: dict = {}\n self.hits = 0\n self.misses = 0\n" + ], + [ + " entry = self._data.get(key)\n if entry is None:\n return None\n", + " entry = self._data.get(key)\n if entry is None:\n self.misses += 1\n return None\n" + ], + [ + " del self._data[key]\n return None\n return value\n", + " del self._data[key]\n self.misses += 1\n return None\n self.hits += 1\n return value\n" + ] + ] + } + }, + { + "message": "deps: bump PyJWT to 2.9.0", + "author": "p.nair@example.com", + "minutes_before_alert": 5, + "files": { + "requirements.txt": [ + [ + "PyJWT==2.8.0", + "PyJWT==2.9.0" + ] + ] + } + } + ], + "label": { + "culprit_index": 2, + "expected_runbook_id": null, + "note": "The new filter strips tzinfo from the cutoff and compares it with timestamptz rows. The export change only formats a file name and never compares datetimes." + } +} diff --git a/eval/cases/sku_lowercase_06.json b/eval/cases/sku_lowercase_06.json deleted file mode 100644 index b525a1b..0000000 --- a/eval/cases/sku_lowercase_06.json +++ /dev/null @@ -1,62 +0,0 @@ -{ - "id": "sku_lowercase_06", - "category": "identifier_format", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "httpx.HTTPStatusError: Client error '404 Not Found' for url 'https://inventory.internal/v2/stock/sku-10293'" - }, - "commits": [ - { - "message": "inventory: normalise SKU keys for cache hits", - "author": "m.okafor@example.com", - "minutes_before_alert": 44, - "files": { - "app/clients/inventory.py": [ - [ - "def get_stock(sku: str) -> int:\n", - "def get_stock(sku: str) -> int:\n sku = sku.strip().lower()\n" - ] - ] - } - }, - { - "message": "cache: track hit and miss counts", - "author": "a.chen@example.com", - "minutes_before_alert": 30, - "files": { - "app/cache.py": [ - [ - " self._data: dict = {}\n", - " self._data: dict = {}\n self.hits = 0\n self.misses = 0\n" - ], - [ - " entry = self._data.get(key)\n if entry is None:\n return None\n", - " entry = self._data.get(key)\n if entry is None:\n self.misses += 1\n return None\n" - ], - [ - " del self._data[key]\n return None\n return value\n", - " del self._data[key]\n self.misses += 1\n return None\n self.hits += 1\n return value\n" - ] - ] - } - }, - { - "message": "config: lower inventory timeout to 2s", - "author": "r.haddad@example.com", - "minutes_before_alert": 14, - "files": { - "config.json": [ - [ - "\"timeout_s\": 3}", - "\"timeout_s\": 2}" - ] - ] - } - } - ], - "label": { - "culprit_index": 0, - "expected_runbook_id": null - } -} diff --git a/eval/cases/tmo_gateway_deadline_02.json b/eval/cases/tmo_gateway_deadline_02.json deleted file mode 100644 index 4712e65..0000000 --- a/eval/cases/tmo_gateway_deadline_02.json +++ /dev/null @@ -1,67 +0,0 @@ -{ - "id": "tmo_gateway_deadline_02", - "category": "latency_timeout", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_504_Gateway_Timeout", - "error_signature": "504 Gateway Timeout: POST /v1/orders exceeded upstream deadline (payments.charge p50 6.2s, unchanged)" - }, - "commits": [ - { - "message": "config: tighten edge deadline to meet p99 target", - "author": "p.nair@example.com", - "minutes_before_alert": 54, - "files": { - "config.json": [ - [ - "\"gateway\": {\"upstream_timeout_s\": 15}", - "\"gateway\": {\"upstream_timeout_s\": 5}" - ] - ] - } - }, - { - "message": "payments: send idempotency key with charges", - "author": "m.okafor@example.com", - "minutes_before_alert": 40, - "files": { - "app/clients/payments.py": [ - [ - " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n", - " json={\"order_id\": order_id, \"amount_cents\": amount_cents, \"card_token\": card_token},\n headers={\"Idempotency-Key\": f\"charge-{order_id}\"},\n" - ] - ] - } - }, - { - "message": "orders: name the stock shortfall list", - "author": "a.chen@example.com", - "minutes_before_alert": 25, - "files": { - "app/handlers/orders.py": [ - [ - " short = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if short:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": short}}\n", - " shortfall = [line[\"sku\"] for line in lines if stock[line[\"sku\"]] < line[\"qty\"]]\n if shortfall:\n return {\"status\": 409, \"body\": {\"error\": \"insufficient stock\", \"skus\": shortfall}}\n" - ] - ] - } - }, - { - "message": "docs: local development notes", - "author": "l.becker@example.com", - "minutes_before_alert": 11, - "files": { - "README.md": [ - [ - "(JSON-encoded)\n", - "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 0, - "expected_runbook_id": "request_timeouts" - } -} diff --git a/eval/cases/tmo_inventory_units_01.json b/eval/cases/tmo_inventory_units_01.json deleted file mode 100644 index 337bd94..0000000 --- a/eval/cases/tmo_inventory_units_01.json +++ /dev/null @@ -1,66 +0,0 @@ -{ - "id": "tmo_inventory_units_01", - "category": "latency_timeout", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "httpx.ConnectTimeout: timed out" - }, - "commits": [ - { - "message": "config: give payments more headroom", - "author": "l.becker@example.com", - "minutes_before_alert": 50, - "files": { - "config.json": [ - [ - "\"timeout_s\": 8, \"retries\": 2", - "\"timeout_s\": 10, \"retries\": 2" - ] - ] - } - }, - { - "message": "inventory: hoist timeout lookup to module level", - "author": "j.silva@example.com", - "minutes_before_alert": 33, - "files": { - "app/clients/inventory.py": [ - [ - "_stock_cache = TTLCache(", - "_TIMEOUT = settings.get(\"inventory\", \"timeout_s\") / 1000 # httpx timeouts are in ms\n_stock_cache = TTLCache(" - ], - [ - " timeout=settings.get(\"inventory\", \"timeout_s\"),\n", - " timeout=_TIMEOUT,\n" - ] - ] - } - }, - { - "message": "worker: shorter idle poll", - "author": "r.haddad@example.com", - "minutes_before_alert": 20, - "files": { - "app/worker.py": [ - [ - "timeout=5)", - "timeout=2)" - ] - ] - } - }, - { - "message": "tests: cover page_bounds", - "author": "a.chen@example.com", - "minutes_before_alert": 6, - "files": { - "tests/test_pagination.py": "from app.pagination import page_bounds\n\n\ndef test_first_page_starts_at_zero():\n assert page_bounds(1, 50) == (0, 50)\n\n\ndef test_second_page():\n assert page_bounds(2, 50) == (50, 100)\n" - } - } - ], - "label": { - "culprit_index": 1, - "expected_runbook_id": "request_timeouts" - } -} diff --git a/eval/cases/tmo_list_live_stock_03.json b/eval/cases/tmo_list_live_stock_03.json deleted file mode 100644 index 2760e27..0000000 --- a/eval/cases/tmo_list_live_stock_03.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "id": "tmo_list_live_stock_03", - "category": "latency_timeout", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_504_Gateway_Timeout", - "error_signature": "504 Gateway Timeout: GET /v1/customers/{customer_id}/orders exceeded upstream deadline" - }, - "commits": [ - { - "message": "config: shorten stock cache ttl", - "author": "r.haddad@example.com", - "minutes_before_alert": 47, - "files": { - "config.json": [ - [ - "\"ttl_s\": 300", - "\"ttl_s\": 120" - ] - ] - } - }, - { - "message": "pagination: document page_size bounds", - "author": "j.silva@example.com", - "minutes_before_alert": 29, - "files": { - "app/pagination.py": [ - [ - "def clamp_page_size(requested: int | None) -> int:\n", - "def clamp_page_size(requested: int | None) -> int:\n \"\"\"Requested size clamped to [1, api.max_page_size]; api.page_size when omitted.\"\"\"\n" - ] - ] - } - }, - { - "message": "orders: show live availability in order history", - "author": "m.okafor@example.com", - "minutes_before_alert": 8, - "files": { - "app/handlers/orders.py": [ - [ - " rows = db.fetch_all(_LIST_SQL, (customer_id,))\n body = paginate([serializers.order_to_dict(r) for r in rows], page, clamp_page_size(page_size))\n", - " rows = db.fetch_all(_LIST_SQL, (customer_id,))\n orders = [serializers.order_to_dict(r) for r in rows]\n for order in orders:\n stock = inventory.get_stock_many([item[\"sku\"] for item in order[\"items\"]])\n for item in order[\"items\"]:\n item[\"available\"] = stock[item[\"sku\"]]\n body = paginate(orders, page, clamp_page_size(page_size))\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 2, - "expected_runbook_id": "request_timeouts" - } -} diff --git a/eval/cases/tz_naive_since_01.json b/eval/cases/tz_naive_since_01.json deleted file mode 100644 index 7ce5902..0000000 --- a/eval/cases/tz_naive_since_01.json +++ /dev/null @@ -1,54 +0,0 @@ -{ - "id": "tz_naive_since_01", - "category": "timezone_mismatch", - "base": "orders_service", - "alert": { - "alert_name": "HTTP_500_Internal_Server_Error", - "error_signature": "TypeError: can't compare offset-naive and offset-aware datetimes" - }, - "commits": [ - { - "message": "orders: filter order history by since", - "author": "j.silva@example.com", - "minutes_before_alert": 45, - "files": { - "app/handlers/orders.py": [ - [ - "def list_orders(customer_id: str, page: int = 1, page_size: int | None = None) -> dict:\n rows = db.fetch_all(_LIST_SQL, (customer_id,))\n", - "def list_orders(\n customer_id: str, page: int = 1, page_size: int | None = None, since: str | None = None\n) -> dict:\n rows = db.fetch_all(_LIST_SQL, (customer_id,))\n if since is not None:\n rows = [r for r in rows if r[\"created_at\"] >= serializers.parse_since(since)]\n" - ] - ] - } - }, - { - "message": "serializers: drop tz info from since for db comparisons", - "author": "r.haddad@example.com", - "minutes_before_alert": 30, - "files": { - "app/serializers.py": [ - [ - " return datetime.fromisoformat(value) if value else None\n", - " return datetime.fromisoformat(value).replace(tzinfo=None) if value else None\n" - ] - ] - } - }, - { - "message": "docs: describe exports module", - "author": "p.nair@example.com", - "minutes_before_alert": 12, - "files": { - "README.md": [ - [ - "- `app/worker.py` fulfilment queue consumer\n", - "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" - ] - ] - } - } - ], - "label": { - "culprit_index": 1, - "expected_runbook_id": null - } -} diff --git a/eval/cases/worker_queue_renamed.json b/eval/cases/worker_queue_renamed.json new file mode 100644 index 0000000..5357025 --- /dev/null +++ b/eval/cases/worker_queue_renamed.json @@ -0,0 +1,81 @@ +{ + "id": "worker_queue_renamed", + "category": "queue_stall", + "base": "orders_service", + "alert": { + "alert_name": "Queue_Depth_Critical", + "error_signature": "fulfilment queue depth 18234 (threshold 5000), consumer throughput 0 jobs/min, worker pods healthy, no errors logged" + }, + "commits": [ + { + "message": "worker: namespace queue keys under orders:", + "author": "m.okafor@example.com", + "minutes_before_alert": 57, + "files": { + "app/worker.py": [ + [ + "QUEUE = \"fulfilment\"\nPROCESSING = \"fulfilment:processing\"\n", + "QUEUE = \"orders:fulfilment\"\nPROCESSING = \"orders:fulfilment:processing\"\n" + ] + ] + } + }, + { + "message": "worker: shorter idle poll", + "author": "j.silva@example.com", + "minutes_before_alert": 44, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } + }, + { + "message": "worker: remove acked job from the tail of the processing list", + "author": "p.nair@example.com", + "minutes_before_alert": 31, + "files": { + "app/worker.py": [ + [ + "_r.lrem(PROCESSING, 1, raw)", + "_r.lrem(PROCESSING, -1, raw)" + ] + ] + } + }, + { + "message": "deps: bump PyJWT to 2.9.0", + "author": "l.becker@example.com", + "minutes_before_alert": 18, + "files": { + "requirements.txt": [ + [ + "PyJWT==2.8.0", + "PyJWT==2.9.0" + ] + ] + } + }, + { + "message": "docs: local development notes", + "author": "r.haddad@example.com", + "minutes_before_alert": 5, + "files": { + "README.md": [ + [ + "(JSON-encoded)\n", + "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + ] + ] + } + } + ], + "label": { + "culprit_index": 0, + "expected_runbook_id": null, + "note": "Producers still push to 'fulfilment'; the worker now blocks on 'orders:fulfilment', which stays empty, so nothing errors and nothing is consumed. The poll and lrem changes still consume the right list." + } +} diff --git a/eval/cases/mem_worker_seen_ids_01.json b/eval/cases/worker_retains_job_payloads.json similarity index 57% rename from eval/cases/mem_worker_seen_ids_01.json rename to eval/cases/worker_retains_job_payloads.json index bfc5f3c..1461c9f 100644 --- a/eval/cases/mem_worker_seen_ids_01.json +++ b/eval/cases/worker_retains_job_payloads.json @@ -1,51 +1,51 @@ { - "id": "mem_worker_seen_ids_01", + "id": "worker_retains_job_payloads", "category": "memory_growth", "base": "orders_service", "alert": { "alert_name": "FulfilmentWorker_Memory_High", - "error_signature": "fulfilment-worker OOMKilled (exit code 137): rss 3.8GiB of 4GiB limit, rss growing linearly with jobs processed" + "error_signature": "fulfilment-worker OOMKilled (exit code 137): rss 0.4GiB -> 3.9GiB over 50 min while processing ~1.2M jobs" }, "commits": [ { - "message": "config: raise stock cache size", - "author": "m.okafor@example.com", - "minutes_before_alert": 54, + "message": "docs: describe exports module", + "author": "r.haddad@example.com", + "minutes_before_alert": 57, "files": { - "config.json": [ + "README.md": [ [ - "\"max_entries\": 5000", - "\"max_entries\": 8000" + "- `app/worker.py` fulfilment queue consumer\n", + "- `app/worker.py` fulfilment queue consumer\n- `app/exports.py` nightly CSV export to `exports.dir`\n" ] ] } }, { - "message": "worker: skip duplicate fulfilment jobs", + "message": "worker: skip fulfilment jobs already handled", "author": "a.chen@example.com", - "minutes_before_alert": 39, + "minutes_before_alert": 44, "files": { "app/worker.py": [ [ "QUEUE = \"fulfilment\"\n", - "_seen_jobs: list[str] = []\nQUEUE = \"fulfilment\"\n" + "_handled: dict[str, dict] = {}\nQUEUE = \"fulfilment\"\n" ], [ " job = json.loads(raw)\n handle(job)\n", - " job = json.loads(raw)\n if job[\"id\"] not in _seen_jobs:\n handle(job)\n _seen_jobs.append(job[\"id\"])\n" + " job = json.loads(raw)\n if job[\"id\"] not in _handled:\n handle(job)\n _handled[job[\"id\"]] = job\n" ] ] } }, { - "message": "docs: local development notes", - "author": "l.becker@example.com", - "minutes_before_alert": 24, + "message": "config: raise stock cache size for catalog growth", + "author": "m.okafor@example.com", + "minutes_before_alert": 31, "files": { - "README.md": [ + "config.json": [ [ - "(JSON-encoded)\n", - "(JSON-encoded)\n\n## Local development\n\n```\npip install -r requirements.txt\nORDERS_DB_HOST='\"localhost\"' python -m app.worker\n```\n" + "\"max_entries\": 5000", + "\"max_entries\": 50000" ] ] } @@ -53,7 +53,7 @@ { "message": "cache: track hit and miss counts", "author": "j.silva@example.com", - "minutes_before_alert": 11, + "minutes_before_alert": 18, "files": { "app/cache.py": [ [ @@ -70,10 +70,24 @@ ] ] } + }, + { + "message": "worker: shorter idle poll", + "author": "p.nair@example.com", + "minutes_before_alert": 5, + "files": { + "app/worker.py": [ + [ + "timeout=5)", + "timeout=2)" + ] + ] + } } ], "label": { "culprit_index": 1, - "expected_runbook_id": "memory_leaks" + "expected_runbook_id": "memory_leaks", + "note": "Every job payload is retained forever (~3KB x 1.2M jobs = ~3.5GiB). A 50k-entry cache of SKU -> int is a few MB at most." } } diff --git a/eval/config.py b/eval/config.py index a614063..049f410 100644 --- a/eval/config.py +++ b/eval/config.py @@ -1,6 +1,7 @@ # Frozen eval inputs. Bump a version whenever its inputs change; never edit after a compared run. RUNBOOK_CORPUS_VERSION = "v1" # sandbox/runbooks/*.md, 10 docs, frozen at Phase 9B -EVAL_SET_VERSION = "v1" # eval/cases/*.json + eval/bases/, 29 cases, frozen at Phase 9C +# v2 replaced v1 after the v1 baseline hit 100% top-1 (ceiling); v1 lives in git history at d6f6bd1. +EVAL_SET_VERSION = "v2" # eval/cases/*.json + eval/bases/, 24 cases, frozen at Phase 9C (rev) ALERT_TS = "2026-09-01T12:00:00+00:00" # fixed so fixture commit dates are reproducible DEFAULT_TRIALS = 3 From 150d678f009d16d099cdf3deefd5347dcec0fcc8 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 15:51:45 -0400 Subject: [PATCH 14/19] feat(eval): baseline run on eval set v2 (65/72 top-1) Co-Authored-By: Claude Opus 5.5 --- eval/results/baseline_v2_2026-10-04.json | 3447 ++++++++++++++++++++++ 1 file changed, 3447 insertions(+) create mode 100644 eval/results/baseline_v2_2026-10-04.json diff --git a/eval/results/baseline_v2_2026-10-04.json b/eval/results/baseline_v2_2026-10-04.json new file mode 100644 index 0000000..15818e6 --- /dev/null +++ b/eval/results/baseline_v2_2026-10-04.json @@ -0,0 +1,3447 @@ +{ + "condition": "baseline", + "dry_run": false, + "eval_set_version": "v2", + "runbook_corpus_version": "v1", + "runbooks_ingested": 10, + "model": "claude-sonnet-5", + "sentinel_git_sha": "2e1eb8d6c9243d9cd742b07f819a6a9c17c6a7f3", + "started_at": "2026-10-04T19:40:05.788931+00:00", + "trials": 3, + "n_cases": 24, + "records": [ + { + "trial": 0, + "case_id": "auth_issuer_mesh_hosts", + "category": "auth_rejection", + "culprit_hash": "5e68f8a", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.317 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "null_field_handling", + "similarity_score": 0.236 + } + ], + "ranking": [ + { + "commit_hash": "5e68f8a", + "confidence_score": 0.97 + }, + { + "commit_hash": "df0ec4c", + "confidence_score": 0.05 + }, + { + "commit_hash": "45a24d5", + "confidence_score": 0.03 + }, + { + "commit_hash": "298be8d", + "confidence_score": 0.02 + }, + { + "commit_hash": "d20025a", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.72 + }, + { + "trial": 0, + "case_id": "cache_contains_recursion", + "category": "infinite_recursion", + "culprit_hash": "4584056", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [ + { + "commit_hash": "4584056", + "confidence_score": 0.97 + }, + { + "commit_hash": "6484609", + "confidence_score": 0.1 + }, + { + "commit_hash": "6cc141a", + "confidence_score": 0.03 + }, + { + "commit_hash": "7e63ea3", + "confidence_score": 0.02 + }, + { + "commit_hash": "a98de00", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.7 + }, + { + "trial": 0, + "case_id": "cart_price_field_v2", + "category": "null_or_missing_field", + "culprit_hash": "d0ab213", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.26 + }, + { + "id": "pagination_errors", + "similarity_score": 0.207 + }, + { + "id": "memory_leaks", + "similarity_score": 0.179 + } + ], + "ranking": [ + { + "commit_hash": "d0ab213", + "confidence_score": 0.95 + }, + { + "commit_hash": "73e1933", + "confidence_score": 0.3 + }, + { + "commit_hash": "4fbde4c", + "confidence_score": 0.05 + }, + { + "commit_hash": "2bbbf3f", + "confidence_score": 0.02 + }, + { + "commit_hash": "f618f59", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.0 + }, + { + "trial": 0, + "case_id": "cart_sku_lowercased", + "category": "identifier_format", + "culprit_hash": "c2ec824", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [ + { + "commit_hash": "c2ec824", + "confidence_score": 0.75 + }, + { + "commit_hash": "64dde29", + "confidence_score": 0.55 + }, + { + "commit_hash": "d0cb0b3", + "confidence_score": 0.05 + }, + { + "commit_hash": "91f8d36", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.75 + }, + { + "trial": 0, + "case_id": "checkout_deadline_extra_call", + "category": "latency_timeout", + "culprit_hash": "efebc07", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.487 + }, + { + "id": "db_failures", + "similarity_score": 0.369 + }, + { + "id": "rate_limiting", + "similarity_score": 0.353 + } + ], + "ranking": [ + { + "commit_hash": "efebc07", + "confidence_score": 0.85 + }, + { + "commit_hash": "a035477", + "confidence_score": 0.55 + }, + { + "commit_hash": "2b765f4", + "confidence_score": 0.3 + }, + { + "commit_hash": "be85cc9", + "confidence_score": 0.1 + }, + { + "commit_hash": "91b8b07", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.33 + }, + { + "trial": 0, + "case_id": "config_reformat_hidden_string", + "category": "config_type_error", + "culprit_hash": "c3629ef", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [ + { + "commit_hash": "c3629ef", + "confidence_score": 0.85 + }, + { + "commit_hash": "3af442c", + "confidence_score": 0.2 + }, + { + "commit_hash": "a6a3c41", + "confidence_score": 0.1 + }, + { + "commit_hash": "976c988", + "confidence_score": 0.05 + }, + { + "commit_hash": "89436d3", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.75 + }, + { + "trial": 0, + "case_id": "db_fetch_one_leak", + "category": "db_pool_exhaustion", + "culprit_hash": "af9b861", + "expected_runbook_id": "db_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "af9b861", + "confidence_score": 0.95 + }, + { + "commit_hash": "f46cd8c", + "confidence_score": 0.45 + }, + { + "commit_hash": "1618815", + "confidence_score": 0.05 + }, + { + "commit_hash": "6d50d20", + "confidence_score": 0.02 + }, + { + "commit_hash": "1bf1024", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.78 + }, + { + "trial": 0, + "case_id": "export_tempfile_never_closed", + "category": "file_handle_leak", + "culprit_hash": "6eea0fc", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.625 + }, + { + "id": "memory_leaks", + "similarity_score": 0.432 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.291 + } + ], + "ranking": [ + { + "commit_hash": "5cedb21", + "confidence_score": 0.75 + }, + { + "commit_hash": "6eea0fc", + "confidence_score": 0.65 + }, + { + "commit_hash": "b245524", + "confidence_score": 0.3 + }, + { + "commit_hash": "0be651d", + "confidence_score": 0.05 + }, + { + "commit_hash": "7453733", + "confidence_score": 0.0 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 5.83 + }, + { + "trial": 0, + "case_id": "guest_email_normalise", + "category": "null_or_missing_field", + "culprit_hash": "7dd25fe", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.241 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.208 + }, + { + "id": "pagination_errors", + "similarity_score": 0.161 + } + ], + "ranking": [ + { + "commit_hash": "7dd25fe", + "confidence_score": 0.97 + }, + { + "commit_hash": "9519064", + "confidence_score": 0.05 + }, + { + "commit_hash": "1c0135b", + "confidence_score": 0.02 + }, + { + "commit_hash": "30a7434", + "confidence_score": 0.01 + }, + { + "commit_hash": "cde392a", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.22 + }, + { + "trial": 0, + "case_id": "httpx_028_drops_proxies", + "category": "dependency_version", + "culprit_hash": "b35322f", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.294 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.285 + }, + { + "id": "memory_leaks", + "similarity_score": 0.263 + } + ], + "ranking": [ + { + "commit_hash": "741f699", + "confidence_score": 0.85 + }, + { + "commit_hash": "b35322f", + "confidence_score": 0.8 + }, + { + "commit_hash": "b45189f", + "confidence_score": 0.02 + }, + { + "commit_hash": "e3601f6", + "confidence_score": 0.01 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 4.8 + }, + { + "trial": 0, + "case_id": "inventory_cache_key_per_request", + "category": "rate_limit_429", + "culprit_hash": "526f8dd", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.382 + }, + { + "id": "auth_failures", + "similarity_score": 0.381 + }, + { + "id": "null_field_handling", + "similarity_score": 0.309 + } + ], + "ranking": [ + { + "commit_hash": "526f8dd", + "confidence_score": 0.97 + }, + { + "commit_hash": "bf4cc5f", + "confidence_score": 0.35 + }, + { + "commit_hash": "93320f2", + "confidence_score": 0.05 + }, + { + "commit_hash": "2e1b06b", + "confidence_score": 0.01 + }, + { + "commit_hash": "d9caa65", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.22 + }, + { + "trial": 0, + "case_id": "inventory_v3_payload", + "category": "null_or_missing_field", + "culprit_hash": "1c29d94", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.33 + }, + { + "id": "memory_leaks", + "similarity_score": 0.281 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "1c29d94", + "confidence_score": 0.85 + }, + { + "commit_hash": "c6f0797", + "confidence_score": 0.05 + }, + { + "commit_hash": "16d2191", + "confidence_score": 0.05 + }, + { + "commit_hash": "5823d1f", + "confidence_score": 0.02 + }, + { + "commit_hash": "fb9b710", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.88 + }, + { + "trial": 0, + "case_id": "jwt_leeway_milliseconds", + "category": "auth_rejection", + "culprit_hash": "2eeb222", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.488 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.198 + }, + { + "id": "request_timeouts", + "similarity_score": 0.159 + } + ], + "ranking": [ + { + "commit_hash": "2eeb222", + "confidence_score": 0.95 + }, + { + "commit_hash": "e8548c5", + "confidence_score": 0.2 + }, + { + "commit_hash": "2e11931", + "confidence_score": 0.15 + }, + { + "commit_hash": "a028c62", + "confidence_score": 0.02 + }, + { + "commit_hash": "716e22d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.7 + }, + { + "trial": 0, + "case_id": "money_returns_decimal", + "category": "json_serialization", + "culprit_hash": "b14897b", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [ + { + "commit_hash": "b14897b", + "confidence_score": 0.97 + }, + { + "commit_hash": "dedc41a", + "confidence_score": 0.05 + }, + { + "commit_hash": "626c22d", + "confidence_score": 0.02 + }, + { + "commit_hash": "cd4a49f", + "confidence_score": 0.01 + }, + { + "commit_hash": "92cb3da", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.56 + }, + { + "trial": 0, + "case_id": "order_read_sync_audit", + "category": "latency_timeout", + "culprit_hash": "8aed86c", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.539 + }, + { + "id": "rate_limiting", + "similarity_score": 0.369 + }, + { + "id": "auth_failures", + "similarity_score": 0.35 + } + ], + "ranking": [ + { + "commit_hash": "8aed86c", + "confidence_score": 0.9 + }, + { + "commit_hash": "d3d6f87", + "confidence_score": 0.4 + }, + { + "commit_hash": "94578cc", + "confidence_score": 0.25 + }, + { + "commit_hash": "5abe2e4", + "confidence_score": 0.05 + }, + { + "commit_hash": "b7ea08d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.28 + }, + { + "trial": 0, + "case_id": "order_sql_missing_param", + "category": "sql_param_mismatch", + "culprit_hash": "9ef79b2", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [ + { + "commit_hash": "9ef79b2", + "confidence_score": 0.85 + }, + { + "commit_hash": "661f871", + "confidence_score": 0.2 + }, + { + "commit_hash": "93ebe2f", + "confidence_score": 0.05 + }, + { + "commit_hash": "66d5f2f", + "confidence_score": 0.02 + }, + { + "commit_hash": "411ddfa", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.47 + }, + { + "trial": 0, + "case_id": "page_size_zero_division", + "category": "pagination_boundary", + "culprit_hash": "6265f24", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [ + { + "commit_hash": "6265f24", + "confidence_score": 0.9 + }, + { + "commit_hash": "ed3861b", + "confidence_score": 0.85 + }, + { + "commit_hash": "0d241ff", + "confidence_score": 0.1 + }, + { + "commit_hash": "ef48acf", + "confidence_score": 0.02 + }, + { + "commit_hash": "debe268", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.86 + }, + { + "trial": 0, + "case_id": "payments_body_without_content_type", + "category": "http_content_type", + "culprit_hash": "33a9e2a", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [ + { + "commit_hash": "33a9e2a", + "confidence_score": 0.97 + }, + { + "commit_hash": "e60edf2", + "confidence_score": 0.03 + }, + { + "commit_hash": "91f3040", + "confidence_score": 0.02 + }, + { + "commit_hash": "5430f37", + "confidence_score": 0.01 + }, + { + "commit_hash": "64ee5a8", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.98 + }, + { + "trial": 0, + "case_id": "payments_retry_loop", + "category": "rate_limit_429", + "culprit_hash": "f07e927", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.357 + }, + { + "id": "auth_failures", + "similarity_score": 0.352 + }, + { + "id": "request_timeouts", + "similarity_score": 0.323 + } + ], + "ranking": [ + { + "commit_hash": "f07e927", + "confidence_score": 0.6 + }, + { + "commit_hash": "07a85f8", + "confidence_score": 0.35 + }, + { + "commit_hash": "c1fc55f", + "confidence_score": 0.02 + }, + { + "commit_hash": "51bc043", + "confidence_score": 0.0 + }, + { + "commit_hash": "429c4dd", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.28 + }, + { + "trial": 0, + "case_id": "redis_legacy_lrem_args", + "category": "dependency_version", + "culprit_hash": "0751831", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [ + { + "commit_hash": "0751831", + "confidence_score": 0.85 + }, + { + "commit_hash": "3059910", + "confidence_score": 0.5 + }, + { + "commit_hash": "fd92e84", + "confidence_score": 0.1 + }, + { + "commit_hash": "7b9dd2c", + "confidence_score": 0.02 + }, + { + "commit_hash": "957a42e", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.55 + }, + { + "trial": 0, + "case_id": "settings_env_quotes", + "category": "config_type_error", + "culprit_hash": "15aba07", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.36 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.278 + }, + { + "id": "memory_leaks", + "similarity_score": 0.241 + } + ], + "ranking": [ + { + "commit_hash": "15aba07", + "confidence_score": 0.85 + }, + { + "commit_hash": "cae9a6c", + "confidence_score": 0.2 + }, + { + "commit_hash": "423c51f", + "confidence_score": 0.05 + }, + { + "commit_hash": "3e36aa4", + "confidence_score": 0.05 + }, + { + "commit_hash": "55cfa0d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.33 + }, + { + "trial": 0, + "case_id": "since_filter_naive_cutoff", + "category": "timezone_mismatch", + "culprit_hash": "266c09c", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [ + { + "commit_hash": "266c09c", + "confidence_score": 0.9 + }, + { + "commit_hash": "f3ad8b7", + "confidence_score": 0.5 + }, + { + "commit_hash": "7f71cc9", + "confidence_score": 0.1 + }, + { + "commit_hash": "ef6b4a1", + "confidence_score": 0.02 + }, + { + "commit_hash": "5948e6d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.3 + }, + { + "trial": 0, + "case_id": "worker_queue_renamed", + "category": "queue_stall", + "culprit_hash": "f8d8258", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.411 + }, + { + "id": "rate_limiting", + "similarity_score": 0.409 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.383 + } + ], + "ranking": [ + { + "commit_hash": "f8d8258", + "confidence_score": 0.92 + }, + { + "commit_hash": "0f497ec", + "confidence_score": 0.15 + }, + { + "commit_hash": "110eec6", + "confidence_score": 0.05 + }, + { + "commit_hash": "aacbcd7", + "confidence_score": 0.03 + }, + { + "commit_hash": "3434d6b", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.16 + }, + { + "trial": 0, + "case_id": "worker_retains_job_payloads", + "category": "memory_growth", + "culprit_hash": "3037583", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.48 + }, + { + "id": "memory_leaks", + "similarity_score": 0.478 + }, + { + "id": "rate_limiting", + "similarity_score": 0.424 + } + ], + "ranking": [ + { + "commit_hash": "3037583", + "confidence_score": 0.95 + }, + { + "commit_hash": "088e964", + "confidence_score": 0.2 + }, + { + "commit_hash": "947ac52", + "confidence_score": 0.05 + }, + { + "commit_hash": "15f34e0", + "confidence_score": 0.02 + }, + { + "commit_hash": "63c2633", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.12 + }, + { + "trial": 1, + "case_id": "auth_issuer_mesh_hosts", + "category": "auth_rejection", + "culprit_hash": "5e68f8a", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.317 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "null_field_handling", + "similarity_score": 0.236 + } + ], + "ranking": [ + { + "commit_hash": "5e68f8a", + "confidence_score": 0.95 + }, + { + "commit_hash": "45a24d5", + "confidence_score": 0.05 + }, + { + "commit_hash": "d20025a", + "confidence_score": 0.02 + }, + { + "commit_hash": "df0ec4c", + "confidence_score": 0.02 + }, + { + "commit_hash": "298be8d", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.78 + }, + { + "trial": 1, + "case_id": "cache_contains_recursion", + "category": "infinite_recursion", + "culprit_hash": "4584056", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [ + { + "commit_hash": "4584056", + "confidence_score": 0.97 + }, + { + "commit_hash": "6484609", + "confidence_score": 0.15 + }, + { + "commit_hash": "7e63ea3", + "confidence_score": 0.05 + }, + { + "commit_hash": "6cc141a", + "confidence_score": 0.02 + }, + { + "commit_hash": "a98de00", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.75 + }, + { + "trial": 1, + "case_id": "cart_price_field_v2", + "category": "null_or_missing_field", + "culprit_hash": "d0ab213", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.26 + }, + { + "id": "pagination_errors", + "similarity_score": 0.207 + }, + { + "id": "memory_leaks", + "similarity_score": 0.179 + } + ], + "ranking": [ + { + "commit_hash": "d0ab213", + "confidence_score": 0.95 + }, + { + "commit_hash": "73e1933", + "confidence_score": 0.55 + }, + { + "commit_hash": "4fbde4c", + "confidence_score": 0.05 + }, + { + "commit_hash": "2bbbf3f", + "confidence_score": 0.02 + }, + { + "commit_hash": "f618f59", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.73 + }, + { + "trial": 1, + "case_id": "cart_sku_lowercased", + "category": "identifier_format", + "culprit_hash": "c2ec824", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [ + { + "commit_hash": "c2ec824", + "confidence_score": 0.75 + }, + { + "commit_hash": "64dde29", + "confidence_score": 0.35 + }, + { + "commit_hash": "d0cb0b3", + "confidence_score": 0.02 + }, + { + "commit_hash": "91f8d36", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.06 + }, + { + "trial": 1, + "case_id": "checkout_deadline_extra_call", + "category": "latency_timeout", + "culprit_hash": "efebc07", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.487 + }, + { + "id": "db_failures", + "similarity_score": 0.369 + }, + { + "id": "rate_limiting", + "similarity_score": 0.353 + } + ], + "ranking": [ + { + "commit_hash": "efebc07", + "confidence_score": 0.75 + }, + { + "commit_hash": "a035477", + "confidence_score": 0.55 + }, + { + "commit_hash": "2b765f4", + "confidence_score": 0.25 + }, + { + "commit_hash": "be85cc9", + "confidence_score": 0.1 + }, + { + "commit_hash": "91b8b07", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.14 + }, + { + "trial": 1, + "case_id": "config_reformat_hidden_string", + "category": "config_type_error", + "culprit_hash": "c3629ef", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [ + { + "commit_hash": "c3629ef", + "confidence_score": 0.85 + }, + { + "commit_hash": "3af442c", + "confidence_score": 0.3 + }, + { + "commit_hash": "976c988", + "confidence_score": 0.1 + }, + { + "commit_hash": "a6a3c41", + "confidence_score": 0.05 + }, + { + "commit_hash": "89436d3", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.58 + }, + { + "trial": 1, + "case_id": "db_fetch_one_leak", + "category": "db_pool_exhaustion", + "culprit_hash": "af9b861", + "expected_runbook_id": "db_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "af9b861", + "confidence_score": 0.95 + }, + { + "commit_hash": "f46cd8c", + "confidence_score": 0.35 + }, + { + "commit_hash": "1618815", + "confidence_score": 0.05 + }, + { + "commit_hash": "6d50d20", + "confidence_score": 0.02 + }, + { + "commit_hash": "1bf1024", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.8 + }, + { + "trial": 1, + "case_id": "export_tempfile_never_closed", + "category": "file_handle_leak", + "culprit_hash": "6eea0fc", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.625 + }, + { + "id": "memory_leaks", + "similarity_score": 0.432 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.291 + } + ], + "ranking": [ + { + "commit_hash": "5cedb21", + "confidence_score": 0.9 + }, + { + "commit_hash": "b245524", + "confidence_score": 0.2 + }, + { + "commit_hash": "6eea0fc", + "confidence_score": 0.1 + }, + { + "commit_hash": "0be651d", + "confidence_score": 0.02 + }, + { + "commit_hash": "7453733", + "confidence_score": 0.0 + } + ], + "culprit_rank": 3, + "error": null, + "rank_latency_s": 5.19 + }, + { + "trial": 1, + "case_id": "guest_email_normalise", + "category": "null_or_missing_field", + "culprit_hash": "7dd25fe", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.241 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.208 + }, + { + "id": "pagination_errors", + "similarity_score": 0.161 + } + ], + "ranking": [ + { + "commit_hash": "7dd25fe", + "confidence_score": 0.97 + }, + { + "commit_hash": "9519064", + "confidence_score": 0.1 + }, + { + "commit_hash": "1c0135b", + "confidence_score": 0.02 + }, + { + "commit_hash": "30a7434", + "confidence_score": 0.01 + }, + { + "commit_hash": "cde392a", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.77 + }, + { + "trial": 1, + "case_id": "httpx_028_drops_proxies", + "category": "dependency_version", + "culprit_hash": "b35322f", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.294 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.285 + }, + { + "id": "memory_leaks", + "similarity_score": 0.263 + } + ], + "ranking": [ + { + "commit_hash": "741f699", + "confidence_score": 0.97 + }, + { + "commit_hash": "b35322f", + "confidence_score": 0.9 + }, + { + "commit_hash": "b45189f", + "confidence_score": 0.02 + }, + { + "commit_hash": "e3601f6", + "confidence_score": 0.01 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 4.33 + }, + { + "trial": 1, + "case_id": "inventory_cache_key_per_request", + "category": "rate_limit_429", + "culprit_hash": "526f8dd", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.382 + }, + { + "id": "auth_failures", + "similarity_score": 0.381 + }, + { + "id": "null_field_handling", + "similarity_score": 0.309 + } + ], + "ranking": [ + { + "commit_hash": "526f8dd", + "confidence_score": 0.95 + }, + { + "commit_hash": "bf4cc5f", + "confidence_score": 0.35 + }, + { + "commit_hash": "93320f2", + "confidence_score": 0.03 + }, + { + "commit_hash": "2e1b06b", + "confidence_score": 0.01 + }, + { + "commit_hash": "d9caa65", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.16 + }, + { + "trial": 1, + "case_id": "inventory_v3_payload", + "category": "null_or_missing_field", + "culprit_hash": "1c29d94", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.33 + }, + { + "id": "memory_leaks", + "similarity_score": 0.281 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "1c29d94", + "confidence_score": 0.85 + }, + { + "commit_hash": "c6f0797", + "confidence_score": 0.05 + }, + { + "commit_hash": "16d2191", + "confidence_score": 0.05 + }, + { + "commit_hash": "5823d1f", + "confidence_score": 0.02 + }, + { + "commit_hash": "fb9b710", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.67 + }, + { + "trial": 1, + "case_id": "jwt_leeway_milliseconds", + "category": "auth_rejection", + "culprit_hash": "2eeb222", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.488 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.198 + }, + { + "id": "request_timeouts", + "similarity_score": 0.159 + } + ], + "ranking": [ + { + "commit_hash": "2eeb222", + "confidence_score": 0.95 + }, + { + "commit_hash": "e8548c5", + "confidence_score": 0.2 + }, + { + "commit_hash": "2e11931", + "confidence_score": 0.1 + }, + { + "commit_hash": "a028c62", + "confidence_score": 0.02 + }, + { + "commit_hash": "716e22d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.94 + }, + { + "trial": 1, + "case_id": "money_returns_decimal", + "category": "json_serialization", + "culprit_hash": "b14897b", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [ + { + "commit_hash": "b14897b", + "confidence_score": 0.98 + }, + { + "commit_hash": "dedc41a", + "confidence_score": 0.1 + }, + { + "commit_hash": "626c22d", + "confidence_score": 0.02 + }, + { + "commit_hash": "cd4a49f", + "confidence_score": 0.01 + }, + { + "commit_hash": "92cb3da", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.27 + }, + { + "trial": 1, + "case_id": "order_read_sync_audit", + "category": "latency_timeout", + "culprit_hash": "8aed86c", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.539 + }, + { + "id": "rate_limiting", + "similarity_score": 0.369 + }, + { + "id": "auth_failures", + "similarity_score": 0.35 + } + ], + "ranking": [ + { + "commit_hash": "8aed86c", + "confidence_score": 0.9 + }, + { + "commit_hash": "d3d6f87", + "confidence_score": 0.35 + }, + { + "commit_hash": "94578cc", + "confidence_score": 0.3 + }, + { + "commit_hash": "5abe2e4", + "confidence_score": 0.05 + }, + { + "commit_hash": "b7ea08d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 8.77 + }, + { + "trial": 1, + "case_id": "order_sql_missing_param", + "category": "sql_param_mismatch", + "culprit_hash": "9ef79b2", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [ + { + "commit_hash": "9ef79b2", + "confidence_score": 0.85 + }, + { + "commit_hash": "661f871", + "confidence_score": 0.2 + }, + { + "commit_hash": "93ebe2f", + "confidence_score": 0.05 + }, + { + "commit_hash": "66d5f2f", + "confidence_score": 0.02 + }, + { + "commit_hash": "411ddfa", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.28 + }, + { + "trial": 1, + "case_id": "page_size_zero_division", + "category": "pagination_boundary", + "culprit_hash": "6265f24", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [ + { + "commit_hash": "ed3861b", + "confidence_score": 0.9 + }, + { + "commit_hash": "6265f24", + "confidence_score": 0.85 + }, + { + "commit_hash": "0d241ff", + "confidence_score": 0.1 + }, + { + "commit_hash": "ef48acf", + "confidence_score": 0.02 + }, + { + "commit_hash": "debe268", + "confidence_score": 0.01 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 5.05 + }, + { + "trial": 1, + "case_id": "payments_body_without_content_type", + "category": "http_content_type", + "culprit_hash": "33a9e2a", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [ + { + "commit_hash": "33a9e2a", + "confidence_score": 0.95 + }, + { + "commit_hash": "e60edf2", + "confidence_score": 0.05 + }, + { + "commit_hash": "5430f37", + "confidence_score": 0.02 + }, + { + "commit_hash": "91f3040", + "confidence_score": 0.02 + }, + { + "commit_hash": "64ee5a8", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.03 + }, + { + "trial": 1, + "case_id": "payments_retry_loop", + "category": "rate_limit_429", + "culprit_hash": "f07e927", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.357 + }, + { + "id": "auth_failures", + "similarity_score": 0.352 + }, + { + "id": "request_timeouts", + "similarity_score": 0.323 + } + ], + "ranking": [ + { + "commit_hash": "f07e927", + "confidence_score": 0.85 + }, + { + "commit_hash": "07a85f8", + "confidence_score": 0.75 + }, + { + "commit_hash": "c1fc55f", + "confidence_score": 0.05 + }, + { + "commit_hash": "429c4dd", + "confidence_score": 0.0 + }, + { + "commit_hash": "51bc043", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.41 + }, + { + "trial": 1, + "case_id": "redis_legacy_lrem_args", + "category": "dependency_version", + "culprit_hash": "0751831", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [ + { + "commit_hash": "0751831", + "confidence_score": 0.85 + }, + { + "commit_hash": "3059910", + "confidence_score": 0.4 + }, + { + "commit_hash": "fd92e84", + "confidence_score": 0.15 + }, + { + "commit_hash": "7b9dd2c", + "confidence_score": 0.03 + }, + { + "commit_hash": "957a42e", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.89 + }, + { + "trial": 1, + "case_id": "settings_env_quotes", + "category": "config_type_error", + "culprit_hash": "15aba07", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.36 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.278 + }, + { + "id": "memory_leaks", + "similarity_score": 0.241 + } + ], + "ranking": [ + { + "commit_hash": "15aba07", + "confidence_score": 0.85 + }, + { + "commit_hash": "cae9a6c", + "confidence_score": 0.1 + }, + { + "commit_hash": "423c51f", + "confidence_score": 0.05 + }, + { + "commit_hash": "3e36aa4", + "confidence_score": 0.05 + }, + { + "commit_hash": "55cfa0d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.3 + }, + { + "trial": 1, + "case_id": "since_filter_naive_cutoff", + "category": "timezone_mismatch", + "culprit_hash": "266c09c", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [ + { + "commit_hash": "266c09c", + "confidence_score": 0.85 + }, + { + "commit_hash": "f3ad8b7", + "confidence_score": 0.45 + }, + { + "commit_hash": "7f71cc9", + "confidence_score": 0.05 + }, + { + "commit_hash": "5948e6d", + "confidence_score": 0.02 + }, + { + "commit_hash": "ef6b4a1", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.12 + }, + { + "trial": 1, + "case_id": "worker_queue_renamed", + "category": "queue_stall", + "culprit_hash": "f8d8258", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.411 + }, + { + "id": "rate_limiting", + "similarity_score": 0.409 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.383 + } + ], + "ranking": [ + { + "commit_hash": "f8d8258", + "confidence_score": 0.9 + }, + { + "commit_hash": "0f497ec", + "confidence_score": 0.15 + }, + { + "commit_hash": "110eec6", + "confidence_score": 0.05 + }, + { + "commit_hash": "aacbcd7", + "confidence_score": 0.03 + }, + { + "commit_hash": "3434d6b", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.3 + }, + { + "trial": 1, + "case_id": "worker_retains_job_payloads", + "category": "memory_growth", + "culprit_hash": "3037583", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.48 + }, + { + "id": "memory_leaks", + "similarity_score": 0.478 + }, + { + "id": "rate_limiting", + "similarity_score": 0.424 + } + ], + "ranking": [ + { + "commit_hash": "3037583", + "confidence_score": 0.95 + }, + { + "commit_hash": "088e964", + "confidence_score": 0.2 + }, + { + "commit_hash": "947ac52", + "confidence_score": 0.05 + }, + { + "commit_hash": "15f34e0", + "confidence_score": 0.02 + }, + { + "commit_hash": "63c2633", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.48 + }, + { + "trial": 2, + "case_id": "auth_issuer_mesh_hosts", + "category": "auth_rejection", + "culprit_hash": "5e68f8a", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.317 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "null_field_handling", + "similarity_score": 0.236 + } + ], + "ranking": [ + { + "commit_hash": "5e68f8a", + "confidence_score": 0.95 + }, + { + "commit_hash": "45a24d5", + "confidence_score": 0.05 + }, + { + "commit_hash": "df0ec4c", + "confidence_score": 0.05 + }, + { + "commit_hash": "298be8d", + "confidence_score": 0.02 + }, + { + "commit_hash": "d20025a", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.59 + }, + { + "trial": 2, + "case_id": "cache_contains_recursion", + "category": "infinite_recursion", + "culprit_hash": "4584056", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [ + { + "commit_hash": "4584056", + "confidence_score": 0.97 + }, + { + "commit_hash": "6484609", + "confidence_score": 0.1 + }, + { + "commit_hash": "6cc141a", + "confidence_score": 0.02 + }, + { + "commit_hash": "7e63ea3", + "confidence_score": 0.02 + }, + { + "commit_hash": "a98de00", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.14 + }, + { + "trial": 2, + "case_id": "cart_price_field_v2", + "category": "null_or_missing_field", + "culprit_hash": "d0ab213", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.26 + }, + { + "id": "pagination_errors", + "similarity_score": 0.207 + }, + { + "id": "memory_leaks", + "similarity_score": 0.179 + } + ], + "ranking": [ + { + "commit_hash": "d0ab213", + "confidence_score": 0.95 + }, + { + "commit_hash": "73e1933", + "confidence_score": 0.5 + }, + { + "commit_hash": "4fbde4c", + "confidence_score": 0.05 + }, + { + "commit_hash": "2bbbf3f", + "confidence_score": 0.02 + }, + { + "commit_hash": "f618f59", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.01 + }, + { + "trial": 2, + "case_id": "cart_sku_lowercased", + "category": "identifier_format", + "culprit_hash": "c2ec824", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [ + { + "commit_hash": "c2ec824", + "confidence_score": 0.8 + }, + { + "commit_hash": "64dde29", + "confidence_score": 0.15 + }, + { + "commit_hash": "d0cb0b3", + "confidence_score": 0.02 + }, + { + "commit_hash": "91f8d36", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.86 + }, + { + "trial": 2, + "case_id": "checkout_deadline_extra_call", + "category": "latency_timeout", + "culprit_hash": "efebc07", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.487 + }, + { + "id": "db_failures", + "similarity_score": 0.369 + }, + { + "id": "rate_limiting", + "similarity_score": 0.353 + } + ], + "ranking": [ + { + "commit_hash": "efebc07", + "confidence_score": 0.85 + }, + { + "commit_hash": "a035477", + "confidence_score": 0.6 + }, + { + "commit_hash": "2b765f4", + "confidence_score": 0.25 + }, + { + "commit_hash": "be85cc9", + "confidence_score": 0.05 + }, + { + "commit_hash": "91b8b07", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.55 + }, + { + "trial": 2, + "case_id": "config_reformat_hidden_string", + "category": "config_type_error", + "culprit_hash": "c3629ef", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [ + { + "commit_hash": "c3629ef", + "confidence_score": 0.9 + }, + { + "commit_hash": "3af442c", + "confidence_score": 0.15 + }, + { + "commit_hash": "976c988", + "confidence_score": 0.1 + }, + { + "commit_hash": "a6a3c41", + "confidence_score": 0.05 + }, + { + "commit_hash": "89436d3", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.55 + }, + { + "trial": 2, + "case_id": "db_fetch_one_leak", + "category": "db_pool_exhaustion", + "culprit_hash": "af9b861", + "expected_runbook_id": "db_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "af9b861", + "confidence_score": 0.95 + }, + { + "commit_hash": "f46cd8c", + "confidence_score": 0.4 + }, + { + "commit_hash": "1618815", + "confidence_score": 0.05 + }, + { + "commit_hash": "6d50d20", + "confidence_score": 0.02 + }, + { + "commit_hash": "1bf1024", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.27 + }, + { + "trial": 2, + "case_id": "export_tempfile_never_closed", + "category": "file_handle_leak", + "culprit_hash": "6eea0fc", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.625 + }, + { + "id": "memory_leaks", + "similarity_score": 0.432 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.291 + } + ], + "ranking": [ + { + "commit_hash": "5cedb21", + "confidence_score": 0.9 + }, + { + "commit_hash": "b245524", + "confidence_score": 0.2 + }, + { + "commit_hash": "6eea0fc", + "confidence_score": 0.1 + }, + { + "commit_hash": "0be651d", + "confidence_score": 0.02 + }, + { + "commit_hash": "7453733", + "confidence_score": 0.01 + } + ], + "culprit_rank": 3, + "error": null, + "rank_latency_s": 5.55 + }, + { + "trial": 2, + "case_id": "guest_email_normalise", + "category": "null_or_missing_field", + "culprit_hash": "7dd25fe", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.241 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.208 + }, + { + "id": "pagination_errors", + "similarity_score": 0.161 + } + ], + "ranking": [ + { + "commit_hash": "7dd25fe", + "confidence_score": 0.97 + }, + { + "commit_hash": "9519064", + "confidence_score": 0.1 + }, + { + "commit_hash": "1c0135b", + "confidence_score": 0.02 + }, + { + "commit_hash": "cde392a", + "confidence_score": 0.01 + }, + { + "commit_hash": "30a7434", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.08 + }, + { + "trial": 2, + "case_id": "httpx_028_drops_proxies", + "category": "dependency_version", + "culprit_hash": "b35322f", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.294 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.285 + }, + { + "id": "memory_leaks", + "similarity_score": 0.263 + } + ], + "ranking": [ + { + "commit_hash": "741f699", + "confidence_score": 0.95 + }, + { + "commit_hash": "b35322f", + "confidence_score": 0.9 + }, + { + "commit_hash": "b45189f", + "confidence_score": 0.02 + }, + { + "commit_hash": "e3601f6", + "confidence_score": 0.01 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 4.5 + }, + { + "trial": 2, + "case_id": "inventory_cache_key_per_request", + "category": "rate_limit_429", + "culprit_hash": "526f8dd", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.382 + }, + { + "id": "auth_failures", + "similarity_score": 0.381 + }, + { + "id": "null_field_handling", + "similarity_score": 0.309 + } + ], + "ranking": [ + { + "commit_hash": "526f8dd", + "confidence_score": 0.85 + }, + { + "commit_hash": "bf4cc5f", + "confidence_score": 0.5 + }, + { + "commit_hash": "93320f2", + "confidence_score": 0.05 + }, + { + "commit_hash": "2e1b06b", + "confidence_score": 0.02 + }, + { + "commit_hash": "d9caa65", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.83 + }, + { + "trial": 2, + "case_id": "inventory_v3_payload", + "category": "null_or_missing_field", + "culprit_hash": "1c29d94", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.33 + }, + { + "id": "memory_leaks", + "similarity_score": 0.281 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "1c29d94", + "confidence_score": 0.85 + }, + { + "commit_hash": "c6f0797", + "confidence_score": 0.05 + }, + { + "commit_hash": "16d2191", + "confidence_score": 0.05 + }, + { + "commit_hash": "5823d1f", + "confidence_score": 0.02 + }, + { + "commit_hash": "fb9b710", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.5 + }, + { + "trial": 2, + "case_id": "jwt_leeway_milliseconds", + "category": "auth_rejection", + "culprit_hash": "2eeb222", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.488 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.198 + }, + { + "id": "request_timeouts", + "similarity_score": 0.159 + } + ], + "ranking": [ + { + "commit_hash": "2eeb222", + "confidence_score": 0.95 + }, + { + "commit_hash": "e8548c5", + "confidence_score": 0.3 + }, + { + "commit_hash": "2e11931", + "confidence_score": 0.2 + }, + { + "commit_hash": "a028c62", + "confidence_score": 0.05 + }, + { + "commit_hash": "716e22d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.45 + }, + { + "trial": 2, + "case_id": "money_returns_decimal", + "category": "json_serialization", + "culprit_hash": "b14897b", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [ + { + "commit_hash": "b14897b", + "confidence_score": 0.98 + }, + { + "commit_hash": "dedc41a", + "confidence_score": 0.05 + }, + { + "commit_hash": "626c22d", + "confidence_score": 0.02 + }, + { + "commit_hash": "cd4a49f", + "confidence_score": 0.01 + }, + { + "commit_hash": "92cb3da", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.22 + }, + { + "trial": 2, + "case_id": "order_read_sync_audit", + "category": "latency_timeout", + "culprit_hash": "8aed86c", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.539 + }, + { + "id": "rate_limiting", + "similarity_score": 0.369 + }, + { + "id": "auth_failures", + "similarity_score": 0.35 + } + ], + "ranking": [ + { + "commit_hash": "8aed86c", + "confidence_score": 0.9 + }, + { + "commit_hash": "94578cc", + "confidence_score": 0.35 + }, + { + "commit_hash": "d3d6f87", + "confidence_score": 0.3 + }, + { + "commit_hash": "5abe2e4", + "confidence_score": 0.05 + }, + { + "commit_hash": "b7ea08d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.03 + }, + { + "trial": 2, + "case_id": "order_sql_missing_param", + "category": "sql_param_mismatch", + "culprit_hash": "9ef79b2", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [ + { + "commit_hash": "9ef79b2", + "confidence_score": 0.85 + }, + { + "commit_hash": "661f871", + "confidence_score": 0.2 + }, + { + "commit_hash": "93ebe2f", + "confidence_score": 0.05 + }, + { + "commit_hash": "66d5f2f", + "confidence_score": 0.02 + }, + { + "commit_hash": "411ddfa", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.77 + }, + { + "trial": 2, + "case_id": "page_size_zero_division", + "category": "pagination_boundary", + "culprit_hash": "6265f24", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [ + { + "commit_hash": "6265f24", + "confidence_score": 0.9 + }, + { + "commit_hash": "ed3861b", + "confidence_score": 0.75 + }, + { + "commit_hash": "0d241ff", + "confidence_score": 0.05 + }, + { + "commit_hash": "ef48acf", + "confidence_score": 0.02 + }, + { + "commit_hash": "debe268", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.78 + }, + { + "trial": 2, + "case_id": "payments_body_without_content_type", + "category": "http_content_type", + "culprit_hash": "33a9e2a", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [ + { + "commit_hash": "33a9e2a", + "confidence_score": 0.97 + }, + { + "commit_hash": "e60edf2", + "confidence_score": 0.05 + }, + { + "commit_hash": "91f3040", + "confidence_score": 0.03 + }, + { + "commit_hash": "5430f37", + "confidence_score": 0.02 + }, + { + "commit_hash": "64ee5a8", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.27 + }, + { + "trial": 2, + "case_id": "payments_retry_loop", + "category": "rate_limit_429", + "culprit_hash": "f07e927", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.357 + }, + { + "id": "auth_failures", + "similarity_score": 0.352 + }, + { + "id": "request_timeouts", + "similarity_score": 0.323 + } + ], + "ranking": [ + { + "commit_hash": "f07e927", + "confidence_score": 0.85 + }, + { + "commit_hash": "07a85f8", + "confidence_score": 0.8 + }, + { + "commit_hash": "c1fc55f", + "confidence_score": 0.05 + }, + { + "commit_hash": "51bc043", + "confidence_score": 0.0 + }, + { + "commit_hash": "429c4dd", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.5 + }, + { + "trial": 2, + "case_id": "redis_legacy_lrem_args", + "category": "dependency_version", + "culprit_hash": "0751831", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [ + { + "commit_hash": "0751831", + "confidence_score": 0.85 + }, + { + "commit_hash": "3059910", + "confidence_score": 0.35 + }, + { + "commit_hash": "fd92e84", + "confidence_score": 0.1 + }, + { + "commit_hash": "7b9dd2c", + "confidence_score": 0.02 + }, + { + "commit_hash": "957a42e", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.77 + }, + { + "trial": 2, + "case_id": "settings_env_quotes", + "category": "config_type_error", + "culprit_hash": "15aba07", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.36 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.278 + }, + { + "id": "memory_leaks", + "similarity_score": 0.241 + } + ], + "ranking": [ + { + "commit_hash": "15aba07", + "confidence_score": 0.85 + }, + { + "commit_hash": "cae9a6c", + "confidence_score": 0.1 + }, + { + "commit_hash": "423c51f", + "confidence_score": 0.05 + }, + { + "commit_hash": "3e36aa4", + "confidence_score": 0.05 + }, + { + "commit_hash": "55cfa0d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.22 + }, + { + "trial": 2, + "case_id": "since_filter_naive_cutoff", + "category": "timezone_mismatch", + "culprit_hash": "266c09c", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [ + { + "commit_hash": "266c09c", + "confidence_score": 0.92 + }, + { + "commit_hash": "f3ad8b7", + "confidence_score": 0.35 + }, + { + "commit_hash": "7f71cc9", + "confidence_score": 0.05 + }, + { + "commit_hash": "ef6b4a1", + "confidence_score": 0.02 + }, + { + "commit_hash": "5948e6d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.11 + }, + { + "trial": 2, + "case_id": "worker_queue_renamed", + "category": "queue_stall", + "culprit_hash": "f8d8258", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.411 + }, + { + "id": "rate_limiting", + "similarity_score": 0.409 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.383 + } + ], + "ranking": [ + { + "commit_hash": "f8d8258", + "confidence_score": 0.9 + }, + { + "commit_hash": "0f497ec", + "confidence_score": 0.2 + }, + { + "commit_hash": "110eec6", + "confidence_score": 0.05 + }, + { + "commit_hash": "aacbcd7", + "confidence_score": 0.02 + }, + { + "commit_hash": "3434d6b", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.61 + }, + { + "trial": 2, + "case_id": "worker_retains_job_payloads", + "category": "memory_growth", + "culprit_hash": "3037583", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.48 + }, + { + "id": "memory_leaks", + "similarity_score": 0.478 + }, + { + "id": "rate_limiting", + "similarity_score": 0.424 + } + ], + "ranking": [ + { + "commit_hash": "3037583", + "confidence_score": 0.95 + }, + { + "commit_hash": "088e964", + "confidence_score": 0.2 + }, + { + "commit_hash": "947ac52", + "confidence_score": 0.05 + }, + { + "commit_hash": "15f34e0", + "confidence_score": 0.02 + }, + { + "commit_hash": "63c2633", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.0 + } + ], + "finished_at": "2026-10-04T19:51:03.625703+00:00" +} \ No newline at end of file From 530b73f1470e8620e26ded1fb96271f477c3309e Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 16:00:28 -0400 Subject: [PATCH 15/19] feat(core): optional runbook context in commit ranking behind SENTINEL_RAG_RANKING - find_matching_runbooks(include_content=True) adds Symptoms + Root Causes (capped) - rank_suspect_commits(runbooks=) inserts a block only above the floor (0.35, frozen) - flag defaults off; prompt without runbooks is byte-identical to the golden baseline - retrieval failure logs and ranks without runbooks instead of reverting the incident - runbook text is prompt-only; persisted matched_runbooks shape unchanged Co-Authored-By: Claude Opus 5.5 --- core/orchestrator.py | 20 ++++- core/services/llm_analyzer.py | 27 +++++- core/services/vector_store.py | 31 +++++-- core/tests/test_rag_ranking.py | 155 +++++++++++++++++++++++++++++++++ eval/config.py | 5 ++ eval/run_eval.py | 3 + 6 files changed, 227 insertions(+), 14 deletions(-) create mode 100644 core/tests/test_rag_ranking.py diff --git a/core/orchestrator.py b/core/orchestrator.py index 5fdec1b..d694ee4 100644 --- a/core/orchestrator.py +++ b/core/orchestrator.py @@ -13,6 +13,8 @@ _DEDUP_WINDOW_S = float(os.getenv("SENTINEL_DEDUP_WINDOW_S", "300")) _DEDUP_THRESHOLD = int(os.getenv("SENTINEL_DEDUP_THRESHOLD", "1")) # Nth hit in window fires _recent_alerts: dict[str, list[float]] = defaultdict(list) # signature -> hit times +# Pass matched runbook text into commit ranking. Default off until the Phase 9 A/B decides (METRICS.md). +_RAG_RANKING = os.getenv("SENTINEL_RAG_RANKING", "0") == "1" _open_signatures: dict[str, str] = {} # signature -> live incident_id @@ -69,12 +71,21 @@ def run_diagnostics(incident_id: str, alert_data: dict): diffs = git_client.get_recent_diffs(REPO_PATH, alert_data["timestamp"]) print(f"[orchestrator] {incident_id} -- {len(diffs)} commits in window") - runbooks = vector_store.find_matching_runbooks(alert_data.get("error_signature", "")) - print(f"[orchestrator] {incident_id} -- {len(runbooks)} runbooks matched") + try: + runbooks = vector_store.find_matching_runbooks( + alert_data.get("error_signature", ""), include_content=_RAG_RANKING + ) + print(f"[orchestrator] {incident_id} -- {len(runbooks)} runbooks matched") + except Exception as e: # retrieval is optional context; never a reason to abort triage + runbooks = [] + print(f"[orchestrator] {incident_id} -- runbook retrieval failed, ranking without: {e}") degraded_reason = None try: - ranked = llm_analyzer.rank_suspect_commits(diffs, alert_data) if diffs else [] + rank_runbooks = runbooks if _RAG_RANKING else None + ranked = ( + llm_analyzer.rank_suspect_commits(diffs, alert_data, runbooks=rank_runbooks) if diffs else [] + ) print(f"[orchestrator] {incident_id} -- LLM ranked {len(ranked)} suspects") except llm_analyzer.LLMUnavailable as e: ranked, degraded_reason = [], str(e) @@ -82,7 +93,8 @@ def run_diagnostics(incident_id: str, alert_data: dict): diagnostics = { "suspect_commits": ranked, - "matched_runbooks": runbooks, + # runbook text is prompt-only; never persisted or sent to Slack + "matched_runbooks": [{k: v for k, v in r.items() if k != "content"} for r in runbooks], "impact_assessment": { "error_rate_delta_pct": alert_data.get("error_rate_pct"), "estimated_affected_users": None, diff --git a/core/services/llm_analyzer.py b/core/services/llm_analyzer.py index 63f8ac5..edc8fdb 100644 --- a/core/services/llm_analyzer.py +++ b/core/services/llm_analyzer.py @@ -1,7 +1,31 @@ +import os + import anthropic from .llm import chat +# Frozen at Phase 9E from the v2 baseline retrieval scores (see METRICS.md, Phase 9). +RUNBOOK_FLOOR = float(os.getenv("SENTINEL_RUNBOOK_FLOOR", "0.35")) + + +def _runbooks_block(runbooks: list[dict] | None) -> str: + """'' unless a runbook clears the floor, so the prompt with no runbooks is byte-identical to baseline.""" + relevant = [r for r in runbooks or [] if r["similarity_score"] >= RUNBOOK_FLOOR and r.get("content")] + if not relevant: + return "" + body = "\n".join( + f'\n{r["content"]}\n' + for r in relevant + ) + return ( + "RUNBOOKS (retrieved by text similarity to the error; they may be irrelevant):\n" + f"\n{body}\n\n" + "Runbooks describe known failure classes and can help interpret the error; " + "ignore any that do not fit it. The diffs are the evidence: rank a commit highly only if " + "its diff plausibly produces this error, not because it touches something a runbook mentions. " + "Runbook root causes are possibilities, not findings.\n\n" + ) + class LLMUnavailable(Exception): """LLM output could not be produced: timeout/network, malformed JSON, or refusal.""" @@ -33,7 +57,7 @@ class LLMUnavailable(Exception): } -def rank_suspect_commits(diffs: list[dict], alert: dict) -> list[dict]: +def rank_suspect_commits(diffs: list[dict], alert: dict, runbooks: list[dict] | None = None) -> list[dict]: if not diffs: return [] @@ -50,6 +74,7 @@ def rank_suspect_commits(diffs: list[dict], alert: dict) -> list[dict]: f" name: {alert['alert_name']}\n" f" error: {alert['error_signature']}\n" f" time: {alert['timestamp']}\n\n" + f"{_runbooks_block(runbooks)}" f"RECENT COMMITS:\n{commits_block}\n\n" f"Use rank_commits to return every commit with a confidence_score (0=unrelated, 1=certain cause) " f"and a one-sentence rationale." diff --git a/core/services/vector_store.py b/core/services/vector_store.py index 41b93e0..b7ab392 100644 --- a/core/services/vector_store.py +++ b/core/services/vector_store.py @@ -1,9 +1,11 @@ import os +import re from pathlib import Path import chromadb CHROMA_PATH = os.getenv("SENTINEL_CHROMA_PATH", ".chroma") COLLECTION = "runbooks" +CONTENT_CAP = 1500 # chars of Symptoms + Root Causes sent to the ranking prompt per runbook def _col(): @@ -23,7 +25,7 @@ def ingest_runbooks(runbooks_dir: str) -> int: return len(docs) -def find_matching_runbooks(error_signature: str, top_k: int = 3) -> list[dict]: +def find_matching_runbooks(error_signature: str, top_k: int = 3, include_content: bool = False) -> list[dict]: col = _col() count = col.count() if count == 0: @@ -37,17 +39,28 @@ def find_matching_runbooks(error_signature: str, top_k: int = 3) -> list[dict]: results["documents"][0], results["distances"][0], ): - out.append( - { - "id": doc_id, - "title": meta["title"], - "similarity_score": round(1 - dist, 3), - "primary_action": _first_action(doc), - } - ) + match = { + "id": doc_id, + "title": meta["title"], + "similarity_score": round(1 - dist, 3), + "primary_action": _first_action(doc), + } + if include_content: # prompt-only; callers must not persist it + match["content"] = _prompt_content(doc) + out.append(match) return out +def _section(doc: str, heading: str) -> str: + m = re.search(rf"^## {heading}\n(.*?)(?=^## |\Z)", doc, flags=re.M | re.S) + return m.group(1).strip() if m else "" + + +def _prompt_content(doc: str) -> str: + text = f"Symptoms:\n{_section(doc, 'Symptoms')}\n\nRoot causes:\n{_section(doc, 'Root Causes')}" + return text[:CONTENT_CAP] + + def _first_action(doc: str) -> str: in_actions = False for line in doc.splitlines(): diff --git a/core/tests/test_rag_ranking.py b/core/tests/test_rag_ranking.py new file mode 100644 index 0000000..61d0511 --- /dev/null +++ b/core/tests/test_rag_ranking.py @@ -0,0 +1,155 @@ +# ponytail: Phase 9E guards: the RAG flag must not move the A/B control; retrieval must never abort triage +import json +from pathlib import Path +from types import SimpleNamespace as NS +from unittest.mock import patch + +import jsonschema + +from core import orchestrator +from core.services import llm_analyzer, vector_store + +SCHEMA = json.loads((Path(__file__).parent.parent.parent / "contract" / "pipeline_schema.json").read_text()) + +_DIFFS = [ + { + "commit_hash": "a1b2c3d", + "author": "dev@example.com", + "timestamp": "2026-09-01T11:30:00+00:00", + "subject": "config: raise pool size", + "diff": '- "pool_size": 10\n+ "pool_size": 12', + }, + { + "commit_hash": "e4f5a6b", + "author": "ops@example.com", + "timestamp": "2026-09-01T11:45:00+00:00", + "subject": "docs: readme", + "diff": "+ more docs", + }, +] +_ALERT = { + "source": "mock_sentry", + "alert_name": "HTTP_500_Internal_Server_Error", + "error_signature": "PoolError: connection pool exhausted", + "timestamp": "2026-09-01T12:00:00+00:00", +} +# Captured from rank_suspect_commits before the runbooks parameter existed (Phase 9E). Never edit: +# the baseline/RAG comparison is only valid while the no-runbook prompt is byte-identical to this. +_GOLDEN_BASELINE_PROMPT = ( + "You are an incident triage engine.\n\nALERT:\n name: HTTP_500_Internal_Server_Error\n" + " error: PoolError: connection pool exhausted\n time: 2026-09-01T12:00:00+00:00\n\n" + "RECENT COMMITS:\nCOMMIT a1b2c3d | dev@example.com | 2026-09-01T11:30:00+00:00\n" + 'Subject: config: raise pool size\nDiff:\n- "pool_size": 10\n+ "pool_size": 12\n\n' + "COMMIT e4f5a6b | ops@example.com | 2026-09-01T11:45:00+00:00\nSubject: docs: readme\nDiff:\n" + "+ more docs\n\nUse rank_commits to return every commit with a confidence_score " + "(0=unrelated, 1=certain cause) and a one-sentence rationale." +) +_RUNBOOK = { + "id": "db_failures", + "title": "Db Failures", + "similarity_score": 0.5, + "primary_action": "Check pool.", + "content": "Symptoms:\n- pool exhausted\n\nRoot causes:\n1. Connection leak", +} + + +def _prompt(runbooks): + captured = [] + + def fake_chat(messages, **_): + captured.append(messages[0]["content"]) + return NS( + stop_reason="tool_use", + content=[NS(type="tool_use", name="rank_commits", input={"ranked_commits": []})], + ) + + with patch.object(llm_analyzer, "chat", fake_chat): + llm_analyzer.rank_suspect_commits(_DIFFS, _ALERT, runbooks=runbooks) + return captured[0] + + +def test_no_runbooks_prompt_is_byte_identical_to_baseline(): + assert _prompt(None) == _GOLDEN_BASELINE_PROMPT + assert _prompt([]) == _GOLDEN_BASELINE_PROMPT + + +def test_runbooks_below_floor_are_omitted_entirely(): + low = {**_RUNBOOK, "similarity_score": llm_analyzer.RUNBOOK_FLOOR - 0.01} + assert _prompt([low]) == _GOLDEN_BASELINE_PROMPT # no empty block + + +def test_runbooks_above_floor_sit_between_alert_and_commits(): + prompt = _prompt([_RUNBOOK]) + assert '' in prompt and "Connection leak" in prompt + assert prompt.index("ALERT:") < prompt.index("") < prompt.index("RECENT COMMITS:") + assert prompt.replace(llm_analyzer._runbooks_block([_RUNBOOK]), "") == _GOLDEN_BASELINE_PROMPT + + +def _run_diagnostics(rag: bool, retrieval: dict): + with ( + patch.object(orchestrator, "_RAG_RANKING", rag), + patch.object(orchestrator, "db") as mock_db, + patch.object(orchestrator.git_client, "get_recent_diffs", return_value=_DIFFS), + patch.object(orchestrator.vector_store, "find_matching_runbooks", **retrieval), + patch.object(orchestrator.llm_analyzer, "rank_suspect_commits", return_value=[]) as rank, + patch.object(orchestrator.notifier, "post_incident_to_slack"), + ): + mock_db.get_incident.return_value = {"id": "inc_t", "status": "triaging"} + orchestrator.run_diagnostics("inc_t", _ALERT) + statuses = [c.args for c in mock_db.update_status.call_args_list] + diagnostics = mock_db.update_diagnostics.call_args.args[1] + return rank, statuses, diagnostics + + +def test_retrieval_failure_still_ranks_and_does_not_revert(): + rank, statuses, diagnostics = _run_diagnostics(True, {"side_effect": RuntimeError("chroma down")}) + rank.assert_called_once() + assert rank.call_args.kwargs["runbooks"] == [] + assert ("inc_t", "triggered") not in statuses + assert diagnostics["matched_runbooks"] == [] + + +def test_flag_off_ranks_without_runbooks(): + rank, _, _ = _run_diagnostics(False, {"return_value": [_RUNBOOK]}) + assert rank.call_args.kwargs["runbooks"] is None + + +def test_persisted_runbooks_carry_no_content_and_match_contract(): + rank, _, diagnostics = _run_diagnostics(True, {"return_value": [_RUNBOOK]}) + assert rank.call_args.kwargs["runbooks"] == [_RUNBOOK] # content reaches the prompt... + assert all("content" not in r for r in diagnostics["matched_runbooks"]) # ...but is never stored + payload = { + "incident_id": "inc_t", + "status": "triaging", + "trigger": _ALERT, + "diagnostics": diagnostics, + "postmortem_draft_url": None, + } + jsonschema.validate(payload, SCHEMA) + + +def test_include_content_returns_symptoms_and_root_causes_capped(): + doc = ( + "# T\n\n## Symptoms\n\n- s1\n\n## Root Causes\n\n1. r1 " + + "x" * 3000 + + "\n\n## Immediate Actions\n\n1. act\n" + ) + + class FakeCol: + def count(self): + return 1 + + def query(self, **_): + return { + "ids": [["rb"]], + "metadatas": [[{"title": "Rb"}]], + "documents": [[doc]], + "distances": [[0.5]], + } + + with patch.object(vector_store, "_col", return_value=FakeCol()): + plain = vector_store.find_matching_runbooks("x")[0] + rich = vector_store.find_matching_runbooks("x", include_content=True)[0] + assert "content" not in plain + assert rich["content"].startswith("Symptoms:\n- s1\n\nRoot causes:\n1. r1") + assert len(rich["content"]) == vector_store.CONTENT_CAP and "Immediate" not in rich["content"] diff --git a/eval/config.py b/eval/config.py index 049f410..9eb896f 100644 --- a/eval/config.py +++ b/eval/config.py @@ -13,3 +13,8 @@ # Words that would announce the culprit. Checked (word-prefix, case-insensitive) in commit # messages and in every file the case writes, including the shared base. BANNED_WORDS = ["bug", "inject", "break", "fault", "fail", "chaos", "oops", "fix", "hotfix", "revert"] + +# Runbook similarity floor for the RAG condition, frozen at 9E before any RAG run. Chosen from the v2 baseline +# retrieval table alone: the candidate maximising (correct runbooks kept - null cases given a runbook): +# 0.15->5, 0.25->3, 0.30->6, 0.35->7, 0.40->5. Must equal core.services.llm_analyzer.RUNBOOK_FLOOR. +RUNBOOK_FLOOR = 0.35 diff --git a/eval/run_eval.py b/eval/run_eval.py index 29a6549..aa06c73 100644 --- a/eval/run_eval.py +++ b/eval/run_eval.py @@ -104,6 +104,8 @@ def main(argv=None): p.add_argument("--dry-run", action="store_true", help="mock chat(); no API calls") args = p.parse_args(argv) + if llm_analyzer.RUNBOOK_FLOOR != config.RUNBOOK_FLOOR: + raise SystemExit(f"runbook floor {llm_analyzer.RUNBOOK_FLOOR} != frozen {config.RUNBOOK_FLOOR}") # isolated Chroma store holding exactly the frozen corpus, never the dev .chroma vector_store.CHROMA_PATH = tempfile.mkdtemp(prefix="sentinel_eval_chroma_") n_runbooks = vector_store.ingest_runbooks(orchestrator.RUNBOOKS_DIR) @@ -115,6 +117,7 @@ def main(argv=None): "eval_set_version": config.EVAL_SET_VERSION, "runbook_corpus_version": config.RUNBOOK_CORPUS_VERSION, "runbooks_ingested": n_runbooks, + "runbook_floor": llm_analyzer.RUNBOOK_FLOOR, "model": llm.MODEL, "sentinel_git_sha": _git_sha(), "started_at": datetime.now(timezone.utc).isoformat(), From 4dc3f1fe419dc23d90755c09b227e18f95c6bfe6 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Sun, 4 Oct 2026 18:10:29 -0400 Subject: [PATCH 16/19] feat(eval): RAG run on eval set v2 (66/72 top-1, floor 0.35) Co-Authored-By: Claude Opus 5.5 --- eval/results/rag_v2_2026-10-04.json | 3448 +++++++++++++++++++++++++++ 1 file changed, 3448 insertions(+) create mode 100644 eval/results/rag_v2_2026-10-04.json diff --git a/eval/results/rag_v2_2026-10-04.json b/eval/results/rag_v2_2026-10-04.json new file mode 100644 index 0000000..ab0584a --- /dev/null +++ b/eval/results/rag_v2_2026-10-04.json @@ -0,0 +1,3448 @@ +{ + "condition": "rag", + "dry_run": false, + "eval_set_version": "v2", + "runbook_corpus_version": "v1", + "runbooks_ingested": 10, + "runbook_floor": 0.35, + "model": "claude-sonnet-5", + "sentinel_git_sha": "530b73f1470e8620e26ded1fb96271f477c3309e", + "started_at": "2026-10-04T21:58:24.034134+00:00", + "trials": 3, + "n_cases": 24, + "records": [ + { + "trial": 0, + "case_id": "auth_issuer_mesh_hosts", + "category": "auth_rejection", + "culprit_hash": "5e68f8a", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.317 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "null_field_handling", + "similarity_score": 0.236 + } + ], + "ranking": [ + { + "commit_hash": "5e68f8a", + "confidence_score": 0.97 + }, + { + "commit_hash": "45a24d5", + "confidence_score": 0.05 + }, + { + "commit_hash": "df0ec4c", + "confidence_score": 0.05 + }, + { + "commit_hash": "298be8d", + "confidence_score": 0.02 + }, + { + "commit_hash": "d20025a", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 8.34 + }, + { + "trial": 0, + "case_id": "cache_contains_recursion", + "category": "infinite_recursion", + "culprit_hash": "4584056", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [ + { + "commit_hash": "4584056", + "confidence_score": 0.97 + }, + { + "commit_hash": "6484609", + "confidence_score": 0.15 + }, + { + "commit_hash": "7e63ea3", + "confidence_score": 0.05 + }, + { + "commit_hash": "6cc141a", + "confidence_score": 0.02 + }, + { + "commit_hash": "a98de00", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.99 + }, + { + "trial": 0, + "case_id": "cart_price_field_v2", + "category": "null_or_missing_field", + "culprit_hash": "d0ab213", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.26 + }, + { + "id": "pagination_errors", + "similarity_score": 0.207 + }, + { + "id": "memory_leaks", + "similarity_score": 0.179 + } + ], + "ranking": [ + { + "commit_hash": "d0ab213", + "confidence_score": 0.9 + }, + { + "commit_hash": "73e1933", + "confidence_score": 0.6 + }, + { + "commit_hash": "4fbde4c", + "confidence_score": 0.05 + }, + { + "commit_hash": "2bbbf3f", + "confidence_score": 0.02 + }, + { + "commit_hash": "f618f59", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.92 + }, + { + "trial": 0, + "case_id": "cart_sku_lowercased", + "category": "identifier_format", + "culprit_hash": "c2ec824", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [ + { + "commit_hash": "c2ec824", + "confidence_score": 0.75 + }, + { + "commit_hash": "64dde29", + "confidence_score": 0.4 + }, + { + "commit_hash": "d0cb0b3", + "confidence_score": 0.05 + }, + { + "commit_hash": "91f8d36", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.02 + }, + { + "trial": 0, + "case_id": "checkout_deadline_extra_call", + "category": "latency_timeout", + "culprit_hash": "efebc07", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.487 + }, + { + "id": "db_failures", + "similarity_score": 0.369 + }, + { + "id": "rate_limiting", + "similarity_score": 0.353 + } + ], + "ranking": [ + { + "commit_hash": "efebc07", + "confidence_score": 0.85 + }, + { + "commit_hash": "a035477", + "confidence_score": 0.55 + }, + { + "commit_hash": "2b765f4", + "confidence_score": 0.2 + }, + { + "commit_hash": "be85cc9", + "confidence_score": 0.1 + }, + { + "commit_hash": "91b8b07", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.58 + }, + { + "trial": 0, + "case_id": "config_reformat_hidden_string", + "category": "config_type_error", + "culprit_hash": "c3629ef", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [ + { + "commit_hash": "c3629ef", + "confidence_score": 0.9 + }, + { + "commit_hash": "3af442c", + "confidence_score": 0.15 + }, + { + "commit_hash": "976c988", + "confidence_score": 0.05 + }, + { + "commit_hash": "a6a3c41", + "confidence_score": 0.05 + }, + { + "commit_hash": "89436d3", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.34 + }, + { + "trial": 0, + "case_id": "db_fetch_one_leak", + "category": "db_pool_exhaustion", + "culprit_hash": "af9b861", + "expected_runbook_id": "db_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "af9b861", + "confidence_score": 0.97 + }, + { + "commit_hash": "f46cd8c", + "confidence_score": 0.4 + }, + { + "commit_hash": "1618815", + "confidence_score": 0.03 + }, + { + "commit_hash": "6d50d20", + "confidence_score": 0.02 + }, + { + "commit_hash": "1bf1024", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.3 + }, + { + "trial": 0, + "case_id": "export_tempfile_never_closed", + "category": "file_handle_leak", + "culprit_hash": "6eea0fc", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.625 + }, + { + "id": "memory_leaks", + "similarity_score": 0.432 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.291 + } + ], + "ranking": [ + { + "commit_hash": "5cedb21", + "confidence_score": 0.75 + }, + { + "commit_hash": "6eea0fc", + "confidence_score": 0.55 + }, + { + "commit_hash": "b245524", + "confidence_score": 0.2 + }, + { + "commit_hash": "0be651d", + "confidence_score": 0.05 + }, + { + "commit_hash": "7453733", + "confidence_score": 0.02 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 7.09 + }, + { + "trial": 0, + "case_id": "guest_email_normalise", + "category": "null_or_missing_field", + "culprit_hash": "7dd25fe", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.241 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.208 + }, + { + "id": "pagination_errors", + "similarity_score": 0.161 + } + ], + "ranking": [ + { + "commit_hash": "7dd25fe", + "confidence_score": 0.97 + }, + { + "commit_hash": "9519064", + "confidence_score": 0.1 + }, + { + "commit_hash": "1c0135b", + "confidence_score": 0.02 + }, + { + "commit_hash": "cde392a", + "confidence_score": 0.01 + }, + { + "commit_hash": "30a7434", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.95 + }, + { + "trial": 0, + "case_id": "httpx_028_drops_proxies", + "category": "dependency_version", + "culprit_hash": "b35322f", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.294 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.285 + }, + { + "id": "memory_leaks", + "similarity_score": 0.263 + } + ], + "ranking": [ + { + "commit_hash": "741f699", + "confidence_score": 0.9 + }, + { + "commit_hash": "b35322f", + "confidence_score": 0.85 + }, + { + "commit_hash": "b45189f", + "confidence_score": 0.03 + }, + { + "commit_hash": "e3601f6", + "confidence_score": 0.02 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 5.12 + }, + { + "trial": 0, + "case_id": "inventory_cache_key_per_request", + "category": "rate_limit_429", + "culprit_hash": "526f8dd", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.382 + }, + { + "id": "auth_failures", + "similarity_score": 0.381 + }, + { + "id": "null_field_handling", + "similarity_score": 0.309 + } + ], + "ranking": [ + { + "commit_hash": "526f8dd", + "confidence_score": 0.92 + }, + { + "commit_hash": "bf4cc5f", + "confidence_score": 0.4 + }, + { + "commit_hash": "93320f2", + "confidence_score": 0.05 + }, + { + "commit_hash": "2e1b06b", + "confidence_score": 0.02 + }, + { + "commit_hash": "d9caa65", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.38 + }, + { + "trial": 0, + "case_id": "inventory_v3_payload", + "category": "null_or_missing_field", + "culprit_hash": "1c29d94", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.33 + }, + { + "id": "memory_leaks", + "similarity_score": 0.281 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "1c29d94", + "confidence_score": 0.85 + }, + { + "commit_hash": "c6f0797", + "confidence_score": 0.1 + }, + { + "commit_hash": "16d2191", + "confidence_score": 0.1 + }, + { + "commit_hash": "5823d1f", + "confidence_score": 0.05 + }, + { + "commit_hash": "fb9b710", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.45 + }, + { + "trial": 0, + "case_id": "jwt_leeway_milliseconds", + "category": "auth_rejection", + "culprit_hash": "2eeb222", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.488 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.198 + }, + { + "id": "request_timeouts", + "similarity_score": 0.159 + } + ], + "ranking": [ + { + "commit_hash": "2eeb222", + "confidence_score": 0.92 + }, + { + "commit_hash": "e8548c5", + "confidence_score": 0.2 + }, + { + "commit_hash": "2e11931", + "confidence_score": 0.15 + }, + { + "commit_hash": "a028c62", + "confidence_score": 0.03 + }, + { + "commit_hash": "716e22d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.09 + }, + { + "trial": 0, + "case_id": "money_returns_decimal", + "category": "json_serialization", + "culprit_hash": "b14897b", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [ + { + "commit_hash": "b14897b", + "confidence_score": 0.97 + }, + { + "commit_hash": "dedc41a", + "confidence_score": 0.05 + }, + { + "commit_hash": "626c22d", + "confidence_score": 0.02 + }, + { + "commit_hash": "cd4a49f", + "confidence_score": 0.01 + }, + { + "commit_hash": "92cb3da", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.88 + }, + { + "trial": 0, + "case_id": "order_read_sync_audit", + "category": "latency_timeout", + "culprit_hash": "8aed86c", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.539 + }, + { + "id": "rate_limiting", + "similarity_score": 0.369 + }, + { + "id": "auth_failures", + "similarity_score": 0.35 + } + ], + "ranking": [ + { + "commit_hash": "8aed86c", + "confidence_score": 0.92 + }, + { + "commit_hash": "94578cc", + "confidence_score": 0.55 + }, + { + "commit_hash": "d3d6f87", + "confidence_score": 0.3 + }, + { + "commit_hash": "5abe2e4", + "confidence_score": 0.05 + }, + { + "commit_hash": "b7ea08d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.08 + }, + { + "trial": 0, + "case_id": "order_sql_missing_param", + "category": "sql_param_mismatch", + "culprit_hash": "9ef79b2", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [ + { + "commit_hash": "9ef79b2", + "confidence_score": 0.85 + }, + { + "commit_hash": "661f871", + "confidence_score": 0.2 + }, + { + "commit_hash": "93ebe2f", + "confidence_score": 0.05 + }, + { + "commit_hash": "66d5f2f", + "confidence_score": 0.03 + }, + { + "commit_hash": "411ddfa", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.58 + }, + { + "trial": 0, + "case_id": "page_size_zero_division", + "category": "pagination_boundary", + "culprit_hash": "6265f24", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [ + { + "commit_hash": "6265f24", + "confidence_score": 0.9 + }, + { + "commit_hash": "ed3861b", + "confidence_score": 0.85 + }, + { + "commit_hash": "0d241ff", + "confidence_score": 0.1 + }, + { + "commit_hash": "ef48acf", + "confidence_score": 0.05 + }, + { + "commit_hash": "debe268", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.98 + }, + { + "trial": 0, + "case_id": "payments_body_without_content_type", + "category": "http_content_type", + "culprit_hash": "33a9e2a", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [ + { + "commit_hash": "33a9e2a", + "confidence_score": 0.97 + }, + { + "commit_hash": "e60edf2", + "confidence_score": 0.05 + }, + { + "commit_hash": "5430f37", + "confidence_score": 0.02 + }, + { + "commit_hash": "91f3040", + "confidence_score": 0.01 + }, + { + "commit_hash": "64ee5a8", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.28 + }, + { + "trial": 0, + "case_id": "payments_retry_loop", + "category": "rate_limit_429", + "culprit_hash": "f07e927", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.357 + }, + { + "id": "auth_failures", + "similarity_score": 0.352 + }, + { + "id": "request_timeouts", + "similarity_score": 0.323 + } + ], + "ranking": [ + { + "commit_hash": "07a85f8", + "confidence_score": 0.85 + }, + { + "commit_hash": "f07e927", + "confidence_score": 0.75 + }, + { + "commit_hash": "c1fc55f", + "confidence_score": 0.05 + }, + { + "commit_hash": "51bc043", + "confidence_score": 0.02 + }, + { + "commit_hash": "429c4dd", + "confidence_score": 0.0 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 5.5 + }, + { + "trial": 0, + "case_id": "redis_legacy_lrem_args", + "category": "dependency_version", + "culprit_hash": "0751831", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [ + { + "commit_hash": "0751831", + "confidence_score": 0.55 + }, + { + "commit_hash": "3059910", + "confidence_score": 0.3 + }, + { + "commit_hash": "fd92e84", + "confidence_score": 0.2 + }, + { + "commit_hash": "7b9dd2c", + "confidence_score": 0.03 + }, + { + "commit_hash": "957a42e", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.45 + }, + { + "trial": 0, + "case_id": "settings_env_quotes", + "category": "config_type_error", + "culprit_hash": "15aba07", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.36 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.278 + }, + { + "id": "memory_leaks", + "similarity_score": 0.241 + } + ], + "ranking": [ + { + "commit_hash": "15aba07", + "confidence_score": 0.85 + }, + { + "commit_hash": "cae9a6c", + "confidence_score": 0.05 + }, + { + "commit_hash": "3e36aa4", + "confidence_score": 0.05 + }, + { + "commit_hash": "423c51f", + "confidence_score": 0.03 + }, + { + "commit_hash": "55cfa0d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.81 + }, + { + "trial": 0, + "case_id": "since_filter_naive_cutoff", + "category": "timezone_mismatch", + "culprit_hash": "266c09c", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [ + { + "commit_hash": "266c09c", + "confidence_score": 0.85 + }, + { + "commit_hash": "f3ad8b7", + "confidence_score": 0.6 + }, + { + "commit_hash": "7f71cc9", + "confidence_score": 0.05 + }, + { + "commit_hash": "ef6b4a1", + "confidence_score": 0.02 + }, + { + "commit_hash": "5948e6d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.67 + }, + { + "trial": 0, + "case_id": "worker_queue_renamed", + "category": "queue_stall", + "culprit_hash": "f8d8258", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.411 + }, + { + "id": "rate_limiting", + "similarity_score": 0.409 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.383 + } + ], + "ranking": [ + { + "commit_hash": "f8d8258", + "confidence_score": 0.85 + }, + { + "commit_hash": "0f497ec", + "confidence_score": 0.25 + }, + { + "commit_hash": "110eec6", + "confidence_score": 0.1 + }, + { + "commit_hash": "aacbcd7", + "confidence_score": 0.03 + }, + { + "commit_hash": "3434d6b", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.98 + }, + { + "trial": 0, + "case_id": "worker_retains_job_payloads", + "category": "memory_growth", + "culprit_hash": "3037583", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.48 + }, + { + "id": "memory_leaks", + "similarity_score": 0.478 + }, + { + "id": "rate_limiting", + "similarity_score": 0.424 + } + ], + "ranking": [ + { + "commit_hash": "3037583", + "confidence_score": 0.93 + }, + { + "commit_hash": "088e964", + "confidence_score": 0.3 + }, + { + "commit_hash": "947ac52", + "confidence_score": 0.05 + }, + { + "commit_hash": "15f34e0", + "confidence_score": 0.03 + }, + { + "commit_hash": "63c2633", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.31 + }, + { + "trial": 1, + "case_id": "auth_issuer_mesh_hosts", + "category": "auth_rejection", + "culprit_hash": "5e68f8a", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.317 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "null_field_handling", + "similarity_score": 0.236 + } + ], + "ranking": [ + { + "commit_hash": "5e68f8a", + "confidence_score": 0.97 + }, + { + "commit_hash": "45a24d5", + "confidence_score": 0.1 + }, + { + "commit_hash": "df0ec4c", + "confidence_score": 0.03 + }, + { + "commit_hash": "d20025a", + "confidence_score": 0.02 + }, + { + "commit_hash": "298be8d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.52 + }, + { + "trial": 1, + "case_id": "cache_contains_recursion", + "category": "infinite_recursion", + "culprit_hash": "4584056", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [ + { + "commit_hash": "4584056", + "confidence_score": 0.97 + }, + { + "commit_hash": "6484609", + "confidence_score": 0.15 + }, + { + "commit_hash": "7e63ea3", + "confidence_score": 0.03 + }, + { + "commit_hash": "6cc141a", + "confidence_score": 0.02 + }, + { + "commit_hash": "a98de00", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.39 + }, + { + "trial": 1, + "case_id": "cart_price_field_v2", + "category": "null_or_missing_field", + "culprit_hash": "d0ab213", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.26 + }, + { + "id": "pagination_errors", + "similarity_score": 0.207 + }, + { + "id": "memory_leaks", + "similarity_score": 0.179 + } + ], + "ranking": [ + { + "commit_hash": "d0ab213", + "confidence_score": 0.95 + }, + { + "commit_hash": "73e1933", + "confidence_score": 0.55 + }, + { + "commit_hash": "4fbde4c", + "confidence_score": 0.05 + }, + { + "commit_hash": "2bbbf3f", + "confidence_score": 0.02 + }, + { + "commit_hash": "f618f59", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.89 + }, + { + "trial": 1, + "case_id": "cart_sku_lowercased", + "category": "identifier_format", + "culprit_hash": "c2ec824", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [ + { + "commit_hash": "c2ec824", + "confidence_score": 0.75 + }, + { + "commit_hash": "64dde29", + "confidence_score": 0.55 + }, + { + "commit_hash": "d0cb0b3", + "confidence_score": 0.05 + }, + { + "commit_hash": "91f8d36", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.23 + }, + { + "trial": 1, + "case_id": "checkout_deadline_extra_call", + "category": "latency_timeout", + "culprit_hash": "efebc07", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.487 + }, + { + "id": "db_failures", + "similarity_score": 0.369 + }, + { + "id": "rate_limiting", + "similarity_score": 0.353 + } + ], + "ranking": [ + { + "commit_hash": "efebc07", + "confidence_score": 0.85 + }, + { + "commit_hash": "a035477", + "confidence_score": 0.5 + }, + { + "commit_hash": "2b765f4", + "confidence_score": 0.2 + }, + { + "commit_hash": "be85cc9", + "confidence_score": 0.1 + }, + { + "commit_hash": "91b8b07", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.81 + }, + { + "trial": 1, + "case_id": "config_reformat_hidden_string", + "category": "config_type_error", + "culprit_hash": "c3629ef", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [ + { + "commit_hash": "c3629ef", + "confidence_score": 0.85 + }, + { + "commit_hash": "3af442c", + "confidence_score": 0.2 + }, + { + "commit_hash": "976c988", + "confidence_score": 0.05 + }, + { + "commit_hash": "a6a3c41", + "confidence_score": 0.05 + }, + { + "commit_hash": "89436d3", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.03 + }, + { + "trial": 1, + "case_id": "db_fetch_one_leak", + "category": "db_pool_exhaustion", + "culprit_hash": "af9b861", + "expected_runbook_id": "db_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "af9b861", + "confidence_score": 0.95 + }, + { + "commit_hash": "f46cd8c", + "confidence_score": 0.4 + }, + { + "commit_hash": "1618815", + "confidence_score": 0.05 + }, + { + "commit_hash": "6d50d20", + "confidence_score": 0.02 + }, + { + "commit_hash": "1bf1024", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.48 + }, + { + "trial": 1, + "case_id": "export_tempfile_never_closed", + "category": "file_handle_leak", + "culprit_hash": "6eea0fc", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.625 + }, + { + "id": "memory_leaks", + "similarity_score": 0.432 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.291 + } + ], + "ranking": [ + { + "commit_hash": "6eea0fc", + "confidence_score": 0.75 + }, + { + "commit_hash": "5cedb21", + "confidence_score": 0.6 + }, + { + "commit_hash": "b245524", + "confidence_score": 0.15 + }, + { + "commit_hash": "0be651d", + "confidence_score": 0.03 + }, + { + "commit_hash": "7453733", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.66 + }, + { + "trial": 1, + "case_id": "guest_email_normalise", + "category": "null_or_missing_field", + "culprit_hash": "7dd25fe", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.241 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.208 + }, + { + "id": "pagination_errors", + "similarity_score": 0.161 + } + ], + "ranking": [ + { + "commit_hash": "7dd25fe", + "confidence_score": 0.97 + }, + { + "commit_hash": "9519064", + "confidence_score": 0.1 + }, + { + "commit_hash": "1c0135b", + "confidence_score": 0.02 + }, + { + "commit_hash": "30a7434", + "confidence_score": 0.01 + }, + { + "commit_hash": "cde392a", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.08 + }, + { + "trial": 1, + "case_id": "httpx_028_drops_proxies", + "category": "dependency_version", + "culprit_hash": "b35322f", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.294 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.285 + }, + { + "id": "memory_leaks", + "similarity_score": 0.263 + } + ], + "ranking": [ + { + "commit_hash": "741f699", + "confidence_score": 0.95 + }, + { + "commit_hash": "b35322f", + "confidence_score": 0.9 + }, + { + "commit_hash": "b45189f", + "confidence_score": 0.03 + }, + { + "commit_hash": "e3601f6", + "confidence_score": 0.02 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 5.25 + }, + { + "trial": 1, + "case_id": "inventory_cache_key_per_request", + "category": "rate_limit_429", + "culprit_hash": "526f8dd", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.382 + }, + { + "id": "auth_failures", + "similarity_score": 0.381 + }, + { + "id": "null_field_handling", + "similarity_score": 0.309 + } + ], + "ranking": [ + { + "commit_hash": "526f8dd", + "confidence_score": 0.95 + }, + { + "commit_hash": "bf4cc5f", + "confidence_score": 0.55 + }, + { + "commit_hash": "93320f2", + "confidence_score": 0.05 + }, + { + "commit_hash": "2e1b06b", + "confidence_score": 0.02 + }, + { + "commit_hash": "d9caa65", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.95 + }, + { + "trial": 1, + "case_id": "inventory_v3_payload", + "category": "null_or_missing_field", + "culprit_hash": "1c29d94", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.33 + }, + { + "id": "memory_leaks", + "similarity_score": 0.281 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "1c29d94", + "confidence_score": 0.85 + }, + { + "commit_hash": "c6f0797", + "confidence_score": 0.1 + }, + { + "commit_hash": "16d2191", + "confidence_score": 0.1 + }, + { + "commit_hash": "5823d1f", + "confidence_score": 0.05 + }, + { + "commit_hash": "fb9b710", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.89 + }, + { + "trial": 1, + "case_id": "jwt_leeway_milliseconds", + "category": "auth_rejection", + "culprit_hash": "2eeb222", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.488 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.198 + }, + { + "id": "request_timeouts", + "similarity_score": 0.159 + } + ], + "ranking": [ + { + "commit_hash": "2eeb222", + "confidence_score": 0.95 + }, + { + "commit_hash": "e8548c5", + "confidence_score": 0.2 + }, + { + "commit_hash": "2e11931", + "confidence_score": 0.15 + }, + { + "commit_hash": "a028c62", + "confidence_score": 0.05 + }, + { + "commit_hash": "716e22d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.22 + }, + { + "trial": 1, + "case_id": "money_returns_decimal", + "category": "json_serialization", + "culprit_hash": "b14897b", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [ + { + "commit_hash": "b14897b", + "confidence_score": 0.98 + }, + { + "commit_hash": "dedc41a", + "confidence_score": 0.05 + }, + { + "commit_hash": "626c22d", + "confidence_score": 0.02 + }, + { + "commit_hash": "cd4a49f", + "confidence_score": 0.0 + }, + { + "commit_hash": "92cb3da", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.47 + }, + { + "trial": 1, + "case_id": "order_read_sync_audit", + "category": "latency_timeout", + "culprit_hash": "8aed86c", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.539 + }, + { + "id": "rate_limiting", + "similarity_score": 0.369 + }, + { + "id": "auth_failures", + "similarity_score": 0.35 + } + ], + "ranking": [ + { + "commit_hash": "8aed86c", + "confidence_score": 0.92 + }, + { + "commit_hash": "d3d6f87", + "confidence_score": 0.4 + }, + { + "commit_hash": "94578cc", + "confidence_score": 0.3 + }, + { + "commit_hash": "5abe2e4", + "confidence_score": 0.05 + }, + { + "commit_hash": "b7ea08d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.58 + }, + { + "trial": 1, + "case_id": "order_sql_missing_param", + "category": "sql_param_mismatch", + "culprit_hash": "9ef79b2", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [ + { + "commit_hash": "9ef79b2", + "confidence_score": 0.85 + }, + { + "commit_hash": "661f871", + "confidence_score": 0.2 + }, + { + "commit_hash": "93ebe2f", + "confidence_score": 0.05 + }, + { + "commit_hash": "66d5f2f", + "confidence_score": 0.02 + }, + { + "commit_hash": "411ddfa", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.97 + }, + { + "trial": 1, + "case_id": "page_size_zero_division", + "category": "pagination_boundary", + "culprit_hash": "6265f24", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [ + { + "commit_hash": "6265f24", + "confidence_score": 0.85 + }, + { + "commit_hash": "ed3861b", + "confidence_score": 0.8 + }, + { + "commit_hash": "0d241ff", + "confidence_score": 0.05 + }, + { + "commit_hash": "ef48acf", + "confidence_score": 0.02 + }, + { + "commit_hash": "debe268", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.56 + }, + { + "trial": 1, + "case_id": "payments_body_without_content_type", + "category": "http_content_type", + "culprit_hash": "33a9e2a", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [ + { + "commit_hash": "33a9e2a", + "confidence_score": 0.97 + }, + { + "commit_hash": "e60edf2", + "confidence_score": 0.05 + }, + { + "commit_hash": "91f3040", + "confidence_score": 0.03 + }, + { + "commit_hash": "5430f37", + "confidence_score": 0.02 + }, + { + "commit_hash": "64ee5a8", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.09 + }, + { + "trial": 1, + "case_id": "payments_retry_loop", + "category": "rate_limit_429", + "culprit_hash": "f07e927", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.357 + }, + { + "id": "auth_failures", + "similarity_score": 0.352 + }, + { + "id": "request_timeouts", + "similarity_score": 0.323 + } + ], + "ranking": [ + { + "commit_hash": "f07e927", + "confidence_score": 0.85 + }, + { + "commit_hash": "07a85f8", + "confidence_score": 0.8 + }, + { + "commit_hash": "c1fc55f", + "confidence_score": 0.05 + }, + { + "commit_hash": "51bc043", + "confidence_score": 0.02 + }, + { + "commit_hash": "429c4dd", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.64 + }, + { + "trial": 1, + "case_id": "redis_legacy_lrem_args", + "category": "dependency_version", + "culprit_hash": "0751831", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [ + { + "commit_hash": "0751831", + "confidence_score": 0.85 + }, + { + "commit_hash": "3059910", + "confidence_score": 0.35 + }, + { + "commit_hash": "fd92e84", + "confidence_score": 0.15 + }, + { + "commit_hash": "7b9dd2c", + "confidence_score": 0.03 + }, + { + "commit_hash": "957a42e", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.22 + }, + { + "trial": 1, + "case_id": "settings_env_quotes", + "category": "config_type_error", + "culprit_hash": "15aba07", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.36 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.278 + }, + { + "id": "memory_leaks", + "similarity_score": 0.241 + } + ], + "ranking": [ + { + "commit_hash": "15aba07", + "confidence_score": 0.85 + }, + { + "commit_hash": "cae9a6c", + "confidence_score": 0.05 + }, + { + "commit_hash": "3e36aa4", + "confidence_score": 0.05 + }, + { + "commit_hash": "423c51f", + "confidence_score": 0.05 + }, + { + "commit_hash": "55cfa0d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.61 + }, + { + "trial": 1, + "case_id": "since_filter_naive_cutoff", + "category": "timezone_mismatch", + "culprit_hash": "266c09c", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [ + { + "commit_hash": "266c09c", + "confidence_score": 0.85 + }, + { + "commit_hash": "f3ad8b7", + "confidence_score": 0.8 + }, + { + "commit_hash": "7f71cc9", + "confidence_score": 0.2 + }, + { + "commit_hash": "5948e6d", + "confidence_score": 0.05 + }, + { + "commit_hash": "ef6b4a1", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.78 + }, + { + "trial": 1, + "case_id": "worker_queue_renamed", + "category": "queue_stall", + "culprit_hash": "f8d8258", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.411 + }, + { + "id": "rate_limiting", + "similarity_score": 0.409 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.383 + } + ], + "ranking": [ + { + "commit_hash": "f8d8258", + "confidence_score": 0.9 + }, + { + "commit_hash": "0f497ec", + "confidence_score": 0.15 + }, + { + "commit_hash": "110eec6", + "confidence_score": 0.05 + }, + { + "commit_hash": "aacbcd7", + "confidence_score": 0.05 + }, + { + "commit_hash": "3434d6b", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.98 + }, + { + "trial": 1, + "case_id": "worker_retains_job_payloads", + "category": "memory_growth", + "culprit_hash": "3037583", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.48 + }, + { + "id": "memory_leaks", + "similarity_score": 0.478 + }, + { + "id": "rate_limiting", + "similarity_score": 0.424 + } + ], + "ranking": [ + { + "commit_hash": "3037583", + "confidence_score": 0.93 + }, + { + "commit_hash": "088e964", + "confidence_score": 0.25 + }, + { + "commit_hash": "15f34e0", + "confidence_score": 0.03 + }, + { + "commit_hash": "947ac52", + "confidence_score": 0.02 + }, + { + "commit_hash": "63c2633", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.64 + }, + { + "trial": 2, + "case_id": "auth_issuer_mesh_hosts", + "category": "auth_rejection", + "culprit_hash": "5e68f8a", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.317 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.245 + }, + { + "id": "null_field_handling", + "similarity_score": 0.236 + } + ], + "ranking": [ + { + "commit_hash": "5e68f8a", + "confidence_score": 0.97 + }, + { + "commit_hash": "45a24d5", + "confidence_score": 0.05 + }, + { + "commit_hash": "df0ec4c", + "confidence_score": 0.03 + }, + { + "commit_hash": "298be8d", + "confidence_score": 0.02 + }, + { + "commit_hash": "d20025a", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.06 + }, + { + "trial": 2, + "case_id": "cache_contains_recursion", + "category": "infinite_recursion", + "culprit_hash": "4584056", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "memory_leaks", + "similarity_score": 0.333 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.31 + }, + { + "id": "db_failures", + "similarity_score": 0.233 + } + ], + "ranking": [ + { + "commit_hash": "4584056", + "confidence_score": 0.97 + }, + { + "commit_hash": "6484609", + "confidence_score": 0.1 + }, + { + "commit_hash": "7e63ea3", + "confidence_score": 0.02 + }, + { + "commit_hash": "6cc141a", + "confidence_score": 0.01 + }, + { + "commit_hash": "a98de00", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.19 + }, + { + "trial": 2, + "case_id": "cart_price_field_v2", + "category": "null_or_missing_field", + "culprit_hash": "d0ab213", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.26 + }, + { + "id": "pagination_errors", + "similarity_score": 0.207 + }, + { + "id": "memory_leaks", + "similarity_score": 0.179 + } + ], + "ranking": [ + { + "commit_hash": "d0ab213", + "confidence_score": 0.95 + }, + { + "commit_hash": "73e1933", + "confidence_score": 0.3 + }, + { + "commit_hash": "4fbde4c", + "confidence_score": 0.05 + }, + { + "commit_hash": "2bbbf3f", + "confidence_score": 0.02 + }, + { + "commit_hash": "f618f59", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.92 + }, + { + "trial": 2, + "case_id": "cart_sku_lowercased", + "category": "identifier_format", + "culprit_hash": "c2ec824", + "expected_runbook_id": null, + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.293 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.247 + }, + { + "id": "null_field_handling", + "similarity_score": 0.229 + } + ], + "ranking": [ + { + "commit_hash": "c2ec824", + "confidence_score": 0.75 + }, + { + "commit_hash": "64dde29", + "confidence_score": 0.1 + }, + { + "commit_hash": "d0cb0b3", + "confidence_score": 0.0 + }, + { + "commit_hash": "91f8d36", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.44 + }, + { + "trial": 2, + "case_id": "checkout_deadline_extra_call", + "category": "latency_timeout", + "culprit_hash": "efebc07", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.487 + }, + { + "id": "db_failures", + "similarity_score": 0.369 + }, + { + "id": "rate_limiting", + "similarity_score": 0.353 + } + ], + "ranking": [ + { + "commit_hash": "efebc07", + "confidence_score": 0.85 + }, + { + "commit_hash": "a035477", + "confidence_score": 0.55 + }, + { + "commit_hash": "2b765f4", + "confidence_score": 0.2 + }, + { + "commit_hash": "be85cc9", + "confidence_score": 0.1 + }, + { + "commit_hash": "91b8b07", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.14 + }, + { + "trial": 2, + "case_id": "config_reformat_hidden_string", + "category": "config_type_error", + "culprit_hash": "c3629ef", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.246 + }, + { + "id": "pagination_errors", + "similarity_score": 0.19 + }, + { + "id": "memory_leaks", + "similarity_score": 0.184 + } + ], + "ranking": [ + { + "commit_hash": "c3629ef", + "confidence_score": 0.85 + }, + { + "commit_hash": "3af442c", + "confidence_score": 0.3 + }, + { + "commit_hash": "976c988", + "confidence_score": 0.1 + }, + { + "commit_hash": "a6a3c41", + "confidence_score": 0.05 + }, + { + "commit_hash": "89436d3", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.34 + }, + { + "trial": 2, + "case_id": "db_fetch_one_leak", + "category": "db_pool_exhaustion", + "culprit_hash": "af9b861", + "expected_runbook_id": "db_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.631 + }, + { + "id": "memory_leaks", + "similarity_score": 0.35 + }, + { + "id": "request_timeouts", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "af9b861", + "confidence_score": 0.95 + }, + { + "commit_hash": "f46cd8c", + "confidence_score": 0.4 + }, + { + "commit_hash": "1618815", + "confidence_score": 0.05 + }, + { + "commit_hash": "6d50d20", + "confidence_score": 0.02 + }, + { + "commit_hash": "1bf1024", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.78 + }, + { + "trial": 2, + "case_id": "export_tempfile_never_closed", + "category": "file_handle_leak", + "culprit_hash": "6eea0fc", + "expected_runbook_id": "file_handle_exhaustion", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.625 + }, + { + "id": "memory_leaks", + "similarity_score": 0.432 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.291 + } + ], + "ranking": [ + { + "commit_hash": "5cedb21", + "confidence_score": 0.65 + }, + { + "commit_hash": "6eea0fc", + "confidence_score": 0.45 + }, + { + "commit_hash": "b245524", + "confidence_score": 0.15 + }, + { + "commit_hash": "0be651d", + "confidence_score": 0.03 + }, + { + "commit_hash": "7453733", + "confidence_score": 0.01 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 7.01 + }, + { + "trial": 2, + "case_id": "guest_email_normalise", + "category": "null_or_missing_field", + "culprit_hash": "7dd25fe", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.241 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.208 + }, + { + "id": "pagination_errors", + "similarity_score": 0.161 + } + ], + "ranking": [ + { + "commit_hash": "7dd25fe", + "confidence_score": 0.97 + }, + { + "commit_hash": "9519064", + "confidence_score": 0.05 + }, + { + "commit_hash": "cde392a", + "confidence_score": 0.0 + }, + { + "commit_hash": "1c0135b", + "confidence_score": 0.0 + }, + { + "commit_hash": "30a7434", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.2 + }, + { + "trial": 2, + "case_id": "httpx_028_drops_proxies", + "category": "dependency_version", + "culprit_hash": "b35322f", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 4, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.294 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.285 + }, + { + "id": "memory_leaks", + "similarity_score": 0.263 + } + ], + "ranking": [ + { + "commit_hash": "741f699", + "confidence_score": 0.9 + }, + { + "commit_hash": "b35322f", + "confidence_score": 0.85 + }, + { + "commit_hash": "b45189f", + "confidence_score": 0.02 + }, + { + "commit_hash": "e3601f6", + "confidence_score": 0.01 + } + ], + "culprit_rank": 2, + "error": null, + "rank_latency_s": 4.17 + }, + { + "trial": 2, + "case_id": "inventory_cache_key_per_request", + "category": "rate_limit_429", + "culprit_hash": "526f8dd", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.382 + }, + { + "id": "auth_failures", + "similarity_score": 0.381 + }, + { + "id": "null_field_handling", + "similarity_score": 0.309 + } + ], + "ranking": [ + { + "commit_hash": "526f8dd", + "confidence_score": 0.95 + }, + { + "commit_hash": "bf4cc5f", + "confidence_score": 0.35 + }, + { + "commit_hash": "93320f2", + "confidence_score": 0.05 + }, + { + "commit_hash": "2e1b06b", + "confidence_score": 0.02 + }, + { + "commit_hash": "d9caa65", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.48 + }, + { + "trial": 2, + "case_id": "inventory_v3_payload", + "category": "null_or_missing_field", + "culprit_hash": "1c29d94", + "expected_runbook_id": "null_field_handling", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.33 + }, + { + "id": "memory_leaks", + "similarity_score": 0.281 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.28 + } + ], + "ranking": [ + { + "commit_hash": "1c29d94", + "confidence_score": 0.85 + }, + { + "commit_hash": "c6f0797", + "confidence_score": 0.05 + }, + { + "commit_hash": "16d2191", + "confidence_score": 0.05 + }, + { + "commit_hash": "5823d1f", + "confidence_score": 0.02 + }, + { + "commit_hash": "fb9b710", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.05 + }, + { + "trial": 2, + "case_id": "jwt_leeway_milliseconds", + "category": "auth_rejection", + "culprit_hash": "2eeb222", + "expected_runbook_id": "auth_failures", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.488 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.198 + }, + { + "id": "request_timeouts", + "similarity_score": 0.159 + } + ], + "ranking": [ + { + "commit_hash": "2eeb222", + "confidence_score": 0.92 + }, + { + "commit_hash": "e8548c5", + "confidence_score": 0.35 + }, + { + "commit_hash": "2e11931", + "confidence_score": 0.15 + }, + { + "commit_hash": "a028c62", + "confidence_score": 0.03 + }, + { + "commit_hash": "716e22d", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.86 + }, + { + "trial": 2, + "case_id": "money_returns_decimal", + "category": "json_serialization", + "culprit_hash": "b14897b", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.298 + }, + { + "id": "null_field_handling", + "similarity_score": 0.151 + }, + { + "id": "auth_failures", + "similarity_score": 0.13 + } + ], + "ranking": [ + { + "commit_hash": "b14897b", + "confidence_score": 0.98 + }, + { + "commit_hash": "dedc41a", + "confidence_score": 0.05 + }, + { + "commit_hash": "626c22d", + "confidence_score": 0.02 + }, + { + "commit_hash": "cd4a49f", + "confidence_score": 0.01 + }, + { + "commit_hash": "92cb3da", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.34 + }, + { + "trial": 2, + "case_id": "order_read_sync_audit", + "category": "latency_timeout", + "culprit_hash": "8aed86c", + "expected_runbook_id": "request_timeouts", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.539 + }, + { + "id": "rate_limiting", + "similarity_score": 0.369 + }, + { + "id": "auth_failures", + "similarity_score": 0.35 + } + ], + "ranking": [ + { + "commit_hash": "8aed86c", + "confidence_score": 0.9 + }, + { + "commit_hash": "d3d6f87", + "confidence_score": 0.4 + }, + { + "commit_hash": "94578cc", + "confidence_score": 0.2 + }, + { + "commit_hash": "5abe2e4", + "confidence_score": 0.05 + }, + { + "commit_hash": "b7ea08d", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.88 + }, + { + "trial": 2, + "case_id": "order_sql_missing_param", + "category": "sql_param_mismatch", + "culprit_hash": "9ef79b2", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.322 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.23 + }, + { + "id": "memory_leaks", + "similarity_score": 0.196 + } + ], + "ranking": [ + { + "commit_hash": "9ef79b2", + "confidence_score": 0.85 + }, + { + "commit_hash": "661f871", + "confidence_score": 0.2 + }, + { + "commit_hash": "93ebe2f", + "confidence_score": 0.05 + }, + { + "commit_hash": "66d5f2f", + "confidence_score": 0.02 + }, + { + "commit_hash": "411ddfa", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.67 + }, + { + "trial": 2, + "case_id": "page_size_zero_division", + "category": "pagination_boundary", + "culprit_hash": "6265f24", + "expected_runbook_id": "pagination_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "config_parse_errors", + "similarity_score": 0.146 + }, + { + "id": "memory_leaks", + "similarity_score": 0.129 + }, + { + "id": "pagination_errors", + "similarity_score": 0.127 + } + ], + "ranking": [ + { + "commit_hash": "6265f24", + "confidence_score": 0.85 + }, + { + "commit_hash": "ed3861b", + "confidence_score": 0.8 + }, + { + "commit_hash": "0d241ff", + "confidence_score": 0.1 + }, + { + "commit_hash": "ef48acf", + "confidence_score": 0.02 + }, + { + "commit_hash": "debe268", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.31 + }, + { + "trial": 2, + "case_id": "payments_body_without_content_type", + "category": "http_content_type", + "culprit_hash": "33a9e2a", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "auth_failures", + "similarity_score": 0.274 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.164 + }, + { + "id": "null_field_handling", + "similarity_score": 0.143 + } + ], + "ranking": [ + { + "commit_hash": "33a9e2a", + "confidence_score": 0.97 + }, + { + "commit_hash": "e60edf2", + "confidence_score": 0.05 + }, + { + "commit_hash": "91f3040", + "confidence_score": 0.03 + }, + { + "commit_hash": "5430f37", + "confidence_score": 0.02 + }, + { + "commit_hash": "64ee5a8", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 4.98 + }, + { + "trial": 2, + "case_id": "payments_retry_loop", + "category": "rate_limit_429", + "culprit_hash": "f07e927", + "expected_runbook_id": "rate_limiting", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "rate_limiting", + "similarity_score": 0.357 + }, + { + "id": "auth_failures", + "similarity_score": 0.352 + }, + { + "id": "request_timeouts", + "similarity_score": 0.323 + } + ], + "ranking": [ + { + "commit_hash": "f07e927", + "confidence_score": 0.85 + }, + { + "commit_hash": "07a85f8", + "confidence_score": 0.8 + }, + { + "commit_hash": "c1fc55f", + "confidence_score": 0.05 + }, + { + "commit_hash": "51bc043", + "confidence_score": 0.02 + }, + { + "commit_hash": "429c4dd", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.95 + }, + { + "trial": 2, + "case_id": "redis_legacy_lrem_args", + "category": "dependency_version", + "culprit_hash": "0751831", + "expected_runbook_id": "dependency_regressions", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "null_field_handling", + "similarity_score": 0.324 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.317 + }, + { + "id": "auth_failures", + "similarity_score": 0.269 + } + ], + "ranking": [ + { + "commit_hash": "0751831", + "confidence_score": 0.85 + }, + { + "commit_hash": "3059910", + "confidence_score": 0.45 + }, + { + "commit_hash": "fd92e84", + "confidence_score": 0.2 + }, + { + "commit_hash": "7b9dd2c", + "confidence_score": 0.02 + }, + { + "commit_hash": "957a42e", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.98 + }, + { + "trial": 2, + "case_id": "settings_env_quotes", + "category": "config_type_error", + "culprit_hash": "15aba07", + "expected_runbook_id": "config_parse_errors", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "db_failures", + "similarity_score": 0.36 + }, + { + "id": "config_parse_errors", + "similarity_score": 0.278 + }, + { + "id": "memory_leaks", + "similarity_score": 0.241 + } + ], + "ranking": [ + { + "commit_hash": "15aba07", + "confidence_score": 0.85 + }, + { + "commit_hash": "cae9a6c", + "confidence_score": 0.05 + }, + { + "commit_hash": "423c51f", + "confidence_score": 0.05 + }, + { + "commit_hash": "3e36aa4", + "confidence_score": 0.02 + }, + { + "commit_hash": "55cfa0d", + "confidence_score": 0.01 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 7.09 + }, + { + "trial": 2, + "case_id": "since_filter_naive_cutoff", + "category": "timezone_mismatch", + "culprit_hash": "266c09c", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "pagination_errors", + "similarity_score": 0.279 + }, + { + "id": "dependency_regressions", + "similarity_score": 0.224 + }, + { + "id": "memory_leaks", + "similarity_score": 0.215 + } + ], + "ranking": [ + { + "commit_hash": "266c09c", + "confidence_score": 0.85 + }, + { + "commit_hash": "f3ad8b7", + "confidence_score": 0.5 + }, + { + "commit_hash": "7f71cc9", + "confidence_score": 0.1 + }, + { + "commit_hash": "ef6b4a1", + "confidence_score": 0.02 + }, + { + "commit_hash": "5948e6d", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.94 + }, + { + "trial": 2, + "case_id": "worker_queue_renamed", + "category": "queue_stall", + "culprit_hash": "f8d8258", + "expected_runbook_id": null, + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "request_timeouts", + "similarity_score": 0.411 + }, + { + "id": "rate_limiting", + "similarity_score": 0.409 + }, + { + "id": "file_handle_exhaustion", + "similarity_score": 0.383 + } + ], + "ranking": [ + { + "commit_hash": "f8d8258", + "confidence_score": 0.9 + }, + { + "commit_hash": "0f497ec", + "confidence_score": 0.3 + }, + { + "commit_hash": "110eec6", + "confidence_score": 0.1 + }, + { + "commit_hash": "aacbcd7", + "confidence_score": 0.05 + }, + { + "commit_hash": "3434d6b", + "confidence_score": 0.02 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 6.24 + }, + { + "trial": 2, + "case_id": "worker_retains_job_payloads", + "category": "memory_growth", + "culprit_hash": "3037583", + "expected_runbook_id": "memory_leaks", + "commits_in_window": 5, + "truncated_diffs": 0, + "retrieved_runbooks": [ + { + "id": "file_handle_exhaustion", + "similarity_score": 0.48 + }, + { + "id": "memory_leaks", + "similarity_score": 0.478 + }, + { + "id": "rate_limiting", + "similarity_score": 0.424 + } + ], + "ranking": [ + { + "commit_hash": "3037583", + "confidence_score": 0.93 + }, + { + "commit_hash": "088e964", + "confidence_score": 0.25 + }, + { + "commit_hash": "947ac52", + "confidence_score": 0.05 + }, + { + "commit_hash": "15f34e0", + "confidence_score": 0.03 + }, + { + "commit_hash": "63c2633", + "confidence_score": 0.0 + } + ], + "culprit_rank": 1, + "error": null, + "rank_latency_s": 5.31 + } + ], + "finished_at": "2026-10-04T22:09:59.928063+00:00" +} \ No newline at end of file From b7b732969c6e0d614ac8299b87b37e9334753dca Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Tue, 6 Oct 2026 02:05:54 -0400 Subject: [PATCH 17/19] docs: Phase 9 eval results, ADR-7, Phase 5 accuracy caveat RAG in ranking measured with no benefit at this corpus size; SENTINEL_RAG_RANKING stays off. README cites only the baseline accuracy from METRICS.md. Co-Authored-By: Claude Opus 5.5 --- ADR.md | 41 +++++++++++++++++++++++ METRICS.md | 96 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ README.md | 1 + 3 files changed, 138 insertions(+) diff --git a/ADR.md b/ADR.md index 38ae3c7..055a1ce 100644 --- a/ADR.md +++ b/ADR.md @@ -104,3 +104,44 @@ Phase 5 verified 3/3 correct #1 rankings *because* ground truth exists. Trade-off: no exposure to messy production alert noise; the mitigation is that the simulation's log and alert shapes mirror production profiles (mock Sentry payloads, structured JSON logs). + +--- + +## ADR-7: Runbook context in commit ranking — built, measured, left off + +**Context.** Sentinel already retrieved matching runbooks (Chroma, cosine +similarity) but only used them in the postmortem and the Slack card. The open +question was whether passing them into `rank_suspect_commits` would help +Claude find the faulty commit. Two other retrieval targets were considered and +rejected: **past incidents**, because a sandbox has too little history for +retrieval over it to matter, and **diff embeddings**, because "what changed +recently" is already answered exactly by the git time window. + +**Decision.** Runbook text (Symptoms + Root Causes, capped at 1500 chars) can +be passed into the ranking prompt behind `SENTINEL_RAG_RANKING`, **default +off**. Only runbooks scoring at or above `SENTINEL_RUNBOOK_FLOOR` (0.35) are +sent; if none clear it, the block is omitted and the prompt is byte-identical +to the no-runbook prompt (guarded by a golden-string test). The prompt tells +the model runbooks may be irrelevant and that diffs are the evidence. Runbook +text is never persisted. A retrieval failure is logged and ranking proceeds +without runbooks. + +The floor was picked from the baseline's retrieval scores before any RAG run +(maximise correct runbooks kept minus null cases given a runbook). + +**Measured result** (METRICS.md, Phase 9; 24 cases × 3 trials): baseline +90.3% top-1, RAG 91.7%. On the 10 cases where runbook text actually reached the +prompt, both scored 27/30; the single extra RAG hit came from a case whose +prompt was unchanged. **No measurable benefit**, and no measurable harm on +null-runbook cases. + +**Consequences.** The flag stays off, so the shipped ranking path is the +measured baseline. The plumbing and the eval harness stay, because the answer +depends on two things that could change: +- **Retrieval quality.** The correct runbook is top-1 in only 10/17 cases, and + at the floor only 8/17 cases receive it. Better retrieval (richer alert text + than a one-line error, or a stronger embedding model) would raise the + ceiling on what runbooks can contribute. +- **Baseline headroom.** At 90% top-1 there is little left to win. A larger, + harder or real incident corpus — especially real runbooks written by the + team that wrote the code — would be the test that could change this decision. diff --git a/METRICS.md b/METRICS.md index 9b5606b..02bc62f 100644 --- a/METRICS.md +++ b/METRICS.md @@ -44,6 +44,11 @@ Instrumented directly in `core/orchestrator.py` (`[metrics]` log lines): **Cite the medians: ~13s to surface the faulty commit, ~14s to a complete postmortem draft.** +> **Caveat (added in Phase 9):** these runs used chaos-CLI commits whose +> messages named the injected bug, with typically one commit in the window; +> they verify the pipeline end-to-end but are not a measure of diagnostic +> accuracy. See [Phase 9](#phase-9--diagnostic-accuracy-and-runbook-context-in-ranking). + ## Resume Reconciliation - The prior `<10s` suspect-commit claim is **not supported** — measured median @@ -76,3 +81,94 @@ python -m sandbox.chaos_cli trigger-bug --type db_failure python -m sandbox.chaos_cli resolve-bug --incident # wait for "[metrics] ... resolve_to_postmortem_s=" in terminal 1 ``` + +## Phase 9 — Diagnostic accuracy and runbook context in ranking + +**Question:** does giving Claude the matched runbooks improve its ability to +find the commit that caused an incident? + +**Answer at this corpus size: no measurable benefit.** `SENTINEL_RAG_RANKING` +stays off by default. Raw results: +[`eval/results/`](eval/results/); every table below is reproduced by +`python -m eval.report eval/results/baseline_v2_2026-10-04.json eval/results/rag_v2_2026-10-04.json`. + +### Methodology + +- **Harness:** `eval/` builds each case as a throwaway git repo (never the + Sentinel repo) and calls the production `get_recent_diffs`, + `find_matching_runbooks` and `rank_suspect_commits`; it never rebuilds the + prompt itself. +- **Model:** `claude-sonnet-5`, Anthropic API direct, provider-default + sampling (non-deterministic), `max_tokens` 16000. **3 trials per case.** +- **Runbook corpus v1:** 10 generic runbooks in `sandbox/runbooks/`, written + before any eval case and frozen at Phase 9B. +- **Eval set v2:** 24 cases, 17 fault categories, 5 commits per case (2 cases + have 4). Culprit position from newest: 0→5 cases, 1→6, 2→6, 3→4, 4→3. + 7/24 cases (29%) deliberately have no matching runbook. + - Every case carries a **decoy** that matches the alert as well as the + culprit at first glance: same config key, same subsystem, or the very + line that raises. + - Commit messages are neutral; a lint test bans words that announce the + answer. A leakage test asserts that no case id, category or label value + reaches the prompt in either condition. +- **Eval set v1 (superseded):** the first 29-case set scored **87/87 top-1** + at baseline (`eval/results/baseline_2026-10-04.json`). It was too easy to + show any difference, so it was replaced by v2 before any RAG run (the one + revision the protocol allows). +- **Similarity floor 0.35:** chosen from the v2 baseline retrieval table alone, + before the RAG run, as the candidate maximising (correct runbooks kept − + null cases given a runbook): 0.15→5, 0.25→3, 0.30→6, **0.35→7**, 0.40→5. + +### Results (eval set v2, n = 24 cases × 3 trials = 72 per condition) + +| Metric | Baseline | RAG (floor 0.35) | +|---|---|---| +| Top-1 (culprit ranked #1) | 90.3% (65/72) | 91.7% (66/72) | +| Top-3 | 100% (72/72) | 100% (72/72) | +| MRR | 0.947 | 0.958 | +| LLM failures (counted as misses) | 0/72 | 0/72 | +| Mean #1 confidence, right / wrong | 0.90 (n=65) / 0.89 (n=7) | 0.89 (n=66) / 0.83 (n=6) | +| Ranking latency, median | 5.8s | 6.0s | + +**Paired per case (top-1 hits out of 3):** RAG wins 2, loses 1, ties 21. + +| Case | Baseline | RAG | Runbook text in RAG prompt? | +|---|---|---|---| +| `export_tempfile_never_closed` | 0/3 | 1/3 | yes (`file_handle_exhaustion`) | +| `page_size_zero_division` | 2/3 | 3/3 | **no** — prompt byte-identical to baseline | +| `payments_retry_loop` | 3/3 | 2/3 | yes (`rate_limiting`) | +| `httpx_028_drops_proxies` | 0/3 | 0/3 | no (best match 0.29) | + +**Where runbooks actually reached the prompt** (10 of 24 cases had a runbook +above the floor): baseline 27/30, RAG 27/30 — identical. The one extra RAG hit +overall came from a case whose prompt was unchanged, i.e. run-to-run noise. + +**Null-runbook cases** (7 cases, 21 trials): 21/21 in both conditions. One of +them (`worker_queue_renamed`) received three irrelevant runbooks above the +floor and was still ranked correctly 3/3, so wrong context did not visibly +hurt either. + +**How large a difference would matter:** a 1/3 swing on a single case occurred +with a byte-identical prompt, so per-case noise is at least that large. With +only 10 cases where the prompt differs, a real effect would need to show up as +several (roughly 3+) net case wins concentrated in those cases. This is a +judgement from the observed noise, not a significance test; none was computed. + +### Retrieval + +- Runbook top-1 on cases with an expected runbook: **10/17**; correct runbook + anywhere in top-3: 13/17. +- At floor 0.35: 1 of 7 null cases still receives a runbook, and only 8 of 17 + expected cases receive the correct one. +- Indexing only title + Symptoms (11/17) or title + Symptoms + Root Causes + (12/17) was measured offline and did not separate correct matches from null + cases either; retrieval was left unchanged. The limiting factor is short + error strings against the default embedding model. + +### Limitations + +- Synthetic cases written by the same author as the system and the runbooks. +- Small n: 24 cases × 3 trials; single model (`claude-sonnet-5`). +- Sandbox repo; each case's diffs are small and fit the 3000-char truncation. +- The baseline is already 90% top-1, leaving little room for runbook context to + help; retrieval only delivers the right runbook in about half the cases. diff --git a/README.md b/README.md index a9b3a06..6054570 100644 --- a/README.md +++ b/README.md @@ -84,6 +84,7 @@ Every hop reads and writes one auditable incident row conforming to ## Docs - [METRICS.md](METRICS.md) — measured end-to-end timings, methodology, and per-run results +- Measured diagnostic accuracy: faulty commit ranked #1 in 90.3% of 72 trials across 24 injected-fault cases with decoy commits ([METRICS.md, Phase 9](METRICS.md#phase-9--diagnostic-accuracy-and-runbook-context-in-ranking)) - [ADR.md](ADR.md) — why explicit state loops over LangChain, polling over WebSockets, Chroma over Pinecone, FastAPI over Flask, tool-call JSON over free-form parsing, sandbox over live infra ## Sandbox reference From e5cb96901b84694b7f55463329ee77fa9fef72c3 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Tue, 6 Oct 2026 02:19:05 -0400 Subject: [PATCH 18/19] docs(core): note the Phase 9 outcome on the RAG ranking flag Co-Authored-By: Claude Opus 5.5 --- core/orchestrator.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/core/orchestrator.py b/core/orchestrator.py index d694ee4..2d5714c 100644 --- a/core/orchestrator.py +++ b/core/orchestrator.py @@ -13,7 +13,7 @@ _DEDUP_WINDOW_S = float(os.getenv("SENTINEL_DEDUP_WINDOW_S", "300")) _DEDUP_THRESHOLD = int(os.getenv("SENTINEL_DEDUP_THRESHOLD", "1")) # Nth hit in window fires _recent_alerts: dict[str, list[float]] = defaultdict(list) # signature -> hit times -# Pass matched runbook text into commit ranking. Default off until the Phase 9 A/B decides (METRICS.md). +# Pass matched runbook text into commit ranking. Off: Phase 9 A/B found no measurable benefit (ADR-7). _RAG_RANKING = os.getenv("SENTINEL_RAG_RANKING", "0") == "1" _open_signatures: dict[str, str] = {} # signature -> live incident_id From 4b5f31d735938162fde4b5824de3e37e02a0e9d2 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Tue, 6 Oct 2026 02:20:55 -0400 Subject: [PATCH 19/19] docs(eval): note where recorded run SHAs live after rebase-merge Co-Authored-By: Claude Opus 5.5 --- eval/results/README.md | 9 +++++++++ 1 file changed, 9 insertions(+) create mode 100644 eval/results/README.md diff --git a/eval/results/README.md b/eval/results/README.md new file mode 100644 index 0000000..22d69cc --- /dev/null +++ b/eval/results/README.md @@ -0,0 +1,9 @@ +# Eval results + +Raw, unedited outputs of `python -m eval.run_eval`; `python -m eval.report` reproduces the METRICS.md tables. + +Each file records `sentinel_git_sha`, the commit the run used. Phase 9 was rebase-merged into `main`, so +those SHAs (`7b4dc90`, `2e1eb8d`, `530b73f`) live on the preserved branch `feat/phase9-rag-eval`, not on +`main`. Do not delete that branch. + +Runs under `aborted/` are kept for the record and are never scored as measurements.