From 7060a1855ee96b9e7c1a8f8c8c566c95439badae Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Wed, 7 Oct 2026 15:06:57 -0400 Subject: [PATCH 1/3] feat(core): flag close-call rankings on the Slack card Raw confidence barely separates right from wrong #1 picks (AUC 0.60 on 144 eval trials); the gap to #2 does (AUC 0.86). When the top two suspects are within 0.1 the card says so and shows both. Co-Authored-By: Claude Opus 5.5 --- core/services/notifier.py | 17 ++++++++++++++++- core/tests/test_phase3.py | 17 +++++++++++++++++ 2 files changed, 33 insertions(+), 1 deletion(-) diff --git a/core/services/notifier.py b/core/services/notifier.py index f2ffe44..f6e4917 100644 --- a/core/services/notifier.py +++ b/core/services/notifier.py @@ -4,6 +4,16 @@ load_dotenv() SLACK_WEBHOOK_URL = os.environ.get("SLACK_WEBHOOK_URL") +# Raw confidence barely separates right from wrong #1 picks (AUC 0.60 on 144 eval trials); the gap to #2 +# does (AUC 0.86). ponytail: 0.1 was picked on those same trials (13 wrong picks), re-check as data grows. +CLOSE_CALL_MARGIN = 0.1 + + +def is_close_call(commits: list[dict]) -> bool: + """Top two suspects within CLOSE_CALL_MARGIN: the #1 pick is not trustworthy on its own.""" + if len(commits) < 2: + return False + return round(commits[0]["confidence_score"] - commits[1]["confidence_score"], 2) <= CLOSE_CALL_MARGIN def _fmt_commit(c: dict) -> str: @@ -49,7 +59,12 @@ def build_incident_card(incident: dict) -> dict: }, } ) - if commits: + if commits and is_close_call(commits): + text = ":scales: *Close call: check both suspects.*\n" + "\n".join( + f"{i}. " + _fmt_commit(c) for i, c in enumerate(commits[:2], 1) + ) + blocks.append({"type": "section", "text": {"type": "mrkdwn", "text": text}}) + elif commits: blocks.append( { "type": "section", diff --git a/core/tests/test_phase3.py b/core/tests/test_phase3.py index e77ce45..0052ddf 100644 --- a/core/tests/test_phase3.py +++ b/core/tests/test_phase3.py @@ -32,6 +32,23 @@ def test_build_incident_card_includes_error_and_top_suspect(): assert "Database Connection Pool Exhaustion" in text +def _with_suspects(*scores): + commits = [ + {"commit_hash": f"c{i}", "confidence_score": s, "rationale": f"r{i}"} for i, s in enumerate(scores) + ] + return {**_INCIDENT, "diagnostics": {**_INCIDENT["diagnostics"], "suspect_commits": commits}} + + +def test_close_call_card_shows_both_suspects(): + text = str(build_incident_card(_with_suspects(0.9, 0.8, 0.1))) # gap 0.1 (float 0.0999..) + assert "Close call" in text and "c0" in text and "c1" in text and "c2" not in text + + +def test_clear_lead_card_shows_only_top_suspect(): + text = str(build_incident_card(_with_suspects(0.9, 0.7))) + assert "Close call" not in text and "c0" in text and "c1" not in text + + def test_post_incident_to_slack_skips_without_webhook_url(): with patch("core.services.notifier.SLACK_WEBHOOK_URL", None): assert post_incident_to_slack(_INCIDENT) is False From d82ad7bd0209767a6779e77c88c1226045056b19 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Wed, 7 Oct 2026 15:06:58 -0400 Subject: [PATCH 2/3] feat(dashboard): close-call banner on the incident page Co-Authored-By: Claude Opus 5.5 --- dashboard/src/app/incidents/[id]/page.tsx | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/dashboard/src/app/incidents/[id]/page.tsx b/dashboard/src/app/incidents/[id]/page.tsx index 644ebd2..9513367 100644 --- a/dashboard/src/app/incidents/[id]/page.tsx +++ b/dashboard/src/app/incidents/[id]/page.tsx @@ -54,6 +54,11 @@ export default function IncidentDetail() {

Suspect commits

{diagnostics.suspect_commits.length === 0 &&

None found in the alert window.

} + {/* mirrors core/services/notifier.is_close_call: the gap to #2 predicts wrong picks, raw confidence doesn't */} + {diagnostics.suspect_commits.length >= 2 && + Math.round((diagnostics.suspect_commits[0].confidence_score - diagnostics.suspect_commits[1].confidence_score) * 100) <= 10 && ( +

Close call: the top two suspects are within 10 points. Check both.

+ )} {diagnostics.suspect_commits.map((c) => (

From d50108a5b87f9c5f8ccb10a5f06bc06ddd24d5c3 Mon Sep 17 00:00:00 2001 From: Lushenwar Date: Wed, 7 Oct 2026 15:06:58 -0400 Subject: [PATCH 3/3] docs(metrics): confidence signal analysis behind the close-call rule Co-Authored-By: Claude Opus 5.5 --- METRICS.md | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/METRICS.md b/METRICS.md index 02bc62f..26d2b9b 100644 --- a/METRICS.md +++ b/METRICS.md @@ -154,6 +154,22 @@ only 10 cases where the prompt differs, a real effect would need to show up as several (roughly 3+) net case wins concentrated in those cases. This is a judgement from the observed noise, not a significance test; none was computed. +### Confidence signal + +Pooling both v2 conditions (144 trials: 131 right #1 picks, 13 wrong): + +| Signal | AUC right-vs-wrong (0.5 = coin flip) | +|---|---| +| Top-1 `confidence_score` | 0.60 | +| Gap between #1 and #2 confidence | **0.86** | + +Raw confidence is near-useless: wrong picks averaged 0.83–0.89. The Slack card +and dashboard therefore flag a **close call** when the top two suspects are +within 0.1, and show both. On these trials that rule flags 9/13 wrong picks +and 9/131 right ones. The 0.1 cut was chosen on the same 144 trials (in-sample, +13 wrong picks), so treat it as a working rule. Two wrong picks had a 0.7 gap: +the signal does not catch confidently wrong rankings. + ### Retrieval - Runbook top-1 on cases with an expected runbook: **10/17**; correct runbook