Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions METRICS.md
Original file line number Diff line number Diff line change
Expand Up @@ -154,6 +154,22 @@ only 10 cases where the prompt differs, a real effect would need to show up as
several (roughly 3+) net case wins concentrated in those cases. This is a
judgement from the observed noise, not a significance test; none was computed.

### Confidence signal

Pooling both v2 conditions (144 trials: 131 right #1 picks, 13 wrong):

| Signal | AUC right-vs-wrong (0.5 = coin flip) |
|---|---|
| Top-1 `confidence_score` | 0.60 |
| Gap between #1 and #2 confidence | **0.86** |

Raw confidence is near-useless: wrong picks averaged 0.83–0.89. The Slack card
and dashboard therefore flag a **close call** when the top two suspects are
within 0.1, and show both. On these trials that rule flags 9/13 wrong picks
and 9/131 right ones. The 0.1 cut was chosen on the same 144 trials (in-sample,
13 wrong picks), so treat it as a working rule. Two wrong picks had a 0.7 gap:
the signal does not catch confidently wrong rankings.

### Retrieval

- Runbook top-1 on cases with an expected runbook: **10/17**; correct runbook
Expand Down
17 changes: 16 additions & 1 deletion core/services/notifier.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,16 @@

load_dotenv()
SLACK_WEBHOOK_URL = os.environ.get("SLACK_WEBHOOK_URL")
# Raw confidence barely separates right from wrong #1 picks (AUC 0.60 on 144 eval trials); the gap to #2
# does (AUC 0.86). ponytail: 0.1 was picked on those same trials (13 wrong picks), re-check as data grows.
CLOSE_CALL_MARGIN = 0.1


def is_close_call(commits: list[dict]) -> bool:
"""Top two suspects within CLOSE_CALL_MARGIN: the #1 pick is not trustworthy on its own."""
if len(commits) < 2:
return False
return round(commits[0]["confidence_score"] - commits[1]["confidence_score"], 2) <= CLOSE_CALL_MARGIN


def _fmt_commit(c: dict) -> str:
Expand Down Expand Up @@ -49,7 +59,12 @@ def build_incident_card(incident: dict) -> dict:
},
}
)
if commits:
if commits and is_close_call(commits):
text = ":scales: *Close call: check both suspects.*\n" + "\n".join(
f"{i}. " + _fmt_commit(c) for i, c in enumerate(commits[:2], 1)
)
blocks.append({"type": "section", "text": {"type": "mrkdwn", "text": text}})
elif commits:
blocks.append(
{
"type": "section",
Expand Down
17 changes: 17 additions & 0 deletions core/tests/test_phase3.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,23 @@ def test_build_incident_card_includes_error_and_top_suspect():
assert "Database Connection Pool Exhaustion" in text


def _with_suspects(*scores):
commits = [
{"commit_hash": f"c{i}", "confidence_score": s, "rationale": f"r{i}"} for i, s in enumerate(scores)
]
return {**_INCIDENT, "diagnostics": {**_INCIDENT["diagnostics"], "suspect_commits": commits}}


def test_close_call_card_shows_both_suspects():
text = str(build_incident_card(_with_suspects(0.9, 0.8, 0.1))) # gap 0.1 (float 0.0999..)
assert "Close call" in text and "c0" in text and "c1" in text and "c2" not in text


def test_clear_lead_card_shows_only_top_suspect():
text = str(build_incident_card(_with_suspects(0.9, 0.7)))
assert "Close call" not in text and "c0" in text and "c1" not in text


def test_post_incident_to_slack_skips_without_webhook_url():
with patch("core.services.notifier.SLACK_WEBHOOK_URL", None):
assert post_incident_to_slack(_INCIDENT) is False
Expand Down
5 changes: 5 additions & 0 deletions dashboard/src/app/incidents/[id]/page.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,11 @@ export default function IncidentDetail() {
<div className="panel">
<h2>Suspect commits</h2>
{diagnostics.suspect_commits.length === 0 && <p>None found in the alert window.</p>}
{/* mirrors core/services/notifier.is_close_call: the gap to #2 predicts wrong picks, raw confidence doesn't */}
{diagnostics.suspect_commits.length >= 2 &&
Math.round((diagnostics.suspect_commits[0].confidence_score - diagnostics.suspect_commits[1].confidence_score) * 100) <= 10 && (
<p className="error-sig">Close call: the top two suspects are within 10 points. Check both.</p>
)}
{diagnostics.suspect_commits.map((c) => (
<div key={c.commit_hash} style={{ marginBottom: 12 }}>
<p>
Expand Down
Loading