-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathcode.py
More file actions
135 lines (110 loc) · 4.87 KB
/
Copy pathcode.py
File metadata and controls
135 lines (110 loc) · 4.87 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
"""Evaluation and Monitoring — measure quality, don't eyeball it.
Two different jobs that get conflated:
evaluation - offline, against a fixed dataset with known answers. Tells you
whether a prompt change helped before you ship it.
monitoring - online, on live traffic with no ground truth. Tells you when
something has started going wrong.
Monitoring can't use accuracy, because production has no answer key. It has to
work from proxies - latency, refusal rate, output length, cost - which is why
they're tracked separately here.
"""
import statistics
import time
from resources.agent import llm
from resources.helper import show_response
# The eval set. Small, but every case has a known-correct answer, which is
# what makes it an eval set rather than a demo.
EVAL_SET = [
("What is the capital of Australia?", "canberra"),
("Who wrote the novel 'Beloved'?", "morrison"),
("What is the chemical symbol for potassium?", "k"),
("In what year did the Berlin Wall fall?", "1989"),
("What is the largest moon of Saturn?", "titan"),
("How many time zones does China officially use?", "one"),
]
# Two prompts to compare. This is the only honest way to know whether prompt
# tinkering helped: run both against the same set and compare numbers.
PROMPTS = {
"baseline": "{q}",
"constrained": "Answer with the single most specific fact, no "
"explanation.\n\n{q}",
}
def grade(answer, expected):
"""Exact-match grading. Deliberately mechanical.
Using an LLM to grade would be more flexible and would also mean the thing
being measured is measuring itself. Keep the ruler independent of the
thing it measures where you possibly can.
"""
return expected.lower() in answer.lower()
def evaluate(name, template):
"""Offline eval: run the whole set, score it, report the metrics."""
correct, latencies, lengths = 0, [], []
for question, expected in EVAL_SET:
start = time.perf_counter()
answer = llm.invoke(template.format(q=question)).content
latencies.append(time.perf_counter() - start)
lengths.append(len(answer.split()))
hit = grade(answer, expected)
correct += hit
print(f" [{'ok ' if hit else 'MISS'}] {question[:44]:<44} "
f"-> {answer.strip()[:32]}")
return {
"accuracy": correct / len(EVAL_SET),
"p50_latency": statistics.median(latencies),
"mean_words": statistics.mean(lengths),
}
def monitor(question):
"""Online monitoring: no ground truth, so track proxies instead.
None of these prove the answer is right. They are early warnings: a spike
in refusals or a collapse in answer length usually means something
upstream changed - a prompt edit, a model swap, a bad deploy.
"""
start = time.perf_counter()
reply = llm.invoke(question)
elapsed = time.perf_counter() - start
text = reply.content
return {
"latency_s": round(elapsed, 2),
"words": len(text.split()),
"tokens": reply.usage_metadata["total_tokens"],
"refused": any(
p in text.lower()
for p in ("i don't have", "i cannot", "i can't", "unable to")
),
}
if __name__ == "__main__":
print(f"{'=' * 60}\nOFFLINE EVALUATION\n{'=' * 60}")
results = {}
for name, template in PROMPTS.items():
print(f"\n prompt: {name}")
results[name] = evaluate(name, template)
print(f"\n{'=' * 60}\nCOMPARISON\n{'=' * 60}")
print(f" {'prompt':<14} {'accuracy':>9} {'p50 latency':>13} {'mean words':>12}")
for name, m in results.items():
print(f" {name:<14} {m['accuracy']:>8.0%} {m['p50_latency']:>12.2f}s "
f"{m['mean_words']:>12.1f}")
# The comparison only supports a decision if the metrics actually differ.
best = max(results, key=lambda n: results[n]["accuracy"])
if results["baseline"]["accuracy"] == results["constrained"]["accuracy"]:
print("\n [verdict] accuracy tied - decide on latency or cost, "
"not on which prompt reads better")
else:
print(f"\n [verdict] '{best}' wins on accuracy")
print(f"\n{'=' * 60}\nONLINE MONITORING (no ground truth)\n{'=' * 60}")
traffic = [
"What is the capital of Australia?",
"What will Bitcoin be worth next March?", # should refuse
"Explain the CAP theorem.",
]
samples = []
for question in traffic:
m = monitor(question)
samples.append(m)
print(f" {question[:40]:<42} latency={m['latency_s']}s "
f"words={m['words']:<4} tokens={m['tokens']:<5} "
f"refused={m['refused']}")
print(f"\n refusal rate: "
f"{sum(s['refused'] for s in samples) / len(samples):.0%}")
print(f" p50 latency: "
f"{statistics.median(s['latency_s'] for s in samples):.2f}s")
print(f" total tokens: {sum(s['tokens'] for s in samples)}")