-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy patheval.py
More file actions
141 lines (124 loc) · 5.18 KB
/
Copy patheval.py
File metadata and controls
141 lines (124 loc) · 5.18 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
"""
Evaluation script: measures how accurate the Gemini triage model is.
Why this matters: in a real system you never just "trust" an LLM's output.
You build a small golden set (tickets where you already know the correct
answer) and periodically check the model against it. This catches:
- Prompt changes that accidentally break accuracy
- Model version upgrades that change behavior
- Silent drift over time
Run it with (backend must be running first):
uv run python eval.py
"""
import time
import requests
API_URL = "http://127.0.0.1:8000/analyze"
# --- Golden test set ---
# Each entry: the ticket text, and the correct (human-labeled) answer.
# Keep this small and clear-cut on purpose — ambiguous cases make a bad eval set.
GOLDEN_SET = [
{
"text": "My order #4521 hasn't arrived in 10 days, I need a refund immediately",
"expected_category": "delivery",
"expected_priority": "urgent",
"expected_sentiment": "frustrated",
},
{
"text": "I can't login to my account, keeps saying invalid password even after reset",
"expected_category": "account",
"expected_priority": "high",
"expected_sentiment": "frustrated",
},
{
"text": "Just wanted to say the new update looks great, love the dark mode!",
"expected_category": "feature-request",
"expected_priority": "low",
"expected_sentiment": "satisfied",
},
{
"text": "Billing charged me twice this month, please fix this",
"expected_category": "billing",
"expected_priority": "high",
"expected_sentiment": "frustrated",
},
{
"text": "Can you add a feature to export my data as CSV?",
"expected_category": "feature-request",
"expected_priority": "low",
"expected_sentiment": "neutral",
},
{
"text": "This is the third time I'm contacting you about my missing package, I want a manager",
"expected_category": "delivery",
"expected_priority": "urgent",
"expected_sentiment": "frustrated",
},
{
"text": "The app crashes every time I try to upload a photo",
"expected_category": "technical",
"expected_priority": "high",
"expected_sentiment": "frustrated",
},
{
"text": "Your customer service replied really fast, thank you!",
"expected_category": "feature-request",
"expected_priority": "low",
"expected_sentiment": "satisfied",
},
]
def run_eval():
total = len(GOLDEN_SET)
category_correct = 0
priority_correct = 0
sentiment_correct = 0
failures = []
for i, case in enumerate(GOLDEN_SET, 1):
print(f"[{i}/{total}] Testing: {case['text'][:50]}...", flush=True)
time.sleep(1) # small pause to stay under free-tier rate limits
try:
response = requests.post(API_URL, json={"text": case["text"]}, timeout=90)
response.raise_for_status()
data = response.json()
except requests.exceptions.HTTPError:
# Show the actual error message from our backend, not just the status code.
try:
detail = response.json().get("detail", response.text)
except Exception:
detail = response.text
print(f" -> FAILED ({response.status_code})", flush=True)
failures.append((case["text"], f"Request failed ({response.status_code}): {detail}"))
continue
except Exception as e:
print(f" -> FAILED: {e}", flush=True)
failures.append((case["text"], f"Request failed: {e}"))
continue
print(" -> done", flush=True)
cat_match = data.get("category") == case["expected_category"]
pri_match = data.get("priority") == case["expected_priority"]
sen_match = data.get("sentiment") == case["expected_sentiment"]
category_correct += cat_match
priority_correct += pri_match
sentiment_correct += sen_match
if not (cat_match and pri_match and sen_match):
failures.append((
case["text"],
f"Expected: category={case['expected_category']}, "
f"priority={case['expected_priority']}, sentiment={case['expected_sentiment']} | "
f"Got: category={data.get('category')}, priority={data.get('priority')}, "
f"sentiment={data.get('sentiment')}",
))
# --- Report ---
print("=" * 60)
print(f"EVAL RESULTS ({total} test cases)")
print("=" * 60)
print(f"Category accuracy: {category_correct}/{total} ({category_correct/total*100:.0f}%)")
print(f"Priority accuracy: {priority_correct}/{total} ({priority_correct/total*100:.0f}%)")
print(f"Sentiment accuracy: {sentiment_correct}/{total} ({sentiment_correct/total*100:.0f}%)")
if failures:
print("\n--- Mismatches ---")
for text, detail in failures:
print(f"\nTicket: {text}")
print(f" {detail}")
else:
print("\nAll test cases matched expected labels.")
if __name__ == "__main__":
run_eval()