Repository navigation
Expand file tree
/
Copy pathcode.py
More file actions
114 lines (94 loc) · 4.7 KB
/
Copy pathcode.py
File metadata and controls
114 lines (94 loc) · 4.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
"""Learning and Adaptation — improve from feedback without retraining.
The model weights never change. What changes is the prompt: outcomes get
recorded, turned into rules, and fed back in on the next attempt. This is
in-context learning, and for most production systems it's the only kind of
"learning" you actually need.
The honest limitation: this learns within a process lifetime. Persist the
lesson store to disk (or a DB) and it survives restarts - see the note at the
bottom.
"""
from resources.agent import llm
from resources.helper import show_response
# Support tickets with a known-correct category, so we can score attempts.
# In production this labelled feedback comes from humans correcting the agent.
#
# The labels are internal team names, which carry no semantic hint about what
# belongs to them. That is deliberate. Earlier drafts used labels like "bug"
# and "regression"; the model scored full marks cold because the label names
# gave the rule away, leaving nothing to learn. Real adaptation is about local
# conventions an outsider cannot infer - so the routing table here can only be
# discovered through feedback:
#
# atlas - money and subscriptions
# borealis - things broken that were never right
# cypress - requests for things that don't exist yet
# drake - things that broke after a release
LABELLED = [
("The invoice charged me twice this month", "atlas"),
("App crashes on launch after the 3.2 update", "drake"),
("Can you add dark mode to the settings page?", "cypress"),
("I was billed after cancelling my subscription", "atlas"),
("Search has returned nothing since yesterday's release", "drake"),
("Please make the export button support CSV too", "cypress"),
("Login screen freezes when I tap 'forgot password'", "borealis"),
("The dashboard has been slow to load since we upgraded", "drake"),
("Dark mode text is unreadable on the billing page", "borealis"),
]
CATEGORIES = ["atlas", "borealis", "cypress", "drake"]
def classify(ticket, lessons):
"""Classify one ticket, with whatever has been learned so far injected."""
learned = (
"\n\nLessons from previous mistakes:\n" + "\n".join(f"- {l}" for l in lessons)
if lessons
else ""
)
prompt = (
f"Route this support ticket to exactly one team: {', '.join(CATEGORIES)}.\n"
f"Reply with the team name only.{learned}\n\nTicket: {ticket}"
)
return llm.invoke(prompt).content.strip().lower()
def learn_from(ticket, predicted, correct):
"""Turn one mistake into a reusable rule.
Note it produces a *generalizable* rule, not a lookup entry for this exact
ticket. A lesson that only fires on the input that produced it teaches
nothing about the next one.
"""
reply = llm.invoke(
"A ticket router made a mistake. Write ONE short general routing rule "
"that would prevent this class of error in future. State it as 'Route "
"X to Y'. Do not mention this specific ticket.\n\n"
f"Ticket: {ticket}\nRouted to: {predicted}\nShould have been: {correct}"
)
return reply.content.strip()
def run_pass(label, lessons):
"""One pass over the whole dataset. Returns (score, new_lessons)."""
print(f"\n{'=' * 60}\n{label}\n{'=' * 60}")
correct_count = 0
new_lessons = list(lessons)
for ticket, truth in LABELLED:
predicted = classify(ticket, new_lessons)
hit = truth in predicted
correct_count += hit
print(f" [{'ok ' if hit else 'MISS'}] {truth:<16} <- {predicted:<20} {ticket[:38]}")
# Only mistakes generate lessons. Correct answers need no correction,
# and adding rules for them just dilutes the prompt.
if not hit:
lesson = learn_from(ticket, predicted, truth)
new_lessons.append(lesson)
print(f" learned: {lesson}")
print(f"\n score: {correct_count}/{len(LABELLED)}")
return correct_count, new_lessons
if __name__ == "__main__":
# Pass 1 starts with no lessons at all - this is the baseline.
baseline, lessons = run_pass("PASS 1 (no prior lessons)", [])
# Pass 2 runs the same tickets with everything learned in pass 1 applied.
improved, lessons = run_pass("PASS 2 (lessons applied)", lessons)
print(f"\n{'=' * 60}\nRESULT\n{'=' * 60}")
print(f" pass 1: {baseline}/{len(LABELLED)}")
print(f" pass 2: {improved}/{len(LABELLED)}")
print(f"\n accumulated lessons ({len(lessons)}):")
for lesson in lessons:
print(f" - {lesson}")
# ponytail: lessons live in a list, so they die with the process. Swap for
# a JSON file or a table when you want them to survive a restart - the
# learning logic above doesn't change, only where `lessons` is loaded from.