agentdynamics 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentdynamics/__init__.py +12 -0
- agentdynamics/__main__.py +322 -0
- agentdynamics/analysis.py +755 -0
- agentdynamics/autotrace.py +697 -0
- agentdynamics/bootstrap/sitecustomize.py +27 -0
- agentdynamics/collectors/__init__.py +0 -0
- agentdynamics/collectors/aegis_audit.py +72 -0
- agentdynamics/collectors/claude_code.py +257 -0
- agentdynamics/collectors/generic.py +103 -0
- agentdynamics/collectors/inbox.py +95 -0
- agentdynamics/collectors/langfuse.py +98 -0
- agentdynamics/collectors/langsmith.py +258 -0
- agentdynamics/collectors/otlp.py +349 -0
- agentdynamics/collectors/spans.py +202 -0
- agentdynamics/config.py +139 -0
- agentdynamics/engine.py +376 -0
- agentdynamics/flows.py +142 -0
- agentdynamics/govern.py +231 -0
- agentdynamics/integrations/__init__.py +1 -0
- agentdynamics/integrations/aegis.py +352 -0
- agentdynamics/phases.py +82 -0
- agentdynamics/pricing.py +69 -0
- agentdynamics/privacy.py +41 -0
- agentdynamics/sdk.py +105 -0
- agentdynamics/server.py +949 -0
- agentdynamics/slo.py +84 -0
- agentdynamics/store.py +228 -0
- agentdynamics/web/app.js +1069 -0
- agentdynamics/web/charts.js +185 -0
- agentdynamics/web/index.html +40 -0
- agentdynamics/web/style.css +244 -0
- agentdynamics-0.4.0.dist-info/METADATA +195 -0
- agentdynamics-0.4.0.dist-info/RECORD +37 -0
- agentdynamics-0.4.0.dist-info/WHEEL +5 -0
- agentdynamics-0.4.0.dist-info/entry_points.txt +2 -0
- agentdynamics-0.4.0.dist-info/licenses/LICENSE +202 -0
- agentdynamics-0.4.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,755 @@
|
|
|
1
|
+
"""Turn normalized runs into tasks, baselines, scores, waste findings and events.
|
|
2
|
+
|
|
3
|
+
AppDynamics -> AgentDynamics mapping implemented here:
|
|
4
|
+
Business Transaction -> Task (one user request, end to end)
|
|
5
|
+
BT type / entry point -> Task type (bugfix, feature, question ...)
|
|
6
|
+
Dynamic baselines -> per-task-type median/p90 of cost, duration, tokens
|
|
7
|
+
Apdex -> Agent Apdex from outcome + cost vs baseline
|
|
8
|
+
Health rules / events -> configurable rules evaluated per task
|
|
9
|
+
Code-level hotspots -> waste findings (redundant reads, loops, error streaks)
|
|
10
|
+
"""
|
|
11
|
+
import math
|
|
12
|
+
import re
|
|
13
|
+
import statistics
|
|
14
|
+
import time
|
|
15
|
+
from collections import Counter, defaultdict
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
# ---------------------------------------------------------------- task typing
|
|
19
|
+
|
|
20
|
+
TYPE_RULES = [
|
|
21
|
+
("review/audit", r"\b(security|audit|review|vulnerab|performance analysis|pen ?test|fuzz)"),
|
|
22
|
+
("bugfix", r"\b(fix|bug|error|broken|not working|doesn'?t work|isn'?t|aren'?t|crash|fail|messed up|wrong|issue|stuck|missing|forbidden|negative value|no data|still (getting|not|seeing)|4\d\d\b|5\d\d\b)"),
|
|
23
|
+
("git/deploy", r"\b(github|commit|push|pull request|\bpr\b|deploy|release|repo\b|merge|publish)"),
|
|
24
|
+
("testing", r"\b(tests?|unit test|coverage|e2e)\b"),
|
|
25
|
+
("refactor", r"\b(refactor|clean ?up|simplif|restructure|rename|reorganiz)"),
|
|
26
|
+
("question/research", r"(\?\s*$|^\s*(what|why|how|is|are|can|does|do|should|would|which|explain|any other|give me (more )?(ideas|options|the values))\b)"),
|
|
27
|
+
("feature/build", r"\b(create|build|add|implement|make|integrate|generate|develop|design|write|extend|enhance|improve)"),
|
|
28
|
+
("change request", r"^\s*(change|switch|use|update|replace|remove|move|star|convert|set|modify|have)\b"),
|
|
29
|
+
("question/research", r"\b(explain|research|compare|ideas?|options|think of|thoughts|opinion|check if|go through)\b"),
|
|
30
|
+
("setup/run", r"\b(install|set ?up|configure|run|start|launch|open)\b"),
|
|
31
|
+
]
|
|
32
|
+
FOLLOWUP = re.compile(
|
|
33
|
+
r"^\s*(continue|go ahead|yes|yep|ok(ay)?|proceed|sure|do it|try( it)? (again|now)|next|both|skip)\b|"
|
|
34
|
+
r"\b(continue|proceed|go ahead)\b|\b(done|is up|is uo|installed|completed)\s*[.!]?\s*$", re.I)
|
|
35
|
+
CORRECTION = re.compile(
|
|
36
|
+
r"^\s*(no\b|nope|wrong|that'?s not|still\b|doesn'?t|didn'?t|not working|it'?s broken|revert|undo|why did you|"
|
|
37
|
+
r"you (forgot|missed|broke|didn'?t)|i don'?t like|dont like)|still (not|doesn'?t|broken|failing|getting)|(completely )?messed up|not what I (asked|wanted)",
|
|
38
|
+
re.I,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def classify_task(prompt_kind, text):
|
|
43
|
+
if prompt_kind == "scheduled":
|
|
44
|
+
return "scheduled job"
|
|
45
|
+
if prompt_kind == "delegated":
|
|
46
|
+
return "subagent"
|
|
47
|
+
t = (text or "").strip()
|
|
48
|
+
low = t.lower()
|
|
49
|
+
if prompt_kind == "command":
|
|
50
|
+
return "slash command"
|
|
51
|
+
if len(t) < 60 and FOLLOWUP.search(t):
|
|
52
|
+
return "follow-up"
|
|
53
|
+
# The earliest intent keyword wins: "Create X, then test it" is a build, not a testing task.
|
|
54
|
+
best = None
|
|
55
|
+
for rank, (name, pat) in enumerate(TYPE_RULES):
|
|
56
|
+
m = re.search(pat, low[:600])
|
|
57
|
+
if m and (best is None or (m.start(), rank) < best[0]):
|
|
58
|
+
best = ((m.start(), rank), name)
|
|
59
|
+
return best[1] if best else "other"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
# ---------------------------------------------------------------- helpers
|
|
63
|
+
|
|
64
|
+
def pct(values, p):
|
|
65
|
+
if not values:
|
|
66
|
+
return None
|
|
67
|
+
v = sorted(values)
|
|
68
|
+
k = (len(v) - 1) * p
|
|
69
|
+
f, c = math.floor(k), math.ceil(k)
|
|
70
|
+
return v[f] if f == c else v[f] + (v[c] - v[f]) * (k - f)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def clamp(x, lo=0.0, hi=100.0):
|
|
74
|
+
return max(lo, min(hi, x))
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
IDLE_CAP = 300 # gaps longer than this (user away, waiting on approval) don't count as agent time
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _active_seconds(tss):
|
|
81
|
+
v = sorted(x for x in tss if x)
|
|
82
|
+
return round(sum(min(b - a, IDLE_CAP) for a, b in zip(v, v[1:])), 1)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
# ---------------------------------------------------------------- segmentation
|
|
86
|
+
|
|
87
|
+
def segment(run):
|
|
88
|
+
"""Split a run's steps into tasks at each prompt step."""
|
|
89
|
+
tasks, cur = [], None
|
|
90
|
+
for s in run["steps"]:
|
|
91
|
+
if s["kind"] == "prompt":
|
|
92
|
+
cur = {"prompt_step": s, "steps": [s]}
|
|
93
|
+
tasks.append(cur)
|
|
94
|
+
else:
|
|
95
|
+
if cur is None:
|
|
96
|
+
cur = {"prompt_step": None, "steps": []}
|
|
97
|
+
tasks.append(cur)
|
|
98
|
+
cur["steps"].append(s)
|
|
99
|
+
return tasks
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
# ---------------------------------------------------------------- per-task metrics
|
|
103
|
+
|
|
104
|
+
def task_metrics(run, idx, seg):
|
|
105
|
+
steps = seg["steps"]
|
|
106
|
+
p = seg["prompt_step"]
|
|
107
|
+
llm = [s for s in steps if s["kind"] == "llm" and not s.get("denied")] # a denied model call was never made
|
|
108
|
+
tools = [s for s in steps if s["kind"] == "tool"]
|
|
109
|
+
notices = [s for s in steps if s["kind"] == "notice"]
|
|
110
|
+
tss = [s["ts"] for s in steps if s.get("ts")] + [s["end_ts"] for s in steps if s.get("end_ts")]
|
|
111
|
+
started = min(tss) if tss else run.get("started")
|
|
112
|
+
ended = max(tss) if tss else started
|
|
113
|
+
|
|
114
|
+
t = {
|
|
115
|
+
"id": f"{run['id']}#{idx}",
|
|
116
|
+
"run_id": run["id"],
|
|
117
|
+
"idx": idx,
|
|
118
|
+
"project": run["project"],
|
|
119
|
+
"source": run["source"],
|
|
120
|
+
"is_subagent": 1 if run.get("is_subagent") else 0,
|
|
121
|
+
"prompt_kind": p["name"] if p else "none",
|
|
122
|
+
"prompt": (p or {}).get("text", "")[:2000],
|
|
123
|
+
"started": started,
|
|
124
|
+
"ended": ended,
|
|
125
|
+
"wall_s": round((ended - started), 1) if started and ended else 0,
|
|
126
|
+
"duration_s": _active_seconds(tss),
|
|
127
|
+
"llm_calls": len(llm),
|
|
128
|
+
"tool_calls": len(tools),
|
|
129
|
+
"tool_errors": sum(1 for s in tools if s.get("is_error") and not s.get("denied")),
|
|
130
|
+
"input_tokens": sum(s.get("input_tokens", 0) for s in llm),
|
|
131
|
+
"output_tokens": sum(s.get("output_tokens", 0) for s in llm),
|
|
132
|
+
"cache_read": sum(s.get("cache_read", 0) for s in llm),
|
|
133
|
+
"cache_write": sum(s.get("cache_write", 0) for s in llm),
|
|
134
|
+
"thinking_tokens": sum(s.get("thinking_tokens", 0) for s in llm),
|
|
135
|
+
"cost": sum(s.get("cost", 0) for s in llm),
|
|
136
|
+
"max_context": max([s.get("context_tokens", 0) for s in llm] or [0]),
|
|
137
|
+
"models": ",".join(sorted({s["model"] for s in llm if s.get("model") and s["model"] != "<synthetic>"})),
|
|
138
|
+
"interrupts": sum(1 for s in notices if s["name"] == "interrupt")
|
|
139
|
+
+ sum(1 for s in tools if s.get("interrupted") or s.get("rejected")),
|
|
140
|
+
"compactions": sum(1 for s in notices if s["name"] == "compaction"),
|
|
141
|
+
"api_errors": sum(1 for s in notices if s["name"] == "api_error"),
|
|
142
|
+
"subagent_cost": 0.0,
|
|
143
|
+
"subagents": 0,
|
|
144
|
+
}
|
|
145
|
+
t["total_tokens"] = t["input_tokens"] + t["output_tokens"] + t["cache_read"] + t["cache_write"]
|
|
146
|
+
in_side = t["input_tokens"] + t["cache_read"] + t["cache_write"]
|
|
147
|
+
t["cache_hit"] = round(t["cache_read"] / in_side, 3) if in_side else None
|
|
148
|
+
t["tool_error_rate"] = round(t["tool_errors"] / len(tools), 3) if tools else 0
|
|
149
|
+
t["task_type"] = classify_task(t["prompt_kind"], t["prompt"])
|
|
150
|
+
if run.get("workflow") and run.get("source") != "claude-code":
|
|
151
|
+
t["task_type"] = run["workflow"] # traced apps: the entry point names the business transaction
|
|
152
|
+
|
|
153
|
+
# --- phase attribution: split each LLM call's cost across the tool calls it issued
|
|
154
|
+
phase_calls = Counter(s["phase"] for s in tools)
|
|
155
|
+
phase_cost = Counter()
|
|
156
|
+
# map llm step -> tool steps that immediately follow it (same llm_msg)
|
|
157
|
+
for i, s in enumerate(steps):
|
|
158
|
+
if s["kind"] != "llm":
|
|
159
|
+
continue
|
|
160
|
+
issued = []
|
|
161
|
+
for s2 in steps[i + 1:]:
|
|
162
|
+
if s2["kind"] == "tool":
|
|
163
|
+
issued.append(s2)
|
|
164
|
+
elif s2["kind"] == "llm":
|
|
165
|
+
break
|
|
166
|
+
else:
|
|
167
|
+
continue
|
|
168
|
+
# only count tool steps sharing the same message id
|
|
169
|
+
if issued:
|
|
170
|
+
mid = issued[0].get("llm_msg")
|
|
171
|
+
issued = [x for x in issued if x.get("llm_msg") == mid]
|
|
172
|
+
c = s.get("cost", 0)
|
|
173
|
+
if issued:
|
|
174
|
+
share = c / len(issued)
|
|
175
|
+
for x in issued:
|
|
176
|
+
x["attributed_cost"] = share
|
|
177
|
+
phase_cost[x["phase"]] += share
|
|
178
|
+
else:
|
|
179
|
+
phase_cost["respond"] += c
|
|
180
|
+
t["phase_calls"] = dict(phase_calls)
|
|
181
|
+
t["phase_cost"] = {k: round(v, 5) for k, v in phase_cost.items()}
|
|
182
|
+
with_tools = [s for s in llm if s.get("tool_calls")]
|
|
183
|
+
t["parallelism"] = round(sum(s["tool_calls"] for s in with_tools) / len(with_tools), 2) if with_tools else 0
|
|
184
|
+
|
|
185
|
+
# --- process analysis & waste detection
|
|
186
|
+
read_seen = {} # input_hash -> seq
|
|
187
|
+
call_seen = {}
|
|
188
|
+
edited_since = set()
|
|
189
|
+
edits_per_file = Counter()
|
|
190
|
+
files_read = set()
|
|
191
|
+
redundant = duplicates = 0
|
|
192
|
+
streak = max_streak = 0
|
|
193
|
+
large = 0
|
|
194
|
+
first_edit_idx = None
|
|
195
|
+
last_edit_i = None
|
|
196
|
+
last_verify_i = None
|
|
197
|
+
waste_cost = 0.0
|
|
198
|
+
for i, s in enumerate(tools):
|
|
199
|
+
s.setdefault("flags", [])
|
|
200
|
+
ph = s["phase"]
|
|
201
|
+
tgt = s.get("target", "")
|
|
202
|
+
if s["name"] in ("Read", "NotebookRead"):
|
|
203
|
+
files_read.add(tgt)
|
|
204
|
+
h = s["input_hash"]
|
|
205
|
+
if h in read_seen and tgt not in edited_since:
|
|
206
|
+
s["flags"].append("redundant_read")
|
|
207
|
+
redundant += 1
|
|
208
|
+
read_seen[h] = i
|
|
209
|
+
edited_since.discard(tgt)
|
|
210
|
+
elif ph == "edit":
|
|
211
|
+
edits_per_file[tgt] += 1
|
|
212
|
+
edited_since.add(tgt)
|
|
213
|
+
# an edit invalidates earlier identical calls
|
|
214
|
+
call_seen.clear()
|
|
215
|
+
if first_edit_idx is None:
|
|
216
|
+
first_edit_idx = i
|
|
217
|
+
last_edit_i = i
|
|
218
|
+
if ph == "verify":
|
|
219
|
+
last_verify_i = i
|
|
220
|
+
if ph not in ("plan", "communicate", "edit") and s["name"] not in ("Read", "NotebookRead"):
|
|
221
|
+
h = s["input_hash"]
|
|
222
|
+
if h in call_seen and not s.get("is_error"):
|
|
223
|
+
s["flags"].append("duplicate_call")
|
|
224
|
+
duplicates += 1
|
|
225
|
+
call_seen[h] = i
|
|
226
|
+
if s.get("is_error"):
|
|
227
|
+
streak += 1
|
|
228
|
+
max_streak = max(max_streak, streak)
|
|
229
|
+
if streak >= 3:
|
|
230
|
+
s["flags"].append("error_streak")
|
|
231
|
+
else:
|
|
232
|
+
streak = 0
|
|
233
|
+
if s.get("output_chars", 0) > 40000:
|
|
234
|
+
s["flags"].append("large_output")
|
|
235
|
+
large += 1
|
|
236
|
+
if s.get("denied"):
|
|
237
|
+
s["flags"].append("denied") # tokens spent generating a call the policy refused
|
|
238
|
+
if any(f in s["flags"] for f in ("redundant_read", "duplicate_call", "error_streak", "denied")):
|
|
239
|
+
waste_cost += s.get("attributed_cost", 0)
|
|
240
|
+
t["files_read"] = len(files_read)
|
|
241
|
+
t["files_edited"] = len(edits_per_file)
|
|
242
|
+
t["edits"] = sum(edits_per_file.values())
|
|
243
|
+
t["max_edits_one_file"] = max(edits_per_file.values() or [0])
|
|
244
|
+
t["churn_file"] = edits_per_file.most_common(1)[0][0] if edits_per_file else None
|
|
245
|
+
t["redundant_reads"] = redundant
|
|
246
|
+
t["duplicate_calls"] = duplicates
|
|
247
|
+
t["max_error_streak"] = max_streak
|
|
248
|
+
t["large_outputs"] = large
|
|
249
|
+
t["explore_ratio"] = round(phase_calls.get("explore", 0) / len(tools), 3) if tools else 0
|
|
250
|
+
t["steps_to_first_edit"] = first_edit_idx
|
|
251
|
+
code_task = t["edits"] > 0 and any(
|
|
252
|
+
re.search(r"\.(py|js|ts|tsx|jsx|go|rs|java|cs|rb|php|c|cpp|h|kt|swift|vue|svelte|sql)$", f or "", re.I)
|
|
253
|
+
for f in edits_per_file
|
|
254
|
+
)
|
|
255
|
+
t["code_changed"] = 1 if code_task else 0
|
|
256
|
+
t["verified"] = None
|
|
257
|
+
if code_task:
|
|
258
|
+
t["verified"] = 1 if (last_verify_i is not None and last_verify_i > last_edit_i) else 0
|
|
259
|
+
t["unverified_edits"] = 1 if t["verified"] == 0 else 0
|
|
260
|
+
t["waste_cost"] = round(waste_cost, 5)
|
|
261
|
+
|
|
262
|
+
# --- final response & outcome signals (outcome finalized later with next prompt)
|
|
263
|
+
last_llm = llm[-1] if llm else None
|
|
264
|
+
t["final_stop"] = last_llm.get("stop_reason") if last_llm else None
|
|
265
|
+
t["final_text"] = (last_llm or {}).get("text", "")[:600]
|
|
266
|
+
last_tool = tools[-1] if tools else None
|
|
267
|
+
t["ended_on_error"] = 1 if (last_tool and last_tool.get("is_error") and (not last_llm or last_llm["seq"] < last_tool["seq"])) else 0
|
|
268
|
+
for s in steps:
|
|
269
|
+
s["task_id"] = t["id"]
|
|
270
|
+
flow_metrics(t, run, steps, llm, tools)
|
|
271
|
+
return t
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def governance_metrics(t, run, steps, tools):
|
|
275
|
+
"""Policy enforcement seen from the task's side (Aegis decisions recorded as steps)."""
|
|
276
|
+
denied = [s for s in steps if s.get("denied")]
|
|
277
|
+
tool_denied = [s for s in tools if s.get("denied")]
|
|
278
|
+
t["governed"] = 1 if (run.get("policy_version") or any(s.get("governed") or s.get("rule") for s in steps)) else 0
|
|
279
|
+
t["policy_version"] = run.get("policy_version")
|
|
280
|
+
t["policy_denials"] = len(tool_denied)
|
|
281
|
+
t["spend_denials"] = sum(1 for s in denied if s["kind"] == "llm")
|
|
282
|
+
t["budget_denials"] = sum(1 for s in denied if str(s.get("rule") or "").startswith("budget."))
|
|
283
|
+
t["revocations"] = sum(1 for s in steps if s["kind"] == "notice" and s.get("name") == "revoked")
|
|
284
|
+
t["blocked_cost"] = round(sum(s.get("attributed_cost") or 0 for s in tool_denied), 6)
|
|
285
|
+
t["denied_rules"] = dict(Counter(s.get("rule") or "unknown" for s in denied))
|
|
286
|
+
streak = best = 0
|
|
287
|
+
last = None
|
|
288
|
+
for s in tools:
|
|
289
|
+
key = s.get("name") if s.get("denied") else None # same tool refused in a row, whatever the rule
|
|
290
|
+
streak = streak + 1 if key is not None and key == last else (1 if key else 0)
|
|
291
|
+
last = key
|
|
292
|
+
best = max(best, streak)
|
|
293
|
+
t["repeated_denials"] = best
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
TRUNCATION = {"max_tokens", "length", "max_output_tokens", "MAX_TOKENS"}
|
|
297
|
+
REFUSAL = {"refusal", "content_filter", "SAFETY", "safety", "blocked"}
|
|
298
|
+
RATE_HINTS = ("429", "rate limit", "rate_limit", "overloaded", "529", "too many requests", "quota")
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def flow_metrics(t, run, steps, llm, tools):
|
|
302
|
+
"""Agent-flow metrics that apply to any framework (graph nodes, handoffs, LLM health, retrieval)."""
|
|
303
|
+
spans = [s for s in steps if s["kind"] == "span"]
|
|
304
|
+
notices = [s for s in steps if s["kind"] == "notice"]
|
|
305
|
+
t["environment"] = run.get("environment") or "default"
|
|
306
|
+
t["framework"] = run.get("framework") or run.get("source")
|
|
307
|
+
t["workflow"] = run.get("workflow") or (t["task_type"] if run.get("source") == "claude-code" else None)
|
|
308
|
+
t["steps_total"] = len(llm) + len(tools)
|
|
309
|
+
t["llm_errors"] = sum(1 for s in llm if s.get("is_error"))
|
|
310
|
+
t["truncations"] = sum(1 for s in llm if s.get("stop_reason") in TRUNCATION)
|
|
311
|
+
t["refusals"] = sum(1 for s in llm if s.get("stop_reason") in REFUSAL)
|
|
312
|
+
t["rate_limited"] = sum(1 for s in llm if s.get("rate_limited")) + sum(
|
|
313
|
+
1 for s in notices if s["name"] == "api_error" and any(h in (s.get("text") or "").lower() for h in RATE_HINTS))
|
|
314
|
+
tt = [s["ttft_ms"] for s in llm if s.get("ttft_ms")]
|
|
315
|
+
t["ttft_ms"] = round(statistics.median(tt)) if tt else None
|
|
316
|
+
tps = [s["output_tokens"] / (s["duration_ms"] / 1000) for s in llm if s.get("duration_ms") and s.get("output_tokens", 0) > 20]
|
|
317
|
+
t["out_tps"] = round(statistics.median(tps), 1) if tps else None
|
|
318
|
+
rets = [s for s in tools if s.get("phase") == "retrieve"]
|
|
319
|
+
t["retrievals"] = len(rets)
|
|
320
|
+
t["empty_retrievals"] = sum(1 for s in rets if s.get("docs") == 0)
|
|
321
|
+
t["unpriced"] = sum(1 for s in llm if s.get("priced") is False and (s.get("input_tokens") or s.get("output_tokens")))
|
|
322
|
+
t["hitl"] = sum(1 for s in spans if s.get("hitl"))
|
|
323
|
+
|
|
324
|
+
# node executions: LangGraph nodes (a span named after its node), explicit node spans, or agent spans
|
|
325
|
+
node_execs = [s for s in spans if s.get("node") and (s.get("name") == s.get("node") or s.get("span_kind") == "node")]
|
|
326
|
+
if not node_execs:
|
|
327
|
+
node_execs = [s for s in spans if s.get("span_kind") == "agent"]
|
|
328
|
+
for s in node_execs:
|
|
329
|
+
if not s.get("node"):
|
|
330
|
+
s["node"] = s.get("agent") or s.get("name")
|
|
331
|
+
if node_execs:
|
|
332
|
+
seq = [s.get("node") or s.get("name") for s in node_execs]
|
|
333
|
+
else:
|
|
334
|
+
# no graph: use the process phases of tool calls (explore -> edit -> verify ...)
|
|
335
|
+
seq = [s["phase"] for s in tools]
|
|
336
|
+
collapsed = [x for i, x in enumerate(seq) if i == 0 or x != seq[i - 1]]
|
|
337
|
+
t["path"] = collapsed[:60]
|
|
338
|
+
visits = Counter(s.get("node") or s.get("name") for s in node_execs)
|
|
339
|
+
t["nodes"] = len(visits)
|
|
340
|
+
t["max_node_visits"] = max(visits.values()) if visits else 0
|
|
341
|
+
t["loop_node"] = visits.most_common(1)[0][0] if visits and t["max_node_visits"] > 1 else None
|
|
342
|
+
# critical node: the node whose executions account for most of the task's elapsed time
|
|
343
|
+
wall = max(t["wall_s"], 0.001)
|
|
344
|
+
dur_by_node = Counter()
|
|
345
|
+
for s in node_execs:
|
|
346
|
+
dur_by_node[s.get("node") or s.get("name")] += (s.get("duration_ms") or 0) / 1000
|
|
347
|
+
if dur_by_node:
|
|
348
|
+
n, d = dur_by_node.most_common(1)[0]
|
|
349
|
+
t["critical_node"], t["critical_share"] = n, round(min(1.0, d / wall), 3)
|
|
350
|
+
else:
|
|
351
|
+
t["critical_node"], t["critical_share"] = None, None
|
|
352
|
+
# multi-agent handoffs
|
|
353
|
+
agents = [s.get("agent") for s in steps if s["kind"] in ("llm", "tool") and s.get("agent")]
|
|
354
|
+
aseq = [a for i, a in enumerate(agents) if i == 0 or a != agents[i - 1]]
|
|
355
|
+
t["handoffs"] = max(0, len(aseq) - 1)
|
|
356
|
+
t["pingpong"] = sum(1 for i in range(2, len(aseq)) if aseq[i] == aseq[i - 2])
|
|
357
|
+
governance_metrics(t, run, steps, tools)
|
|
358
|
+
fb = [f["score"] for f in run.get("feedback") or [] if isinstance(f.get("score"), (int, float))]
|
|
359
|
+
t["feedback_score"] = round(statistics.mean(fb), 3) if fb and t["idx"] <= 1 else None
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
# ---------------------------------------------------------------- scoring
|
|
363
|
+
|
|
364
|
+
def score_task(t, base):
|
|
365
|
+
s = {}
|
|
366
|
+
b = base.get(t["task_type"]) or base.get("__all__") or {}
|
|
367
|
+
med_cost = b.get("cost_p50") or 0
|
|
368
|
+
ratio = (t["cost"] + t["subagent_cost"]) / med_cost if med_cost else 1
|
|
369
|
+
t["cost_vs_baseline"] = round(ratio, 2)
|
|
370
|
+
med_dur = b.get("duration_p50") or 0
|
|
371
|
+
t["duration_vs_baseline"] = round(t["duration_s"] / med_dur, 2) if med_dur else 1
|
|
372
|
+
waste_share = t["waste_cost"] / t["cost"] if t["cost"] else 0
|
|
373
|
+
s["efficiency"] = clamp(100 - 35 * math.log2(max(ratio, 1)) - 100 * waste_share)
|
|
374
|
+
focus = 100 - 6 * t["redundant_reads"] - 10 * t["duplicate_calls"] - 4 * max(0, t["max_edits_one_file"] - 4)
|
|
375
|
+
if t["edits"] and t["explore_ratio"] > 0.75:
|
|
376
|
+
focus -= 15
|
|
377
|
+
focus -= 8 * max(0, (t.get("max_node_visits") or 0) - 3) + 10 * (t.get("pingpong") or 0)
|
|
378
|
+
s["focus"] = clamp(focus) if (t["tool_calls"] or t.get("nodes")) else None
|
|
379
|
+
llm_pen = 8 * min(t.get("llm_errors") or 0, 5) + 6 * min(t.get("truncations") or 0, 5) + 10 * min(t.get("refusals") or 0, 3)
|
|
380
|
+
llm_pen += 40 if t.get("outcome") == "failed" else 0
|
|
381
|
+
if t["tool_calls"]:
|
|
382
|
+
s["reliability"] = clamp(100 * (1 - t["tool_error_rate"]) - 12 * max(0, t["max_error_streak"] - 1) - 15 * t["api_errors"] - llm_pen)
|
|
383
|
+
else:
|
|
384
|
+
s["reliability"] = clamp(100 - 15 * t["api_errors"] - llm_pen)
|
|
385
|
+
s["verification"] = None if t["verified"] is None else (100.0 if t["verified"] else 25.0)
|
|
386
|
+
if t["cache_hit"] is not None and t["llm_calls"] >= 3:
|
|
387
|
+
ctx = 100 * min(1, t["cache_hit"] / 0.9)
|
|
388
|
+
if t["max_context"] > 200_000:
|
|
389
|
+
ctx -= 20
|
|
390
|
+
ctx -= 15 * t["compactions"]
|
|
391
|
+
s["context"] = clamp(ctx)
|
|
392
|
+
else:
|
|
393
|
+
s["context"] = None
|
|
394
|
+
s["autonomy"] = clamp(100 - 45 * min(t["interrupts"], 2) - (35 if t.get("outcome") == "rework" else 0))
|
|
395
|
+
if t.get("governed"):
|
|
396
|
+
s["compliance"] = clamp(100 - 12 * (t.get("policy_denials") or 0) - 20 * max(0, (t.get("repeated_denials") or 0) - 1)
|
|
397
|
+
- 40 * (t.get("revocations") or 0) - 15 * (t.get("budget_denials") or 0))
|
|
398
|
+
else:
|
|
399
|
+
s["compliance"] = None
|
|
400
|
+
weights = {"efficiency": 0.25, "focus": 0.15, "reliability": 0.2, "verification": 0.15, "context": 0.1, "autonomy": 0.15,
|
|
401
|
+
"compliance": 0.15}
|
|
402
|
+
tot = sum(weights[k] for k, v in s.items() if v is not None)
|
|
403
|
+
s["overall"] = round(sum(weights[k] * v for k, v in s.items() if v is not None) / tot, 1) if tot else None
|
|
404
|
+
t["scores"] = {k: (round(v, 1) if v is not None else None) for k, v in s.items()}
|
|
405
|
+
t["score"] = t["scores"]["overall"]
|
|
406
|
+
|
|
407
|
+
# Agent Apdex: T = 1.5x median cost of this task type
|
|
408
|
+
T = 1.5 * med_cost if med_cost else None
|
|
409
|
+
total = t["cost"] + t["subagent_cost"]
|
|
410
|
+
if t["outcome"] in ("interrupted", "rework", "failed") or t["max_error_streak"] >= 4:
|
|
411
|
+
t["apdex"] = "frustrated"
|
|
412
|
+
elif T is None or total <= T:
|
|
413
|
+
t["apdex"] = "satisfied"
|
|
414
|
+
elif total <= 4 * T:
|
|
415
|
+
t["apdex"] = "tolerating"
|
|
416
|
+
else:
|
|
417
|
+
t["apdex"] = "frustrated"
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def apdex_score(tasks):
|
|
421
|
+
if not tasks:
|
|
422
|
+
return None
|
|
423
|
+
c = Counter(t["apdex"] for t in tasks)
|
|
424
|
+
return round((c["satisfied"] + c["tolerating"] / 2) / len(tasks), 3)
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
# ---------------------------------------------------------------- health rules
|
|
428
|
+
|
|
429
|
+
DEFAULT_RULES = [
|
|
430
|
+
{"id": "cost_spike", "name": "Cost above baseline", "metric": "cost_vs_baseline", "op": ">", "value": 3, "severity": "warning",
|
|
431
|
+
"message": "Cost {v}x the median for '{task_type}' tasks"},
|
|
432
|
+
{"id": "cost_critical", "name": "Runaway cost", "metric": "cost_vs_baseline", "op": ">", "value": 8, "severity": "critical",
|
|
433
|
+
"message": "Cost {v}x the median for '{task_type}' tasks"},
|
|
434
|
+
{"id": "error_rate", "name": "High tool error rate", "metric": "tool_error_rate", "op": ">", "value": 0.25, "severity": "warning",
|
|
435
|
+
"guard": {"tool_calls": 4}, "message": "{v:.0%} of tool calls failed"},
|
|
436
|
+
{"id": "error_streak", "name": "Error retry loop", "metric": "max_error_streak", "op": ">=", "value": 3, "severity": "critical",
|
|
437
|
+
"message": "{v} consecutive failing tool calls"},
|
|
438
|
+
{"id": "loop", "name": "Repeated identical calls", "metric": "duplicate_calls", "op": ">=", "value": 3, "severity": "warning",
|
|
439
|
+
"message": "{v} identical tool calls repeated without an intervening edit"},
|
|
440
|
+
{"id": "redundant_reads", "name": "Redundant file reads", "metric": "redundant_reads", "op": ">=", "value": 3, "severity": "info",
|
|
441
|
+
"message": "{v} files re-read with no change in between"},
|
|
442
|
+
{"id": "unverified", "name": "Code changed but not verified", "metric": "unverified_edits", "op": "==", "value": 1, "severity": "warning",
|
|
443
|
+
"message": "Edited code but never ran tests/build/app after the last edit"},
|
|
444
|
+
{"id": "context", "name": "Context window pressure", "metric": "max_context", "op": ">", "value": 250000, "severity": "warning",
|
|
445
|
+
"message": "Context reached {v:,} tokens"},
|
|
446
|
+
{"id": "compaction", "name": "Context compacted", "metric": "compactions", "op": ">=", "value": 1, "severity": "info",
|
|
447
|
+
"message": "Conversation was compacted {v} time(s); earlier detail was lost"},
|
|
448
|
+
{"id": "interrupted", "name": "User interrupted agent", "metric": "interrupts", "op": ">=", "value": 1, "severity": "critical",
|
|
449
|
+
"message": "User stopped or rejected the agent {v} time(s)"},
|
|
450
|
+
{"id": "rework", "name": "User had to correct result", "metric": "rework", "op": "==", "value": 1, "severity": "warning",
|
|
451
|
+
"message": "Next message was a correction: \"{next_prompt}\""},
|
|
452
|
+
{"id": "low_cache", "name": "Poor prompt-cache reuse", "metric": "cache_hit", "op": "<", "value": 0.5, "severity": "info",
|
|
453
|
+
"guard": {"llm_calls": 5}, "message": "Only {v:.0%} of input tokens came from cache"},
|
|
454
|
+
{"id": "api_error", "name": "Model API errors", "metric": "api_errors", "op": ">=", "value": 1, "severity": "warning",
|
|
455
|
+
"message": "{v} API error(s) (overload, rate limit, timeout)"},
|
|
456
|
+
{"id": "slow", "name": "Slow task", "metric": "duration_vs_baseline", "op": ">", "value": 5, "severity": "info",
|
|
457
|
+
"guard": {"duration_s": 120}, "message": "Took {v}x the usual time for '{task_type}' tasks"},
|
|
458
|
+
# --- agent-flow rules (graphs, multi-agent, LLM health)
|
|
459
|
+
{"id": "run_failed", "name": "Run failed", "metric": "failed", "op": "==", "value": 1, "severity": "critical",
|
|
460
|
+
"message": "Run ended in error: {root_error}"},
|
|
461
|
+
{"id": "graph_loop", "name": "Node loop", "metric": "max_node_visits", "op": ">=", "value": 5, "severity": "warning",
|
|
462
|
+
"message": "Node '{loop_node}' ran {v} times in one run (possible loop / recursion)"},
|
|
463
|
+
{"id": "pingpong", "name": "Agent handoff ping-pong", "metric": "pingpong", "op": ">=", "value": 2, "severity": "warning",
|
|
464
|
+
"message": "Agents handed work back and forth {v} times"},
|
|
465
|
+
{"id": "truncation", "name": "Output truncated", "metric": "truncations", "op": ">=", "value": 1, "severity": "warning",
|
|
466
|
+
"message": "{v} model response(s) hit the max-token limit"},
|
|
467
|
+
{"id": "refusal", "name": "Model refusal", "metric": "refusals", "op": ">=", "value": 1, "severity": "warning",
|
|
468
|
+
"message": "{v} model response(s) refused or filtered"},
|
|
469
|
+
{"id": "rate_limit", "name": "Rate limited / overloaded", "metric": "rate_limited", "op": ">=", "value": 1, "severity": "warning",
|
|
470
|
+
"message": "{v} model call(s) rate-limited or overloaded"},
|
|
471
|
+
{"id": "llm_errors", "name": "Model call errors", "metric": "llm_errors", "op": ">=", "value": 2, "severity": "warning",
|
|
472
|
+
"message": "{v} model calls failed"},
|
|
473
|
+
{"id": "empty_retrieval", "name": "Retrieval returned nothing", "metric": "empty_retrievals", "op": ">=", "value": 1,
|
|
474
|
+
"severity": "info", "message": "{v} retrieval(s) returned no documents"},
|
|
475
|
+
{"id": "negative_feedback", "name": "Negative user feedback", "metric": "feedback_score", "op": "<", "value": 0.5,
|
|
476
|
+
"severity": "warning", "message": "Feedback score {v}"},
|
|
477
|
+
# --- governance rules (Aegis)
|
|
478
|
+
{"id": "policy_denials", "name": "Actions blocked by policy", "metric": "policy_denials", "op": ">=", "value": 3,
|
|
479
|
+
"severity": "warning", "message": "{v} tool calls were refused by the policy"},
|
|
480
|
+
{"id": "repeated_denials", "name": "Agent probing a boundary", "metric": "repeated_denials", "op": ">=", "value": 3,
|
|
481
|
+
"severity": "critical", "message": "Same forbidden call attempted {v} times in a row (prompt injection or stuck agent)"},
|
|
482
|
+
{"id": "revoked", "name": "Grant revoked", "metric": "revocations", "op": ">=", "value": 1, "severity": "critical",
|
|
483
|
+
"message": "The agent's authority was revoked mid-run"},
|
|
484
|
+
{"id": "budget_stop", "name": "Budget stop", "metric": "budget_denials", "op": ">=", "value": 1, "severity": "warning",
|
|
485
|
+
"message": "Budget limit reached {v} time(s); work was stopped"},
|
|
486
|
+
{"id": "slow_ttft", "name": "Slow first token", "metric": "ttft_ms", "op": ">", "value": 8000, "severity": "info",
|
|
487
|
+
"message": "Median time to first token {v:,.0f} ms"},
|
|
488
|
+
]
|
|
489
|
+
|
|
490
|
+
OPS = {">": lambda a, b: a > b, ">=": lambda a, b: a >= b, "<": lambda a, b: a < b, "<=": lambda a, b: a <= b,
|
|
491
|
+
"==": lambda a, b: a == b}
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def evaluate_rules(t, rules):
|
|
495
|
+
events = []
|
|
496
|
+
for r in rules:
|
|
497
|
+
if not r.get("enabled", True):
|
|
498
|
+
continue
|
|
499
|
+
v = t.get(r["metric"])
|
|
500
|
+
if v is None:
|
|
501
|
+
continue
|
|
502
|
+
guard = r.get("guard") or {}
|
|
503
|
+
if any((t.get(k) or 0) < gv for k, gv in guard.items()):
|
|
504
|
+
continue
|
|
505
|
+
if OPS[r["op"]](v, r["value"]):
|
|
506
|
+
ctx = dict(t)
|
|
507
|
+
ctx["v"] = v
|
|
508
|
+
ctx["next_prompt"] = (t.get("next_prompt") or "")[:80]
|
|
509
|
+
try:
|
|
510
|
+
msg = r["message"].format(**ctx)
|
|
511
|
+
except (KeyError, ValueError, IndexError):
|
|
512
|
+
msg = f"{r['metric']}={v}"
|
|
513
|
+
events.append({
|
|
514
|
+
"id": f"{t['id']}:{r['id']}", "ts": t["ended"], "rule_id": r["id"], "rule": r["name"],
|
|
515
|
+
"severity": r["severity"], "task_id": t["id"], "run_id": t["run_id"], "project": t["project"],
|
|
516
|
+
"task_type": t["task_type"], "message": msg, "value": v,
|
|
517
|
+
})
|
|
518
|
+
return events
|
|
519
|
+
|
|
520
|
+
|
|
521
|
+
# ---------------------------------------------------------------- orchestration
|
|
522
|
+
|
|
523
|
+
def run_tasks(run):
|
|
524
|
+
"""Per-run analysis (cacheable): segment into tasks and compute task metrics."""
|
|
525
|
+
return [task_metrics(run, i, seg) for i, seg in enumerate(segment(run))]
|
|
526
|
+
|
|
527
|
+
|
|
528
|
+
def _outcomes_claude(ts, now):
|
|
529
|
+
for i, t in enumerate(ts):
|
|
530
|
+
# "continue" / "yes" carries on the previous request, so it belongs to that task type
|
|
531
|
+
if t["task_type"] == "follow-up" and i > 0 and ts[i - 1]["task_type"] not in ("follow-up",):
|
|
532
|
+
t["task_type"] = ts[i - 1]["task_type"]
|
|
533
|
+
if t["source"] == "claude-code":
|
|
534
|
+
t["workflow"] = t["task_type"]
|
|
535
|
+
nxt = next((x for x in ts[i + 1:] if x["prompt_kind"] in ("human", "command")), None)
|
|
536
|
+
t["next_prompt"] = nxt["prompt"][:300] if nxt else None
|
|
537
|
+
t["rework"] = 1 if nxt and CORRECTION.search(nxt["prompt"][:200]) else 0
|
|
538
|
+
if t["interrupts"]:
|
|
539
|
+
t["outcome"] = "interrupted"
|
|
540
|
+
elif t["rework"]:
|
|
541
|
+
t["outcome"] = "rework"
|
|
542
|
+
elif t["ended_on_error"] or (t["api_errors"] and t["final_stop"] != "end_turn"):
|
|
543
|
+
t["outcome"] = "failed"
|
|
544
|
+
elif nxt is None and t["source"] == "claude-code" and t["ended"] and now - t["ended"] < 600:
|
|
545
|
+
t["outcome"] = "in progress"
|
|
546
|
+
elif t["final_stop"] in ("end_turn", "stop_sequence") or t["llm_calls"]:
|
|
547
|
+
t["outcome"] = "completed"
|
|
548
|
+
else:
|
|
549
|
+
t["outcome"] = "unknown"
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
def _outcome_trace(t, run, now):
|
|
553
|
+
t.setdefault("next_prompt", None)
|
|
554
|
+
t.setdefault("rework", 0)
|
|
555
|
+
if run.get("root_status") == "error" or t["ended_on_error"]:
|
|
556
|
+
t["outcome"] = "failed"
|
|
557
|
+
elif t["hitl"] and not run.get("complete"):
|
|
558
|
+
t["outcome"] = "interrupted"
|
|
559
|
+
elif t["rework"] or (t["feedback_score"] is not None and t["feedback_score"] < 0.5):
|
|
560
|
+
t["outcome"] = "rework"
|
|
561
|
+
elif run.get("complete") is False:
|
|
562
|
+
t["outcome"] = "in progress" if t["ended"] and now - t["ended"] < 600 else "unknown"
|
|
563
|
+
else:
|
|
564
|
+
t["outcome"] = "completed"
|
|
565
|
+
t["root_error"] = run.get("root_error")
|
|
566
|
+
|
|
567
|
+
|
|
568
|
+
def finalize(runs, tasks_by_run, rules=None, now=None):
|
|
569
|
+
"""Cross-run analysis: outcomes, subagent roll-up, baselines, scores, events."""
|
|
570
|
+
rules = rules or DEFAULT_RULES
|
|
571
|
+
now = now or time.time()
|
|
572
|
+
all_tasks = [t for r in runs for t in tasks_by_run[r["id"]]]
|
|
573
|
+
for t in all_tasks:
|
|
574
|
+
t["subagent_cost"], t["subagents"], t["parent_task_id"] = 0.0, 0, None
|
|
575
|
+
|
|
576
|
+
threads = defaultdict(list)
|
|
577
|
+
for run in runs:
|
|
578
|
+
ts = tasks_by_run[run["id"]]
|
|
579
|
+
if run["source"] == "claude-code" or not run.get("workflow"):
|
|
580
|
+
_outcomes_claude(ts, now)
|
|
581
|
+
else:
|
|
582
|
+
if run.get("thread_id"):
|
|
583
|
+
threads[(run["source"], run["thread_id"])].extend(ts)
|
|
584
|
+
# traced conversations: the next trace in the same thread plays the role of "your next message"
|
|
585
|
+
for group in threads.values():
|
|
586
|
+
group.sort(key=lambda x: x["started"] or 0)
|
|
587
|
+
for a, b in zip(group, group[1:]):
|
|
588
|
+
a["next_prompt"] = b["prompt"][:300]
|
|
589
|
+
a["rework"] = 1 if CORRECTION.search(b["prompt"][:200]) else 0
|
|
590
|
+
for run in runs:
|
|
591
|
+
if not (run["source"] == "claude-code" or not run.get("workflow")):
|
|
592
|
+
for t in tasks_by_run[run["id"]]:
|
|
593
|
+
_outcome_trace(t, run, now)
|
|
594
|
+
|
|
595
|
+
# subagent roll-up: link child runs to the parent task that spawned them
|
|
596
|
+
task_by_id = {t["id"]: t for t in all_tasks}
|
|
597
|
+
parent_tasks = defaultdict(list)
|
|
598
|
+
for t in all_tasks:
|
|
599
|
+
if not t["is_subagent"]:
|
|
600
|
+
parent_tasks[t["run_id"]].append(t)
|
|
601
|
+
spawn_map = {}
|
|
602
|
+
for run in runs:
|
|
603
|
+
if run.get("is_subagent"):
|
|
604
|
+
continue
|
|
605
|
+
for s in run["steps"]:
|
|
606
|
+
if s["kind"] == "tool" and s.get("subagent_id"):
|
|
607
|
+
spawn_map[s["subagent_id"]] = s.get("task_id")
|
|
608
|
+
for run in runs:
|
|
609
|
+
if not run.get("is_subagent"):
|
|
610
|
+
continue
|
|
611
|
+
agent_id = run["id"].split(":")[-1].replace("agent-", "")
|
|
612
|
+
parent_task_id = spawn_map.get(agent_id)
|
|
613
|
+
if not parent_task_id:
|
|
614
|
+
cands = [t for t in parent_tasks.get(run["parent_id"], []) if t["started"] and run["started"] and t["started"] <= run["started"]]
|
|
615
|
+
parent_task_id = cands[-1]["id"] if cands else None
|
|
616
|
+
run["parent_task_id"] = parent_task_id
|
|
617
|
+
for t in tasks_by_run[run["id"]]:
|
|
618
|
+
t["parent_task_id"] = parent_task_id
|
|
619
|
+
pt = task_by_id.get(parent_task_id)
|
|
620
|
+
if pt:
|
|
621
|
+
pt["subagent_cost"] += t["cost"]
|
|
622
|
+
pt["subagents"] += 1
|
|
623
|
+
|
|
624
|
+
# baselines per task type (top-level tasks only)
|
|
625
|
+
main = [t for t in all_tasks if not t["is_subagent"] and t["llm_calls"] > 0]
|
|
626
|
+
groups = defaultdict(list)
|
|
627
|
+
for t in main:
|
|
628
|
+
groups[t["task_type"]].append(t)
|
|
629
|
+
groups["__all__"] = main
|
|
630
|
+
sub = [t for t in all_tasks if t["is_subagent"] and t["llm_calls"] > 0]
|
|
631
|
+
if sub:
|
|
632
|
+
groups["subagent"] = sub
|
|
633
|
+
baselines = {}
|
|
634
|
+
for k, g in groups.items():
|
|
635
|
+
if len(g) < 3 and k != "__all__":
|
|
636
|
+
continue
|
|
637
|
+
costs = [t["cost"] + t["subagent_cost"] for t in g]
|
|
638
|
+
durs = [t["duration_s"] for t in g]
|
|
639
|
+
baselines[k] = {
|
|
640
|
+
"n": len(g),
|
|
641
|
+
"cost_p50": pct(costs, 0.5), "cost_p90": pct(costs, 0.9),
|
|
642
|
+
"duration_p50": pct(durs, 0.5), "duration_p90": pct(durs, 0.9),
|
|
643
|
+
"tokens_p50": pct([t["total_tokens"] for t in g], 0.5),
|
|
644
|
+
"tool_calls_p50": pct([t["tool_calls"] for t in g], 0.5),
|
|
645
|
+
"steps_p50": pct([t["steps_total"] for t in g], 0.5),
|
|
646
|
+
}
|
|
647
|
+
|
|
648
|
+
events = []
|
|
649
|
+
for t in all_tasks:
|
|
650
|
+
t["failed"] = 1 if t.get("outcome") == "failed" else 0
|
|
651
|
+
if t["llm_calls"] == 0 and t["tool_calls"] == 0:
|
|
652
|
+
t.update({"scores": {}, "score": None, "apdex": None, "cost_vs_baseline": None, "duration_vs_baseline": None})
|
|
653
|
+
continue
|
|
654
|
+
score_task(t, baselines)
|
|
655
|
+
events.extend(evaluate_rules(t, rules))
|
|
656
|
+
return all_tasks, baselines, events
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
def analyze(runs, rules=None, now=None):
|
|
660
|
+
"""One-shot analysis of a list of runs (used by tests and the CLI)."""
|
|
661
|
+
return finalize(runs, {r["id"]: run_tasks(r) for r in runs}, rules, now)
|
|
662
|
+
|
|
663
|
+
|
|
664
|
+
# ---------------------------------------------------------------- process review (coaching)
|
|
665
|
+
|
|
666
|
+
def process_insights(tasks):
|
|
667
|
+
"""Aggregate findings meant to help a human assess how the agent works."""
|
|
668
|
+
main = [t for t in tasks if not t["is_subagent"] and t["llm_calls"] > 0]
|
|
669
|
+
if not main:
|
|
670
|
+
return []
|
|
671
|
+
out = []
|
|
672
|
+
code = [t for t in main if t["code_changed"]]
|
|
673
|
+
if code:
|
|
674
|
+
unv = [t for t in code if not t["verified"]]
|
|
675
|
+
out.append({
|
|
676
|
+
"id": "verification", "title": "Verification after code changes",
|
|
677
|
+
"metric": f"{100 * (1 - len(unv) / len(code)):.0f}% verified",
|
|
678
|
+
"detail": f"{len(unv)} of {len(code)} code-changing tasks ended without running tests, a build, or the app after the last edit.",
|
|
679
|
+
"severity": "warning" if len(unv) / len(code) > 0.3 else "ok",
|
|
680
|
+
"advice": "Add to CLAUDE.md: 'After editing code, always run the relevant tests or build before reporting done.'",
|
|
681
|
+
"examples": [t["id"] for t in sorted(unv, key=lambda x: -x["cost"])[:5]],
|
|
682
|
+
})
|
|
683
|
+
total_cost = sum(t["cost"] for t in main) or 1
|
|
684
|
+
waste = sum(t["waste_cost"] for t in main)
|
|
685
|
+
out.append({
|
|
686
|
+
"id": "waste", "title": "Avoidable work (waste)",
|
|
687
|
+
"metric": f"${waste:.2f} ({100 * waste / total_cost:.1f}%)",
|
|
688
|
+
"detail": f"Redundant reads: {sum(t['redundant_reads'] for t in main)}, duplicate calls: {sum(t['duplicate_calls'] for t in main)}, "
|
|
689
|
+
f"calls inside error streaks: {sum(max(0, t['max_error_streak'] - 2) for t in main)}.",
|
|
690
|
+
"severity": "warning" if waste / total_cost > 0.05 else "ok",
|
|
691
|
+
"advice": "Large re-reads usually mean context was compacted or the agent lost track; smaller, focused tasks help.",
|
|
692
|
+
"examples": [t["id"] for t in sorted(main, key=lambda x: -x["waste_cost"])[:5] if t["waste_cost"] > 0],
|
|
693
|
+
})
|
|
694
|
+
errs = sum(t["tool_errors"] for t in main)
|
|
695
|
+
calls = sum(t["tool_calls"] for t in main) or 1
|
|
696
|
+
streaky = [t for t in main if t["max_error_streak"] >= 3]
|
|
697
|
+
out.append({
|
|
698
|
+
"id": "errors", "title": "Tool reliability",
|
|
699
|
+
"metric": f"{100 * errs / calls:.1f}% tool calls failed",
|
|
700
|
+
"detail": f"{errs} failed calls; {len(streaky)} tasks had 3+ consecutive failures (the agent kept retrying).",
|
|
701
|
+
"severity": "warning" if errs / calls > 0.08 or streaky else "ok",
|
|
702
|
+
"advice": "Look at the error streak examples: repeated shell-quoting or path errors can be fixed with a CLAUDE.md note about the environment (e.g. Windows paths, PowerShell vs Bash).",
|
|
703
|
+
"examples": [t["id"] for t in sorted(streaky, key=lambda x: -x["max_error_streak"])[:5]],
|
|
704
|
+
})
|
|
705
|
+
ph = Counter()
|
|
706
|
+
for t in main:
|
|
707
|
+
ph.update(t["phase_cost"])
|
|
708
|
+
tot = sum(ph.values()) or 1
|
|
709
|
+
out.append({
|
|
710
|
+
"id": "phases", "title": "Where the effort goes",
|
|
711
|
+
"metric": ", ".join(f"{k} {100 * v / tot:.0f}%" for k, v in ph.most_common(4)),
|
|
712
|
+
"detail": "Share of model spend attributed to each phase (explore, edit, verify, ...). 'respond' is spend on turns that only produced text.",
|
|
713
|
+
"severity": "info",
|
|
714
|
+
"advice": "High explore share on small tasks suggests missing project context; a CLAUDE.md with architecture notes reduces it.",
|
|
715
|
+
"examples": [],
|
|
716
|
+
"breakdown": {k: round(v, 4) for k, v in ph.items()},
|
|
717
|
+
})
|
|
718
|
+
rework = [t for t in main if t["outcome"] in ("rework", "interrupted")]
|
|
719
|
+
out.append({
|
|
720
|
+
"id": "rework", "title": "First-time-right rate",
|
|
721
|
+
"metric": f"{100 * (1 - len(rework) / len(main)):.0f}%",
|
|
722
|
+
"detail": f"{len(rework)} of {len(main)} tasks were interrupted or followed by a correction from you.",
|
|
723
|
+
"severity": "warning" if len(rework) / len(main) > 0.15 else "ok",
|
|
724
|
+
"advice": "Open these tasks and compare the prompt to the final answer: ambiguity in the request vs. agent error.",
|
|
725
|
+
"examples": [t["id"] for t in rework[:8]],
|
|
726
|
+
})
|
|
727
|
+
ctx = [t for t in main if t["max_context"] > 250_000 or t["compactions"]]
|
|
728
|
+
ch = [t["cache_hit"] for t in main if t["cache_hit"] is not None]
|
|
729
|
+
out.append({
|
|
730
|
+
"id": "context", "title": "Context & cache hygiene",
|
|
731
|
+
"metric": f"median cache hit {100 * (statistics.median(ch) if ch else 0):.0f}%",
|
|
732
|
+
"detail": f"{len(ctx)} tasks pushed context past 250k tokens or triggered compaction.",
|
|
733
|
+
"severity": "warning" if ctx else "ok",
|
|
734
|
+
"advice": "Very long sessions get expensive per turn. Start a fresh session for unrelated work.",
|
|
735
|
+
"examples": [t["id"] for t in sorted(ctx, key=lambda x: -x["max_context"])[:5]],
|
|
736
|
+
})
|
|
737
|
+
par = [t["parallelism"] for t in main if t["parallelism"]]
|
|
738
|
+
out.append({
|
|
739
|
+
"id": "parallel", "title": "Parallel tool use",
|
|
740
|
+
"metric": f"{statistics.mean(par) if par else 0:.2f} tools per model turn",
|
|
741
|
+
"detail": "Each model turn re-sends the whole context. Batching independent tool calls into one turn cuts turns and cost.",
|
|
742
|
+
"severity": "info",
|
|
743
|
+
"advice": "Values near 1.0 mean strictly sequential work.",
|
|
744
|
+
"examples": [],
|
|
745
|
+
})
|
|
746
|
+
churn = [t for t in main if t["max_edits_one_file"] >= 8]
|
|
747
|
+
out.append({
|
|
748
|
+
"id": "churn", "title": "Edit churn",
|
|
749
|
+
"metric": f"{len(churn)} tasks with 8+ edits to one file",
|
|
750
|
+
"detail": "Many small edits to the same file often mean trial-and-error instead of a plan.",
|
|
751
|
+
"severity": "info" if not churn else "warning",
|
|
752
|
+
"advice": "Asking for a plan first (plan mode) on bigger changes usually reduces churn.",
|
|
753
|
+
"examples": [t["id"] for t in sorted(churn, key=lambda x: -x["max_edits_one_file"])[:5]],
|
|
754
|
+
})
|
|
755
|
+
return out
|