agentdynamics 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,755 @@
1
+ """Turn normalized runs into tasks, baselines, scores, waste findings and events.
2
+
3
+ AppDynamics -> AgentDynamics mapping implemented here:
4
+ Business Transaction -> Task (one user request, end to end)
5
+ BT type / entry point -> Task type (bugfix, feature, question ...)
6
+ Dynamic baselines -> per-task-type median/p90 of cost, duration, tokens
7
+ Apdex -> Agent Apdex from outcome + cost vs baseline
8
+ Health rules / events -> configurable rules evaluated per task
9
+ Code-level hotspots -> waste findings (redundant reads, loops, error streaks)
10
+ """
11
+ import math
12
+ import re
13
+ import statistics
14
+ import time
15
+ from collections import Counter, defaultdict
16
+
17
+
18
+ # ---------------------------------------------------------------- task typing
19
+
20
+ TYPE_RULES = [
21
+ ("review/audit", r"\b(security|audit|review|vulnerab|performance analysis|pen ?test|fuzz)"),
22
+ ("bugfix", r"\b(fix|bug|error|broken|not working|doesn'?t work|isn'?t|aren'?t|crash|fail|messed up|wrong|issue|stuck|missing|forbidden|negative value|no data|still (getting|not|seeing)|4\d\d\b|5\d\d\b)"),
23
+ ("git/deploy", r"\b(github|commit|push|pull request|\bpr\b|deploy|release|repo\b|merge|publish)"),
24
+ ("testing", r"\b(tests?|unit test|coverage|e2e)\b"),
25
+ ("refactor", r"\b(refactor|clean ?up|simplif|restructure|rename|reorganiz)"),
26
+ ("question/research", r"(\?\s*$|^\s*(what|why|how|is|are|can|does|do|should|would|which|explain|any other|give me (more )?(ideas|options|the values))\b)"),
27
+ ("feature/build", r"\b(create|build|add|implement|make|integrate|generate|develop|design|write|extend|enhance|improve)"),
28
+ ("change request", r"^\s*(change|switch|use|update|replace|remove|move|star|convert|set|modify|have)\b"),
29
+ ("question/research", r"\b(explain|research|compare|ideas?|options|think of|thoughts|opinion|check if|go through)\b"),
30
+ ("setup/run", r"\b(install|set ?up|configure|run|start|launch|open)\b"),
31
+ ]
32
+ FOLLOWUP = re.compile(
33
+ r"^\s*(continue|go ahead|yes|yep|ok(ay)?|proceed|sure|do it|try( it)? (again|now)|next|both|skip)\b|"
34
+ r"\b(continue|proceed|go ahead)\b|\b(done|is up|is uo|installed|completed)\s*[.!]?\s*$", re.I)
35
+ CORRECTION = re.compile(
36
+ r"^\s*(no\b|nope|wrong|that'?s not|still\b|doesn'?t|didn'?t|not working|it'?s broken|revert|undo|why did you|"
37
+ r"you (forgot|missed|broke|didn'?t)|i don'?t like|dont like)|still (not|doesn'?t|broken|failing|getting)|(completely )?messed up|not what I (asked|wanted)",
38
+ re.I,
39
+ )
40
+
41
+
42
+ def classify_task(prompt_kind, text):
43
+ if prompt_kind == "scheduled":
44
+ return "scheduled job"
45
+ if prompt_kind == "delegated":
46
+ return "subagent"
47
+ t = (text or "").strip()
48
+ low = t.lower()
49
+ if prompt_kind == "command":
50
+ return "slash command"
51
+ if len(t) < 60 and FOLLOWUP.search(t):
52
+ return "follow-up"
53
+ # The earliest intent keyword wins: "Create X, then test it" is a build, not a testing task.
54
+ best = None
55
+ for rank, (name, pat) in enumerate(TYPE_RULES):
56
+ m = re.search(pat, low[:600])
57
+ if m and (best is None or (m.start(), rank) < best[0]):
58
+ best = ((m.start(), rank), name)
59
+ return best[1] if best else "other"
60
+
61
+
62
+ # ---------------------------------------------------------------- helpers
63
+
64
+ def pct(values, p):
65
+ if not values:
66
+ return None
67
+ v = sorted(values)
68
+ k = (len(v) - 1) * p
69
+ f, c = math.floor(k), math.ceil(k)
70
+ return v[f] if f == c else v[f] + (v[c] - v[f]) * (k - f)
71
+
72
+
73
+ def clamp(x, lo=0.0, hi=100.0):
74
+ return max(lo, min(hi, x))
75
+
76
+
77
+ IDLE_CAP = 300 # gaps longer than this (user away, waiting on approval) don't count as agent time
78
+
79
+
80
+ def _active_seconds(tss):
81
+ v = sorted(x for x in tss if x)
82
+ return round(sum(min(b - a, IDLE_CAP) for a, b in zip(v, v[1:])), 1)
83
+
84
+
85
+ # ---------------------------------------------------------------- segmentation
86
+
87
+ def segment(run):
88
+ """Split a run's steps into tasks at each prompt step."""
89
+ tasks, cur = [], None
90
+ for s in run["steps"]:
91
+ if s["kind"] == "prompt":
92
+ cur = {"prompt_step": s, "steps": [s]}
93
+ tasks.append(cur)
94
+ else:
95
+ if cur is None:
96
+ cur = {"prompt_step": None, "steps": []}
97
+ tasks.append(cur)
98
+ cur["steps"].append(s)
99
+ return tasks
100
+
101
+
102
+ # ---------------------------------------------------------------- per-task metrics
103
+
104
+ def task_metrics(run, idx, seg):
105
+ steps = seg["steps"]
106
+ p = seg["prompt_step"]
107
+ llm = [s for s in steps if s["kind"] == "llm" and not s.get("denied")] # a denied model call was never made
108
+ tools = [s for s in steps if s["kind"] == "tool"]
109
+ notices = [s for s in steps if s["kind"] == "notice"]
110
+ tss = [s["ts"] for s in steps if s.get("ts")] + [s["end_ts"] for s in steps if s.get("end_ts")]
111
+ started = min(tss) if tss else run.get("started")
112
+ ended = max(tss) if tss else started
113
+
114
+ t = {
115
+ "id": f"{run['id']}#{idx}",
116
+ "run_id": run["id"],
117
+ "idx": idx,
118
+ "project": run["project"],
119
+ "source": run["source"],
120
+ "is_subagent": 1 if run.get("is_subagent") else 0,
121
+ "prompt_kind": p["name"] if p else "none",
122
+ "prompt": (p or {}).get("text", "")[:2000],
123
+ "started": started,
124
+ "ended": ended,
125
+ "wall_s": round((ended - started), 1) if started and ended else 0,
126
+ "duration_s": _active_seconds(tss),
127
+ "llm_calls": len(llm),
128
+ "tool_calls": len(tools),
129
+ "tool_errors": sum(1 for s in tools if s.get("is_error") and not s.get("denied")),
130
+ "input_tokens": sum(s.get("input_tokens", 0) for s in llm),
131
+ "output_tokens": sum(s.get("output_tokens", 0) for s in llm),
132
+ "cache_read": sum(s.get("cache_read", 0) for s in llm),
133
+ "cache_write": sum(s.get("cache_write", 0) for s in llm),
134
+ "thinking_tokens": sum(s.get("thinking_tokens", 0) for s in llm),
135
+ "cost": sum(s.get("cost", 0) for s in llm),
136
+ "max_context": max([s.get("context_tokens", 0) for s in llm] or [0]),
137
+ "models": ",".join(sorted({s["model"] for s in llm if s.get("model") and s["model"] != "<synthetic>"})),
138
+ "interrupts": sum(1 for s in notices if s["name"] == "interrupt")
139
+ + sum(1 for s in tools if s.get("interrupted") or s.get("rejected")),
140
+ "compactions": sum(1 for s in notices if s["name"] == "compaction"),
141
+ "api_errors": sum(1 for s in notices if s["name"] == "api_error"),
142
+ "subagent_cost": 0.0,
143
+ "subagents": 0,
144
+ }
145
+ t["total_tokens"] = t["input_tokens"] + t["output_tokens"] + t["cache_read"] + t["cache_write"]
146
+ in_side = t["input_tokens"] + t["cache_read"] + t["cache_write"]
147
+ t["cache_hit"] = round(t["cache_read"] / in_side, 3) if in_side else None
148
+ t["tool_error_rate"] = round(t["tool_errors"] / len(tools), 3) if tools else 0
149
+ t["task_type"] = classify_task(t["prompt_kind"], t["prompt"])
150
+ if run.get("workflow") and run.get("source") != "claude-code":
151
+ t["task_type"] = run["workflow"] # traced apps: the entry point names the business transaction
152
+
153
+ # --- phase attribution: split each LLM call's cost across the tool calls it issued
154
+ phase_calls = Counter(s["phase"] for s in tools)
155
+ phase_cost = Counter()
156
+ # map llm step -> tool steps that immediately follow it (same llm_msg)
157
+ for i, s in enumerate(steps):
158
+ if s["kind"] != "llm":
159
+ continue
160
+ issued = []
161
+ for s2 in steps[i + 1:]:
162
+ if s2["kind"] == "tool":
163
+ issued.append(s2)
164
+ elif s2["kind"] == "llm":
165
+ break
166
+ else:
167
+ continue
168
+ # only count tool steps sharing the same message id
169
+ if issued:
170
+ mid = issued[0].get("llm_msg")
171
+ issued = [x for x in issued if x.get("llm_msg") == mid]
172
+ c = s.get("cost", 0)
173
+ if issued:
174
+ share = c / len(issued)
175
+ for x in issued:
176
+ x["attributed_cost"] = share
177
+ phase_cost[x["phase"]] += share
178
+ else:
179
+ phase_cost["respond"] += c
180
+ t["phase_calls"] = dict(phase_calls)
181
+ t["phase_cost"] = {k: round(v, 5) for k, v in phase_cost.items()}
182
+ with_tools = [s for s in llm if s.get("tool_calls")]
183
+ t["parallelism"] = round(sum(s["tool_calls"] for s in with_tools) / len(with_tools), 2) if with_tools else 0
184
+
185
+ # --- process analysis & waste detection
186
+ read_seen = {} # input_hash -> seq
187
+ call_seen = {}
188
+ edited_since = set()
189
+ edits_per_file = Counter()
190
+ files_read = set()
191
+ redundant = duplicates = 0
192
+ streak = max_streak = 0
193
+ large = 0
194
+ first_edit_idx = None
195
+ last_edit_i = None
196
+ last_verify_i = None
197
+ waste_cost = 0.0
198
+ for i, s in enumerate(tools):
199
+ s.setdefault("flags", [])
200
+ ph = s["phase"]
201
+ tgt = s.get("target", "")
202
+ if s["name"] in ("Read", "NotebookRead"):
203
+ files_read.add(tgt)
204
+ h = s["input_hash"]
205
+ if h in read_seen and tgt not in edited_since:
206
+ s["flags"].append("redundant_read")
207
+ redundant += 1
208
+ read_seen[h] = i
209
+ edited_since.discard(tgt)
210
+ elif ph == "edit":
211
+ edits_per_file[tgt] += 1
212
+ edited_since.add(tgt)
213
+ # an edit invalidates earlier identical calls
214
+ call_seen.clear()
215
+ if first_edit_idx is None:
216
+ first_edit_idx = i
217
+ last_edit_i = i
218
+ if ph == "verify":
219
+ last_verify_i = i
220
+ if ph not in ("plan", "communicate", "edit") and s["name"] not in ("Read", "NotebookRead"):
221
+ h = s["input_hash"]
222
+ if h in call_seen and not s.get("is_error"):
223
+ s["flags"].append("duplicate_call")
224
+ duplicates += 1
225
+ call_seen[h] = i
226
+ if s.get("is_error"):
227
+ streak += 1
228
+ max_streak = max(max_streak, streak)
229
+ if streak >= 3:
230
+ s["flags"].append("error_streak")
231
+ else:
232
+ streak = 0
233
+ if s.get("output_chars", 0) > 40000:
234
+ s["flags"].append("large_output")
235
+ large += 1
236
+ if s.get("denied"):
237
+ s["flags"].append("denied") # tokens spent generating a call the policy refused
238
+ if any(f in s["flags"] for f in ("redundant_read", "duplicate_call", "error_streak", "denied")):
239
+ waste_cost += s.get("attributed_cost", 0)
240
+ t["files_read"] = len(files_read)
241
+ t["files_edited"] = len(edits_per_file)
242
+ t["edits"] = sum(edits_per_file.values())
243
+ t["max_edits_one_file"] = max(edits_per_file.values() or [0])
244
+ t["churn_file"] = edits_per_file.most_common(1)[0][0] if edits_per_file else None
245
+ t["redundant_reads"] = redundant
246
+ t["duplicate_calls"] = duplicates
247
+ t["max_error_streak"] = max_streak
248
+ t["large_outputs"] = large
249
+ t["explore_ratio"] = round(phase_calls.get("explore", 0) / len(tools), 3) if tools else 0
250
+ t["steps_to_first_edit"] = first_edit_idx
251
+ code_task = t["edits"] > 0 and any(
252
+ re.search(r"\.(py|js|ts|tsx|jsx|go|rs|java|cs|rb|php|c|cpp|h|kt|swift|vue|svelte|sql)$", f or "", re.I)
253
+ for f in edits_per_file
254
+ )
255
+ t["code_changed"] = 1 if code_task else 0
256
+ t["verified"] = None
257
+ if code_task:
258
+ t["verified"] = 1 if (last_verify_i is not None and last_verify_i > last_edit_i) else 0
259
+ t["unverified_edits"] = 1 if t["verified"] == 0 else 0
260
+ t["waste_cost"] = round(waste_cost, 5)
261
+
262
+ # --- final response & outcome signals (outcome finalized later with next prompt)
263
+ last_llm = llm[-1] if llm else None
264
+ t["final_stop"] = last_llm.get("stop_reason") if last_llm else None
265
+ t["final_text"] = (last_llm or {}).get("text", "")[:600]
266
+ last_tool = tools[-1] if tools else None
267
+ t["ended_on_error"] = 1 if (last_tool and last_tool.get("is_error") and (not last_llm or last_llm["seq"] < last_tool["seq"])) else 0
268
+ for s in steps:
269
+ s["task_id"] = t["id"]
270
+ flow_metrics(t, run, steps, llm, tools)
271
+ return t
272
+
273
+
274
+ def governance_metrics(t, run, steps, tools):
275
+ """Policy enforcement seen from the task's side (Aegis decisions recorded as steps)."""
276
+ denied = [s for s in steps if s.get("denied")]
277
+ tool_denied = [s for s in tools if s.get("denied")]
278
+ t["governed"] = 1 if (run.get("policy_version") or any(s.get("governed") or s.get("rule") for s in steps)) else 0
279
+ t["policy_version"] = run.get("policy_version")
280
+ t["policy_denials"] = len(tool_denied)
281
+ t["spend_denials"] = sum(1 for s in denied if s["kind"] == "llm")
282
+ t["budget_denials"] = sum(1 for s in denied if str(s.get("rule") or "").startswith("budget."))
283
+ t["revocations"] = sum(1 for s in steps if s["kind"] == "notice" and s.get("name") == "revoked")
284
+ t["blocked_cost"] = round(sum(s.get("attributed_cost") or 0 for s in tool_denied), 6)
285
+ t["denied_rules"] = dict(Counter(s.get("rule") or "unknown" for s in denied))
286
+ streak = best = 0
287
+ last = None
288
+ for s in tools:
289
+ key = s.get("name") if s.get("denied") else None # same tool refused in a row, whatever the rule
290
+ streak = streak + 1 if key is not None and key == last else (1 if key else 0)
291
+ last = key
292
+ best = max(best, streak)
293
+ t["repeated_denials"] = best
294
+
295
+
296
+ TRUNCATION = {"max_tokens", "length", "max_output_tokens", "MAX_TOKENS"}
297
+ REFUSAL = {"refusal", "content_filter", "SAFETY", "safety", "blocked"}
298
+ RATE_HINTS = ("429", "rate limit", "rate_limit", "overloaded", "529", "too many requests", "quota")
299
+
300
+
301
+ def flow_metrics(t, run, steps, llm, tools):
302
+ """Agent-flow metrics that apply to any framework (graph nodes, handoffs, LLM health, retrieval)."""
303
+ spans = [s for s in steps if s["kind"] == "span"]
304
+ notices = [s for s in steps if s["kind"] == "notice"]
305
+ t["environment"] = run.get("environment") or "default"
306
+ t["framework"] = run.get("framework") or run.get("source")
307
+ t["workflow"] = run.get("workflow") or (t["task_type"] if run.get("source") == "claude-code" else None)
308
+ t["steps_total"] = len(llm) + len(tools)
309
+ t["llm_errors"] = sum(1 for s in llm if s.get("is_error"))
310
+ t["truncations"] = sum(1 for s in llm if s.get("stop_reason") in TRUNCATION)
311
+ t["refusals"] = sum(1 for s in llm if s.get("stop_reason") in REFUSAL)
312
+ t["rate_limited"] = sum(1 for s in llm if s.get("rate_limited")) + sum(
313
+ 1 for s in notices if s["name"] == "api_error" and any(h in (s.get("text") or "").lower() for h in RATE_HINTS))
314
+ tt = [s["ttft_ms"] for s in llm if s.get("ttft_ms")]
315
+ t["ttft_ms"] = round(statistics.median(tt)) if tt else None
316
+ tps = [s["output_tokens"] / (s["duration_ms"] / 1000) for s in llm if s.get("duration_ms") and s.get("output_tokens", 0) > 20]
317
+ t["out_tps"] = round(statistics.median(tps), 1) if tps else None
318
+ rets = [s for s in tools if s.get("phase") == "retrieve"]
319
+ t["retrievals"] = len(rets)
320
+ t["empty_retrievals"] = sum(1 for s in rets if s.get("docs") == 0)
321
+ t["unpriced"] = sum(1 for s in llm if s.get("priced") is False and (s.get("input_tokens") or s.get("output_tokens")))
322
+ t["hitl"] = sum(1 for s in spans if s.get("hitl"))
323
+
324
+ # node executions: LangGraph nodes (a span named after its node), explicit node spans, or agent spans
325
+ node_execs = [s for s in spans if s.get("node") and (s.get("name") == s.get("node") or s.get("span_kind") == "node")]
326
+ if not node_execs:
327
+ node_execs = [s for s in spans if s.get("span_kind") == "agent"]
328
+ for s in node_execs:
329
+ if not s.get("node"):
330
+ s["node"] = s.get("agent") or s.get("name")
331
+ if node_execs:
332
+ seq = [s.get("node") or s.get("name") for s in node_execs]
333
+ else:
334
+ # no graph: use the process phases of tool calls (explore -> edit -> verify ...)
335
+ seq = [s["phase"] for s in tools]
336
+ collapsed = [x for i, x in enumerate(seq) if i == 0 or x != seq[i - 1]]
337
+ t["path"] = collapsed[:60]
338
+ visits = Counter(s.get("node") or s.get("name") for s in node_execs)
339
+ t["nodes"] = len(visits)
340
+ t["max_node_visits"] = max(visits.values()) if visits else 0
341
+ t["loop_node"] = visits.most_common(1)[0][0] if visits and t["max_node_visits"] > 1 else None
342
+ # critical node: the node whose executions account for most of the task's elapsed time
343
+ wall = max(t["wall_s"], 0.001)
344
+ dur_by_node = Counter()
345
+ for s in node_execs:
346
+ dur_by_node[s.get("node") or s.get("name")] += (s.get("duration_ms") or 0) / 1000
347
+ if dur_by_node:
348
+ n, d = dur_by_node.most_common(1)[0]
349
+ t["critical_node"], t["critical_share"] = n, round(min(1.0, d / wall), 3)
350
+ else:
351
+ t["critical_node"], t["critical_share"] = None, None
352
+ # multi-agent handoffs
353
+ agents = [s.get("agent") for s in steps if s["kind"] in ("llm", "tool") and s.get("agent")]
354
+ aseq = [a for i, a in enumerate(agents) if i == 0 or a != agents[i - 1]]
355
+ t["handoffs"] = max(0, len(aseq) - 1)
356
+ t["pingpong"] = sum(1 for i in range(2, len(aseq)) if aseq[i] == aseq[i - 2])
357
+ governance_metrics(t, run, steps, tools)
358
+ fb = [f["score"] for f in run.get("feedback") or [] if isinstance(f.get("score"), (int, float))]
359
+ t["feedback_score"] = round(statistics.mean(fb), 3) if fb and t["idx"] <= 1 else None
360
+
361
+
362
+ # ---------------------------------------------------------------- scoring
363
+
364
+ def score_task(t, base):
365
+ s = {}
366
+ b = base.get(t["task_type"]) or base.get("__all__") or {}
367
+ med_cost = b.get("cost_p50") or 0
368
+ ratio = (t["cost"] + t["subagent_cost"]) / med_cost if med_cost else 1
369
+ t["cost_vs_baseline"] = round(ratio, 2)
370
+ med_dur = b.get("duration_p50") or 0
371
+ t["duration_vs_baseline"] = round(t["duration_s"] / med_dur, 2) if med_dur else 1
372
+ waste_share = t["waste_cost"] / t["cost"] if t["cost"] else 0
373
+ s["efficiency"] = clamp(100 - 35 * math.log2(max(ratio, 1)) - 100 * waste_share)
374
+ focus = 100 - 6 * t["redundant_reads"] - 10 * t["duplicate_calls"] - 4 * max(0, t["max_edits_one_file"] - 4)
375
+ if t["edits"] and t["explore_ratio"] > 0.75:
376
+ focus -= 15
377
+ focus -= 8 * max(0, (t.get("max_node_visits") or 0) - 3) + 10 * (t.get("pingpong") or 0)
378
+ s["focus"] = clamp(focus) if (t["tool_calls"] or t.get("nodes")) else None
379
+ llm_pen = 8 * min(t.get("llm_errors") or 0, 5) + 6 * min(t.get("truncations") or 0, 5) + 10 * min(t.get("refusals") or 0, 3)
380
+ llm_pen += 40 if t.get("outcome") == "failed" else 0
381
+ if t["tool_calls"]:
382
+ s["reliability"] = clamp(100 * (1 - t["tool_error_rate"]) - 12 * max(0, t["max_error_streak"] - 1) - 15 * t["api_errors"] - llm_pen)
383
+ else:
384
+ s["reliability"] = clamp(100 - 15 * t["api_errors"] - llm_pen)
385
+ s["verification"] = None if t["verified"] is None else (100.0 if t["verified"] else 25.0)
386
+ if t["cache_hit"] is not None and t["llm_calls"] >= 3:
387
+ ctx = 100 * min(1, t["cache_hit"] / 0.9)
388
+ if t["max_context"] > 200_000:
389
+ ctx -= 20
390
+ ctx -= 15 * t["compactions"]
391
+ s["context"] = clamp(ctx)
392
+ else:
393
+ s["context"] = None
394
+ s["autonomy"] = clamp(100 - 45 * min(t["interrupts"], 2) - (35 if t.get("outcome") == "rework" else 0))
395
+ if t.get("governed"):
396
+ s["compliance"] = clamp(100 - 12 * (t.get("policy_denials") or 0) - 20 * max(0, (t.get("repeated_denials") or 0) - 1)
397
+ - 40 * (t.get("revocations") or 0) - 15 * (t.get("budget_denials") or 0))
398
+ else:
399
+ s["compliance"] = None
400
+ weights = {"efficiency": 0.25, "focus": 0.15, "reliability": 0.2, "verification": 0.15, "context": 0.1, "autonomy": 0.15,
401
+ "compliance": 0.15}
402
+ tot = sum(weights[k] for k, v in s.items() if v is not None)
403
+ s["overall"] = round(sum(weights[k] * v for k, v in s.items() if v is not None) / tot, 1) if tot else None
404
+ t["scores"] = {k: (round(v, 1) if v is not None else None) for k, v in s.items()}
405
+ t["score"] = t["scores"]["overall"]
406
+
407
+ # Agent Apdex: T = 1.5x median cost of this task type
408
+ T = 1.5 * med_cost if med_cost else None
409
+ total = t["cost"] + t["subagent_cost"]
410
+ if t["outcome"] in ("interrupted", "rework", "failed") or t["max_error_streak"] >= 4:
411
+ t["apdex"] = "frustrated"
412
+ elif T is None or total <= T:
413
+ t["apdex"] = "satisfied"
414
+ elif total <= 4 * T:
415
+ t["apdex"] = "tolerating"
416
+ else:
417
+ t["apdex"] = "frustrated"
418
+
419
+
420
+ def apdex_score(tasks):
421
+ if not tasks:
422
+ return None
423
+ c = Counter(t["apdex"] for t in tasks)
424
+ return round((c["satisfied"] + c["tolerating"] / 2) / len(tasks), 3)
425
+
426
+
427
+ # ---------------------------------------------------------------- health rules
428
+
429
+ DEFAULT_RULES = [
430
+ {"id": "cost_spike", "name": "Cost above baseline", "metric": "cost_vs_baseline", "op": ">", "value": 3, "severity": "warning",
431
+ "message": "Cost {v}x the median for '{task_type}' tasks"},
432
+ {"id": "cost_critical", "name": "Runaway cost", "metric": "cost_vs_baseline", "op": ">", "value": 8, "severity": "critical",
433
+ "message": "Cost {v}x the median for '{task_type}' tasks"},
434
+ {"id": "error_rate", "name": "High tool error rate", "metric": "tool_error_rate", "op": ">", "value": 0.25, "severity": "warning",
435
+ "guard": {"tool_calls": 4}, "message": "{v:.0%} of tool calls failed"},
436
+ {"id": "error_streak", "name": "Error retry loop", "metric": "max_error_streak", "op": ">=", "value": 3, "severity": "critical",
437
+ "message": "{v} consecutive failing tool calls"},
438
+ {"id": "loop", "name": "Repeated identical calls", "metric": "duplicate_calls", "op": ">=", "value": 3, "severity": "warning",
439
+ "message": "{v} identical tool calls repeated without an intervening edit"},
440
+ {"id": "redundant_reads", "name": "Redundant file reads", "metric": "redundant_reads", "op": ">=", "value": 3, "severity": "info",
441
+ "message": "{v} files re-read with no change in between"},
442
+ {"id": "unverified", "name": "Code changed but not verified", "metric": "unverified_edits", "op": "==", "value": 1, "severity": "warning",
443
+ "message": "Edited code but never ran tests/build/app after the last edit"},
444
+ {"id": "context", "name": "Context window pressure", "metric": "max_context", "op": ">", "value": 250000, "severity": "warning",
445
+ "message": "Context reached {v:,} tokens"},
446
+ {"id": "compaction", "name": "Context compacted", "metric": "compactions", "op": ">=", "value": 1, "severity": "info",
447
+ "message": "Conversation was compacted {v} time(s); earlier detail was lost"},
448
+ {"id": "interrupted", "name": "User interrupted agent", "metric": "interrupts", "op": ">=", "value": 1, "severity": "critical",
449
+ "message": "User stopped or rejected the agent {v} time(s)"},
450
+ {"id": "rework", "name": "User had to correct result", "metric": "rework", "op": "==", "value": 1, "severity": "warning",
451
+ "message": "Next message was a correction: \"{next_prompt}\""},
452
+ {"id": "low_cache", "name": "Poor prompt-cache reuse", "metric": "cache_hit", "op": "<", "value": 0.5, "severity": "info",
453
+ "guard": {"llm_calls": 5}, "message": "Only {v:.0%} of input tokens came from cache"},
454
+ {"id": "api_error", "name": "Model API errors", "metric": "api_errors", "op": ">=", "value": 1, "severity": "warning",
455
+ "message": "{v} API error(s) (overload, rate limit, timeout)"},
456
+ {"id": "slow", "name": "Slow task", "metric": "duration_vs_baseline", "op": ">", "value": 5, "severity": "info",
457
+ "guard": {"duration_s": 120}, "message": "Took {v}x the usual time for '{task_type}' tasks"},
458
+ # --- agent-flow rules (graphs, multi-agent, LLM health)
459
+ {"id": "run_failed", "name": "Run failed", "metric": "failed", "op": "==", "value": 1, "severity": "critical",
460
+ "message": "Run ended in error: {root_error}"},
461
+ {"id": "graph_loop", "name": "Node loop", "metric": "max_node_visits", "op": ">=", "value": 5, "severity": "warning",
462
+ "message": "Node '{loop_node}' ran {v} times in one run (possible loop / recursion)"},
463
+ {"id": "pingpong", "name": "Agent handoff ping-pong", "metric": "pingpong", "op": ">=", "value": 2, "severity": "warning",
464
+ "message": "Agents handed work back and forth {v} times"},
465
+ {"id": "truncation", "name": "Output truncated", "metric": "truncations", "op": ">=", "value": 1, "severity": "warning",
466
+ "message": "{v} model response(s) hit the max-token limit"},
467
+ {"id": "refusal", "name": "Model refusal", "metric": "refusals", "op": ">=", "value": 1, "severity": "warning",
468
+ "message": "{v} model response(s) refused or filtered"},
469
+ {"id": "rate_limit", "name": "Rate limited / overloaded", "metric": "rate_limited", "op": ">=", "value": 1, "severity": "warning",
470
+ "message": "{v} model call(s) rate-limited or overloaded"},
471
+ {"id": "llm_errors", "name": "Model call errors", "metric": "llm_errors", "op": ">=", "value": 2, "severity": "warning",
472
+ "message": "{v} model calls failed"},
473
+ {"id": "empty_retrieval", "name": "Retrieval returned nothing", "metric": "empty_retrievals", "op": ">=", "value": 1,
474
+ "severity": "info", "message": "{v} retrieval(s) returned no documents"},
475
+ {"id": "negative_feedback", "name": "Negative user feedback", "metric": "feedback_score", "op": "<", "value": 0.5,
476
+ "severity": "warning", "message": "Feedback score {v}"},
477
+ # --- governance rules (Aegis)
478
+ {"id": "policy_denials", "name": "Actions blocked by policy", "metric": "policy_denials", "op": ">=", "value": 3,
479
+ "severity": "warning", "message": "{v} tool calls were refused by the policy"},
480
+ {"id": "repeated_denials", "name": "Agent probing a boundary", "metric": "repeated_denials", "op": ">=", "value": 3,
481
+ "severity": "critical", "message": "Same forbidden call attempted {v} times in a row (prompt injection or stuck agent)"},
482
+ {"id": "revoked", "name": "Grant revoked", "metric": "revocations", "op": ">=", "value": 1, "severity": "critical",
483
+ "message": "The agent's authority was revoked mid-run"},
484
+ {"id": "budget_stop", "name": "Budget stop", "metric": "budget_denials", "op": ">=", "value": 1, "severity": "warning",
485
+ "message": "Budget limit reached {v} time(s); work was stopped"},
486
+ {"id": "slow_ttft", "name": "Slow first token", "metric": "ttft_ms", "op": ">", "value": 8000, "severity": "info",
487
+ "message": "Median time to first token {v:,.0f} ms"},
488
+ ]
489
+
490
+ OPS = {">": lambda a, b: a > b, ">=": lambda a, b: a >= b, "<": lambda a, b: a < b, "<=": lambda a, b: a <= b,
491
+ "==": lambda a, b: a == b}
492
+
493
+
494
+ def evaluate_rules(t, rules):
495
+ events = []
496
+ for r in rules:
497
+ if not r.get("enabled", True):
498
+ continue
499
+ v = t.get(r["metric"])
500
+ if v is None:
501
+ continue
502
+ guard = r.get("guard") or {}
503
+ if any((t.get(k) or 0) < gv for k, gv in guard.items()):
504
+ continue
505
+ if OPS[r["op"]](v, r["value"]):
506
+ ctx = dict(t)
507
+ ctx["v"] = v
508
+ ctx["next_prompt"] = (t.get("next_prompt") or "")[:80]
509
+ try:
510
+ msg = r["message"].format(**ctx)
511
+ except (KeyError, ValueError, IndexError):
512
+ msg = f"{r['metric']}={v}"
513
+ events.append({
514
+ "id": f"{t['id']}:{r['id']}", "ts": t["ended"], "rule_id": r["id"], "rule": r["name"],
515
+ "severity": r["severity"], "task_id": t["id"], "run_id": t["run_id"], "project": t["project"],
516
+ "task_type": t["task_type"], "message": msg, "value": v,
517
+ })
518
+ return events
519
+
520
+
521
+ # ---------------------------------------------------------------- orchestration
522
+
523
+ def run_tasks(run):
524
+ """Per-run analysis (cacheable): segment into tasks and compute task metrics."""
525
+ return [task_metrics(run, i, seg) for i, seg in enumerate(segment(run))]
526
+
527
+
528
+ def _outcomes_claude(ts, now):
529
+ for i, t in enumerate(ts):
530
+ # "continue" / "yes" carries on the previous request, so it belongs to that task type
531
+ if t["task_type"] == "follow-up" and i > 0 and ts[i - 1]["task_type"] not in ("follow-up",):
532
+ t["task_type"] = ts[i - 1]["task_type"]
533
+ if t["source"] == "claude-code":
534
+ t["workflow"] = t["task_type"]
535
+ nxt = next((x for x in ts[i + 1:] if x["prompt_kind"] in ("human", "command")), None)
536
+ t["next_prompt"] = nxt["prompt"][:300] if nxt else None
537
+ t["rework"] = 1 if nxt and CORRECTION.search(nxt["prompt"][:200]) else 0
538
+ if t["interrupts"]:
539
+ t["outcome"] = "interrupted"
540
+ elif t["rework"]:
541
+ t["outcome"] = "rework"
542
+ elif t["ended_on_error"] or (t["api_errors"] and t["final_stop"] != "end_turn"):
543
+ t["outcome"] = "failed"
544
+ elif nxt is None and t["source"] == "claude-code" and t["ended"] and now - t["ended"] < 600:
545
+ t["outcome"] = "in progress"
546
+ elif t["final_stop"] in ("end_turn", "stop_sequence") or t["llm_calls"]:
547
+ t["outcome"] = "completed"
548
+ else:
549
+ t["outcome"] = "unknown"
550
+
551
+
552
+ def _outcome_trace(t, run, now):
553
+ t.setdefault("next_prompt", None)
554
+ t.setdefault("rework", 0)
555
+ if run.get("root_status") == "error" or t["ended_on_error"]:
556
+ t["outcome"] = "failed"
557
+ elif t["hitl"] and not run.get("complete"):
558
+ t["outcome"] = "interrupted"
559
+ elif t["rework"] or (t["feedback_score"] is not None and t["feedback_score"] < 0.5):
560
+ t["outcome"] = "rework"
561
+ elif run.get("complete") is False:
562
+ t["outcome"] = "in progress" if t["ended"] and now - t["ended"] < 600 else "unknown"
563
+ else:
564
+ t["outcome"] = "completed"
565
+ t["root_error"] = run.get("root_error")
566
+
567
+
568
+ def finalize(runs, tasks_by_run, rules=None, now=None):
569
+ """Cross-run analysis: outcomes, subagent roll-up, baselines, scores, events."""
570
+ rules = rules or DEFAULT_RULES
571
+ now = now or time.time()
572
+ all_tasks = [t for r in runs for t in tasks_by_run[r["id"]]]
573
+ for t in all_tasks:
574
+ t["subagent_cost"], t["subagents"], t["parent_task_id"] = 0.0, 0, None
575
+
576
+ threads = defaultdict(list)
577
+ for run in runs:
578
+ ts = tasks_by_run[run["id"]]
579
+ if run["source"] == "claude-code" or not run.get("workflow"):
580
+ _outcomes_claude(ts, now)
581
+ else:
582
+ if run.get("thread_id"):
583
+ threads[(run["source"], run["thread_id"])].extend(ts)
584
+ # traced conversations: the next trace in the same thread plays the role of "your next message"
585
+ for group in threads.values():
586
+ group.sort(key=lambda x: x["started"] or 0)
587
+ for a, b in zip(group, group[1:]):
588
+ a["next_prompt"] = b["prompt"][:300]
589
+ a["rework"] = 1 if CORRECTION.search(b["prompt"][:200]) else 0
590
+ for run in runs:
591
+ if not (run["source"] == "claude-code" or not run.get("workflow")):
592
+ for t in tasks_by_run[run["id"]]:
593
+ _outcome_trace(t, run, now)
594
+
595
+ # subagent roll-up: link child runs to the parent task that spawned them
596
+ task_by_id = {t["id"]: t for t in all_tasks}
597
+ parent_tasks = defaultdict(list)
598
+ for t in all_tasks:
599
+ if not t["is_subagent"]:
600
+ parent_tasks[t["run_id"]].append(t)
601
+ spawn_map = {}
602
+ for run in runs:
603
+ if run.get("is_subagent"):
604
+ continue
605
+ for s in run["steps"]:
606
+ if s["kind"] == "tool" and s.get("subagent_id"):
607
+ spawn_map[s["subagent_id"]] = s.get("task_id")
608
+ for run in runs:
609
+ if not run.get("is_subagent"):
610
+ continue
611
+ agent_id = run["id"].split(":")[-1].replace("agent-", "")
612
+ parent_task_id = spawn_map.get(agent_id)
613
+ if not parent_task_id:
614
+ cands = [t for t in parent_tasks.get(run["parent_id"], []) if t["started"] and run["started"] and t["started"] <= run["started"]]
615
+ parent_task_id = cands[-1]["id"] if cands else None
616
+ run["parent_task_id"] = parent_task_id
617
+ for t in tasks_by_run[run["id"]]:
618
+ t["parent_task_id"] = parent_task_id
619
+ pt = task_by_id.get(parent_task_id)
620
+ if pt:
621
+ pt["subagent_cost"] += t["cost"]
622
+ pt["subagents"] += 1
623
+
624
+ # baselines per task type (top-level tasks only)
625
+ main = [t for t in all_tasks if not t["is_subagent"] and t["llm_calls"] > 0]
626
+ groups = defaultdict(list)
627
+ for t in main:
628
+ groups[t["task_type"]].append(t)
629
+ groups["__all__"] = main
630
+ sub = [t for t in all_tasks if t["is_subagent"] and t["llm_calls"] > 0]
631
+ if sub:
632
+ groups["subagent"] = sub
633
+ baselines = {}
634
+ for k, g in groups.items():
635
+ if len(g) < 3 and k != "__all__":
636
+ continue
637
+ costs = [t["cost"] + t["subagent_cost"] for t in g]
638
+ durs = [t["duration_s"] for t in g]
639
+ baselines[k] = {
640
+ "n": len(g),
641
+ "cost_p50": pct(costs, 0.5), "cost_p90": pct(costs, 0.9),
642
+ "duration_p50": pct(durs, 0.5), "duration_p90": pct(durs, 0.9),
643
+ "tokens_p50": pct([t["total_tokens"] for t in g], 0.5),
644
+ "tool_calls_p50": pct([t["tool_calls"] for t in g], 0.5),
645
+ "steps_p50": pct([t["steps_total"] for t in g], 0.5),
646
+ }
647
+
648
+ events = []
649
+ for t in all_tasks:
650
+ t["failed"] = 1 if t.get("outcome") == "failed" else 0
651
+ if t["llm_calls"] == 0 and t["tool_calls"] == 0:
652
+ t.update({"scores": {}, "score": None, "apdex": None, "cost_vs_baseline": None, "duration_vs_baseline": None})
653
+ continue
654
+ score_task(t, baselines)
655
+ events.extend(evaluate_rules(t, rules))
656
+ return all_tasks, baselines, events
657
+
658
+
659
+ def analyze(runs, rules=None, now=None):
660
+ """One-shot analysis of a list of runs (used by tests and the CLI)."""
661
+ return finalize(runs, {r["id"]: run_tasks(r) for r in runs}, rules, now)
662
+
663
+
664
+ # ---------------------------------------------------------------- process review (coaching)
665
+
666
+ def process_insights(tasks):
667
+ """Aggregate findings meant to help a human assess how the agent works."""
668
+ main = [t for t in tasks if not t["is_subagent"] and t["llm_calls"] > 0]
669
+ if not main:
670
+ return []
671
+ out = []
672
+ code = [t for t in main if t["code_changed"]]
673
+ if code:
674
+ unv = [t for t in code if not t["verified"]]
675
+ out.append({
676
+ "id": "verification", "title": "Verification after code changes",
677
+ "metric": f"{100 * (1 - len(unv) / len(code)):.0f}% verified",
678
+ "detail": f"{len(unv)} of {len(code)} code-changing tasks ended without running tests, a build, or the app after the last edit.",
679
+ "severity": "warning" if len(unv) / len(code) > 0.3 else "ok",
680
+ "advice": "Add to CLAUDE.md: 'After editing code, always run the relevant tests or build before reporting done.'",
681
+ "examples": [t["id"] for t in sorted(unv, key=lambda x: -x["cost"])[:5]],
682
+ })
683
+ total_cost = sum(t["cost"] for t in main) or 1
684
+ waste = sum(t["waste_cost"] for t in main)
685
+ out.append({
686
+ "id": "waste", "title": "Avoidable work (waste)",
687
+ "metric": f"${waste:.2f} ({100 * waste / total_cost:.1f}%)",
688
+ "detail": f"Redundant reads: {sum(t['redundant_reads'] for t in main)}, duplicate calls: {sum(t['duplicate_calls'] for t in main)}, "
689
+ f"calls inside error streaks: {sum(max(0, t['max_error_streak'] - 2) for t in main)}.",
690
+ "severity": "warning" if waste / total_cost > 0.05 else "ok",
691
+ "advice": "Large re-reads usually mean context was compacted or the agent lost track; smaller, focused tasks help.",
692
+ "examples": [t["id"] for t in sorted(main, key=lambda x: -x["waste_cost"])[:5] if t["waste_cost"] > 0],
693
+ })
694
+ errs = sum(t["tool_errors"] for t in main)
695
+ calls = sum(t["tool_calls"] for t in main) or 1
696
+ streaky = [t for t in main if t["max_error_streak"] >= 3]
697
+ out.append({
698
+ "id": "errors", "title": "Tool reliability",
699
+ "metric": f"{100 * errs / calls:.1f}% tool calls failed",
700
+ "detail": f"{errs} failed calls; {len(streaky)} tasks had 3+ consecutive failures (the agent kept retrying).",
701
+ "severity": "warning" if errs / calls > 0.08 or streaky else "ok",
702
+ "advice": "Look at the error streak examples: repeated shell-quoting or path errors can be fixed with a CLAUDE.md note about the environment (e.g. Windows paths, PowerShell vs Bash).",
703
+ "examples": [t["id"] for t in sorted(streaky, key=lambda x: -x["max_error_streak"])[:5]],
704
+ })
705
+ ph = Counter()
706
+ for t in main:
707
+ ph.update(t["phase_cost"])
708
+ tot = sum(ph.values()) or 1
709
+ out.append({
710
+ "id": "phases", "title": "Where the effort goes",
711
+ "metric": ", ".join(f"{k} {100 * v / tot:.0f}%" for k, v in ph.most_common(4)),
712
+ "detail": "Share of model spend attributed to each phase (explore, edit, verify, ...). 'respond' is spend on turns that only produced text.",
713
+ "severity": "info",
714
+ "advice": "High explore share on small tasks suggests missing project context; a CLAUDE.md with architecture notes reduces it.",
715
+ "examples": [],
716
+ "breakdown": {k: round(v, 4) for k, v in ph.items()},
717
+ })
718
+ rework = [t for t in main if t["outcome"] in ("rework", "interrupted")]
719
+ out.append({
720
+ "id": "rework", "title": "First-time-right rate",
721
+ "metric": f"{100 * (1 - len(rework) / len(main)):.0f}%",
722
+ "detail": f"{len(rework)} of {len(main)} tasks were interrupted or followed by a correction from you.",
723
+ "severity": "warning" if len(rework) / len(main) > 0.15 else "ok",
724
+ "advice": "Open these tasks and compare the prompt to the final answer: ambiguity in the request vs. agent error.",
725
+ "examples": [t["id"] for t in rework[:8]],
726
+ })
727
+ ctx = [t for t in main if t["max_context"] > 250_000 or t["compactions"]]
728
+ ch = [t["cache_hit"] for t in main if t["cache_hit"] is not None]
729
+ out.append({
730
+ "id": "context", "title": "Context & cache hygiene",
731
+ "metric": f"median cache hit {100 * (statistics.median(ch) if ch else 0):.0f}%",
732
+ "detail": f"{len(ctx)} tasks pushed context past 250k tokens or triggered compaction.",
733
+ "severity": "warning" if ctx else "ok",
734
+ "advice": "Very long sessions get expensive per turn. Start a fresh session for unrelated work.",
735
+ "examples": [t["id"] for t in sorted(ctx, key=lambda x: -x["max_context"])[:5]],
736
+ })
737
+ par = [t["parallelism"] for t in main if t["parallelism"]]
738
+ out.append({
739
+ "id": "parallel", "title": "Parallel tool use",
740
+ "metric": f"{statistics.mean(par) if par else 0:.2f} tools per model turn",
741
+ "detail": "Each model turn re-sends the whole context. Batching independent tool calls into one turn cuts turns and cost.",
742
+ "severity": "info",
743
+ "advice": "Values near 1.0 mean strictly sequential work.",
744
+ "examples": [],
745
+ })
746
+ churn = [t for t in main if t["max_edits_one_file"] >= 8]
747
+ out.append({
748
+ "id": "churn", "title": "Edit churn",
749
+ "metric": f"{len(churn)} tasks with 8+ edits to one file",
750
+ "detail": "Many small edits to the same file often mean trial-and-error instead of a plan.",
751
+ "severity": "info" if not churn else "warning",
752
+ "advice": "Asking for a plan first (plan mode) on bigger changes usually reduces churn.",
753
+ "examples": [t["id"] for t in sorted(churn, key=lambda x: -x["max_edits_one_file"])[:5]],
754
+ })
755
+ return out