agentprof 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentprof/__init__.py +7 -0
- agentprof/adapters/__init__.py +15 -0
- agentprof/adapters/base.py +79 -0
- agentprof/adapters/claude_code/__init__.py +3 -0
- agentprof/adapters/claude_code/adapter.py +133 -0
- agentprof/adapters/claude_code/discovery.py +79 -0
- agentprof/adapters/claude_code/prompts.py +44 -0
- agentprof/adapters/claude_code/tools.py +102 -0
- agentprof/adapters/claude_code/transcript.py +264 -0
- agentprof/adapters/claude_code/tree.py +492 -0
- agentprof/adapters/codex/__init__.py +3 -0
- agentprof/adapters/codex/adapter.py +164 -0
- agentprof/adapters/codex/discovery.py +117 -0
- agentprof/adapters/codex/prompts.py +16 -0
- agentprof/adapters/codex/rollout.py +328 -0
- agentprof/adapters/codex/tools.py +97 -0
- agentprof/adapters/codex/tree.py +311 -0
- agentprof/adapters/copilot_vscode/__init__.py +3 -0
- agentprof/adapters/copilot_vscode/adapter.py +166 -0
- agentprof/adapters/copilot_vscode/debuglog.py +136 -0
- agentprof/adapters/copilot_vscode/discovery.py +153 -0
- agentprof/adapters/copilot_vscode/session.py +281 -0
- agentprof/adapters/copilot_vscode/tools.py +127 -0
- agentprof/adapters/copilot_vscode/transcript.py +176 -0
- agentprof/adapters/copilot_vscode/tree.py +325 -0
- agentprof/adapters/execution.py +64 -0
- agentprof/adapters/timestamps.py +13 -0
- agentprof/adapters/turns.py +24 -0
- agentprof/analysis/__init__.py +3 -0
- agentprof/analysis/agent_summary.py +161 -0
- agentprof/analysis/call_context.py +101 -0
- agentprof/analysis/evidence_findings.py +151 -0
- agentprof/analysis/execution.py +39 -0
- agentprof/analysis/heuristics.py +306 -0
- agentprof/analysis/pipeline.py +31 -0
- agentprof/analysis/rollup.py +196 -0
- agentprof/cli.py +141 -0
- agentprof/model.py +341 -0
- agentprof/pricing.json +23 -0
- agentprof/pricing.py +120 -0
- agentprof/registry.py +304 -0
- agentprof/server/__init__.py +3 -0
- agentprof/server/app.py +399 -0
- agentprof/server/openapi.py +31 -0
- agentprof/server/pricing_schema.py +40 -0
- agentprof/server/schemas.py +579 -0
- agentprof/server/static/assets/index-C4HEqUVf.css +1 -0
- agentprof/server/static/assets/index-ZsPn1H3d.js +58 -0
- agentprof/server/static/index.html +14 -0
- agentprof-0.1.0.dist-info/METADATA +177 -0
- agentprof-0.1.0.dist-info/RECORD +54 -0
- agentprof-0.1.0.dist-info/WHEEL +4 -0
- agentprof-0.1.0.dist-info/entry_points.txt +8 -0
- agentprof-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,306 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
# Copyright (c) 2026 epicodic
|
|
3
|
+
"""Waste heuristics over the neutral model. Findings are hints, not verdicts.
|
|
4
|
+
|
|
5
|
+
Heuristics only look at tool categories and normalised arguments, never at native tool ids.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
from collections.abc import Callable
|
|
10
|
+
from typing import NamedTuple
|
|
11
|
+
|
|
12
|
+
from agentprof.model import Finding, LlmCall, Node, NodeKind, ToolCategory, iter_nodes
|
|
13
|
+
|
|
14
|
+
_MIN_REPEATED_READS = 3
|
|
15
|
+
_MIN_REPEATED_COMMANDS = 3
|
|
16
|
+
_MAX_FAILURE_RATE = 0.2
|
|
17
|
+
_MIN_POLLING_STREAK = 5
|
|
18
|
+
_MIN_SIBLINGS_FOR_COST_OUTLIER = 4
|
|
19
|
+
_COST_OUTLIER_PERCENTILE = 0.9
|
|
20
|
+
_REACQUIRE_WINDOW_MS = 10 * 60 * 1000
|
|
21
|
+
_MIN_TOPIC_SIMILARITY = 0.6
|
|
22
|
+
_MAX_MEAN_UNCACHED_PROMPT_TOKENS = 100_000
|
|
23
|
+
_MIN_CACHE_HIT_RATIO = 0.5
|
|
24
|
+
_MIN_CALLS_FOR_CACHE_RATIO = 3
|
|
25
|
+
_WORD = re.compile(r"\w+")
|
|
26
|
+
_NUMBER = re.compile(r"\d+")
|
|
27
|
+
_AGENT_KINDS = (NodeKind.TURN, NodeKind.AGENT)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class _Read(NamedTuple):
|
|
31
|
+
path: str
|
|
32
|
+
line_range: tuple[int, int] | None
|
|
33
|
+
node: Node
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _reads(agent: Node) -> list[_Read]:
|
|
37
|
+
reads: list[_Read] = []
|
|
38
|
+
for child in agent.children:
|
|
39
|
+
tool = child.tool
|
|
40
|
+
if tool is None or tool.category is not ToolCategory.READ:
|
|
41
|
+
continue
|
|
42
|
+
if (path := tool.path) is not None:
|
|
43
|
+
reads.append(_Read(path=path, line_range=tool.line_range, node=child))
|
|
44
|
+
return reads
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _is_category(node: Node, category: ToolCategory) -> bool:
|
|
48
|
+
return node.tool is not None and node.tool.category is category
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _percentile(sorted_values: list[float], quantile: float) -> float:
|
|
52
|
+
"""Linear-interpolation percentile (numpy's default `linear` method)."""
|
|
53
|
+
if len(sorted_values) == 1:
|
|
54
|
+
return sorted_values[0]
|
|
55
|
+
position = quantile * (len(sorted_values) - 1)
|
|
56
|
+
lower_index = int(position)
|
|
57
|
+
upper_index = min(lower_index + 1, len(sorted_values) - 1)
|
|
58
|
+
fraction = position - lower_index
|
|
59
|
+
return sorted_values[lower_index] + (sorted_values[upper_index] - sorted_values[lower_index]) * fraction
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _overlaps(first: tuple[int, int] | None, second: tuple[int, int] | None) -> bool:
|
|
63
|
+
"""Unknown line ranges mean the whole file and overlap everything."""
|
|
64
|
+
if first is None or second is None:
|
|
65
|
+
return True
|
|
66
|
+
return first[0] <= second[1] and second[0] <= first[1]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _check_repeated_reads(agent: Node) -> list[Finding]:
|
|
70
|
+
"""W1: the same file read with overlapping line ranges at least three times within one agent."""
|
|
71
|
+
ranges_by_path: dict[str, list[tuple[int, int] | None]] = {}
|
|
72
|
+
for read in _reads(agent):
|
|
73
|
+
ranges_by_path.setdefault(read.path, []).append(read.line_range)
|
|
74
|
+
findings: list[Finding] = []
|
|
75
|
+
for path, ranges in ranges_by_path.items():
|
|
76
|
+
count = max(sum(1 for other in ranges if _overlaps(current, other)) for current in ranges)
|
|
77
|
+
if count >= _MIN_REPEATED_READS:
|
|
78
|
+
findings.append(
|
|
79
|
+
Finding(
|
|
80
|
+
heuristic_id="W1",
|
|
81
|
+
node_id=agent.node_id,
|
|
82
|
+
severity="low",
|
|
83
|
+
message=f"{path!r} read {count} times with overlapping line ranges",
|
|
84
|
+
evidence={"path": path, "count": count},
|
|
85
|
+
)
|
|
86
|
+
)
|
|
87
|
+
return findings
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _check_repeated_or_failed_commands(agent: Node) -> list[Finding]:
|
|
91
|
+
"""W4: the same shell command at least three times, or a tool failure rate of 20 % or more."""
|
|
92
|
+
findings: list[Finding] = []
|
|
93
|
+
counts: dict[str, int] = {}
|
|
94
|
+
for child in agent.children:
|
|
95
|
+
tool = child.tool
|
|
96
|
+
if tool is None or tool.category is not ToolCategory.SHELL:
|
|
97
|
+
continue
|
|
98
|
+
if (command := tool.command) is not None:
|
|
99
|
+
counts[command] = counts.get(command, 0) + 1
|
|
100
|
+
for command, count in counts.items():
|
|
101
|
+
if count >= _MIN_REPEATED_COMMANDS:
|
|
102
|
+
findings.append(
|
|
103
|
+
Finding(
|
|
104
|
+
heuristic_id="W4",
|
|
105
|
+
node_id=agent.node_id,
|
|
106
|
+
severity="medium",
|
|
107
|
+
message=f"repeated command run {count} times: {command!r}",
|
|
108
|
+
evidence={"command": command, "count": count},
|
|
109
|
+
)
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
tool_children = [child for child in agent.children if child.kind is NodeKind.TOOL]
|
|
113
|
+
if tool_children:
|
|
114
|
+
failures = sum(1 for child in tool_children if child.success is False)
|
|
115
|
+
rate = failures / len(tool_children)
|
|
116
|
+
if rate >= _MAX_FAILURE_RATE:
|
|
117
|
+
findings.append(
|
|
118
|
+
Finding(
|
|
119
|
+
heuristic_id="W4",
|
|
120
|
+
node_id=agent.node_id,
|
|
121
|
+
severity="medium",
|
|
122
|
+
message=f"tool failure rate {rate:.0%} ({failures}/{len(tool_children)})",
|
|
123
|
+
evidence={"failure_rate": rate},
|
|
124
|
+
)
|
|
125
|
+
)
|
|
126
|
+
return findings
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _check_polling_loops(agent: Node) -> list[Finding]:
|
|
130
|
+
"""W5: at least five consecutive shell polling calls without other tool calls."""
|
|
131
|
+
streak = 0
|
|
132
|
+
for child in agent.children:
|
|
133
|
+
if not _is_category(child, ToolCategory.SHELL_POLL):
|
|
134
|
+
streak = 0
|
|
135
|
+
continue
|
|
136
|
+
streak += 1
|
|
137
|
+
if streak == _MIN_POLLING_STREAK:
|
|
138
|
+
return [
|
|
139
|
+
Finding(
|
|
140
|
+
heuristic_id="W5",
|
|
141
|
+
node_id=agent.node_id,
|
|
142
|
+
severity="low",
|
|
143
|
+
message=f"{streak} consecutive polling calls",
|
|
144
|
+
evidence={"streak": streak},
|
|
145
|
+
)
|
|
146
|
+
]
|
|
147
|
+
return []
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _check_cost_outliers(parent: Node) -> list[Finding]:
|
|
151
|
+
"""W7: an agent above the 90th percentile of its siblings' cost (at least four siblings, one unit)."""
|
|
152
|
+
siblings = [c for c in parent.children if c.kind is NodeKind.AGENT and c.cost_total.value is not None]
|
|
153
|
+
if len(siblings) < _MIN_SIBLINGS_FOR_COST_OUTLIER:
|
|
154
|
+
return []
|
|
155
|
+
units = {sibling.cost_total.unit for sibling in siblings}
|
|
156
|
+
if len(units) != 1:
|
|
157
|
+
return []
|
|
158
|
+
unit = units.pop()
|
|
159
|
+
threshold = _percentile(sorted(float(s.cost_total.value or 0.0) for s in siblings), _COST_OUTLIER_PERCENTILE)
|
|
160
|
+
findings: list[Finding] = []
|
|
161
|
+
for sibling in siblings:
|
|
162
|
+
cost = float(sibling.cost_total.value or 0.0)
|
|
163
|
+
if cost > threshold:
|
|
164
|
+
findings.append(
|
|
165
|
+
Finding(
|
|
166
|
+
heuristic_id="W7",
|
|
167
|
+
node_id=sibling.node_id,
|
|
168
|
+
severity="medium",
|
|
169
|
+
message=f"cost {cost:.1f} {unit} above the 90th percentile of its siblings",
|
|
170
|
+
evidence={"cost": cost, "unit": unit, "threshold": threshold},
|
|
171
|
+
)
|
|
172
|
+
)
|
|
173
|
+
return findings
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _check_reacquired_context(parent: Node) -> list[Finding]:
|
|
177
|
+
"""W2: a child agent reads a file its parent read within the preceding 10 minutes."""
|
|
178
|
+
parent_read_ends = [
|
|
179
|
+
(read.path, read.node.end.number()) for read in _reads(parent) if read.node.end.value is not None
|
|
180
|
+
]
|
|
181
|
+
findings: list[Finding] = []
|
|
182
|
+
for child_agent in parent.children:
|
|
183
|
+
if child_agent.kind is not NodeKind.AGENT:
|
|
184
|
+
continue
|
|
185
|
+
flagged: set[str] = set()
|
|
186
|
+
for read in _reads(child_agent):
|
|
187
|
+
if read.path in flagged or read.node.start.value is None:
|
|
188
|
+
continue
|
|
189
|
+
start = read.node.start.number()
|
|
190
|
+
if any(path == read.path and 0 <= start - end <= _REACQUIRE_WINDOW_MS for path, end in parent_read_ends):
|
|
191
|
+
flagged.add(read.path)
|
|
192
|
+
findings.append(
|
|
193
|
+
Finding(
|
|
194
|
+
heuristic_id="W2",
|
|
195
|
+
node_id=child_agent.node_id,
|
|
196
|
+
severity="low",
|
|
197
|
+
message=f"{read.path!r} was already read by the parent agent",
|
|
198
|
+
evidence={"path": read.path},
|
|
199
|
+
)
|
|
200
|
+
)
|
|
201
|
+
return findings
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _words(topic: str) -> set[str]:
|
|
205
|
+
return set(_WORD.findall(topic.lower()))
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _numbers(topic: str) -> set[str]:
|
|
209
|
+
return set(_NUMBER.findall(topic))
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _check_retry_chains(parent: Node) -> list[Finding]:
|
|
213
|
+
"""W3: sibling agents with a word-set Jaccard topic similarity of at least 0.6, unless their numbers differ."""
|
|
214
|
+
agents = [child for child in parent.children if child.kind is NodeKind.AGENT]
|
|
215
|
+
findings: list[Finding] = []
|
|
216
|
+
for index, later in enumerate(agents):
|
|
217
|
+
later_words = _words(later.topic)
|
|
218
|
+
for earlier in agents[:index]:
|
|
219
|
+
earlier_words = _words(earlier.topic)
|
|
220
|
+
later_numbers, earlier_numbers = _numbers(later.topic), _numbers(earlier.topic)
|
|
221
|
+
if later_numbers and earlier_numbers and later_numbers != earlier_numbers:
|
|
222
|
+
continue # numbered variants of one task ("Task 2", "Task 3") are not retries
|
|
223
|
+
union = later_words | earlier_words
|
|
224
|
+
if not union:
|
|
225
|
+
continue
|
|
226
|
+
similarity = len(later_words & earlier_words) / len(union)
|
|
227
|
+
if similarity >= _MIN_TOPIC_SIMILARITY:
|
|
228
|
+
findings.append(
|
|
229
|
+
Finding(
|
|
230
|
+
heuristic_id="W3",
|
|
231
|
+
node_id=later.node_id,
|
|
232
|
+
severity="medium",
|
|
233
|
+
message=f"topic {similarity:.0%} similar to earlier sibling {earlier.node_id!r}",
|
|
234
|
+
evidence={"earlier": earlier.node_id, "similarity": similarity},
|
|
235
|
+
)
|
|
236
|
+
)
|
|
237
|
+
break
|
|
238
|
+
return findings
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _prompt_and_cached(call: LlmCall) -> tuple[float, float] | None:
|
|
242
|
+
"""Prompt size and cached part of one call, or `None` if the counts are unavailable."""
|
|
243
|
+
tokens = call.tokens
|
|
244
|
+
if (uncached := tokens.input.value) is None or (cached := tokens.cache_read.value) is None:
|
|
245
|
+
return None
|
|
246
|
+
cache_write = tokens.cache_write.value or 0
|
|
247
|
+
return float(uncached + cached + cache_write), float(cached)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _check_context_bloat(agent: Node) -> list[Finding]:
|
|
251
|
+
"""W8: large mean uncached prompt per LLM call, or a low cache hit ratio.
|
|
252
|
+
|
|
253
|
+
Only the uncached part (`input + cache_write`) counts towards the size rule: cached tokens cost a fraction
|
|
254
|
+
of that, so a large but well-cached context is not waste.
|
|
255
|
+
"""
|
|
256
|
+
measured = [pair for call in agent.llm_calls if (pair := _prompt_and_cached(call)) is not None]
|
|
257
|
+
if not measured:
|
|
258
|
+
return []
|
|
259
|
+
findings: list[Finding] = []
|
|
260
|
+
total_prompt = sum(prompt for prompt, _ in measured)
|
|
261
|
+
mean_uncached = sum(prompt - cached for prompt, cached in measured) / len(measured)
|
|
262
|
+
if mean_uncached > _MAX_MEAN_UNCACHED_PROMPT_TOKENS:
|
|
263
|
+
findings.append(
|
|
264
|
+
Finding(
|
|
265
|
+
heuristic_id="W8",
|
|
266
|
+
node_id=agent.node_id,
|
|
267
|
+
severity="medium",
|
|
268
|
+
message=f"mean uncached prompt of {mean_uncached:,.0f} tokens per LLM call",
|
|
269
|
+
evidence={"mean_uncached_prompt_tokens": mean_uncached},
|
|
270
|
+
)
|
|
271
|
+
)
|
|
272
|
+
if len(measured) >= _MIN_CALLS_FOR_CACHE_RATIO and total_prompt > 0:
|
|
273
|
+
ratio = sum(cached for _, cached in measured) / total_prompt
|
|
274
|
+
if ratio < _MIN_CACHE_HIT_RATIO:
|
|
275
|
+
findings.append(
|
|
276
|
+
Finding(
|
|
277
|
+
heuristic_id="W8",
|
|
278
|
+
node_id=agent.node_id,
|
|
279
|
+
severity="low",
|
|
280
|
+
message=f"cache hit ratio {ratio:.0%}",
|
|
281
|
+
evidence={"cache_hit_ratio": ratio},
|
|
282
|
+
)
|
|
283
|
+
)
|
|
284
|
+
return findings
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
_CHECKS: tuple[Callable[[Node], list[Finding]], ...] = (
|
|
288
|
+
_check_repeated_reads,
|
|
289
|
+
_check_reacquired_context,
|
|
290
|
+
_check_retry_chains,
|
|
291
|
+
_check_repeated_or_failed_commands,
|
|
292
|
+
_check_polling_loops,
|
|
293
|
+
_check_cost_outliers,
|
|
294
|
+
_check_context_bloat,
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def find_findings(root: Node) -> list[Finding]:
|
|
299
|
+
"""Run all heuristics on every turn and agent node of the tree rooted at `root`."""
|
|
300
|
+
return [
|
|
301
|
+
finding
|
|
302
|
+
for node in iter_nodes(root)
|
|
303
|
+
if node.kind in _AGENT_KINDS
|
|
304
|
+
for check in _CHECKS
|
|
305
|
+
for finding in check(node)
|
|
306
|
+
]
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
# Copyright (c) 2026 epicodic
|
|
3
|
+
"""Full analysis of one session: adapter parse, roll-ups and heuristics."""
|
|
4
|
+
|
|
5
|
+
from agentprof.adapters.base import AgentAdapter, SessionRef
|
|
6
|
+
from agentprof.analysis.call_context import annotate_call_context
|
|
7
|
+
from agentprof.analysis.evidence_findings import find_evidence_findings
|
|
8
|
+
from agentprof.analysis.execution import resolve_execution_links
|
|
9
|
+
from agentprof.analysis.heuristics import find_findings
|
|
10
|
+
from agentprof.analysis.rollup import roll_up
|
|
11
|
+
from agentprof.model import Finding, Node, Session, iter_nodes
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def attach_findings(root: Node, findings: list[Finding]) -> None:
|
|
15
|
+
"""Append each finding to the `findings` of the node it names."""
|
|
16
|
+
nodes_by_id = {node.node_id: node for node in iter_nodes(root)}
|
|
17
|
+
for finding in findings:
|
|
18
|
+
node = nodes_by_id.get(finding.node_id)
|
|
19
|
+
if node is None:
|
|
20
|
+
raise ValueError(f"finding {finding.heuristic_id} references unknown node {finding.node_id!r}")
|
|
21
|
+
node.findings.append(finding)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def analyze(adapter: AgentAdapter, ref: SessionRef) -> Session:
|
|
25
|
+
"""Parse `ref` with `adapter`, fill derived metrics, and attach waste findings."""
|
|
26
|
+
session = adapter.analyze(ref)
|
|
27
|
+
resolve_execution_links(session.root)
|
|
28
|
+
roll_up(session.root)
|
|
29
|
+
annotate_call_context(session.root)
|
|
30
|
+
attach_findings(session.root, [*find_findings(session.root), *find_evidence_findings(session.root)])
|
|
31
|
+
return session
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
# Copyright (c) 2026 epicodic
|
|
3
|
+
"""Generic roll-ups over a neutral tree: tokens, time spans, cost, tool call counts and context.
|
|
4
|
+
|
|
5
|
+
Adapters set only what they know; these rules fill in the rest the same way for every agent.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from collections.abc import Callable
|
|
9
|
+
from itertools import pairwise
|
|
10
|
+
|
|
11
|
+
from agentprof.model import (
|
|
12
|
+
CostMetric,
|
|
13
|
+
LlmCall,
|
|
14
|
+
Metric,
|
|
15
|
+
Node,
|
|
16
|
+
NodeKind,
|
|
17
|
+
Provenance,
|
|
18
|
+
Tokens,
|
|
19
|
+
context_size,
|
|
20
|
+
sum_costs,
|
|
21
|
+
worst_provenance,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
# Node kinds that own LLM calls and therefore tokens and cost; tool nodes and the session root never do.
|
|
25
|
+
_WORKER_KINDS = (NodeKind.TURN, NodeKind.AGENT)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _sum_metrics(metrics: list[Metric]) -> Metric:
|
|
29
|
+
available = [metric for metric in metrics if metric.value is not None]
|
|
30
|
+
if not available:
|
|
31
|
+
return Metric.not_available()
|
|
32
|
+
provenance = worst_provenance(*(metric.provenance for metric in available))
|
|
33
|
+
if len(available) < len(metrics):
|
|
34
|
+
provenance = worst_provenance(provenance, Provenance.ESTIMATED)
|
|
35
|
+
return Metric(value=sum(metric.number() for metric in available), provenance=provenance)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def sum_tokens(tokens: list[Tokens]) -> Tokens:
|
|
39
|
+
"""Add token counts field by field."""
|
|
40
|
+
return Tokens(
|
|
41
|
+
input=_sum_metrics([t.input for t in tokens]),
|
|
42
|
+
output=_sum_metrics([t.output for t in tokens]),
|
|
43
|
+
cache_read=_sum_metrics([t.cache_read for t in tokens]),
|
|
44
|
+
cache_write=_sum_metrics([t.cache_write for t in tokens]),
|
|
45
|
+
cache_write_5m=_sum_metrics([t.cache_write_5m for t in tokens]),
|
|
46
|
+
cache_write_1h=_sum_metrics([t.cache_write_1h for t in tokens]),
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _call_end(call: LlmCall) -> Metric:
|
|
51
|
+
if call.start.value is None:
|
|
52
|
+
return Metric.not_available()
|
|
53
|
+
if call.duration.value is None:
|
|
54
|
+
return Metric(value=call.start.value, provenance=worst_provenance(call.start.provenance, Provenance.ESTIMATED))
|
|
55
|
+
return Metric(
|
|
56
|
+
value=call.start.number() + call.duration.number(),
|
|
57
|
+
provenance=worst_provenance(call.start.provenance, call.duration.provenance),
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _extreme(metrics: list[Metric], pick: Callable[..., Metric]) -> Metric:
|
|
62
|
+
available = [metric for metric in metrics if metric.value is not None]
|
|
63
|
+
if not available:
|
|
64
|
+
return Metric.not_available()
|
|
65
|
+
chosen = pick(available, key=Metric.number)
|
|
66
|
+
if len(available) < len(metrics):
|
|
67
|
+
return Metric(value=chosen.value, provenance=worst_provenance(chosen.provenance, Provenance.ESTIMATED))
|
|
68
|
+
return chosen
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _roll_up_span(node: Node) -> None:
|
|
72
|
+
if node.start.value is None:
|
|
73
|
+
node.start = _extreme([c.start for c in node.children] + [call.start for call in node.llm_calls], min)
|
|
74
|
+
if node.end.value is None:
|
|
75
|
+
derived_end = _extreme([c.end for c in node.children] + [_call_end(call) for call in node.llm_calls], max)
|
|
76
|
+
node.end = Metric.estimated(derived_end.number()) if derived_end.value is not None else Metric.not_available()
|
|
77
|
+
if node.duration.value is None and node.start.value is not None and node.end.value is not None:
|
|
78
|
+
node.duration = Metric(
|
|
79
|
+
value=node.end.number() - node.start.number(),
|
|
80
|
+
provenance=worst_provenance(node.start.provenance, node.end.provenance),
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _roll_up_tokens(node: Node) -> None:
|
|
85
|
+
if node.llm_calls:
|
|
86
|
+
node.tokens = sum_tokens([call.tokens for call in node.llm_calls])
|
|
87
|
+
parts = [child.tokens_total for child in node.children if child.kind in _WORKER_KINDS]
|
|
88
|
+
if node.kind in _WORKER_KINDS:
|
|
89
|
+
parts.append(node.tokens)
|
|
90
|
+
if parts:
|
|
91
|
+
node.tokens_total = sum_tokens(parts)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _subtract_children(node: Node) -> None:
|
|
95
|
+
"""`cost_own` = known `cost_total` minus the direct agent children's totals of the same unit."""
|
|
96
|
+
total = node.cost_total
|
|
97
|
+
if total.value is None:
|
|
98
|
+
return
|
|
99
|
+
agent_costs = [child.cost_total for child in node.children if child.kind is NodeKind.AGENT]
|
|
100
|
+
subtractable = [value for cost in agent_costs if cost.unit == total.unit and (value := cost.value) is not None]
|
|
101
|
+
own = total.value - sum(subtractable)
|
|
102
|
+
if own < 0:
|
|
103
|
+
return # children report more than the total (seen in real Copilot data): own cost is unknowable
|
|
104
|
+
exact = len(subtractable) == len(agent_costs) and total.provenance is Provenance.EXACT
|
|
105
|
+
node.cost_own = CostMetric(
|
|
106
|
+
value=own,
|
|
107
|
+
unit=total.unit,
|
|
108
|
+
provenance=Provenance.EXACT if exact else Provenance.ESTIMATED,
|
|
109
|
+
usd_per_unit=total.usd_per_unit,
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _roll_up_cost(node: Node) -> None:
|
|
114
|
+
if node.cost_own.value is None and node.llm_calls:
|
|
115
|
+
node.cost_own = sum_costs([call.cost for call in node.llm_calls])
|
|
116
|
+
if node.cost_total.value is not None:
|
|
117
|
+
if node.cost_own.value is None:
|
|
118
|
+
_subtract_children(node)
|
|
119
|
+
return
|
|
120
|
+
parts = [child.cost_total for child in node.children if child.kind in _WORKER_KINDS]
|
|
121
|
+
if node.llm_calls:
|
|
122
|
+
parts.append(node.cost_own)
|
|
123
|
+
if parts:
|
|
124
|
+
node.cost_total = sum_costs(parts)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
# A call whose context is smaller than this share of the previous call's marks a compaction.
|
|
128
|
+
_COMPACTION_DROP = 0.5
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _sized_calls(calls: list[LlmCall]) -> list[tuple[LlmCall, Metric]]:
|
|
132
|
+
"""Calls in the agent's context with a start and a known context size, in time order."""
|
|
133
|
+
sized: list[tuple[LlmCall, Metric]] = []
|
|
134
|
+
for call in calls:
|
|
135
|
+
size = context_size(call.tokens)
|
|
136
|
+
if call.in_context and call.start.value is not None and size.value is not None:
|
|
137
|
+
sized.append((call, size))
|
|
138
|
+
return sorted(sized, key=lambda item: item[0].start.number())
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _drops(sized: list[tuple[LlmCall, Metric]]) -> list[LlmCall]:
|
|
142
|
+
"""Calls whose context is below `_COMPACTION_DROP` of the previous call's: the first call after a compaction."""
|
|
143
|
+
return [
|
|
144
|
+
later for (_, before), (later, after) in pairwise(sized) if after.number() < before.number() * _COMPACTION_DROP
|
|
145
|
+
]
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _roll_up_main_agent_context(session: Node) -> None:
|
|
149
|
+
"""The main agent's context runs through all turns, so a compaction may take effect at a turn boundary.
|
|
150
|
+
|
|
151
|
+
The session node reports the whole context: the largest peak and every compaction of its turns.
|
|
152
|
+
Its calls stay on the turns, so tokens and cost are not counted twice.
|
|
153
|
+
"""
|
|
154
|
+
turns = [child for child in session.children if child.kind is NodeKind.TURN]
|
|
155
|
+
owners = {id(call): turn for turn in turns for call in turn.llm_calls}
|
|
156
|
+
recorded = {id(turn) for turn in turns if turn.compactions}
|
|
157
|
+
for later in _drops(_sized_calls([call for turn in turns for call in turn.llm_calls])):
|
|
158
|
+
turn = owners[id(later)]
|
|
159
|
+
if id(turn) not in recorded:
|
|
160
|
+
turn.compactions.append(Metric.estimated(later.start.number()))
|
|
161
|
+
peaks = [turn.context_peak for turn in turns if turn.context_peak.value is not None]
|
|
162
|
+
if peaks:
|
|
163
|
+
session.context_peak = max(peaks, key=Metric.number)
|
|
164
|
+
session.compactions = sorted(
|
|
165
|
+
(compaction for turn in turns for compaction in turn.compactions if compaction.value is not None),
|
|
166
|
+
key=Metric.number,
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _roll_up_context(node: Node) -> None:
|
|
171
|
+
"""Context peak over a node's own calls; compactions detected unless the adapter recorded them.
|
|
172
|
+
|
|
173
|
+
Sub-agents own their whole context; the main agent's context spans all turns and is checked on the session.
|
|
174
|
+
"""
|
|
175
|
+
if node.kind is NodeKind.SESSION:
|
|
176
|
+
_roll_up_main_agent_context(node)
|
|
177
|
+
return
|
|
178
|
+
if node.kind not in _WORKER_KINDS:
|
|
179
|
+
return
|
|
180
|
+
sized = _sized_calls(node.llm_calls)
|
|
181
|
+
if not sized:
|
|
182
|
+
return
|
|
183
|
+
node.context_peak = max((size for _, size in sized), key=Metric.number)
|
|
184
|
+
if node.kind is NodeKind.AGENT and not node.compactions:
|
|
185
|
+
node.compactions = [Metric.estimated(later.start.number()) for later in _drops(sized)]
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def roll_up(node: Node) -> None:
|
|
189
|
+
"""Fill derived metrics bottom-up, following the design spec's "Roll-ups" section."""
|
|
190
|
+
for child in node.children:
|
|
191
|
+
roll_up(child)
|
|
192
|
+
node.tool_call_count = Metric.exact(len(node.children))
|
|
193
|
+
_roll_up_tokens(node)
|
|
194
|
+
_roll_up_span(node)
|
|
195
|
+
_roll_up_cost(node)
|
|
196
|
+
_roll_up_context(node)
|
agentprof/cli.py
ADDED
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
# Copyright (c) 2026 epicodic
|
|
3
|
+
"""The `agentprof` command: start the registry and the local server, and open the browser."""
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import contextlib
|
|
7
|
+
import ipaddress
|
|
8
|
+
import socket
|
|
9
|
+
import sys
|
|
10
|
+
import webbrowser
|
|
11
|
+
from collections.abc import Callable
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from types import FrameType
|
|
14
|
+
from urllib.parse import quote
|
|
15
|
+
|
|
16
|
+
import uvicorn
|
|
17
|
+
from fastapi import FastAPI
|
|
18
|
+
|
|
19
|
+
from agentprof.adapters import load_adapters
|
|
20
|
+
from agentprof.adapters.base import AdapterConfig
|
|
21
|
+
from agentprof.pricing import PriceTable
|
|
22
|
+
from agentprof.registry import Registry
|
|
23
|
+
from agentprof.server.app import create_app
|
|
24
|
+
|
|
25
|
+
_DEFAULT_HOST = "127.0.0.1"
|
|
26
|
+
_DEFAULT_PORT = 8765
|
|
27
|
+
_WILDCARD_HOSTS = ("0.0.0.0", "::")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
31
|
+
parser = argparse.ArgumentParser(prog="agentprof", description="Analyse AI coding agent sessions in the browser.")
|
|
32
|
+
parser.add_argument("session", nargs="?", help="session id ('<agent>:<id>' or a bare id) or session file to open")
|
|
33
|
+
parser.add_argument("--host", default=_DEFAULT_HOST, help="interface to bind (default: %(default)s)")
|
|
34
|
+
parser.add_argument("--port", type=int, default=_DEFAULT_PORT, help="port to bind (default: %(default)s)")
|
|
35
|
+
parser.add_argument("--no-browser", action="store_true", help="do not open a browser")
|
|
36
|
+
parser.add_argument("--copilot-root", type=Path, help="VS Code workspaceStorage directory")
|
|
37
|
+
parser.add_argument("--claude-root", type=Path, help="Claude Code projects directory")
|
|
38
|
+
parser.add_argument("--codex-root", type=Path, help="Codex CLI sessions directory")
|
|
39
|
+
parser.add_argument("--pricing", type=Path, help="JSON price table replacing the bundled one")
|
|
40
|
+
return parser
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def adapter_config(args: argparse.Namespace) -> AdapterConfig:
|
|
44
|
+
roots: dict[str, Path] = {}
|
|
45
|
+
if args.copilot_root is not None:
|
|
46
|
+
roots["copilot-vscode"] = args.copilot_root
|
|
47
|
+
if args.claude_root is not None:
|
|
48
|
+
roots["claude-code"] = args.claude_root
|
|
49
|
+
if args.codex_root is not None:
|
|
50
|
+
roots["codex"] = args.codex_root
|
|
51
|
+
return AdapterConfig(roots=roots, pricing_file=args.pricing)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def is_loopback(host: str) -> bool:
|
|
55
|
+
if host == "localhost":
|
|
56
|
+
return True
|
|
57
|
+
try:
|
|
58
|
+
return ipaddress.ip_address(host).is_loopback
|
|
59
|
+
except ValueError:
|
|
60
|
+
return False
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def browser_url(host: str, port: int, session_id: str | None) -> str:
|
|
64
|
+
shown_host = "localhost" if host in _WILDCARD_HOSTS else host
|
|
65
|
+
base = f"http://{shown_host}:{port}/"
|
|
66
|
+
return f"{base}sessions/{quote(session_id, safe='')}" if session_id is not None else base
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class _Server(uvicorn.Server):
|
|
70
|
+
"""A server that calls `on_started` once it accepts connections and ends open event streams on the first Ctrl+C.
|
|
71
|
+
|
|
72
|
+
Without the latter, the shutdown would wait for the browser to disconnect.
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
def __init__(self, config: uvicorn.Config, end_streams: Callable[[], None], on_started: Callable[[], None]) -> None:
|
|
76
|
+
super().__init__(config)
|
|
77
|
+
self._end_streams = end_streams
|
|
78
|
+
self._on_started = on_started
|
|
79
|
+
|
|
80
|
+
async def startup(self, sockets: list[socket.socket] | None = None) -> None:
|
|
81
|
+
await super().startup(sockets)
|
|
82
|
+
if not self.should_exit:
|
|
83
|
+
self._on_started()
|
|
84
|
+
|
|
85
|
+
def handle_exit(self, sig: int, frame: FrameType | None) -> None:
|
|
86
|
+
self._end_streams()
|
|
87
|
+
super().handle_exit(sig, frame)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _serve(app: FastAPI, host: str, port: int, on_started: Callable[[], None]) -> None:
|
|
91
|
+
config = uvicorn.Config(app, host=host, port=port, log_level="warning")
|
|
92
|
+
with contextlib.suppress(KeyboardInterrupt): # uvicorn re-raises the signal that stopped it, as `uvicorn.run` does
|
|
93
|
+
_Server(config, app.state.end_streams, on_started).run()
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def main(
|
|
97
|
+
argv: list[str] | None = None,
|
|
98
|
+
*,
|
|
99
|
+
serve: Callable[[FastAPI, str, int, Callable[[], None]], None] = _serve,
|
|
100
|
+
open_browser: Callable[[str], object] = webbrowser.open,
|
|
101
|
+
) -> int:
|
|
102
|
+
"""Run agentprof until interrupted; returns the process exit code."""
|
|
103
|
+
args = build_parser().parse_args(argv)
|
|
104
|
+
try:
|
|
105
|
+
registry = Registry(load_adapters(adapter_config(args)))
|
|
106
|
+
pricing = PriceTable.load(args.pricing)
|
|
107
|
+
except (OSError, ValueError) as error: # e.g. an unreadable or invalid --pricing file
|
|
108
|
+
print(f"error: {error}", file=sys.stderr)
|
|
109
|
+
return 2
|
|
110
|
+
registry.refresh()
|
|
111
|
+
|
|
112
|
+
session_id: str | None = None
|
|
113
|
+
if args.session is not None:
|
|
114
|
+
try:
|
|
115
|
+
session_id = registry.resolve(args.session)
|
|
116
|
+
except LookupError as error:
|
|
117
|
+
print(f"error: {error}", file=sys.stderr)
|
|
118
|
+
return 2
|
|
119
|
+
|
|
120
|
+
if not is_loopback(args.host):
|
|
121
|
+
print(
|
|
122
|
+
f"warning: serving on {args.host} without authentication; anyone who can reach it can read your sessions",
|
|
123
|
+
file=sys.stderr,
|
|
124
|
+
)
|
|
125
|
+
url = browser_url(args.host, args.port, session_id)
|
|
126
|
+
|
|
127
|
+
def _on_started() -> None:
|
|
128
|
+
if not args.no_browser:
|
|
129
|
+
open_browser(url)
|
|
130
|
+
|
|
131
|
+
app = create_app(registry, pricing=pricing)
|
|
132
|
+
print(f"agentprof: {url}", flush=True)
|
|
133
|
+
registry.start()
|
|
134
|
+
try:
|
|
135
|
+
serve(app, args.host, args.port, _on_started)
|
|
136
|
+
except OSError as error: # e.g. the port is already in use
|
|
137
|
+
print(f"error: cannot serve on {args.host}:{args.port}: {error}", file=sys.stderr)
|
|
138
|
+
return 2
|
|
139
|
+
finally:
|
|
140
|
+
registry.stop()
|
|
141
|
+
return 0
|