agentprof 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. agentprof/__init__.py +7 -0
  2. agentprof/adapters/__init__.py +15 -0
  3. agentprof/adapters/base.py +79 -0
  4. agentprof/adapters/claude_code/__init__.py +3 -0
  5. agentprof/adapters/claude_code/adapter.py +133 -0
  6. agentprof/adapters/claude_code/discovery.py +79 -0
  7. agentprof/adapters/claude_code/prompts.py +44 -0
  8. agentprof/adapters/claude_code/tools.py +102 -0
  9. agentprof/adapters/claude_code/transcript.py +264 -0
  10. agentprof/adapters/claude_code/tree.py +492 -0
  11. agentprof/adapters/codex/__init__.py +3 -0
  12. agentprof/adapters/codex/adapter.py +164 -0
  13. agentprof/adapters/codex/discovery.py +117 -0
  14. agentprof/adapters/codex/prompts.py +16 -0
  15. agentprof/adapters/codex/rollout.py +328 -0
  16. agentprof/adapters/codex/tools.py +97 -0
  17. agentprof/adapters/codex/tree.py +311 -0
  18. agentprof/adapters/copilot_vscode/__init__.py +3 -0
  19. agentprof/adapters/copilot_vscode/adapter.py +166 -0
  20. agentprof/adapters/copilot_vscode/debuglog.py +136 -0
  21. agentprof/adapters/copilot_vscode/discovery.py +153 -0
  22. agentprof/adapters/copilot_vscode/session.py +281 -0
  23. agentprof/adapters/copilot_vscode/tools.py +127 -0
  24. agentprof/adapters/copilot_vscode/transcript.py +176 -0
  25. agentprof/adapters/copilot_vscode/tree.py +325 -0
  26. agentprof/adapters/execution.py +64 -0
  27. agentprof/adapters/timestamps.py +13 -0
  28. agentprof/adapters/turns.py +24 -0
  29. agentprof/analysis/__init__.py +3 -0
  30. agentprof/analysis/agent_summary.py +161 -0
  31. agentprof/analysis/call_context.py +101 -0
  32. agentprof/analysis/evidence_findings.py +151 -0
  33. agentprof/analysis/execution.py +39 -0
  34. agentprof/analysis/heuristics.py +306 -0
  35. agentprof/analysis/pipeline.py +31 -0
  36. agentprof/analysis/rollup.py +196 -0
  37. agentprof/cli.py +141 -0
  38. agentprof/model.py +341 -0
  39. agentprof/pricing.json +23 -0
  40. agentprof/pricing.py +120 -0
  41. agentprof/registry.py +304 -0
  42. agentprof/server/__init__.py +3 -0
  43. agentprof/server/app.py +399 -0
  44. agentprof/server/openapi.py +31 -0
  45. agentprof/server/pricing_schema.py +40 -0
  46. agentprof/server/schemas.py +579 -0
  47. agentprof/server/static/assets/index-C4HEqUVf.css +1 -0
  48. agentprof/server/static/assets/index-ZsPn1H3d.js +58 -0
  49. agentprof/server/static/index.html +14 -0
  50. agentprof-0.1.0.dist-info/METADATA +177 -0
  51. agentprof-0.1.0.dist-info/RECORD +54 -0
  52. agentprof-0.1.0.dist-info/WHEEL +4 -0
  53. agentprof-0.1.0.dist-info/entry_points.txt +8 -0
  54. agentprof-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,492 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2026 epicodic
3
+ """Build the neutral node tree from a Claude Code main transcript and its subagent transcripts."""
4
+
5
+ import json
6
+ import re
7
+ from typing import Any
8
+
9
+ from agentprof.adapters.claude_code.discovery import Subagent
10
+ from agentprof.adapters.claude_code.prompts import first_line, prompt_topic
11
+ from agentprof.adapters.claude_code.tools import tool_info
12
+ from agentprof.adapters.claude_code.transcript import AssistantMessage, Prompt, ToolResult, ToolUse, Transcript
13
+ from agentprof.adapters.execution import tool_execution_events
14
+ from agentprof.adapters.turns import split_by_turn
15
+ from agentprof.model import CostMetric, ExecutionEvent, LlmCall, Metric, Node, NodeKind, Tokens, ToolCategory
16
+ from agentprof.pricing import PriceTable, Usage
17
+
18
+ _ASK_USER_TOOL = "AskUserQuestion"
19
+ _TOPIC_KEYS = ("description", "file_path", "notebook_path", "command", "pattern", "url", "query")
20
+ _TOPIC_LENGTH = 200
21
+ _SYNTHETIC_TURN_ID = "turn-1"
22
+ # Claude Code's own placeholder messages (e.g. an interrupted request); they carry no real context.
23
+ _SYNTHETIC_MODEL = "<synthetic>"
24
+ _ASYNC_AGENT_LAUNCH = "Async agent launched successfully."
25
+ _SEND_MESSAGE_TOOL = "sendmessage"
26
+ _MESSAGE_PREVIEW_LENGTH = 80
27
+ _TARGET_PREVIEW_LENGTH = 60
28
+
29
+
30
+ def _count(source: object, key: str) -> int:
31
+ value = source.get(key) if isinstance(source, dict) else None
32
+ return value if isinstance(value, int) and not isinstance(value, bool) else 0
33
+
34
+
35
+ def usage_of(usage: dict[str, Any]) -> Usage:
36
+ """Token counts of one call; cache writes without a 5m/1h split are counted at the 5-minute rate."""
37
+ cache_write = _count(usage, "cache_creation_input_tokens")
38
+ one_hour = _count(usage.get("cache_creation"), "ephemeral_1h_input_tokens")
39
+ return Usage(
40
+ input=_count(usage, "input_tokens"),
41
+ output=_count(usage, "output_tokens"),
42
+ cache_read=_count(usage, "cache_read_input_tokens"),
43
+ cache_write_5m=max(0, cache_write - one_hour),
44
+ cache_write_1h=one_hour,
45
+ )
46
+
47
+
48
+ def message_cost(message: AssistantMessage, prices: PriceTable) -> CostMetric:
49
+ """Estimated cost of one assistant message; `n/a` without usage or a price for its model."""
50
+ if not message.usage:
51
+ return CostMetric.not_available()
52
+ return prices.cost(message.model, usage_of(message.usage))
53
+
54
+
55
+ def _llm_call(message: AssistantMessage, prices: PriceTable, source_stream_id: str | None = None) -> LlmCall:
56
+ call = LlmCall(
57
+ start=Metric.exact(message.start_ms),
58
+ model=message.model,
59
+ in_context=message.model != _SYNTHETIC_MODEL,
60
+ call_id=message.message_id,
61
+ source_request_id=message.message_id,
62
+ source_order=message.source_order,
63
+ source_stream_id=source_stream_id,
64
+ timing_basis="assistant_message",
65
+ price_prefix=prices.matching_prefix(message.model),
66
+ )
67
+ if not message.usage:
68
+ return call
69
+ usage = usage_of(message.usage)
70
+ call.tokens = Tokens(
71
+ input=Metric.exact(usage.input),
72
+ output=Metric.exact(usage.output),
73
+ cache_read=Metric.exact(usage.cache_read),
74
+ cache_write=Metric.exact(usage.cache_write_5m + usage.cache_write_1h),
75
+ )
76
+ split = message.usage.get("cache_creation")
77
+ if (
78
+ isinstance(split, dict)
79
+ and isinstance(split.get("ephemeral_5m_input_tokens"), int)
80
+ and isinstance(split.get("ephemeral_1h_input_tokens"), int)
81
+ and not isinstance(split["ephemeral_5m_input_tokens"], bool)
82
+ and not isinstance(split["ephemeral_1h_input_tokens"], bool)
83
+ and split["ephemeral_5m_input_tokens"] + split["ephemeral_1h_input_tokens"]
84
+ == _count(message.usage, "cache_creation_input_tokens")
85
+ ):
86
+ call.tokens.cache_write_5m = Metric.exact(split["ephemeral_5m_input_tokens"])
87
+ call.tokens.cache_write_1h = Metric.exact(split["ephemeral_1h_input_tokens"])
88
+ call.cost = message_cost(message, prices)
89
+ call.cost_parts = prices.cost_parts(message.model, usage)
90
+ if call.tokens.cache_write_5m.value is None:
91
+ call.cost_parts["cache_write_5m"] = CostMetric.not_available()
92
+ call.cost_parts["cache_write_1h"] = CostMetric.not_available()
93
+ return call
94
+
95
+
96
+ def _topic(tool_use: ToolUse) -> str:
97
+ if tool_use.name.casefold() == _SEND_MESSAGE_TOOL:
98
+ target = tool_use.input.get("target") or tool_use.input.get("to") or tool_use.input.get("recipient")
99
+ message = tool_use.input.get("message") or tool_use.input.get("content")
100
+ target_preview = first_line(target, _TARGET_PREVIEW_LENGTH) if isinstance(target, str) else ""
101
+ message_preview = first_line(message, _MESSAGE_PREVIEW_LENGTH) if isinstance(message, str) else ""
102
+ topic = f"SendMessage to {target_preview}" if target_preview else "SendMessage"
103
+ return f"{topic}: {message_preview}" if message_preview else topic
104
+ for key in _TOPIC_KEYS:
105
+ value = tool_use.input.get(key)
106
+ if isinstance(value, str) and value:
107
+ return f"{tool_use.name}: {first_line(value, _TOPIC_LENGTH)}"
108
+ return tool_use.name
109
+
110
+
111
+ def _resume_agent_id(text: str, target: str | None) -> str | None:
112
+ """Return the exact harness ID confirmed by a SendMessage result, when known."""
113
+ match = re.search(r"(?:^|\n)Resuming agent ([A-Za-z0-9_-]+)(?:\s|$)", text)
114
+ if match is not None:
115
+ return match.group(1)
116
+ try:
117
+ result = json.loads(text)
118
+ except ValueError:
119
+ return None
120
+ if isinstance(result, dict) and (result.get("resumed") is True or result.get("status") == "resumed"):
121
+ return target
122
+ if isinstance(result, dict) and result.get("success") is True:
123
+ resumed_id = result.get("resumedAgentId")
124
+ if isinstance(resumed_id, str) and resumed_id:
125
+ return resumed_id
126
+ return None
127
+
128
+
129
+ def _tool_node(tool_use: ToolUse, transcript: Transcript) -> Node:
130
+ node = Node(
131
+ node_id=tool_use.tool_use_id,
132
+ kind=NodeKind.TOOL,
133
+ topic=_topic(tool_use),
134
+ tool=tool_info(tool_use.name, tool_use.input),
135
+ start=Metric.exact(tool_use.start_ms),
136
+ )
137
+ result = transcript.tool_results.get(tool_use.tool_use_id)
138
+ if node.tool is not None and tool_use.name.casefold() == _SEND_MESSAGE_TOOL:
139
+ target = tool_use.input.get("target") or tool_use.input.get("to") or tool_use.input.get("recipient")
140
+ node.tool.target_agent_id = target if isinstance(target, str) and target else None
141
+ if result is not None and not result.is_error:
142
+ resumed_id = _resume_agent_id(result.text, node.tool.target_agent_id)
143
+ node.tool.is_resume = resumed_id is not None
144
+ if resumed_id is not None:
145
+ node.tool.target_agent_id = resumed_id
146
+ if result is not None:
147
+ node.end = Metric.exact(result.end_ms)
148
+ node.duration = Metric.exact(result.end_ms - tool_use.start_ms)
149
+ node.success = not result.is_error
150
+ node.result = result.text
151
+ return node
152
+
153
+
154
+ def _last_activity_ms(transcript: Transcript) -> int | None:
155
+ times = [message.start_ms for message in transcript.messages]
156
+ times += [result.end_ms for result in transcript.tool_results.values()]
157
+ return max(times, default=None)
158
+
159
+
160
+ class _TreeBuilder:
161
+ """Creates one tool node per tool call across all transcripts, then links them into agents.
162
+
163
+ Every tool call of every transcript is indexed up front, so `attach` can find a spawning `Agent` call in any
164
+ file. `attach` turns that tool node into an agent node in place; the node may already sit in another agent's
165
+ `children`, which is intended: the tree shares these node objects instead of copying them.
166
+ """
167
+
168
+ def __init__(self, transcripts: list[Transcript], prices: PriceTable) -> None:
169
+ self._prices = prices
170
+ self._nodes: dict[str, Node] = {}
171
+ self._placed: set[str] = set()
172
+ self._agents_by_harness_id: dict[str, list[Node]] = {}
173
+ self._transcript_owners: dict[int, Node] = {}
174
+ self.orphans: list[Node] = []
175
+ for transcript in transcripts:
176
+ for tool_use in transcript.tool_uses:
177
+ self._nodes.setdefault(tool_use.tool_use_id, _tool_node(tool_use, transcript))
178
+
179
+ def take(self, transcript: Transcript) -> list[Node]:
180
+ """The tool nodes of `transcript` that no other agent has claimed yet."""
181
+ taken: list[Node] = []
182
+ for tool_use in transcript.tool_uses:
183
+ if tool_use.tool_use_id not in self._placed:
184
+ self._placed.add(tool_use.tool_use_id)
185
+ taken.append(self._nodes[tool_use.tool_use_id])
186
+ return taken
187
+
188
+ def calls(self, transcript: Transcript) -> list[LlmCall]:
189
+ return [_llm_call(message, self._prices, transcript.source_stream_id) for message in transcript.messages]
190
+
191
+ def attach(self, subagent: Subagent) -> None:
192
+ """Turn the spawning tool node into the subagent's agent node, or create an orphan agent node."""
193
+ meta, transcript = subagent.meta, subagent.transcript
194
+ own_ids = {tool_use.tool_use_id for tool_use in transcript.tool_uses}
195
+ node = self._nodes.get(meta.tool_use_id) if meta.tool_use_id and meta.tool_use_id not in own_ids else None
196
+ if node is None:
197
+ node = Node(node_id=f"agent-{meta.agent_id}", kind=NodeKind.AGENT, topic=meta.agent_id)
198
+ if transcript.first_timestamp_ms is not None:
199
+ node.start = Metric.exact(transcript.first_timestamp_ms)
200
+ self.orphans.append(node)
201
+ elif node.result.startswith(_ASYNC_AGENT_LAUNCH):
202
+ # This is an acknowledgement that a background agent started, not its terminal result.
203
+ node.end = Metric.not_available()
204
+ node.duration = Metric.not_available()
205
+ node.kind = NodeKind.AGENT
206
+ self._transcript_owners[id(transcript)] = node
207
+ node.agent_uuid = meta.agent_id
208
+ self._agents_by_harness_id.setdefault(meta.agent_id, []).append(node)
209
+ if meta.description:
210
+ node.topic = meta.description
211
+ node.llm_calls = self.calls(transcript)
212
+ node.llm_call_count = Metric.exact(len(node.llm_calls))
213
+ node.compactions = [Metric.exact(ms) for ms in transcript.compactions_ms]
214
+ node.execution_events.extend(
215
+ ExecutionEvent(
216
+ kind="compaction",
217
+ subject_node_id=node.node_id,
218
+ start=Metric.exact(ms),
219
+ source_stream_id=transcript.source_stream_id,
220
+ )
221
+ for ms in transcript.compactions_ms
222
+ )
223
+ node.model = next((call.model for call in node.llm_calls if call.model), None) or meta.model
224
+ if transcript.prompts:
225
+ node.prompt = transcript.prompts[0].text
226
+ node.children = self.take(transcript)
227
+ last_ms = _last_activity_ms(transcript)
228
+ if last_ms is not None and node.end.value is not None and last_ms > node.end.value:
229
+ node.end = Metric.exact(last_ms)
230
+ node.duration = Metric.exact(last_ms - node.start.number())
231
+
232
+ def link_resumes(self) -> None:
233
+ """Link confirmed messages only when their exact harness target identifies one agent."""
234
+ for node in self._nodes.values():
235
+ info = node.tool
236
+ if info is None or not info.is_resume or info.target_agent_id is None:
237
+ continue
238
+ matches = self._agents_by_harness_id.get(info.target_agent_id, [])
239
+ if len(matches) != 1:
240
+ continue
241
+ agent = matches[0]
242
+ info.linked_agent_node_id = agent.node_id
243
+ agent.resume_times.append(node.start)
244
+ agent.execution_events.append(
245
+ ExecutionEvent(kind="resume", subject_node_id=agent.node_id, start=node.start)
246
+ )
247
+
248
+
249
+ def _item_start(item: Node | LlmCall) -> float | None:
250
+ return item.start.value
251
+
252
+
253
+ def _ms_start(ms: int) -> float:
254
+ return float(ms)
255
+
256
+
257
+ def _turn(
258
+ prompt: Prompt,
259
+ children: list[Node],
260
+ calls: list[LlmCall],
261
+ compactions: list[int],
262
+ source_stream_id: str | None,
263
+ ) -> Node:
264
+ waits = [
265
+ child.duration.number()
266
+ for child in children
267
+ if child.tool is not None and child.tool.native_id == _ASK_USER_TOOL and child.duration.value is not None
268
+ ]
269
+ return Node(
270
+ node_id=prompt.uuid,
271
+ kind=NodeKind.TURN,
272
+ topic=prompt_topic(prompt.text),
273
+ prompt=prompt.text,
274
+ model=next((call.model for call in calls if call.model), None),
275
+ start=Metric.exact(prompt.start_ms),
276
+ children=children,
277
+ llm_calls=calls,
278
+ llm_call_count=Metric.exact(len(calls)),
279
+ user_wait=Metric.estimated(sum(waits)),
280
+ compactions=[Metric.exact(ms) for ms in compactions],
281
+ execution_events=[
282
+ ExecutionEvent(
283
+ kind="compaction",
284
+ subject_node_id=prompt.uuid,
285
+ start=Metric.exact(ms),
286
+ source_stream_id=source_stream_id,
287
+ )
288
+ for ms in compactions
289
+ ],
290
+ )
291
+
292
+
293
+ def _parent_nodes(root: Node) -> dict[int, Node]:
294
+ parents: dict[int, Node] = {}
295
+
296
+ def visit(parent: Node) -> None:
297
+ for child in parent.children:
298
+ parents[id(child)] = parent
299
+ visit(child)
300
+
301
+ visit(root)
302
+ return parents
303
+
304
+
305
+ def _turn_at(turns: list[Node], timestamp_ms: int) -> Node | None:
306
+ candidates = [turn for turn in turns if turn.start.value is not None and turn.start.number() <= timestamp_ms]
307
+ return candidates[-1] if candidates else (turns[0] if turns else None)
308
+
309
+
310
+ def _emit_tool_events(
311
+ root: Node,
312
+ transcripts: list[Transcript],
313
+ builder: _TreeBuilder,
314
+ unique_ids: set[str],
315
+ ) -> None:
316
+ parents = _parent_nodes(root)
317
+ turns = [node for node in root.children if node.kind is NodeKind.TURN]
318
+
319
+ def owner_for(transcript: Transcript, timestamp_ms: int, node: Node | None) -> Node | None:
320
+ if node is not None and id(node) in parents:
321
+ return parents[id(node)]
322
+ if any(transcript is candidate for candidate in transcripts[1:]):
323
+ return builder._transcript_owners.get(id(transcript))
324
+ return _turn_at(turns, timestamp_ms) or root
325
+
326
+ for transcript in transcripts:
327
+ record_list = transcript.tool_result_records or list(transcript.tool_results.items())
328
+ records_by_id: dict[str, list[tuple[int, ToolResult]]] = {}
329
+ for record_index, (invocation_id, result) in enumerate(record_list):
330
+ records_by_id.setdefault(invocation_id, []).append((record_index, result))
331
+
332
+ used_result_records: set[int] = set()
333
+ for tool_use in transcript.tool_uses:
334
+ invocation_id = tool_use.tool_use_id
335
+ node = builder._nodes.get(invocation_id)
336
+ subject_node_id = node.node_id if node is not None and invocation_id in unique_ids else None
337
+ results = records_by_id.get(invocation_id, [])
338
+ last_result = results[-1][1] if results else None
339
+ start = Metric.exact(tool_use.start_ms)
340
+ end = Metric.exact(last_result.end_ms) if last_result is not None else Metric.not_available()
341
+ owner = owner_for(transcript, tool_use.start_ms, node if subject_node_id is not None else None)
342
+ if owner is None:
343
+ continue
344
+ source_tool = tool_info(tool_use.name, tool_use.input)
345
+ delegation = source_tool.category is ToolCategory.SUBAGENT
346
+ start_event = tool_execution_events(
347
+ invocation_id=invocation_id,
348
+ subject_node_id=subject_node_id,
349
+ start=start,
350
+ end=end,
351
+ result_recorded=False,
352
+ delegation=delegation,
353
+ start_order=tool_use.source_order,
354
+ source_stream_id=transcript.source_stream_id,
355
+ request_id=tool_use.request_message_id,
356
+ )[0]
357
+ owner.execution_events.append(start_event)
358
+ for record_index, result in results:
359
+ _, result_event = tool_execution_events(
360
+ invocation_id=invocation_id,
361
+ subject_node_id=subject_node_id,
362
+ start=start,
363
+ end=Metric.exact(result.end_ms),
364
+ result_recorded=True,
365
+ success=not result.is_error,
366
+ delegation=delegation,
367
+ start_order=tool_use.source_order,
368
+ result_order=result.source_order,
369
+ source_stream_id=transcript.source_stream_id,
370
+ request_id=tool_use.request_message_id,
371
+ next_id=result.next_message_id,
372
+ )
373
+ owner.execution_events.append(result_event)
374
+ used_result_records.add(record_index)
375
+
376
+ for record_index, (invocation_id, result) in enumerate(record_list):
377
+ if record_index in used_result_records:
378
+ continue
379
+ owner = owner_for(transcript, result.end_ms, None)
380
+ if owner is None:
381
+ continue
382
+ result_event = tool_execution_events(
383
+ invocation_id=invocation_id,
384
+ subject_node_id=None,
385
+ start=Metric.not_available(),
386
+ end=Metric.exact(result.end_ms),
387
+ result_recorded=True,
388
+ success=not result.is_error,
389
+ result_order=result.source_order,
390
+ source_stream_id=transcript.source_stream_id,
391
+ next_id=result.next_message_id,
392
+ )[1]
393
+ owner.execution_events.append(result_event)
394
+
395
+ def sort_events(node: Node) -> None:
396
+ node.execution_events.sort(key=lambda event: (event.source_order is None, event.source_order or 0))
397
+ for child in node.children:
398
+ sort_events(child)
399
+
400
+ sort_events(root)
401
+
402
+
403
+ def build_root(main: Transcript, subagents: list[Subagent], prices: PriceTable, title: str) -> Node:
404
+ """Build session -> turn -> agent -> tool from the main transcript and all subagent transcripts."""
405
+ builder = _TreeBuilder([main, *(subagent.transcript for subagent in subagents)], prices)
406
+ for subagent in subagents:
407
+ builder.attach(subagent)
408
+ builder.link_resumes()
409
+ main_nodes = builder.take(main) + builder.orphans
410
+ main_calls = builder.calls(main)
411
+
412
+ prompts = main.prompts
413
+ synthetic_prompt = False
414
+ if not prompts:
415
+ starts = [start for item in (*main_nodes, *main_calls) if (start := _item_start(item)) is not None]
416
+ if not starts:
417
+ root = Node(node_id="session", kind=NodeKind.SESSION, topic=title)
418
+ transcripts = [main, *(subagent.transcript for subagent in subagents)]
419
+ id_counts: dict[str, int] = {}
420
+ for transcript in transcripts:
421
+ for tool_use in transcript.tool_uses:
422
+ id_counts[tool_use.tool_use_id] = id_counts.get(tool_use.tool_use_id, 0) + 1
423
+ unique_ids = {invocation_id for invocation_id, count in id_counts.items() if count == 1}
424
+ for transcript in transcripts:
425
+ root.execution_events.extend(
426
+ ExecutionEvent(
427
+ kind="compaction", start=Metric.exact(ms), source_stream_id=transcript.source_stream_id
428
+ )
429
+ for ms in transcript.compactions_ms
430
+ )
431
+ _emit_tool_events(root, transcripts, builder, unique_ids)
432
+ return root
433
+ prompts = [Prompt(uuid=_SYNTHETIC_TURN_ID, start_ms=int(min(starts)), text="")]
434
+ synthetic_prompt = True
435
+
436
+ turn_starts = [prompt.start_ms for prompt in prompts]
437
+ node_groups = split_by_turn(main_nodes, _item_start, turn_starts)
438
+ call_groups = split_by_turn(main_calls, _item_start, turn_starts)
439
+ compaction_groups = split_by_turn(main.compactions_ms, _ms_start, turn_starts)
440
+ turns = [
441
+ _turn(prompt, nodes, calls, compactions, main.source_stream_id)
442
+ for prompt, nodes, calls, compactions in zip(prompts, node_groups, call_groups, compaction_groups, strict=True)
443
+ ]
444
+ for prompt, turn in zip(prompts, turns, strict=True):
445
+ if not synthetic_prompt and prompt in main.prompts:
446
+ turn.execution_events.append(
447
+ ExecutionEvent(
448
+ kind="user_input",
449
+ event_id=f"user-input:{prompt.uuid}",
450
+ subject_node_id=turn.node_id,
451
+ start=Metric.exact(prompt.start_ms),
452
+ source_order=prompt.source_order,
453
+ source_stream_id=main.source_stream_id,
454
+ )
455
+ )
456
+ for subagent in subagents:
457
+ owner = builder._transcript_owners.get(id(subagent.transcript))
458
+ if owner is not None:
459
+ for prompt in subagent.transcript.prompts:
460
+ owner.execution_events.append(
461
+ ExecutionEvent(
462
+ kind="user_input",
463
+ event_id=f"user-input:{prompt.uuid}",
464
+ subject_node_id=owner.node_id,
465
+ start=Metric.exact(prompt.start_ms),
466
+ source_order=prompt.source_order,
467
+ source_stream_id=subagent.transcript.source_stream_id,
468
+ )
469
+ )
470
+
471
+ visible_turns: list[Node] = []
472
+ pending_compactions: list[Metric] = []
473
+ for turn in turns:
474
+ if not turn.llm_calls and not turn.children and not turn.execution_events:
475
+ pending_compactions.extend(turn.compactions)
476
+ continue
477
+ turn.compactions = sorted([*pending_compactions, *turn.compactions], key=Metric.number)
478
+ pending_compactions.clear()
479
+ visible_turns.append(turn)
480
+ if pending_compactions and visible_turns:
481
+ visible_turns[-1].compactions = sorted(
482
+ [*visible_turns[-1].compactions, *pending_compactions], key=Metric.number
483
+ )
484
+ root = Node(node_id="session", kind=NodeKind.SESSION, topic=title, children=visible_turns)
485
+ transcripts = [main, *(subagent.transcript for subagent in subagents)]
486
+ id_counts: dict[str, int] = {}
487
+ for transcript in transcripts:
488
+ for tool_use in transcript.tool_uses:
489
+ id_counts[tool_use.tool_use_id] = id_counts.get(tool_use.tool_use_id, 0) + 1
490
+ unique_ids = {invocation_id for invocation_id, count in id_counts.items() if count == 1}
491
+ _emit_tool_events(root, transcripts, builder, unique_ids)
492
+ return root
@@ -0,0 +1,3 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2026 epicodic
3
+ """Read local Codex rollout files."""
@@ -0,0 +1,164 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2026 epicodic
3
+ """The Codex CLI rollout adapter."""
4
+
5
+ from collections import Counter
6
+ from collections.abc import Iterator
7
+ from pathlib import Path
8
+
9
+ from agentprof.adapters.base import AdapterConfig, SessionRef, SessionSummary, latest_mtime
10
+ from agentprof.adapters.codex.discovery import SessionFiles, default_sessions_root, read_metadata, session_files
11
+ from agentprof.adapters.codex.prompts import first_line
12
+ from agentprof.adapters.codex.rollout import Rollout, UsageRecord, load_rollout
13
+ from agentprof.adapters.codex.tools import is_known_tool_id
14
+ from agentprof.adapters.codex.tree import build_root
15
+ from agentprof.model import CostMetric, Diagnostics, Session, sum_costs
16
+ from agentprof.pricing import PriceTable, Usage
17
+
18
+ _TITLE_LENGTH = 80
19
+
20
+
21
+ def _merge_main_rollouts(rollouts: list[Rollout]) -> Rollout:
22
+ """Merge root continuation rollouts into one chronological main conversation."""
23
+ if not rollouts:
24
+ return Rollout()
25
+ main = rollouts[0]
26
+ for continuation in rollouts[1:]:
27
+ main.turns.extend(continuation.turns)
28
+ main.malformed_lines += continuation.malformed_lines
29
+ if continuation.last_message_ms is not None:
30
+ main.last_message_ms = max(
31
+ main.last_message_ms or continuation.last_message_ms, continuation.last_message_ms
32
+ )
33
+ main.turns.sort(key=lambda turn: turn.start_ms if turn.start_ms is not None else float("inf"))
34
+ return main
35
+
36
+
37
+ def _title(main: Rollout, fallback: str) -> str:
38
+ """Return the first root-user prompt topic, or the stable native id."""
39
+ for turn in main.turns:
40
+ if turn.prompt:
41
+ return first_line(turn.prompt, _TITLE_LENGTH) or fallback
42
+ return fallback
43
+
44
+
45
+ def _usage(usage: UsageRecord) -> Usage:
46
+ """Convert Codex token categories to the shared pricing representation."""
47
+ return Usage(
48
+ input=usage.input,
49
+ output=usage.output,
50
+ cache_read=usage.cache_read,
51
+ cache_write_5m=usage.cache_write,
52
+ cache_write_1h=0,
53
+ )
54
+
55
+
56
+ def _costs(rollouts: list[Rollout], prices: PriceTable) -> list[CostMetric]:
57
+ """Return one estimated cost for every native usage record."""
58
+ return [
59
+ prices.cost(turn.model, _usage(record.usage))
60
+ for rollout in rollouts
61
+ for turn in rollout.turns
62
+ for record in turn.token_usages
63
+ ]
64
+
65
+
66
+ class CodexAdapter:
67
+ """Reads local Codex CLI rollout sessions from ``~/.codex/sessions``."""
68
+
69
+ name = "codex"
70
+
71
+ def __init__(self, config: AdapterConfig) -> None:
72
+ override = config.roots.get(self.name)
73
+ self._root = override if override is not None else default_sessions_root(Path.home())
74
+ self._prices = PriceTable.load(config.pricing_file)
75
+
76
+ def _ref(self, files: SessionFiles) -> SessionRef:
77
+ path = files.main if files.main is not None else files.files[0]
78
+ return SessionRef(
79
+ agent=self.name,
80
+ native_id=files.session_id,
81
+ path=path,
82
+ mtime=latest_mtime(files.files),
83
+ )
84
+
85
+ def _files_for_id(self, session_id: str) -> SessionFiles | None:
86
+ return next((files for files in session_files(self._root) if files.session_id == session_id), None)
87
+
88
+ def _load(self, ref: SessionRef) -> tuple[SessionFiles, Rollout, list[Rollout]]:
89
+ files = self._files_for_id(ref.native_id)
90
+ if files is None:
91
+ metadata = read_metadata(ref.path)
92
+ if metadata is None:
93
+ raise ValueError(f"cannot read Codex rollout {ref.path}")
94
+ files = SessionFiles(metadata.session_id, ref.path, [ref.path], [metadata])
95
+ roots = [rollout.path for rollout in files.rollouts if rollout.parent_thread_id is None]
96
+ main_paths = roots or [ref.path]
97
+ main = _merge_main_rollouts([load_rollout(path) for path in main_paths])
98
+ subagents = [load_rollout(rollout.path) for rollout in files.rollouts if rollout.path not in main_paths]
99
+ return files, main, subagents
100
+
101
+ def discover(self) -> Iterator[SessionRef]:
102
+ for files in session_files(self._root):
103
+ yield self._ref(files)
104
+
105
+ def open_path(self, path: Path) -> SessionRef | None:
106
+ metadata = read_metadata(path) if path.suffix == ".jsonl" and path.is_file() else None
107
+ if metadata is None:
108
+ return None
109
+ files = self._files_for_id(metadata.session_id)
110
+ if files is not None:
111
+ return self._ref(files)
112
+ return SessionRef(agent=self.name, native_id=metadata.session_id, path=path, mtime=latest_mtime([path]))
113
+
114
+ def summarize(self, ref: SessionRef) -> SessionSummary:
115
+ files, main, subagents = self._load(ref)
116
+ rollouts = [main, *subagents]
117
+ start_ms = next((turn.start_ms for turn in main.turns if turn.start_ms is not None), None)
118
+ return SessionSummary(
119
+ id=ref.id,
120
+ agent=self.name,
121
+ title=_title(main, ref.native_id),
122
+ workspace=main.cwd,
123
+ start_ms=start_ms if start_ms is not None else ref.mtime * 1000,
124
+ end_ms=ref.mtime * 1000,
125
+ file_size=sum(path.stat().st_size for path in files.files),
126
+ last_activity_ms=max(
127
+ (timestamp for rollout in rollouts if (timestamp := rollout.last_message_ms) is not None),
128
+ default=None,
129
+ ),
130
+ cost_total=sum_costs(_costs(rollouts, self._prices)),
131
+ )
132
+
133
+ def analyze(self, ref: SessionRef) -> Session:
134
+ _, main, subagents = self._load(ref)
135
+ rollouts = [main, *subagents]
136
+ diagnostics = Diagnostics(malformed_lines=sum(rollout.malformed_lines for rollout in rollouts))
137
+ diagnostics.unknown_tool_ids = dict(
138
+ Counter(
139
+ tool.name
140
+ for rollout in rollouts
141
+ for turn in rollout.turns
142
+ for tool in turn.tools
143
+ if not is_known_tool_id(tool.name)
144
+ )
145
+ )
146
+ unpriced = sorted(
147
+ {
148
+ turn.model
149
+ for rollout in rollouts
150
+ for turn in rollout.turns
151
+ if turn.model and turn.token_usages and self._prices.price_for(turn.model) is None
152
+ }
153
+ )
154
+ diagnostics.warnings.extend(f"No price for model {model!r}: its cost is unavailable." for model in unpriced)
155
+ title = _title(main, ref.native_id)
156
+ return Session(
157
+ id=ref.id,
158
+ agent=self.name,
159
+ title=title,
160
+ workspace=main.cwd,
161
+ root=build_root(main, subagents, self._prices, title),
162
+ sources=["rollout", *(["subagents"] if subagents else [])],
163
+ diagnostics=diagnostics,
164
+ )