agentprof 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. agentprof/__init__.py +7 -0
  2. agentprof/adapters/__init__.py +15 -0
  3. agentprof/adapters/base.py +79 -0
  4. agentprof/adapters/claude_code/__init__.py +3 -0
  5. agentprof/adapters/claude_code/adapter.py +133 -0
  6. agentprof/adapters/claude_code/discovery.py +79 -0
  7. agentprof/adapters/claude_code/prompts.py +44 -0
  8. agentprof/adapters/claude_code/tools.py +102 -0
  9. agentprof/adapters/claude_code/transcript.py +264 -0
  10. agentprof/adapters/claude_code/tree.py +492 -0
  11. agentprof/adapters/codex/__init__.py +3 -0
  12. agentprof/adapters/codex/adapter.py +164 -0
  13. agentprof/adapters/codex/discovery.py +117 -0
  14. agentprof/adapters/codex/prompts.py +16 -0
  15. agentprof/adapters/codex/rollout.py +328 -0
  16. agentprof/adapters/codex/tools.py +97 -0
  17. agentprof/adapters/codex/tree.py +311 -0
  18. agentprof/adapters/copilot_vscode/__init__.py +3 -0
  19. agentprof/adapters/copilot_vscode/adapter.py +166 -0
  20. agentprof/adapters/copilot_vscode/debuglog.py +136 -0
  21. agentprof/adapters/copilot_vscode/discovery.py +153 -0
  22. agentprof/adapters/copilot_vscode/session.py +281 -0
  23. agentprof/adapters/copilot_vscode/tools.py +127 -0
  24. agentprof/adapters/copilot_vscode/transcript.py +176 -0
  25. agentprof/adapters/copilot_vscode/tree.py +325 -0
  26. agentprof/adapters/execution.py +64 -0
  27. agentprof/adapters/timestamps.py +13 -0
  28. agentprof/adapters/turns.py +24 -0
  29. agentprof/analysis/__init__.py +3 -0
  30. agentprof/analysis/agent_summary.py +161 -0
  31. agentprof/analysis/call_context.py +101 -0
  32. agentprof/analysis/evidence_findings.py +151 -0
  33. agentprof/analysis/execution.py +39 -0
  34. agentprof/analysis/heuristics.py +306 -0
  35. agentprof/analysis/pipeline.py +31 -0
  36. agentprof/analysis/rollup.py +196 -0
  37. agentprof/cli.py +141 -0
  38. agentprof/model.py +341 -0
  39. agentprof/pricing.json +23 -0
  40. agentprof/pricing.py +120 -0
  41. agentprof/registry.py +304 -0
  42. agentprof/server/__init__.py +3 -0
  43. agentprof/server/app.py +399 -0
  44. agentprof/server/openapi.py +31 -0
  45. agentprof/server/pricing_schema.py +40 -0
  46. agentprof/server/schemas.py +579 -0
  47. agentprof/server/static/assets/index-C4HEqUVf.css +1 -0
  48. agentprof/server/static/assets/index-ZsPn1H3d.js +58 -0
  49. agentprof/server/static/index.html +14 -0
  50. agentprof-0.1.0.dist-info/METADATA +177 -0
  51. agentprof-0.1.0.dist-info/RECORD +54 -0
  52. agentprof-0.1.0.dist-info/WHEEL +4 -0
  53. agentprof-0.1.0.dist-info/entry_points.txt +8 -0
  54. agentprof-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,311 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2026 epicodic
3
+ """Build neutral session trees from Codex rollout records."""
4
+
5
+ import json
6
+
7
+ from agentprof.adapters.codex.prompts import prompt_topic
8
+ from agentprof.adapters.codex.rollout import Rollout, ToolCall, Turn, UsageRecord
9
+ from agentprof.adapters.codex.tools import tool_info
10
+ from agentprof.adapters.execution import tool_execution_events
11
+ from agentprof.adapters.turns import split_by_turn
12
+ from agentprof.model import CostMetric, ExecutionEvent, LlmCall, Metric, Node, NodeKind, Tokens
13
+ from agentprof.pricing import PriceTable, Usage
14
+
15
+ _ENCRYPTED_HANDOVER = "Handover unavailable: Codex stored it encrypted."
16
+
17
+
18
+ def _arguments(raw_input: str | None) -> dict[str, object]:
19
+ """Decode native tool input when it is a JSON object, otherwise retain no structured arguments."""
20
+ if raw_input is None:
21
+ return {}
22
+ try:
23
+ arguments = json.loads(raw_input)
24
+ except json.JSONDecodeError:
25
+ return {"patch": raw_input} if "*** Update File:" in raw_input else {}
26
+ return arguments if isinstance(arguments, dict) else {}
27
+
28
+
29
+ def _usage(usage: UsageRecord) -> Usage:
30
+ """Convert a native Codex usage record to the shared price-table representation."""
31
+ return Usage(
32
+ input=usage.input,
33
+ output=usage.output,
34
+ cache_read=usage.cache_read,
35
+ cache_write_5m=usage.cache_write,
36
+ cache_write_1h=0,
37
+ )
38
+
39
+
40
+ def _llm_calls(turn: Turn, prices: PriceTable, source_stream_id: str | None = None) -> list[LlmCall]:
41
+ """Build one timed LLM call for every native usage record in a turn."""
42
+ calls: list[LlmCall] = []
43
+ for record in turn.token_usages:
44
+ start = record.timestamp_ms if record.timestamp_ms is not None else turn.start_ms
45
+ call = LlmCall(
46
+ start=Metric.exact(start) if start is not None else Metric.not_available(),
47
+ model=turn.model,
48
+ source_request_id=record.response_id,
49
+ source_order=record.source_order,
50
+ source_stream_id=source_stream_id,
51
+ timing_basis="usage_report" if record.timestamp_ms is not None else "turn_start",
52
+ )
53
+ usage = record.usage
54
+ call.tokens = Tokens(
55
+ input=Metric.exact(usage.input),
56
+ output=Metric.exact(usage.output),
57
+ cache_read=Metric.exact(usage.cache_read),
58
+ cache_write=Metric.exact(usage.cache_write),
59
+ )
60
+ call.cost = prices.cost(turn.model, _usage(usage))
61
+ call.price_prefix = prices.matching_prefix(turn.model)
62
+ call.cost_parts = prices.cost_parts(turn.model, _usage(usage))
63
+ call.cost_parts["cache_write_5m"] = CostMetric.not_available()
64
+ call.cost_parts["cache_write_1h"] = CostMetric.not_available()
65
+ calls.append(call)
66
+ return calls
67
+
68
+
69
+ def _tool_node(tool: ToolCall) -> Node:
70
+ """Translate one native tool call and its optional output to a leaf node."""
71
+ arguments = _arguments(tool.input)
72
+ info = tool_info(tool.name, arguments)
73
+ node = Node(
74
+ node_id=tool.call_id or f"tool-{tool.name}-{tool.start_ms}",
75
+ kind=NodeKind.TOOL,
76
+ topic=_tool_topic(tool.name, arguments),
77
+ tool=info,
78
+ start=Metric.exact(tool.start_ms) if tool.start_ms is not None else Metric.not_available(),
79
+ )
80
+ if tool.output is not None:
81
+ node.result = tool.output
82
+ node.success = _success(tool)
83
+ if tool.end_ms is not None:
84
+ node.end = Metric.exact(tool.end_ms)
85
+ if tool.start_ms is not None:
86
+ node.duration = Metric.exact(tool.end_ms - tool.start_ms)
87
+ return node
88
+
89
+
90
+ def _success(tool: ToolCall) -> bool:
91
+ """Whether native call and output statuses do not report a failure."""
92
+ return all(status not in {"failed", "error", "cancelled"} for status in (tool.status, tool.output_status) if status)
93
+
94
+
95
+ def _tool_events(tool: ToolCall, node: Node, stream_id: str | None) -> list[ExecutionEvent]:
96
+ is_delegation = node.tool is not None and node.tool.category.value == "subagent"
97
+ return tool_execution_events(
98
+ invocation_id=tool.call_id,
99
+ subject_node_id=tool.call_id,
100
+ start=Metric.exact(tool.start_ms) if tool.start_ms is not None else Metric.not_available(),
101
+ end=Metric.exact(tool.end_ms) if tool.end_ms is not None else Metric.not_available(),
102
+ result_recorded=tool.result_recorded,
103
+ success=_success(tool) if tool.result_recorded else None,
104
+ delegation=is_delegation,
105
+ start_order=tool.start_order,
106
+ result_order=tool.result_order,
107
+ source_stream_id=stream_id,
108
+ )
109
+
110
+
111
+ def _tool_topic(name: str, arguments: dict[str, object]) -> str:
112
+ """Return an informative label without losing the native name for unfamiliar tools."""
113
+ for key in ("task", "description", "path", "file_path", "cmd", "command", "query", "url"):
114
+ value = arguments.get(key)
115
+ if isinstance(value, str) and value:
116
+ return f"{name}: {value.splitlines()[0][:200]}"
117
+ return name
118
+
119
+
120
+ def _turn(turn: Turn, prices: PriceTable, source_stream_id: str | None) -> Node:
121
+ """Build one neutral turn from its direct tools and LLM usage records."""
122
+ calls = _llm_calls(turn, prices, source_stream_id)
123
+ children = [_tool_node(tool) for tool in turn.tools]
124
+ events = [
125
+ ExecutionEvent(
126
+ kind="user_input",
127
+ event_id=f"user-input:{record.message_id}" if record.message_id else None,
128
+ start=Metric.exact(record.timestamp_ms) if record.timestamp_ms is not None else Metric.not_available(),
129
+ source_order=record.source_order,
130
+ source_stream_id=source_stream_id,
131
+ )
132
+ for record in turn.user_inputs
133
+ ]
134
+ for tool, child in zip(turn.tools, children, strict=True):
135
+ events.extend(_tool_events(tool, child, source_stream_id))
136
+ for result in turn.unmatched_tool_results:
137
+ result_events = tool_execution_events(
138
+ invocation_id=result.call_id,
139
+ subject_node_id=None,
140
+ start=Metric.not_available(),
141
+ end=Metric.exact(result.timestamp_ms) if result.timestamp_ms is not None else Metric.not_available(),
142
+ result_recorded=True,
143
+ success=result.status not in {"failed", "error", "cancelled"} if result.status is not None else None,
144
+ result_order=result.source_order,
145
+ source_stream_id=source_stream_id,
146
+ )
147
+ events.append(result_events[1])
148
+ if turn.completion_recorded:
149
+ events.append(
150
+ ExecutionEvent(
151
+ kind="completion",
152
+ event_id=f"completion:{turn.turn_id}",
153
+ start=Metric.exact(turn.end_ms) if turn.end_ms is not None else Metric.not_available(),
154
+ source_order=turn.completion_order,
155
+ source_stream_id=source_stream_id,
156
+ )
157
+ )
158
+ events.sort(key=lambda event: (event.source_order is None, event.source_order or 0))
159
+ node = Node(
160
+ node_id=turn.turn_id,
161
+ kind=NodeKind.TURN,
162
+ topic=prompt_topic(turn.prompt),
163
+ prompt=turn.prompt or "",
164
+ model=turn.model,
165
+ children=children,
166
+ start=Metric.exact(turn.start_ms) if turn.start_ms is not None else Metric.not_available(),
167
+ end=Metric.exact(turn.end_ms) if turn.end_ms is not None else Metric.not_available(),
168
+ llm_calls=calls,
169
+ llm_call_count=Metric.exact(len(calls)),
170
+ execution_events=events,
171
+ )
172
+ if turn.start_ms is not None and turn.end_ms is not None:
173
+ node.duration = Metric.exact(turn.end_ms - turn.start_ms)
174
+ return node
175
+
176
+
177
+ def _agent_node(rollout: Rollout, prices: PriceTable) -> Node:
178
+ """Build an agent node whose own calls and tools span every rollout turn."""
179
+ stream_id = rollout.thread_id or rollout.session_id
180
+ calls = [call for turn in rollout.turns for call in _llm_calls(turn, prices, stream_id)]
181
+ tools = [_tool_node(tool) for turn in rollout.turns for tool in turn.tools]
182
+ starts = [value for turn in rollout.turns if (value := turn.start_ms) is not None]
183
+ ends = [value for turn in rollout.turns if (value := turn.end_ms) is not None]
184
+ complete = bool(rollout.turns) and all(turn.end_ms is not None for turn in rollout.turns)
185
+ prompt = (
186
+ _ENCRYPTED_HANDOVER
187
+ if _parent_thread_id(rollout) is not None
188
+ else next((turn.prompt for turn in rollout.turns if turn.prompt), "")
189
+ )
190
+ topic = rollout.agent_nickname or (rollout.thread_spawn.agent_path if rollout.thread_spawn else None) or "(orphan)"
191
+ node = Node(
192
+ node_id=f"agent-{rollout.thread_id or rollout.session_id or 'unknown'}",
193
+ kind=NodeKind.AGENT,
194
+ topic=topic,
195
+ agent_uuid=rollout.thread_id or rollout.session_id,
196
+ prompt=prompt,
197
+ result=rollout.result or "",
198
+ model=next((turn.model for turn in rollout.turns if turn.model), rollout.model),
199
+ children=tools,
200
+ start=Metric.exact(min(starts)) if starts else Metric.not_available(),
201
+ end=Metric.exact(max(ends)) if ends and complete else Metric.not_available(),
202
+ llm_calls=calls,
203
+ llm_call_count=Metric.exact(len(calls)),
204
+ execution_events=sorted(
205
+ (event for turn in rollout.turns for event in _turn(turn, prices, stream_id).execution_events),
206
+ key=lambda event: (event.source_order is None, event.source_order or 0),
207
+ ),
208
+ )
209
+ if starts and ends and complete:
210
+ node.duration = Metric.exact(max(ends) - min(starts))
211
+ return node
212
+
213
+
214
+ def _link_names(rollout: Rollout) -> set[str]:
215
+ """Return metadata labels that can identify a spawning delegation tool."""
216
+ values = [rollout.agent_nickname]
217
+ if rollout.thread_spawn is not None:
218
+ values.extend((rollout.thread_spawn.agent_nickname, rollout.thread_spawn.agent_path))
219
+ return {value.lower() for value in values if value}
220
+
221
+
222
+ def _delegation_matches(node: Node, names: set[str]) -> bool:
223
+ if node.tool is None or node.tool.category.value != "subagent":
224
+ return False
225
+ values = (value for value in node.tool.arguments.values() if isinstance(value, str))
226
+ return any(value.lower() in names for value in values)
227
+
228
+
229
+ def _attach_agent(parent: Node, agent: Node, names: set[str]) -> bool:
230
+ """Replace a matching delegation leaf with its corresponding subagent node."""
231
+ for index, child in enumerate(parent.children):
232
+ if child.kind is NodeKind.TOOL and _delegation_matches(child, names):
233
+ agent.tool = child.tool
234
+ if not agent.result:
235
+ agent.result = child.result
236
+ agent.success = child.success
237
+ task = child.tool.arguments.get("task") if child.tool is not None else None
238
+ if isinstance(task, str) and task:
239
+ agent.prompt = task
240
+ message = child.tool.arguments.get("message") if child.tool is not None else None
241
+ if isinstance(message, str) and message.startswith("gAAAA"):
242
+ agent.prompt = _ENCRYPTED_HANDOVER
243
+ if child.start.value is not None:
244
+ agent.start = child.start
245
+ if agent.end.value is not None and agent.start.value is not None:
246
+ agent.duration = Metric.exact(agent.end.number() - agent.start.number())
247
+ parent.children[index] = agent
248
+ _append_child_completion(parent, agent)
249
+ return True
250
+ if _attach_agent(child, agent, names):
251
+ return True
252
+ return False
253
+
254
+
255
+ def _append_child_completion(parent: Node, child: Node) -> None:
256
+ """Copy a source-recorded child terminal event onto its direct calling parent."""
257
+ terminal = next((event for event in reversed(child.execution_events) if event.kind == "completion"), None)
258
+ if terminal is None:
259
+ return
260
+ event_id = None
261
+ if terminal.event_id is not None and child.agent_uuid is not None:
262
+ event_id = f"child-completion:{child.agent_uuid}:{terminal.event_id}"
263
+ parent_stream_id = next(
264
+ (event.source_stream_id for event in parent.execution_events if event.source_stream_id is not None), None
265
+ )
266
+ parent.execution_events.append(
267
+ ExecutionEvent(
268
+ kind="child_completion",
269
+ event_id=event_id,
270
+ subject_node_id=child.node_id,
271
+ start=terminal.start,
272
+ source_stream_id=parent_stream_id or parent.agent_uuid,
273
+ )
274
+ )
275
+
276
+
277
+ def _parent_thread_id(rollout: Rollout) -> str | None:
278
+ """Return the parent thread identifier recorded for a subagent rollout."""
279
+ if rollout.thread_spawn is not None and rollout.thread_spawn.parent_thread_id is not None:
280
+ return rollout.thread_spawn.parent_thread_id
281
+ return rollout.parent_thread_id
282
+
283
+
284
+ def _node_start(node: Node) -> float | None:
285
+ """Return a node's start value for shared turn-splitting logic."""
286
+ return node.start.value
287
+
288
+
289
+ def build_root(main: Rollout, subagents: list[Rollout], prices: PriceTable, title: str) -> Node:
290
+ """Build a session tree, retaining unlinked subagents under the matching main turn."""
291
+ main_stream_id = main.thread_id or main.session_id
292
+ turns = [_turn(turn, prices, main_stream_id) for turn in main.turns]
293
+ root = Node(node_id="session", kind=NodeKind.SESSION, topic=title, children=turns)
294
+ parent_nodes: dict[str, Node] = {main.thread_id or main.session_id or "": root}
295
+ agents = [(rollout, _agent_node(rollout, prices)) for rollout in subagents]
296
+ parent_nodes.update({rollout.thread_id: agent for rollout, agent in agents if rollout.thread_id is not None})
297
+ unlinked: list[Node] = []
298
+
299
+ for rollout, agent in agents:
300
+ parent_id = _parent_thread_id(rollout)
301
+ parent = parent_nodes.get(parent_id or "", root)
302
+ if not _attach_agent(parent, agent, _link_names(rollout)):
303
+ unlinked.append(agent)
304
+ timed_turns = [turn for turn in turns if turn.start.value is not None]
305
+ groups = split_by_turn(unlinked, _node_start, [turn.start.number() for turn in timed_turns])
306
+ if groups:
307
+ for turn, agents_for_turn in zip(timed_turns, groups, strict=True):
308
+ turn.children.extend(agents_for_turn)
309
+ else:
310
+ root.children.extend(unlinked)
311
+ return root
@@ -0,0 +1,3 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2026 epicodic
3
+ """VS Code Copilot Chat adapter."""
@@ -0,0 +1,166 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2026 epicodic
3
+ """The VS Code Copilot Chat adapter."""
4
+
5
+ import json
6
+ import os
7
+ import sys
8
+ from collections import Counter
9
+ from collections.abc import Iterator
10
+ from pathlib import Path
11
+
12
+ from agentprof.adapters.base import AdapterConfig, SessionRef, SessionSummary, latest_mtime
13
+ from agentprof.adapters.copilot_vscode.debuglog import DebugLog, load_debug_log_dir
14
+ from agentprof.adapters.copilot_vscode.discovery import (
15
+ SideFiles,
16
+ default_storage_roots,
17
+ find_side_files,
18
+ read_session_state,
19
+ session_files,
20
+ side_files_in_workspace,
21
+ workspace_folder,
22
+ )
23
+ from agentprof.adapters.copilot_vscode.session import RawSession, load_session, parse_session_data
24
+ from agentprof.adapters.copilot_vscode.tools import is_known_tool_id
25
+ from agentprof.adapters.copilot_vscode.transcript import Transcript, load_transcript
26
+ from agentprof.adapters.copilot_vscode.tree import build_root, credits_cost
27
+ from agentprof.model import Diagnostics, Session, sum_costs
28
+ from agentprof.pricing import PriceTable
29
+
30
+ _TITLE_LENGTH = 80
31
+ _LIVE_SESSION_DIR = "chatSessions"
32
+
33
+
34
+ def _title(raw: RawSession, fallback: str) -> str:
35
+ if raw.title:
36
+ return raw.title
37
+ for request in raw.requests:
38
+ first_line = request.text.strip().splitlines()[0] if request.text.strip() else ""
39
+ if first_line:
40
+ return first_line[:_TITLE_LENGTH]
41
+ return fallback
42
+
43
+
44
+ class CopilotVscodeAdapter:
45
+ """Reads VS Code Copilot Chat sessions from `workspaceStorage`, or single exports."""
46
+
47
+ name = "copilot-vscode"
48
+
49
+ def __init__(self, config: AdapterConfig) -> None:
50
+ self._usd_per_credit = PriceTable.load(config.pricing_file).usd_per_credit
51
+ override = config.roots.get(self.name)
52
+ if override is not None:
53
+ self._roots = [override]
54
+ else:
55
+ appdata = os.environ.get("APPDATA")
56
+ self._roots = default_storage_roots(sys.platform, Path.home(), Path(appdata) if appdata else None)
57
+
58
+ def _ref(self, path: Path, native_id: str, side: SideFiles) -> SessionRef:
59
+ files = [path]
60
+ if side.transcript_file is not None:
61
+ files.append(side.transcript_file)
62
+ if side.debug_log_dir is not None:
63
+ files.extend(side.debug_log_dir.glob("*.jsonl"))
64
+ return SessionRef(agent=self.name, native_id=native_id, path=path, mtime=latest_mtime(files))
65
+
66
+ def discover(self) -> Iterator[SessionRef]:
67
+ for root in self._roots:
68
+ for path in session_files(root):
69
+ yield self._ref(path, path.stem, side_files_in_workspace(path.parent.parent, path.stem))
70
+
71
+ def open_path(self, path: Path) -> SessionRef | None:
72
+ if path.suffix == ".jsonl" and path.parent.name == _LIVE_SESSION_DIR and path.is_file():
73
+ return self._ref(path, path.stem, side_files_in_workspace(path.parent.parent, path.stem))
74
+ if path.suffix != ".json":
75
+ return None
76
+ try:
77
+ data = json.loads(path.read_text())
78
+ except (OSError, ValueError):
79
+ return None
80
+ if not isinstance(data, dict) or not isinstance(data.get("requests"), list):
81
+ return None
82
+ try:
83
+ raw = parse_session_data(data)
84
+ except (KeyError, TypeError, AttributeError):
85
+ return None
86
+ native_id = raw.session_id or path.stem
87
+ return self._ref(path, native_id, find_side_files(self._roots, native_id))
88
+
89
+ def summarize(self, ref: SessionRef) -> SessionSummary:
90
+ if ref.path.suffix == ".jsonl":
91
+ state = read_session_state(ref.path)
92
+ if state is None:
93
+ raise ValueError(f"cannot read session file {ref.path}")
94
+ title, start_ms, credits = state.title, state.created_ms, state.credits
95
+ last_activity_ms = state.last_activity_ms
96
+ workspace = workspace_folder(ref.path.parent.parent)
97
+ else:
98
+ raw = load_session(ref.path)
99
+ title, start_ms, credits = raw.title, raw.creation_ms, [request.credits for request in raw.requests]
100
+ last_activity_ms = max(
101
+ (
102
+ max(request.timestamp_ms, request.response_timestamp_ms or request.timestamp_ms)
103
+ for request in raw.requests
104
+ ),
105
+ default=None,
106
+ )
107
+ if start_ms is None and raw.requests:
108
+ start_ms = raw.requests[0].timestamp_ms
109
+ workspace = None
110
+ return SessionSummary(
111
+ id=ref.id,
112
+ agent=self.name,
113
+ title=title or ref.native_id,
114
+ workspace=workspace,
115
+ start_ms=start_ms if start_ms is not None else ref.mtime * 1000,
116
+ end_ms=ref.mtime * 1000,
117
+ file_size=ref.path.stat().st_size,
118
+ last_activity_ms=last_activity_ms,
119
+ cost_total=sum_costs([credits_cost(value, self._usd_per_credit) for value in credits]),
120
+ )
121
+
122
+ def analyze(self, ref: SessionRef) -> Session:
123
+ raw = load_session(ref.path)
124
+ is_live = ref.path.parent.name == _LIVE_SESSION_DIR
125
+ side = self._side_files(ref, is_live)
126
+ diagnostics = Diagnostics(malformed_lines=raw.malformed_lines)
127
+ sources = ["live-session" if is_live else "export"]
128
+
129
+ transcript: Transcript | None = None
130
+ if side.transcript_file is not None:
131
+ transcript = load_transcript(side.transcript_file)
132
+ sources.append("transcript")
133
+ diagnostics.malformed_lines += transcript.malformed_lines
134
+ else:
135
+ diagnostics.warnings.append("No transcript found: tool timings and LLM call counts are unavailable.")
136
+
137
+ debug_log: DebugLog | None = None
138
+ if side.debug_log_dir is not None:
139
+ debug_log = load_debug_log_dir(side.debug_log_dir, root_session_id=side.debug_log_dir.name)
140
+ sources.append("debug-log")
141
+ diagnostics.malformed_lines += debug_log.malformed_lines
142
+ else:
143
+ diagnostics.warnings.append("No debug log found: token counts are unavailable.")
144
+
145
+ unknown = Counter(
146
+ tool_call.tool_id
147
+ for request in raw.requests
148
+ for tool_call in request.tool_calls
149
+ if not is_known_tool_id(tool_call.tool_id)
150
+ )
151
+ diagnostics.unknown_tool_ids = dict(unknown)
152
+
153
+ return Session(
154
+ id=ref.id,
155
+ agent=self.name,
156
+ title=_title(raw, ref.native_id),
157
+ workspace=workspace_folder(ref.path.parent.parent) if is_live else None,
158
+ root=build_root(raw, transcript, debug_log, self._usd_per_credit),
159
+ sources=sources,
160
+ diagnostics=diagnostics,
161
+ )
162
+
163
+ def _side_files(self, ref: SessionRef, is_live: bool) -> SideFiles:
164
+ if is_live:
165
+ return side_files_in_workspace(ref.path.parent.parent, ref.native_id)
166
+ return find_side_files(self._roots, ref.native_id)
@@ -0,0 +1,136 @@
1
+ # SPDX-License-Identifier: MIT
2
+ # Copyright (c) 2026 epicodic
3
+ """Parse a `GitHub.copilot-chat/debug-logs/<sid>/` directory into LLM calls per agent."""
4
+
5
+ import json
6
+ import re
7
+ from dataclasses import dataclass, field, replace
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class DebugLlmCall:
14
+ """One `llm_request` span. `input_tokens` includes `cached_tokens`, as Copilot logs it.
15
+
16
+ Token counts are `None` when the span does not log them (e.g. a failed or still running request).
17
+ `debug_name` names the request's purpose, e.g. `panel/editAgent` or `summarizeConversationHistory`.
18
+ """
19
+
20
+ start_ms: int
21
+ duration_ms: int
22
+ input_tokens: int | None
23
+ output_tokens: int | None
24
+ cached_tokens: int | None
25
+ debug_name: str | None = None
26
+ source_request_id: str | None = None
27
+ requested_tool_ids: tuple[str, ...] = ()
28
+ consumed_tool_ids: tuple[str, ...] = ()
29
+ model: str | None = None
30
+ usage_nano_aiu: int | None = None
31
+
32
+
33
+ @dataclass
34
+ class DebugLog:
35
+ """LLM calls keyed by the owning agent's `toolCallId` (`None` for the main agent)."""
36
+
37
+ calls: dict[str | None, list[DebugLlmCall]] = field(default_factory=dict)
38
+ malformed_lines: int = 0
39
+
40
+
41
+ def _count(attrs: dict[str, Any], key: str) -> int | None:
42
+ value = attrs.get(key)
43
+ return None if value is None else int(value)
44
+
45
+
46
+ def _tool_part_ids(messages: object, part_type: str) -> tuple[str, ...]:
47
+ """Read typed tool IDs, including complete identity headers before Copilot's truncation marker."""
48
+ if isinstance(messages, str):
49
+ encoded = messages
50
+ try:
51
+ messages = json.loads(encoded)
52
+ except json.JSONDecodeError:
53
+ if not encoded.startswith("[") or not encoded.endswith("[truncated]"):
54
+ return ()
55
+ pattern = r'\{"type"\s*:\s*"' + re.escape(part_type) + r'"\s*,\s*"id"\s*:\s*"([^"\\]+)"'
56
+ return tuple(dict.fromkeys(re.findall(pattern, encoded)))
57
+ if not isinstance(messages, list):
58
+ return ()
59
+ ids: list[str] = []
60
+ for message in messages:
61
+ if not isinstance(message, dict) or not isinstance(message.get("parts"), list):
62
+ continue
63
+ for part in message["parts"]:
64
+ if not isinstance(part, dict) or part.get("type") != part_type:
65
+ continue
66
+ tool_id = part.get("id")
67
+ if isinstance(tool_id, str) and tool_id and tool_id not in ids:
68
+ ids.append(tool_id)
69
+ return tuple(ids)
70
+
71
+
72
+ def _parse_call(entry: dict[str, Any]) -> DebugLlmCall:
73
+ attrs = entry["attrs"]
74
+ debug_name = attrs.get("debugName")
75
+ return DebugLlmCall(
76
+ start_ms=int(entry["ts"]),
77
+ duration_ms=int(entry.get("dur", 0)),
78
+ input_tokens=_count(attrs, "inputTokens"),
79
+ output_tokens=_count(attrs, "outputTokens"),
80
+ cached_tokens=_count(attrs, "cachedTokens"),
81
+ debug_name=debug_name if isinstance(debug_name, str) else None,
82
+ model=attrs.get("model") if isinstance(attrs.get("model"), str) else None,
83
+ usage_nano_aiu=_count(attrs, "copilotUsageNanoAiu"),
84
+ source_request_id=entry.get("spanId") if isinstance(entry.get("spanId"), str) else None,
85
+ consumed_tool_ids=_tool_part_ids(attrs.get("inputMessages"), "tool_call_response"),
86
+ )
87
+
88
+
89
+ def load_debug_log_dir(directory: Path, root_session_id: str) -> DebugLog:
90
+ """Collect `llm_request` spans from every file in a debug-log directory.
91
+
92
+ Every line carries a top-level `sid`: the root session id for the main agent's file, otherwise the
93
+ subagent's own `toolCallId`, regardless of nesting depth.
94
+
95
+ Args:
96
+ directory: A `<sessionId>/` debug-log directory containing `*.jsonl` files.
97
+ root_session_id: The `sid` used by the main agent's lines.
98
+
99
+ Returns:
100
+ The LLM calls per agent and the number of malformed lines skipped.
101
+ """
102
+ log = DebugLog()
103
+ responses: dict[tuple[str | None, str], tuple[str, ...]] = {}
104
+ for file_path in sorted(directory.glob("*.jsonl")):
105
+ for line in file_path.read_text().splitlines():
106
+ if not line.strip():
107
+ continue
108
+ try:
109
+ entry = json.loads(line)
110
+ if entry.get("type") not in ("llm_request", "agent_response"):
111
+ continue
112
+ sid = entry["sid"]
113
+ key = None if sid == root_session_id else sid
114
+ if entry["type"] == "agent_response":
115
+ span_id = entry.get("spanId")
116
+ if isinstance(span_id, str) and span_id.startswith("agent-msg-"):
117
+ request_id = span_id.removeprefix("agent-msg-")
118
+ if request_id:
119
+ ids = _tool_part_ids(entry["attrs"].get("response"), "tool_call")
120
+ previous = responses.get((key, request_id), ())
121
+ responses[key, request_id] = tuple(dict.fromkeys((*previous, *ids)))
122
+ continue
123
+ call = _parse_call(entry)
124
+ except (json.JSONDecodeError, AttributeError, KeyError, TypeError, ValueError):
125
+ log.malformed_lines += 1
126
+ continue
127
+ key = None if sid == root_session_id else sid
128
+ log.calls.setdefault(key, []).append(call)
129
+ for owner, calls in log.calls.items():
130
+ log.calls[owner] = [
131
+ replace(call, requested_tool_ids=responses.get((owner, call.source_request_id), ()))
132
+ if call.source_request_id is not None
133
+ else call
134
+ for call in calls
135
+ ]
136
+ return log