agentprof 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentprof/__init__.py +7 -0
- agentprof/adapters/__init__.py +15 -0
- agentprof/adapters/base.py +79 -0
- agentprof/adapters/claude_code/__init__.py +3 -0
- agentprof/adapters/claude_code/adapter.py +133 -0
- agentprof/adapters/claude_code/discovery.py +79 -0
- agentprof/adapters/claude_code/prompts.py +44 -0
- agentprof/adapters/claude_code/tools.py +102 -0
- agentprof/adapters/claude_code/transcript.py +264 -0
- agentprof/adapters/claude_code/tree.py +492 -0
- agentprof/adapters/codex/__init__.py +3 -0
- agentprof/adapters/codex/adapter.py +164 -0
- agentprof/adapters/codex/discovery.py +117 -0
- agentprof/adapters/codex/prompts.py +16 -0
- agentprof/adapters/codex/rollout.py +328 -0
- agentprof/adapters/codex/tools.py +97 -0
- agentprof/adapters/codex/tree.py +311 -0
- agentprof/adapters/copilot_vscode/__init__.py +3 -0
- agentprof/adapters/copilot_vscode/adapter.py +166 -0
- agentprof/adapters/copilot_vscode/debuglog.py +136 -0
- agentprof/adapters/copilot_vscode/discovery.py +153 -0
- agentprof/adapters/copilot_vscode/session.py +281 -0
- agentprof/adapters/copilot_vscode/tools.py +127 -0
- agentprof/adapters/copilot_vscode/transcript.py +176 -0
- agentprof/adapters/copilot_vscode/tree.py +325 -0
- agentprof/adapters/execution.py +64 -0
- agentprof/adapters/timestamps.py +13 -0
- agentprof/adapters/turns.py +24 -0
- agentprof/analysis/__init__.py +3 -0
- agentprof/analysis/agent_summary.py +161 -0
- agentprof/analysis/call_context.py +101 -0
- agentprof/analysis/evidence_findings.py +151 -0
- agentprof/analysis/execution.py +39 -0
- agentprof/analysis/heuristics.py +306 -0
- agentprof/analysis/pipeline.py +31 -0
- agentprof/analysis/rollup.py +196 -0
- agentprof/cli.py +141 -0
- agentprof/model.py +341 -0
- agentprof/pricing.json +23 -0
- agentprof/pricing.py +120 -0
- agentprof/registry.py +304 -0
- agentprof/server/__init__.py +3 -0
- agentprof/server/app.py +399 -0
- agentprof/server/openapi.py +31 -0
- agentprof/server/pricing_schema.py +40 -0
- agentprof/server/schemas.py +579 -0
- agentprof/server/static/assets/index-C4HEqUVf.css +1 -0
- agentprof/server/static/assets/index-ZsPn1H3d.js +58 -0
- agentprof/server/static/index.html +14 -0
- agentprof-0.1.0.dist-info/METADATA +177 -0
- agentprof-0.1.0.dist-info/RECORD +54 -0
- agentprof-0.1.0.dist-info/WHEEL +4 -0
- agentprof-0.1.0.dist-info/entry_points.txt +8 -0
- agentprof-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,311 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
# Copyright (c) 2026 epicodic
|
|
3
|
+
"""Build neutral session trees from Codex rollout records."""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
|
|
7
|
+
from agentprof.adapters.codex.prompts import prompt_topic
|
|
8
|
+
from agentprof.adapters.codex.rollout import Rollout, ToolCall, Turn, UsageRecord
|
|
9
|
+
from agentprof.adapters.codex.tools import tool_info
|
|
10
|
+
from agentprof.adapters.execution import tool_execution_events
|
|
11
|
+
from agentprof.adapters.turns import split_by_turn
|
|
12
|
+
from agentprof.model import CostMetric, ExecutionEvent, LlmCall, Metric, Node, NodeKind, Tokens
|
|
13
|
+
from agentprof.pricing import PriceTable, Usage
|
|
14
|
+
|
|
15
|
+
_ENCRYPTED_HANDOVER = "Handover unavailable: Codex stored it encrypted."
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _arguments(raw_input: str | None) -> dict[str, object]:
|
|
19
|
+
"""Decode native tool input when it is a JSON object, otherwise retain no structured arguments."""
|
|
20
|
+
if raw_input is None:
|
|
21
|
+
return {}
|
|
22
|
+
try:
|
|
23
|
+
arguments = json.loads(raw_input)
|
|
24
|
+
except json.JSONDecodeError:
|
|
25
|
+
return {"patch": raw_input} if "*** Update File:" in raw_input else {}
|
|
26
|
+
return arguments if isinstance(arguments, dict) else {}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _usage(usage: UsageRecord) -> Usage:
|
|
30
|
+
"""Convert a native Codex usage record to the shared price-table representation."""
|
|
31
|
+
return Usage(
|
|
32
|
+
input=usage.input,
|
|
33
|
+
output=usage.output,
|
|
34
|
+
cache_read=usage.cache_read,
|
|
35
|
+
cache_write_5m=usage.cache_write,
|
|
36
|
+
cache_write_1h=0,
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _llm_calls(turn: Turn, prices: PriceTable, source_stream_id: str | None = None) -> list[LlmCall]:
|
|
41
|
+
"""Build one timed LLM call for every native usage record in a turn."""
|
|
42
|
+
calls: list[LlmCall] = []
|
|
43
|
+
for record in turn.token_usages:
|
|
44
|
+
start = record.timestamp_ms if record.timestamp_ms is not None else turn.start_ms
|
|
45
|
+
call = LlmCall(
|
|
46
|
+
start=Metric.exact(start) if start is not None else Metric.not_available(),
|
|
47
|
+
model=turn.model,
|
|
48
|
+
source_request_id=record.response_id,
|
|
49
|
+
source_order=record.source_order,
|
|
50
|
+
source_stream_id=source_stream_id,
|
|
51
|
+
timing_basis="usage_report" if record.timestamp_ms is not None else "turn_start",
|
|
52
|
+
)
|
|
53
|
+
usage = record.usage
|
|
54
|
+
call.tokens = Tokens(
|
|
55
|
+
input=Metric.exact(usage.input),
|
|
56
|
+
output=Metric.exact(usage.output),
|
|
57
|
+
cache_read=Metric.exact(usage.cache_read),
|
|
58
|
+
cache_write=Metric.exact(usage.cache_write),
|
|
59
|
+
)
|
|
60
|
+
call.cost = prices.cost(turn.model, _usage(usage))
|
|
61
|
+
call.price_prefix = prices.matching_prefix(turn.model)
|
|
62
|
+
call.cost_parts = prices.cost_parts(turn.model, _usage(usage))
|
|
63
|
+
call.cost_parts["cache_write_5m"] = CostMetric.not_available()
|
|
64
|
+
call.cost_parts["cache_write_1h"] = CostMetric.not_available()
|
|
65
|
+
calls.append(call)
|
|
66
|
+
return calls
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _tool_node(tool: ToolCall) -> Node:
|
|
70
|
+
"""Translate one native tool call and its optional output to a leaf node."""
|
|
71
|
+
arguments = _arguments(tool.input)
|
|
72
|
+
info = tool_info(tool.name, arguments)
|
|
73
|
+
node = Node(
|
|
74
|
+
node_id=tool.call_id or f"tool-{tool.name}-{tool.start_ms}",
|
|
75
|
+
kind=NodeKind.TOOL,
|
|
76
|
+
topic=_tool_topic(tool.name, arguments),
|
|
77
|
+
tool=info,
|
|
78
|
+
start=Metric.exact(tool.start_ms) if tool.start_ms is not None else Metric.not_available(),
|
|
79
|
+
)
|
|
80
|
+
if tool.output is not None:
|
|
81
|
+
node.result = tool.output
|
|
82
|
+
node.success = _success(tool)
|
|
83
|
+
if tool.end_ms is not None:
|
|
84
|
+
node.end = Metric.exact(tool.end_ms)
|
|
85
|
+
if tool.start_ms is not None:
|
|
86
|
+
node.duration = Metric.exact(tool.end_ms - tool.start_ms)
|
|
87
|
+
return node
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _success(tool: ToolCall) -> bool:
|
|
91
|
+
"""Whether native call and output statuses do not report a failure."""
|
|
92
|
+
return all(status not in {"failed", "error", "cancelled"} for status in (tool.status, tool.output_status) if status)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _tool_events(tool: ToolCall, node: Node, stream_id: str | None) -> list[ExecutionEvent]:
|
|
96
|
+
is_delegation = node.tool is not None and node.tool.category.value == "subagent"
|
|
97
|
+
return tool_execution_events(
|
|
98
|
+
invocation_id=tool.call_id,
|
|
99
|
+
subject_node_id=tool.call_id,
|
|
100
|
+
start=Metric.exact(tool.start_ms) if tool.start_ms is not None else Metric.not_available(),
|
|
101
|
+
end=Metric.exact(tool.end_ms) if tool.end_ms is not None else Metric.not_available(),
|
|
102
|
+
result_recorded=tool.result_recorded,
|
|
103
|
+
success=_success(tool) if tool.result_recorded else None,
|
|
104
|
+
delegation=is_delegation,
|
|
105
|
+
start_order=tool.start_order,
|
|
106
|
+
result_order=tool.result_order,
|
|
107
|
+
source_stream_id=stream_id,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _tool_topic(name: str, arguments: dict[str, object]) -> str:
|
|
112
|
+
"""Return an informative label without losing the native name for unfamiliar tools."""
|
|
113
|
+
for key in ("task", "description", "path", "file_path", "cmd", "command", "query", "url"):
|
|
114
|
+
value = arguments.get(key)
|
|
115
|
+
if isinstance(value, str) and value:
|
|
116
|
+
return f"{name}: {value.splitlines()[0][:200]}"
|
|
117
|
+
return name
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _turn(turn: Turn, prices: PriceTable, source_stream_id: str | None) -> Node:
|
|
121
|
+
"""Build one neutral turn from its direct tools and LLM usage records."""
|
|
122
|
+
calls = _llm_calls(turn, prices, source_stream_id)
|
|
123
|
+
children = [_tool_node(tool) for tool in turn.tools]
|
|
124
|
+
events = [
|
|
125
|
+
ExecutionEvent(
|
|
126
|
+
kind="user_input",
|
|
127
|
+
event_id=f"user-input:{record.message_id}" if record.message_id else None,
|
|
128
|
+
start=Metric.exact(record.timestamp_ms) if record.timestamp_ms is not None else Metric.not_available(),
|
|
129
|
+
source_order=record.source_order,
|
|
130
|
+
source_stream_id=source_stream_id,
|
|
131
|
+
)
|
|
132
|
+
for record in turn.user_inputs
|
|
133
|
+
]
|
|
134
|
+
for tool, child in zip(turn.tools, children, strict=True):
|
|
135
|
+
events.extend(_tool_events(tool, child, source_stream_id))
|
|
136
|
+
for result in turn.unmatched_tool_results:
|
|
137
|
+
result_events = tool_execution_events(
|
|
138
|
+
invocation_id=result.call_id,
|
|
139
|
+
subject_node_id=None,
|
|
140
|
+
start=Metric.not_available(),
|
|
141
|
+
end=Metric.exact(result.timestamp_ms) if result.timestamp_ms is not None else Metric.not_available(),
|
|
142
|
+
result_recorded=True,
|
|
143
|
+
success=result.status not in {"failed", "error", "cancelled"} if result.status is not None else None,
|
|
144
|
+
result_order=result.source_order,
|
|
145
|
+
source_stream_id=source_stream_id,
|
|
146
|
+
)
|
|
147
|
+
events.append(result_events[1])
|
|
148
|
+
if turn.completion_recorded:
|
|
149
|
+
events.append(
|
|
150
|
+
ExecutionEvent(
|
|
151
|
+
kind="completion",
|
|
152
|
+
event_id=f"completion:{turn.turn_id}",
|
|
153
|
+
start=Metric.exact(turn.end_ms) if turn.end_ms is not None else Metric.not_available(),
|
|
154
|
+
source_order=turn.completion_order,
|
|
155
|
+
source_stream_id=source_stream_id,
|
|
156
|
+
)
|
|
157
|
+
)
|
|
158
|
+
events.sort(key=lambda event: (event.source_order is None, event.source_order or 0))
|
|
159
|
+
node = Node(
|
|
160
|
+
node_id=turn.turn_id,
|
|
161
|
+
kind=NodeKind.TURN,
|
|
162
|
+
topic=prompt_topic(turn.prompt),
|
|
163
|
+
prompt=turn.prompt or "",
|
|
164
|
+
model=turn.model,
|
|
165
|
+
children=children,
|
|
166
|
+
start=Metric.exact(turn.start_ms) if turn.start_ms is not None else Metric.not_available(),
|
|
167
|
+
end=Metric.exact(turn.end_ms) if turn.end_ms is not None else Metric.not_available(),
|
|
168
|
+
llm_calls=calls,
|
|
169
|
+
llm_call_count=Metric.exact(len(calls)),
|
|
170
|
+
execution_events=events,
|
|
171
|
+
)
|
|
172
|
+
if turn.start_ms is not None and turn.end_ms is not None:
|
|
173
|
+
node.duration = Metric.exact(turn.end_ms - turn.start_ms)
|
|
174
|
+
return node
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _agent_node(rollout: Rollout, prices: PriceTable) -> Node:
|
|
178
|
+
"""Build an agent node whose own calls and tools span every rollout turn."""
|
|
179
|
+
stream_id = rollout.thread_id or rollout.session_id
|
|
180
|
+
calls = [call for turn in rollout.turns for call in _llm_calls(turn, prices, stream_id)]
|
|
181
|
+
tools = [_tool_node(tool) for turn in rollout.turns for tool in turn.tools]
|
|
182
|
+
starts = [value for turn in rollout.turns if (value := turn.start_ms) is not None]
|
|
183
|
+
ends = [value for turn in rollout.turns if (value := turn.end_ms) is not None]
|
|
184
|
+
complete = bool(rollout.turns) and all(turn.end_ms is not None for turn in rollout.turns)
|
|
185
|
+
prompt = (
|
|
186
|
+
_ENCRYPTED_HANDOVER
|
|
187
|
+
if _parent_thread_id(rollout) is not None
|
|
188
|
+
else next((turn.prompt for turn in rollout.turns if turn.prompt), "")
|
|
189
|
+
)
|
|
190
|
+
topic = rollout.agent_nickname or (rollout.thread_spawn.agent_path if rollout.thread_spawn else None) or "(orphan)"
|
|
191
|
+
node = Node(
|
|
192
|
+
node_id=f"agent-{rollout.thread_id or rollout.session_id or 'unknown'}",
|
|
193
|
+
kind=NodeKind.AGENT,
|
|
194
|
+
topic=topic,
|
|
195
|
+
agent_uuid=rollout.thread_id or rollout.session_id,
|
|
196
|
+
prompt=prompt,
|
|
197
|
+
result=rollout.result or "",
|
|
198
|
+
model=next((turn.model for turn in rollout.turns if turn.model), rollout.model),
|
|
199
|
+
children=tools,
|
|
200
|
+
start=Metric.exact(min(starts)) if starts else Metric.not_available(),
|
|
201
|
+
end=Metric.exact(max(ends)) if ends and complete else Metric.not_available(),
|
|
202
|
+
llm_calls=calls,
|
|
203
|
+
llm_call_count=Metric.exact(len(calls)),
|
|
204
|
+
execution_events=sorted(
|
|
205
|
+
(event for turn in rollout.turns for event in _turn(turn, prices, stream_id).execution_events),
|
|
206
|
+
key=lambda event: (event.source_order is None, event.source_order or 0),
|
|
207
|
+
),
|
|
208
|
+
)
|
|
209
|
+
if starts and ends and complete:
|
|
210
|
+
node.duration = Metric.exact(max(ends) - min(starts))
|
|
211
|
+
return node
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _link_names(rollout: Rollout) -> set[str]:
|
|
215
|
+
"""Return metadata labels that can identify a spawning delegation tool."""
|
|
216
|
+
values = [rollout.agent_nickname]
|
|
217
|
+
if rollout.thread_spawn is not None:
|
|
218
|
+
values.extend((rollout.thread_spawn.agent_nickname, rollout.thread_spawn.agent_path))
|
|
219
|
+
return {value.lower() for value in values if value}
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _delegation_matches(node: Node, names: set[str]) -> bool:
|
|
223
|
+
if node.tool is None or node.tool.category.value != "subagent":
|
|
224
|
+
return False
|
|
225
|
+
values = (value for value in node.tool.arguments.values() if isinstance(value, str))
|
|
226
|
+
return any(value.lower() in names for value in values)
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _attach_agent(parent: Node, agent: Node, names: set[str]) -> bool:
|
|
230
|
+
"""Replace a matching delegation leaf with its corresponding subagent node."""
|
|
231
|
+
for index, child in enumerate(parent.children):
|
|
232
|
+
if child.kind is NodeKind.TOOL and _delegation_matches(child, names):
|
|
233
|
+
agent.tool = child.tool
|
|
234
|
+
if not agent.result:
|
|
235
|
+
agent.result = child.result
|
|
236
|
+
agent.success = child.success
|
|
237
|
+
task = child.tool.arguments.get("task") if child.tool is not None else None
|
|
238
|
+
if isinstance(task, str) and task:
|
|
239
|
+
agent.prompt = task
|
|
240
|
+
message = child.tool.arguments.get("message") if child.tool is not None else None
|
|
241
|
+
if isinstance(message, str) and message.startswith("gAAAA"):
|
|
242
|
+
agent.prompt = _ENCRYPTED_HANDOVER
|
|
243
|
+
if child.start.value is not None:
|
|
244
|
+
agent.start = child.start
|
|
245
|
+
if agent.end.value is not None and agent.start.value is not None:
|
|
246
|
+
agent.duration = Metric.exact(agent.end.number() - agent.start.number())
|
|
247
|
+
parent.children[index] = agent
|
|
248
|
+
_append_child_completion(parent, agent)
|
|
249
|
+
return True
|
|
250
|
+
if _attach_agent(child, agent, names):
|
|
251
|
+
return True
|
|
252
|
+
return False
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _append_child_completion(parent: Node, child: Node) -> None:
|
|
256
|
+
"""Copy a source-recorded child terminal event onto its direct calling parent."""
|
|
257
|
+
terminal = next((event for event in reversed(child.execution_events) if event.kind == "completion"), None)
|
|
258
|
+
if terminal is None:
|
|
259
|
+
return
|
|
260
|
+
event_id = None
|
|
261
|
+
if terminal.event_id is not None and child.agent_uuid is not None:
|
|
262
|
+
event_id = f"child-completion:{child.agent_uuid}:{terminal.event_id}"
|
|
263
|
+
parent_stream_id = next(
|
|
264
|
+
(event.source_stream_id for event in parent.execution_events if event.source_stream_id is not None), None
|
|
265
|
+
)
|
|
266
|
+
parent.execution_events.append(
|
|
267
|
+
ExecutionEvent(
|
|
268
|
+
kind="child_completion",
|
|
269
|
+
event_id=event_id,
|
|
270
|
+
subject_node_id=child.node_id,
|
|
271
|
+
start=terminal.start,
|
|
272
|
+
source_stream_id=parent_stream_id or parent.agent_uuid,
|
|
273
|
+
)
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _parent_thread_id(rollout: Rollout) -> str | None:
|
|
278
|
+
"""Return the parent thread identifier recorded for a subagent rollout."""
|
|
279
|
+
if rollout.thread_spawn is not None and rollout.thread_spawn.parent_thread_id is not None:
|
|
280
|
+
return rollout.thread_spawn.parent_thread_id
|
|
281
|
+
return rollout.parent_thread_id
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _node_start(node: Node) -> float | None:
|
|
285
|
+
"""Return a node's start value for shared turn-splitting logic."""
|
|
286
|
+
return node.start.value
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def build_root(main: Rollout, subagents: list[Rollout], prices: PriceTable, title: str) -> Node:
|
|
290
|
+
"""Build a session tree, retaining unlinked subagents under the matching main turn."""
|
|
291
|
+
main_stream_id = main.thread_id or main.session_id
|
|
292
|
+
turns = [_turn(turn, prices, main_stream_id) for turn in main.turns]
|
|
293
|
+
root = Node(node_id="session", kind=NodeKind.SESSION, topic=title, children=turns)
|
|
294
|
+
parent_nodes: dict[str, Node] = {main.thread_id or main.session_id or "": root}
|
|
295
|
+
agents = [(rollout, _agent_node(rollout, prices)) for rollout in subagents]
|
|
296
|
+
parent_nodes.update({rollout.thread_id: agent for rollout, agent in agents if rollout.thread_id is not None})
|
|
297
|
+
unlinked: list[Node] = []
|
|
298
|
+
|
|
299
|
+
for rollout, agent in agents:
|
|
300
|
+
parent_id = _parent_thread_id(rollout)
|
|
301
|
+
parent = parent_nodes.get(parent_id or "", root)
|
|
302
|
+
if not _attach_agent(parent, agent, _link_names(rollout)):
|
|
303
|
+
unlinked.append(agent)
|
|
304
|
+
timed_turns = [turn for turn in turns if turn.start.value is not None]
|
|
305
|
+
groups = split_by_turn(unlinked, _node_start, [turn.start.number() for turn in timed_turns])
|
|
306
|
+
if groups:
|
|
307
|
+
for turn, agents_for_turn in zip(timed_turns, groups, strict=True):
|
|
308
|
+
turn.children.extend(agents_for_turn)
|
|
309
|
+
else:
|
|
310
|
+
root.children.extend(unlinked)
|
|
311
|
+
return root
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
# Copyright (c) 2026 epicodic
|
|
3
|
+
"""The VS Code Copilot Chat adapter."""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
from collections import Counter
|
|
9
|
+
from collections.abc import Iterator
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from agentprof.adapters.base import AdapterConfig, SessionRef, SessionSummary, latest_mtime
|
|
13
|
+
from agentprof.adapters.copilot_vscode.debuglog import DebugLog, load_debug_log_dir
|
|
14
|
+
from agentprof.adapters.copilot_vscode.discovery import (
|
|
15
|
+
SideFiles,
|
|
16
|
+
default_storage_roots,
|
|
17
|
+
find_side_files,
|
|
18
|
+
read_session_state,
|
|
19
|
+
session_files,
|
|
20
|
+
side_files_in_workspace,
|
|
21
|
+
workspace_folder,
|
|
22
|
+
)
|
|
23
|
+
from agentprof.adapters.copilot_vscode.session import RawSession, load_session, parse_session_data
|
|
24
|
+
from agentprof.adapters.copilot_vscode.tools import is_known_tool_id
|
|
25
|
+
from agentprof.adapters.copilot_vscode.transcript import Transcript, load_transcript
|
|
26
|
+
from agentprof.adapters.copilot_vscode.tree import build_root, credits_cost
|
|
27
|
+
from agentprof.model import Diagnostics, Session, sum_costs
|
|
28
|
+
from agentprof.pricing import PriceTable
|
|
29
|
+
|
|
30
|
+
_TITLE_LENGTH = 80
|
|
31
|
+
_LIVE_SESSION_DIR = "chatSessions"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _title(raw: RawSession, fallback: str) -> str:
|
|
35
|
+
if raw.title:
|
|
36
|
+
return raw.title
|
|
37
|
+
for request in raw.requests:
|
|
38
|
+
first_line = request.text.strip().splitlines()[0] if request.text.strip() else ""
|
|
39
|
+
if first_line:
|
|
40
|
+
return first_line[:_TITLE_LENGTH]
|
|
41
|
+
return fallback
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class CopilotVscodeAdapter:
|
|
45
|
+
"""Reads VS Code Copilot Chat sessions from `workspaceStorage`, or single exports."""
|
|
46
|
+
|
|
47
|
+
name = "copilot-vscode"
|
|
48
|
+
|
|
49
|
+
def __init__(self, config: AdapterConfig) -> None:
|
|
50
|
+
self._usd_per_credit = PriceTable.load(config.pricing_file).usd_per_credit
|
|
51
|
+
override = config.roots.get(self.name)
|
|
52
|
+
if override is not None:
|
|
53
|
+
self._roots = [override]
|
|
54
|
+
else:
|
|
55
|
+
appdata = os.environ.get("APPDATA")
|
|
56
|
+
self._roots = default_storage_roots(sys.platform, Path.home(), Path(appdata) if appdata else None)
|
|
57
|
+
|
|
58
|
+
def _ref(self, path: Path, native_id: str, side: SideFiles) -> SessionRef:
|
|
59
|
+
files = [path]
|
|
60
|
+
if side.transcript_file is not None:
|
|
61
|
+
files.append(side.transcript_file)
|
|
62
|
+
if side.debug_log_dir is not None:
|
|
63
|
+
files.extend(side.debug_log_dir.glob("*.jsonl"))
|
|
64
|
+
return SessionRef(agent=self.name, native_id=native_id, path=path, mtime=latest_mtime(files))
|
|
65
|
+
|
|
66
|
+
def discover(self) -> Iterator[SessionRef]:
|
|
67
|
+
for root in self._roots:
|
|
68
|
+
for path in session_files(root):
|
|
69
|
+
yield self._ref(path, path.stem, side_files_in_workspace(path.parent.parent, path.stem))
|
|
70
|
+
|
|
71
|
+
def open_path(self, path: Path) -> SessionRef | None:
|
|
72
|
+
if path.suffix == ".jsonl" and path.parent.name == _LIVE_SESSION_DIR and path.is_file():
|
|
73
|
+
return self._ref(path, path.stem, side_files_in_workspace(path.parent.parent, path.stem))
|
|
74
|
+
if path.suffix != ".json":
|
|
75
|
+
return None
|
|
76
|
+
try:
|
|
77
|
+
data = json.loads(path.read_text())
|
|
78
|
+
except (OSError, ValueError):
|
|
79
|
+
return None
|
|
80
|
+
if not isinstance(data, dict) or not isinstance(data.get("requests"), list):
|
|
81
|
+
return None
|
|
82
|
+
try:
|
|
83
|
+
raw = parse_session_data(data)
|
|
84
|
+
except (KeyError, TypeError, AttributeError):
|
|
85
|
+
return None
|
|
86
|
+
native_id = raw.session_id or path.stem
|
|
87
|
+
return self._ref(path, native_id, find_side_files(self._roots, native_id))
|
|
88
|
+
|
|
89
|
+
def summarize(self, ref: SessionRef) -> SessionSummary:
|
|
90
|
+
if ref.path.suffix == ".jsonl":
|
|
91
|
+
state = read_session_state(ref.path)
|
|
92
|
+
if state is None:
|
|
93
|
+
raise ValueError(f"cannot read session file {ref.path}")
|
|
94
|
+
title, start_ms, credits = state.title, state.created_ms, state.credits
|
|
95
|
+
last_activity_ms = state.last_activity_ms
|
|
96
|
+
workspace = workspace_folder(ref.path.parent.parent)
|
|
97
|
+
else:
|
|
98
|
+
raw = load_session(ref.path)
|
|
99
|
+
title, start_ms, credits = raw.title, raw.creation_ms, [request.credits for request in raw.requests]
|
|
100
|
+
last_activity_ms = max(
|
|
101
|
+
(
|
|
102
|
+
max(request.timestamp_ms, request.response_timestamp_ms or request.timestamp_ms)
|
|
103
|
+
for request in raw.requests
|
|
104
|
+
),
|
|
105
|
+
default=None,
|
|
106
|
+
)
|
|
107
|
+
if start_ms is None and raw.requests:
|
|
108
|
+
start_ms = raw.requests[0].timestamp_ms
|
|
109
|
+
workspace = None
|
|
110
|
+
return SessionSummary(
|
|
111
|
+
id=ref.id,
|
|
112
|
+
agent=self.name,
|
|
113
|
+
title=title or ref.native_id,
|
|
114
|
+
workspace=workspace,
|
|
115
|
+
start_ms=start_ms if start_ms is not None else ref.mtime * 1000,
|
|
116
|
+
end_ms=ref.mtime * 1000,
|
|
117
|
+
file_size=ref.path.stat().st_size,
|
|
118
|
+
last_activity_ms=last_activity_ms,
|
|
119
|
+
cost_total=sum_costs([credits_cost(value, self._usd_per_credit) for value in credits]),
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
def analyze(self, ref: SessionRef) -> Session:
|
|
123
|
+
raw = load_session(ref.path)
|
|
124
|
+
is_live = ref.path.parent.name == _LIVE_SESSION_DIR
|
|
125
|
+
side = self._side_files(ref, is_live)
|
|
126
|
+
diagnostics = Diagnostics(malformed_lines=raw.malformed_lines)
|
|
127
|
+
sources = ["live-session" if is_live else "export"]
|
|
128
|
+
|
|
129
|
+
transcript: Transcript | None = None
|
|
130
|
+
if side.transcript_file is not None:
|
|
131
|
+
transcript = load_transcript(side.transcript_file)
|
|
132
|
+
sources.append("transcript")
|
|
133
|
+
diagnostics.malformed_lines += transcript.malformed_lines
|
|
134
|
+
else:
|
|
135
|
+
diagnostics.warnings.append("No transcript found: tool timings and LLM call counts are unavailable.")
|
|
136
|
+
|
|
137
|
+
debug_log: DebugLog | None = None
|
|
138
|
+
if side.debug_log_dir is not None:
|
|
139
|
+
debug_log = load_debug_log_dir(side.debug_log_dir, root_session_id=side.debug_log_dir.name)
|
|
140
|
+
sources.append("debug-log")
|
|
141
|
+
diagnostics.malformed_lines += debug_log.malformed_lines
|
|
142
|
+
else:
|
|
143
|
+
diagnostics.warnings.append("No debug log found: token counts are unavailable.")
|
|
144
|
+
|
|
145
|
+
unknown = Counter(
|
|
146
|
+
tool_call.tool_id
|
|
147
|
+
for request in raw.requests
|
|
148
|
+
for tool_call in request.tool_calls
|
|
149
|
+
if not is_known_tool_id(tool_call.tool_id)
|
|
150
|
+
)
|
|
151
|
+
diagnostics.unknown_tool_ids = dict(unknown)
|
|
152
|
+
|
|
153
|
+
return Session(
|
|
154
|
+
id=ref.id,
|
|
155
|
+
agent=self.name,
|
|
156
|
+
title=_title(raw, ref.native_id),
|
|
157
|
+
workspace=workspace_folder(ref.path.parent.parent) if is_live else None,
|
|
158
|
+
root=build_root(raw, transcript, debug_log, self._usd_per_credit),
|
|
159
|
+
sources=sources,
|
|
160
|
+
diagnostics=diagnostics,
|
|
161
|
+
)
|
|
162
|
+
|
|
163
|
+
def _side_files(self, ref: SessionRef, is_live: bool) -> SideFiles:
|
|
164
|
+
if is_live:
|
|
165
|
+
return side_files_in_workspace(ref.path.parent.parent, ref.native_id)
|
|
166
|
+
return find_side_files(self._roots, ref.native_id)
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
# Copyright (c) 2026 epicodic
|
|
3
|
+
"""Parse a `GitHub.copilot-chat/debug-logs/<sid>/` directory into LLM calls per agent."""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass, field, replace
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class DebugLlmCall:
|
|
14
|
+
"""One `llm_request` span. `input_tokens` includes `cached_tokens`, as Copilot logs it.
|
|
15
|
+
|
|
16
|
+
Token counts are `None` when the span does not log them (e.g. a failed or still running request).
|
|
17
|
+
`debug_name` names the request's purpose, e.g. `panel/editAgent` or `summarizeConversationHistory`.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
start_ms: int
|
|
21
|
+
duration_ms: int
|
|
22
|
+
input_tokens: int | None
|
|
23
|
+
output_tokens: int | None
|
|
24
|
+
cached_tokens: int | None
|
|
25
|
+
debug_name: str | None = None
|
|
26
|
+
source_request_id: str | None = None
|
|
27
|
+
requested_tool_ids: tuple[str, ...] = ()
|
|
28
|
+
consumed_tool_ids: tuple[str, ...] = ()
|
|
29
|
+
model: str | None = None
|
|
30
|
+
usage_nano_aiu: int | None = None
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass
|
|
34
|
+
class DebugLog:
|
|
35
|
+
"""LLM calls keyed by the owning agent's `toolCallId` (`None` for the main agent)."""
|
|
36
|
+
|
|
37
|
+
calls: dict[str | None, list[DebugLlmCall]] = field(default_factory=dict)
|
|
38
|
+
malformed_lines: int = 0
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _count(attrs: dict[str, Any], key: str) -> int | None:
|
|
42
|
+
value = attrs.get(key)
|
|
43
|
+
return None if value is None else int(value)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _tool_part_ids(messages: object, part_type: str) -> tuple[str, ...]:
|
|
47
|
+
"""Read typed tool IDs, including complete identity headers before Copilot's truncation marker."""
|
|
48
|
+
if isinstance(messages, str):
|
|
49
|
+
encoded = messages
|
|
50
|
+
try:
|
|
51
|
+
messages = json.loads(encoded)
|
|
52
|
+
except json.JSONDecodeError:
|
|
53
|
+
if not encoded.startswith("[") or not encoded.endswith("[truncated]"):
|
|
54
|
+
return ()
|
|
55
|
+
pattern = r'\{"type"\s*:\s*"' + re.escape(part_type) + r'"\s*,\s*"id"\s*:\s*"([^"\\]+)"'
|
|
56
|
+
return tuple(dict.fromkeys(re.findall(pattern, encoded)))
|
|
57
|
+
if not isinstance(messages, list):
|
|
58
|
+
return ()
|
|
59
|
+
ids: list[str] = []
|
|
60
|
+
for message in messages:
|
|
61
|
+
if not isinstance(message, dict) or not isinstance(message.get("parts"), list):
|
|
62
|
+
continue
|
|
63
|
+
for part in message["parts"]:
|
|
64
|
+
if not isinstance(part, dict) or part.get("type") != part_type:
|
|
65
|
+
continue
|
|
66
|
+
tool_id = part.get("id")
|
|
67
|
+
if isinstance(tool_id, str) and tool_id and tool_id not in ids:
|
|
68
|
+
ids.append(tool_id)
|
|
69
|
+
return tuple(ids)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _parse_call(entry: dict[str, Any]) -> DebugLlmCall:
|
|
73
|
+
attrs = entry["attrs"]
|
|
74
|
+
debug_name = attrs.get("debugName")
|
|
75
|
+
return DebugLlmCall(
|
|
76
|
+
start_ms=int(entry["ts"]),
|
|
77
|
+
duration_ms=int(entry.get("dur", 0)),
|
|
78
|
+
input_tokens=_count(attrs, "inputTokens"),
|
|
79
|
+
output_tokens=_count(attrs, "outputTokens"),
|
|
80
|
+
cached_tokens=_count(attrs, "cachedTokens"),
|
|
81
|
+
debug_name=debug_name if isinstance(debug_name, str) else None,
|
|
82
|
+
model=attrs.get("model") if isinstance(attrs.get("model"), str) else None,
|
|
83
|
+
usage_nano_aiu=_count(attrs, "copilotUsageNanoAiu"),
|
|
84
|
+
source_request_id=entry.get("spanId") if isinstance(entry.get("spanId"), str) else None,
|
|
85
|
+
consumed_tool_ids=_tool_part_ids(attrs.get("inputMessages"), "tool_call_response"),
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def load_debug_log_dir(directory: Path, root_session_id: str) -> DebugLog:
|
|
90
|
+
"""Collect `llm_request` spans from every file in a debug-log directory.
|
|
91
|
+
|
|
92
|
+
Every line carries a top-level `sid`: the root session id for the main agent's file, otherwise the
|
|
93
|
+
subagent's own `toolCallId`, regardless of nesting depth.
|
|
94
|
+
|
|
95
|
+
Args:
|
|
96
|
+
directory: A `<sessionId>/` debug-log directory containing `*.jsonl` files.
|
|
97
|
+
root_session_id: The `sid` used by the main agent's lines.
|
|
98
|
+
|
|
99
|
+
Returns:
|
|
100
|
+
The LLM calls per agent and the number of malformed lines skipped.
|
|
101
|
+
"""
|
|
102
|
+
log = DebugLog()
|
|
103
|
+
responses: dict[tuple[str | None, str], tuple[str, ...]] = {}
|
|
104
|
+
for file_path in sorted(directory.glob("*.jsonl")):
|
|
105
|
+
for line in file_path.read_text().splitlines():
|
|
106
|
+
if not line.strip():
|
|
107
|
+
continue
|
|
108
|
+
try:
|
|
109
|
+
entry = json.loads(line)
|
|
110
|
+
if entry.get("type") not in ("llm_request", "agent_response"):
|
|
111
|
+
continue
|
|
112
|
+
sid = entry["sid"]
|
|
113
|
+
key = None if sid == root_session_id else sid
|
|
114
|
+
if entry["type"] == "agent_response":
|
|
115
|
+
span_id = entry.get("spanId")
|
|
116
|
+
if isinstance(span_id, str) and span_id.startswith("agent-msg-"):
|
|
117
|
+
request_id = span_id.removeprefix("agent-msg-")
|
|
118
|
+
if request_id:
|
|
119
|
+
ids = _tool_part_ids(entry["attrs"].get("response"), "tool_call")
|
|
120
|
+
previous = responses.get((key, request_id), ())
|
|
121
|
+
responses[key, request_id] = tuple(dict.fromkeys((*previous, *ids)))
|
|
122
|
+
continue
|
|
123
|
+
call = _parse_call(entry)
|
|
124
|
+
except (json.JSONDecodeError, AttributeError, KeyError, TypeError, ValueError):
|
|
125
|
+
log.malformed_lines += 1
|
|
126
|
+
continue
|
|
127
|
+
key = None if sid == root_session_id else sid
|
|
128
|
+
log.calls.setdefault(key, []).append(call)
|
|
129
|
+
for owner, calls in log.calls.items():
|
|
130
|
+
log.calls[owner] = [
|
|
131
|
+
replace(call, requested_tool_ids=responses.get((owner, call.source_request_id), ()))
|
|
132
|
+
if call.source_request_id is not None
|
|
133
|
+
else call
|
|
134
|
+
for call in calls
|
|
135
|
+
]
|
|
136
|
+
return log
|