agentprof 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentprof/__init__.py +7 -0
- agentprof/adapters/__init__.py +15 -0
- agentprof/adapters/base.py +79 -0
- agentprof/adapters/claude_code/__init__.py +3 -0
- agentprof/adapters/claude_code/adapter.py +133 -0
- agentprof/adapters/claude_code/discovery.py +79 -0
- agentprof/adapters/claude_code/prompts.py +44 -0
- agentprof/adapters/claude_code/tools.py +102 -0
- agentprof/adapters/claude_code/transcript.py +264 -0
- agentprof/adapters/claude_code/tree.py +492 -0
- agentprof/adapters/codex/__init__.py +3 -0
- agentprof/adapters/codex/adapter.py +164 -0
- agentprof/adapters/codex/discovery.py +117 -0
- agentprof/adapters/codex/prompts.py +16 -0
- agentprof/adapters/codex/rollout.py +328 -0
- agentprof/adapters/codex/tools.py +97 -0
- agentprof/adapters/codex/tree.py +311 -0
- agentprof/adapters/copilot_vscode/__init__.py +3 -0
- agentprof/adapters/copilot_vscode/adapter.py +166 -0
- agentprof/adapters/copilot_vscode/debuglog.py +136 -0
- agentprof/adapters/copilot_vscode/discovery.py +153 -0
- agentprof/adapters/copilot_vscode/session.py +281 -0
- agentprof/adapters/copilot_vscode/tools.py +127 -0
- agentprof/adapters/copilot_vscode/transcript.py +176 -0
- agentprof/adapters/copilot_vscode/tree.py +325 -0
- agentprof/adapters/execution.py +64 -0
- agentprof/adapters/timestamps.py +13 -0
- agentprof/adapters/turns.py +24 -0
- agentprof/analysis/__init__.py +3 -0
- agentprof/analysis/agent_summary.py +161 -0
- agentprof/analysis/call_context.py +101 -0
- agentprof/analysis/evidence_findings.py +151 -0
- agentprof/analysis/execution.py +39 -0
- agentprof/analysis/heuristics.py +306 -0
- agentprof/analysis/pipeline.py +31 -0
- agentprof/analysis/rollup.py +196 -0
- agentprof/cli.py +141 -0
- agentprof/model.py +341 -0
- agentprof/pricing.json +23 -0
- agentprof/pricing.py +120 -0
- agentprof/registry.py +304 -0
- agentprof/server/__init__.py +3 -0
- agentprof/server/app.py +399 -0
- agentprof/server/openapi.py +31 -0
- agentprof/server/pricing_schema.py +40 -0
- agentprof/server/schemas.py +579 -0
- agentprof/server/static/assets/index-C4HEqUVf.css +1 -0
- agentprof/server/static/assets/index-ZsPn1H3d.js +58 -0
- agentprof/server/static/index.html +14 -0
- agentprof-0.1.0.dist-info/METADATA +177 -0
- agentprof-0.1.0.dist-info/RECORD +54 -0
- agentprof-0.1.0.dist-info/WHEEL +4 -0
- agentprof-0.1.0.dist-info/entry_points.txt +8 -0
- agentprof-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
# Copyright (c) 2026 epicodic
|
|
3
|
+
"""Locate Copilot session files and read the few fields the session list needs, cheaply."""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
from urllib.parse import urlparse
|
|
11
|
+
|
|
12
|
+
_PRODUCTS = ("Code", "Code - Insiders")
|
|
13
|
+
|
|
14
|
+
# The only patch lines that change the summary; everything else is skipped without JSON-parsing it.
|
|
15
|
+
_SUMMARY_PATCH = re.compile(r'\{"kind":\s*[12],\s*"k":\s*\["(requests|customTitle)"\]')
|
|
16
|
+
_CREDITS_PATCH = re.compile(r'\{"kind":\s*1,\s*"k":\s*\["requests",\s*(\d+),\s*"copilotCredits"\]')
|
|
17
|
+
_ACTIVITY_PATCH = re.compile(r'\{"kind":\s*1,\s*"k":\s*\["requests",\s*(\d+),\s*"(timestamp|responseTimestamp)"\]')
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class SideFiles:
|
|
22
|
+
"""Optional sources that belong to one session."""
|
|
23
|
+
|
|
24
|
+
transcript_file: Path | None
|
|
25
|
+
debug_log_dir: Path | None
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class SessionState:
|
|
30
|
+
"""Summary fields of a live session file; `credits` holds each request's `copilotCredits`, if known yet."""
|
|
31
|
+
|
|
32
|
+
created_ms: float | None
|
|
33
|
+
title: str
|
|
34
|
+
credits: list[float | None]
|
|
35
|
+
activity_ms: list[float | None]
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def last_activity_ms(self) -> float | None:
|
|
39
|
+
return max((value for value in self.activity_ms if value is not None), default=None)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _credits_of(requests: list[dict[str, Any]]) -> list[float | None]:
|
|
43
|
+
return [request.get("copilotCredits") for request in requests]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _activity_of(requests: list[dict[str, Any]]) -> list[float | None]:
|
|
47
|
+
return [
|
|
48
|
+
max(
|
|
49
|
+
(value for key in ("timestamp", "responseTimestamp") if (value := request.get(key)) is not None),
|
|
50
|
+
default=None,
|
|
51
|
+
)
|
|
52
|
+
for request in requests
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def default_storage_roots(platform: str, home: Path, appdata: Path | None) -> list[Path]:
|
|
57
|
+
"""The `workspaceStorage` directories of VS Code and VS Code Insiders on `platform` (`sys.platform`)."""
|
|
58
|
+
if platform == "darwin":
|
|
59
|
+
base = home / "Library" / "Application Support"
|
|
60
|
+
elif platform == "win32":
|
|
61
|
+
if appdata is None:
|
|
62
|
+
return []
|
|
63
|
+
base = appdata
|
|
64
|
+
else:
|
|
65
|
+
base = home / ".config"
|
|
66
|
+
return [base / product / "User" / "workspaceStorage" for product in _PRODUCTS]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def session_files(root: Path) -> list[Path]:
|
|
70
|
+
"""All live session files under one `workspaceStorage` root."""
|
|
71
|
+
return sorted(root.glob("*/chatSessions/*.jsonl"))
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def workspace_folder(workspace_dir: Path) -> str | None:
|
|
75
|
+
"""The folder a workspace storage directory belongs to, from its `workspace.json`."""
|
|
76
|
+
workspace_json = workspace_dir / "workspace.json"
|
|
77
|
+
if not workspace_json.is_file():
|
|
78
|
+
return None
|
|
79
|
+
try:
|
|
80
|
+
folder_uri = json.loads(workspace_json.read_text()).get("folder")
|
|
81
|
+
except json.JSONDecodeError:
|
|
82
|
+
return None
|
|
83
|
+
if not folder_uri:
|
|
84
|
+
return None
|
|
85
|
+
return urlparse(folder_uri).path
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def read_session_state(session_file: Path) -> SessionState | None:
|
|
89
|
+
"""Read the snapshot line plus the request, title and credit patches, or `None` if the file is malformed."""
|
|
90
|
+
with session_file.open() as lines:
|
|
91
|
+
try:
|
|
92
|
+
snapshot = json.loads(next(lines, ""))["v"]
|
|
93
|
+
except (json.JSONDecodeError, KeyError, TypeError):
|
|
94
|
+
return None
|
|
95
|
+
state = SessionState(
|
|
96
|
+
created_ms=snapshot.get("creationDate"),
|
|
97
|
+
title=str(snapshot.get("customTitle", "")),
|
|
98
|
+
credits=_credits_of(snapshot.get("requests", [])),
|
|
99
|
+
activity_ms=_activity_of(snapshot.get("requests", [])),
|
|
100
|
+
)
|
|
101
|
+
for line in lines:
|
|
102
|
+
_apply_summary_patch(state, line)
|
|
103
|
+
return state
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _apply_summary_patch(state: SessionState, line: str) -> None:
|
|
107
|
+
summary, credits, activity = _SUMMARY_PATCH.match(line), _CREDITS_PATCH.match(line), _ACTIVITY_PATCH.match(line)
|
|
108
|
+
if summary is None and credits is None and activity is None:
|
|
109
|
+
return
|
|
110
|
+
try:
|
|
111
|
+
patch = json.loads(line)
|
|
112
|
+
except json.JSONDecodeError:
|
|
113
|
+
return # in-progress write of the last line
|
|
114
|
+
if credits is not None:
|
|
115
|
+
index = int(credits.group(1))
|
|
116
|
+
if index < len(state.credits):
|
|
117
|
+
state.credits[index] = patch["v"]
|
|
118
|
+
elif activity is not None:
|
|
119
|
+
index = int(activity.group(1))
|
|
120
|
+
if index < len(state.activity_ms):
|
|
121
|
+
state.activity_ms[index] = max(state.activity_ms[index] or patch["v"], patch["v"])
|
|
122
|
+
elif summary is not None and summary.group(1) == "customTitle":
|
|
123
|
+
state.title = str(patch["v"])
|
|
124
|
+
elif patch["kind"] == 1:
|
|
125
|
+
state.credits = _credits_of(patch["v"])
|
|
126
|
+
state.activity_ms = _activity_of(patch["v"])
|
|
127
|
+
else:
|
|
128
|
+
start = patch.get("i")
|
|
129
|
+
state.credits[len(state.credits) if start is None else start :] = _credits_of(patch["v"])
|
|
130
|
+
state.activity_ms[len(state.activity_ms) if start is None else start :] = _activity_of(patch["v"])
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def side_files_in_workspace(workspace_dir: Path, session_id: str) -> SideFiles:
|
|
134
|
+
"""Transcript and debug log of `session_id` inside one workspace storage directory."""
|
|
135
|
+
copilot_dir = workspace_dir / "GitHub.copilot-chat"
|
|
136
|
+
transcript_file = copilot_dir / "transcripts" / f"{session_id}.jsonl"
|
|
137
|
+
debug_log_dir = copilot_dir / "debug-logs" / session_id
|
|
138
|
+
return SideFiles(
|
|
139
|
+
transcript_file=transcript_file if transcript_file.is_file() else None,
|
|
140
|
+
debug_log_dir=debug_log_dir if debug_log_dir.is_dir() else None,
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def find_side_files(roots: list[Path], session_id: str) -> SideFiles:
|
|
145
|
+
"""Search every workspace under `roots` for the transcript and debug log of `session_id`."""
|
|
146
|
+
for root in roots:
|
|
147
|
+
if not root.is_dir():
|
|
148
|
+
continue
|
|
149
|
+
for workspace_dir in sorted(path for path in root.iterdir() if path.is_dir()):
|
|
150
|
+
side = side_files_in_workspace(workspace_dir, session_id)
|
|
151
|
+
if side.transcript_file is not None or side.debug_log_dir is not None:
|
|
152
|
+
return side
|
|
153
|
+
return SideFiles(transcript_file=None, debug_log_dir=None)
|
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
# Copyright (c) 2026 epicodic
|
|
3
|
+
"""Parse a Copilot Chat export (`.json`) or live session (`chatSessions/<sid>.jsonl`)."""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
import string
|
|
8
|
+
from collections.abc import Iterable
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
from urllib.parse import unquote
|
|
13
|
+
|
|
14
|
+
from agentprof.adapters.copilot_vscode.tools import is_subagent_tool_id
|
|
15
|
+
|
|
16
|
+
_VSCODE_ID_SUFFIX = re.compile(r"__vscode-\d+$")
|
|
17
|
+
_MARKDOWN_LINK = re.compile(r"\[([^\]]*)\]\(([^)]*)\)")
|
|
18
|
+
_BACKSLASH_ESCAPE = re.compile(f"\\\\([{re.escape(string.punctuation)}])")
|
|
19
|
+
_FILE_SCHEME = "file://"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class RawToolCall:
|
|
24
|
+
"""One `toolInvocationSerialized` response part."""
|
|
25
|
+
|
|
26
|
+
tool_call_id: str
|
|
27
|
+
tool_id: str
|
|
28
|
+
topic: str
|
|
29
|
+
parent_tool_call_id: str | None
|
|
30
|
+
is_agent: bool = False
|
|
31
|
+
subagent_description: str | None = None
|
|
32
|
+
subagent_prompt: str | None = None
|
|
33
|
+
subagent_model: str | None = None
|
|
34
|
+
subagent_result: str | None = None
|
|
35
|
+
subagent_credits: float | None = None
|
|
36
|
+
arguments: dict[str, object] = field(default_factory=dict)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class RawRequest:
|
|
41
|
+
"""One user turn, with its tool calls in response order."""
|
|
42
|
+
|
|
43
|
+
request_id: str
|
|
44
|
+
text: str
|
|
45
|
+
timestamp_ms: int
|
|
46
|
+
model_id: str | None
|
|
47
|
+
session_id: str | None
|
|
48
|
+
credits: float | None
|
|
49
|
+
prompt_tokens: int | None
|
|
50
|
+
completion_tokens: int | None
|
|
51
|
+
time_spent_waiting_ms: int | None
|
|
52
|
+
elapsed_ms: int | None
|
|
53
|
+
response_timestamp_ms: int | None = None
|
|
54
|
+
round_timestamps_ms: list[int] = field(default_factory=list)
|
|
55
|
+
tool_calls: list[RawToolCall] = field(default_factory=list)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class RawSession:
|
|
60
|
+
"""A parsed export or replayed live session."""
|
|
61
|
+
|
|
62
|
+
session_id: str | None
|
|
63
|
+
title: str
|
|
64
|
+
creation_ms: int | None
|
|
65
|
+
requests: list[RawRequest]
|
|
66
|
+
malformed_lines: int = 0
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _markdown_link_replacement(match: re.Match[str]) -> str:
|
|
70
|
+
label, target = match.group(1), match.group(2)
|
|
71
|
+
if label:
|
|
72
|
+
return label
|
|
73
|
+
if target.startswith(_FILE_SCHEME):
|
|
74
|
+
path = target[len(_FILE_SCHEME) :].split("#", 1)[0]
|
|
75
|
+
return unquote(path)
|
|
76
|
+
return target
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def clean_topic(text: str) -> str:
|
|
80
|
+
"""Render Copilot's Markdown `invocationMessage` down to a plain topic string."""
|
|
81
|
+
text = _MARKDOWN_LINK.sub(_markdown_link_replacement, text)
|
|
82
|
+
return _BACKSLASH_ESCAPE.sub(r"\1", text)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _topic_of(part: dict[str, Any]) -> str:
|
|
86
|
+
message = part.get("invocationMessage")
|
|
87
|
+
if isinstance(message, dict):
|
|
88
|
+
return clean_topic(str(message.get("value", "")))
|
|
89
|
+
return clean_topic(str(message or ""))
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _parse_tool_call(part: dict[str, Any], round_arguments: dict[str, dict[str, object]]) -> RawToolCall:
|
|
93
|
+
tool_specific = part.get("toolSpecificData") or {}
|
|
94
|
+
return RawToolCall(
|
|
95
|
+
tool_call_id=part["toolCallId"],
|
|
96
|
+
tool_id=part["toolId"],
|
|
97
|
+
topic=_topic_of(part),
|
|
98
|
+
parent_tool_call_id=part.get("subAgentInvocationId"),
|
|
99
|
+
subagent_description=tool_specific.get("description"),
|
|
100
|
+
subagent_prompt=tool_specific.get("prompt"),
|
|
101
|
+
subagent_model=tool_specific.get("modelName"),
|
|
102
|
+
subagent_result=tool_specific.get("result"),
|
|
103
|
+
subagent_credits=tool_specific.get("credits"),
|
|
104
|
+
arguments=round_arguments.get(part["toolCallId"], {}),
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _deduplicate_tool_calls(tool_calls: Iterable[RawToolCall]) -> list[RawToolCall]:
|
|
109
|
+
"""Copilot re-serializes live-updating tool calls; keep the first occurrence's position but the last's data."""
|
|
110
|
+
by_id: dict[str, RawToolCall] = {}
|
|
111
|
+
for tool_call in tool_calls:
|
|
112
|
+
by_id[tool_call.tool_call_id] = tool_call
|
|
113
|
+
return list(by_id.values())
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _parse_round_arguments(rounds: list[dict[str, Any]]) -> dict[str, dict[str, object]]:
|
|
117
|
+
"""Map each round-dispatched tool call's id to its parsed arguments (a JSON string in the raw data).
|
|
118
|
+
|
|
119
|
+
Live sessions suffix the round id with `__vscode-<digits>` (e.g. `toolu_...__vscode-1779116386890`) while
|
|
120
|
+
the response part's `toolCallId` carries the bare id; that suffix is stripped so the two line up.
|
|
121
|
+
"""
|
|
122
|
+
arguments_by_id: dict[str, dict[str, object]] = {}
|
|
123
|
+
for round_ in rounds:
|
|
124
|
+
for tool_call in round_.get("toolCalls", []):
|
|
125
|
+
raw_arguments = tool_call.get("arguments")
|
|
126
|
+
if not raw_arguments:
|
|
127
|
+
continue
|
|
128
|
+
try:
|
|
129
|
+
parsed = json.loads(raw_arguments)
|
|
130
|
+
except json.JSONDecodeError:
|
|
131
|
+
continue
|
|
132
|
+
arguments_by_id[_VSCODE_ID_SUFFIX.sub("", tool_call["id"])] = parsed
|
|
133
|
+
return arguments_by_id
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _parse_request(raw: dict[str, Any]) -> RawRequest:
|
|
137
|
+
metadata = raw.get("result", {}).get("metadata", {})
|
|
138
|
+
rounds = metadata.get("toolCallRounds", [])
|
|
139
|
+
round_arguments = _parse_round_arguments(rounds)
|
|
140
|
+
|
|
141
|
+
tool_calls = _deduplicate_tool_calls(
|
|
142
|
+
_parse_tool_call(part, round_arguments)
|
|
143
|
+
for part in raw.get("response", [])
|
|
144
|
+
if part.get("kind") == "toolInvocationSerialized" and "toolCallId" in part
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
# A tool call is an agent if its id says so, or if another tool call names it as its parent.
|
|
148
|
+
parent_ids = {tc.parent_tool_call_id for tc in tool_calls if tc.parent_tool_call_id}
|
|
149
|
+
for tool_call in tool_calls:
|
|
150
|
+
if is_subagent_tool_id(tool_call.tool_id) or tool_call.tool_call_id in parent_ids:
|
|
151
|
+
tool_call.is_agent = True
|
|
152
|
+
|
|
153
|
+
return RawRequest(
|
|
154
|
+
request_id=raw["requestId"],
|
|
155
|
+
text=raw.get("message", {}).get("text", ""),
|
|
156
|
+
timestamp_ms=raw["timestamp"],
|
|
157
|
+
model_id=raw.get("modelId"),
|
|
158
|
+
session_id=metadata.get("sessionId"),
|
|
159
|
+
credits=raw.get("copilotCredits"),
|
|
160
|
+
prompt_tokens=raw.get("promptTokens"),
|
|
161
|
+
completion_tokens=raw.get("completionTokens"),
|
|
162
|
+
time_spent_waiting_ms=raw.get("timeSpentWaiting"),
|
|
163
|
+
elapsed_ms=raw.get("elapsedMs"),
|
|
164
|
+
response_timestamp_ms=raw.get("responseTimestamp"),
|
|
165
|
+
round_timestamps_ms=[r["timestamp"] for r in rounds if "timestamp" in r],
|
|
166
|
+
tool_calls=tool_calls,
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def parse_export(data: dict[str, Any]) -> list[RawRequest]:
|
|
171
|
+
"""Parse the `requests` of an export (or a replayed live session)."""
|
|
172
|
+
return [_parse_request(raw) for raw in data.get("requests", [])]
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _deduplicate_across_requests(requests: list[RawRequest]) -> None:
|
|
176
|
+
"""Drop a tool call from a later request if its id already appeared in an earlier request of the session.
|
|
177
|
+
|
|
178
|
+
Copilot occasionally re-lists a long-running tool call (e.g. a background terminal command) under the
|
|
179
|
+
same `toolCallId` in a follow-up turn; the request that first listed it keeps it.
|
|
180
|
+
"""
|
|
181
|
+
seen_ids: set[str] = set()
|
|
182
|
+
for request in requests:
|
|
183
|
+
request.tool_calls = [tc for tc in request.tool_calls if tc.tool_call_id not in seen_ids]
|
|
184
|
+
seen_ids.update(tc.tool_call_id for tc in request.tool_calls)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def parse_session_data(data: dict[str, Any], malformed_lines: int = 0) -> RawSession:
|
|
188
|
+
"""Parse a whole export (or replayed live session) including its session-level fields."""
|
|
189
|
+
requests = parse_export(data)
|
|
190
|
+
_deduplicate_across_requests(requests)
|
|
191
|
+
session_id = next((r.session_id for r in requests if r.session_id), None) or data.get("sessionId")
|
|
192
|
+
return RawSession(
|
|
193
|
+
session_id=session_id,
|
|
194
|
+
title=str(data.get("customTitle") or ""),
|
|
195
|
+
creation_ms=data.get("creationDate"),
|
|
196
|
+
requests=requests,
|
|
197
|
+
malformed_lines=malformed_lines,
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _apply_patch(snapshot: dict[str, Any], patch: dict[str, Any]) -> None:
|
|
202
|
+
"""Apply one entry: `kind` 1 replaces the value at `k`; `kind` 2 appends to (or splices via `i`) the array."""
|
|
203
|
+
path = patch["k"]
|
|
204
|
+
target: Any = snapshot
|
|
205
|
+
for key in path[:-1]:
|
|
206
|
+
target = target[key]
|
|
207
|
+
last_key = path[-1]
|
|
208
|
+
|
|
209
|
+
if patch["kind"] == 1:
|
|
210
|
+
target[last_key] = patch["v"]
|
|
211
|
+
return
|
|
212
|
+
|
|
213
|
+
array = target[last_key]
|
|
214
|
+
start = patch.get("i")
|
|
215
|
+
if start is None:
|
|
216
|
+
array.extend(patch.get("v", []))
|
|
217
|
+
else:
|
|
218
|
+
array[start:] = patch.get("v", [])
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _decode_line(line: str) -> tuple[list[dict[str, Any]], bool]:
|
|
222
|
+
"""Decode a JSONL line that may hold several JSON objects written back to back without a separator.
|
|
223
|
+
|
|
224
|
+
Returns the decoded objects and whether the line was malformed: either an undecodable remainder was
|
|
225
|
+
left on the line (objects decoded before it are kept), or a decoded value was not a JSON object.
|
|
226
|
+
"""
|
|
227
|
+
decoder = json.JSONDecoder()
|
|
228
|
+
objects: list[dict[str, Any]] = []
|
|
229
|
+
malformed = False
|
|
230
|
+
index = 0
|
|
231
|
+
length = len(line)
|
|
232
|
+
while index < length:
|
|
233
|
+
while index < length and line[index].isspace():
|
|
234
|
+
index += 1
|
|
235
|
+
if index >= length:
|
|
236
|
+
break
|
|
237
|
+
try:
|
|
238
|
+
value, index = decoder.raw_decode(line, index)
|
|
239
|
+
except json.JSONDecodeError:
|
|
240
|
+
malformed = True
|
|
241
|
+
break
|
|
242
|
+
if not isinstance(value, dict):
|
|
243
|
+
malformed = True
|
|
244
|
+
break
|
|
245
|
+
objects.append(value)
|
|
246
|
+
return objects, malformed
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _replay_live_session(lines: list[str]) -> tuple[dict[str, Any], int]:
|
|
250
|
+
"""Replay snapshot and patches; return the reconstructed data and the number of malformed lines skipped."""
|
|
251
|
+
entries: list[dict[str, Any]] = []
|
|
252
|
+
malformed_lines = 0
|
|
253
|
+
for line in lines:
|
|
254
|
+
if not line.strip():
|
|
255
|
+
continue
|
|
256
|
+
objects, malformed = _decode_line(line)
|
|
257
|
+
entries.extend(objects)
|
|
258
|
+
if malformed:
|
|
259
|
+
malformed_lines += 1 # e.g. an in-progress write, or two records concatenated without a newline
|
|
260
|
+
|
|
261
|
+
if not entries or entries[0].get("kind") != 0:
|
|
262
|
+
raise ValueError("live session file must start with a kind:0 snapshot")
|
|
263
|
+
|
|
264
|
+
snapshot = entries[0]["v"]
|
|
265
|
+
for patch in entries[1:]:
|
|
266
|
+
# Other patch kinds are deliberately ignored: not needed to reconstruct request data.
|
|
267
|
+
if patch.get("kind") not in (1, 2):
|
|
268
|
+
continue
|
|
269
|
+
try:
|
|
270
|
+
_apply_patch(snapshot, patch)
|
|
271
|
+
except (KeyError, IndexError, TypeError):
|
|
272
|
+
malformed_lines += 1 # e.g. a patch targeting a path dropped by an earlier malformed line
|
|
273
|
+
return snapshot, malformed_lines
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def load_session(path: Path) -> RawSession:
|
|
277
|
+
"""Load an export (`.json`) or a live `chatSessions` file (`.jsonl`)."""
|
|
278
|
+
if path.suffix == ".json":
|
|
279
|
+
return parse_session_data(json.loads(path.read_text()))
|
|
280
|
+
snapshot, malformed_lines = _replay_live_session(path.read_text().splitlines())
|
|
281
|
+
return parse_session_data(snapshot, malformed_lines=malformed_lines)
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
# SPDX-License-Identifier: MIT
|
|
2
|
+
# Copyright (c) 2026 epicodic
|
|
3
|
+
"""Map Copilot tool ids to neutral tool categories and normalised arguments.
|
|
4
|
+
|
|
5
|
+
Copilot uses two names per tool: the `toolId` in the session file (e.g. `copilot_readFile`) and the function
|
|
6
|
+
name in `toolCallRounds` and the transcript (e.g. `read_file`). Both are mapped.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from agentprof.model import ToolCategory, ToolInfo
|
|
10
|
+
|
|
11
|
+
_CATEGORIES: dict[str, ToolCategory] = {
|
|
12
|
+
"copilot_readFile": ToolCategory.READ,
|
|
13
|
+
"read_file": ToolCategory.READ,
|
|
14
|
+
"copilot_replaceString": ToolCategory.EDIT,
|
|
15
|
+
"replace_string_in_file": ToolCategory.EDIT,
|
|
16
|
+
"copilot_multiReplaceString": ToolCategory.EDIT,
|
|
17
|
+
"multi_replace_string_in_file": ToolCategory.EDIT,
|
|
18
|
+
"copilot_createFile": ToolCategory.EDIT,
|
|
19
|
+
"create_file": ToolCategory.EDIT,
|
|
20
|
+
"copilot_applyPatch": ToolCategory.EDIT,
|
|
21
|
+
"apply_patch": ToolCategory.EDIT,
|
|
22
|
+
"copilot_insertEdit": ToolCategory.EDIT,
|
|
23
|
+
"insert_edit_into_file": ToolCategory.EDIT,
|
|
24
|
+
"copilot_findTextInFiles": ToolCategory.SEARCH,
|
|
25
|
+
"grep_search": ToolCategory.SEARCH,
|
|
26
|
+
"copilot_findFiles": ToolCategory.SEARCH,
|
|
27
|
+
"file_search": ToolCategory.SEARCH,
|
|
28
|
+
"copilot_listDirectory": ToolCategory.SEARCH,
|
|
29
|
+
"list_dir": ToolCategory.SEARCH,
|
|
30
|
+
"copilot_searchCodebase": ToolCategory.SEARCH,
|
|
31
|
+
"semantic_search": ToolCategory.SEARCH,
|
|
32
|
+
"run_in_terminal": ToolCategory.SHELL,
|
|
33
|
+
"run_task": ToolCategory.SHELL,
|
|
34
|
+
"runTests": ToolCategory.SHELL,
|
|
35
|
+
"get_terminal_output": ToolCategory.SHELL_POLL,
|
|
36
|
+
"send_to_terminal": ToolCategory.SHELL_POLL,
|
|
37
|
+
"terminal_last_command": ToolCategory.SHELL_POLL,
|
|
38
|
+
"copilot_fetchWebPage": ToolCategory.WEB,
|
|
39
|
+
"fetch_webpage": ToolCategory.WEB,
|
|
40
|
+
"vscode_fetchWebPage_internal": ToolCategory.WEB,
|
|
41
|
+
"vscode_renameSymbol": ToolCategory.EDIT,
|
|
42
|
+
"vscode_listCodeUsages": ToolCategory.SEARCH,
|
|
43
|
+
"copilot_getChangedFiles": ToolCategory.SEARCH,
|
|
44
|
+
"Run in Terminal": ToolCategory.SHELL,
|
|
45
|
+
"copilot_viewImage": ToolCategory.READ,
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
# Expected tools without a dedicated category; they map to `other` but are not reported as unknown.
|
|
49
|
+
_KNOWN_OTHER_TOOL_IDS = {
|
|
50
|
+
"manage_todo_list",
|
|
51
|
+
"tool_search",
|
|
52
|
+
"vscode_askQuestions",
|
|
53
|
+
"copilot_getErrors",
|
|
54
|
+
"get_errors",
|
|
55
|
+
"copilot_memory",
|
|
56
|
+
"copilot_sessionStoreSql",
|
|
57
|
+
"kill_terminal",
|
|
58
|
+
"configure_python_environment",
|
|
59
|
+
"copilot_runVscodeCommand",
|
|
60
|
+
"run_playwright_code",
|
|
61
|
+
"screenshot_page",
|
|
62
|
+
"open_browser_page",
|
|
63
|
+
}
|
|
64
|
+
_MCP_PREFIX = "mcp_"
|
|
65
|
+
_GITHUB_PULL_REQUEST_PREFIX = "github-pull-request_"
|
|
66
|
+
_FILE_CATEGORIES = (ToolCategory.READ, ToolCategory.EDIT)
|
|
67
|
+
_WHOLE_FILE_WRITERS = {"copilot_createFile", "create_file"}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def is_subagent_tool_id(tool_id: str) -> bool:
|
|
71
|
+
"""Whether `tool_id` always denotes a subagent dispatch."""
|
|
72
|
+
return tool_id == "runSubagent" or tool_id.endswith("_subagent")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def is_known_tool_id(tool_id: str) -> bool:
|
|
76
|
+
"""Whether `tool_id` is expected; unknown ids are counted in the session diagnostics."""
|
|
77
|
+
return (
|
|
78
|
+
tool_id in _CATEGORIES
|
|
79
|
+
or tool_id in _KNOWN_OTHER_TOOL_IDS
|
|
80
|
+
or tool_id.startswith(_MCP_PREFIX)
|
|
81
|
+
or tool_id.startswith(_GITHUB_PULL_REQUEST_PREFIX)
|
|
82
|
+
or is_subagent_tool_id(tool_id)
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _category(tool_id: str) -> ToolCategory:
|
|
87
|
+
if is_subagent_tool_id(tool_id):
|
|
88
|
+
return ToolCategory.SUBAGENT
|
|
89
|
+
return _CATEGORIES.get(tool_id, ToolCategory.OTHER)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _int_or_none(value: object) -> int | None:
|
|
93
|
+
return value if isinstance(value, int) and not isinstance(value, bool) else None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _paths(arguments: dict[str, object]) -> tuple[str, ...]:
|
|
97
|
+
"""The files named by the arguments: a top-level path, or each `replacements` entry's (multi-replace)."""
|
|
98
|
+
path = arguments.get("filePath") or arguments.get("path")
|
|
99
|
+
if path:
|
|
100
|
+
return (str(path),)
|
|
101
|
+
replacements = arguments.get("replacements")
|
|
102
|
+
if not isinstance(replacements, list):
|
|
103
|
+
return ()
|
|
104
|
+
paths = (entry.get("filePath") for entry in replacements if isinstance(entry, dict))
|
|
105
|
+
return tuple(dict.fromkeys(str(path) for path in paths if path))
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def tool_info(tool_id: str, arguments: dict[str, object]) -> ToolInfo:
|
|
109
|
+
"""Build the neutral `ToolInfo` for one Copilot tool call."""
|
|
110
|
+
category = _category(tool_id)
|
|
111
|
+
paths = _paths(arguments) if category in _FILE_CATEGORIES else ()
|
|
112
|
+
start_line = _int_or_none(arguments.get("startLine"))
|
|
113
|
+
end_line = _int_or_none(arguments.get("endLine"))
|
|
114
|
+
command = arguments.get("command")
|
|
115
|
+
line_range = None
|
|
116
|
+
if category is ToolCategory.READ and start_line is not None and end_line is not None:
|
|
117
|
+
line_range = (start_line, end_line)
|
|
118
|
+
return ToolInfo(
|
|
119
|
+
native_id=tool_id,
|
|
120
|
+
category=category,
|
|
121
|
+
path=paths[0] if paths else None,
|
|
122
|
+
line_range=line_range,
|
|
123
|
+
command=str(command) if command and category is ToolCategory.SHELL else None,
|
|
124
|
+
writes_file=tool_id in _WHOLE_FILE_WRITERS,
|
|
125
|
+
arguments=arguments,
|
|
126
|
+
paths=paths,
|
|
127
|
+
)
|