davinci-resolve-mcp 2.205.2 → 2.207.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +103 -0
- package/README.md +30 -1
- package/README.zh-CN.md +14 -2
- package/docs/SKILL.md +93 -0
- package/install.py +1 -1
- package/package.json +1 -1
- package/src/granular/common.py +1 -1
- package/src/server.py +159 -6
- package/src/utils/execution_trace.py +795 -0
- package/src/utils/operation_result.py +18 -0
|
@@ -0,0 +1,795 @@
|
|
|
1
|
+
"""Agent execution tracing and observability ("Why did the editor do this?").
|
|
2
|
+
|
|
3
|
+
Translates agent execution tracing concepts to DaVinci Resolve MCP.
|
|
4
|
+
Connects multi-step tool calls, timing (duration_ms), semantic changes, and
|
|
5
|
+
readback verifications under unified execution traces so human editors and AI
|
|
6
|
+
agents can inspect, debug, and understand AI editorial workflows.
|
|
7
|
+
|
|
8
|
+
Key capabilities:
|
|
9
|
+
1. Correlated execution traces across multi-step agent actions.
|
|
10
|
+
2. Per-tool timing (duration_ms) and invocation counts.
|
|
11
|
+
3. Semantic change aggregation (e.g. items_deleted, items_added).
|
|
12
|
+
4. Cumulative verification rollup (passed, checks, contradictions).
|
|
13
|
+
5. In-memory thread-safe ring buffer with fast queries:
|
|
14
|
+
- get_execution_trace(execution_id?) / get_execution(id)
|
|
15
|
+
- list_recent_executions(limit?)
|
|
16
|
+
- begin_execution(request?, execution_id?)
|
|
17
|
+
- end_execution(execution_id?, verification?)
|
|
18
|
+
6. Best-effort append-only persistence beside the server's own log.
|
|
19
|
+
|
|
20
|
+
Adapted from the design contributed in PR #183.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import collections
|
|
26
|
+
import json
|
|
27
|
+
import logging
|
|
28
|
+
import os
|
|
29
|
+
import re
|
|
30
|
+
import threading
|
|
31
|
+
import time
|
|
32
|
+
import uuid
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
from typing import Any, Deque, Dict, List, Optional, Union
|
|
35
|
+
|
|
36
|
+
logger = logging.getLogger("resolve-mcp.execution-trace")
|
|
37
|
+
|
|
38
|
+
MAX_RECENT_EXECUTIONS = 100
|
|
39
|
+
|
|
40
|
+
#: Where traces are written, unless RESOLVE_MCP_TRACE_FILE overrides it.
|
|
41
|
+
#:
|
|
42
|
+
#: Anchored to the repository root, the way `server.log`,
|
|
43
|
+
#: `media-analysis-preferences.json` and `server-preferences.json` all are.
|
|
44
|
+
#: Deriving it from `os.getcwd()` instead put the file wherever the MCP client
|
|
45
|
+
#: happened to launch the server from — and since the generated client configs
|
|
46
|
+
#: set no `cwd`, that is usually a directory with no `logs/` in it, where the
|
|
47
|
+
#: original code returned None and wrote nothing at all. Silently. A feature
|
|
48
|
+
#: whose whole purpose is answering "why did the editor do this?" is worth
|
|
49
|
+
#: rather more than a file that may or may not exist depending on the launcher.
|
|
50
|
+
_REPO_ROOT = Path(__file__).resolve().parents[2]
|
|
51
|
+
|
|
52
|
+
#: Size at which the trace log is rotated, and how many old files are kept.
|
|
53
|
+
#:
|
|
54
|
+
#: The in-memory ring is capped at 100 executions; the file had no bound at all,
|
|
55
|
+
#: and it takes one append per tool call. A default-on log that grows without
|
|
56
|
+
#: limit on a working editorial machine is a slow leak, so it rolls over to
|
|
57
|
+
#: `.1` and starts fresh — one generation back is enough for "what did the
|
|
58
|
+
#: agent just do", which is the whole question this feature answers.
|
|
59
|
+
MAX_TRACE_LOG_BYTES = 8 * 1024 * 1024
|
|
60
|
+
TRACE_LOG_GENERATIONS = 1
|
|
61
|
+
|
|
62
|
+
_LOCK = threading.Lock()
|
|
63
|
+
_EXECUTIONS_BY_ID: Dict[str, Dict[str, Any]] = {}
|
|
64
|
+
_RECENT_ORDER: Deque[str] = collections.deque(maxlen=MAX_RECENT_EXECUTIONS)
|
|
65
|
+
_ACTIVE_EXECUTION_ID: Optional[str] = None
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _now_iso() -> str:
|
|
69
|
+
return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def new_execution_id() -> str:
|
|
73
|
+
"""Generate a correlated execution ID with prefix 'exec_'."""
|
|
74
|
+
return f"exec_{uuid.uuid4().hex[:12]}"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def current_execution_id() -> Optional[str]:
|
|
78
|
+
"""Return the active multi-turn execution ID for this session, if any."""
|
|
79
|
+
with _LOCK:
|
|
80
|
+
return _ACTIVE_EXECUTION_ID
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _clean_trace_dict(trace: Dict[str, Any]) -> Dict[str, Any]:
|
|
84
|
+
"""Return a deep copy of the execution trace for safe serialization."""
|
|
85
|
+
return json.loads(json.dumps(trace))
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _aggregate_changes(
|
|
89
|
+
cumulative: Optional[Dict[str, Any]],
|
|
90
|
+
delta: Optional[Dict[str, Any]],
|
|
91
|
+
) -> Optional[Dict[str, Any]]:
|
|
92
|
+
if not delta:
|
|
93
|
+
return cumulative
|
|
94
|
+
if cumulative is None:
|
|
95
|
+
return dict(delta)
|
|
96
|
+
|
|
97
|
+
out = dict(cumulative)
|
|
98
|
+
for k, v in delta.items():
|
|
99
|
+
if isinstance(v, (int, float)):
|
|
100
|
+
existing = out.get(k, 0)
|
|
101
|
+
if isinstance(existing, (int, float)):
|
|
102
|
+
out[k] = existing + v
|
|
103
|
+
else:
|
|
104
|
+
out[k] = v
|
|
105
|
+
else:
|
|
106
|
+
out[k] = v
|
|
107
|
+
return out
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _merge_verification(
|
|
111
|
+
current: Dict[str, Any],
|
|
112
|
+
step_verif: Optional[Dict[str, Any]],
|
|
113
|
+
) -> Dict[str, Any]:
|
|
114
|
+
if not isinstance(step_verif, dict):
|
|
115
|
+
return current
|
|
116
|
+
|
|
117
|
+
checks = list(current.get("checks", []))
|
|
118
|
+
new_checks = step_verif.get("checks", [])
|
|
119
|
+
if isinstance(new_checks, list):
|
|
120
|
+
checks.extend(new_checks)
|
|
121
|
+
|
|
122
|
+
contradiction = bool(current.get("contradiction")) or bool(step_verif.get("contradiction"))
|
|
123
|
+
|
|
124
|
+
# Determine status priority: contradiction > failed > partial > passed > unverified
|
|
125
|
+
statuses = {current.get("status", "unverified"), step_verif.get("status", "unverified")}
|
|
126
|
+
if contradiction or "contradiction" in statuses:
|
|
127
|
+
status = "contradiction"
|
|
128
|
+
passed = False
|
|
129
|
+
elif "failed" in statuses:
|
|
130
|
+
status = "failed"
|
|
131
|
+
passed = False
|
|
132
|
+
elif "partial" in statuses:
|
|
133
|
+
status = "partial"
|
|
134
|
+
passed = False
|
|
135
|
+
elif "passed" in statuses:
|
|
136
|
+
status = "passed"
|
|
137
|
+
passed = True
|
|
138
|
+
else:
|
|
139
|
+
# None, not True. Nothing reported any evidence either way, and
|
|
140
|
+
# collapsing that into a boolean makes the rollup assert a pass it
|
|
141
|
+
# never observed — which then reaches a human as "Passed: yes" in an
|
|
142
|
+
# exported audit report.
|
|
143
|
+
status = "unverified"
|
|
144
|
+
passed = None
|
|
145
|
+
|
|
146
|
+
return {
|
|
147
|
+
"status": status,
|
|
148
|
+
"passed": passed,
|
|
149
|
+
"contradiction": contradiction,
|
|
150
|
+
"checks": checks,
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def begin_execution(
|
|
155
|
+
request: Optional[str] = None,
|
|
156
|
+
*,
|
|
157
|
+
execution_id: Optional[str] = None,
|
|
158
|
+
initiator: Optional[str] = None,
|
|
159
|
+
) -> Dict[str, Any]:
|
|
160
|
+
"""Start an active multi-step execution trace for the session.
|
|
161
|
+
|
|
162
|
+
Subsequent tool calls will automatically thread under this execution ID
|
|
163
|
+
until end_execution() is called.
|
|
164
|
+
"""
|
|
165
|
+
global _ACTIVE_EXECUTION_ID
|
|
166
|
+
|
|
167
|
+
exec_id = execution_id or new_execution_id()
|
|
168
|
+
now = _now_iso()
|
|
169
|
+
|
|
170
|
+
trace: Dict[str, Any] = {
|
|
171
|
+
"execution_id": exec_id,
|
|
172
|
+
"request": str(request).strip() if request else None,
|
|
173
|
+
"status": "running",
|
|
174
|
+
"started_at": now,
|
|
175
|
+
"ended_at": None,
|
|
176
|
+
"duration_ms": 0,
|
|
177
|
+
"tools": [],
|
|
178
|
+
"steps": [],
|
|
179
|
+
"changes": None,
|
|
180
|
+
"verification": {
|
|
181
|
+
"status": "unverified",
|
|
182
|
+
# None, not True. "Nothing was checked" is not "everything passed",
|
|
183
|
+
# and the difference reaches a human in an exported audit report —
|
|
184
|
+
# see _execution_report_markdown.
|
|
185
|
+
"passed": None,
|
|
186
|
+
"contradiction": False,
|
|
187
|
+
"checks": [],
|
|
188
|
+
},
|
|
189
|
+
"warnings": [],
|
|
190
|
+
"initiator": initiator or "agent",
|
|
191
|
+
"is_active": True,
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
with _LOCK:
|
|
195
|
+
_EXECUTIONS_BY_ID[exec_id] = trace
|
|
196
|
+
if exec_id in _RECENT_ORDER:
|
|
197
|
+
_RECENT_ORDER.remove(exec_id)
|
|
198
|
+
_RECENT_ORDER.appendleft(exec_id)
|
|
199
|
+
_ACTIVE_EXECUTION_ID = exec_id
|
|
200
|
+
|
|
201
|
+
_persist_trace_event("begin", trace)
|
|
202
|
+
return {
|
|
203
|
+
"success": True,
|
|
204
|
+
"execution_id": exec_id,
|
|
205
|
+
"started_at": now,
|
|
206
|
+
"request": trace["request"],
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def end_execution(
|
|
211
|
+
execution_id: Optional[str] = None,
|
|
212
|
+
*,
|
|
213
|
+
verification: Optional[Dict[str, Any]] = None,
|
|
214
|
+
status: Optional[str] = None,
|
|
215
|
+
notes: Optional[str] = None,
|
|
216
|
+
) -> Optional[Dict[str, Any]]:
|
|
217
|
+
"""End an active execution trace and calculate final rollups."""
|
|
218
|
+
global _ACTIVE_EXECUTION_ID
|
|
219
|
+
|
|
220
|
+
target_id = execution_id or _ACTIVE_EXECUTION_ID
|
|
221
|
+
if not target_id:
|
|
222
|
+
return None
|
|
223
|
+
|
|
224
|
+
now = _now_iso()
|
|
225
|
+
with _LOCK:
|
|
226
|
+
trace = _EXECUTIONS_BY_ID.get(target_id)
|
|
227
|
+
if not trace:
|
|
228
|
+
return None
|
|
229
|
+
|
|
230
|
+
trace["ended_at"] = now
|
|
231
|
+
trace["is_active"] = False
|
|
232
|
+
|
|
233
|
+
if verification:
|
|
234
|
+
trace["verification"] = _merge_verification(trace["verification"], verification)
|
|
235
|
+
|
|
236
|
+
if notes:
|
|
237
|
+
trace["notes"] = str(notes).strip()
|
|
238
|
+
|
|
239
|
+
# Deduce overall status if not explicitly passed
|
|
240
|
+
if status:
|
|
241
|
+
trace["status"] = status
|
|
242
|
+
else:
|
|
243
|
+
if trace["verification"].get("contradiction"):
|
|
244
|
+
trace["status"] = "failed"
|
|
245
|
+
elif any(s.get("status") == "failed" for s in trace.get("steps", [])):
|
|
246
|
+
trace["status"] = "failed"
|
|
247
|
+
elif any(s.get("status") == "partial" for s in trace.get("steps", [])):
|
|
248
|
+
trace["status"] = "partial"
|
|
249
|
+
elif any(s.get("status") == "blocked" for s in trace.get("steps", [])):
|
|
250
|
+
trace["status"] = "blocked"
|
|
251
|
+
else:
|
|
252
|
+
trace["status"] = "success"
|
|
253
|
+
|
|
254
|
+
# If this was the active session execution, clear it
|
|
255
|
+
if _ACTIVE_EXECUTION_ID == target_id:
|
|
256
|
+
_ACTIVE_EXECUTION_ID = None
|
|
257
|
+
|
|
258
|
+
copy_trace = _clean_trace_dict(trace)
|
|
259
|
+
|
|
260
|
+
_persist_trace_event("end", copy_trace)
|
|
261
|
+
return copy_trace
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def record_step(
|
|
265
|
+
tool: str,
|
|
266
|
+
action: str,
|
|
267
|
+
params: Optional[Dict[str, Any]],
|
|
268
|
+
raw_result: Any,
|
|
269
|
+
duration_ms: int,
|
|
270
|
+
*,
|
|
271
|
+
execution_id: Optional[str] = None,
|
|
272
|
+
status: Optional[str] = None,
|
|
273
|
+
verification: Optional[Dict[str, Any]] = None,
|
|
274
|
+
changes: Optional[Dict[str, Any]] = None,
|
|
275
|
+
warnings: Optional[List[str]] = None,
|
|
276
|
+
) -> Dict[str, Any]:
|
|
277
|
+
"""Record a single tool call step into an execution trace.
|
|
278
|
+
|
|
279
|
+
If execution_id is provided or an active execution exists, appends to it.
|
|
280
|
+
Otherwise, creates a self-contained single-step execution trace.
|
|
281
|
+
"""
|
|
282
|
+
op_name = f"{tool}.{action}"
|
|
283
|
+
step_status = status or ("failed" if isinstance(raw_result, dict) and raw_result.get("error") else "success")
|
|
284
|
+
now = _now_iso()
|
|
285
|
+
|
|
286
|
+
step_record: Dict[str, Any] = {
|
|
287
|
+
"seq": 1,
|
|
288
|
+
"tool": tool,
|
|
289
|
+
"action": action,
|
|
290
|
+
"operation": op_name,
|
|
291
|
+
"duration_ms": max(0, int(duration_ms)),
|
|
292
|
+
"status": step_status,
|
|
293
|
+
"timestamp": now,
|
|
294
|
+
}
|
|
295
|
+
if verification:
|
|
296
|
+
step_record["verification"] = verification
|
|
297
|
+
if changes:
|
|
298
|
+
step_record["changes"] = changes
|
|
299
|
+
if warnings:
|
|
300
|
+
step_record["warnings"] = warnings
|
|
301
|
+
|
|
302
|
+
with _LOCK:
|
|
303
|
+
target_id = execution_id or _ACTIVE_EXECUTION_ID
|
|
304
|
+
is_single_step = False
|
|
305
|
+
|
|
306
|
+
if not target_id:
|
|
307
|
+
target_id = new_execution_id()
|
|
308
|
+
is_single_step = True
|
|
309
|
+
request_text = None
|
|
310
|
+
if isinstance(params, dict):
|
|
311
|
+
request_text = params.get("request") or params.get("prompt") or params.get("reason")
|
|
312
|
+
trace: Dict[str, Any] = {
|
|
313
|
+
"execution_id": target_id,
|
|
314
|
+
"request": str(request_text).strip() if request_text else None,
|
|
315
|
+
"status": step_status,
|
|
316
|
+
"started_at": now,
|
|
317
|
+
"ended_at": now,
|
|
318
|
+
"duration_ms": 0,
|
|
319
|
+
"tools": [],
|
|
320
|
+
"steps": [],
|
|
321
|
+
"changes": None,
|
|
322
|
+
"verification": {
|
|
323
|
+
"status": "unverified",
|
|
324
|
+
"passed": None, # see above: unknown, not passed
|
|
325
|
+
"contradiction": False,
|
|
326
|
+
"checks": [],
|
|
327
|
+
},
|
|
328
|
+
"warnings": [],
|
|
329
|
+
"initiator": "tool_call",
|
|
330
|
+
"is_active": False,
|
|
331
|
+
}
|
|
332
|
+
_EXECUTIONS_BY_ID[target_id] = trace
|
|
333
|
+
_RECENT_ORDER.appendleft(target_id)
|
|
334
|
+
else:
|
|
335
|
+
trace = _EXECUTIONS_BY_ID.get(target_id)
|
|
336
|
+
if not trace:
|
|
337
|
+
trace = {
|
|
338
|
+
"execution_id": target_id,
|
|
339
|
+
"request": None,
|
|
340
|
+
"status": "running",
|
|
341
|
+
"started_at": now,
|
|
342
|
+
"ended_at": None,
|
|
343
|
+
"duration_ms": 0,
|
|
344
|
+
"tools": [],
|
|
345
|
+
"steps": [],
|
|
346
|
+
"changes": None,
|
|
347
|
+
"verification": {
|
|
348
|
+
"status": "unverified",
|
|
349
|
+
"passed": None, # see above: unknown, not passed
|
|
350
|
+
"contradiction": False,
|
|
351
|
+
"checks": [],
|
|
352
|
+
},
|
|
353
|
+
"warnings": [],
|
|
354
|
+
"initiator": "agent",
|
|
355
|
+
"is_active": True,
|
|
356
|
+
}
|
|
357
|
+
_EXECUTIONS_BY_ID[target_id] = trace
|
|
358
|
+
_RECENT_ORDER.appendleft(target_id)
|
|
359
|
+
|
|
360
|
+
# Update trace request if provided in params and currently empty
|
|
361
|
+
if not trace.get("request") and isinstance(params, dict):
|
|
362
|
+
req = params.get("request") or params.get("prompt") or params.get("reason")
|
|
363
|
+
if req:
|
|
364
|
+
trace["request"] = str(req).strip()
|
|
365
|
+
|
|
366
|
+
step_record["seq"] = len(trace["steps"]) + 1
|
|
367
|
+
trace["steps"].append(step_record)
|
|
368
|
+
trace["duration_ms"] = int(trace.get("duration_ms", 0)) + max(0, int(duration_ms))
|
|
369
|
+
|
|
370
|
+
# Update aggregated tools entry (matching user format)
|
|
371
|
+
found_tool = None
|
|
372
|
+
for t in trace["tools"]:
|
|
373
|
+
if t.get("tool") == op_name or t.get("tool") == action:
|
|
374
|
+
found_tool = t
|
|
375
|
+
break
|
|
376
|
+
|
|
377
|
+
if found_tool:
|
|
378
|
+
found_tool["count"] = int(found_tool.get("count", 1)) + 1
|
|
379
|
+
found_tool["duration_ms"] = int(found_tool.get("duration_ms", 0)) + max(0, int(duration_ms))
|
|
380
|
+
else:
|
|
381
|
+
trace["tools"].append({
|
|
382
|
+
"tool": op_name,
|
|
383
|
+
"count": 1,
|
|
384
|
+
"duration_ms": max(0, int(duration_ms)),
|
|
385
|
+
})
|
|
386
|
+
|
|
387
|
+
# Aggregate changes
|
|
388
|
+
if changes:
|
|
389
|
+
trace["changes"] = _aggregate_changes(trace.get("changes"), changes)
|
|
390
|
+
|
|
391
|
+
# Merge verification
|
|
392
|
+
if verification:
|
|
393
|
+
trace["verification"] = _merge_verification(trace["verification"], verification)
|
|
394
|
+
|
|
395
|
+
# Merge warnings
|
|
396
|
+
if warnings:
|
|
397
|
+
existing_warnings = set(trace.get("warnings", []))
|
|
398
|
+
for w in warnings:
|
|
399
|
+
w_str = str(w).strip()
|
|
400
|
+
if w_str and w_str not in existing_warnings:
|
|
401
|
+
trace["warnings"].append(w_str)
|
|
402
|
+
existing_warnings.add(w_str)
|
|
403
|
+
|
|
404
|
+
if is_single_step:
|
|
405
|
+
trace["status"] = step_status
|
|
406
|
+
trace["ended_at"] = now
|
|
407
|
+
|
|
408
|
+
result_copy = _clean_trace_dict(trace)
|
|
409
|
+
|
|
410
|
+
_persist_trace_event("step", {"execution_id": target_id, "step": step_record})
|
|
411
|
+
return result_copy
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def get_execution_trace(execution_id: Optional[str] = None) -> Optional[Dict[str, Any]]:
|
|
415
|
+
"""Look up an execution trace by ID.
|
|
416
|
+
|
|
417
|
+
If execution_id is omitted, returns the most recent execution trace.
|
|
418
|
+
"""
|
|
419
|
+
with _LOCK:
|
|
420
|
+
if not execution_id:
|
|
421
|
+
if not _RECENT_ORDER:
|
|
422
|
+
return None
|
|
423
|
+
target_id = _RECENT_ORDER[0]
|
|
424
|
+
else:
|
|
425
|
+
target_id = execution_id
|
|
426
|
+
|
|
427
|
+
trace = _EXECUTIONS_BY_ID.get(target_id)
|
|
428
|
+
if not trace:
|
|
429
|
+
return None
|
|
430
|
+
return _clean_trace_dict(trace)
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
def get_execution(execution_id: str) -> Optional[Dict[str, Any]]:
|
|
434
|
+
"""Alias for get_execution_trace(execution_id)."""
|
|
435
|
+
return get_execution_trace(execution_id)
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def list_recent_executions(limit: int = 20) -> List[Dict[str, Any]]:
|
|
439
|
+
"""List recent executions (newest first) with summary information."""
|
|
440
|
+
max_count = max(1, min(int(limit), MAX_RECENT_EXECUTIONS))
|
|
441
|
+
out: List[Dict[str, Any]] = []
|
|
442
|
+
|
|
443
|
+
with _LOCK:
|
|
444
|
+
for exec_id in list(_RECENT_ORDER)[:max_count]:
|
|
445
|
+
trace = _EXECUTIONS_BY_ID.get(exec_id)
|
|
446
|
+
if not trace:
|
|
447
|
+
continue
|
|
448
|
+
summary = {
|
|
449
|
+
"execution_id": trace["execution_id"],
|
|
450
|
+
"request": trace.get("request"),
|
|
451
|
+
"status": trace.get("status"),
|
|
452
|
+
"started_at": trace.get("started_at"),
|
|
453
|
+
"ended_at": trace.get("ended_at"),
|
|
454
|
+
"duration_ms": trace.get("duration_ms", 0),
|
|
455
|
+
"tool_count": len(trace.get("tools", [])),
|
|
456
|
+
"step_count": len(trace.get("steps", [])),
|
|
457
|
+
"tools": trace.get("tools", []),
|
|
458
|
+
"verification": trace.get("verification", {}),
|
|
459
|
+
"changes": trace.get("changes"),
|
|
460
|
+
}
|
|
461
|
+
out.append(summary)
|
|
462
|
+
|
|
463
|
+
return out
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def clear_executions() -> Dict[str, Any]:
|
|
467
|
+
"""Clear in-memory execution traces and active execution ID."""
|
|
468
|
+
global _ACTIVE_EXECUTION_ID
|
|
469
|
+
with _LOCK:
|
|
470
|
+
count = len(_EXECUTIONS_BY_ID)
|
|
471
|
+
_EXECUTIONS_BY_ID.clear()
|
|
472
|
+
_RECENT_ORDER.clear()
|
|
473
|
+
_ACTIVE_EXECUTION_ID = None
|
|
474
|
+
return {"success": True, "cleared": count}
|
|
475
|
+
|
|
476
|
+
|
|
477
|
+
# ── Audit Reports ───────────────────────────────────────────────────────────
|
|
478
|
+
|
|
479
|
+
def _report_format(report_format: str) -> str:
|
|
480
|
+
fmt = str(report_format or "markdown").strip().lower()
|
|
481
|
+
if fmt in {"md", "markdown"}:
|
|
482
|
+
return "markdown"
|
|
483
|
+
if fmt == "json":
|
|
484
|
+
return "json"
|
|
485
|
+
raise ValueError("format must be markdown or json")
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def _report_extension(report_format: str) -> str:
|
|
489
|
+
return ".md" if _report_format(report_format) == "markdown" else ".json"
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
def _report_filename(execution_id: str, report_format: str) -> str:
|
|
493
|
+
safe_id = re.sub(r"[^A-Za-z0-9_.-]+", "_", str(execution_id or "execution")).strip("._")
|
|
494
|
+
if not safe_id:
|
|
495
|
+
safe_id = "execution"
|
|
496
|
+
return f"{safe_id}{_report_extension(report_format)}"
|
|
497
|
+
|
|
498
|
+
|
|
499
|
+
def execution_report_dir() -> str:
|
|
500
|
+
"""Directory used for generated execution audit reports."""
|
|
501
|
+
override = os.environ.get("RESOLVE_MCP_TRACE_REPORT_DIR")
|
|
502
|
+
if override:
|
|
503
|
+
return os.path.realpath(os.path.abspath(os.path.expanduser(override)))
|
|
504
|
+
return str(_REPO_ROOT / "logs" / "execution-reports")
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def _default_report_path(trace: Dict[str, Any], report_format: str) -> str:
|
|
508
|
+
return str(Path(execution_report_dir()) / _report_filename(trace["execution_id"], report_format))
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
def _format_ms(value: Any) -> str:
|
|
512
|
+
try:
|
|
513
|
+
return f"{max(0, int(value))} ms"
|
|
514
|
+
except (TypeError, ValueError):
|
|
515
|
+
return "0 ms"
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def _scalar(value: Any) -> str:
|
|
519
|
+
if value is None:
|
|
520
|
+
return ""
|
|
521
|
+
if isinstance(value, bool):
|
|
522
|
+
return "yes" if value else "no"
|
|
523
|
+
if isinstance(value, (int, float)):
|
|
524
|
+
return str(value)
|
|
525
|
+
if isinstance(value, (dict, list)):
|
|
526
|
+
return json.dumps(value, sort_keys=True, ensure_ascii=False)
|
|
527
|
+
return str(value)
|
|
528
|
+
|
|
529
|
+
|
|
530
|
+
def _markdown_row(*cells: Any) -> str:
|
|
531
|
+
escaped = []
|
|
532
|
+
for cell in cells:
|
|
533
|
+
text = _scalar(cell).replace("\n", " ").replace("|", "\\|")
|
|
534
|
+
escaped.append(text)
|
|
535
|
+
return "| " + " | ".join(escaped) + " |"
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
def _execution_report_json(trace: Dict[str, Any], *, include_steps: bool = True) -> Dict[str, Any]:
|
|
539
|
+
"""Return a stable, compact report object derived from safe trace summaries."""
|
|
540
|
+
out = {
|
|
541
|
+
"execution_id": trace.get("execution_id"),
|
|
542
|
+
"request": trace.get("request"),
|
|
543
|
+
"status": trace.get("status"),
|
|
544
|
+
"started_at": trace.get("started_at"),
|
|
545
|
+
"ended_at": trace.get("ended_at"),
|
|
546
|
+
"duration_ms": trace.get("duration_ms", 0),
|
|
547
|
+
"initiator": trace.get("initiator"),
|
|
548
|
+
"is_active": trace.get("is_active", False),
|
|
549
|
+
"tools": trace.get("tools", []),
|
|
550
|
+
"changes": trace.get("changes"),
|
|
551
|
+
"verification": trace.get("verification", {}),
|
|
552
|
+
"warnings": trace.get("warnings", []),
|
|
553
|
+
}
|
|
554
|
+
if trace.get("notes"):
|
|
555
|
+
out["notes"] = trace.get("notes")
|
|
556
|
+
if include_steps:
|
|
557
|
+
out["steps"] = trace.get("steps", [])
|
|
558
|
+
return out
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def _execution_report_markdown(trace: Dict[str, Any], *, include_steps: bool = True) -> str:
|
|
562
|
+
report = _execution_report_json(trace, include_steps=include_steps)
|
|
563
|
+
lines = [
|
|
564
|
+
"# Execution Audit Report",
|
|
565
|
+
"",
|
|
566
|
+
_markdown_row("Field", "Value"),
|
|
567
|
+
_markdown_row("---", "---"),
|
|
568
|
+
_markdown_row("Execution ID", report.get("execution_id")),
|
|
569
|
+
_markdown_row("Request", report.get("request")),
|
|
570
|
+
_markdown_row("Status", report.get("status")),
|
|
571
|
+
_markdown_row("Started", report.get("started_at")),
|
|
572
|
+
_markdown_row("Ended", report.get("ended_at")),
|
|
573
|
+
_markdown_row("Duration", _format_ms(report.get("duration_ms"))),
|
|
574
|
+
_markdown_row("Initiator", report.get("initiator")),
|
|
575
|
+
"",
|
|
576
|
+
"## Tool Summary",
|
|
577
|
+
"",
|
|
578
|
+
]
|
|
579
|
+
|
|
580
|
+
tools = report.get("tools") or []
|
|
581
|
+
if tools:
|
|
582
|
+
lines.extend([
|
|
583
|
+
_markdown_row("Tool", "Calls", "Duration"),
|
|
584
|
+
_markdown_row("---", "---:", "---:"),
|
|
585
|
+
])
|
|
586
|
+
for tool in tools:
|
|
587
|
+
lines.append(_markdown_row(tool.get("tool"), tool.get("count", 0), _format_ms(tool.get("duration_ms"))))
|
|
588
|
+
else:
|
|
589
|
+
lines.append("No tool calls recorded.")
|
|
590
|
+
|
|
591
|
+
lines.extend(["", "## Changes", ""])
|
|
592
|
+
changes = report.get("changes")
|
|
593
|
+
if isinstance(changes, dict) and changes:
|
|
594
|
+
lines.extend([
|
|
595
|
+
_markdown_row("Change", "Value"),
|
|
596
|
+
_markdown_row("---", "---"),
|
|
597
|
+
])
|
|
598
|
+
for key in sorted(changes):
|
|
599
|
+
lines.append(_markdown_row(key, changes[key]))
|
|
600
|
+
else:
|
|
601
|
+
lines.append("No semantic changes recorded.")
|
|
602
|
+
|
|
603
|
+
verification = report.get("verification") or {}
|
|
604
|
+
lines.extend([
|
|
605
|
+
"",
|
|
606
|
+
"## Verification",
|
|
607
|
+
"",
|
|
608
|
+
_markdown_row("Field", "Value"),
|
|
609
|
+
_markdown_row("---", "---"),
|
|
610
|
+
_markdown_row("Status", verification.get("status")),
|
|
611
|
+
_markdown_row("Passed", _verification_passed_label(verification)),
|
|
612
|
+
_markdown_row("Contradiction", verification.get("contradiction")),
|
|
613
|
+
_markdown_row("Checks", len(verification.get("checks") or [])),
|
|
614
|
+
])
|
|
615
|
+
|
|
616
|
+
warnings = report.get("warnings") or []
|
|
617
|
+
lines.extend(["", "## Warnings", ""])
|
|
618
|
+
if warnings:
|
|
619
|
+
lines.extend(f"- {_scalar(w)}" for w in warnings)
|
|
620
|
+
else:
|
|
621
|
+
lines.append("No warnings recorded.")
|
|
622
|
+
|
|
623
|
+
if report.get("notes"):
|
|
624
|
+
lines.extend(["", "## Notes", "", _scalar(report["notes"])])
|
|
625
|
+
|
|
626
|
+
if include_steps:
|
|
627
|
+
lines.extend(["", "## Steps", ""])
|
|
628
|
+
steps = report.get("steps") or []
|
|
629
|
+
if steps:
|
|
630
|
+
lines.extend([
|
|
631
|
+
_markdown_row("#", "Timestamp", "Operation", "Status", "Duration"),
|
|
632
|
+
_markdown_row("---:", "---", "---", "---", "---:"),
|
|
633
|
+
])
|
|
634
|
+
for step in steps:
|
|
635
|
+
lines.append(_markdown_row(
|
|
636
|
+
step.get("seq"),
|
|
637
|
+
step.get("timestamp"),
|
|
638
|
+
step.get("operation"),
|
|
639
|
+
step.get("status"),
|
|
640
|
+
_format_ms(step.get("duration_ms")),
|
|
641
|
+
))
|
|
642
|
+
else:
|
|
643
|
+
lines.append("No steps recorded.")
|
|
644
|
+
|
|
645
|
+
lines.append("")
|
|
646
|
+
return "\n".join(lines)
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
def _verification_passed_label(verification: Dict[str, Any]) -> str:
|
|
650
|
+
"""How the Passed row reads when nothing was actually verified.
|
|
651
|
+
|
|
652
|
+
A report that prints `Status: unverified` beside `Passed: yes` is read by a
|
|
653
|
+
human as "it passed" — the two lines are inches apart and only one of them
|
|
654
|
+
is scanned. Unverified means no evidence was reported, which is a question
|
|
655
|
+
still open, not a clean bill of health; an audit document is the last place
|
|
656
|
+
that distinction should be left to the reader.
|
|
657
|
+
"""
|
|
658
|
+
passed = verification.get("passed")
|
|
659
|
+
if passed is None:
|
|
660
|
+
return "not established — no checks recorded"
|
|
661
|
+
return "yes" if passed else "no"
|
|
662
|
+
|
|
663
|
+
|
|
664
|
+
def render_execution_report(
|
|
665
|
+
trace: Dict[str, Any],
|
|
666
|
+
*,
|
|
667
|
+
report_format: str = "markdown",
|
|
668
|
+
include_steps: bool = True,
|
|
669
|
+
) -> Union[str, Dict[str, Any]]:
|
|
670
|
+
"""Render an execution trace as a Markdown or JSON audit report."""
|
|
671
|
+
fmt = _report_format(report_format)
|
|
672
|
+
if fmt == "json":
|
|
673
|
+
return _execution_report_json(trace, include_steps=include_steps)
|
|
674
|
+
return _execution_report_markdown(trace, include_steps=include_steps)
|
|
675
|
+
|
|
676
|
+
|
|
677
|
+
def export_execution_report(
|
|
678
|
+
execution_id: Optional[str] = None,
|
|
679
|
+
*,
|
|
680
|
+
report_format: str = "markdown",
|
|
681
|
+
output_path: Optional[str] = None,
|
|
682
|
+
overwrite: bool = False,
|
|
683
|
+
include_steps: bool = True,
|
|
684
|
+
) -> Optional[Dict[str, Any]]:
|
|
685
|
+
"""Write an execution audit report to disk.
|
|
686
|
+
|
|
687
|
+
The export is built from trace summaries, not raw tool arguments/results.
|
|
688
|
+
"""
|
|
689
|
+
trace = get_execution_trace(execution_id)
|
|
690
|
+
if not trace:
|
|
691
|
+
return None
|
|
692
|
+
|
|
693
|
+
fmt = _report_format(report_format)
|
|
694
|
+
path = output_path or _default_report_path(trace, fmt)
|
|
695
|
+
real_path = os.path.realpath(os.path.abspath(os.path.expanduser(path)))
|
|
696
|
+
if os.path.exists(real_path) and not overwrite:
|
|
697
|
+
raise FileExistsError(f"Report already exists: {real_path}")
|
|
698
|
+
|
|
699
|
+
rendered = render_execution_report(trace, report_format=fmt, include_steps=include_steps)
|
|
700
|
+
os.makedirs(os.path.dirname(real_path), exist_ok=True)
|
|
701
|
+
if fmt == "json":
|
|
702
|
+
payload = json.dumps(rendered, indent=2, sort_keys=True, ensure_ascii=False) + "\n"
|
|
703
|
+
else:
|
|
704
|
+
payload = str(rendered)
|
|
705
|
+
with open(real_path, "w", encoding="utf-8") as fh:
|
|
706
|
+
fh.write(payload)
|
|
707
|
+
|
|
708
|
+
return {
|
|
709
|
+
"success": True,
|
|
710
|
+
"execution_id": trace["execution_id"],
|
|
711
|
+
"format": fmt,
|
|
712
|
+
"path": real_path,
|
|
713
|
+
"bytes": len(payload.encode("utf-8")),
|
|
714
|
+
"included_steps": bool(include_steps),
|
|
715
|
+
}
|
|
716
|
+
|
|
717
|
+
|
|
718
|
+
# ── Persistence (Best-Effort) ────────────────────────────────────────────────
|
|
719
|
+
|
|
720
|
+
def trace_log_path() -> str:
|
|
721
|
+
"""The file traces are appended to. Always a path, never None.
|
|
722
|
+
|
|
723
|
+
Returning None when a directory did not happen to exist made persistence
|
|
724
|
+
an invisible coin flip; the directory is created on first write instead.
|
|
725
|
+
"""
|
|
726
|
+
override = os.environ.get("RESOLVE_MCP_TRACE_FILE")
|
|
727
|
+
if override:
|
|
728
|
+
return os.path.realpath(os.path.abspath(os.path.expanduser(override)))
|
|
729
|
+
return str(_REPO_ROOT / "logs" / "execution-traces.jsonl")
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
def persistence_status() -> Dict[str, Any]:
|
|
733
|
+
"""Whether traces are reaching disk, and where.
|
|
734
|
+
|
|
735
|
+
Reported alongside every query so "no traces in the file" is answerable
|
|
736
|
+
without reading this module: an unwritable path says so here rather than
|
|
737
|
+
being swallowed by the best-effort append.
|
|
738
|
+
"""
|
|
739
|
+
path = trace_log_path()
|
|
740
|
+
status: Dict[str, Any] = {"path": path, "writable": False, "exists": False,
|
|
741
|
+
"reason": None}
|
|
742
|
+
try:
|
|
743
|
+
status["exists"] = os.path.isfile(path)
|
|
744
|
+
directory = os.path.dirname(path)
|
|
745
|
+
os.makedirs(directory, exist_ok=True)
|
|
746
|
+
status["writable"] = os.access(directory, os.W_OK)
|
|
747
|
+
if not status["writable"]:
|
|
748
|
+
status["reason"] = f"{directory} is not writable"
|
|
749
|
+
except OSError as exc:
|
|
750
|
+
status["reason"] = str(exc)
|
|
751
|
+
return status
|
|
752
|
+
|
|
753
|
+
|
|
754
|
+
# Kept for callers that used the private name; the public one is the path itself.
|
|
755
|
+
_trace_log_path = trace_log_path
|
|
756
|
+
|
|
757
|
+
|
|
758
|
+
def _rotate_if_oversized(path: str) -> None:
|
|
759
|
+
"""Roll the trace log over once it passes the size cap. Never raises."""
|
|
760
|
+
try:
|
|
761
|
+
if os.path.getsize(path) < MAX_TRACE_LOG_BYTES:
|
|
762
|
+
return
|
|
763
|
+
except OSError:
|
|
764
|
+
return
|
|
765
|
+
try:
|
|
766
|
+
previous = f"{path}.{TRACE_LOG_GENERATIONS}"
|
|
767
|
+
if os.path.exists(previous):
|
|
768
|
+
os.remove(previous)
|
|
769
|
+
os.replace(path, previous)
|
|
770
|
+
except OSError as exc:
|
|
771
|
+
logger.debug("Could not rotate the execution trace log: %s", exc)
|
|
772
|
+
|
|
773
|
+
|
|
774
|
+
def _persist_trace_event(event_type: str, data: Any) -> None:
|
|
775
|
+
"""Best-effort append to the trace log. Never raises.
|
|
776
|
+
|
|
777
|
+
Synchronous, on the calling thread — not "non-blocking", as this was
|
|
778
|
+
originally described. It is a buffered append of a few hundred bytes and
|
|
779
|
+
measures at ~0.07ms per tool call on local disk, which is immaterial next
|
|
780
|
+
to any Resolve round-trip, but the accurate word for it is *cheap*. It runs
|
|
781
|
+
outside `_LOCK` so a slow filesystem cannot serialize concurrent tool calls.
|
|
782
|
+
"""
|
|
783
|
+
try:
|
|
784
|
+
path = trace_log_path()
|
|
785
|
+
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
786
|
+
_rotate_if_oversized(path)
|
|
787
|
+
line = json.dumps({
|
|
788
|
+
"event": event_type,
|
|
789
|
+
"timestamp": _now_iso(),
|
|
790
|
+
"data": data,
|
|
791
|
+
}) + "\n"
|
|
792
|
+
with open(path, "a", encoding="utf-8") as fh:
|
|
793
|
+
fh.write(line)
|
|
794
|
+
except Exception as exc: # pragma: no cover
|
|
795
|
+
logger.debug("Failed to persist execution trace event: %s", exc)
|