loopview 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- loopview/__init__.py +3 -0
- loopview/app.py +252 -0
- loopview/cli.py +128 -0
- loopview/cost/__init__.py +1 -0
- loopview/cost/pricing.json +62 -0
- loopview/cost/pricing.py +74 -0
- loopview/cost/split.py +228 -0
- loopview/demo_data/flagship.otlp.jsonl +13 -0
- loopview/devtools/__init__.py +1 -0
- loopview/devtools/dump_normalized.py +211 -0
- loopview/ingest/__init__.py +1 -0
- loopview/ingest/otlp.py +305 -0
- loopview/ingest/raw.py +47 -0
- loopview/live.py +154 -0
- loopview/normalize/__init__.py +1 -0
- loopview/normalize/adapters/__init__.py +14 -0
- loopview/normalize/adapters/base.py +87 -0
- loopview/normalize/adapters/gen_ai.py +252 -0
- loopview/normalize/adapters/generic.py +22 -0
- loopview/normalize/adapters/openinference.py +367 -0
- loopview/normalize/derived_tools.py +90 -0
- loopview/normalize/loop_nodes.py +193 -0
- loopview/normalize/normalizer.py +326 -0
- loopview/normalize/schema.py +161 -0
- loopview/normalize/transitions.py +106 -0
- loopview/py.typed +0 -0
- loopview/static/assets/elk-worker.min-DfmSo98M.js +22 -0
- loopview/static/assets/index-CEGppBgk.css +1 -0
- loopview/static/assets/index-DzC96LiQ.js +21 -0
- loopview/static/assets/inter-cyrillic-ext-wght-normal-BOeWTOD4.woff2 +0 -0
- loopview/static/assets/inter-cyrillic-wght-normal-DqGufNeO.woff2 +0 -0
- loopview/static/assets/inter-greek-ext-wght-normal-DlzME5K_.woff2 +0 -0
- loopview/static/assets/inter-greek-wght-normal-CkhJZR-_.woff2 +0 -0
- loopview/static/assets/inter-latin-ext-wght-normal-DO1Apj_S.woff2 +0 -0
- loopview/static/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
- loopview/static/assets/inter-vietnamese-wght-normal-CBcvBZtf.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-cyrillic-wght-normal-D73BlboJ.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-greek-wght-normal-Bw9x6K1M.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-latin-ext-wght-normal-DBQx-q_a.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-latin-wght-normal-B9CIFXIH.woff2 +0 -0
- loopview/static/assets/jetbrains-mono-vietnamese-wght-normal-Bt-aOZkq.woff2 +0 -0
- loopview/static/index.html +13 -0
- loopview/store/__init__.py +1 -0
- loopview/store/capture.py +112 -0
- loopview/store/memory.py +165 -0
- loopview/tools/__init__.py +1 -0
- loopview/tools/report.py +347 -0
- loopview-0.1.0.dist-info/METADATA +72 -0
- loopview-0.1.0.dist-info/RECORD +51 -0
- loopview-0.1.0.dist-info/WHEEL +4 -0
- loopview-0.1.0.dist-info/entry_points.txt +3 -0
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
<!doctype html>
|
|
2
|
+
<html lang="en">
|
|
3
|
+
<head>
|
|
4
|
+
<meta charset="UTF-8" />
|
|
5
|
+
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
|
6
|
+
<title>loopview</title>
|
|
7
|
+
<script type="module" crossorigin src="/assets/index-DzC96LiQ.js"></script>
|
|
8
|
+
<link rel="stylesheet" crossorigin href="/assets/index-CEGppBgk.css">
|
|
9
|
+
</head>
|
|
10
|
+
<body>
|
|
11
|
+
<div id="root"></div>
|
|
12
|
+
</body>
|
|
13
|
+
</html>
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Storage: the in-memory ring buffer of runs and JSONL capture files."""
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Capture files: OTLP export requests saved as JSON Lines.
|
|
2
|
+
|
|
3
|
+
One line per export request, exactly as the receiver got it (after gzip is
|
|
4
|
+
undone), plus the time it arrived:
|
|
5
|
+
|
|
6
|
+
{"v": 1, "received_at_ns": 17..., "content_type": "application/json", "body_json": {...}}
|
|
7
|
+
{"v": 1, "received_at_ns": 17..., "content_type": "application/x-protobuf", "body_base64": ".."}
|
|
8
|
+
|
|
9
|
+
JSON bodies are stored as JSON so the file stays readable; protobuf bodies as base64.
|
|
10
|
+
|
|
11
|
+
The same format is used for three things:
|
|
12
|
+
- opt-in persistence (`loopview --persist FILE`): appended to as requests arrive,
|
|
13
|
+
reloaded on the next start;
|
|
14
|
+
- test fixtures: real framework output, replayed byte for byte into the receiver;
|
|
15
|
+
- the flagship demo recording.
|
|
16
|
+
|
|
17
|
+
Storing raw requests rather than decoded spans means a capture can be re-decoded
|
|
18
|
+
by a newer version of loopview, and the arrival timing is kept for liveness tests.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
import base64
|
|
22
|
+
import json
|
|
23
|
+
import logging
|
|
24
|
+
from collections.abc import Iterator
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import IO
|
|
28
|
+
|
|
29
|
+
from loopview.ingest.otlp import JSON, OtlpDecodeError, decode_body
|
|
30
|
+
from loopview.store.memory import TraceStore
|
|
31
|
+
|
|
32
|
+
FORMAT_VERSION = 1
|
|
33
|
+
|
|
34
|
+
log = logging.getLogger(__name__)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass(frozen=True)
|
|
38
|
+
class CapturedRequest:
|
|
39
|
+
received_at_ns: int
|
|
40
|
+
content_type: str
|
|
41
|
+
body: bytes
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def to_line(request: CapturedRequest) -> str:
|
|
45
|
+
record: dict[str, object] = {
|
|
46
|
+
"v": FORMAT_VERSION,
|
|
47
|
+
"received_at_ns": request.received_at_ns,
|
|
48
|
+
"content_type": request.content_type,
|
|
49
|
+
}
|
|
50
|
+
if request.content_type.lower().startswith(JSON):
|
|
51
|
+
record["body_json"] = json.loads(request.body)
|
|
52
|
+
else:
|
|
53
|
+
record["body_base64"] = base64.b64encode(request.body).decode()
|
|
54
|
+
return json.dumps(record, separators=(",", ":"))
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def from_line(line: str) -> CapturedRequest:
|
|
58
|
+
record = json.loads(line)
|
|
59
|
+
if record.get("v") != FORMAT_VERSION:
|
|
60
|
+
raise ValueError(f"unsupported capture format version: {record.get('v')!r}")
|
|
61
|
+
if "body_json" in record:
|
|
62
|
+
body = json.dumps(record["body_json"]).encode()
|
|
63
|
+
else:
|
|
64
|
+
body = base64.b64decode(record["body_base64"])
|
|
65
|
+
return CapturedRequest(record["received_at_ns"], record["content_type"], body)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def read_capture(path: Path) -> Iterator[CapturedRequest]:
|
|
69
|
+
"""Yield every request in a capture file, skipping lines that don't parse.
|
|
70
|
+
|
|
71
|
+
Skipping (with a warning) matters for persistence: if loopview was killed
|
|
72
|
+
mid-write, the last line may be truncated and should not block the next start.
|
|
73
|
+
"""
|
|
74
|
+
with path.open(encoding="utf-8") as f:
|
|
75
|
+
for number, line in enumerate(f, start=1):
|
|
76
|
+
if not line.strip():
|
|
77
|
+
continue
|
|
78
|
+
try:
|
|
79
|
+
yield from_line(line)
|
|
80
|
+
except (ValueError, KeyError, TypeError) as exc:
|
|
81
|
+
log.warning("skipping line %d of %s: %s", number, path, exc)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def load_capture(store: TraceStore, path: Path) -> int:
|
|
85
|
+
"""Replay a capture file into the store, keeping the recorded arrival times.
|
|
86
|
+
Returns the number of requests loaded."""
|
|
87
|
+
loaded = 0
|
|
88
|
+
for request in read_capture(path):
|
|
89
|
+
try:
|
|
90
|
+
spans = decode_body(request.body, request.content_type)
|
|
91
|
+
except OtlpDecodeError as exc:
|
|
92
|
+
log.warning("skipping undecodable request in %s: %s", path, exc)
|
|
93
|
+
continue
|
|
94
|
+
store.add_spans(spans, received_at_ns=request.received_at_ns)
|
|
95
|
+
loaded += 1
|
|
96
|
+
return loaded
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class CaptureWriter:
|
|
100
|
+
"""Appends requests to a capture file, flushing after each one."""
|
|
101
|
+
|
|
102
|
+
def __init__(self, path: Path) -> None:
|
|
103
|
+
self.path = path
|
|
104
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
105
|
+
self._file: IO[str] = path.open("a", encoding="utf-8")
|
|
106
|
+
|
|
107
|
+
def write(self, request: CapturedRequest) -> None:
|
|
108
|
+
self._file.write(to_line(request) + "\n")
|
|
109
|
+
self._file.flush()
|
|
110
|
+
|
|
111
|
+
def close(self) -> None:
|
|
112
|
+
self._file.close()
|
loopview/store/memory.py
ADDED
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""In-memory store of recent traces: a bounded ring buffer of runs.
|
|
2
|
+
|
|
3
|
+
- A run is one trace: every span sharing a trace id.
|
|
4
|
+
- A session is several runs sharing a conversation or session id attribute.
|
|
5
|
+
- The store keeps at most `max_runs` runs. When it is full, the run that was
|
|
6
|
+
updated least recently is dropped, so a run that is still receiving spans is
|
|
7
|
+
never evicted in favour of an idle one.
|
|
8
|
+
|
|
9
|
+
No locking: the store is only touched from the asyncio event loop (the receiver
|
|
10
|
+
endpoint is `async def`), so calls never overlap.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import time
|
|
14
|
+
from collections import OrderedDict
|
|
15
|
+
from collections.abc import Callable, Iterable
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
|
|
18
|
+
from pydantic import BaseModel
|
|
19
|
+
|
|
20
|
+
from loopview.ingest.raw import RawSpan
|
|
21
|
+
|
|
22
|
+
# Attributes that tie runs into a session, in priority order:
|
|
23
|
+
# OTel GenAI (gen_ai.conversation.id) and OpenInference (session.id).
|
|
24
|
+
SESSION_ID_ATTRIBUTES = ("gen_ai.conversation.id", "session.id")
|
|
25
|
+
|
|
26
|
+
DEFAULT_MAX_RUNS = 200
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class Run:
|
|
31
|
+
trace_id: str
|
|
32
|
+
first_received_ns: int
|
|
33
|
+
last_received_ns: int
|
|
34
|
+
spans: dict[str, RawSpan] = field(default_factory=dict) # by span id
|
|
35
|
+
# When each span arrived at the server. Arrival order and timing are what the
|
|
36
|
+
# liveness inference (M4) uses as a heartbeat.
|
|
37
|
+
received_ns: dict[str, int] = field(default_factory=dict)
|
|
38
|
+
# Spans reported as started (by loopview-sdk) that have not ended yet. Only
|
|
39
|
+
# in-progress spans are kept here; the ended span replaces its start report.
|
|
40
|
+
started: dict[str, RawSpan] = field(default_factory=dict)
|
|
41
|
+
|
|
42
|
+
def add(self, span: RawSpan, received_at_ns: int) -> None:
|
|
43
|
+
# A re-sent span (exporter retry) replaces the old copy but keeps its first
|
|
44
|
+
# arrival time.
|
|
45
|
+
self.received_ns.setdefault(span.span_id, received_at_ns)
|
|
46
|
+
self.spans[span.span_id] = span
|
|
47
|
+
self.started.pop(span.span_id, None)
|
|
48
|
+
self.last_received_ns = max(self.last_received_ns, received_at_ns)
|
|
49
|
+
|
|
50
|
+
def add_started(self, span: RawSpan, received_at_ns: int) -> None:
|
|
51
|
+
if span.span_id not in self.spans: # the end can overtake the start report
|
|
52
|
+
self.started[span.span_id] = span
|
|
53
|
+
self.last_received_ns = max(self.last_received_ns, received_at_ns)
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def session_id(self) -> str | None:
|
|
57
|
+
"""The session id of the shallowest span that has one.
|
|
58
|
+
|
|
59
|
+
Shallowest, because nested agents can carry their own conversation ids and
|
|
60
|
+
the outermost one describes the run. Computed on demand, since spans arrive
|
|
61
|
+
in any order (children usually first)."""
|
|
62
|
+
best: tuple[int, str] | None = None
|
|
63
|
+
for span in self.spans.values():
|
|
64
|
+
session = _session_id(span)
|
|
65
|
+
if session is None:
|
|
66
|
+
continue
|
|
67
|
+
depth, parent = 0, span.parent_span_id
|
|
68
|
+
while parent in self.spans:
|
|
69
|
+
depth, parent = depth + 1, self.spans[parent].parent_span_id
|
|
70
|
+
if best is None or depth < best[0]:
|
|
71
|
+
best = (depth, session)
|
|
72
|
+
return best[1] if best else None
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def root(self) -> RawSpan | None:
|
|
76
|
+
"""The span with no parent, or None if it hasn't arrived yet.
|
|
77
|
+
|
|
78
|
+
Spans are exported when they end, and a root ends last, so a run that is
|
|
79
|
+
still going usually has no root yet.
|
|
80
|
+
"""
|
|
81
|
+
for span in self.spans.values():
|
|
82
|
+
if span.parent_span_id is None:
|
|
83
|
+
return span
|
|
84
|
+
return None
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class SessionSummary(BaseModel):
|
|
88
|
+
session_id: str
|
|
89
|
+
trace_ids: list[str] # oldest run first
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class TraceStore:
|
|
93
|
+
def __init__(
|
|
94
|
+
self, max_runs: int = DEFAULT_MAX_RUNS, clock: Callable[[], int] = time.time_ns
|
|
95
|
+
) -> None:
|
|
96
|
+
self.max_runs = max_runs
|
|
97
|
+
self._clock = clock
|
|
98
|
+
# Ordered from least to most recently updated, so eviction is popitem(last=False).
|
|
99
|
+
self._runs: OrderedDict[str, Run] = OrderedDict()
|
|
100
|
+
|
|
101
|
+
def add_spans(self, spans: Iterable[RawSpan], received_at_ns: int | None = None) -> list[str]:
|
|
102
|
+
"""Store spans from one export request. Returns the trace ids that changed."""
|
|
103
|
+
now = self._clock() if received_at_ns is None else received_at_ns
|
|
104
|
+
changed: list[str] = []
|
|
105
|
+
for span in spans:
|
|
106
|
+
run = self._runs.get(span.trace_id)
|
|
107
|
+
if run is None:
|
|
108
|
+
run = Run(span.trace_id, first_received_ns=now, last_received_ns=now)
|
|
109
|
+
self._runs[span.trace_id] = run
|
|
110
|
+
run.add(span, now)
|
|
111
|
+
self._runs.move_to_end(span.trace_id)
|
|
112
|
+
if span.trace_id not in changed:
|
|
113
|
+
changed.append(span.trace_id)
|
|
114
|
+
while len(self._runs) > self.max_runs:
|
|
115
|
+
self._runs.popitem(last=False)
|
|
116
|
+
return changed
|
|
117
|
+
|
|
118
|
+
def add_started_spans(
|
|
119
|
+
self, spans: Iterable[RawSpan], received_at_ns: int | None = None
|
|
120
|
+
) -> list[str]:
|
|
121
|
+
"""Store start reports from loopview-sdk. Returns the trace ids that changed."""
|
|
122
|
+
now = self._clock() if received_at_ns is None else received_at_ns
|
|
123
|
+
changed: list[str] = []
|
|
124
|
+
for span in spans:
|
|
125
|
+
run = self._runs.get(span.trace_id)
|
|
126
|
+
if run is None:
|
|
127
|
+
run = Run(span.trace_id, first_received_ns=now, last_received_ns=now)
|
|
128
|
+
self._runs[span.trace_id] = run
|
|
129
|
+
run.add_started(span, now)
|
|
130
|
+
self._runs.move_to_end(span.trace_id)
|
|
131
|
+
if span.trace_id not in changed:
|
|
132
|
+
changed.append(span.trace_id)
|
|
133
|
+
while len(self._runs) > self.max_runs:
|
|
134
|
+
self._runs.popitem(last=False)
|
|
135
|
+
return changed
|
|
136
|
+
|
|
137
|
+
def get_run(self, trace_id: str) -> Run | None:
|
|
138
|
+
return self._runs.get(trace_id)
|
|
139
|
+
|
|
140
|
+
def runs(self) -> list[Run]:
|
|
141
|
+
"""All runs, most recently started first."""
|
|
142
|
+
return sorted(self._runs.values(), key=lambda r: r.first_received_ns, reverse=True)
|
|
143
|
+
|
|
144
|
+
def sessions(self) -> list[SessionSummary]:
|
|
145
|
+
"""Sessions, derived on demand. With a few hundred runs this is cheap and
|
|
146
|
+
avoids keeping a second index in sync with eviction."""
|
|
147
|
+
by_session: dict[str, list[Run]] = {}
|
|
148
|
+
for run in self._runs.values():
|
|
149
|
+
if run.session_id is not None:
|
|
150
|
+
by_session.setdefault(run.session_id, []).append(run)
|
|
151
|
+
return [
|
|
152
|
+
SessionSummary(
|
|
153
|
+
session_id=session_id,
|
|
154
|
+
trace_ids=[r.trace_id for r in sorted(runs, key=lambda r: r.first_received_ns)],
|
|
155
|
+
)
|
|
156
|
+
for session_id, runs in by_session.items()
|
|
157
|
+
]
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _session_id(span: RawSpan) -> str | None:
|
|
161
|
+
for key in SESSION_ID_ATTRIBUTES:
|
|
162
|
+
value = span.attributes.get(key)
|
|
163
|
+
if value:
|
|
164
|
+
return str(value)
|
|
165
|
+
return None
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""The Tools tab: how tools behave across many runs (report.py)."""
|
loopview/tools/report.py
ADDED
|
@@ -0,0 +1,347 @@
|
|
|
1
|
+
"""Which tools an agent struggles with, across many runs.
|
|
2
|
+
|
|
3
|
+
Pure functions over normalized runs: no I/O, no store. The API picks the runs.
|
|
4
|
+
|
|
5
|
+
Definitions (also in DECISIONS.md D44):
|
|
6
|
+
|
|
7
|
+
- Tool call: a finished `tool_call` step. Running calls are left out.
|
|
8
|
+
- Error: the call is marked failed by the trace: span status ERROR (with or
|
|
9
|
+
without an exception event), or an MCP result with `isError: true`. Errors are
|
|
10
|
+
never guessed from the result text.
|
|
11
|
+
- Canonical arguments: the arguments as JSON with sorted keys, so identical calls
|
|
12
|
+
compare equal.
|
|
13
|
+
- Agent: the nearest agent above the call, by name, within its run.
|
|
14
|
+
- Next move after an error: what the agent does once the error is back in front
|
|
15
|
+
of the model. Models often request several tools in one turn, so the call that
|
|
16
|
+
happens to start next was usually requested before the model saw the error.
|
|
17
|
+
So each tool call is assigned to the model call that requested it (the
|
|
18
|
+
agent's latest model call that had ended when the tool started), and we look
|
|
19
|
+
at the turns whose model call started after the failed call ended:
|
|
20
|
+
blind_retry the same tool with identical canonical arguments
|
|
21
|
+
fixed the same tool with different arguments (and whether that worked)
|
|
22
|
+
switched a different tool, recorded as the pair (failed -> next)
|
|
23
|
+
gave_up no further tool call by that agent in that run
|
|
24
|
+
Calls to tools that also failed in the failed call's own turn are skipped:
|
|
25
|
+
they are those tools' own retries. When an agent has no recorded model calls,
|
|
26
|
+
the next move is simply its next tool call started after the error ended.
|
|
27
|
+
- Error group: the error message with UUIDs, hex ids, long quoted values and
|
|
28
|
+
numbers replaced by placeholders, trimmed to 200 characters.
|
|
29
|
+
- Confused pair: (A -> B) from `switched`, counted across runs; reported from two
|
|
30
|
+
occurrences.
|
|
31
|
+
- Offered tools: the tool definitions recorded on model calls. If no model call
|
|
32
|
+
in the input records any, the tool list is "not recorded" and nothing is
|
|
33
|
+
inferred about unused tools.
|
|
34
|
+
- Never called: offered at least once, called zero times in the input runs.
|
|
35
|
+
- Definition cost (estimate): a definition's JSON length / 4, times the number of
|
|
36
|
+
model calls that carried it.
|
|
37
|
+
- Result size (estimate): the average result length / 4 over successful calls
|
|
38
|
+
(Pydantic AI records its retry prompt as the result of a failed call).
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
import json
|
|
42
|
+
import math
|
|
43
|
+
import re
|
|
44
|
+
from collections import Counter
|
|
45
|
+
from dataclasses import dataclass, field
|
|
46
|
+
from typing import Any, Literal
|
|
47
|
+
|
|
48
|
+
from loopview.normalize.schema import NormalizedRun, Step
|
|
49
|
+
|
|
50
|
+
CHARS_PER_TOKEN = 4
|
|
51
|
+
ERROR_GROUP_MAX = 200
|
|
52
|
+
TOP_ERRORS = 3
|
|
53
|
+
MIN_CONFUSED = 2
|
|
54
|
+
|
|
55
|
+
Move = Literal["blind_retry", "fixed", "switched", "gave_up"]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
# --- small helpers -------------------------------------------------------------------
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def canonical_args(arguments: Any) -> str:
|
|
62
|
+
return json.dumps(arguments, sort_keys=True, default=str, ensure_ascii=False)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def estimate_tokens(value: Any) -> int:
|
|
66
|
+
text = value if isinstance(value, str) else canonical_args(value)
|
|
67
|
+
return math.ceil(len(text) / CHARS_PER_TOKEN) # anything non-empty is a token
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
_UUID = re.compile(
|
|
71
|
+
r"\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b"
|
|
72
|
+
)
|
|
73
|
+
# 0x-prefixed, or a run of 8+ hex characters that contains a digit (ids, hashes)
|
|
74
|
+
_HEX = re.compile(r"\b0x[0-9a-fA-F]+\b|\b(?=[0-9a-fA-F]*\d)[0-9a-fA-F]{8,}\b")
|
|
75
|
+
_LONG_QUOTED = re.compile(r"'[^']{13,}'|\"[^\"]{13,}\"")
|
|
76
|
+
_NUMBER = re.compile(r"\d+(?:\.\d+)?")
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def error_group(message: str) -> str:
|
|
80
|
+
text = _UUID.sub("<uuid>", message)
|
|
81
|
+
text = _HEX.sub("<id>", text)
|
|
82
|
+
text = _LONG_QUOTED.sub("<value>", text)
|
|
83
|
+
text = _NUMBER.sub("<n>", text)
|
|
84
|
+
return " ".join(text.split())[:ERROR_GROUP_MAX]
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _mcp_is_error(result: Any) -> bool:
|
|
88
|
+
return isinstance(result, dict) and result.get("isError") is True
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _mcp_error_text(result: dict[str, Any]) -> str:
|
|
92
|
+
texts = [
|
|
93
|
+
str(c.get("text"))
|
|
94
|
+
for c in result.get("content") or []
|
|
95
|
+
if isinstance(c, dict) and c.get("text")
|
|
96
|
+
]
|
|
97
|
+
return "\n".join(texts) or "isError"
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def definition_name(definition: Any) -> str | None:
|
|
101
|
+
"""Tool definitions come as {"name": ...} (GenAI, Anthropic, OpenInference) or
|
|
102
|
+
OpenAI's {"type": "function", "function": {"name": ...}}."""
|
|
103
|
+
if not isinstance(definition, dict):
|
|
104
|
+
return None
|
|
105
|
+
name = definition.get("name")
|
|
106
|
+
if name is None and isinstance(definition.get("function"), dict):
|
|
107
|
+
name = definition["function"].get("name")
|
|
108
|
+
return str(name) if name else None
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
# --- calls -----------------------------------------------------------------------------
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
@dataclass
|
|
115
|
+
class Call:
|
|
116
|
+
run_id: str
|
|
117
|
+
step_id: str
|
|
118
|
+
agent: str
|
|
119
|
+
name: str
|
|
120
|
+
start: int
|
|
121
|
+
end: int
|
|
122
|
+
args: Any
|
|
123
|
+
canonical: str
|
|
124
|
+
result: Any
|
|
125
|
+
error: str | None # None when the call succeeded
|
|
126
|
+
|
|
127
|
+
@property
|
|
128
|
+
def failed(self) -> bool:
|
|
129
|
+
return self.error is not None
|
|
130
|
+
|
|
131
|
+
def ref(self) -> dict[str, str]:
|
|
132
|
+
return {"run_id": self.run_id, "step_id": self.step_id}
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _agent_of(step: Step, by_id: dict[str, Step]) -> str:
|
|
136
|
+
parent = by_id.get(step.parent_id) if step.parent_id else None
|
|
137
|
+
while parent is not None:
|
|
138
|
+
# A synthetic `tools` step can be a group (it holds sub-agents); it isn't an agent.
|
|
139
|
+
if parent.kind == "agent" and not parent.synthetic:
|
|
140
|
+
return parent.name
|
|
141
|
+
parent = by_id.get(parent.parent_id) if parent.parent_id else None
|
|
142
|
+
return "" # a call outside any agent: the run itself is its agent
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _calls(run: NormalizedRun, by_id: dict[str, Step]) -> list[Call]:
|
|
146
|
+
calls = []
|
|
147
|
+
for s in run.steps:
|
|
148
|
+
if s.kind != "tool_call" or s.tool is None or s.end_ns is None or s.status == "running":
|
|
149
|
+
continue
|
|
150
|
+
error = None
|
|
151
|
+
if s.status == "error":
|
|
152
|
+
error = s.error or "error"
|
|
153
|
+
elif _mcp_is_error(s.tool.result):
|
|
154
|
+
error = _mcp_error_text(s.tool.result)
|
|
155
|
+
calls.append(
|
|
156
|
+
Call(
|
|
157
|
+
run_id=s.run_id,
|
|
158
|
+
step_id=s.id,
|
|
159
|
+
agent=_agent_of(s, by_id),
|
|
160
|
+
name=s.tool.name or s.name,
|
|
161
|
+
start=s.start_ns,
|
|
162
|
+
end=s.end_ns,
|
|
163
|
+
args=s.tool.arguments,
|
|
164
|
+
canonical=canonical_args(s.tool.arguments),
|
|
165
|
+
result=s.tool.result,
|
|
166
|
+
error=error,
|
|
167
|
+
)
|
|
168
|
+
)
|
|
169
|
+
return sorted(calls, key=lambda c: (c.start, c.step_id))
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
# --- next move after an error -----------------------------------------------------------
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
@dataclass
|
|
176
|
+
class NextMove:
|
|
177
|
+
move: Move
|
|
178
|
+
next_tool: str | None = None # for switched
|
|
179
|
+
succeeded: bool | None = None # for fixed
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _turn_of(call: Call, model_calls: list[tuple[int, int]]) -> int:
|
|
183
|
+
"""Index of the model call that requested `call`: the latest one that had
|
|
184
|
+
ended when the call started. Comparing with the end, not the start, keeps
|
|
185
|
+
this right when timestamps tie (a coarse clock, DECISIONS D17). -1: none."""
|
|
186
|
+
turn = -1
|
|
187
|
+
for i, (_, end) in enumerate(model_calls):
|
|
188
|
+
if end <= call.start:
|
|
189
|
+
turn = i
|
|
190
|
+
return turn
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def next_move(
|
|
194
|
+
failed: Call, agent_calls: list[Call], model_calls: list[tuple[int, int]]
|
|
195
|
+
) -> NextMove:
|
|
196
|
+
"""What the agent did after `failed` (see the module docstring).
|
|
197
|
+
`model_calls` are the agent's model calls as (start, end), sorted by start."""
|
|
198
|
+
others = [c for c in agent_calls if c is not failed]
|
|
199
|
+
if not model_calls:
|
|
200
|
+
later = [c for c in others if c.start >= failed.end]
|
|
201
|
+
turns = [later[:1]] if later else []
|
|
202
|
+
skip: set[str] = set()
|
|
203
|
+
else:
|
|
204
|
+
own = _turn_of(failed, model_calls)
|
|
205
|
+
by_turn: dict[int, list[Call]] = {}
|
|
206
|
+
for c in others:
|
|
207
|
+
by_turn.setdefault(_turn_of(c, model_calls), []).append(c)
|
|
208
|
+
# Turns whose model call started once the error was back: it saw the error.
|
|
209
|
+
turns = [by_turn[i] for i in sorted(by_turn) if i > own and model_calls[i][0] >= failed.end]
|
|
210
|
+
skip = {c.name for c in by_turn.get(own, []) if c.failed and c.name != failed.name}
|
|
211
|
+
for turn in turns:
|
|
212
|
+
same = [c for c in turn if c.name == failed.name]
|
|
213
|
+
if any(c.canonical == failed.canonical for c in same):
|
|
214
|
+
return NextMove("blind_retry")
|
|
215
|
+
if same:
|
|
216
|
+
return NextMove("fixed", succeeded=not same[0].failed)
|
|
217
|
+
switched = [c for c in turn if c.name not in skip]
|
|
218
|
+
if switched:
|
|
219
|
+
return NextMove("switched", next_tool=switched[0].name)
|
|
220
|
+
return NextMove("gave_up")
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
# --- the report ----------------------------------------------------------------------------
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
@dataclass
|
|
227
|
+
class _ToolStats:
|
|
228
|
+
calls: list[Call] = field(default_factory=list)
|
|
229
|
+
moves: Counter[str] = field(default_factory=Counter)
|
|
230
|
+
switched_to: Counter[str] = field(default_factory=Counter)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def tools_report(runs: list[NormalizedRun]) -> dict[str, Any]:
|
|
234
|
+
stats: dict[str, _ToolStats] = {}
|
|
235
|
+
pairs: Counter[tuple[str, str]] = Counter()
|
|
236
|
+
offered_size: dict[str, int] = {} # tool -> definition size in tokens (largest seen)
|
|
237
|
+
carried: Counter[str] = Counter() # tool -> model calls that carried its definition
|
|
238
|
+
tool_list_recorded = False
|
|
239
|
+
all_calls: list[Call] = []
|
|
240
|
+
|
|
241
|
+
for run in runs:
|
|
242
|
+
by_id = {s.id: s for s in run.steps}
|
|
243
|
+
calls = _calls(run, by_id)
|
|
244
|
+
all_calls += calls
|
|
245
|
+
model_calls: dict[str, list[tuple[int, int]]] = {}
|
|
246
|
+
for s in run.steps:
|
|
247
|
+
if s.kind != "model_call":
|
|
248
|
+
continue
|
|
249
|
+
# A running model call hasn't requested anything yet.
|
|
250
|
+
end = s.end_ns if s.end_ns is not None else 2**63
|
|
251
|
+
model_calls.setdefault(_agent_of(s, by_id), []).append((s.start_ns, end))
|
|
252
|
+
definitions = s.model.tool_definitions if s.model else []
|
|
253
|
+
if definitions:
|
|
254
|
+
tool_list_recorded = True
|
|
255
|
+
for d in definitions:
|
|
256
|
+
name = definition_name(d)
|
|
257
|
+
if name:
|
|
258
|
+
carried[name] += 1
|
|
259
|
+
offered_size[name] = max(offered_size.get(name, 0), estimate_tokens(d))
|
|
260
|
+
for spans in model_calls.values():
|
|
261
|
+
spans.sort()
|
|
262
|
+
|
|
263
|
+
by_agent: dict[str, list[Call]] = {}
|
|
264
|
+
for c in calls:
|
|
265
|
+
by_agent.setdefault(c.agent, []).append(c)
|
|
266
|
+
stats.setdefault(c.name, _ToolStats()).calls.append(c)
|
|
267
|
+
for c in calls:
|
|
268
|
+
if not c.failed:
|
|
269
|
+
continue
|
|
270
|
+
move = next_move(c, by_agent[c.agent], model_calls.get(c.agent, []))
|
|
271
|
+
tool = stats[c.name]
|
|
272
|
+
tool.moves[move.move] += 1
|
|
273
|
+
if move.move == "fixed" and move.succeeded:
|
|
274
|
+
tool.moves["fixed_succeeded"] += 1
|
|
275
|
+
if move.move == "switched" and move.next_tool:
|
|
276
|
+
tool.switched_to[move.next_tool] += 1
|
|
277
|
+
pairs[(c.name, move.next_tool)] += 1
|
|
278
|
+
|
|
279
|
+
tools = [_tool_entry(name, s, pairs) for name, s in stats.items()]
|
|
280
|
+
tools.sort(key=lambda t: (-t["errors"], -t["calls"], t["name"]))
|
|
281
|
+
|
|
282
|
+
called = set(stats)
|
|
283
|
+
never_called = []
|
|
284
|
+
if tool_list_recorded:
|
|
285
|
+
never_called = [
|
|
286
|
+
{
|
|
287
|
+
"name": name,
|
|
288
|
+
"definition_tokens_estimate": offered_size[name] * carried[name],
|
|
289
|
+
"carried_by_model_calls": carried[name],
|
|
290
|
+
}
|
|
291
|
+
for name in sorted(set(carried) - called)
|
|
292
|
+
]
|
|
293
|
+
never_called.sort(key=lambda t: (-t["definition_tokens_estimate"], t["name"]))
|
|
294
|
+
|
|
295
|
+
errors = sum(t["errors"] for t in tools)
|
|
296
|
+
top_two = sum(sorted((t["errors"] for t in tools), reverse=True)[:2])
|
|
297
|
+
never_tokens = sum(t["definition_tokens_estimate"] for t in never_called)
|
|
298
|
+
return {
|
|
299
|
+
"summary": {
|
|
300
|
+
"runs": len(runs),
|
|
301
|
+
"tool_calls": len(all_calls),
|
|
302
|
+
"errors": errors,
|
|
303
|
+
"share_of_errors_from_top_2_tools": round(top_two / errors, 3) if errors else None,
|
|
304
|
+
"never_called_count": len(never_called),
|
|
305
|
+
"never_called_tokens_estimate": never_tokens,
|
|
306
|
+
"never_called_tokens_per_run_estimate": round(never_tokens / len(runs)) if runs else 0,
|
|
307
|
+
},
|
|
308
|
+
"tools": tools,
|
|
309
|
+
"never_called": never_called,
|
|
310
|
+
"tool_list_recorded": tool_list_recorded,
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _tool_entry(name: str, s: _ToolStats, pairs: Counter[tuple[str, str]]) -> dict[str, Any]:
|
|
315
|
+
failures = [c for c in s.calls if c.failed]
|
|
316
|
+
groups: dict[str, list[Call]] = {}
|
|
317
|
+
for c in failures:
|
|
318
|
+
groups.setdefault(error_group(c.error or ""), []).append(c)
|
|
319
|
+
top = sorted(groups.items(), key=lambda g: (-len(g[1]), g[0]))[:TOP_ERRORS]
|
|
320
|
+
results = [estimate_tokens(c.result) for c in s.calls if not c.failed and c.result is not None]
|
|
321
|
+
return {
|
|
322
|
+
"name": name,
|
|
323
|
+
"calls": len(s.calls),
|
|
324
|
+
"errors": len(failures),
|
|
325
|
+
"error_rate": round(len(failures) / len(s.calls), 3),
|
|
326
|
+
"after_error": {
|
|
327
|
+
k: s.moves[k]
|
|
328
|
+
for k in ("blind_retry", "fixed", "fixed_succeeded", "switched", "gave_up")
|
|
329
|
+
},
|
|
330
|
+
"top_errors": [
|
|
331
|
+
{
|
|
332
|
+
"message_group": group,
|
|
333
|
+
"count": len(examples),
|
|
334
|
+
"example_args": examples[0].args,
|
|
335
|
+
"example_message": examples[0].error,
|
|
336
|
+
"step_ref": examples[0].ref(),
|
|
337
|
+
}
|
|
338
|
+
for group, examples in top
|
|
339
|
+
],
|
|
340
|
+
"confused_with": [
|
|
341
|
+
{"tool": other, "count": n}
|
|
342
|
+
for (a, other), n in pairs.most_common()
|
|
343
|
+
if a == name and n >= MIN_CONFUSED
|
|
344
|
+
],
|
|
345
|
+
"avg_result_tokens_estimate": round(sum(results) / len(results)) if results else None,
|
|
346
|
+
"step_refs": [c.ref() for c in s.calls],
|
|
347
|
+
}
|