loopview 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. loopview/__init__.py +3 -0
  2. loopview/app.py +252 -0
  3. loopview/cli.py +128 -0
  4. loopview/cost/__init__.py +1 -0
  5. loopview/cost/pricing.json +62 -0
  6. loopview/cost/pricing.py +74 -0
  7. loopview/cost/split.py +228 -0
  8. loopview/demo_data/flagship.otlp.jsonl +13 -0
  9. loopview/devtools/__init__.py +1 -0
  10. loopview/devtools/dump_normalized.py +211 -0
  11. loopview/ingest/__init__.py +1 -0
  12. loopview/ingest/otlp.py +305 -0
  13. loopview/ingest/raw.py +47 -0
  14. loopview/live.py +154 -0
  15. loopview/normalize/__init__.py +1 -0
  16. loopview/normalize/adapters/__init__.py +14 -0
  17. loopview/normalize/adapters/base.py +87 -0
  18. loopview/normalize/adapters/gen_ai.py +252 -0
  19. loopview/normalize/adapters/generic.py +22 -0
  20. loopview/normalize/adapters/openinference.py +367 -0
  21. loopview/normalize/derived_tools.py +90 -0
  22. loopview/normalize/loop_nodes.py +193 -0
  23. loopview/normalize/normalizer.py +326 -0
  24. loopview/normalize/schema.py +161 -0
  25. loopview/normalize/transitions.py +106 -0
  26. loopview/py.typed +0 -0
  27. loopview/static/assets/elk-worker.min-DfmSo98M.js +22 -0
  28. loopview/static/assets/index-CEGppBgk.css +1 -0
  29. loopview/static/assets/index-DzC96LiQ.js +21 -0
  30. loopview/static/assets/inter-cyrillic-ext-wght-normal-BOeWTOD4.woff2 +0 -0
  31. loopview/static/assets/inter-cyrillic-wght-normal-DqGufNeO.woff2 +0 -0
  32. loopview/static/assets/inter-greek-ext-wght-normal-DlzME5K_.woff2 +0 -0
  33. loopview/static/assets/inter-greek-wght-normal-CkhJZR-_.woff2 +0 -0
  34. loopview/static/assets/inter-latin-ext-wght-normal-DO1Apj_S.woff2 +0 -0
  35. loopview/static/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
  36. loopview/static/assets/inter-vietnamese-wght-normal-CBcvBZtf.woff2 +0 -0
  37. loopview/static/assets/jetbrains-mono-cyrillic-wght-normal-D73BlboJ.woff2 +0 -0
  38. loopview/static/assets/jetbrains-mono-greek-wght-normal-Bw9x6K1M.woff2 +0 -0
  39. loopview/static/assets/jetbrains-mono-latin-ext-wght-normal-DBQx-q_a.woff2 +0 -0
  40. loopview/static/assets/jetbrains-mono-latin-wght-normal-B9CIFXIH.woff2 +0 -0
  41. loopview/static/assets/jetbrains-mono-vietnamese-wght-normal-Bt-aOZkq.woff2 +0 -0
  42. loopview/static/index.html +13 -0
  43. loopview/store/__init__.py +1 -0
  44. loopview/store/capture.py +112 -0
  45. loopview/store/memory.py +165 -0
  46. loopview/tools/__init__.py +1 -0
  47. loopview/tools/report.py +347 -0
  48. loopview-0.1.0.dist-info/METADATA +72 -0
  49. loopview-0.1.0.dist-info/RECORD +51 -0
  50. loopview-0.1.0.dist-info/WHEEL +4 -0
  51. loopview-0.1.0.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,13 @@
1
+ <!doctype html>
2
+ <html lang="en">
3
+ <head>
4
+ <meta charset="UTF-8" />
5
+ <meta name="viewport" content="width=device-width, initial-scale=1.0" />
6
+ <title>loopview</title>
7
+ <script type="module" crossorigin src="/assets/index-DzC96LiQ.js"></script>
8
+ <link rel="stylesheet" crossorigin href="/assets/index-CEGppBgk.css">
9
+ </head>
10
+ <body>
11
+ <div id="root"></div>
12
+ </body>
13
+ </html>
@@ -0,0 +1 @@
1
+ """Storage: the in-memory ring buffer of runs and JSONL capture files."""
@@ -0,0 +1,112 @@
1
+ """Capture files: OTLP export requests saved as JSON Lines.
2
+
3
+ One line per export request, exactly as the receiver got it (after gzip is
4
+ undone), plus the time it arrived:
5
+
6
+ {"v": 1, "received_at_ns": 17..., "content_type": "application/json", "body_json": {...}}
7
+ {"v": 1, "received_at_ns": 17..., "content_type": "application/x-protobuf", "body_base64": ".."}
8
+
9
+ JSON bodies are stored as JSON so the file stays readable; protobuf bodies as base64.
10
+
11
+ The same format is used for three things:
12
+ - opt-in persistence (`loopview --persist FILE`): appended to as requests arrive,
13
+ reloaded on the next start;
14
+ - test fixtures: real framework output, replayed byte for byte into the receiver;
15
+ - the flagship demo recording.
16
+
17
+ Storing raw requests rather than decoded spans means a capture can be re-decoded
18
+ by a newer version of loopview, and the arrival timing is kept for liveness tests.
19
+ """
20
+
21
+ import base64
22
+ import json
23
+ import logging
24
+ from collections.abc import Iterator
25
+ from dataclasses import dataclass
26
+ from pathlib import Path
27
+ from typing import IO
28
+
29
+ from loopview.ingest.otlp import JSON, OtlpDecodeError, decode_body
30
+ from loopview.store.memory import TraceStore
31
+
32
+ FORMAT_VERSION = 1
33
+
34
+ log = logging.getLogger(__name__)
35
+
36
+
37
+ @dataclass(frozen=True)
38
+ class CapturedRequest:
39
+ received_at_ns: int
40
+ content_type: str
41
+ body: bytes
42
+
43
+
44
+ def to_line(request: CapturedRequest) -> str:
45
+ record: dict[str, object] = {
46
+ "v": FORMAT_VERSION,
47
+ "received_at_ns": request.received_at_ns,
48
+ "content_type": request.content_type,
49
+ }
50
+ if request.content_type.lower().startswith(JSON):
51
+ record["body_json"] = json.loads(request.body)
52
+ else:
53
+ record["body_base64"] = base64.b64encode(request.body).decode()
54
+ return json.dumps(record, separators=(",", ":"))
55
+
56
+
57
+ def from_line(line: str) -> CapturedRequest:
58
+ record = json.loads(line)
59
+ if record.get("v") != FORMAT_VERSION:
60
+ raise ValueError(f"unsupported capture format version: {record.get('v')!r}")
61
+ if "body_json" in record:
62
+ body = json.dumps(record["body_json"]).encode()
63
+ else:
64
+ body = base64.b64decode(record["body_base64"])
65
+ return CapturedRequest(record["received_at_ns"], record["content_type"], body)
66
+
67
+
68
+ def read_capture(path: Path) -> Iterator[CapturedRequest]:
69
+ """Yield every request in a capture file, skipping lines that don't parse.
70
+
71
+ Skipping (with a warning) matters for persistence: if loopview was killed
72
+ mid-write, the last line may be truncated and should not block the next start.
73
+ """
74
+ with path.open(encoding="utf-8") as f:
75
+ for number, line in enumerate(f, start=1):
76
+ if not line.strip():
77
+ continue
78
+ try:
79
+ yield from_line(line)
80
+ except (ValueError, KeyError, TypeError) as exc:
81
+ log.warning("skipping line %d of %s: %s", number, path, exc)
82
+
83
+
84
+ def load_capture(store: TraceStore, path: Path) -> int:
85
+ """Replay a capture file into the store, keeping the recorded arrival times.
86
+ Returns the number of requests loaded."""
87
+ loaded = 0
88
+ for request in read_capture(path):
89
+ try:
90
+ spans = decode_body(request.body, request.content_type)
91
+ except OtlpDecodeError as exc:
92
+ log.warning("skipping undecodable request in %s: %s", path, exc)
93
+ continue
94
+ store.add_spans(spans, received_at_ns=request.received_at_ns)
95
+ loaded += 1
96
+ return loaded
97
+
98
+
99
+ class CaptureWriter:
100
+ """Appends requests to a capture file, flushing after each one."""
101
+
102
+ def __init__(self, path: Path) -> None:
103
+ self.path = path
104
+ path.parent.mkdir(parents=True, exist_ok=True)
105
+ self._file: IO[str] = path.open("a", encoding="utf-8")
106
+
107
+ def write(self, request: CapturedRequest) -> None:
108
+ self._file.write(to_line(request) + "\n")
109
+ self._file.flush()
110
+
111
+ def close(self) -> None:
112
+ self._file.close()
@@ -0,0 +1,165 @@
1
+ """In-memory store of recent traces: a bounded ring buffer of runs.
2
+
3
+ - A run is one trace: every span sharing a trace id.
4
+ - A session is several runs sharing a conversation or session id attribute.
5
+ - The store keeps at most `max_runs` runs. When it is full, the run that was
6
+ updated least recently is dropped, so a run that is still receiving spans is
7
+ never evicted in favour of an idle one.
8
+
9
+ No locking: the store is only touched from the asyncio event loop (the receiver
10
+ endpoint is `async def`), so calls never overlap.
11
+ """
12
+
13
+ import time
14
+ from collections import OrderedDict
15
+ from collections.abc import Callable, Iterable
16
+ from dataclasses import dataclass, field
17
+
18
+ from pydantic import BaseModel
19
+
20
+ from loopview.ingest.raw import RawSpan
21
+
22
+ # Attributes that tie runs into a session, in priority order:
23
+ # OTel GenAI (gen_ai.conversation.id) and OpenInference (session.id).
24
+ SESSION_ID_ATTRIBUTES = ("gen_ai.conversation.id", "session.id")
25
+
26
+ DEFAULT_MAX_RUNS = 200
27
+
28
+
29
+ @dataclass
30
+ class Run:
31
+ trace_id: str
32
+ first_received_ns: int
33
+ last_received_ns: int
34
+ spans: dict[str, RawSpan] = field(default_factory=dict) # by span id
35
+ # When each span arrived at the server. Arrival order and timing are what the
36
+ # liveness inference (M4) uses as a heartbeat.
37
+ received_ns: dict[str, int] = field(default_factory=dict)
38
+ # Spans reported as started (by loopview-sdk) that have not ended yet. Only
39
+ # in-progress spans are kept here; the ended span replaces its start report.
40
+ started: dict[str, RawSpan] = field(default_factory=dict)
41
+
42
+ def add(self, span: RawSpan, received_at_ns: int) -> None:
43
+ # A re-sent span (exporter retry) replaces the old copy but keeps its first
44
+ # arrival time.
45
+ self.received_ns.setdefault(span.span_id, received_at_ns)
46
+ self.spans[span.span_id] = span
47
+ self.started.pop(span.span_id, None)
48
+ self.last_received_ns = max(self.last_received_ns, received_at_ns)
49
+
50
+ def add_started(self, span: RawSpan, received_at_ns: int) -> None:
51
+ if span.span_id not in self.spans: # the end can overtake the start report
52
+ self.started[span.span_id] = span
53
+ self.last_received_ns = max(self.last_received_ns, received_at_ns)
54
+
55
+ @property
56
+ def session_id(self) -> str | None:
57
+ """The session id of the shallowest span that has one.
58
+
59
+ Shallowest, because nested agents can carry their own conversation ids and
60
+ the outermost one describes the run. Computed on demand, since spans arrive
61
+ in any order (children usually first)."""
62
+ best: tuple[int, str] | None = None
63
+ for span in self.spans.values():
64
+ session = _session_id(span)
65
+ if session is None:
66
+ continue
67
+ depth, parent = 0, span.parent_span_id
68
+ while parent in self.spans:
69
+ depth, parent = depth + 1, self.spans[parent].parent_span_id
70
+ if best is None or depth < best[0]:
71
+ best = (depth, session)
72
+ return best[1] if best else None
73
+
74
+ @property
75
+ def root(self) -> RawSpan | None:
76
+ """The span with no parent, or None if it hasn't arrived yet.
77
+
78
+ Spans are exported when they end, and a root ends last, so a run that is
79
+ still going usually has no root yet.
80
+ """
81
+ for span in self.spans.values():
82
+ if span.parent_span_id is None:
83
+ return span
84
+ return None
85
+
86
+
87
+ class SessionSummary(BaseModel):
88
+ session_id: str
89
+ trace_ids: list[str] # oldest run first
90
+
91
+
92
+ class TraceStore:
93
+ def __init__(
94
+ self, max_runs: int = DEFAULT_MAX_RUNS, clock: Callable[[], int] = time.time_ns
95
+ ) -> None:
96
+ self.max_runs = max_runs
97
+ self._clock = clock
98
+ # Ordered from least to most recently updated, so eviction is popitem(last=False).
99
+ self._runs: OrderedDict[str, Run] = OrderedDict()
100
+
101
+ def add_spans(self, spans: Iterable[RawSpan], received_at_ns: int | None = None) -> list[str]:
102
+ """Store spans from one export request. Returns the trace ids that changed."""
103
+ now = self._clock() if received_at_ns is None else received_at_ns
104
+ changed: list[str] = []
105
+ for span in spans:
106
+ run = self._runs.get(span.trace_id)
107
+ if run is None:
108
+ run = Run(span.trace_id, first_received_ns=now, last_received_ns=now)
109
+ self._runs[span.trace_id] = run
110
+ run.add(span, now)
111
+ self._runs.move_to_end(span.trace_id)
112
+ if span.trace_id not in changed:
113
+ changed.append(span.trace_id)
114
+ while len(self._runs) > self.max_runs:
115
+ self._runs.popitem(last=False)
116
+ return changed
117
+
118
+ def add_started_spans(
119
+ self, spans: Iterable[RawSpan], received_at_ns: int | None = None
120
+ ) -> list[str]:
121
+ """Store start reports from loopview-sdk. Returns the trace ids that changed."""
122
+ now = self._clock() if received_at_ns is None else received_at_ns
123
+ changed: list[str] = []
124
+ for span in spans:
125
+ run = self._runs.get(span.trace_id)
126
+ if run is None:
127
+ run = Run(span.trace_id, first_received_ns=now, last_received_ns=now)
128
+ self._runs[span.trace_id] = run
129
+ run.add_started(span, now)
130
+ self._runs.move_to_end(span.trace_id)
131
+ if span.trace_id not in changed:
132
+ changed.append(span.trace_id)
133
+ while len(self._runs) > self.max_runs:
134
+ self._runs.popitem(last=False)
135
+ return changed
136
+
137
+ def get_run(self, trace_id: str) -> Run | None:
138
+ return self._runs.get(trace_id)
139
+
140
+ def runs(self) -> list[Run]:
141
+ """All runs, most recently started first."""
142
+ return sorted(self._runs.values(), key=lambda r: r.first_received_ns, reverse=True)
143
+
144
+ def sessions(self) -> list[SessionSummary]:
145
+ """Sessions, derived on demand. With a few hundred runs this is cheap and
146
+ avoids keeping a second index in sync with eviction."""
147
+ by_session: dict[str, list[Run]] = {}
148
+ for run in self._runs.values():
149
+ if run.session_id is not None:
150
+ by_session.setdefault(run.session_id, []).append(run)
151
+ return [
152
+ SessionSummary(
153
+ session_id=session_id,
154
+ trace_ids=[r.trace_id for r in sorted(runs, key=lambda r: r.first_received_ns)],
155
+ )
156
+ for session_id, runs in by_session.items()
157
+ ]
158
+
159
+
160
+ def _session_id(span: RawSpan) -> str | None:
161
+ for key in SESSION_ID_ATTRIBUTES:
162
+ value = span.attributes.get(key)
163
+ if value:
164
+ return str(value)
165
+ return None
@@ -0,0 +1 @@
1
+ """The Tools tab: how tools behave across many runs (report.py)."""
@@ -0,0 +1,347 @@
1
+ """Which tools an agent struggles with, across many runs.
2
+
3
+ Pure functions over normalized runs: no I/O, no store. The API picks the runs.
4
+
5
+ Definitions (also in DECISIONS.md D44):
6
+
7
+ - Tool call: a finished `tool_call` step. Running calls are left out.
8
+ - Error: the call is marked failed by the trace: span status ERROR (with or
9
+ without an exception event), or an MCP result with `isError: true`. Errors are
10
+ never guessed from the result text.
11
+ - Canonical arguments: the arguments as JSON with sorted keys, so identical calls
12
+ compare equal.
13
+ - Agent: the nearest agent above the call, by name, within its run.
14
+ - Next move after an error: what the agent does once the error is back in front
15
+ of the model. Models often request several tools in one turn, so the call that
16
+ happens to start next was usually requested before the model saw the error.
17
+ So each tool call is assigned to the model call that requested it (the
18
+ agent's latest model call that had ended when the tool started), and we look
19
+ at the turns whose model call started after the failed call ended:
20
+ blind_retry the same tool with identical canonical arguments
21
+ fixed the same tool with different arguments (and whether that worked)
22
+ switched a different tool, recorded as the pair (failed -> next)
23
+ gave_up no further tool call by that agent in that run
24
+ Calls to tools that also failed in the failed call's own turn are skipped:
25
+ they are those tools' own retries. When an agent has no recorded model calls,
26
+ the next move is simply its next tool call started after the error ended.
27
+ - Error group: the error message with UUIDs, hex ids, long quoted values and
28
+ numbers replaced by placeholders, trimmed to 200 characters.
29
+ - Confused pair: (A -> B) from `switched`, counted across runs; reported from two
30
+ occurrences.
31
+ - Offered tools: the tool definitions recorded on model calls. If no model call
32
+ in the input records any, the tool list is "not recorded" and nothing is
33
+ inferred about unused tools.
34
+ - Never called: offered at least once, called zero times in the input runs.
35
+ - Definition cost (estimate): a definition's JSON length / 4, times the number of
36
+ model calls that carried it.
37
+ - Result size (estimate): the average result length / 4 over successful calls
38
+ (Pydantic AI records its retry prompt as the result of a failed call).
39
+ """
40
+
41
+ import json
42
+ import math
43
+ import re
44
+ from collections import Counter
45
+ from dataclasses import dataclass, field
46
+ from typing import Any, Literal
47
+
48
+ from loopview.normalize.schema import NormalizedRun, Step
49
+
50
+ CHARS_PER_TOKEN = 4
51
+ ERROR_GROUP_MAX = 200
52
+ TOP_ERRORS = 3
53
+ MIN_CONFUSED = 2
54
+
55
+ Move = Literal["blind_retry", "fixed", "switched", "gave_up"]
56
+
57
+
58
+ # --- small helpers -------------------------------------------------------------------
59
+
60
+
61
+ def canonical_args(arguments: Any) -> str:
62
+ return json.dumps(arguments, sort_keys=True, default=str, ensure_ascii=False)
63
+
64
+
65
+ def estimate_tokens(value: Any) -> int:
66
+ text = value if isinstance(value, str) else canonical_args(value)
67
+ return math.ceil(len(text) / CHARS_PER_TOKEN) # anything non-empty is a token
68
+
69
+
70
+ _UUID = re.compile(
71
+ r"\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b"
72
+ )
73
+ # 0x-prefixed, or a run of 8+ hex characters that contains a digit (ids, hashes)
74
+ _HEX = re.compile(r"\b0x[0-9a-fA-F]+\b|\b(?=[0-9a-fA-F]*\d)[0-9a-fA-F]{8,}\b")
75
+ _LONG_QUOTED = re.compile(r"'[^']{13,}'|\"[^\"]{13,}\"")
76
+ _NUMBER = re.compile(r"\d+(?:\.\d+)?")
77
+
78
+
79
+ def error_group(message: str) -> str:
80
+ text = _UUID.sub("<uuid>", message)
81
+ text = _HEX.sub("<id>", text)
82
+ text = _LONG_QUOTED.sub("<value>", text)
83
+ text = _NUMBER.sub("<n>", text)
84
+ return " ".join(text.split())[:ERROR_GROUP_MAX]
85
+
86
+
87
+ def _mcp_is_error(result: Any) -> bool:
88
+ return isinstance(result, dict) and result.get("isError") is True
89
+
90
+
91
+ def _mcp_error_text(result: dict[str, Any]) -> str:
92
+ texts = [
93
+ str(c.get("text"))
94
+ for c in result.get("content") or []
95
+ if isinstance(c, dict) and c.get("text")
96
+ ]
97
+ return "\n".join(texts) or "isError"
98
+
99
+
100
+ def definition_name(definition: Any) -> str | None:
101
+ """Tool definitions come as {"name": ...} (GenAI, Anthropic, OpenInference) or
102
+ OpenAI's {"type": "function", "function": {"name": ...}}."""
103
+ if not isinstance(definition, dict):
104
+ return None
105
+ name = definition.get("name")
106
+ if name is None and isinstance(definition.get("function"), dict):
107
+ name = definition["function"].get("name")
108
+ return str(name) if name else None
109
+
110
+
111
+ # --- calls -----------------------------------------------------------------------------
112
+
113
+
114
+ @dataclass
115
+ class Call:
116
+ run_id: str
117
+ step_id: str
118
+ agent: str
119
+ name: str
120
+ start: int
121
+ end: int
122
+ args: Any
123
+ canonical: str
124
+ result: Any
125
+ error: str | None # None when the call succeeded
126
+
127
+ @property
128
+ def failed(self) -> bool:
129
+ return self.error is not None
130
+
131
+ def ref(self) -> dict[str, str]:
132
+ return {"run_id": self.run_id, "step_id": self.step_id}
133
+
134
+
135
+ def _agent_of(step: Step, by_id: dict[str, Step]) -> str:
136
+ parent = by_id.get(step.parent_id) if step.parent_id else None
137
+ while parent is not None:
138
+ # A synthetic `tools` step can be a group (it holds sub-agents); it isn't an agent.
139
+ if parent.kind == "agent" and not parent.synthetic:
140
+ return parent.name
141
+ parent = by_id.get(parent.parent_id) if parent.parent_id else None
142
+ return "" # a call outside any agent: the run itself is its agent
143
+
144
+
145
+ def _calls(run: NormalizedRun, by_id: dict[str, Step]) -> list[Call]:
146
+ calls = []
147
+ for s in run.steps:
148
+ if s.kind != "tool_call" or s.tool is None or s.end_ns is None or s.status == "running":
149
+ continue
150
+ error = None
151
+ if s.status == "error":
152
+ error = s.error or "error"
153
+ elif _mcp_is_error(s.tool.result):
154
+ error = _mcp_error_text(s.tool.result)
155
+ calls.append(
156
+ Call(
157
+ run_id=s.run_id,
158
+ step_id=s.id,
159
+ agent=_agent_of(s, by_id),
160
+ name=s.tool.name or s.name,
161
+ start=s.start_ns,
162
+ end=s.end_ns,
163
+ args=s.tool.arguments,
164
+ canonical=canonical_args(s.tool.arguments),
165
+ result=s.tool.result,
166
+ error=error,
167
+ )
168
+ )
169
+ return sorted(calls, key=lambda c: (c.start, c.step_id))
170
+
171
+
172
+ # --- next move after an error -----------------------------------------------------------
173
+
174
+
175
+ @dataclass
176
+ class NextMove:
177
+ move: Move
178
+ next_tool: str | None = None # for switched
179
+ succeeded: bool | None = None # for fixed
180
+
181
+
182
+ def _turn_of(call: Call, model_calls: list[tuple[int, int]]) -> int:
183
+ """Index of the model call that requested `call`: the latest one that had
184
+ ended when the call started. Comparing with the end, not the start, keeps
185
+ this right when timestamps tie (a coarse clock, DECISIONS D17). -1: none."""
186
+ turn = -1
187
+ for i, (_, end) in enumerate(model_calls):
188
+ if end <= call.start:
189
+ turn = i
190
+ return turn
191
+
192
+
193
+ def next_move(
194
+ failed: Call, agent_calls: list[Call], model_calls: list[tuple[int, int]]
195
+ ) -> NextMove:
196
+ """What the agent did after `failed` (see the module docstring).
197
+ `model_calls` are the agent's model calls as (start, end), sorted by start."""
198
+ others = [c for c in agent_calls if c is not failed]
199
+ if not model_calls:
200
+ later = [c for c in others if c.start >= failed.end]
201
+ turns = [later[:1]] if later else []
202
+ skip: set[str] = set()
203
+ else:
204
+ own = _turn_of(failed, model_calls)
205
+ by_turn: dict[int, list[Call]] = {}
206
+ for c in others:
207
+ by_turn.setdefault(_turn_of(c, model_calls), []).append(c)
208
+ # Turns whose model call started once the error was back: it saw the error.
209
+ turns = [by_turn[i] for i in sorted(by_turn) if i > own and model_calls[i][0] >= failed.end]
210
+ skip = {c.name for c in by_turn.get(own, []) if c.failed and c.name != failed.name}
211
+ for turn in turns:
212
+ same = [c for c in turn if c.name == failed.name]
213
+ if any(c.canonical == failed.canonical for c in same):
214
+ return NextMove("blind_retry")
215
+ if same:
216
+ return NextMove("fixed", succeeded=not same[0].failed)
217
+ switched = [c for c in turn if c.name not in skip]
218
+ if switched:
219
+ return NextMove("switched", next_tool=switched[0].name)
220
+ return NextMove("gave_up")
221
+
222
+
223
+ # --- the report ----------------------------------------------------------------------------
224
+
225
+
226
+ @dataclass
227
+ class _ToolStats:
228
+ calls: list[Call] = field(default_factory=list)
229
+ moves: Counter[str] = field(default_factory=Counter)
230
+ switched_to: Counter[str] = field(default_factory=Counter)
231
+
232
+
233
+ def tools_report(runs: list[NormalizedRun]) -> dict[str, Any]:
234
+ stats: dict[str, _ToolStats] = {}
235
+ pairs: Counter[tuple[str, str]] = Counter()
236
+ offered_size: dict[str, int] = {} # tool -> definition size in tokens (largest seen)
237
+ carried: Counter[str] = Counter() # tool -> model calls that carried its definition
238
+ tool_list_recorded = False
239
+ all_calls: list[Call] = []
240
+
241
+ for run in runs:
242
+ by_id = {s.id: s for s in run.steps}
243
+ calls = _calls(run, by_id)
244
+ all_calls += calls
245
+ model_calls: dict[str, list[tuple[int, int]]] = {}
246
+ for s in run.steps:
247
+ if s.kind != "model_call":
248
+ continue
249
+ # A running model call hasn't requested anything yet.
250
+ end = s.end_ns if s.end_ns is not None else 2**63
251
+ model_calls.setdefault(_agent_of(s, by_id), []).append((s.start_ns, end))
252
+ definitions = s.model.tool_definitions if s.model else []
253
+ if definitions:
254
+ tool_list_recorded = True
255
+ for d in definitions:
256
+ name = definition_name(d)
257
+ if name:
258
+ carried[name] += 1
259
+ offered_size[name] = max(offered_size.get(name, 0), estimate_tokens(d))
260
+ for spans in model_calls.values():
261
+ spans.sort()
262
+
263
+ by_agent: dict[str, list[Call]] = {}
264
+ for c in calls:
265
+ by_agent.setdefault(c.agent, []).append(c)
266
+ stats.setdefault(c.name, _ToolStats()).calls.append(c)
267
+ for c in calls:
268
+ if not c.failed:
269
+ continue
270
+ move = next_move(c, by_agent[c.agent], model_calls.get(c.agent, []))
271
+ tool = stats[c.name]
272
+ tool.moves[move.move] += 1
273
+ if move.move == "fixed" and move.succeeded:
274
+ tool.moves["fixed_succeeded"] += 1
275
+ if move.move == "switched" and move.next_tool:
276
+ tool.switched_to[move.next_tool] += 1
277
+ pairs[(c.name, move.next_tool)] += 1
278
+
279
+ tools = [_tool_entry(name, s, pairs) for name, s in stats.items()]
280
+ tools.sort(key=lambda t: (-t["errors"], -t["calls"], t["name"]))
281
+
282
+ called = set(stats)
283
+ never_called = []
284
+ if tool_list_recorded:
285
+ never_called = [
286
+ {
287
+ "name": name,
288
+ "definition_tokens_estimate": offered_size[name] * carried[name],
289
+ "carried_by_model_calls": carried[name],
290
+ }
291
+ for name in sorted(set(carried) - called)
292
+ ]
293
+ never_called.sort(key=lambda t: (-t["definition_tokens_estimate"], t["name"]))
294
+
295
+ errors = sum(t["errors"] for t in tools)
296
+ top_two = sum(sorted((t["errors"] for t in tools), reverse=True)[:2])
297
+ never_tokens = sum(t["definition_tokens_estimate"] for t in never_called)
298
+ return {
299
+ "summary": {
300
+ "runs": len(runs),
301
+ "tool_calls": len(all_calls),
302
+ "errors": errors,
303
+ "share_of_errors_from_top_2_tools": round(top_two / errors, 3) if errors else None,
304
+ "never_called_count": len(never_called),
305
+ "never_called_tokens_estimate": never_tokens,
306
+ "never_called_tokens_per_run_estimate": round(never_tokens / len(runs)) if runs else 0,
307
+ },
308
+ "tools": tools,
309
+ "never_called": never_called,
310
+ "tool_list_recorded": tool_list_recorded,
311
+ }
312
+
313
+
314
+ def _tool_entry(name: str, s: _ToolStats, pairs: Counter[tuple[str, str]]) -> dict[str, Any]:
315
+ failures = [c for c in s.calls if c.failed]
316
+ groups: dict[str, list[Call]] = {}
317
+ for c in failures:
318
+ groups.setdefault(error_group(c.error or ""), []).append(c)
319
+ top = sorted(groups.items(), key=lambda g: (-len(g[1]), g[0]))[:TOP_ERRORS]
320
+ results = [estimate_tokens(c.result) for c in s.calls if not c.failed and c.result is not None]
321
+ return {
322
+ "name": name,
323
+ "calls": len(s.calls),
324
+ "errors": len(failures),
325
+ "error_rate": round(len(failures) / len(s.calls), 3),
326
+ "after_error": {
327
+ k: s.moves[k]
328
+ for k in ("blind_retry", "fixed", "fixed_succeeded", "switched", "gave_up")
329
+ },
330
+ "top_errors": [
331
+ {
332
+ "message_group": group,
333
+ "count": len(examples),
334
+ "example_args": examples[0].args,
335
+ "example_message": examples[0].error,
336
+ "step_ref": examples[0].ref(),
337
+ }
338
+ for group, examples in top
339
+ ],
340
+ "confused_with": [
341
+ {"tool": other, "count": n}
342
+ for (a, other), n in pairs.most_common()
343
+ if a == name and n >= MIN_CONFUSED
344
+ ],
345
+ "avg_result_tokens_estimate": round(sum(results) / len(results)) if results else None,
346
+ "step_refs": [c.ref() for c in s.calls],
347
+ }