@coreplane/switchboard 0.0.0 → 1.18.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +17 -1
  3. package/dist/assets/.dockerignore +27 -0
  4. package/dist/assets/.env.example +33 -0
  5. package/dist/assets/Dockerfile +111 -0
  6. package/dist/assets/config/config.example.yaml +359 -0
  7. package/dist/assets/deploy/bin/build-stamp.d.mts +15 -0
  8. package/dist/assets/deploy/bin/build-stamp.mjs +98 -0
  9. package/dist/assets/deploy/bin/cf-logs +32 -0
  10. package/dist/assets/deploy/cloudflare/package.json +29 -0
  11. package/dist/assets/deploy/cloudflare/preflight.mjs +243 -0
  12. package/dist/assets/deploy/cloudflare/tsconfig.json +18 -0
  13. package/dist/assets/deploy/cloudflare/worker.ts +382 -0
  14. package/dist/assets/deploy/cloudflare/wrangler.template.jsonc +67 -0
  15. package/dist/assets/deploy/cloudflare/write-build.d.mts +7 -0
  16. package/dist/assets/deploy/cloudflare/write-build.mjs +53 -0
  17. package/dist/assets/deploy/cloudflare-docs/package.json +18 -0
  18. package/dist/assets/deploy/cloudflare-docs/wrangler.template.jsonc +30 -0
  19. package/dist/assets/deploy/cloudflare-memory/package.json +25 -0
  20. package/dist/assets/deploy/cloudflare-memory/tsconfig.json +17 -0
  21. package/dist/assets/deploy/cloudflare-memory/worker.ts +2635 -0
  22. package/dist/assets/deploy/cloudflare-memory/wrangler.template.jsonc +50 -0
  23. package/dist/assets/deploy/cloudflare-resident/Dockerfile +91 -0
  24. package/dist/assets/deploy/cloudflare-resident/gc.ts +287 -0
  25. package/dist/assets/deploy/cloudflare-resident/node-async-hooks.d.ts +11 -0
  26. package/dist/assets/deploy/cloudflare-resident/package.json +29 -0
  27. package/dist/assets/deploy/cloudflare-resident/preflight.mjs +224 -0
  28. package/dist/assets/deploy/cloudflare-resident/tsconfig.json +19 -0
  29. package/dist/assets/deploy/cloudflare-resident/worker.ts +6637 -0
  30. package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +120 -0
  31. package/dist/assets/deploy/cloudflare-sandbox/Dockerfile +67 -0
  32. package/dist/assets/deploy/cloudflare-sandbox/docker-wrapper.sh +37 -0
  33. package/dist/assets/deploy/cloudflare-sandbox/package.json +26 -0
  34. package/dist/assets/deploy/cloudflare-sandbox/tsconfig.json +20 -0
  35. package/dist/assets/deploy/cloudflare-sandbox/worker.ts +410 -0
  36. package/dist/assets/deploy/cloudflare-sandbox/wrangler.template.jsonc +67 -0
  37. package/dist/assets/deploy/profile.example.json +13 -0
  38. package/dist/assets/deploy/secrets.manifest.json +108 -0
  39. package/dist/assets/docker-entrypoint.sh +15 -0
  40. package/dist/assets/package-lock.json +18407 -0
  41. package/dist/assets/package.json +104 -0
  42. package/dist/assets/project.json +219 -0
  43. package/dist/assets/source.json +5 -0
  44. package/dist/assets/src/core/authz/actor.ts +100 -0
  45. package/dist/assets/src/core/authz/authorize.ts +169 -0
  46. package/dist/assets/src/core/authz/grants.ts +347 -0
  47. package/dist/assets/src/core/authz/policy.ts +281 -0
  48. package/dist/assets/src/core/authz/resource.ts +147 -0
  49. package/dist/assets/src/core/authz/types.ts +164 -0
  50. package/dist/assets/src/core/drain.ts +54 -0
  51. package/dist/assets/src/core/ingressTokens.ts +64 -0
  52. package/dist/assets/src/core/memory/engine.ts +115 -0
  53. package/dist/assets/src/core/memory/scorer.ts +147 -0
  54. package/dist/assets/src/core/memory/types.ts +120 -0
  55. package/dist/assets/src/core/normalizeSpans.ts +299 -0
  56. package/dist/assets/src/core/prDescriptionTypes.ts +54 -0
  57. package/dist/assets/src/core/redact.ts +113 -0
  58. package/dist/assets/src/core/runEvents.ts +537 -0
  59. package/dist/assets/src/core/runFriction.ts +665 -0
  60. package/dist/assets/src/core/runLedger/decisions.ts +126 -0
  61. package/dist/assets/src/core/runLedger/types.ts +177 -0
  62. package/dist/assets/src/core/runRecord.ts +627 -0
  63. package/dist/assets/src/core/runShape.ts +61 -0
  64. package/dist/assets/src/core/schedules.ts +452 -0
  65. package/dist/assets/src/core/time/formatDuration.ts +61 -0
  66. package/dist/assets/src/core/trace/attrs.ts +203 -0
  67. package/dist/assets/src/core/trace/classify.ts +49 -0
  68. package/dist/assets/src/core/trace/clock.ts +6 -0
  69. package/dist/assets/src/core/trace/context.ts +9 -0
  70. package/dist/assets/src/core/trace/ids.ts +23 -0
  71. package/dist/assets/src/core/trace/partition.ts +235 -0
  72. package/dist/assets/src/core/trace/sinks.ts +68 -0
  73. package/dist/assets/src/core/trace/streamSpans.ts +163 -0
  74. package/dist/assets/src/core/trace/traceparent.ts +29 -0
  75. package/dist/assets/src/core/trace/tracer.ts +247 -0
  76. package/dist/assets/src/core/trace/types.ts +125 -0
  77. package/dist/assets/src/core/trace/workerTrace.ts +97 -0
  78. package/dist/assets/src/deploy/buildStamp.ts +93 -0
  79. package/dist/assets/src/deploy/liveGate.ts +203 -0
  80. package/dist/assets/src/deploy/profile.ts +162 -0
  81. package/dist/assets/src/deploy/restart.ts +393 -0
  82. package/dist/assets/src/effort.ts +17 -0
  83. package/dist/assets/src/execution/bashTimeout.ts +78 -0
  84. package/dist/assets/src/execution/bindingPurge.ts +43 -0
  85. package/dist/assets/src/execution/residentBackupTransfer.ts +50 -0
  86. package/dist/assets/src/execution/residentCleanliness.ts +95 -0
  87. package/dist/assets/src/execution/residentCredentials.ts +81 -0
  88. package/dist/assets/src/execution/residentDepCache.ts +321 -0
  89. package/dist/assets/src/execution/residentDepsStore.ts +326 -0
  90. package/dist/assets/src/execution/residentDetach.ts +48 -0
  91. package/dist/assets/src/execution/residentDisk.ts +107 -0
  92. package/dist/assets/src/execution/residentDiskBudget.ts +448 -0
  93. package/dist/assets/src/execution/residentExecWrap.ts +100 -0
  94. package/dist/assets/src/execution/residentHead.ts +85 -0
  95. package/dist/assets/src/execution/residentReadonly.ts +72 -0
  96. package/dist/assets/src/execution/residentRefresh.ts +429 -0
  97. package/dist/assets/src/execution/residentRestoreExtract.ts +130 -0
  98. package/dist/assets/src/execution/residentState.ts +47 -0
  99. package/dist/assets/src/execution/residentStepReport.ts +98 -0
  100. package/dist/assets/src/execution/residentStepTrace.ts +97 -0
  101. package/dist/assets/src/execution/residentSteps.ts +99 -0
  102. package/dist/assets/src/execution/residentText.ts +83 -0
  103. package/dist/assets/src/execution/residentTrace.ts +119 -0
  104. package/dist/assets/src/execution/sandboxEnv.ts +42 -0
  105. package/dist/assets/src/execution/sandboxErrors.ts +159 -0
  106. package/dist/assets/src/execution/sandboxKeepalive.ts +118 -0
  107. package/dist/assets/src/execution/shellQuote.ts +8 -0
  108. package/dist/assets/src/mcp/registry.ts +242 -0
  109. package/dist/assets/src/providers/types.ts +152 -0
  110. package/dist/assets/web/dist/.vite/manifest.json +176 -0
  111. package/dist/assets/web/dist/assets/AppShell-Bk2gbvet.js +1 -0
  112. package/dist/assets/web/dist/assets/CostsPage-CTZcMYYx.js +1 -0
  113. package/dist/assets/web/dist/assets/NotFoundPage-C-BuaSm8.js +1 -0
  114. package/dist/assets/web/dist/assets/ResidentDetailPage-DvQ05AGa.js +1 -0
  115. package/dist/assets/web/dist/assets/ResidentsIndexPage-B3uxKUne.js +1 -0
  116. package/dist/assets/web/dist/assets/RunRoutePage-XVFj0XDc.css +1 -0
  117. package/dist/assets/web/dist/assets/RunRoutePage-ty94olNM.js +126 -0
  118. package/dist/assets/web/dist/assets/RunsIndexPage-CM-qxyQm.js +1 -0
  119. package/dist/assets/web/dist/assets/RunsTabs-C4krAL9o.js +1 -0
  120. package/dist/assets/web/dist/assets/ScheduledPage-C1psvLD4.js +1 -0
  121. package/dist/assets/web/dist/assets/StatusDot-DuoQnQeU.js +1 -0
  122. package/dist/assets/web/dist/assets/Tooltip-BfLPyxQy.js +1 -0
  123. package/dist/assets/web/dist/assets/favicon-DL1rdWJt.js +1 -0
  124. package/dist/assets/web/dist/assets/localIso-L06jV29p.js +1 -0
  125. package/dist/assets/web/dist/assets/main-Bnbk_Rsg.js +28 -0
  126. package/dist/assets/web/dist/assets/main-BsBGUyMH.css +2 -0
  127. package/dist/assets/web/dist/assets/residentDiskBudget-BMBKlYRH.js +1 -0
  128. package/dist/assets/web/dist/assets/seed-BglCRKLA.js +6 -0
  129. package/dist/assets/web/dist/assets/wallClock-Ckv3sKoR.js +1 -0
  130. package/dist/cli.js +34494 -0
  131. package/package.json +43 -10
@@ -0,0 +1,665 @@
1
+ import type { RunEvent } from "./runEvents.js";
2
+ import { isSpanRecord } from "./runEvents.js";
3
+ import { lossesFromStream, normalizeSpans, spansFromEvents } from "./normalizeSpans.js";
4
+ import { formatShape } from "./runShape.js";
5
+ import { formatDuration } from "./time/formatDuration.js";
6
+ import { partition, type LossInterval, type Partition, type Window } from "./trace/partition.js";
7
+ import type { RunOwner } from "./trace/streamSpans.js";
8
+ import type { SpanRecord } from "./trace/types.js";
9
+
10
+ // Run-friction analyzer (docs/reference/specs/run-friction.md): a PURE, deterministic
11
+ // function from a run's RunEvent stream to a structured diagnosis of what cost
12
+ // the run time or made it stumble — slow/failed tool calls, slow model turns,
13
+ // retries, setup/install time, wrap-up, budget hits, exec-infrastructure
14
+ // failures — and, for a finished run with a window, its shape: how the window
15
+ // splits into getting ready, thinking, tools, finishing up and overhead
16
+ // (docs/reference/specs/tracing.md item 5). It is the observe→diagnose half of the
17
+ // self-improvement loop; proposing fix PRs from a diagnosis is a later piece
18
+ // and deliberately NOT here. No clock, no I/O: the same events always yield the
19
+ // same diagnosis, so it runs identically over a live backlog
20
+ // (`/runs/:id/friction`), a saved JSONL stream (`friction analyze`), or a test
21
+ // fixture.
22
+ //
23
+ // Every duration comes from ONE span set: the stream normalized through
24
+ // `normalizeSpans` (a tool pair whose twin span the record budget dropped gets
25
+ // it back) and paired by `spansFromEvents`. Model time is the sum of the
26
+ // `model.turn` spans, tool time the sum of the `tool.*` spans, and each timed
27
+ // finding carries the duration of the span it is about — so a tool-denominated
28
+ // finding is a summand of tool time and a slow turn of model time, and no
29
+ // share can exceed 100 %. A stream with no spans (a hand-written capture, a
30
+ // record from before spans) has no durations: the same classification, every
31
+ // timed field absent — one code path, nothing special-cased.
32
+
33
+ export type FrictionCategory =
34
+ | "slow_tool"
35
+ | "slow_model_turn"
36
+ | "failed_tool"
37
+ | "retry"
38
+ | "setup_install"
39
+ | "wrap_up"
40
+ | "budget_hit"
41
+ | "infra_failure";
42
+
43
+ export const FRICTION_CATEGORIES: readonly FrictionCategory[] = [
44
+ "slow_tool",
45
+ "slow_model_turn",
46
+ "failed_tool",
47
+ "retry",
48
+ "setup_install",
49
+ "wrap_up",
50
+ "budget_hit",
51
+ "infra_failure",
52
+ ];
53
+
54
+ /** Human labels — the verdict line, the report's table and its finding lines. */
55
+ export const CATEGORY_LABEL: Record<FrictionCategory, string> = {
56
+ slow_tool: "slow tool calls",
57
+ slow_model_turn: "slow model turns",
58
+ failed_tool: "failed tool calls",
59
+ retry: "retries",
60
+ setup_install: "the repo's setup/install",
61
+ wrap_up: "agent wind-down",
62
+ budget_hit: "budget hits",
63
+ infra_failure: "infra failures",
64
+ };
65
+
66
+ /** What a category's time is a share OF: tool time for the categories whose
67
+ * findings are tool calls (each a summand of `toolTimeMs`), run time for the
68
+ * rest. A new category must choose, so no share can exceed 100 %. */
69
+ export const DENOMINATOR_OF: Record<FrictionCategory, "tool" | "run"> = {
70
+ slow_tool: "tool",
71
+ failed_tool: "tool",
72
+ retry: "tool",
73
+ setup_install: "tool",
74
+ slow_model_turn: "run",
75
+ wrap_up: "run",
76
+ budget_hit: "run",
77
+ infra_failure: "run",
78
+ };
79
+
80
+ export type FrictionSeverity = "low" | "medium" | "high";
81
+
82
+ export interface FrictionFinding {
83
+ category: FrictionCategory;
84
+ severity: FrictionSeverity;
85
+ /** One line: what happened. Derived from event summaries, which are already redacted upstream. */
86
+ summary: string;
87
+ /** Tool involved, for tool-anchored findings. */
88
+ tool?: string;
89
+ /** Wall time attributed to this finding, when the events carried timestamps:
90
+ * the duration of the span it is about. */
91
+ durationMs?: number;
92
+ /** The finding's own interval, for the run-denominated categories whose
93
+ * category time is the UNION of their findings (three calls of one batch
94
+ * dying together are one interval, not three). Absent on a note-anchored
95
+ * finding with no extent (a budget hit, a dead sandbox). */
96
+ interval?: { start: number; end: number };
97
+ /** Index into the input stream of the event the finding anchors to: the
98
+ * tool_call for tool findings, the note for note findings, the event a model
99
+ * turn produced (its `span_end`, or the synthesized twin's terminator) for a
100
+ * slow turn. */
101
+ eventIndex: number;
102
+ }
103
+
104
+ export interface CategoryTotals {
105
+ count: number;
106
+ /** The category's attributed time: the sum of its findings' `durationMs` for
107
+ * the tool-denominated categories and `slow_model_turn` (whose turns are
108
+ * disjoint), the union of its findings' intervals for the other
109
+ * run-denominated ones; 0 without timings. */
110
+ durationMs: number;
111
+ }
112
+
113
+ /** The run's shape (docs/reference/specs/tracing.md item 5): the seven terms of a finished
114
+ * window. Absent while a run is live or when no window was given. */
115
+ export type RunShape = Omit<Partition, "backgroundOnlyMs">;
116
+
117
+ export interface FrictionDiagnosis {
118
+ eventCount: number;
119
+ toolCalls: number;
120
+ /** The window (`receivedAt` → `finishedAt`, or now) when one was given; else
121
+ * first→last content stamp; absent when no content event carried one. */
122
+ runMs?: number;
123
+ /** Sum of the `tool.*` span durations, when the stream is timed (`runMs`). */
124
+ toolTimeMs?: number;
125
+ /** Sum of the `model.turn` span durations, when the stream had a turn and
126
+ * timings. Thinking and tool time are the counted terms of the shape; the
127
+ * rest of the window is setup, finishing up and Switchboard overhead. */
128
+ modelTimeMs?: number;
129
+ /** Every category present, zeroed when absent. */
130
+ byCategory: Record<FrictionCategory, CategoryTotals>;
131
+ /** In detection order (stream order). */
132
+ findings: FrictionFinding[];
133
+ /** One-line headline: the dominant cause, or "no friction detected". */
134
+ verdict: string;
135
+ /** Present (true) when the analyzed stream lost records — the registry's
136
+ * bounded backlog or the record budget dropped some before the diagnosis
137
+ * ran — so the counts and timings describe part of the run, not all of it.
138
+ * Optional: a stored diagnosis from before this field reads as complete. */
139
+ truncatedInput?: true;
140
+ /** The finished window's shape, when a window was given and the run finished. */
141
+ shape?: RunShape;
142
+ }
143
+
144
+ export interface FrictionOptions {
145
+ /** A paired tool call taking at least this long is a `slow_tool`. Default 30s. */
146
+ slowToolMs?: number;
147
+ /** A model turn taking at least this long is a `slow_model_turn`. Default 60s. */
148
+ slowModelTurnMs?: number;
149
+ /** Whether the stream is complete (default true). For an in-flight run a
150
+ * trailing tool_call without a result is simply still running — only in a
151
+ * finished stream is it evidence the run died mid-tool. */
152
+ finished?: boolean;
153
+ /** Whether `events` is a truncated stream (`RunSnapshot.truncated`): the
154
+ * diagnosis is then stamped `truncatedInput: true`. Default false. */
155
+ truncated?: boolean;
156
+ /** The run's window (docs/reference/specs/tracing.md): `receivedAt` to `finishedAt` (a
157
+ * record), or to now (a live read). With it `runMs` is the window and a
158
+ * finished diagnosis carries `shape`; open spans run to its end while live.
159
+ * Absent (a stdin capture): `runMs` is first→last over the content events
160
+ * and there is no shape. */
161
+ window?: Window;
162
+ /** Who owns the run: `run.command` counts as tools for a command run and as
163
+ * getting ready for an agent run. Default `agent`. */
164
+ owner?: RunOwner;
165
+ }
166
+
167
+ const DEFAULT_SLOW_TOOL_MS = 30_000;
168
+ const DEFAULT_SLOW_MODEL_TURN_MS = 60_000;
169
+
170
+ // Setup/install commands, matched at the START of a shell segment (after `$ `,
171
+ // `&&`, `;`, `|`), so `echo npm install` and `npm test` don't match while
172
+ // `cd repo && npm install` does. `(?=\s|$)` (not `\b`) ends each subcommand
173
+ // word, so `npm ci-lockfile-report` is not `npm ci`. Optional wrappers: `sudo`,
174
+ // `corepack`, `python -m` (for pip).
175
+ const INSTALL_SEGMENT =
176
+ /(?:^|&&|;|\|\|?)\s*(?:sudo\s+)?(?:corepack\s+)?(?:(?:npm|pnpm|bun)\s+(?:install|i|ci|add)(?=\s|$)|yarn(?:\s+(?:install|add)(?=\s|$)|\s*$)|npx\s+playwright\s+install(?=\s|$)|(?:python3?\s+-m\s+)?pip3?\s+install(?=\s|$)|poetry\s+install(?=\s|$)|uv\s+(?:sync|pip\s+install)(?=\s|$)|apt(?:-get)?\s+install(?=\s|$)|apk\s+add(?=\s|$)|brew\s+install(?=\s|$)|bundle\s+install(?=\s|$)|gem\s+install(?=\s|$)|cargo\s+(?:fetch|build)(?=\s|$)|go\s+mod\s+download(?=\s|$)|git\s+clone(?=\s|$)|make\s+(?:deps|install|setup)(?=\s|$))/;
177
+
178
+ // Quoted string literals ("…" / '…'): prose to the shell, not command segments.
179
+ const QUOTED = /"(?:[^"\\]|\\.)*"|'[^']*'/g;
180
+
181
+ /** True for a bash tool_call summary (`$ <command>`) that is a setup/install
182
+ * step. Quoted text is blanked first so `echo "cd x && npm install"` is not
183
+ * an install. Tolerates a non-string (a corrupted external capture) → false. */
184
+ export function isSetupInstallCommand(summary: unknown): boolean {
185
+ if (typeof summary !== "string") return false;
186
+ const cmd = summary.startsWith("$ ") ? summary.slice(2) : summary;
187
+ return INSTALL_SEGMENT.test(cmd.replace(QUOTED, '""'));
188
+ }
189
+
190
+ interface PendingCall {
191
+ index: number;
192
+ event: Extract<RunEvent, { type: "tool_call" }>;
193
+ }
194
+
195
+ /** The event types that tell the run's story rather than its steps — the
196
+ * narrative (`input`/`context`/`assistant`/`answer`) and what the run is about. */
197
+ type NarrativeEvent = Extract<RunEvent, { type: "input" | "context" | "assistant" | "answer" | "run_meta" }>;
198
+ function isNarrative(ev: RunEvent): ev is NarrativeEvent {
199
+ return (
200
+ ev.type === "input" ||
201
+ ev.type === "context" ||
202
+ ev.type === "assistant" ||
203
+ ev.type === "answer" ||
204
+ ev.type === "run_meta"
205
+ );
206
+ }
207
+
208
+ /** The span set the analyzer reads, with two indices per span end into the
209
+ * ORIGINAL stream: `anchorOf` — the event the span ended at (its own `span_end`
210
+ * when the stream carried one, else the first original event after the
211
+ * synthesized end) — and `producedOf` — the first content event after it, the
212
+ * event a model turn produced (its tool call, its narration, the answer). */
213
+ function spanSet(events: readonly RunEvent[]) {
214
+ const normalized = normalizeSpans(events);
215
+ const spans = spansFromEvents(normalized, "run");
216
+ const originalIndex = new Map<RunEvent, number>();
217
+ events.forEach((e, i) => originalIndex.set(e, i));
218
+ const anchorOf = new Map<string, number>();
219
+ const producedOf = new Map<string, number>();
220
+ // Walk from the end so each span end learns the originals that follow it.
221
+ let nextOriginal = events.length - 1;
222
+ let nextContent: number | undefined;
223
+ for (let p = normalized.length - 1; p >= 0; p--) {
224
+ const x = normalized[p]!;
225
+ const own = originalIndex.get(x);
226
+ if (x.type === "span_end") {
227
+ anchorOf.set(x.spanId, own ?? nextOriginal);
228
+ if (nextContent !== undefined) producedOf.set(x.spanId, nextContent);
229
+ }
230
+ if (own !== undefined) {
231
+ nextOriginal = own;
232
+ if (!isSpanRecord(x)) nextContent = own;
233
+ }
234
+ }
235
+ return { spans, anchorOf, producedOf };
236
+ }
237
+
238
+ function unionMs(intervals: ReadonlyArray<{ start: number; end: number }>): number {
239
+ const sorted = intervals
240
+ .filter((i) => i.end > i.start)
241
+ .map((i) => ({ ...i }))
242
+ .sort((a, b) => a.start - b.start);
243
+ let total = 0;
244
+ let cur: { start: number; end: number } | undefined;
245
+ for (const i of sorted) {
246
+ if (cur && i.start <= cur.end) {
247
+ cur.end = Math.max(cur.end, i.end);
248
+ continue;
249
+ }
250
+ if (cur) total += cur.end - cur.start;
251
+ cur = i;
252
+ }
253
+ if (cur) total += cur.end - cur.start;
254
+ return total;
255
+ }
256
+
257
+ /** Analyze a run's event stream. Pure and deterministic; never mutates `events`. */
258
+ export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOptions = {}): FrictionDiagnosis {
259
+ const slowToolMs = opts.slowToolMs ?? DEFAULT_SLOW_TOOL_MS;
260
+ const slowModelTurnMs = opts.slowModelTurnMs ?? DEFAULT_SLOW_MODEL_TURN_MS;
261
+ const finished = opts.finished ?? true;
262
+ const owner: RunOwner = opts.owner ?? "agent";
263
+ const window = opts.window;
264
+
265
+ // ---- the content pass: what happened, in stream order -------------------
266
+ const findings: FrictionFinding[] = [];
267
+ // Pending calls awaiting their result, FIFO per tool name (the runner emits
268
+ // call→result sequentially; pairing per tool is robust to interleaving).
269
+ const pending = new Map<string, PendingCall[]>();
270
+ // `tool summary` keys that have failed at least once → a later identical call is a retry.
271
+ const failedCalls = new Set<string>();
272
+ let toolCalls = 0;
273
+ let firstAt: number | undefined;
274
+ let lastAt: number | undefined;
275
+ // The content events' stamps: `context` is replayed thread history published
276
+ // at run start with timestamps of its own, so it never moves the clock; span
277
+ // records are timing, not steps, and never move it either.
278
+ for (const ev of events) {
279
+ if (isSpanRecord(ev) || ev.type === "context") continue;
280
+ if (ev.at !== undefined) {
281
+ firstAt ??= ev.at;
282
+ lastAt = ev.at;
283
+ }
284
+ }
285
+ const windowEnd = window?.end ?? lastAt;
286
+
287
+ // ---- the span set: every duration comes from here -------------------------
288
+ const { spans, anchorOf, producedOf } = spanSet(events);
289
+ const losses: LossInterval[] = window ? lossesFromStream(events, { windowStart: window.start }) : [];
290
+ const lossStarts = losses.map((l) => l.from).sort((a, b) => a - b);
291
+ /** A span's end: its own, or — open — the window's end while live, the next
292
+ * loss after its start once finished (nothing can still be running; the end
293
+ * was lost), else the last content stamp. */
294
+ const endOf = (s: SpanRecord): number | undefined => {
295
+ if (s.endedAt !== undefined) return s.endedAt;
296
+ if (!finished) return windowEnd;
297
+ const next = lossStarts.find((at) => at > s.startedAt);
298
+ return next ?? windowEnd;
299
+ };
300
+ const durationOfSpan = (s: SpanRecord): number | undefined => {
301
+ const end = endOf(s);
302
+ return end === undefined ? undefined : Math.max(0, end - s.startedAt);
303
+ };
304
+ const toolSpans = spans.filter((s) => s.name.startsWith("tool."));
305
+ const turnSpans = spans.filter((s) => s.name === "model.turn");
306
+ const toolSpanByCallId = new Map<string, SpanRecord>();
307
+ const toolSpanById = new Map<string, SpanRecord>();
308
+ for (const s of toolSpans) {
309
+ toolSpanById.set(s.spanId, s);
310
+ const callId = s.attrs.callId;
311
+ if (typeof callId === "string") toolSpanByCallId.set(callId, s);
312
+ }
313
+ /** The span a call/result pair is about: by the stamped `spanId`, by
314
+ * `callId`, or by the synthesized id the Adapter minted from the call's index. */
315
+ const spanOfCall = (
316
+ call: PendingCall | undefined,
317
+ result: Extract<RunEvent, { type: "tool_result" }>,
318
+ ): SpanRecord | undefined => {
319
+ const stamped = result.spanId ?? call?.event.spanId;
320
+ if (stamped !== undefined && toolSpanById.has(stamped)) return toolSpanById.get(stamped);
321
+ const callId = result.callId ?? call?.event.callId;
322
+ if (callId !== undefined && toolSpanByCallId.has(callId)) return toolSpanByCallId.get(callId);
323
+ if (callId !== undefined && toolSpanById.has(`synth:${callId}`)) return toolSpanById.get(`synth:${callId}`);
324
+ return call ? toolSpanById.get(`synth:${call.index}`) : undefined;
325
+ };
326
+
327
+ // Slow turns, keyed by the event they produced, so they land in stream order
328
+ // before that event's own findings (a turn is over before its tool starts).
329
+ const turnFindingsAt = new Map<number, FrictionFinding[]>();
330
+ let modelTimeMs: number | undefined;
331
+ for (const s of turnSpans) {
332
+ const durationMs = durationOfSpan(s);
333
+ if (durationMs === undefined) continue;
334
+ modelTimeMs = (modelTimeMs ?? 0) + durationMs;
335
+ if (durationMs < slowModelTurnMs) continue;
336
+ const anchor = anchorOf.get(s.spanId) ?? events.length - 1;
337
+ const produced = events[producedOf.get(s.spanId) ?? anchor];
338
+ const what =
339
+ produced?.type === "tool_call"
340
+ ? typeof produced.summary === "string"
341
+ ? produced.summary
342
+ : produced.tool
343
+ : produced?.type === "assistant"
344
+ ? "(narration)"
345
+ : produced?.type === "answer"
346
+ ? "(answer)"
347
+ : typeof s.attrs.stopReason === "string"
348
+ ? `(${s.attrs.stopReason})`
349
+ : "(a model turn)";
350
+ const finding: FrictionFinding = {
351
+ category: "slow_model_turn",
352
+ severity: durationMs >= 2 * slowModelTurnMs ? "high" : "medium",
353
+ summary: `model turn took ${formatDuration(durationMs, "report")} before: ${what}`,
354
+ durationMs,
355
+ interval: { start: s.startedAt, end: s.startedAt + durationMs },
356
+ eventIndex: anchor,
357
+ };
358
+ turnFindingsAt.set(anchor, [...(turnFindingsAt.get(anchor) ?? []), finding]);
359
+ }
360
+ const flushTurns = (index: number) => {
361
+ const list = turnFindingsAt.get(index);
362
+ if (list) {
363
+ findings.push(...list);
364
+ turnFindingsAt.delete(index);
365
+ }
366
+ };
367
+
368
+ // The narrative events — the request (`input`), the thread context fed to
369
+ // the model (`context`), the model's prose between tools (`assistant`), the
370
+ // final answer (`answer`) — are the run's story, not its steps: none counts
371
+ // toward `eventCount`.
372
+ let narrativeEvents = 0;
373
+ let sideFactEvents = 0; // skill_use / review_artifact / pr_description / pr_opened / ship_round: facts about the run, not steps
374
+ let spanEvents = 0; // span_start / span_end (docs/reference/specs/tracing.md): timing records, not steps
375
+ let wrapUp: { index: number; at?: number } | undefined;
376
+ events.forEach((ev, index) => {
377
+ flushTurns(index);
378
+ if (isSpanRecord(ev)) {
379
+ spanEvents++;
380
+ return;
381
+ }
382
+ if (isNarrative(ev)) {
383
+ narrativeEvents++; // the narrative and `run_meta` are neither steps nor findings
384
+ return;
385
+ }
386
+ // Side facts about the run, not steps: skill_use rides beside a use_skill
387
+ // call that already produced its own tool pair; review_artifact,
388
+ // pr_description, pr_opened and the ship_round boundaries are published
389
+ // by the dispatcher/pipeline outside the model loop entirely. Counting
390
+ // any of them would distort the story.
391
+ if (
392
+ ev.type === "skill_use" ||
393
+ ev.type === "review_artifact" ||
394
+ ev.type === "pr_description" ||
395
+ ev.type === "pr_opened" ||
396
+ ev.type === "ship_round"
397
+ ) {
398
+ sideFactEvents++;
399
+ return;
400
+ }
401
+
402
+ if (ev.type === "tool_call") {
403
+ toolCalls++;
404
+ if (failedCalls.has(`${ev.tool} ${ev.summary}`)) {
405
+ findings.push({
406
+ category: "retry",
407
+ severity: "low",
408
+ summary: `retried after failure: ${ev.summary}`,
409
+ tool: ev.tool,
410
+ eventIndex: index,
411
+ });
412
+ }
413
+ const queue = pending.get(ev.tool) ?? [];
414
+ queue.push({ index, event: ev });
415
+ pending.set(ev.tool, queue);
416
+ return;
417
+ }
418
+
419
+ if (ev.type === "tool_result") {
420
+ const callEntry = pending.get(ev.tool)?.shift();
421
+ const anchor = callEntry?.index ?? index;
422
+ const callSummary = callEntry?.event.summary ?? ev.tool;
423
+ const span = spanOfCall(callEntry, ev);
424
+ const durationMs = span ? durationOfSpan(span) : undefined;
425
+ const timed = (f: FrictionFinding): FrictionFinding => (durationMs !== undefined ? { ...f, durationMs } : f);
426
+ const interval =
427
+ span && durationMs !== undefined
428
+ ? { interval: { start: span.startedAt, end: span.startedAt + durationMs } }
429
+ : {};
430
+
431
+ if (!ev.ok) failedCalls.add(`${ev.tool} ${callSummary}`);
432
+
433
+ if (ev.infra) {
434
+ // An infra-level failure is the sandbox, not the command: classify once,
435
+ // as infra — but name the command that was running, so "the sandbox died
436
+ // during installs" is readable from the findings alone.
437
+ findings.push(
438
+ timed({
439
+ category: "infra_failure",
440
+ severity: "high",
441
+ summary: `exec infrastructure failed during ${callSummary} → ${ev.summary}`,
442
+ tool: ev.tool,
443
+ ...interval,
444
+ eventIndex: anchor,
445
+ }),
446
+ );
447
+ return;
448
+ }
449
+ if (ev.tool === "bash" && isSetupInstallCommand(callSummary)) {
450
+ // Setup/install is reported once, as setup — a slow or failed install is
451
+ // still setup cost — so the category total is the true install bill.
452
+ const slow = durationMs !== undefined && durationMs >= slowToolMs;
453
+ const label = !ev.ok ? "install failed" : slow ? "slow install" : "install";
454
+ findings.push(
455
+ timed({
456
+ category: "setup_install",
457
+ severity: !ev.ok ? "high" : slow ? "medium" : "low",
458
+ summary: `${label}: ${callSummary}${!ev.ok ? ` → ${ev.summary}` : ""}`,
459
+ tool: ev.tool,
460
+ eventIndex: anchor,
461
+ }),
462
+ );
463
+ return;
464
+ }
465
+ if (!ev.ok) {
466
+ // A failure that was ALSO slow cost more than a fast one: high once it
467
+ // crosses the slow threshold (the same bar slow_tool uses).
468
+ const slowFailure = durationMs !== undefined && durationMs >= slowToolMs;
469
+ findings.push(
470
+ timed({
471
+ category: "failed_tool",
472
+ severity: slowFailure ? "high" : "medium",
473
+ summary: `${callSummary} → ${ev.summary}`,
474
+ tool: ev.tool,
475
+ eventIndex: anchor,
476
+ }),
477
+ );
478
+ return;
479
+ }
480
+ if (durationMs !== undefined && durationMs >= slowToolMs) {
481
+ findings.push(
482
+ timed({
483
+ category: "slow_tool",
484
+ severity: durationMs >= 2 * slowToolMs ? "high" : "medium",
485
+ summary: `took ${formatDuration(durationMs, "report")}: ${callSummary}`,
486
+ tool: ev.tool,
487
+ eventIndex: anchor,
488
+ }),
489
+ );
490
+ }
491
+ return;
492
+ }
493
+
494
+ switch (ev.kind) {
495
+ case "wrap_up":
496
+ // Its extent is the time from the warning to the end of the window (how
497
+ // long the wind-down actually took); filled in after the loop.
498
+ wrapUp = { index, at: ev.at };
499
+ findings.push({ category: "wrap_up", severity: "medium", summary: ev.summary, eventIndex: index });
500
+ return;
501
+ case "time_budget_exhausted":
502
+ findings.push({
503
+ category: "budget_hit",
504
+ severity: "high",
505
+ summary: `budget hit (time): ${ev.summary}`,
506
+ eventIndex: index,
507
+ });
508
+ return;
509
+ case "turn_budget_exhausted":
510
+ findings.push({
511
+ category: "budget_hit",
512
+ severity: "high",
513
+ summary: `budget hit (turns): ${ev.summary}`,
514
+ eventIndex: index,
515
+ });
516
+ return;
517
+ case "sandbox_dead":
518
+ findings.push({
519
+ category: "infra_failure",
520
+ severity: "high",
521
+ summary: `sandbox dead: ${ev.summary}`,
522
+ eventIndex: index,
523
+ });
524
+ return;
525
+ case "fleet_busy":
526
+ // Capacity, not a dead sandbox: the run went on, but the minutes spent
527
+ // waiting for an instance are friction the fleet's sizing owns.
528
+ findings.push({
529
+ category: "infra_failure",
530
+ severity: "medium",
531
+ summary: `fleet busy: ${ev.summary}`,
532
+ eventIndex: index,
533
+ });
534
+ return;
535
+ }
536
+ });
537
+ // A slow turn whose produced event is past the stream (a turn that ended the run).
538
+ for (const index of [...turnFindingsAt.keys()].sort((a, b) => a - b)) flushTurns(index);
539
+
540
+ // In a FINISHED stream, a call with no result means the run ended mid-tool
541
+ // (process died, stream cut) — infrastructure friction that must not vanish
542
+ // silently. Mid-run it is just the tool still executing. Its extent is the
543
+ // open tool span's: to the next loss, else to the window's end.
544
+ for (const queue of finished ? pending.values() : []) {
545
+ for (const { index, event } of queue) {
546
+ const open =
547
+ (event.spanId !== undefined ? toolSpanById.get(event.spanId) : undefined) ??
548
+ (event.callId !== undefined ? toolSpanByCallId.get(event.callId) : undefined) ??
549
+ toolSpanById.get(`synth:${index}`);
550
+ const durationMs = open ? durationOfSpan(open) : undefined;
551
+ findings.push({
552
+ category: "infra_failure",
553
+ severity: "high",
554
+ summary: `no result for tool call (run ended mid-tool): ${event.summary}`,
555
+ tool: event.tool,
556
+ eventIndex: index,
557
+ ...(durationMs !== undefined ? { durationMs } : {}),
558
+ ...(open && durationMs !== undefined
559
+ ? { interval: { start: open.startedAt, end: open.startedAt + durationMs } }
560
+ : {}),
561
+ });
562
+ }
563
+ }
564
+ if (wrapUp !== undefined) {
565
+ const { index, at } = wrapUp;
566
+ const f = findings.find((x) => x.category === "wrap_up" && x.eventIndex === index);
567
+ if (at !== undefined && windowEnd !== undefined && f) {
568
+ f.durationMs = Math.max(0, windowEnd - at);
569
+ f.interval = { start: at, end: Math.max(at, windowEnd) };
570
+ }
571
+ }
572
+
573
+ // ---- totals ----------------------------------------------------------------
574
+ let toolTimeMs: number | undefined;
575
+ for (const s of toolSpans) {
576
+ const d = durationOfSpan(s);
577
+ if (d !== undefined) toolTimeMs = (toolTimeMs ?? 0) + d;
578
+ }
579
+ const byCategory = Object.fromEntries(
580
+ FRICTION_CATEGORIES.map((c) => [c, { count: 0, durationMs: 0 } satisfies CategoryTotals]),
581
+ ) as Record<FrictionCategory, CategoryTotals>;
582
+ for (const f of findings) byCategory[f.category].count++;
583
+ for (const c of FRICTION_CATEGORIES) {
584
+ const own = findings.filter((f) => f.category === c);
585
+ // Turns are disjoint (one loop runs at a time), so their sum is their union.
586
+ byCategory[c].durationMs =
587
+ DENOMINATOR_OF[c] === "run" && c !== "slow_model_turn"
588
+ ? unionMs(own.flatMap((f) => (f.interval ? [f.interval] : [])))
589
+ : own.reduce((sum, f) => sum + (f.durationMs ?? 0), 0);
590
+ }
591
+ const runMs =
592
+ window !== undefined
593
+ ? Math.max(0, window.end - window.start)
594
+ : firstAt !== undefined && lastAt !== undefined
595
+ ? lastAt - firstAt
596
+ : undefined;
597
+ let shape: RunShape | undefined;
598
+ if (window !== undefined && finished) {
599
+ const { backgroundOnlyMs: _background, ...terms } = partition(spans, { window, owner, finished, losses });
600
+ shape = terms;
601
+ }
602
+
603
+ const diagnosis: FrictionDiagnosis = {
604
+ eventCount: events.length - narrativeEvents - sideFactEvents - spanEvents,
605
+ toolCalls,
606
+ ...(runMs !== undefined ? { runMs, toolTimeMs: toolTimeMs ?? 0 } : {}),
607
+ ...(runMs !== undefined && modelTimeMs !== undefined ? { modelTimeMs } : {}),
608
+ byCategory,
609
+ findings,
610
+ verdict: "",
611
+ ...(opts.truncated ? { truncatedInput: true as const } : {}),
612
+ ...(shape ? { shape } : {}),
613
+ };
614
+ diagnosis.verdict = verdictOf(diagnosis);
615
+ return diagnosis;
616
+ }
617
+
618
+ /** The dominant cause: most attributed time when timed (ties → most findings →
619
+ * category order), else most findings. Names the share of its denominator
620
+ * (`DENOMINATOR_OF`) so the reader knows whether the cause is the whole story. */
621
+ function verdictOf(d: FrictionDiagnosis): string {
622
+ if (d.findings.length === 0) return "no friction detected";
623
+ const ranked = FRICTION_CATEGORIES.filter((c) => d.byCategory[c].count > 0).sort(
624
+ (a, b) => d.byCategory[b].durationMs - d.byCategory[a].durationMs || d.byCategory[b].count - d.byCategory[a].count,
625
+ );
626
+ const top = ranked[0];
627
+ const t = d.byCategory[top];
628
+ const what = `${CATEGORY_LABEL[top]} dominated: ${t.count} finding${t.count === 1 ? "" : "s"}`;
629
+ if (t.durationMs === 0) return what;
630
+ const [denominator, of] = DENOMINATOR_OF[top] === "run" ? [d.runMs, "run time"] : [d.toolTimeMs, "tool time"];
631
+ const share = denominator ? ` (${Math.round((t.durationMs / denominator) * 100)}% of ${of})` : "";
632
+ return `${what}, ${formatDuration(t.durationMs, "report")}${share}`;
633
+ }
634
+
635
+ /** Plain-text report for the CLI / terminal. */
636
+ export function formatFrictionReport(d: FrictionDiagnosis): string {
637
+ const lines: string[] = [`verdict: ${d.verdict}`];
638
+ const totals = [`events: ${d.eventCount}`];
639
+ if (d.toolCalls > 0) totals.push(`tool calls: ${d.toolCalls}`);
640
+ if (d.runMs !== undefined) totals.push(`run: ${formatDuration(d.runMs, "report")}`);
641
+ if (d.toolCalls > 0 && d.toolTimeMs !== undefined)
642
+ totals.push(`tool time: ${formatDuration(d.toolTimeMs, "report")}`);
643
+ if (d.modelTimeMs !== undefined) totals.push(`model time: ${formatDuration(d.modelTimeMs, "report")}`);
644
+ if (d.truncatedInput) totals.push("(input truncated — some records were dropped before analysis)");
645
+ lines.push(totals.join(" · "));
646
+ if (d.shape) {
647
+ const shape = formatShape(d.shape);
648
+ lines.push(`shape: ${shape ?? `${formatDuration(d.shape.windowMs, "report")} (one bucket)`}`);
649
+ }
650
+ const width = Math.max(...FRICTION_CATEGORIES.map((c) => CATEGORY_LABEL[c].length), "category".length);
651
+ lines.push("", `${"category".padEnd(width)} count time`);
652
+ for (const c of FRICTION_CATEGORIES) {
653
+ const t = d.byCategory[c];
654
+ const time = t.count && t.durationMs > 0 ? formatDuration(t.durationMs, "report") : "-";
655
+ lines.push(`${CATEGORY_LABEL[c].padEnd(width)} ${String(t.count).padStart(5)} ${time}`);
656
+ }
657
+ if (d.findings.length > 0) {
658
+ lines.push("", "findings (stream order):");
659
+ for (const f of d.findings) {
660
+ const dur = f.durationMs !== undefined ? ` (${formatDuration(f.durationMs, "report")})` : "";
661
+ lines.push(` #${f.eventIndex} [${CATEGORY_LABEL[f.category]}] ${f.severity}${dur} ${f.summary}`);
662
+ }
663
+ }
664
+ return lines.join("\n");
665
+ }