@coreplane/switchboard 0.0.0 → 1.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +18 -1
- package/dist/assets/.dockerignore +27 -0
- package/dist/assets/.env.example +33 -0
- package/dist/assets/Dockerfile +111 -0
- package/dist/assets/config/config.example.yaml +359 -0
- package/dist/assets/deploy/bin/build-stamp.d.mts +15 -0
- package/dist/assets/deploy/bin/build-stamp.mjs +98 -0
- package/dist/assets/deploy/bin/cf-logs +32 -0
- package/dist/assets/deploy/cloudflare/package.json +29 -0
- package/dist/assets/deploy/cloudflare/preflight.mjs +243 -0
- package/dist/assets/deploy/cloudflare/tsconfig.json +18 -0
- package/dist/assets/deploy/cloudflare/worker.ts +382 -0
- package/dist/assets/deploy/cloudflare/wrangler.template.jsonc +67 -0
- package/dist/assets/deploy/cloudflare/write-build.d.mts +7 -0
- package/dist/assets/deploy/cloudflare/write-build.mjs +53 -0
- package/dist/assets/deploy/cloudflare-docs/package.json +18 -0
- package/dist/assets/deploy/cloudflare-docs/wrangler.template.jsonc +30 -0
- package/dist/assets/deploy/cloudflare-memory/package.json +25 -0
- package/dist/assets/deploy/cloudflare-memory/tsconfig.json +17 -0
- package/dist/assets/deploy/cloudflare-memory/worker.ts +2635 -0
- package/dist/assets/deploy/cloudflare-memory/wrangler.template.jsonc +50 -0
- package/dist/assets/deploy/cloudflare-resident/Dockerfile +91 -0
- package/dist/assets/deploy/cloudflare-resident/gc.ts +287 -0
- package/dist/assets/deploy/cloudflare-resident/node-async-hooks.d.ts +11 -0
- package/dist/assets/deploy/cloudflare-resident/package.json +29 -0
- package/dist/assets/deploy/cloudflare-resident/preflight.mjs +224 -0
- package/dist/assets/deploy/cloudflare-resident/tsconfig.json +19 -0
- package/dist/assets/deploy/cloudflare-resident/worker.ts +6637 -0
- package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +120 -0
- package/dist/assets/deploy/cloudflare-sandbox/Dockerfile +67 -0
- package/dist/assets/deploy/cloudflare-sandbox/docker-wrapper.sh +37 -0
- package/dist/assets/deploy/cloudflare-sandbox/package.json +26 -0
- package/dist/assets/deploy/cloudflare-sandbox/tsconfig.json +20 -0
- package/dist/assets/deploy/cloudflare-sandbox/worker.ts +410 -0
- package/dist/assets/deploy/cloudflare-sandbox/wrangler.template.jsonc +67 -0
- package/dist/assets/deploy/profile.example.json +13 -0
- package/dist/assets/deploy/secrets.manifest.json +108 -0
- package/dist/assets/docker-entrypoint.sh +15 -0
- package/dist/assets/package-lock.json +18407 -0
- package/dist/assets/package.json +104 -0
- package/dist/assets/project.json +219 -0
- package/dist/assets/source.json +5 -0
- package/dist/assets/src/core/authz/actor.ts +100 -0
- package/dist/assets/src/core/authz/authorize.ts +169 -0
- package/dist/assets/src/core/authz/grants.ts +347 -0
- package/dist/assets/src/core/authz/policy.ts +281 -0
- package/dist/assets/src/core/authz/resource.ts +147 -0
- package/dist/assets/src/core/authz/types.ts +164 -0
- package/dist/assets/src/core/drain.ts +54 -0
- package/dist/assets/src/core/ingressTokens.ts +64 -0
- package/dist/assets/src/core/memory/engine.ts +115 -0
- package/dist/assets/src/core/memory/scorer.ts +147 -0
- package/dist/assets/src/core/memory/types.ts +120 -0
- package/dist/assets/src/core/normalizeSpans.ts +299 -0
- package/dist/assets/src/core/prDescriptionTypes.ts +54 -0
- package/dist/assets/src/core/redact.ts +113 -0
- package/dist/assets/src/core/runEvents.ts +537 -0
- package/dist/assets/src/core/runFriction.ts +665 -0
- package/dist/assets/src/core/runLedger/decisions.ts +126 -0
- package/dist/assets/src/core/runLedger/types.ts +177 -0
- package/dist/assets/src/core/runRecord.ts +627 -0
- package/dist/assets/src/core/runShape.ts +61 -0
- package/dist/assets/src/core/schedules.ts +452 -0
- package/dist/assets/src/core/time/formatDuration.ts +61 -0
- package/dist/assets/src/core/trace/attrs.ts +203 -0
- package/dist/assets/src/core/trace/classify.ts +49 -0
- package/dist/assets/src/core/trace/clock.ts +6 -0
- package/dist/assets/src/core/trace/context.ts +9 -0
- package/dist/assets/src/core/trace/ids.ts +23 -0
- package/dist/assets/src/core/trace/partition.ts +235 -0
- package/dist/assets/src/core/trace/sinks.ts +68 -0
- package/dist/assets/src/core/trace/streamSpans.ts +163 -0
- package/dist/assets/src/core/trace/traceparent.ts +29 -0
- package/dist/assets/src/core/trace/tracer.ts +247 -0
- package/dist/assets/src/core/trace/types.ts +125 -0
- package/dist/assets/src/core/trace/workerTrace.ts +97 -0
- package/dist/assets/src/deploy/buildStamp.ts +93 -0
- package/dist/assets/src/deploy/liveGate.ts +203 -0
- package/dist/assets/src/deploy/profile.ts +162 -0
- package/dist/assets/src/deploy/restart.ts +393 -0
- package/dist/assets/src/effort.ts +17 -0
- package/dist/assets/src/execution/bashTimeout.ts +78 -0
- package/dist/assets/src/execution/bindingPurge.ts +43 -0
- package/dist/assets/src/execution/residentBackupTransfer.ts +50 -0
- package/dist/assets/src/execution/residentCleanliness.ts +95 -0
- package/dist/assets/src/execution/residentCredentials.ts +81 -0
- package/dist/assets/src/execution/residentDepCache.ts +321 -0
- package/dist/assets/src/execution/residentDepsStore.ts +326 -0
- package/dist/assets/src/execution/residentDetach.ts +48 -0
- package/dist/assets/src/execution/residentDisk.ts +107 -0
- package/dist/assets/src/execution/residentDiskBudget.ts +448 -0
- package/dist/assets/src/execution/residentExecWrap.ts +100 -0
- package/dist/assets/src/execution/residentHead.ts +85 -0
- package/dist/assets/src/execution/residentReadonly.ts +72 -0
- package/dist/assets/src/execution/residentRefresh.ts +429 -0
- package/dist/assets/src/execution/residentRestoreExtract.ts +130 -0
- package/dist/assets/src/execution/residentState.ts +47 -0
- package/dist/assets/src/execution/residentStepReport.ts +98 -0
- package/dist/assets/src/execution/residentStepTrace.ts +97 -0
- package/dist/assets/src/execution/residentSteps.ts +99 -0
- package/dist/assets/src/execution/residentText.ts +83 -0
- package/dist/assets/src/execution/residentTrace.ts +119 -0
- package/dist/assets/src/execution/sandboxEnv.ts +42 -0
- package/dist/assets/src/execution/sandboxErrors.ts +159 -0
- package/dist/assets/src/execution/sandboxKeepalive.ts +118 -0
- package/dist/assets/src/execution/shellQuote.ts +8 -0
- package/dist/assets/src/mcp/registry.ts +242 -0
- package/dist/assets/src/providers/types.ts +152 -0
- package/dist/assets/web/dist/.vite/manifest.json +176 -0
- package/dist/assets/web/dist/assets/AppShell-Bk2gbvet.js +1 -0
- package/dist/assets/web/dist/assets/CostsPage-CTZcMYYx.js +1 -0
- package/dist/assets/web/dist/assets/NotFoundPage-C-BuaSm8.js +1 -0
- package/dist/assets/web/dist/assets/ResidentDetailPage-D3shEnzl.js +1 -0
- package/dist/assets/web/dist/assets/ResidentsIndexPage-DWIubQ05.js +1 -0
- package/dist/assets/web/dist/assets/RunRoutePage-BMjuE-oX.js +126 -0
- package/dist/assets/web/dist/assets/RunRoutePage-XVFj0XDc.css +1 -0
- package/dist/assets/web/dist/assets/RunsIndexPage-C3_jYIo0.js +1 -0
- package/dist/assets/web/dist/assets/RunsTabs-C4krAL9o.js +1 -0
- package/dist/assets/web/dist/assets/ScheduledPage-g1W58mtN.js +1 -0
- package/dist/assets/web/dist/assets/StatusDot-DcPRw3zu.js +1 -0
- package/dist/assets/web/dist/assets/Tooltip-DJUkMYjo.js +1 -0
- package/dist/assets/web/dist/assets/favicon-DL1rdWJt.js +1 -0
- package/dist/assets/web/dist/assets/localIso-L06jV29p.js +1 -0
- package/dist/assets/web/dist/assets/main-BsBGUyMH.css +2 -0
- package/dist/assets/web/dist/assets/main-CyM5f4JC.js +28 -0
- package/dist/assets/web/dist/assets/residentDiskBudget-BMBKlYRH.js +1 -0
- package/dist/assets/web/dist/assets/seed-BglCRKLA.js +6 -0
- package/dist/assets/web/dist/assets/wallClock-Ckv3sKoR.js +1 -0
- package/dist/cli.js +34494 -0
- package/package.json +43 -10
|
@@ -0,0 +1,665 @@
|
|
|
1
|
+
import type { RunEvent } from "./runEvents.js";
|
|
2
|
+
import { isSpanRecord } from "./runEvents.js";
|
|
3
|
+
import { lossesFromStream, normalizeSpans, spansFromEvents } from "./normalizeSpans.js";
|
|
4
|
+
import { formatShape } from "./runShape.js";
|
|
5
|
+
import { formatDuration } from "./time/formatDuration.js";
|
|
6
|
+
import { partition, type LossInterval, type Partition, type Window } from "./trace/partition.js";
|
|
7
|
+
import type { RunOwner } from "./trace/streamSpans.js";
|
|
8
|
+
import type { SpanRecord } from "./trace/types.js";
|
|
9
|
+
|
|
10
|
+
// Run-friction analyzer (docs/reference/specs/run-friction.md): a PURE, deterministic
|
|
11
|
+
// function from a run's RunEvent stream to a structured diagnosis of what cost
|
|
12
|
+
// the run time or made it stumble — slow/failed tool calls, slow model turns,
|
|
13
|
+
// retries, setup/install time, wrap-up, budget hits, exec-infrastructure
|
|
14
|
+
// failures — and, for a finished run with a window, its shape: how the window
|
|
15
|
+
// splits into getting ready, thinking, tools, finishing up and overhead
|
|
16
|
+
// (docs/reference/specs/tracing.md item 5). It is the observe→diagnose half of the
|
|
17
|
+
// self-improvement loop; proposing fix PRs from a diagnosis is a later piece
|
|
18
|
+
// and deliberately NOT here. No clock, no I/O: the same events always yield the
|
|
19
|
+
// same diagnosis, so it runs identically over a live backlog
|
|
20
|
+
// (`/runs/:id/friction`), a saved JSONL stream (`friction analyze`), or a test
|
|
21
|
+
// fixture.
|
|
22
|
+
//
|
|
23
|
+
// Every duration comes from ONE span set: the stream normalized through
|
|
24
|
+
// `normalizeSpans` (a tool pair whose twin span the record budget dropped gets
|
|
25
|
+
// it back) and paired by `spansFromEvents`. Model time is the sum of the
|
|
26
|
+
// `model.turn` spans, tool time the sum of the `tool.*` spans, and each timed
|
|
27
|
+
// finding carries the duration of the span it is about — so a tool-denominated
|
|
28
|
+
// finding is a summand of tool time and a slow turn of model time, and no
|
|
29
|
+
// share can exceed 100 %. A stream with no spans (a hand-written capture, a
|
|
30
|
+
// record from before spans) has no durations: the same classification, every
|
|
31
|
+
// timed field absent — one code path, nothing special-cased.
|
|
32
|
+
|
|
33
|
+
export type FrictionCategory =
|
|
34
|
+
| "slow_tool"
|
|
35
|
+
| "slow_model_turn"
|
|
36
|
+
| "failed_tool"
|
|
37
|
+
| "retry"
|
|
38
|
+
| "setup_install"
|
|
39
|
+
| "wrap_up"
|
|
40
|
+
| "budget_hit"
|
|
41
|
+
| "infra_failure";
|
|
42
|
+
|
|
43
|
+
export const FRICTION_CATEGORIES: readonly FrictionCategory[] = [
|
|
44
|
+
"slow_tool",
|
|
45
|
+
"slow_model_turn",
|
|
46
|
+
"failed_tool",
|
|
47
|
+
"retry",
|
|
48
|
+
"setup_install",
|
|
49
|
+
"wrap_up",
|
|
50
|
+
"budget_hit",
|
|
51
|
+
"infra_failure",
|
|
52
|
+
];
|
|
53
|
+
|
|
54
|
+
/** Human labels — the verdict line, the report's table and its finding lines. */
|
|
55
|
+
export const CATEGORY_LABEL: Record<FrictionCategory, string> = {
|
|
56
|
+
slow_tool: "slow tool calls",
|
|
57
|
+
slow_model_turn: "slow model turns",
|
|
58
|
+
failed_tool: "failed tool calls",
|
|
59
|
+
retry: "retries",
|
|
60
|
+
setup_install: "the repo's setup/install",
|
|
61
|
+
wrap_up: "agent wind-down",
|
|
62
|
+
budget_hit: "budget hits",
|
|
63
|
+
infra_failure: "infra failures",
|
|
64
|
+
};
|
|
65
|
+
|
|
66
|
+
/** What a category's time is a share OF: tool time for the categories whose
|
|
67
|
+
* findings are tool calls (each a summand of `toolTimeMs`), run time for the
|
|
68
|
+
* rest. A new category must choose, so no share can exceed 100 %. */
|
|
69
|
+
export const DENOMINATOR_OF: Record<FrictionCategory, "tool" | "run"> = {
|
|
70
|
+
slow_tool: "tool",
|
|
71
|
+
failed_tool: "tool",
|
|
72
|
+
retry: "tool",
|
|
73
|
+
setup_install: "tool",
|
|
74
|
+
slow_model_turn: "run",
|
|
75
|
+
wrap_up: "run",
|
|
76
|
+
budget_hit: "run",
|
|
77
|
+
infra_failure: "run",
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
export type FrictionSeverity = "low" | "medium" | "high";
|
|
81
|
+
|
|
82
|
+
export interface FrictionFinding {
|
|
83
|
+
category: FrictionCategory;
|
|
84
|
+
severity: FrictionSeverity;
|
|
85
|
+
/** One line: what happened. Derived from event summaries, which are already redacted upstream. */
|
|
86
|
+
summary: string;
|
|
87
|
+
/** Tool involved, for tool-anchored findings. */
|
|
88
|
+
tool?: string;
|
|
89
|
+
/** Wall time attributed to this finding, when the events carried timestamps:
|
|
90
|
+
* the duration of the span it is about. */
|
|
91
|
+
durationMs?: number;
|
|
92
|
+
/** The finding's own interval, for the run-denominated categories whose
|
|
93
|
+
* category time is the UNION of their findings (three calls of one batch
|
|
94
|
+
* dying together are one interval, not three). Absent on a note-anchored
|
|
95
|
+
* finding with no extent (a budget hit, a dead sandbox). */
|
|
96
|
+
interval?: { start: number; end: number };
|
|
97
|
+
/** Index into the input stream of the event the finding anchors to: the
|
|
98
|
+
* tool_call for tool findings, the note for note findings, the event a model
|
|
99
|
+
* turn produced (its `span_end`, or the synthesized twin's terminator) for a
|
|
100
|
+
* slow turn. */
|
|
101
|
+
eventIndex: number;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export interface CategoryTotals {
|
|
105
|
+
count: number;
|
|
106
|
+
/** The category's attributed time: the sum of its findings' `durationMs` for
|
|
107
|
+
* the tool-denominated categories and `slow_model_turn` (whose turns are
|
|
108
|
+
* disjoint), the union of its findings' intervals for the other
|
|
109
|
+
* run-denominated ones; 0 without timings. */
|
|
110
|
+
durationMs: number;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** The run's shape (docs/reference/specs/tracing.md item 5): the seven terms of a finished
|
|
114
|
+
* window. Absent while a run is live or when no window was given. */
|
|
115
|
+
export type RunShape = Omit<Partition, "backgroundOnlyMs">;
|
|
116
|
+
|
|
117
|
+
export interface FrictionDiagnosis {
|
|
118
|
+
eventCount: number;
|
|
119
|
+
toolCalls: number;
|
|
120
|
+
/** The window (`receivedAt` → `finishedAt`, or now) when one was given; else
|
|
121
|
+
* first→last content stamp; absent when no content event carried one. */
|
|
122
|
+
runMs?: number;
|
|
123
|
+
/** Sum of the `tool.*` span durations, when the stream is timed (`runMs`). */
|
|
124
|
+
toolTimeMs?: number;
|
|
125
|
+
/** Sum of the `model.turn` span durations, when the stream had a turn and
|
|
126
|
+
* timings. Thinking and tool time are the counted terms of the shape; the
|
|
127
|
+
* rest of the window is setup, finishing up and Switchboard overhead. */
|
|
128
|
+
modelTimeMs?: number;
|
|
129
|
+
/** Every category present, zeroed when absent. */
|
|
130
|
+
byCategory: Record<FrictionCategory, CategoryTotals>;
|
|
131
|
+
/** In detection order (stream order). */
|
|
132
|
+
findings: FrictionFinding[];
|
|
133
|
+
/** One-line headline: the dominant cause, or "no friction detected". */
|
|
134
|
+
verdict: string;
|
|
135
|
+
/** Present (true) when the analyzed stream lost records — the registry's
|
|
136
|
+
* bounded backlog or the record budget dropped some before the diagnosis
|
|
137
|
+
* ran — so the counts and timings describe part of the run, not all of it.
|
|
138
|
+
* Optional: a stored diagnosis from before this field reads as complete. */
|
|
139
|
+
truncatedInput?: true;
|
|
140
|
+
/** The finished window's shape, when a window was given and the run finished. */
|
|
141
|
+
shape?: RunShape;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
export interface FrictionOptions {
|
|
145
|
+
/** A paired tool call taking at least this long is a `slow_tool`. Default 30s. */
|
|
146
|
+
slowToolMs?: number;
|
|
147
|
+
/** A model turn taking at least this long is a `slow_model_turn`. Default 60s. */
|
|
148
|
+
slowModelTurnMs?: number;
|
|
149
|
+
/** Whether the stream is complete (default true). For an in-flight run a
|
|
150
|
+
* trailing tool_call without a result is simply still running — only in a
|
|
151
|
+
* finished stream is it evidence the run died mid-tool. */
|
|
152
|
+
finished?: boolean;
|
|
153
|
+
/** Whether `events` is a truncated stream (`RunSnapshot.truncated`): the
|
|
154
|
+
* diagnosis is then stamped `truncatedInput: true`. Default false. */
|
|
155
|
+
truncated?: boolean;
|
|
156
|
+
/** The run's window (docs/reference/specs/tracing.md): `receivedAt` to `finishedAt` (a
|
|
157
|
+
* record), or to now (a live read). With it `runMs` is the window and a
|
|
158
|
+
* finished diagnosis carries `shape`; open spans run to its end while live.
|
|
159
|
+
* Absent (a stdin capture): `runMs` is first→last over the content events
|
|
160
|
+
* and there is no shape. */
|
|
161
|
+
window?: Window;
|
|
162
|
+
/** Who owns the run: `run.command` counts as tools for a command run and as
|
|
163
|
+
* getting ready for an agent run. Default `agent`. */
|
|
164
|
+
owner?: RunOwner;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
const DEFAULT_SLOW_TOOL_MS = 30_000;
|
|
168
|
+
const DEFAULT_SLOW_MODEL_TURN_MS = 60_000;
|
|
169
|
+
|
|
170
|
+
// Setup/install commands, matched at the START of a shell segment (after `$ `,
|
|
171
|
+
// `&&`, `;`, `|`), so `echo npm install` and `npm test` don't match while
|
|
172
|
+
// `cd repo && npm install` does. `(?=\s|$)` (not `\b`) ends each subcommand
|
|
173
|
+
// word, so `npm ci-lockfile-report` is not `npm ci`. Optional wrappers: `sudo`,
|
|
174
|
+
// `corepack`, `python -m` (for pip).
|
|
175
|
+
const INSTALL_SEGMENT =
|
|
176
|
+
/(?:^|&&|;|\|\|?)\s*(?:sudo\s+)?(?:corepack\s+)?(?:(?:npm|pnpm|bun)\s+(?:install|i|ci|add)(?=\s|$)|yarn(?:\s+(?:install|add)(?=\s|$)|\s*$)|npx\s+playwright\s+install(?=\s|$)|(?:python3?\s+-m\s+)?pip3?\s+install(?=\s|$)|poetry\s+install(?=\s|$)|uv\s+(?:sync|pip\s+install)(?=\s|$)|apt(?:-get)?\s+install(?=\s|$)|apk\s+add(?=\s|$)|brew\s+install(?=\s|$)|bundle\s+install(?=\s|$)|gem\s+install(?=\s|$)|cargo\s+(?:fetch|build)(?=\s|$)|go\s+mod\s+download(?=\s|$)|git\s+clone(?=\s|$)|make\s+(?:deps|install|setup)(?=\s|$))/;
|
|
177
|
+
|
|
178
|
+
// Quoted string literals ("…" / '…'): prose to the shell, not command segments.
|
|
179
|
+
const QUOTED = /"(?:[^"\\]|\\.)*"|'[^']*'/g;
|
|
180
|
+
|
|
181
|
+
/** True for a bash tool_call summary (`$ <command>`) that is a setup/install
|
|
182
|
+
* step. Quoted text is blanked first so `echo "cd x && npm install"` is not
|
|
183
|
+
* an install. Tolerates a non-string (a corrupted external capture) → false. */
|
|
184
|
+
export function isSetupInstallCommand(summary: unknown): boolean {
|
|
185
|
+
if (typeof summary !== "string") return false;
|
|
186
|
+
const cmd = summary.startsWith("$ ") ? summary.slice(2) : summary;
|
|
187
|
+
return INSTALL_SEGMENT.test(cmd.replace(QUOTED, '""'));
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
interface PendingCall {
|
|
191
|
+
index: number;
|
|
192
|
+
event: Extract<RunEvent, { type: "tool_call" }>;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** The event types that tell the run's story rather than its steps — the
|
|
196
|
+
* narrative (`input`/`context`/`assistant`/`answer`) and what the run is about. */
|
|
197
|
+
type NarrativeEvent = Extract<RunEvent, { type: "input" | "context" | "assistant" | "answer" | "run_meta" }>;
|
|
198
|
+
function isNarrative(ev: RunEvent): ev is NarrativeEvent {
|
|
199
|
+
return (
|
|
200
|
+
ev.type === "input" ||
|
|
201
|
+
ev.type === "context" ||
|
|
202
|
+
ev.type === "assistant" ||
|
|
203
|
+
ev.type === "answer" ||
|
|
204
|
+
ev.type === "run_meta"
|
|
205
|
+
);
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/** The span set the analyzer reads, with two indices per span end into the
|
|
209
|
+
* ORIGINAL stream: `anchorOf` — the event the span ended at (its own `span_end`
|
|
210
|
+
* when the stream carried one, else the first original event after the
|
|
211
|
+
* synthesized end) — and `producedOf` — the first content event after it, the
|
|
212
|
+
* event a model turn produced (its tool call, its narration, the answer). */
|
|
213
|
+
function spanSet(events: readonly RunEvent[]) {
|
|
214
|
+
const normalized = normalizeSpans(events);
|
|
215
|
+
const spans = spansFromEvents(normalized, "run");
|
|
216
|
+
const originalIndex = new Map<RunEvent, number>();
|
|
217
|
+
events.forEach((e, i) => originalIndex.set(e, i));
|
|
218
|
+
const anchorOf = new Map<string, number>();
|
|
219
|
+
const producedOf = new Map<string, number>();
|
|
220
|
+
// Walk from the end so each span end learns the originals that follow it.
|
|
221
|
+
let nextOriginal = events.length - 1;
|
|
222
|
+
let nextContent: number | undefined;
|
|
223
|
+
for (let p = normalized.length - 1; p >= 0; p--) {
|
|
224
|
+
const x = normalized[p]!;
|
|
225
|
+
const own = originalIndex.get(x);
|
|
226
|
+
if (x.type === "span_end") {
|
|
227
|
+
anchorOf.set(x.spanId, own ?? nextOriginal);
|
|
228
|
+
if (nextContent !== undefined) producedOf.set(x.spanId, nextContent);
|
|
229
|
+
}
|
|
230
|
+
if (own !== undefined) {
|
|
231
|
+
nextOriginal = own;
|
|
232
|
+
if (!isSpanRecord(x)) nextContent = own;
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
return { spans, anchorOf, producedOf };
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
function unionMs(intervals: ReadonlyArray<{ start: number; end: number }>): number {
|
|
239
|
+
const sorted = intervals
|
|
240
|
+
.filter((i) => i.end > i.start)
|
|
241
|
+
.map((i) => ({ ...i }))
|
|
242
|
+
.sort((a, b) => a.start - b.start);
|
|
243
|
+
let total = 0;
|
|
244
|
+
let cur: { start: number; end: number } | undefined;
|
|
245
|
+
for (const i of sorted) {
|
|
246
|
+
if (cur && i.start <= cur.end) {
|
|
247
|
+
cur.end = Math.max(cur.end, i.end);
|
|
248
|
+
continue;
|
|
249
|
+
}
|
|
250
|
+
if (cur) total += cur.end - cur.start;
|
|
251
|
+
cur = i;
|
|
252
|
+
}
|
|
253
|
+
if (cur) total += cur.end - cur.start;
|
|
254
|
+
return total;
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/** Analyze a run's event stream. Pure and deterministic; never mutates `events`. */
|
|
258
|
+
export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOptions = {}): FrictionDiagnosis {
|
|
259
|
+
const slowToolMs = opts.slowToolMs ?? DEFAULT_SLOW_TOOL_MS;
|
|
260
|
+
const slowModelTurnMs = opts.slowModelTurnMs ?? DEFAULT_SLOW_MODEL_TURN_MS;
|
|
261
|
+
const finished = opts.finished ?? true;
|
|
262
|
+
const owner: RunOwner = opts.owner ?? "agent";
|
|
263
|
+
const window = opts.window;
|
|
264
|
+
|
|
265
|
+
// ---- the content pass: what happened, in stream order -------------------
|
|
266
|
+
const findings: FrictionFinding[] = [];
|
|
267
|
+
// Pending calls awaiting their result, FIFO per tool name (the runner emits
|
|
268
|
+
// call→result sequentially; pairing per tool is robust to interleaving).
|
|
269
|
+
const pending = new Map<string, PendingCall[]>();
|
|
270
|
+
// `tool summary` keys that have failed at least once → a later identical call is a retry.
|
|
271
|
+
const failedCalls = new Set<string>();
|
|
272
|
+
let toolCalls = 0;
|
|
273
|
+
let firstAt: number | undefined;
|
|
274
|
+
let lastAt: number | undefined;
|
|
275
|
+
// The content events' stamps: `context` is replayed thread history published
|
|
276
|
+
// at run start with timestamps of its own, so it never moves the clock; span
|
|
277
|
+
// records are timing, not steps, and never move it either.
|
|
278
|
+
for (const ev of events) {
|
|
279
|
+
if (isSpanRecord(ev) || ev.type === "context") continue;
|
|
280
|
+
if (ev.at !== undefined) {
|
|
281
|
+
firstAt ??= ev.at;
|
|
282
|
+
lastAt = ev.at;
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
const windowEnd = window?.end ?? lastAt;
|
|
286
|
+
|
|
287
|
+
// ---- the span set: every duration comes from here -------------------------
|
|
288
|
+
const { spans, anchorOf, producedOf } = spanSet(events);
|
|
289
|
+
const losses: LossInterval[] = window ? lossesFromStream(events, { windowStart: window.start }) : [];
|
|
290
|
+
const lossStarts = losses.map((l) => l.from).sort((a, b) => a - b);
|
|
291
|
+
/** A span's end: its own, or — open — the window's end while live, the next
|
|
292
|
+
* loss after its start once finished (nothing can still be running; the end
|
|
293
|
+
* was lost), else the last content stamp. */
|
|
294
|
+
const endOf = (s: SpanRecord): number | undefined => {
|
|
295
|
+
if (s.endedAt !== undefined) return s.endedAt;
|
|
296
|
+
if (!finished) return windowEnd;
|
|
297
|
+
const next = lossStarts.find((at) => at > s.startedAt);
|
|
298
|
+
return next ?? windowEnd;
|
|
299
|
+
};
|
|
300
|
+
const durationOfSpan = (s: SpanRecord): number | undefined => {
|
|
301
|
+
const end = endOf(s);
|
|
302
|
+
return end === undefined ? undefined : Math.max(0, end - s.startedAt);
|
|
303
|
+
};
|
|
304
|
+
const toolSpans = spans.filter((s) => s.name.startsWith("tool."));
|
|
305
|
+
const turnSpans = spans.filter((s) => s.name === "model.turn");
|
|
306
|
+
const toolSpanByCallId = new Map<string, SpanRecord>();
|
|
307
|
+
const toolSpanById = new Map<string, SpanRecord>();
|
|
308
|
+
for (const s of toolSpans) {
|
|
309
|
+
toolSpanById.set(s.spanId, s);
|
|
310
|
+
const callId = s.attrs.callId;
|
|
311
|
+
if (typeof callId === "string") toolSpanByCallId.set(callId, s);
|
|
312
|
+
}
|
|
313
|
+
/** The span a call/result pair is about: by the stamped `spanId`, by
|
|
314
|
+
* `callId`, or by the synthesized id the Adapter minted from the call's index. */
|
|
315
|
+
const spanOfCall = (
|
|
316
|
+
call: PendingCall | undefined,
|
|
317
|
+
result: Extract<RunEvent, { type: "tool_result" }>,
|
|
318
|
+
): SpanRecord | undefined => {
|
|
319
|
+
const stamped = result.spanId ?? call?.event.spanId;
|
|
320
|
+
if (stamped !== undefined && toolSpanById.has(stamped)) return toolSpanById.get(stamped);
|
|
321
|
+
const callId = result.callId ?? call?.event.callId;
|
|
322
|
+
if (callId !== undefined && toolSpanByCallId.has(callId)) return toolSpanByCallId.get(callId);
|
|
323
|
+
if (callId !== undefined && toolSpanById.has(`synth:${callId}`)) return toolSpanById.get(`synth:${callId}`);
|
|
324
|
+
return call ? toolSpanById.get(`synth:${call.index}`) : undefined;
|
|
325
|
+
};
|
|
326
|
+
|
|
327
|
+
// Slow turns, keyed by the event they produced, so they land in stream order
|
|
328
|
+
// before that event's own findings (a turn is over before its tool starts).
|
|
329
|
+
const turnFindingsAt = new Map<number, FrictionFinding[]>();
|
|
330
|
+
let modelTimeMs: number | undefined;
|
|
331
|
+
for (const s of turnSpans) {
|
|
332
|
+
const durationMs = durationOfSpan(s);
|
|
333
|
+
if (durationMs === undefined) continue;
|
|
334
|
+
modelTimeMs = (modelTimeMs ?? 0) + durationMs;
|
|
335
|
+
if (durationMs < slowModelTurnMs) continue;
|
|
336
|
+
const anchor = anchorOf.get(s.spanId) ?? events.length - 1;
|
|
337
|
+
const produced = events[producedOf.get(s.spanId) ?? anchor];
|
|
338
|
+
const what =
|
|
339
|
+
produced?.type === "tool_call"
|
|
340
|
+
? typeof produced.summary === "string"
|
|
341
|
+
? produced.summary
|
|
342
|
+
: produced.tool
|
|
343
|
+
: produced?.type === "assistant"
|
|
344
|
+
? "(narration)"
|
|
345
|
+
: produced?.type === "answer"
|
|
346
|
+
? "(answer)"
|
|
347
|
+
: typeof s.attrs.stopReason === "string"
|
|
348
|
+
? `(${s.attrs.stopReason})`
|
|
349
|
+
: "(a model turn)";
|
|
350
|
+
const finding: FrictionFinding = {
|
|
351
|
+
category: "slow_model_turn",
|
|
352
|
+
severity: durationMs >= 2 * slowModelTurnMs ? "high" : "medium",
|
|
353
|
+
summary: `model turn took ${formatDuration(durationMs, "report")} before: ${what}`,
|
|
354
|
+
durationMs,
|
|
355
|
+
interval: { start: s.startedAt, end: s.startedAt + durationMs },
|
|
356
|
+
eventIndex: anchor,
|
|
357
|
+
};
|
|
358
|
+
turnFindingsAt.set(anchor, [...(turnFindingsAt.get(anchor) ?? []), finding]);
|
|
359
|
+
}
|
|
360
|
+
const flushTurns = (index: number) => {
|
|
361
|
+
const list = turnFindingsAt.get(index);
|
|
362
|
+
if (list) {
|
|
363
|
+
findings.push(...list);
|
|
364
|
+
turnFindingsAt.delete(index);
|
|
365
|
+
}
|
|
366
|
+
};
|
|
367
|
+
|
|
368
|
+
// The narrative events — the request (`input`), the thread context fed to
|
|
369
|
+
// the model (`context`), the model's prose between tools (`assistant`), the
|
|
370
|
+
// final answer (`answer`) — are the run's story, not its steps: none counts
|
|
371
|
+
// toward `eventCount`.
|
|
372
|
+
let narrativeEvents = 0;
|
|
373
|
+
let sideFactEvents = 0; // skill_use / review_artifact / pr_description / pr_opened / ship_round: facts about the run, not steps
|
|
374
|
+
let spanEvents = 0; // span_start / span_end (docs/reference/specs/tracing.md): timing records, not steps
|
|
375
|
+
let wrapUp: { index: number; at?: number } | undefined;
|
|
376
|
+
events.forEach((ev, index) => {
|
|
377
|
+
flushTurns(index);
|
|
378
|
+
if (isSpanRecord(ev)) {
|
|
379
|
+
spanEvents++;
|
|
380
|
+
return;
|
|
381
|
+
}
|
|
382
|
+
if (isNarrative(ev)) {
|
|
383
|
+
narrativeEvents++; // the narrative and `run_meta` are neither steps nor findings
|
|
384
|
+
return;
|
|
385
|
+
}
|
|
386
|
+
// Side facts about the run, not steps: skill_use rides beside a use_skill
|
|
387
|
+
// call that already produced its own tool pair; review_artifact,
|
|
388
|
+
// pr_description, pr_opened and the ship_round boundaries are published
|
|
389
|
+
// by the dispatcher/pipeline outside the model loop entirely. Counting
|
|
390
|
+
// any of them would distort the story.
|
|
391
|
+
if (
|
|
392
|
+
ev.type === "skill_use" ||
|
|
393
|
+
ev.type === "review_artifact" ||
|
|
394
|
+
ev.type === "pr_description" ||
|
|
395
|
+
ev.type === "pr_opened" ||
|
|
396
|
+
ev.type === "ship_round"
|
|
397
|
+
) {
|
|
398
|
+
sideFactEvents++;
|
|
399
|
+
return;
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
if (ev.type === "tool_call") {
|
|
403
|
+
toolCalls++;
|
|
404
|
+
if (failedCalls.has(`${ev.tool} ${ev.summary}`)) {
|
|
405
|
+
findings.push({
|
|
406
|
+
category: "retry",
|
|
407
|
+
severity: "low",
|
|
408
|
+
summary: `retried after failure: ${ev.summary}`,
|
|
409
|
+
tool: ev.tool,
|
|
410
|
+
eventIndex: index,
|
|
411
|
+
});
|
|
412
|
+
}
|
|
413
|
+
const queue = pending.get(ev.tool) ?? [];
|
|
414
|
+
queue.push({ index, event: ev });
|
|
415
|
+
pending.set(ev.tool, queue);
|
|
416
|
+
return;
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
if (ev.type === "tool_result") {
|
|
420
|
+
const callEntry = pending.get(ev.tool)?.shift();
|
|
421
|
+
const anchor = callEntry?.index ?? index;
|
|
422
|
+
const callSummary = callEntry?.event.summary ?? ev.tool;
|
|
423
|
+
const span = spanOfCall(callEntry, ev);
|
|
424
|
+
const durationMs = span ? durationOfSpan(span) : undefined;
|
|
425
|
+
const timed = (f: FrictionFinding): FrictionFinding => (durationMs !== undefined ? { ...f, durationMs } : f);
|
|
426
|
+
const interval =
|
|
427
|
+
span && durationMs !== undefined
|
|
428
|
+
? { interval: { start: span.startedAt, end: span.startedAt + durationMs } }
|
|
429
|
+
: {};
|
|
430
|
+
|
|
431
|
+
if (!ev.ok) failedCalls.add(`${ev.tool} ${callSummary}`);
|
|
432
|
+
|
|
433
|
+
if (ev.infra) {
|
|
434
|
+
// An infra-level failure is the sandbox, not the command: classify once,
|
|
435
|
+
// as infra — but name the command that was running, so "the sandbox died
|
|
436
|
+
// during installs" is readable from the findings alone.
|
|
437
|
+
findings.push(
|
|
438
|
+
timed({
|
|
439
|
+
category: "infra_failure",
|
|
440
|
+
severity: "high",
|
|
441
|
+
summary: `exec infrastructure failed during ${callSummary} → ${ev.summary}`,
|
|
442
|
+
tool: ev.tool,
|
|
443
|
+
...interval,
|
|
444
|
+
eventIndex: anchor,
|
|
445
|
+
}),
|
|
446
|
+
);
|
|
447
|
+
return;
|
|
448
|
+
}
|
|
449
|
+
if (ev.tool === "bash" && isSetupInstallCommand(callSummary)) {
|
|
450
|
+
// Setup/install is reported once, as setup — a slow or failed install is
|
|
451
|
+
// still setup cost — so the category total is the true install bill.
|
|
452
|
+
const slow = durationMs !== undefined && durationMs >= slowToolMs;
|
|
453
|
+
const label = !ev.ok ? "install failed" : slow ? "slow install" : "install";
|
|
454
|
+
findings.push(
|
|
455
|
+
timed({
|
|
456
|
+
category: "setup_install",
|
|
457
|
+
severity: !ev.ok ? "high" : slow ? "medium" : "low",
|
|
458
|
+
summary: `${label}: ${callSummary}${!ev.ok ? ` → ${ev.summary}` : ""}`,
|
|
459
|
+
tool: ev.tool,
|
|
460
|
+
eventIndex: anchor,
|
|
461
|
+
}),
|
|
462
|
+
);
|
|
463
|
+
return;
|
|
464
|
+
}
|
|
465
|
+
if (!ev.ok) {
|
|
466
|
+
// A failure that was ALSO slow cost more than a fast one: high once it
|
|
467
|
+
// crosses the slow threshold (the same bar slow_tool uses).
|
|
468
|
+
const slowFailure = durationMs !== undefined && durationMs >= slowToolMs;
|
|
469
|
+
findings.push(
|
|
470
|
+
timed({
|
|
471
|
+
category: "failed_tool",
|
|
472
|
+
severity: slowFailure ? "high" : "medium",
|
|
473
|
+
summary: `${callSummary} → ${ev.summary}`,
|
|
474
|
+
tool: ev.tool,
|
|
475
|
+
eventIndex: anchor,
|
|
476
|
+
}),
|
|
477
|
+
);
|
|
478
|
+
return;
|
|
479
|
+
}
|
|
480
|
+
if (durationMs !== undefined && durationMs >= slowToolMs) {
|
|
481
|
+
findings.push(
|
|
482
|
+
timed({
|
|
483
|
+
category: "slow_tool",
|
|
484
|
+
severity: durationMs >= 2 * slowToolMs ? "high" : "medium",
|
|
485
|
+
summary: `took ${formatDuration(durationMs, "report")}: ${callSummary}`,
|
|
486
|
+
tool: ev.tool,
|
|
487
|
+
eventIndex: anchor,
|
|
488
|
+
}),
|
|
489
|
+
);
|
|
490
|
+
}
|
|
491
|
+
return;
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
switch (ev.kind) {
|
|
495
|
+
case "wrap_up":
|
|
496
|
+
// Its extent is the time from the warning to the end of the window (how
|
|
497
|
+
// long the wind-down actually took); filled in after the loop.
|
|
498
|
+
wrapUp = { index, at: ev.at };
|
|
499
|
+
findings.push({ category: "wrap_up", severity: "medium", summary: ev.summary, eventIndex: index });
|
|
500
|
+
return;
|
|
501
|
+
case "time_budget_exhausted":
|
|
502
|
+
findings.push({
|
|
503
|
+
category: "budget_hit",
|
|
504
|
+
severity: "high",
|
|
505
|
+
summary: `budget hit (time): ${ev.summary}`,
|
|
506
|
+
eventIndex: index,
|
|
507
|
+
});
|
|
508
|
+
return;
|
|
509
|
+
case "turn_budget_exhausted":
|
|
510
|
+
findings.push({
|
|
511
|
+
category: "budget_hit",
|
|
512
|
+
severity: "high",
|
|
513
|
+
summary: `budget hit (turns): ${ev.summary}`,
|
|
514
|
+
eventIndex: index,
|
|
515
|
+
});
|
|
516
|
+
return;
|
|
517
|
+
case "sandbox_dead":
|
|
518
|
+
findings.push({
|
|
519
|
+
category: "infra_failure",
|
|
520
|
+
severity: "high",
|
|
521
|
+
summary: `sandbox dead: ${ev.summary}`,
|
|
522
|
+
eventIndex: index,
|
|
523
|
+
});
|
|
524
|
+
return;
|
|
525
|
+
case "fleet_busy":
|
|
526
|
+
// Capacity, not a dead sandbox: the run went on, but the minutes spent
|
|
527
|
+
// waiting for an instance are friction the fleet's sizing owns.
|
|
528
|
+
findings.push({
|
|
529
|
+
category: "infra_failure",
|
|
530
|
+
severity: "medium",
|
|
531
|
+
summary: `fleet busy: ${ev.summary}`,
|
|
532
|
+
eventIndex: index,
|
|
533
|
+
});
|
|
534
|
+
return;
|
|
535
|
+
}
|
|
536
|
+
});
|
|
537
|
+
// A slow turn whose produced event is past the stream (a turn that ended the run).
|
|
538
|
+
for (const index of [...turnFindingsAt.keys()].sort((a, b) => a - b)) flushTurns(index);
|
|
539
|
+
|
|
540
|
+
// In a FINISHED stream, a call with no result means the run ended mid-tool
|
|
541
|
+
// (process died, stream cut) — infrastructure friction that must not vanish
|
|
542
|
+
// silently. Mid-run it is just the tool still executing. Its extent is the
|
|
543
|
+
// open tool span's: to the next loss, else to the window's end.
|
|
544
|
+
for (const queue of finished ? pending.values() : []) {
|
|
545
|
+
for (const { index, event } of queue) {
|
|
546
|
+
const open =
|
|
547
|
+
(event.spanId !== undefined ? toolSpanById.get(event.spanId) : undefined) ??
|
|
548
|
+
(event.callId !== undefined ? toolSpanByCallId.get(event.callId) : undefined) ??
|
|
549
|
+
toolSpanById.get(`synth:${index}`);
|
|
550
|
+
const durationMs = open ? durationOfSpan(open) : undefined;
|
|
551
|
+
findings.push({
|
|
552
|
+
category: "infra_failure",
|
|
553
|
+
severity: "high",
|
|
554
|
+
summary: `no result for tool call (run ended mid-tool): ${event.summary}`,
|
|
555
|
+
tool: event.tool,
|
|
556
|
+
eventIndex: index,
|
|
557
|
+
...(durationMs !== undefined ? { durationMs } : {}),
|
|
558
|
+
...(open && durationMs !== undefined
|
|
559
|
+
? { interval: { start: open.startedAt, end: open.startedAt + durationMs } }
|
|
560
|
+
: {}),
|
|
561
|
+
});
|
|
562
|
+
}
|
|
563
|
+
}
|
|
564
|
+
if (wrapUp !== undefined) {
|
|
565
|
+
const { index, at } = wrapUp;
|
|
566
|
+
const f = findings.find((x) => x.category === "wrap_up" && x.eventIndex === index);
|
|
567
|
+
if (at !== undefined && windowEnd !== undefined && f) {
|
|
568
|
+
f.durationMs = Math.max(0, windowEnd - at);
|
|
569
|
+
f.interval = { start: at, end: Math.max(at, windowEnd) };
|
|
570
|
+
}
|
|
571
|
+
}
|
|
572
|
+
|
|
573
|
+
// ---- totals ----------------------------------------------------------------
|
|
574
|
+
let toolTimeMs: number | undefined;
|
|
575
|
+
for (const s of toolSpans) {
|
|
576
|
+
const d = durationOfSpan(s);
|
|
577
|
+
if (d !== undefined) toolTimeMs = (toolTimeMs ?? 0) + d;
|
|
578
|
+
}
|
|
579
|
+
const byCategory = Object.fromEntries(
|
|
580
|
+
FRICTION_CATEGORIES.map((c) => [c, { count: 0, durationMs: 0 } satisfies CategoryTotals]),
|
|
581
|
+
) as Record<FrictionCategory, CategoryTotals>;
|
|
582
|
+
for (const f of findings) byCategory[f.category].count++;
|
|
583
|
+
for (const c of FRICTION_CATEGORIES) {
|
|
584
|
+
const own = findings.filter((f) => f.category === c);
|
|
585
|
+
// Turns are disjoint (one loop runs at a time), so their sum is their union.
|
|
586
|
+
byCategory[c].durationMs =
|
|
587
|
+
DENOMINATOR_OF[c] === "run" && c !== "slow_model_turn"
|
|
588
|
+
? unionMs(own.flatMap((f) => (f.interval ? [f.interval] : [])))
|
|
589
|
+
: own.reduce((sum, f) => sum + (f.durationMs ?? 0), 0);
|
|
590
|
+
}
|
|
591
|
+
const runMs =
|
|
592
|
+
window !== undefined
|
|
593
|
+
? Math.max(0, window.end - window.start)
|
|
594
|
+
: firstAt !== undefined && lastAt !== undefined
|
|
595
|
+
? lastAt - firstAt
|
|
596
|
+
: undefined;
|
|
597
|
+
let shape: RunShape | undefined;
|
|
598
|
+
if (window !== undefined && finished) {
|
|
599
|
+
const { backgroundOnlyMs: _background, ...terms } = partition(spans, { window, owner, finished, losses });
|
|
600
|
+
shape = terms;
|
|
601
|
+
}
|
|
602
|
+
|
|
603
|
+
const diagnosis: FrictionDiagnosis = {
|
|
604
|
+
eventCount: events.length - narrativeEvents - sideFactEvents - spanEvents,
|
|
605
|
+
toolCalls,
|
|
606
|
+
...(runMs !== undefined ? { runMs, toolTimeMs: toolTimeMs ?? 0 } : {}),
|
|
607
|
+
...(runMs !== undefined && modelTimeMs !== undefined ? { modelTimeMs } : {}),
|
|
608
|
+
byCategory,
|
|
609
|
+
findings,
|
|
610
|
+
verdict: "",
|
|
611
|
+
...(opts.truncated ? { truncatedInput: true as const } : {}),
|
|
612
|
+
...(shape ? { shape } : {}),
|
|
613
|
+
};
|
|
614
|
+
diagnosis.verdict = verdictOf(diagnosis);
|
|
615
|
+
return diagnosis;
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
/** The dominant cause: most attributed time when timed (ties → most findings →
|
|
619
|
+
* category order), else most findings. Names the share of its denominator
|
|
620
|
+
* (`DENOMINATOR_OF`) so the reader knows whether the cause is the whole story. */
|
|
621
|
+
function verdictOf(d: FrictionDiagnosis): string {
|
|
622
|
+
if (d.findings.length === 0) return "no friction detected";
|
|
623
|
+
const ranked = FRICTION_CATEGORIES.filter((c) => d.byCategory[c].count > 0).sort(
|
|
624
|
+
(a, b) => d.byCategory[b].durationMs - d.byCategory[a].durationMs || d.byCategory[b].count - d.byCategory[a].count,
|
|
625
|
+
);
|
|
626
|
+
const top = ranked[0];
|
|
627
|
+
const t = d.byCategory[top];
|
|
628
|
+
const what = `${CATEGORY_LABEL[top]} dominated: ${t.count} finding${t.count === 1 ? "" : "s"}`;
|
|
629
|
+
if (t.durationMs === 0) return what;
|
|
630
|
+
const [denominator, of] = DENOMINATOR_OF[top] === "run" ? [d.runMs, "run time"] : [d.toolTimeMs, "tool time"];
|
|
631
|
+
const share = denominator ? ` (${Math.round((t.durationMs / denominator) * 100)}% of ${of})` : "";
|
|
632
|
+
return `${what}, ${formatDuration(t.durationMs, "report")}${share}`;
|
|
633
|
+
}
|
|
634
|
+
|
|
635
|
+
/** Plain-text report for the CLI / terminal. */
|
|
636
|
+
export function formatFrictionReport(d: FrictionDiagnosis): string {
|
|
637
|
+
const lines: string[] = [`verdict: ${d.verdict}`];
|
|
638
|
+
const totals = [`events: ${d.eventCount}`];
|
|
639
|
+
if (d.toolCalls > 0) totals.push(`tool calls: ${d.toolCalls}`);
|
|
640
|
+
if (d.runMs !== undefined) totals.push(`run: ${formatDuration(d.runMs, "report")}`);
|
|
641
|
+
if (d.toolCalls > 0 && d.toolTimeMs !== undefined)
|
|
642
|
+
totals.push(`tool time: ${formatDuration(d.toolTimeMs, "report")}`);
|
|
643
|
+
if (d.modelTimeMs !== undefined) totals.push(`model time: ${formatDuration(d.modelTimeMs, "report")}`);
|
|
644
|
+
if (d.truncatedInput) totals.push("(input truncated — some records were dropped before analysis)");
|
|
645
|
+
lines.push(totals.join(" · "));
|
|
646
|
+
if (d.shape) {
|
|
647
|
+
const shape = formatShape(d.shape);
|
|
648
|
+
lines.push(`shape: ${shape ?? `${formatDuration(d.shape.windowMs, "report")} (one bucket)`}`);
|
|
649
|
+
}
|
|
650
|
+
const width = Math.max(...FRICTION_CATEGORIES.map((c) => CATEGORY_LABEL[c].length), "category".length);
|
|
651
|
+
lines.push("", `${"category".padEnd(width)} count time`);
|
|
652
|
+
for (const c of FRICTION_CATEGORIES) {
|
|
653
|
+
const t = d.byCategory[c];
|
|
654
|
+
const time = t.count && t.durationMs > 0 ? formatDuration(t.durationMs, "report") : "-";
|
|
655
|
+
lines.push(`${CATEGORY_LABEL[c].padEnd(width)} ${String(t.count).padStart(5)} ${time}`);
|
|
656
|
+
}
|
|
657
|
+
if (d.findings.length > 0) {
|
|
658
|
+
lines.push("", "findings (stream order):");
|
|
659
|
+
for (const f of d.findings) {
|
|
660
|
+
const dur = f.durationMs !== undefined ? ` (${formatDuration(f.durationMs, "report")})` : "";
|
|
661
|
+
lines.push(` #${f.eventIndex} [${CATEGORY_LABEL[f.category]}] ${f.severity}${dur} ${f.summary}`);
|
|
662
|
+
}
|
|
663
|
+
}
|
|
664
|
+
return lines.join("\n");
|
|
665
|
+
}
|