@spunto/build 0.3.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -102,6 +102,52 @@ fields it has no feature for** — shared volumes are a Lite concept, the task s
102
102
  both optional. Dropping a field the target cannot honour is the correct outcome; refusing the file
103
103
  is not.
104
104
 
105
+ ### `@spunto/build/script` — turning a project spec into the shell a worker runs
106
+
107
+ Four generators, one recipe: `buildImageScript` bakes the project image, `buildSetupScript` is the
108
+ first boot, `buildStartScript` is every boot, `buildWorkerScript` assembles the container CMD.
109
+
110
+ ```ts
111
+ import { buildImageScript, IMAGE_RECIPE_VERSION } from "@spunto/build/script"
112
+
113
+ const { script, steps, hasDinD } = buildImageScript({ features, vscodeExtensions })
114
+ ```
115
+
116
+ No filesystem, no clock: the one install script the platform ships itself is embedded as a string
117
+ (`local-features.ts`), and every path a generated script writes to comes from `./naming` rather
118
+ than a literal.
119
+
120
+ `buildContainerScript` and `buildSetupPlan` are shaped by one product's model — an agent that
121
+ orchestrates setup phase by phase instead of a container running one long CMD. They live here
122
+ because they share every helper in the file; splitting them out would fork the helpers, which is
123
+ the duplication this package exists to remove.
124
+
125
+ ### `@spunto/build/catalogs` — what a project picker offers
126
+
127
+ `AVAILABLE_IMAGES`, `AVAILABLE_FEATURES`, `SUGGESTED_EXTENSIONS`. Data, but shared data: the two
128
+ hand-maintained copies had already drifted, one of them advertising Oh My Zsh in a feature the
129
+ image recipe explicitly disables.
130
+
131
+ Project *templates* are deliberately absent — one product's clone a starter repository and carry
132
+ onboarding-gallery presentation, the other's only configure an environment. They answer different
133
+ questions.
134
+
135
+ ### `@spunto/build/types` — the shapes both schemas point at
136
+
137
+ `ProjectFeature`, `Repository`, `SetupStatus`. Declared here so each product's ORM can point its
138
+ JSON columns at them with `$type<…>()`, instead of a generator having to import a database schema.
139
+ `SetupStatus` is what a shell script writes into a container and a control plane reads back minutes
140
+ later: changing it is a migration, not an edit.
141
+
142
+ ### `@spunto/build/agent-stream` — reading an agent session as it happens
143
+
144
+ One adapter per harness dialect, one vocabulary out. A CLI's JSON output is an output format, not
145
+ an API; adapters normalise into a small set of event types so stored history doesn't date the first
146
+ time a vendor reshuffles a field.
147
+
148
+ Here for the reason that inverts the rest of the package: **nothing has forked this yet.** Putting
149
+ it in the shared package now costs nothing and means the second product never writes its own.
150
+
105
151
  ## The rule
106
152
 
107
153
  A module belongs in this package only if it imports **no** ORM schema, **no** HTTP framework, **no**
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@spunto/build",
3
- "version": "0.3.1",
3
+ "version": "0.4.0",
4
4
  "description": "Spunto's shared Build engine \u2014 the devcontainer image protocol and VS Code extension registry clients, with no database, no HTTP framework and no UI.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -31,22 +31,39 @@
31
31
  "types": "./src/steps/index.ts",
32
32
  "import": "./src/steps/index.ts"
33
33
  },
34
- "./extensions": {
35
- "types": "./src/extensions/index.ts",
36
- "import": "./src/extensions/index.ts"
37
- },
38
34
  "./naming": {
39
35
  "types": "./src/naming/index.ts",
40
36
  "import": "./src/naming/index.ts"
41
37
  },
38
+ "./types": {
39
+ "types": "./src/types/index.ts",
40
+ "import": "./src/types/index.ts"
41
+ },
42
+ "./catalogs": {
43
+ "types": "./src/catalogs/index.ts",
44
+ "import": "./src/catalogs/index.ts"
45
+ },
46
+ "./extensions": {
47
+ "types": "./src/extensions/index.ts",
48
+ "import": "./src/extensions/index.ts"
49
+ },
42
50
  "./spec": {
43
51
  "types": "./src/spec/index.ts",
44
52
  "import": "./src/spec/index.ts"
53
+ },
54
+ "./script": {
55
+ "types": "./src/script/index.ts",
56
+ "import": "./src/script/index.ts"
57
+ },
58
+ "./agent-stream": {
59
+ "types": "./src/agent-stream/index.ts",
60
+ "import": "./src/agent-stream/index.ts"
45
61
  }
46
62
  },
47
63
  "files": [
48
64
  "src",
49
65
  "!src/**/*.test.ts",
66
+ "!src/**/__snapshots__",
50
67
  "README.md"
51
68
  ],
52
69
  "scripts": {
@@ -0,0 +1,561 @@
1
+ /**
2
+ * Reading an agent session as it happens (RFC 0020, phase 1).
3
+ *
4
+ * A task's harness is launched with the prompt on stdin and nothing else agreed — that poverty is
5
+ * what makes `taskAgentCommand` able to hold Claude Code, Codex or a shell script. Streaming needs
6
+ * strictly more than that, so it is **opt-in per project** (`projects.taskAgentProtocol`) and the
7
+ * extra knowledge lives here, in one adapter per dialect, and nowhere else.
8
+ *
9
+ * Two rules the rest of the code depends on:
10
+ *
11
+ * 1. **Our vocabulary is the stored one.** A harness's JSON shape is an output format, not an
12
+ * API; storing it would date the history. Adapters translate into the small set below, and
13
+ * anything unmapped becomes `raw` rather than being dropped — same escape hatch as
14
+ * `tasks.meta`, which has already earned its keep twice.
15
+ * 2. **A malformed line is never fatal.** This parses the output of a third-party CLI on a read
16
+ * path. Garbage in a line costs that line, not the task.
17
+ *
18
+ * Pure functions only: no database, no worker, no clock beyond what a caller passes in. The
19
+ * ingestion loop lives in task-events.service.ts.
20
+ */
21
+
22
+ import { contextWindowOf } from "./model-window"
23
+
24
+ export const AGENT_PROTOCOLS = ["none", "claude-json", "claude-stream", "jsonl"] as const
25
+ export type AgentProtocol = (typeof AGENT_PROTOCOLS)[number]
26
+
27
+ /**
28
+ * The normalized vocabulary. Deliberately short — it has to be translatable *from* several
29
+ * harnesses, so anything specific to one of them does not belong in it.
30
+ */
31
+ export const TASK_EVENT_TYPES = [
32
+ "session.started", // { harness, model?, cwd?, tools? }
33
+ "session.title", // { title } — the name the harness gave the conversation
34
+ "message", // { role, text }
35
+ "thinking", // { text }
36
+ "tool.call", // { callId, name, input }
37
+ "tool.result", // { callId, isError, output }
38
+ "plan", // { steps: [{ text, status }] }
39
+ // { scope: "call" | "session", model?, inputTokens?, outputTokens?, cacheReadTokens?,
40
+ // cacheCreationTokens?, contextTokens?, costUsd?, turns?, durationMs? } — see `summarizeUsage`
41
+ "usage",
42
+ "session.ended", // { status, stopReason?, summary? }
43
+ "raw", // anything an adapter could not place
44
+ ] as const
45
+ export type TaskEventType = (typeof TASK_EVENT_TYPES)[number]
46
+
47
+ export type NormalizedEvent = {
48
+ type: TaskEventType
49
+ payload: Record<string, unknown>
50
+ }
51
+
52
+ /**
53
+ * Per-field ceiling on stored text. A single `tool.result` can be a 2 MB file read; the timeline
54
+ * needs enough of it to be readable, not all of it. `truncated` is set so the UI can say so
55
+ * instead of quietly lying about what the agent saw.
56
+ */
57
+ const MAX_TEXT = 8_000
58
+
59
+ function clip(value: unknown): { text: string; truncated: boolean } {
60
+ const text =
61
+ typeof value === "string" ? value : value === undefined || value === null ? "" : safeStringify(value)
62
+ return text.length > MAX_TEXT ? { text: text.slice(0, MAX_TEXT), truncated: true } : { text, truncated: false }
63
+ }
64
+
65
+ function safeStringify(value: unknown): string {
66
+ try {
67
+ return JSON.stringify(value) ?? ""
68
+ } catch {
69
+ return String(value)
70
+ }
71
+ }
72
+
73
+ /**
74
+ * The command actually run, given the project's harness and its protocol.
75
+ *
76
+ * The platform appends the flags rather than asking the project to, because getting them wrong is
77
+ * silent: a missing `--verbose` makes Claude Code refuse `stream-json` outright, and a missing
78
+ * `--output-format` produces prose that parses into nothing at all.
79
+ *
80
+ * Left alone when the command already names an output format — a project whose harness is a
81
+ * pipeline, or who wants different flags, keeps full control by saying so.
82
+ */
83
+ export function applyProtocol(command: string, protocol: AgentProtocol): string {
84
+ if (protocol !== "claude-json" && protocol !== "claude-stream") return command
85
+ if (/--output-format/.test(command)) return command
86
+ const flags = ["--output-format stream-json --verbose"]
87
+ // Interactive mode. `--input-format stream-json` is what keeps the harness alive between
88
+ // turns — it reads instructions as a stream instead of exiting after one prompt — which is the
89
+ // precondition for both answering it mid-flight and interrupting it. Measured in a worker
90
+ // before it was built: two turns, one session id, and an interrupt the process survives.
91
+ if (protocol === "claude-stream") flags.push("--input-format stream-json")
92
+ return `${command} ${flags.join(" ")}`
93
+ }
94
+
95
+ /** Does this protocol keep the harness alive, listening on stdin, between turns? */
96
+ export function isInteractive(protocol: AgentProtocol): boolean {
97
+ return protocol === "claude-stream"
98
+ }
99
+
100
+ /**
101
+ * One turn of the conversation, as the harness expects it on stdin.
102
+ *
103
+ * Newline-terminated and emitted as a single write: under PIPE_BUF that is atomic, which is what
104
+ * stops two messages from interleaving into one unparseable line.
105
+ */
106
+ export function userTurnLine(protocol: AgentProtocol, text: string): string {
107
+ if (protocol !== "claude-stream") return `${text}\n`
108
+ return `${JSON.stringify({ type: "user", message: { role: "user", content: [{ type: "text", text }] } })}\n`
109
+ }
110
+
111
+ /**
112
+ * Stop what the harness is doing, without killing it.
113
+ *
114
+ * Measured: Claude Code answers a `control_request`/`interrupt` with a `control_response`
115
+ * success, ends the turn with `result/error_during_execution`, and **stays alive** — so the next
116
+ * message is a normal turn rather than a resume.
117
+ */
118
+ export function interruptLine(protocol: AgentProtocol, requestId: string): string | null {
119
+ if (protocol !== "claude-stream") return null
120
+ return `${JSON.stringify({ type: "control_request", request_id: requestId, request: { subtype: "interrupt" } })}\n`
121
+ }
122
+
123
+ /**
124
+ * A shell that prints the line where the harness wrote **the name it gave the conversation**, or
125
+ * null for a dialect that has no such thing.
126
+ *
127
+ * This exists because Claude Code names its own conversations — the title you see in the terminal
128
+ * and in `/resume` — but does **not** put that name on stdout: measured in a worker, a
129
+ * `--output-format stream-json` session emits `system/init`, `assistant`, `user` and `result`,
130
+ * and nothing else. The name is written to the session transcript, as one
131
+ * `{"type":"ai-title","aiTitle":…}` line per revision. So the only way to have it is to go and
132
+ * read it, once, out of band — which is why this returns a *shell* rather than a parser: it runs
133
+ * in the worker, where the transcript is.
134
+ *
135
+ * Deliberately not a `--name` handed to the harness (the flag exists): a task created without a
136
+ * title has no name to impose, and imposing the prompt's first line would just be our own
137
+ * placeholder read back to us.
138
+ *
139
+ * Found by file name rather than by rebuilding the per-directory path Claude Code derives from
140
+ * the cwd: that encoding is an internal detail, `find` by session id is not. A harness run with
141
+ * `--no-session-persistence` writes no transcript, and then this finds nothing — which is a
142
+ * normal answer, not an error.
143
+ *
144
+ * `jsonl` gets no probe on purpose: a harness speaking our vocabulary says the name in band,
145
+ * `{"type":"session.title","title":"…"}`, and pays nothing for it.
146
+ */
147
+ export function sessionTitleProbe(protocol: AgentProtocol, sessionId: string): string | null {
148
+ if (!readsTitleOutOfBand(protocol)) return null
149
+ // The id comes from the harness's own output, and it is about to be interpolated into a shell
150
+ // run inside the worker. Nothing outside this alphabet has any business in a session id, and
151
+ // refusing is cheaper than trusting a CLI's field to be what it was yesterday. It also means
152
+ // the single quotes below are enough: the set excludes the quote itself.
153
+ if (!/^[A-Za-z0-9._-]{1,128}$/.test(sessionId)) return null
154
+ return [
155
+ `root="\${CLAUDE_CONFIG_DIR:-$HOME/.claude}/projects"`,
156
+ `f=$(find "$root" -name '${sessionId}.jsonl' -type f 2>/dev/null | head -1)`,
157
+ `[ -n "$f" ] || exit 0`,
158
+ // The last few candidates, not just the last one: the pattern is a substring match on a file
159
+ // the agent itself can write into (a tool result quoting these very lines would match), so
160
+ // the reader below takes the last line that is actually a title record.
161
+ `grep -a '"type":"ai-title"' "$f" | tail -5`,
162
+ ].join("\n")
163
+ }
164
+
165
+ /**
166
+ * Does this dialect's name have to be fetched, rather than read off the stream? Asked *before* a
167
+ * session id is looked up, so a harness that says its name in band pays nothing for the question.
168
+ */
169
+ export function readsTitleOutOfBand(protocol: AgentProtocol): boolean {
170
+ return protocol === "claude-json" || protocol === "claude-stream"
171
+ }
172
+
173
+ /** Longest name we keep. The column allows 200; a harness that writes an essay gets cut. */
174
+ const MAX_TITLE = 200
175
+
176
+ /**
177
+ * The name, out of what the probe printed. Newest first, and the first line that is really a
178
+ * title record wins — see the `tail -5` above.
179
+ */
180
+ export function parseSessionTitle(protocol: AgentProtocol, output: string): { title: string; source: string } | null {
181
+ if (!readsTitleOutOfBand(protocol)) return null
182
+ const lines = output.split("\n").map((l) => l.trim()).filter(Boolean)
183
+ for (let i = lines.length - 1; i >= 0; i--) {
184
+ let parsed: unknown
185
+ try {
186
+ parsed = JSON.parse(lines[i])
187
+ } catch {
188
+ continue
189
+ }
190
+ if (!parsed || typeof parsed !== "object") continue
191
+ const obj = parsed as Record<string, unknown>
192
+ if (obj.type !== "ai-title") continue
193
+ const title = cleanTitle(obj.aiTitle)
194
+ if (title) return { title, source: lines[i] }
195
+ }
196
+ return null
197
+ }
198
+
199
+ /** One line, no runs of whitespace, bounded — a title goes in a list row and in a page header. */
200
+ export function cleanTitle(value: unknown): string | null {
201
+ if (typeof value !== "string") return null
202
+ const title = value.replace(/\s+/g, " ").trim().slice(0, MAX_TITLE).trim()
203
+ return title || null
204
+ }
205
+
206
+ /**
207
+ * Parse one line of a harness's stream into zero or more normalized events.
208
+ *
209
+ * Zero is a normal answer: a blank line, a line of prose from a harness that also logs to stdout,
210
+ * or an event type we deliberately have no use for.
211
+ */
212
+ export function parseStreamLine(protocol: AgentProtocol, line: string): NormalizedEvent[] {
213
+ const trimmed = line.trim()
214
+ if (!trimmed) return []
215
+ let parsed: unknown
216
+ try {
217
+ parsed = JSON.parse(trimmed)
218
+ } catch {
219
+ // Not JSON. A harness that also writes human lines to stdout is common enough (a banner, a
220
+ // warning) that this must not be an error — but it must not be invisible either.
221
+ return [{ type: "raw", payload: { text: clip(trimmed).text } }]
222
+ }
223
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return []
224
+ const obj = parsed as Record<string, unknown>
225
+ return protocol === "claude-json" || protocol === "claude-stream" ? adaptClaudeJson(obj) : adaptPassthrough(obj)
226
+ }
227
+
228
+ /**
229
+ * `jsonl` — the project's harness already speaks our vocabulary. The escape hatch that means a
230
+ * home-made agent has nothing to learn beyond one page of documentation, and the thing that keeps
231
+ * the vocabulary honest: it is a format someone can write to, not just one we read from.
232
+ */
233
+ function adaptPassthrough(obj: Record<string, unknown>): NormalizedEvent[] {
234
+ const type = typeof obj.type === "string" ? obj.type : null
235
+ if (!type) return [{ type: "raw", payload: { json: obj } }]
236
+ if (!(TASK_EVENT_TYPES as readonly string[]).includes(type)) {
237
+ return [{ type: "raw", payload: { json: obj } }]
238
+ }
239
+ const { type: _t, ...payload } = obj
240
+ return [{ type: type as TaskEventType, payload }]
241
+ }
242
+
243
+ /**
244
+ * `claude-json` — Claude Code's `--output-format stream-json --verbose`.
245
+ *
246
+ * Written against output measured in a real worker, not against a spec: the shapes below are
247
+ * what the CLI emitted, and `task-events.service.test.ts` replays recorded samples so a change in
248
+ * that output shows up as a failing test rather than as an empty timeline.
249
+ */
250
+ function adaptClaudeJson(obj: Record<string, unknown>): NormalizedEvent[] {
251
+ const type = obj.type
252
+
253
+ if (type === "system" && obj.subtype === "init") {
254
+ const model = typeof obj.model === "string" ? obj.model : null
255
+ return [
256
+ {
257
+ type: "session.started",
258
+ payload: {
259
+ harness: "claude-code",
260
+ model,
261
+ // Read here and nowhere else: this is the only line that spells the model the way the
262
+ // *session* was opened (`claude-opus-5[1m]`), suffix included — the assistant messages
263
+ // that follow carry the plain API id. From the catalogue only, so this stays a pure
264
+ // function; the ingestion loop replaces it with the model's own answer when the task
265
+ // has a key to ask with (`resolveContextWindow`).
266
+ contextWindow: contextWindowOf(model),
267
+ cwd: typeof obj.cwd === "string" ? obj.cwd : null,
268
+ sessionId: typeof obj.session_id === "string" ? obj.session_id : null,
269
+ tools: Array.isArray(obj.tools) ? obj.tools.filter((t) => typeof t === "string") : [],
270
+ },
271
+ },
272
+ ]
273
+ }
274
+
275
+ if (type === "assistant" || type === "user") {
276
+ const message = obj.message as Record<string, unknown> | undefined
277
+ const content = Array.isArray(message?.content) ? message!.content : []
278
+ const role = type === "assistant" ? "assistant" : "user"
279
+ const events: NormalizedEvent[] = []
280
+ for (const rawBlock of content) {
281
+ if (!rawBlock || typeof rawBlock !== "object") continue
282
+ const block = rawBlock as Record<string, unknown>
283
+ switch (block.type) {
284
+ case "text": {
285
+ const { text, truncated } = clip(block.text)
286
+ if (text) events.push({ type: "message", payload: { role, text, truncated } })
287
+ break
288
+ }
289
+ case "thinking": {
290
+ const { text, truncated } = clip(block.thinking)
291
+ if (text) events.push({ type: "thinking", payload: { text, truncated } })
292
+ break
293
+ }
294
+ case "tool_use": {
295
+ const name = typeof block.name === "string" ? block.name : "tool"
296
+ const { text: input, truncated } = clip(block.input)
297
+ events.push({
298
+ type: "tool.call",
299
+ payload: { callId: typeof block.id === "string" ? block.id : null, name, input, truncated },
300
+ })
301
+ // A todo list is a plan, and a plan is the single most useful thing to show while a
302
+ // session runs. Emitted *alongside* the tool call rather than instead of it: the call
303
+ // is what happened, the plan is what it means.
304
+ const plan = todoPlan(block.input)
305
+ if (plan) events.push(plan)
306
+ break
307
+ }
308
+ case "tool_result": {
309
+ const { text: output, truncated } = clip(
310
+ typeof block.content === "string" ? block.content : block.content ?? "",
311
+ )
312
+ events.push({
313
+ type: "tool.result",
314
+ payload: {
315
+ callId: typeof block.tool_use_id === "string" ? block.tool_use_id : null,
316
+ isError: block.is_error === true,
317
+ output,
318
+ truncated,
319
+ },
320
+ })
321
+ break
322
+ }
323
+ }
324
+ }
325
+ // What the call that produced this message cost, and how full the context was when it ran.
326
+ // Only assistant messages carry it — a `user` line is tool results being handed back.
327
+ if (type === "assistant") {
328
+ const measured = callUsage(message)
329
+ if (measured) events.push(measured)
330
+ }
331
+ return events
332
+ }
333
+
334
+ if (type === "result") {
335
+ const usage = (obj.usage ?? {}) as Record<string, unknown>
336
+ const events: NormalizedEvent[] = [
337
+ {
338
+ type: "usage",
339
+ payload: {
340
+ // The harness's own running total, **cumulative over the whole process** — with an
341
+ // interactive protocol a `result` lands at the end of every turn and each one restates
342
+ // the session so far. Summing them would bill the first turn once per turn that
343
+ // followed; the latest one is the answer. (Claude Code's own SDK reader does the same:
344
+ // it overwrites `total_cost_usd`/`num_turns` on each result rather than adding.)
345
+ scope: "session",
346
+ inputTokens: num(usage.input_tokens),
347
+ outputTokens: num(usage.output_tokens),
348
+ cacheReadTokens: num(usage.cache_read_input_tokens),
349
+ cacheCreationTokens: num(usage.cache_creation_input_tokens),
350
+ costUsd: num(obj.total_cost_usd),
351
+ turns: num(obj.num_turns),
352
+ durationMs: num(obj.duration_ms),
353
+ },
354
+ },
355
+ {
356
+ type: "session.ended",
357
+ payload: {
358
+ status: obj.subtype === "success" ? "succeeded" : "failed",
359
+ stopReason: typeof obj.stop_reason === "string" ? obj.stop_reason : null,
360
+ summary: clip(obj.result).text,
361
+ },
362
+ },
363
+ ]
364
+ return events
365
+ }
366
+
367
+ // Known-but-uninteresting (`rate_limit_event`, `system/thinking_tokens`) and anything the CLI
368
+ // grows next land here rather than being dropped: an unknown event is information about the
369
+ // harness, and the UI keeps it behind a toggle.
370
+ return [{ type: "raw", payload: { json: obj } }]
371
+ }
372
+
373
+ function num(value: unknown): number | null {
374
+ return typeof value === "number" && Number.isFinite(value) ? value : null
375
+ }
376
+
377
+ /**
378
+ * What **one model call** consumed, read off an assistant message.
379
+ *
380
+ * Two different questions are answered by these numbers, and only the first is obvious:
381
+ *
382
+ * - added up over the session, they are what it has spent in tokens — and unlike the `result`
383
+ * line, they exist *during* a turn, which is when someone is watching;
384
+ * - the latest one alone is the **context**: `input + cache_read + cache_creation` is the prompt
385
+ * the harness just sent, so it is how full the window is right now. Same formula Claude Code
386
+ * uses for its own context meter.
387
+ *
388
+ * Skipped when every counter is zero: an API error makes the CLI emit a synthetic assistant
389
+ * message with an all-zero usage, and drawing a context that just fell to zero would be a lie
390
+ * about the session rather than a fact about the error.
391
+ */
392
+ function callUsage(message: Record<string, unknown> | undefined): NormalizedEvent | null {
393
+ const usage = message?.usage
394
+ if (!usage || typeof usage !== "object") return null
395
+ const u = usage as Record<string, unknown>
396
+ const inputTokens = num(u.input_tokens) ?? 0
397
+ const outputTokens = num(u.output_tokens) ?? 0
398
+ const cacheReadTokens = num(u.cache_read_input_tokens) ?? 0
399
+ const cacheCreationTokens = num(u.cache_creation_input_tokens) ?? 0
400
+ if (inputTokens + outputTokens + cacheReadTokens + cacheCreationTokens === 0) return null
401
+ // `<synthetic>` is what the CLI puts there when no model ran at all.
402
+ const model = typeof message?.model === "string" && !message.model.startsWith("<") ? message.model : null
403
+ return {
404
+ type: "usage",
405
+ payload: {
406
+ scope: "call",
407
+ model,
408
+ inputTokens,
409
+ outputTokens,
410
+ cacheReadTokens,
411
+ cacheCreationTokens,
412
+ contextTokens: inputTokens + cacheReadTokens + cacheCreationTokens,
413
+ },
414
+ }
415
+ }
416
+
417
+ /** What a session has consumed so far — the recap the cockpit puts beside the timeline. */
418
+ export type SessionUsage = {
419
+ /** The model that ran, as the harness named it. */
420
+ model: string | null
421
+ /** Tokens in the prompt of the **latest** call: how full the window is right now. */
422
+ contextTokens: number | null
423
+ /** What that window holds, when it can be told from the model. */
424
+ contextWindow: number | null
425
+ inputTokens: number | null
426
+ outputTokens: number | null
427
+ cacheReadTokens: number | null
428
+ cacheCreationTokens: number | null
429
+ /** Only the harness can price a call, so this is always its own number, never ours. */
430
+ costUsd: number | null
431
+ turns: number | null
432
+ durationMs: number | null
433
+ }
434
+
435
+ /** The measurements `summarizeUsage` reduces, however the caller got hold of them. */
436
+ export type UsageInput = {
437
+ /** Summed over every `scope: "call"` event of the task. */
438
+ calls: {
439
+ count: number
440
+ inputTokens: number
441
+ outputTokens: number
442
+ cacheReadTokens: number
443
+ cacheCreationTokens: number
444
+ }
445
+ /** The most recent `scope: "call"` payload — the one that carries the live context. */
446
+ latestCall: Record<string, unknown> | null
447
+ /** The most recent `scope: "session"` payload — the harness's running total. */
448
+ latestSession: Record<string, unknown> | null
449
+ /** The most recent `session.started` payload, for the model and the window it implies. */
450
+ started: Record<string, unknown> | null
451
+ }
452
+
453
+ /**
454
+ * Fold a task's usage events into the one recap a reader wants.
455
+ *
456
+ * Which number comes from where is the whole content of this function:
457
+ *
458
+ * - **tokens** are summed from the per-call events rather than taken from the harness's total,
459
+ * because a turn in flight has not produced a total yet — and "0 tokens" under a session that
460
+ * has been working for ten minutes is the one answer that is never true. The harness's total
461
+ * is the fallback for a dialect that only reports at the end.
462
+ * - **money, turns and duration** can only come from the harness: pricing a call needs a rate
463
+ * card, and a platform that made one up would be quoting a number it cannot stand behind.
464
+ * - **context** is the latest call's prompt, against the window the *session* was opened with.
465
+ *
466
+ * `null` when a session has emitted no usage at all — nothing measured is not zero measured, and
467
+ * the difference is visible: the panel stays away instead of claiming a free session.
468
+ */
469
+ export function summarizeUsage(input: UsageInput): SessionUsage | null {
470
+ const { calls, latestCall, latestSession, started } = input
471
+ if (calls.count === 0 && !latestSession) return null
472
+ const measured = calls.count > 0
473
+ const fallback = (key: string): number | null => num(latestSession?.[key])
474
+ return {
475
+ model:
476
+ (typeof latestCall?.model === "string" ? latestCall.model : null) ??
477
+ (typeof started?.model === "string" ? started.model : null),
478
+ contextTokens: num(latestCall?.contextTokens),
479
+ // Read off the line that opened the session; older events predate the field, so the model
480
+ // string is re-read as a fallback rather than leaving the meter blank on a live task.
481
+ contextWindow: num(started?.contextWindow) ?? contextWindowOf(started?.model as string | undefined),
482
+ inputTokens: measured ? calls.inputTokens : fallback("inputTokens"),
483
+ outputTokens: measured ? calls.outputTokens : fallback("outputTokens"),
484
+ cacheReadTokens: measured ? calls.cacheReadTokens : fallback("cacheReadTokens"),
485
+ cacheCreationTokens: measured ? calls.cacheCreationTokens : fallback("cacheCreationTokens"),
486
+ costUsd: fallback("costUsd"),
487
+ turns: fallback("turns"),
488
+ durationMs: fallback("durationMs"),
489
+ }
490
+ }
491
+
492
+ /** Claude Code's TodoWrite input, as a `plan` — `{ todos: [{ content, status }] }`. */
493
+ function todoPlan(input: unknown): NormalizedEvent | null {
494
+ if (!input || typeof input !== "object") return null
495
+ const todos = (input as Record<string, unknown>).todos
496
+ if (!Array.isArray(todos) || todos.length === 0) return null
497
+ const steps = todos
498
+ .filter((t): t is Record<string, unknown> => !!t && typeof t === "object")
499
+ .map((t) => ({
500
+ text: clip(t.content ?? t.activeForm ?? "").text,
501
+ status: typeof t.status === "string" ? t.status : "pending",
502
+ }))
503
+ .filter((s) => s.text)
504
+ return steps.length > 0 ? { type: "plan", payload: { steps } } : null
505
+ }
506
+
507
+ /** An event and the harness line it came from. One line can yield several. */
508
+ export type SourcedEvent = NormalizedEvent & { source: string }
509
+
510
+ /**
511
+ * Parse whole lines, keeping each event attached to the line it was derived from.
512
+ *
513
+ * The pairing lives here, next to the adapters, because it is the thing that makes normalization
514
+ * reversible: `type`/`payload` is our reading of the line, `source` is the line. Get the reading
515
+ * wrong and the line is still there to read again.
516
+ *
517
+ * `budget` bounds how many events one pass may produce, so a runaway session cannot blow past the
518
+ * per-task ceiling in a single chunk. An adapter that throws costs its line, not the pass.
519
+ */
520
+ export function parseLines(protocol: AgentProtocol, lines: string[], budget: number): SourcedEvent[] {
521
+ const out: SourcedEvent[] = []
522
+ for (const line of lines) {
523
+ if (out.length >= budget) break
524
+ let events: NormalizedEvent[]
525
+ try {
526
+ events = parseStreamLine(protocol, line)
527
+ } catch (err) {
528
+ console.warn(`[tasks] agent stream adapter threw: ${err}`)
529
+ continue
530
+ }
531
+ const source = line.trim()
532
+ for (const event of events) {
533
+ if (out.length >= budget) break
534
+ out.push({ ...event, source })
535
+ }
536
+ }
537
+ return out
538
+ }
539
+
540
+ /**
541
+ * Split a chunk of bytes into whole lines, and say how many bytes that consumed.
542
+ *
543
+ * The cursor only ever advances past a **complete** line: a read that lands mid-line leaves the
544
+ * tail for the next one. Without this, a 3 KB JSON event straddling two reads would be parsed as
545
+ * two halves — twice — and neither would be valid JSON.
546
+ */
547
+ export function splitCompleteLines(chunk: Uint8Array): { lines: string[]; consumed: number } {
548
+ // `Uint8Array` rather than Node's `Buffer`: a Buffer *is* one, so every caller compiles
549
+ // unchanged, and the package stays runnable anywhere — see the rule in the README. That costs
550
+ // `lastIndexOf`, which Uint8Array does not have, hence the backwards scan.
551
+ let lastNewline = -1
552
+ for (let i = chunk.length - 1; i >= 0; i--) {
553
+ if (chunk[i] === 0x0a) {
554
+ lastNewline = i
555
+ break
556
+ }
557
+ }
558
+ if (lastNewline === -1) return { lines: [], consumed: 0 }
559
+ const consumed = lastNewline + 1
560
+ return { lines: new TextDecoder().decode(chunk.subarray(0, consumed)).split("\n"), consumed }
561
+ }