@spunto/build 0.3.1 → 0.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +51 -4
- package/package.json +22 -5
- package/src/agent-stream/agent-stream.ts +561 -0
- package/src/agent-stream/index.ts +54 -0
- package/src/agent-stream/model-window.ts +189 -0
- package/src/catalogs/index.ts +216 -0
- package/src/index.ts +4 -0
- package/src/script/index.ts +30 -0
- package/src/script/local-features.ts +39 -0
- package/src/script/setup-script.ts +1932 -0
- package/src/types/index.ts +68 -0
package/README.md
CHANGED
|
@@ -102,6 +102,52 @@ fields it has no feature for** — shared volumes are a Lite concept, the task s
|
|
|
102
102
|
both optional. Dropping a field the target cannot honour is the correct outcome; refusing the file
|
|
103
103
|
is not.
|
|
104
104
|
|
|
105
|
+
### `@spunto/build/script` — turning a project spec into the shell a worker runs
|
|
106
|
+
|
|
107
|
+
Four generators, one recipe: `buildImageScript` bakes the project image, `buildSetupScript` is the
|
|
108
|
+
first boot, `buildStartScript` is every boot, `buildWorkerScript` assembles the container CMD.
|
|
109
|
+
|
|
110
|
+
```ts
|
|
111
|
+
import { buildImageScript, IMAGE_RECIPE_VERSION } from "@spunto/build/script"
|
|
112
|
+
|
|
113
|
+
const { script, steps, hasDinD } = buildImageScript({ features, vscodeExtensions })
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
No filesystem, no clock: the one install script the platform ships itself is embedded as a string
|
|
117
|
+
(`local-features.ts`), and every path a generated script writes to comes from `./naming` rather
|
|
118
|
+
than a literal.
|
|
119
|
+
|
|
120
|
+
`buildContainerScript` and `buildSetupPlan` are shaped by one product's model — an agent that
|
|
121
|
+
orchestrates setup phase by phase instead of a container running one long CMD. They live here
|
|
122
|
+
because they share every helper in the file; splitting them out would fork the helpers, which is
|
|
123
|
+
the duplication this package exists to remove.
|
|
124
|
+
|
|
125
|
+
### `@spunto/build/catalogs` — what a project picker offers
|
|
126
|
+
|
|
127
|
+
`AVAILABLE_IMAGES`, `AVAILABLE_FEATURES`, `SUGGESTED_EXTENSIONS`. Data, but shared data: the two
|
|
128
|
+
hand-maintained copies had already drifted, one of them advertising Oh My Zsh in a feature the
|
|
129
|
+
image recipe explicitly disables.
|
|
130
|
+
|
|
131
|
+
Project *templates* are deliberately absent — one product's clone a starter repository and carry
|
|
132
|
+
onboarding-gallery presentation, the other's only configure an environment. They answer different
|
|
133
|
+
questions.
|
|
134
|
+
|
|
135
|
+
### `@spunto/build/types` — the shapes both schemas point at
|
|
136
|
+
|
|
137
|
+
`ProjectFeature`, `Repository`, `SetupStatus`. Declared here so each product's ORM can point its
|
|
138
|
+
JSON columns at them with `$type<…>()`, instead of a generator having to import a database schema.
|
|
139
|
+
`SetupStatus` is what a shell script writes into a container and a control plane reads back minutes
|
|
140
|
+
later: changing it is a migration, not an edit.
|
|
141
|
+
|
|
142
|
+
### `@spunto/build/agent-stream` — reading an agent session as it happens
|
|
143
|
+
|
|
144
|
+
One adapter per harness dialect, one vocabulary out. A CLI's JSON output is an output format, not
|
|
145
|
+
an API; adapters normalise into a small set of event types so stored history doesn't date the first
|
|
146
|
+
time a vendor reshuffles a field.
|
|
147
|
+
|
|
148
|
+
Here for the reason that inverts the rest of the package: **nothing has forked this yet.** Putting
|
|
149
|
+
it in the shared package now costs nothing and means the second product never writes its own.
|
|
150
|
+
|
|
105
151
|
## The rule
|
|
106
152
|
|
|
107
153
|
A module belongs in this package only if it imports **no** ORM schema, **no** HTTP framework, **no**
|
|
@@ -117,10 +163,11 @@ installed on its behalf. Import `./spec` without zod and resolution fails loudly
|
|
|
117
163
|
right trade for not taxing every other entry point.
|
|
118
164
|
|
|
119
165
|
Concretely: **no platform I/O**. No database, no Docker socket, no WebSocket. This package produces
|
|
120
|
-
strings and parses strings
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
166
|
+
strings and parses strings. Its only network calls are outbound HTTP on the caller's behalf — a
|
|
167
|
+
public extension registry, and a model vendor's catalogue endpoint when you hand
|
|
168
|
+
`resolveContextWindow` a credential — and it holds no credential of its own. Configuration arrives
|
|
169
|
+
as function parameters, never read from the environment, so that one control plane can scope a
|
|
170
|
+
setting per organization and another per process without either shape leaking in here.
|
|
124
171
|
|
|
125
172
|
That rule is what makes the package testable, safe to import from a node agent as well as from an
|
|
126
173
|
API, and the reason it can't simply be folded into the design system: a script generator needs
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@spunto/build",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.4.1",
|
|
4
4
|
"description": "Spunto's shared Build engine \u2014 the devcontainer image protocol and VS Code extension registry clients, with no database, no HTTP framework and no UI.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -31,22 +31,39 @@
|
|
|
31
31
|
"types": "./src/steps/index.ts",
|
|
32
32
|
"import": "./src/steps/index.ts"
|
|
33
33
|
},
|
|
34
|
-
"./extensions": {
|
|
35
|
-
"types": "./src/extensions/index.ts",
|
|
36
|
-
"import": "./src/extensions/index.ts"
|
|
37
|
-
},
|
|
38
34
|
"./naming": {
|
|
39
35
|
"types": "./src/naming/index.ts",
|
|
40
36
|
"import": "./src/naming/index.ts"
|
|
41
37
|
},
|
|
38
|
+
"./types": {
|
|
39
|
+
"types": "./src/types/index.ts",
|
|
40
|
+
"import": "./src/types/index.ts"
|
|
41
|
+
},
|
|
42
|
+
"./catalogs": {
|
|
43
|
+
"types": "./src/catalogs/index.ts",
|
|
44
|
+
"import": "./src/catalogs/index.ts"
|
|
45
|
+
},
|
|
46
|
+
"./extensions": {
|
|
47
|
+
"types": "./src/extensions/index.ts",
|
|
48
|
+
"import": "./src/extensions/index.ts"
|
|
49
|
+
},
|
|
42
50
|
"./spec": {
|
|
43
51
|
"types": "./src/spec/index.ts",
|
|
44
52
|
"import": "./src/spec/index.ts"
|
|
53
|
+
},
|
|
54
|
+
"./script": {
|
|
55
|
+
"types": "./src/script/index.ts",
|
|
56
|
+
"import": "./src/script/index.ts"
|
|
57
|
+
},
|
|
58
|
+
"./agent-stream": {
|
|
59
|
+
"types": "./src/agent-stream/index.ts",
|
|
60
|
+
"import": "./src/agent-stream/index.ts"
|
|
45
61
|
}
|
|
46
62
|
},
|
|
47
63
|
"files": [
|
|
48
64
|
"src",
|
|
49
65
|
"!src/**/*.test.ts",
|
|
66
|
+
"!src/**/__snapshots__",
|
|
50
67
|
"README.md"
|
|
51
68
|
],
|
|
52
69
|
"scripts": {
|
|
@@ -0,0 +1,561 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Reading an agent session as it happens (RFC 0020, phase 1).
|
|
3
|
+
*
|
|
4
|
+
* A task's harness is launched with the prompt on stdin and nothing else agreed — that poverty is
|
|
5
|
+
* what makes `taskAgentCommand` able to hold Claude Code, Codex or a shell script. Streaming needs
|
|
6
|
+
* strictly more than that, so it is **opt-in per project** (`projects.taskAgentProtocol`) and the
|
|
7
|
+
* extra knowledge lives here, in one adapter per dialect, and nowhere else.
|
|
8
|
+
*
|
|
9
|
+
* Two rules the rest of the code depends on:
|
|
10
|
+
*
|
|
11
|
+
* 1. **Our vocabulary is the stored one.** A harness's JSON shape is an output format, not an
|
|
12
|
+
* API; storing it would date the history. Adapters translate into the small set below, and
|
|
13
|
+
* anything unmapped becomes `raw` rather than being dropped — same escape hatch as
|
|
14
|
+
* `tasks.meta`, which has already earned its keep twice.
|
|
15
|
+
* 2. **A malformed line is never fatal.** This parses the output of a third-party CLI on a read
|
|
16
|
+
* path. Garbage in a line costs that line, not the task.
|
|
17
|
+
*
|
|
18
|
+
* Pure functions only: no database, no worker, no clock beyond what a caller passes in. The
|
|
19
|
+
* ingestion loop lives in task-events.service.ts.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import { contextWindowOf } from "./model-window"
|
|
23
|
+
|
|
24
|
+
export const AGENT_PROTOCOLS = ["none", "claude-json", "claude-stream", "jsonl"] as const
|
|
25
|
+
export type AgentProtocol = (typeof AGENT_PROTOCOLS)[number]
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* The normalized vocabulary. Deliberately short — it has to be translatable *from* several
|
|
29
|
+
* harnesses, so anything specific to one of them does not belong in it.
|
|
30
|
+
*/
|
|
31
|
+
export const TASK_EVENT_TYPES = [
|
|
32
|
+
"session.started", // { harness, model?, cwd?, tools? }
|
|
33
|
+
"session.title", // { title } — the name the harness gave the conversation
|
|
34
|
+
"message", // { role, text }
|
|
35
|
+
"thinking", // { text }
|
|
36
|
+
"tool.call", // { callId, name, input }
|
|
37
|
+
"tool.result", // { callId, isError, output }
|
|
38
|
+
"plan", // { steps: [{ text, status }] }
|
|
39
|
+
// { scope: "call" | "session", model?, inputTokens?, outputTokens?, cacheReadTokens?,
|
|
40
|
+
// cacheCreationTokens?, contextTokens?, costUsd?, turns?, durationMs? } — see `summarizeUsage`
|
|
41
|
+
"usage",
|
|
42
|
+
"session.ended", // { status, stopReason?, summary? }
|
|
43
|
+
"raw", // anything an adapter could not place
|
|
44
|
+
] as const
|
|
45
|
+
export type TaskEventType = (typeof TASK_EVENT_TYPES)[number]
|
|
46
|
+
|
|
47
|
+
export type NormalizedEvent = {
|
|
48
|
+
type: TaskEventType
|
|
49
|
+
payload: Record<string, unknown>
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Per-field ceiling on stored text. A single `tool.result` can be a 2 MB file read; the timeline
|
|
54
|
+
* needs enough of it to be readable, not all of it. `truncated` is set so the UI can say so
|
|
55
|
+
* instead of quietly lying about what the agent saw.
|
|
56
|
+
*/
|
|
57
|
+
const MAX_TEXT = 8_000
|
|
58
|
+
|
|
59
|
+
function clip(value: unknown): { text: string; truncated: boolean } {
|
|
60
|
+
const text =
|
|
61
|
+
typeof value === "string" ? value : value === undefined || value === null ? "" : safeStringify(value)
|
|
62
|
+
return text.length > MAX_TEXT ? { text: text.slice(0, MAX_TEXT), truncated: true } : { text, truncated: false }
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function safeStringify(value: unknown): string {
|
|
66
|
+
try {
|
|
67
|
+
return JSON.stringify(value) ?? ""
|
|
68
|
+
} catch {
|
|
69
|
+
return String(value)
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* The command actually run, given the project's harness and its protocol.
|
|
75
|
+
*
|
|
76
|
+
* The platform appends the flags rather than asking the project to, because getting them wrong is
|
|
77
|
+
* silent: a missing `--verbose` makes Claude Code refuse `stream-json` outright, and a missing
|
|
78
|
+
* `--output-format` produces prose that parses into nothing at all.
|
|
79
|
+
*
|
|
80
|
+
* Left alone when the command already names an output format — a project whose harness is a
|
|
81
|
+
* pipeline, or who wants different flags, keeps full control by saying so.
|
|
82
|
+
*/
|
|
83
|
+
export function applyProtocol(command: string, protocol: AgentProtocol): string {
|
|
84
|
+
if (protocol !== "claude-json" && protocol !== "claude-stream") return command
|
|
85
|
+
if (/--output-format/.test(command)) return command
|
|
86
|
+
const flags = ["--output-format stream-json --verbose"]
|
|
87
|
+
// Interactive mode. `--input-format stream-json` is what keeps the harness alive between
|
|
88
|
+
// turns — it reads instructions as a stream instead of exiting after one prompt — which is the
|
|
89
|
+
// precondition for both answering it mid-flight and interrupting it. Measured in a worker
|
|
90
|
+
// before it was built: two turns, one session id, and an interrupt the process survives.
|
|
91
|
+
if (protocol === "claude-stream") flags.push("--input-format stream-json")
|
|
92
|
+
return `${command} ${flags.join(" ")}`
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** Does this protocol keep the harness alive, listening on stdin, between turns? */
|
|
96
|
+
export function isInteractive(protocol: AgentProtocol): boolean {
|
|
97
|
+
return protocol === "claude-stream"
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* One turn of the conversation, as the harness expects it on stdin.
|
|
102
|
+
*
|
|
103
|
+
* Newline-terminated and emitted as a single write: under PIPE_BUF that is atomic, which is what
|
|
104
|
+
* stops two messages from interleaving into one unparseable line.
|
|
105
|
+
*/
|
|
106
|
+
export function userTurnLine(protocol: AgentProtocol, text: string): string {
|
|
107
|
+
if (protocol !== "claude-stream") return `${text}\n`
|
|
108
|
+
return `${JSON.stringify({ type: "user", message: { role: "user", content: [{ type: "text", text }] } })}\n`
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* Stop what the harness is doing, without killing it.
|
|
113
|
+
*
|
|
114
|
+
* Measured: Claude Code answers a `control_request`/`interrupt` with a `control_response`
|
|
115
|
+
* success, ends the turn with `result/error_during_execution`, and **stays alive** — so the next
|
|
116
|
+
* message is a normal turn rather than a resume.
|
|
117
|
+
*/
|
|
118
|
+
export function interruptLine(protocol: AgentProtocol, requestId: string): string | null {
|
|
119
|
+
if (protocol !== "claude-stream") return null
|
|
120
|
+
return `${JSON.stringify({ type: "control_request", request_id: requestId, request: { subtype: "interrupt" } })}\n`
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* A shell that prints the line where the harness wrote **the name it gave the conversation**, or
|
|
125
|
+
* null for a dialect that has no such thing.
|
|
126
|
+
*
|
|
127
|
+
* This exists because Claude Code names its own conversations — the title you see in the terminal
|
|
128
|
+
* and in `/resume` — but does **not** put that name on stdout: measured in a worker, a
|
|
129
|
+
* `--output-format stream-json` session emits `system/init`, `assistant`, `user` and `result`,
|
|
130
|
+
* and nothing else. The name is written to the session transcript, as one
|
|
131
|
+
* `{"type":"ai-title","aiTitle":…}` line per revision. So the only way to have it is to go and
|
|
132
|
+
* read it, once, out of band — which is why this returns a *shell* rather than a parser: it runs
|
|
133
|
+
* in the worker, where the transcript is.
|
|
134
|
+
*
|
|
135
|
+
* Deliberately not a `--name` handed to the harness (the flag exists): a task created without a
|
|
136
|
+
* title has no name to impose, and imposing the prompt's first line would just be our own
|
|
137
|
+
* placeholder read back to us.
|
|
138
|
+
*
|
|
139
|
+
* Found by file name rather than by rebuilding the per-directory path Claude Code derives from
|
|
140
|
+
* the cwd: that encoding is an internal detail, `find` by session id is not. A harness run with
|
|
141
|
+
* `--no-session-persistence` writes no transcript, and then this finds nothing — which is a
|
|
142
|
+
* normal answer, not an error.
|
|
143
|
+
*
|
|
144
|
+
* `jsonl` gets no probe on purpose: a harness speaking our vocabulary says the name in band,
|
|
145
|
+
* `{"type":"session.title","title":"…"}`, and pays nothing for it.
|
|
146
|
+
*/
|
|
147
|
+
export function sessionTitleProbe(protocol: AgentProtocol, sessionId: string): string | null {
|
|
148
|
+
if (!readsTitleOutOfBand(protocol)) return null
|
|
149
|
+
// The id comes from the harness's own output, and it is about to be interpolated into a shell
|
|
150
|
+
// run inside the worker. Nothing outside this alphabet has any business in a session id, and
|
|
151
|
+
// refusing is cheaper than trusting a CLI's field to be what it was yesterday. It also means
|
|
152
|
+
// the single quotes below are enough: the set excludes the quote itself.
|
|
153
|
+
if (!/^[A-Za-z0-9._-]{1,128}$/.test(sessionId)) return null
|
|
154
|
+
return [
|
|
155
|
+
`root="\${CLAUDE_CONFIG_DIR:-$HOME/.claude}/projects"`,
|
|
156
|
+
`f=$(find "$root" -name '${sessionId}.jsonl' -type f 2>/dev/null | head -1)`,
|
|
157
|
+
`[ -n "$f" ] || exit 0`,
|
|
158
|
+
// The last few candidates, not just the last one: the pattern is a substring match on a file
|
|
159
|
+
// the agent itself can write into (a tool result quoting these very lines would match), so
|
|
160
|
+
// the reader below takes the last line that is actually a title record.
|
|
161
|
+
`grep -a '"type":"ai-title"' "$f" | tail -5`,
|
|
162
|
+
].join("\n")
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* Does this dialect's name have to be fetched, rather than read off the stream? Asked *before* a
|
|
167
|
+
* session id is looked up, so a harness that says its name in band pays nothing for the question.
|
|
168
|
+
*/
|
|
169
|
+
export function readsTitleOutOfBand(protocol: AgentProtocol): boolean {
|
|
170
|
+
return protocol === "claude-json" || protocol === "claude-stream"
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
/** Longest name we keep. The column allows 200; a harness that writes an essay gets cut. */
|
|
174
|
+
const MAX_TITLE = 200
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* The name, out of what the probe printed. Newest first, and the first line that is really a
|
|
178
|
+
* title record wins — see the `tail -5` above.
|
|
179
|
+
*/
|
|
180
|
+
export function parseSessionTitle(protocol: AgentProtocol, output: string): { title: string; source: string } | null {
|
|
181
|
+
if (!readsTitleOutOfBand(protocol)) return null
|
|
182
|
+
const lines = output.split("\n").map((l) => l.trim()).filter(Boolean)
|
|
183
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
184
|
+
let parsed: unknown
|
|
185
|
+
try {
|
|
186
|
+
parsed = JSON.parse(lines[i])
|
|
187
|
+
} catch {
|
|
188
|
+
continue
|
|
189
|
+
}
|
|
190
|
+
if (!parsed || typeof parsed !== "object") continue
|
|
191
|
+
const obj = parsed as Record<string, unknown>
|
|
192
|
+
if (obj.type !== "ai-title") continue
|
|
193
|
+
const title = cleanTitle(obj.aiTitle)
|
|
194
|
+
if (title) return { title, source: lines[i] }
|
|
195
|
+
}
|
|
196
|
+
return null
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/** One line, no runs of whitespace, bounded — a title goes in a list row and in a page header. */
|
|
200
|
+
export function cleanTitle(value: unknown): string | null {
|
|
201
|
+
if (typeof value !== "string") return null
|
|
202
|
+
const title = value.replace(/\s+/g, " ").trim().slice(0, MAX_TITLE).trim()
|
|
203
|
+
return title || null
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/**
|
|
207
|
+
* Parse one line of a harness's stream into zero or more normalized events.
|
|
208
|
+
*
|
|
209
|
+
* Zero is a normal answer: a blank line, a line of prose from a harness that also logs to stdout,
|
|
210
|
+
* or an event type we deliberately have no use for.
|
|
211
|
+
*/
|
|
212
|
+
export function parseStreamLine(protocol: AgentProtocol, line: string): NormalizedEvent[] {
|
|
213
|
+
const trimmed = line.trim()
|
|
214
|
+
if (!trimmed) return []
|
|
215
|
+
let parsed: unknown
|
|
216
|
+
try {
|
|
217
|
+
parsed = JSON.parse(trimmed)
|
|
218
|
+
} catch {
|
|
219
|
+
// Not JSON. A harness that also writes human lines to stdout is common enough (a banner, a
|
|
220
|
+
// warning) that this must not be an error — but it must not be invisible either.
|
|
221
|
+
return [{ type: "raw", payload: { text: clip(trimmed).text } }]
|
|
222
|
+
}
|
|
223
|
+
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return []
|
|
224
|
+
const obj = parsed as Record<string, unknown>
|
|
225
|
+
return protocol === "claude-json" || protocol === "claude-stream" ? adaptClaudeJson(obj) : adaptPassthrough(obj)
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/**
|
|
229
|
+
* `jsonl` — the project's harness already speaks our vocabulary. The escape hatch that means a
|
|
230
|
+
* home-made agent has nothing to learn beyond one page of documentation, and the thing that keeps
|
|
231
|
+
* the vocabulary honest: it is a format someone can write to, not just one we read from.
|
|
232
|
+
*/
|
|
233
|
+
function adaptPassthrough(obj: Record<string, unknown>): NormalizedEvent[] {
|
|
234
|
+
const type = typeof obj.type === "string" ? obj.type : null
|
|
235
|
+
if (!type) return [{ type: "raw", payload: { json: obj } }]
|
|
236
|
+
if (!(TASK_EVENT_TYPES as readonly string[]).includes(type)) {
|
|
237
|
+
return [{ type: "raw", payload: { json: obj } }]
|
|
238
|
+
}
|
|
239
|
+
const { type: _t, ...payload } = obj
|
|
240
|
+
return [{ type: type as TaskEventType, payload }]
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
/**
|
|
244
|
+
* `claude-json` — Claude Code's `--output-format stream-json --verbose`.
|
|
245
|
+
*
|
|
246
|
+
* Written against output measured in a real worker, not against a spec: the shapes below are
|
|
247
|
+
* what the CLI emitted, and `task-events.service.test.ts` replays recorded samples so a change in
|
|
248
|
+
* that output shows up as a failing test rather than as an empty timeline.
|
|
249
|
+
*/
|
|
250
|
+
function adaptClaudeJson(obj: Record<string, unknown>): NormalizedEvent[] {
|
|
251
|
+
const type = obj.type
|
|
252
|
+
|
|
253
|
+
if (type === "system" && obj.subtype === "init") {
|
|
254
|
+
const model = typeof obj.model === "string" ? obj.model : null
|
|
255
|
+
return [
|
|
256
|
+
{
|
|
257
|
+
type: "session.started",
|
|
258
|
+
payload: {
|
|
259
|
+
harness: "claude-code",
|
|
260
|
+
model,
|
|
261
|
+
// Read here and nowhere else: this is the only line that spells the model the way the
|
|
262
|
+
// *session* was opened (`claude-opus-5[1m]`), suffix included — the assistant messages
|
|
263
|
+
// that follow carry the plain API id. From the catalogue only, so this stays a pure
|
|
264
|
+
// function; the ingestion loop replaces it with the model's own answer when the task
|
|
265
|
+
// has a key to ask with (`resolveContextWindow`).
|
|
266
|
+
contextWindow: contextWindowOf(model),
|
|
267
|
+
cwd: typeof obj.cwd === "string" ? obj.cwd : null,
|
|
268
|
+
sessionId: typeof obj.session_id === "string" ? obj.session_id : null,
|
|
269
|
+
tools: Array.isArray(obj.tools) ? obj.tools.filter((t) => typeof t === "string") : [],
|
|
270
|
+
},
|
|
271
|
+
},
|
|
272
|
+
]
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
if (type === "assistant" || type === "user") {
|
|
276
|
+
const message = obj.message as Record<string, unknown> | undefined
|
|
277
|
+
const content = Array.isArray(message?.content) ? message!.content : []
|
|
278
|
+
const role = type === "assistant" ? "assistant" : "user"
|
|
279
|
+
const events: NormalizedEvent[] = []
|
|
280
|
+
for (const rawBlock of content) {
|
|
281
|
+
if (!rawBlock || typeof rawBlock !== "object") continue
|
|
282
|
+
const block = rawBlock as Record<string, unknown>
|
|
283
|
+
switch (block.type) {
|
|
284
|
+
case "text": {
|
|
285
|
+
const { text, truncated } = clip(block.text)
|
|
286
|
+
if (text) events.push({ type: "message", payload: { role, text, truncated } })
|
|
287
|
+
break
|
|
288
|
+
}
|
|
289
|
+
case "thinking": {
|
|
290
|
+
const { text, truncated } = clip(block.thinking)
|
|
291
|
+
if (text) events.push({ type: "thinking", payload: { text, truncated } })
|
|
292
|
+
break
|
|
293
|
+
}
|
|
294
|
+
case "tool_use": {
|
|
295
|
+
const name = typeof block.name === "string" ? block.name : "tool"
|
|
296
|
+
const { text: input, truncated } = clip(block.input)
|
|
297
|
+
events.push({
|
|
298
|
+
type: "tool.call",
|
|
299
|
+
payload: { callId: typeof block.id === "string" ? block.id : null, name, input, truncated },
|
|
300
|
+
})
|
|
301
|
+
// A todo list is a plan, and a plan is the single most useful thing to show while a
|
|
302
|
+
// session runs. Emitted *alongside* the tool call rather than instead of it: the call
|
|
303
|
+
// is what happened, the plan is what it means.
|
|
304
|
+
const plan = todoPlan(block.input)
|
|
305
|
+
if (plan) events.push(plan)
|
|
306
|
+
break
|
|
307
|
+
}
|
|
308
|
+
case "tool_result": {
|
|
309
|
+
const { text: output, truncated } = clip(
|
|
310
|
+
typeof block.content === "string" ? block.content : block.content ?? "",
|
|
311
|
+
)
|
|
312
|
+
events.push({
|
|
313
|
+
type: "tool.result",
|
|
314
|
+
payload: {
|
|
315
|
+
callId: typeof block.tool_use_id === "string" ? block.tool_use_id : null,
|
|
316
|
+
isError: block.is_error === true,
|
|
317
|
+
output,
|
|
318
|
+
truncated,
|
|
319
|
+
},
|
|
320
|
+
})
|
|
321
|
+
break
|
|
322
|
+
}
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
// What the call that produced this message cost, and how full the context was when it ran.
|
|
326
|
+
// Only assistant messages carry it — a `user` line is tool results being handed back.
|
|
327
|
+
if (type === "assistant") {
|
|
328
|
+
const measured = callUsage(message)
|
|
329
|
+
if (measured) events.push(measured)
|
|
330
|
+
}
|
|
331
|
+
return events
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
if (type === "result") {
|
|
335
|
+
const usage = (obj.usage ?? {}) as Record<string, unknown>
|
|
336
|
+
const events: NormalizedEvent[] = [
|
|
337
|
+
{
|
|
338
|
+
type: "usage",
|
|
339
|
+
payload: {
|
|
340
|
+
// The harness's own running total, **cumulative over the whole process** — with an
|
|
341
|
+
// interactive protocol a `result` lands at the end of every turn and each one restates
|
|
342
|
+
// the session so far. Summing them would bill the first turn once per turn that
|
|
343
|
+
// followed; the latest one is the answer. (Claude Code's own SDK reader does the same:
|
|
344
|
+
// it overwrites `total_cost_usd`/`num_turns` on each result rather than adding.)
|
|
345
|
+
scope: "session",
|
|
346
|
+
inputTokens: num(usage.input_tokens),
|
|
347
|
+
outputTokens: num(usage.output_tokens),
|
|
348
|
+
cacheReadTokens: num(usage.cache_read_input_tokens),
|
|
349
|
+
cacheCreationTokens: num(usage.cache_creation_input_tokens),
|
|
350
|
+
costUsd: num(obj.total_cost_usd),
|
|
351
|
+
turns: num(obj.num_turns),
|
|
352
|
+
durationMs: num(obj.duration_ms),
|
|
353
|
+
},
|
|
354
|
+
},
|
|
355
|
+
{
|
|
356
|
+
type: "session.ended",
|
|
357
|
+
payload: {
|
|
358
|
+
status: obj.subtype === "success" ? "succeeded" : "failed",
|
|
359
|
+
stopReason: typeof obj.stop_reason === "string" ? obj.stop_reason : null,
|
|
360
|
+
summary: clip(obj.result).text,
|
|
361
|
+
},
|
|
362
|
+
},
|
|
363
|
+
]
|
|
364
|
+
return events
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
// Known-but-uninteresting (`rate_limit_event`, `system/thinking_tokens`) and anything the CLI
|
|
368
|
+
// grows next land here rather than being dropped: an unknown event is information about the
|
|
369
|
+
// harness, and the UI keeps it behind a toggle.
|
|
370
|
+
return [{ type: "raw", payload: { json: obj } }]
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
function num(value: unknown): number | null {
|
|
374
|
+
return typeof value === "number" && Number.isFinite(value) ? value : null
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
/**
|
|
378
|
+
* What **one model call** consumed, read off an assistant message.
|
|
379
|
+
*
|
|
380
|
+
* Two different questions are answered by these numbers, and only the first is obvious:
|
|
381
|
+
*
|
|
382
|
+
* - added up over the session, they are what it has spent in tokens — and unlike the `result`
|
|
383
|
+
* line, they exist *during* a turn, which is when someone is watching;
|
|
384
|
+
* - the latest one alone is the **context**: `input + cache_read + cache_creation` is the prompt
|
|
385
|
+
* the harness just sent, so it is how full the window is right now. Same formula Claude Code
|
|
386
|
+
* uses for its own context meter.
|
|
387
|
+
*
|
|
388
|
+
* Skipped when every counter is zero: an API error makes the CLI emit a synthetic assistant
|
|
389
|
+
* message with an all-zero usage, and drawing a context that just fell to zero would be a lie
|
|
390
|
+
* about the session rather than a fact about the error.
|
|
391
|
+
*/
|
|
392
|
+
function callUsage(message: Record<string, unknown> | undefined): NormalizedEvent | null {
|
|
393
|
+
const usage = message?.usage
|
|
394
|
+
if (!usage || typeof usage !== "object") return null
|
|
395
|
+
const u = usage as Record<string, unknown>
|
|
396
|
+
const inputTokens = num(u.input_tokens) ?? 0
|
|
397
|
+
const outputTokens = num(u.output_tokens) ?? 0
|
|
398
|
+
const cacheReadTokens = num(u.cache_read_input_tokens) ?? 0
|
|
399
|
+
const cacheCreationTokens = num(u.cache_creation_input_tokens) ?? 0
|
|
400
|
+
if (inputTokens + outputTokens + cacheReadTokens + cacheCreationTokens === 0) return null
|
|
401
|
+
// `<synthetic>` is what the CLI puts there when no model ran at all.
|
|
402
|
+
const model = typeof message?.model === "string" && !message.model.startsWith("<") ? message.model : null
|
|
403
|
+
return {
|
|
404
|
+
type: "usage",
|
|
405
|
+
payload: {
|
|
406
|
+
scope: "call",
|
|
407
|
+
model,
|
|
408
|
+
inputTokens,
|
|
409
|
+
outputTokens,
|
|
410
|
+
cacheReadTokens,
|
|
411
|
+
cacheCreationTokens,
|
|
412
|
+
contextTokens: inputTokens + cacheReadTokens + cacheCreationTokens,
|
|
413
|
+
},
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
/** What a session has consumed so far — the recap the cockpit puts beside the timeline. */
|
|
418
|
+
export type SessionUsage = {
|
|
419
|
+
/** The model that ran, as the harness named it. */
|
|
420
|
+
model: string | null
|
|
421
|
+
/** Tokens in the prompt of the **latest** call: how full the window is right now. */
|
|
422
|
+
contextTokens: number | null
|
|
423
|
+
/** What that window holds, when it can be told from the model. */
|
|
424
|
+
contextWindow: number | null
|
|
425
|
+
inputTokens: number | null
|
|
426
|
+
outputTokens: number | null
|
|
427
|
+
cacheReadTokens: number | null
|
|
428
|
+
cacheCreationTokens: number | null
|
|
429
|
+
/** Only the harness can price a call, so this is always its own number, never ours. */
|
|
430
|
+
costUsd: number | null
|
|
431
|
+
turns: number | null
|
|
432
|
+
durationMs: number | null
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
/** The measurements `summarizeUsage` reduces, however the caller got hold of them. */
|
|
436
|
+
export type UsageInput = {
|
|
437
|
+
/** Summed over every `scope: "call"` event of the task. */
|
|
438
|
+
calls: {
|
|
439
|
+
count: number
|
|
440
|
+
inputTokens: number
|
|
441
|
+
outputTokens: number
|
|
442
|
+
cacheReadTokens: number
|
|
443
|
+
cacheCreationTokens: number
|
|
444
|
+
}
|
|
445
|
+
/** The most recent `scope: "call"` payload — the one that carries the live context. */
|
|
446
|
+
latestCall: Record<string, unknown> | null
|
|
447
|
+
/** The most recent `scope: "session"` payload — the harness's running total. */
|
|
448
|
+
latestSession: Record<string, unknown> | null
|
|
449
|
+
/** The most recent `session.started` payload, for the model and the window it implies. */
|
|
450
|
+
started: Record<string, unknown> | null
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
/**
|
|
454
|
+
* Fold a task's usage events into the one recap a reader wants.
|
|
455
|
+
*
|
|
456
|
+
* Which number comes from where is the whole content of this function:
|
|
457
|
+
*
|
|
458
|
+
* - **tokens** are summed from the per-call events rather than taken from the harness's total,
|
|
459
|
+
* because a turn in flight has not produced a total yet — and "0 tokens" under a session that
|
|
460
|
+
* has been working for ten minutes is the one answer that is never true. The harness's total
|
|
461
|
+
* is the fallback for a dialect that only reports at the end.
|
|
462
|
+
* - **money, turns and duration** can only come from the harness: pricing a call needs a rate
|
|
463
|
+
* card, and a platform that made one up would be quoting a number it cannot stand behind.
|
|
464
|
+
* - **context** is the latest call's prompt, against the window the *session* was opened with.
|
|
465
|
+
*
|
|
466
|
+
* `null` when a session has emitted no usage at all — nothing measured is not zero measured, and
|
|
467
|
+
* the difference is visible: the panel stays away instead of claiming a free session.
|
|
468
|
+
*/
|
|
469
|
+
export function summarizeUsage(input: UsageInput): SessionUsage | null {
|
|
470
|
+
const { calls, latestCall, latestSession, started } = input
|
|
471
|
+
if (calls.count === 0 && !latestSession) return null
|
|
472
|
+
const measured = calls.count > 0
|
|
473
|
+
const fallback = (key: string): number | null => num(latestSession?.[key])
|
|
474
|
+
return {
|
|
475
|
+
model:
|
|
476
|
+
(typeof latestCall?.model === "string" ? latestCall.model : null) ??
|
|
477
|
+
(typeof started?.model === "string" ? started.model : null),
|
|
478
|
+
contextTokens: num(latestCall?.contextTokens),
|
|
479
|
+
// Read off the line that opened the session; older events predate the field, so the model
|
|
480
|
+
// string is re-read as a fallback rather than leaving the meter blank on a live task.
|
|
481
|
+
contextWindow: num(started?.contextWindow) ?? contextWindowOf(started?.model as string | undefined),
|
|
482
|
+
inputTokens: measured ? calls.inputTokens : fallback("inputTokens"),
|
|
483
|
+
outputTokens: measured ? calls.outputTokens : fallback("outputTokens"),
|
|
484
|
+
cacheReadTokens: measured ? calls.cacheReadTokens : fallback("cacheReadTokens"),
|
|
485
|
+
cacheCreationTokens: measured ? calls.cacheCreationTokens : fallback("cacheCreationTokens"),
|
|
486
|
+
costUsd: fallback("costUsd"),
|
|
487
|
+
turns: fallback("turns"),
|
|
488
|
+
durationMs: fallback("durationMs"),
|
|
489
|
+
}
|
|
490
|
+
}
|
|
491
|
+
|
|
492
|
+
/** Claude Code's TodoWrite input, as a `plan` — `{ todos: [{ content, status }] }`. */
|
|
493
|
+
function todoPlan(input: unknown): NormalizedEvent | null {
|
|
494
|
+
if (!input || typeof input !== "object") return null
|
|
495
|
+
const todos = (input as Record<string, unknown>).todos
|
|
496
|
+
if (!Array.isArray(todos) || todos.length === 0) return null
|
|
497
|
+
const steps = todos
|
|
498
|
+
.filter((t): t is Record<string, unknown> => !!t && typeof t === "object")
|
|
499
|
+
.map((t) => ({
|
|
500
|
+
text: clip(t.content ?? t.activeForm ?? "").text,
|
|
501
|
+
status: typeof t.status === "string" ? t.status : "pending",
|
|
502
|
+
}))
|
|
503
|
+
.filter((s) => s.text)
|
|
504
|
+
return steps.length > 0 ? { type: "plan", payload: { steps } } : null
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
/** An event and the harness line it came from. One line can yield several. */
|
|
508
|
+
export type SourcedEvent = NormalizedEvent & { source: string }
|
|
509
|
+
|
|
510
|
+
/**
|
|
511
|
+
* Parse whole lines, keeping each event attached to the line it was derived from.
|
|
512
|
+
*
|
|
513
|
+
* The pairing lives here, next to the adapters, because it is the thing that makes normalization
|
|
514
|
+
* reversible: `type`/`payload` is our reading of the line, `source` is the line. Get the reading
|
|
515
|
+
* wrong and the line is still there to read again.
|
|
516
|
+
*
|
|
517
|
+
* `budget` bounds how many events one pass may produce, so a runaway session cannot blow past the
|
|
518
|
+
* per-task ceiling in a single chunk. An adapter that throws costs its line, not the pass.
|
|
519
|
+
*/
|
|
520
|
+
export function parseLines(protocol: AgentProtocol, lines: string[], budget: number): SourcedEvent[] {
|
|
521
|
+
const out: SourcedEvent[] = []
|
|
522
|
+
for (const line of lines) {
|
|
523
|
+
if (out.length >= budget) break
|
|
524
|
+
let events: NormalizedEvent[]
|
|
525
|
+
try {
|
|
526
|
+
events = parseStreamLine(protocol, line)
|
|
527
|
+
} catch (err) {
|
|
528
|
+
console.warn(`[tasks] agent stream adapter threw: ${err}`)
|
|
529
|
+
continue
|
|
530
|
+
}
|
|
531
|
+
const source = line.trim()
|
|
532
|
+
for (const event of events) {
|
|
533
|
+
if (out.length >= budget) break
|
|
534
|
+
out.push({ ...event, source })
|
|
535
|
+
}
|
|
536
|
+
}
|
|
537
|
+
return out
|
|
538
|
+
}
|
|
539
|
+
|
|
540
|
+
/**
|
|
541
|
+
* Split a chunk of bytes into whole lines, and say how many bytes that consumed.
|
|
542
|
+
*
|
|
543
|
+
* The cursor only ever advances past a **complete** line: a read that lands mid-line leaves the
|
|
544
|
+
* tail for the next one. Without this, a 3 KB JSON event straddling two reads would be parsed as
|
|
545
|
+
* two halves — twice — and neither would be valid JSON.
|
|
546
|
+
*/
|
|
547
|
+
export function splitCompleteLines(chunk: Uint8Array): { lines: string[]; consumed: number } {
|
|
548
|
+
// `Uint8Array` rather than Node's `Buffer`: a Buffer *is* one, so every caller compiles
|
|
549
|
+
// unchanged, and the package stays runnable anywhere — see the rule in the README. That costs
|
|
550
|
+
// `lastIndexOf`, which Uint8Array does not have, hence the backwards scan.
|
|
551
|
+
let lastNewline = -1
|
|
552
|
+
for (let i = chunk.length - 1; i >= 0; i--) {
|
|
553
|
+
if (chunk[i] === 0x0a) {
|
|
554
|
+
lastNewline = i
|
|
555
|
+
break
|
|
556
|
+
}
|
|
557
|
+
}
|
|
558
|
+
if (lastNewline === -1) return { lines: [], consumed: 0 }
|
|
559
|
+
const consumed = lastNewline + 1
|
|
560
|
+
return { lines: new TextDecoder().decode(chunk.subarray(0, consumed)).split("\n"), consumed }
|
|
561
|
+
}
|