@zvada/agent-server 0.2.2 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +317 -0
- package/README.md +20 -4
- package/docs/consuming.md +269 -0
- package/docs/deploy.md +80 -0
- package/docs/harnesses.md +64 -0
- package/docs/rfds/0001-deterministic-echo-ids.md +44 -0
- package/package.json +23 -3
- package/src/client/client.ts +143 -50
- package/src/core/agents/acp/acp-agent.ts +9 -0
- package/src/core/agents/acp/mappings.ts +3 -3
- package/src/core/agents/base.ts +9 -1
- package/src/core/agents/claude-code/adapter.ts +116 -26
- package/src/core/agents/claude-code/claude-agent.ts +31 -3
- package/src/core/agents/claude-code/generator-session.ts +16 -5
- package/src/core/agents/claude-code/options.ts +13 -3
- package/src/core/agents/claude-code/session-manager.ts +9 -4
- package/src/core/agents/codex-app-server/codex-app-server-agent.ts +17 -3
- package/src/core/agents/codex-sdk/codex-sdk-agent.ts +3 -3
- package/src/core/agents/types.ts +1 -1
- package/src/core/diagnostics.ts +59 -0
- package/src/core/index.ts +3 -1
- package/src/core/presets.ts +15 -2
- package/src/core/provision/pins.ts +5 -1
- package/src/core/proxy/anthropic-proxy.ts +12 -0
- package/src/core/runtime/agent-runtime.ts +66 -18
- package/src/core/runtime/event-processor.ts +51 -26
- package/src/protocol/config.ts +8 -6
- package/src/{core/agents/error-classifier.ts → protocol/errors.ts} +33 -7
- package/src/protocol/factories.ts +106 -10
- package/src/protocol/guards.ts +53 -0
- package/src/protocol/index.ts +10 -0
- package/src/protocol/lifecycle.ts +289 -112
- package/src/protocol/meta.ts +14 -0
- package/src/protocol/part-input.ts +56 -7
- package/src/protocol/parts.ts +125 -10
- package/src/protocol/reduce.ts +749 -0
- package/src/protocol/selectors.ts +162 -0
- package/src/protocol/seq-cursor.ts +87 -0
- package/src/protocol/stop-reasons.ts +45 -0
- package/src/protocol/time.ts +23 -0
- package/src/protocol/tokens.ts +23 -0
- package/src/protocol/tool-state.ts +85 -25
- package/src/protocol/verify.ts +440 -0
- package/src/protocol/vocabulary.ts +18 -0
- package/src/protocol/wire.ts +81 -13
- package/src/server/acp/binding.ts +23 -2
- package/src/server/acp/translate.ts +51 -14
- package/src/server/agent-server.ts +85 -6
- package/AGENTS.md +0 -21
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
// Read-only projections of a folded `ConversationState`.
|
|
2
|
+
//
|
|
3
|
+
// `reduceConversation` says what the stream MEANS; these say what a UI or an
|
|
4
|
+
// API layer then ASKS of it. Every one of them was written independently — and
|
|
5
|
+
// differently — in each product before it landed here: the turn grouping (with
|
|
6
|
+
// the guard that keeps a finished turn out of streaming UI), the subagent
|
|
7
|
+
// nesting map, the "what is the agent doing right now" indicator, and the
|
|
8
|
+
// tool-result text every renderer falls back to.
|
|
9
|
+
//
|
|
10
|
+
// Pure, allocation-light, and derived — never cached in state, never a source
|
|
11
|
+
// of truth. Call them per render.
|
|
12
|
+
|
|
13
|
+
import type { ConversationMessage, ConversationState, TimelineEntry } from "./reduce.ts";
|
|
14
|
+
import type { ToolState } from "./tool-state.ts";
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* One rendered turn card: a run of consecutive timeline entries that belong to
|
|
18
|
+
* the same turn and the same speaker.
|
|
19
|
+
*/
|
|
20
|
+
export interface ConversationTurnGroup {
|
|
21
|
+
turnId: string;
|
|
22
|
+
role: "user" | "assistant";
|
|
23
|
+
/** Messages (and compactions) in stream order. */
|
|
24
|
+
entries: TimelineEntry[];
|
|
25
|
+
/** True for the final group only — see `groupIntoTurns`. */
|
|
26
|
+
isLatest: boolean;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Group the timeline into turn cards: consecutive entries break into a new
|
|
31
|
+
* group when the speaker changes or the turn changes. (A group carries one
|
|
32
|
+
* `turnId`, so two back-to-back user echoes from different turns stay
|
|
33
|
+
* separate even though the role is unchanged.) Compactions are engine-side
|
|
34
|
+
* content and group as `assistant` within their turn — a divider that renders
|
|
35
|
+
* inside the run it interrupted, which is where it happened.
|
|
36
|
+
*
|
|
37
|
+
* `isLatest` is TRUE FOR THE FINAL GROUP ONLY, and only while nothing follows
|
|
38
|
+
* it. That guard is the whole point: a consumer renders the latest group in
|
|
39
|
+
* streaming mode (expanded, live cursor), and the moment the next turn's user
|
|
40
|
+
* echo lands, the previous group MUST lose the flag. Without it, the gap
|
|
41
|
+
* between "user submits" and "first assistant part arrives" re-opens the
|
|
42
|
+
* finished turn in streaming mode — the completed answer visibly reverts to
|
|
43
|
+
* "working". Combine with your own busy flag (`isLatest && role ===
|
|
44
|
+
* "assistant" && isWorking`), never with position alone.
|
|
45
|
+
*/
|
|
46
|
+
export function groupIntoTurns(state: ConversationState): ConversationTurnGroup[] {
|
|
47
|
+
const groups: ConversationTurnGroup[] = [];
|
|
48
|
+
let current: ConversationTurnGroup | undefined;
|
|
49
|
+
for (const entry of state.timeline) {
|
|
50
|
+
const role = entry.kind === "message" ? entry.role : "assistant";
|
|
51
|
+
if (!current || current.turnId !== entry.turnId || current.role !== role) {
|
|
52
|
+
current = { turnId: entry.turnId, role, entries: [], isLatest: false };
|
|
53
|
+
groups.push(current);
|
|
54
|
+
}
|
|
55
|
+
current.entries.push(entry);
|
|
56
|
+
}
|
|
57
|
+
const last = groups[groups.length - 1];
|
|
58
|
+
if (last) last.isLatest = true;
|
|
59
|
+
return groups;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Subagent output, keyed by the `toolCallId` of the tool that spawned it.
|
|
64
|
+
*
|
|
65
|
+
* A message with `parentToolCallId` is NOT top-level model output — rendering
|
|
66
|
+
* it in the main flow double-prints the subagent's work next to the tool card
|
|
67
|
+
* that owns it. Consumers nest `subagentGroups(state).get(part.toolCallId)`
|
|
68
|
+
* under the tool part and filter parented messages out of the main timeline
|
|
69
|
+
* (`groupIntoTurns` keeps them, because "which turn did this belong to" is
|
|
70
|
+
* still true — the filtering is a rendering choice).
|
|
71
|
+
*/
|
|
72
|
+
export function subagentGroups(state: ConversationState): Map<string, ConversationMessage[]> {
|
|
73
|
+
const groups = new Map<string, ConversationMessage[]>();
|
|
74
|
+
for (const entry of state.timeline) {
|
|
75
|
+
if (entry.kind !== "message" || entry.parentToolCallId === undefined) continue;
|
|
76
|
+
const group = groups.get(entry.parentToolCallId);
|
|
77
|
+
if (group) group.push(entry);
|
|
78
|
+
else groups.set(entry.parentToolCallId, [entry]);
|
|
79
|
+
}
|
|
80
|
+
return groups;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/** What the agent is doing right now, for a live activity indicator. */
|
|
84
|
+
export type AgentActivity = "thinking" | "generating" | "tool_running" | "tool_failed" | "idle";
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Derive the live activity from the LAST part of the LAST top-level assistant
|
|
88
|
+
* message of the LAST ACTIVE turn: reasoning → `thinking`, text →
|
|
89
|
+
* `generating`, a pending/in-progress tool → `tool_running`, a failed tool →
|
|
90
|
+
* `tool_failed`.
|
|
91
|
+
*
|
|
92
|
+
* Everything else is `idle` — no active turn, no assistant output yet, an
|
|
93
|
+
* unknown part type, or a part that says nothing about what happens next (a
|
|
94
|
+
* completed tool: the model is about to speak, but the stream has not said so
|
|
95
|
+
* yet). `idle` DURING an active turn means "between observable activities",
|
|
96
|
+
* not "finished": a consumer with its own busy flag renders its default
|
|
97
|
+
* working state there rather than nothing.
|
|
98
|
+
*
|
|
99
|
+
* Subagent messages (`parentToolCallId`) are skipped: the main thread's
|
|
100
|
+
* activity is the spawning tool call, which stays `tool_running` for as long
|
|
101
|
+
* as the subagent works — otherwise the indicator would flicker on the
|
|
102
|
+
* subagent's internal steps.
|
|
103
|
+
*/
|
|
104
|
+
export function agentActivity(state: ConversationState): AgentActivity {
|
|
105
|
+
let activeTurnId: string | undefined;
|
|
106
|
+
for (let i = state.turns.length - 1; i >= 0; i--) {
|
|
107
|
+
const turn = state.turns[i];
|
|
108
|
+
if (turn?.status === "active") {
|
|
109
|
+
activeTurnId = turn.turnId;
|
|
110
|
+
break;
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
if (activeTurnId === undefined) return "idle";
|
|
114
|
+
|
|
115
|
+
for (let i = state.timeline.length - 1; i >= 0; i--) {
|
|
116
|
+
const entry = state.timeline[i];
|
|
117
|
+
if (entry?.kind !== "message") continue;
|
|
118
|
+
if (entry.turnId !== activeTurnId) continue;
|
|
119
|
+
if (entry.role !== "assistant" || entry.parentToolCallId !== undefined) continue;
|
|
120
|
+
// ONLY the latest such message speaks for "now". Falling through to an
|
|
121
|
+
// older one would resurrect its last state — a finished message's text
|
|
122
|
+
// part reading as `generating` minutes after it closed.
|
|
123
|
+
const part = entry.parts[entry.parts.length - 1];
|
|
124
|
+
// A just-opened message with no parts yet: between activities.
|
|
125
|
+
if (!part) return "idle";
|
|
126
|
+
// Unknown part types carry a `raw` bag and nothing we can interpret.
|
|
127
|
+
if ("raw" in part) return "idle";
|
|
128
|
+
if (part.type === "tool") {
|
|
129
|
+
// Tools may settle AFTER their message ends (the contract's late
|
|
130
|
+
// re-statement), so tool activity outranks the message bracket.
|
|
131
|
+
const status = part.state.status;
|
|
132
|
+
if (status === "failed") return "tool_failed";
|
|
133
|
+
if (status === "pending" || status === "in_progress") return "tool_running";
|
|
134
|
+
return "idle";
|
|
135
|
+
}
|
|
136
|
+
// Text/reasoning only speak while their message is still open — an
|
|
137
|
+
// ended message's last words are history, not activity.
|
|
138
|
+
if (entry.endedAt !== undefined) return "idle";
|
|
139
|
+
if (part.type === "reasoning") return "thinking";
|
|
140
|
+
if (part.type === "text") return "generating";
|
|
141
|
+
return "idle";
|
|
142
|
+
}
|
|
143
|
+
return "idle";
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* The tool result as display text: a completed call's `output`, a failed
|
|
148
|
+
* call's `error`, and "" for everything that has not produced one yet
|
|
149
|
+
* (`pending`, `in_progress`) or never will (`cancelled`).
|
|
150
|
+
*
|
|
151
|
+
* The empty string is deliberate and load-bearing: a renderer that shows this
|
|
152
|
+
* unconditionally shows nothing for a call in flight, instead of the
|
|
153
|
+
* placeholder prose ("Running…", "Cancelled") each consumer invented
|
|
154
|
+
* differently. Structured output (`state.content`: diffs, images, terminals)
|
|
155
|
+
* is a richer view a renderer opts into — `output` stays the universal
|
|
156
|
+
* fallback, per the three-audience doctrine.
|
|
157
|
+
*/
|
|
158
|
+
export function toolResultText(state: ToolState): string {
|
|
159
|
+
if (state.status === "completed") return state.output;
|
|
160
|
+
if (state.status === "failed") return state.error;
|
|
161
|
+
return "";
|
|
162
|
+
}
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What one envelope's `seq` means for a session's cursor.
|
|
3
|
+
*
|
|
4
|
+
* - `deliver` — the expected next envelope; the cursor advanced to it.
|
|
5
|
+
* - `duplicate` — already delivered (a replay overlap, a resync); drop it.
|
|
6
|
+
* The cursor does NOT move.
|
|
7
|
+
* - `gap` — envelopes were missed; hold this one and heal (replay).
|
|
8
|
+
* The cursor does NOT move, so the healed events still land
|
|
9
|
+
* in order.
|
|
10
|
+
* - `reset` — `seq` restarted at 1 while the cursor was past it: the
|
|
11
|
+
* session log began again (a new server process behind a
|
|
12
|
+
* reconnecting transport). The cursor resets and DELIVERS
|
|
13
|
+
* this envelope — treating it as a duplicate silently drops
|
|
14
|
+
* the entire post-restart stream, which is exactly the bug
|
|
15
|
+
* this verdict exists to prevent. The caller must drop the
|
|
16
|
+
* state it was holding for the dead log (held-back envelopes,
|
|
17
|
+
* partial reconstruction); the fresh log is authoritative.
|
|
18
|
+
*
|
|
19
|
+
* The reset rule is a HEURISTIC with two documented blind spots — at the seq
|
|
20
|
+
* layer, a restart is indistinguishable from a redelivery in exactly these:
|
|
21
|
+
* a restart seen while the cursor sits at 1 (the fresh seq 1 reads as a
|
|
22
|
+
* duplicate), and a restart first contacted mid-stream (fresh seq k ≤ last
|
|
23
|
+
* reads as a duplicate). Callers with a second signal close them: the wire
|
|
24
|
+
* client does not guess at all — `initialize` returns a per-process
|
|
25
|
+
* `instanceId` and a reconnect re-handshake adopts the fresh logs outright;
|
|
26
|
+
* a consumer without a handshake (a browser folding forwarded envelopes)
|
|
27
|
+
* treats any backwards seq beyond the immediately-last frame as a
|
|
28
|
+
* replacement.
|
|
29
|
+
*/
|
|
30
|
+
export type SeqVerdict = "deliver" | "duplicate" | "gap" | "reset";
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Per-session `seq` bookkeeping — the whole of "did I already see this, did I
|
|
34
|
+
* miss something, did the server restart", with no transport, no schemas and
|
|
35
|
+
* no dependencies.
|
|
36
|
+
*
|
|
37
|
+
* This is the wire client's own discipline, hoisted so the consumers that
|
|
38
|
+
* cannot use the client (a backend folding envelopes forwarded over its own
|
|
39
|
+
* WebSocket, a Durable Object replaying a log) get the same answers instead of
|
|
40
|
+
* a fourth hand-rolled variant. `AgentServerClient` is implemented on it.
|
|
41
|
+
*/
|
|
42
|
+
export interface SeqCursor {
|
|
43
|
+
/** Classify `seq` for `sessionId`, advancing the cursor when it delivers. */
|
|
44
|
+
advance(sessionId: string, seq: number): SeqVerdict;
|
|
45
|
+
/** Forget the session entirely — the next `seq` starts a fresh stream. */
|
|
46
|
+
reset(sessionId: string): void;
|
|
47
|
+
/**
|
|
48
|
+
* Force the watermark (an evicted prefix accepted as lost, a snapshot
|
|
49
|
+
* restored from durable state). The next contiguous envelope is `seq + 1`.
|
|
50
|
+
*/
|
|
51
|
+
seek(sessionId: string, seq: number): void;
|
|
52
|
+
/** Highest delivered `seq` for the session; 0 when none is known. */
|
|
53
|
+
last(sessionId: string): number;
|
|
54
|
+
/** The sessions with a cursor — what to resync after a reconnect. */
|
|
55
|
+
sessions(): string[];
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export function createSeqCursor(): SeqCursor {
|
|
59
|
+
const lastSeq = new Map<string, number>();
|
|
60
|
+
return {
|
|
61
|
+
advance(sessionId, seq) {
|
|
62
|
+
const last = lastSeq.get(sessionId) ?? 0;
|
|
63
|
+
if (seq === 1 && last > 1) {
|
|
64
|
+
lastSeq.set(sessionId, seq);
|
|
65
|
+
return "reset";
|
|
66
|
+
}
|
|
67
|
+
if (seq <= last) return "duplicate";
|
|
68
|
+
if (seq === last + 1) {
|
|
69
|
+
lastSeq.set(sessionId, seq);
|
|
70
|
+
return "deliver";
|
|
71
|
+
}
|
|
72
|
+
return "gap";
|
|
73
|
+
},
|
|
74
|
+
reset(sessionId) {
|
|
75
|
+
lastSeq.delete(sessionId);
|
|
76
|
+
},
|
|
77
|
+
seek(sessionId, seq) {
|
|
78
|
+
lastSeq.set(sessionId, seq);
|
|
79
|
+
},
|
|
80
|
+
last(sessionId) {
|
|
81
|
+
return lastSeq.get(sessionId) ?? 0;
|
|
82
|
+
},
|
|
83
|
+
sessions() {
|
|
84
|
+
return [...lastSeq.keys()];
|
|
85
|
+
},
|
|
86
|
+
};
|
|
87
|
+
}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import type { StopReason } from "./lifecycle.ts";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* The stop reasons that mean "the turn finished the way it was supposed to".
|
|
5
|
+
* A subset of `STOP_REASONS` — everything except `error`.
|
|
6
|
+
*
|
|
7
|
+
* `cancelled` is CLEAN: the user asked for the turn to stop and it stopped.
|
|
8
|
+
* Products that treat it as a failure paint an error banner over the user's
|
|
9
|
+
* own action (all three copies of this set in the wild got this right, and
|
|
10
|
+
* two of them then drifted on `refusal` and `max_turn_requests`).
|
|
11
|
+
*
|
|
12
|
+
* Zod-free and dependency-free on purpose: this is a membership test, and a
|
|
13
|
+
* consumer branching on a stop reason should not have to load the schemas.
|
|
14
|
+
*/
|
|
15
|
+
export const CLEAN_STOP_REASONS = [
|
|
16
|
+
"end_turn",
|
|
17
|
+
"max_tokens",
|
|
18
|
+
"max_turn_requests",
|
|
19
|
+
"refusal",
|
|
20
|
+
"cancelled",
|
|
21
|
+
] as const;
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* True when the turn stopped for a reason this build knows to be clean.
|
|
25
|
+
*
|
|
26
|
+
* `StopReason` is an OPEN vocabulary (Law 3): a newer engine's value, or an
|
|
27
|
+
* adapter's `_`-prefixed extension, reaches an older consumer unchanged. Such
|
|
28
|
+
* a value is classified as NOT clean — the conservative direction. Showing a
|
|
29
|
+
* failure affordance for an outcome we cannot interpret is recoverable (the
|
|
30
|
+
* user retries); silently rendering an unknown outcome as success is not (the
|
|
31
|
+
* turn looks complete, the work is gone, nothing points at the cause).
|
|
32
|
+
*/
|
|
33
|
+
export function isCleanStop(reason: StopReason | undefined): boolean {
|
|
34
|
+
return (CLEAN_STOP_REASONS as readonly string[]).includes(reason as string);
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* The complement of `isCleanStop`: `error`, plus every stop reason this build
|
|
39
|
+
* does not know (see the conservative-classification note above). `undefined`
|
|
40
|
+
* — a turn with no stop reason at all — is a failure too: a turn that ended
|
|
41
|
+
* without saying why did not end well.
|
|
42
|
+
*/
|
|
43
|
+
export function isFailureStopReason(reason: StopReason | undefined): boolean {
|
|
44
|
+
return !isCleanStop(reason);
|
|
45
|
+
}
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Law 5 — every instant on this wire is epoch MILLISECONDS, as an integer.
|
|
5
|
+
* One schema, used by every `timestamp` and every nested `time.start`/
|
|
6
|
+
* `time.end`, so the boundary validation is real rather than prose: a bare
|
|
7
|
+
* `z.number()` accepts `-5`, `1.5` and `-0`, all of which round-trip silently
|
|
8
|
+
* and then sort, subtract and render wrong (a seconds-based stamp lands in
|
|
9
|
+
* 1970; `-0` breaks `Object.is` keyed caches).
|
|
10
|
+
*/
|
|
11
|
+
export const EpochMsSchema = z
|
|
12
|
+
.number()
|
|
13
|
+
.int()
|
|
14
|
+
.nonnegative()
|
|
15
|
+
// -0 passes `>= 0`; it is not a legal instant, and it survives JSON as `0`
|
|
16
|
+
// only by accident of the serializer.
|
|
17
|
+
.refine((value) => !Object.is(value, -0), { message: "epoch-ms must not be -0" });
|
|
18
|
+
export type EpochMs = z.infer<typeof EpochMsSchema>;
|
|
19
|
+
|
|
20
|
+
/** A start/end span in epoch ms (tool executions, thinking blocks). */
|
|
21
|
+
export const TimeSpanSchema = z.object({ start: EpochMsSchema, end: EpochMsSchema });
|
|
22
|
+
/** A span whose end is unknown while the thing is still running. */
|
|
23
|
+
export const OpenTimeSpanSchema = z.object({ start: EpochMsSchema, end: EpochMsSchema.optional() });
|
package/src/protocol/tokens.ts
CHANGED
|
@@ -4,6 +4,14 @@ import { z } from "zod";
|
|
|
4
4
|
* Normalized token accounting. Every harness reports usage differently
|
|
5
5
|
* (Claude: input/output/cache_creation/cache_read; Codex: input/output/
|
|
6
6
|
* reasoning/cached). Adapters fold those into this single shape.
|
|
7
|
+
*
|
|
8
|
+
* INVARIANTS (normative — so consumers never subtract):
|
|
9
|
+
* - `input` is NON-CACHED input tokens ONLY; billed input =
|
|
10
|
+
* `input + cache.read + cache.write` (the three are disjoint).
|
|
11
|
+
* - `reasoning` is a subset of `output` (output already includes it).
|
|
12
|
+
* - `writeEphemeral5m`/`writeEphemeral1h` are subsets of `cache.write`.
|
|
13
|
+
* Providers that fold cache hits into a total prompt count are unfolded once,
|
|
14
|
+
* at the adapter — never downstream.
|
|
7
15
|
*/
|
|
8
16
|
export const TokenUsageSchema = z.object({
|
|
9
17
|
input: z.number().default(0),
|
|
@@ -13,6 +21,10 @@ export const TokenUsageSchema = z.object({
|
|
|
13
21
|
.object({
|
|
14
22
|
read: z.number().default(0),
|
|
15
23
|
write: z.number().default(0),
|
|
24
|
+
/** Anthropic 5m-TTL cache-creation bucket, when reported. */
|
|
25
|
+
writeEphemeral5m: z.number().optional(),
|
|
26
|
+
/** Anthropic 1h-TTL cache-creation bucket, when reported. */
|
|
27
|
+
writeEphemeral1h: z.number().optional(),
|
|
16
28
|
})
|
|
17
29
|
.optional(),
|
|
18
30
|
});
|
|
@@ -26,6 +38,11 @@ export const DEFAULT_TOKEN_USAGE: TokenUsage = {
|
|
|
26
38
|
cache: { read: 0, write: 0 },
|
|
27
39
|
};
|
|
28
40
|
|
|
41
|
+
function addOptional(a: number | undefined, b: number | undefined): number | undefined {
|
|
42
|
+
if (a === undefined && b === undefined) return undefined;
|
|
43
|
+
return (a ?? 0) + (b ?? 0);
|
|
44
|
+
}
|
|
45
|
+
|
|
29
46
|
export function addTokenUsage(a: TokenUsage, b: TokenUsage): TokenUsage {
|
|
30
47
|
return {
|
|
31
48
|
input: a.input + b.input,
|
|
@@ -34,6 +51,12 @@ export function addTokenUsage(a: TokenUsage, b: TokenUsage): TokenUsage {
|
|
|
34
51
|
cache: {
|
|
35
52
|
read: (a.cache?.read ?? 0) + (b.cache?.read ?? 0),
|
|
36
53
|
write: (a.cache?.write ?? 0) + (b.cache?.write ?? 0),
|
|
54
|
+
...(addOptional(a.cache?.writeEphemeral5m, b.cache?.writeEphemeral5m) !== undefined
|
|
55
|
+
? { writeEphemeral5m: addOptional(a.cache?.writeEphemeral5m, b.cache?.writeEphemeral5m) }
|
|
56
|
+
: {}),
|
|
57
|
+
...(addOptional(a.cache?.writeEphemeral1h, b.cache?.writeEphemeral1h) !== undefined
|
|
58
|
+
? { writeEphemeral1h: addOptional(a.cache?.writeEphemeral1h, b.cache?.writeEphemeral1h) }
|
|
59
|
+
: {}),
|
|
37
60
|
},
|
|
38
61
|
};
|
|
39
62
|
}
|
|
@@ -1,19 +1,39 @@
|
|
|
1
1
|
import { z } from "zod";
|
|
2
|
+
import { MetaSchema } from "./meta.ts";
|
|
3
|
+
import { EpochMsSchema, TimeSpanSchema } from "./time.ts";
|
|
4
|
+
import { openVocabulary } from "./vocabulary.ts";
|
|
2
5
|
|
|
3
6
|
/**
|
|
4
7
|
* Runtime tool lifecycle: a tool call moves pending -> in_progress ->
|
|
5
|
-
* completed|failed (ACP
|
|
6
|
-
* (streaming) JSON input before it
|
|
7
|
-
* fully-parsed input object.
|
|
8
|
+
* completed | failed | cancelled (ACP ToolCallStatus verbatim, incl. v2's
|
|
9
|
+
* `cancelled`). `pending` carries the partial (streaming) JSON input before it
|
|
10
|
+
* parses; the other states carry the fully-parsed input object. `cancelled` is
|
|
11
|
+
* the terminal for tools in flight when a turn is cancelled — the reducer
|
|
12
|
+
* applies it implicitly at `turn.ended{stopReason:"cancelled"}`; producers MAY
|
|
13
|
+
* also emit the terminal snapshot explicitly.
|
|
14
|
+
*
|
|
15
|
+
* The status set is a STRUCTURAL discriminator (each status has its own
|
|
16
|
+
* payload), so the union below is closed; new statuses are protocol additions,
|
|
17
|
+
* not extensions.
|
|
8
18
|
*/
|
|
9
|
-
export const RUNTIME_TOOL_STATUSES = [
|
|
19
|
+
export const RUNTIME_TOOL_STATUSES = [
|
|
20
|
+
"pending",
|
|
21
|
+
"in_progress",
|
|
22
|
+
"completed",
|
|
23
|
+
"failed",
|
|
24
|
+
"cancelled",
|
|
25
|
+
] as const;
|
|
10
26
|
export type RuntimeToolStatus = (typeof RUNTIME_TOOL_STATUSES)[number];
|
|
11
|
-
export const RuntimeToolStatusSchema = z.enum(RUNTIME_TOOL_STATUSES);
|
|
12
27
|
|
|
13
28
|
/**
|
|
14
|
-
* What a tool call does, normalized across harnesses
|
|
15
|
-
*
|
|
16
|
-
* provider-native tool names.
|
|
29
|
+
* What a tool call does, normalized across harnesses — ACP's ToolKind taxonomy
|
|
30
|
+
* verbatim, plus `task` (subagent spawn; ACP has no subagent concept). Lets a
|
|
31
|
+
* consumer pick an icon / verb without knowing provider-native tool names.
|
|
32
|
+
*
|
|
33
|
+
* OPEN vocabulary (Law 3): unknown non-`_` values are reserved for this
|
|
34
|
+
* protocol and MUST be preserved; `_`-prefixed values are adapter/product
|
|
35
|
+
* extensions. Display groupings (e.g. edit|delete|move -> a "write" card) are
|
|
36
|
+
* consumer maps, not wire vocabulary.
|
|
17
37
|
*/
|
|
18
38
|
export const TOOL_KINDS = [
|
|
19
39
|
"read",
|
|
@@ -25,31 +45,56 @@ export const TOOL_KINDS = [
|
|
|
25
45
|
"think",
|
|
26
46
|
"fetch",
|
|
27
47
|
"switch_mode",
|
|
48
|
+
"task",
|
|
28
49
|
"other",
|
|
29
50
|
] as const;
|
|
30
|
-
export type ToolKind = (typeof TOOL_KINDS)[number];
|
|
31
|
-
export const ToolKindSchema =
|
|
51
|
+
export type ToolKind = (typeof TOOL_KINDS)[number] | (string & {});
|
|
52
|
+
export const ToolKindSchema = openVocabulary<ToolKind>();
|
|
32
53
|
|
|
33
|
-
/** A file location a tool call touches — lets UIs "follow along". */
|
|
54
|
+
/** A file location a tool call touches — lets UIs "follow along". (ACP shape.) */
|
|
34
55
|
export const ToolLocationSchema = z.object({
|
|
35
56
|
path: z.string(),
|
|
36
57
|
line: z.number().optional(),
|
|
37
58
|
});
|
|
38
59
|
export type ToolLocation = z.infer<typeof ToolLocationSchema>;
|
|
39
60
|
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
})
|
|
48
|
-
|
|
61
|
+
/**
|
|
62
|
+
* Structured, display-grade tool output. Three-audience doctrine:
|
|
63
|
+
* `ToolStateCompleted.output` (string) is the MODEL-facing factual record and
|
|
64
|
+
* universal rendering fallback; `content` is the display-grade structured
|
|
65
|
+
* view; `metadata` is the machine channel (exitCode, ...).
|
|
66
|
+
*/
|
|
67
|
+
export const ToolResultContentSchema = z.discriminatedUnion("type", [
|
|
68
|
+
z.object({ type: z.literal("text"), text: z.string(), _meta: MetaSchema }),
|
|
69
|
+
z.object({
|
|
70
|
+
type: z.literal("image"),
|
|
71
|
+
/** Base64-encoded bytes. */
|
|
72
|
+
data: z.string(),
|
|
73
|
+
mimeType: z.string(),
|
|
74
|
+
_meta: MetaSchema,
|
|
75
|
+
}),
|
|
76
|
+
z.object({
|
|
77
|
+
type: z.literal("diff"),
|
|
78
|
+
path: z.string(),
|
|
79
|
+
/** Absent for file creation. */
|
|
80
|
+
oldText: z.string().optional(),
|
|
81
|
+
newText: z.string(),
|
|
82
|
+
_meta: MetaSchema,
|
|
83
|
+
}),
|
|
84
|
+
z.object({
|
|
85
|
+
type: z.literal("terminal"),
|
|
86
|
+
/** Display anchor only — a separate ID domain from toolCallId. */
|
|
87
|
+
terminalId: z.string(),
|
|
88
|
+
_meta: MetaSchema,
|
|
89
|
+
}),
|
|
90
|
+
]);
|
|
91
|
+
export type ToolResultContent = z.infer<typeof ToolResultContentSchema>;
|
|
49
92
|
|
|
50
93
|
export const ToolStatePendingSchema = z.object({
|
|
51
94
|
status: z.literal("pending"),
|
|
95
|
+
/** Raw streamed JSON input — possibly invalid; parse best-effort for early render. */
|
|
52
96
|
partialInput: z.string(),
|
|
97
|
+
_meta: MetaSchema,
|
|
53
98
|
});
|
|
54
99
|
export type ToolStatePending = z.infer<typeof ToolStatePendingSchema>;
|
|
55
100
|
|
|
@@ -57,18 +102,23 @@ export const ToolStateInProgressSchema = z.object({
|
|
|
57
102
|
status: z.literal("in_progress"),
|
|
58
103
|
input: z.record(z.string(), z.unknown()),
|
|
59
104
|
title: z.string().optional(),
|
|
60
|
-
time: z.object({ start:
|
|
105
|
+
time: z.object({ start: EpochMsSchema }),
|
|
106
|
+
_meta: MetaSchema,
|
|
61
107
|
});
|
|
62
108
|
export type ToolStateInProgress = z.infer<typeof ToolStateInProgressSchema>;
|
|
63
109
|
|
|
64
110
|
export const ToolStateCompletedSchema = z.object({
|
|
65
111
|
status: z.literal("completed"),
|
|
66
112
|
input: z.record(z.string(), z.unknown()),
|
|
113
|
+
/** Model-facing factual record; always present as the rendering fallback. */
|
|
67
114
|
output: z.string(),
|
|
68
|
-
|
|
115
|
+
/** Display-grade structured output, when expressible (diffs, images, terminals). */
|
|
116
|
+
content: z.array(ToolResultContentSchema).optional(),
|
|
117
|
+
/** Machine channel: harness-native extras (e.g. `exitCode`). */
|
|
69
118
|
metadata: z.record(z.string(), z.unknown()).optional(),
|
|
70
|
-
|
|
71
|
-
|
|
119
|
+
title: z.string().optional(),
|
|
120
|
+
time: TimeSpanSchema,
|
|
121
|
+
_meta: MetaSchema,
|
|
72
122
|
});
|
|
73
123
|
export type ToolStateCompleted = z.infer<typeof ToolStateCompletedSchema>;
|
|
74
124
|
|
|
@@ -76,14 +126,24 @@ export const ToolStateFailedSchema = z.object({
|
|
|
76
126
|
status: z.literal("failed"),
|
|
77
127
|
input: z.record(z.string(), z.unknown()),
|
|
78
128
|
error: z.string(),
|
|
79
|
-
time:
|
|
129
|
+
time: TimeSpanSchema,
|
|
130
|
+
_meta: MetaSchema,
|
|
80
131
|
});
|
|
81
132
|
export type ToolStateFailed = z.infer<typeof ToolStateFailedSchema>;
|
|
82
133
|
|
|
134
|
+
export const ToolStateCancelledSchema = z.object({
|
|
135
|
+
status: z.literal("cancelled"),
|
|
136
|
+
input: z.record(z.string(), z.unknown()),
|
|
137
|
+
time: TimeSpanSchema,
|
|
138
|
+
_meta: MetaSchema,
|
|
139
|
+
});
|
|
140
|
+
export type ToolStateCancelled = z.infer<typeof ToolStateCancelledSchema>;
|
|
141
|
+
|
|
83
142
|
export const ToolStateSchema = z.discriminatedUnion("status", [
|
|
84
143
|
ToolStatePendingSchema,
|
|
85
144
|
ToolStateInProgressSchema,
|
|
86
145
|
ToolStateCompletedSchema,
|
|
87
146
|
ToolStateFailedSchema,
|
|
147
|
+
ToolStateCancelledSchema,
|
|
88
148
|
]);
|
|
89
149
|
export type ToolState = z.infer<typeof ToolStateSchema>;
|