@zvada/agent-server 0.3.7-replay.1 → 0.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,6 +1,13 @@
1
1
  # Changelog
2
2
 
3
- ## 0.3.7-replay.1
3
+ ## 0.3.7
4
+
5
+ - Retain Claude's reported 5-minute and 1-hour cache-creation token counts in the existing token usage contract. Missing durations stay absent and reported zeroes are preserved.
6
+ - Document the shared cache accounting contract for custom harness adapters, including totals versus subsets and reported cost.
7
+
8
+ - Preserve per-turn harness, configured model and thinking level in lifecycle events, run summaries and the canonical conversation fold.
9
+ - Keep primary model reports separate from requested aliases; collect Claude message models and Codex routing targets without attributing subagent models to the main run.
10
+ - Execution metadata is an explicit allowlist and contains no credentials, environment variables or working-directory paths.
4
11
 
5
12
  - Add `AgentServerClient.onEventGap` for consumers using `startTurn` and `onEvent`.
6
13
  Missing replay data and server restarts are reported before surviving events.
package/docs/harnesses.md CHANGED
@@ -82,6 +82,38 @@ The protocol was checked against Codex CLI 0.153.4 and the
82
82
  harness's unparsed event, for migration/debugging/fixture-recording. `data`
83
83
  has **no stability guarantees** and is off by default.
84
84
 
85
+ ## Token and cache accounting
86
+
87
+ Every adapter uses the same `TokenUsage` on `turn.ended.tokens`. A custom
88
+ adapter maps its provider's counts into this contract; storage and consumers
89
+ do not need a new schema for each harness.
90
+
91
+ | Field | Meaning |
92
+ | --- | --- |
93
+ | `input` | Uncached input tokens |
94
+ | `output` | Output tokens, including any reasoning tokens |
95
+ | `reasoning` | Reported reasoning portion of output, when available |
96
+ | `cache.read` | Input tokens reused from the provider's cache |
97
+ | `cache.write` | Input tokens written to the provider's cache |
98
+ | `cache.writeEphemeral5m` | Reported 5-minute portion of cache writes |
99
+ | `cache.writeEphemeral1h` | Reported 1-hour portion of cache writes |
100
+ | `turn.ended.cost` | Harness-reported USD cost, when available |
101
+
102
+ Input, cache reads and cache writes are disjoint. If the native input total
103
+ already includes cached tokens, subtract them once in the adapter. Reasoning
104
+ is part of output and duration buckets are parts of cache writes; do not add
105
+ those subsets to the totals again. Use the shipped `addTokenUsage` helper.
106
+
107
+ Cache values count tokens, not cache entries. Duration buckets describe creation
108
+ usage reported for that turn, not whether a cache entry is still alive. Leave
109
+ unreported duration buckets and cost absent; a reported zero stays zero. Do not
110
+ infer cache durations, savings or prices from the harness name. Final native
111
+ turn totals take precedence over intermediate message usage.
112
+
113
+ `session.usage` describes context occupancy and is not a source for per-turn
114
+ cache accounting. An ACP/custom agent that exposes only this gauge cannot
115
+ provide the cache breakdown without an additional native usage report.
116
+
85
117
  ## Known limitations (roadmap)
86
118
 
87
119
  - **MCP servers** are wired for Claude only; Codex MCP passthrough is pending
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@zvada/agent-server",
3
- "version": "0.3.7-replay.1",
3
+ "version": "0.3.7",
4
4
  "description": "Harness-agnostic agent execution engine: run Claude Code, Codex (SDK/CLI + app-server), and any ACP agent behind one interface with a normalized event stream, multi-turn sessions, and resume. Root export is the wire contract; /core, /server, /client are the seats.",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -55,6 +55,10 @@ interface RawUsage {
55
55
  output_tokens?: number;
56
56
  cache_read_input_tokens?: number;
57
57
  cache_creation_input_tokens?: number;
58
+ cache_creation?: {
59
+ ephemeral_5m_input_tokens?: number;
60
+ ephemeral_1h_input_tokens?: number;
61
+ };
58
62
  }
59
63
  interface ContentBlock {
60
64
  type: string;
@@ -79,7 +83,7 @@ type ClaudeMessage =
79
83
  | { type: "turn_interrupted" }
80
84
  | {
81
85
  type: "assistant";
82
- message?: { content?: ContentBlock[]; usage?: RawUsage };
86
+ message?: { content?: ContentBlock[]; usage?: RawUsage; model?: string };
83
87
  parent_tool_use_id?: string | null;
84
88
  }
85
89
  | { type: "user"; message?: { content?: ContentBlock[] | string } }
@@ -227,6 +231,7 @@ export class ClaudeCodeTransformer implements EventTransformer<unknown> {
227
231
  private hasStreamed = false;
228
232
 
229
233
  private usage: TokenUsage = { ...DEFAULT_TOKEN_USAGE };
234
+ private readonly reportedModels = new Set<string>();
230
235
  private stopReason?: StopReason;
231
236
  private finishReason?: string;
232
237
  private cost?: number;
@@ -287,6 +292,7 @@ export class ClaudeCodeTransformer implements EventTransformer<unknown> {
287
292
  : undefined);
288
293
  return {
289
294
  usage: this.usage,
295
+ ...(this.reportedModels.size && { reportedModels: [...this.reportedModels] }),
290
296
  stopReason: this.stopReason,
291
297
  finishReason: this.finishReason,
292
298
  cost: this.cost,
@@ -308,6 +314,7 @@ export class ClaudeCodeTransformer implements EventTransformer<unknown> {
308
314
  * skip the duplicate. Otherwise emit its content blocks as finalized parts.
309
315
  */
310
316
  private handleAssistant(msg: Extract<ClaudeMessage, { type: "assistant" }>): AdapterEvent[] {
317
+ if (!msg.parent_tool_use_id && msg.message?.model) this.reportedModels.add(msg.message.model);
311
318
  // Context gauge: every top-level model message carries fresh usage (the
312
319
  // sub-agent's context is its own and would misreport the session's).
313
320
  const gauge: AdapterEvent[] =
@@ -496,12 +503,19 @@ export class ClaudeCodeTransformer implements EventTransformer<unknown> {
496
503
  private captureResult(msg: Extract<ClaudeMessage, { type: "result" }>): AdapterEvent[] {
497
504
  this.sawResult = true;
498
505
  if (msg.usage) {
506
+ const creation = msg.usage.cache_creation;
499
507
  this.usage = {
500
508
  input: msg.usage.input_tokens ?? 0,
501
509
  output: msg.usage.output_tokens ?? this.usage.output,
502
510
  cache: {
503
511
  read: msg.usage.cache_read_input_tokens ?? 0,
504
512
  write: msg.usage.cache_creation_input_tokens ?? 0,
513
+ ...(creation?.ephemeral_5m_input_tokens !== undefined && {
514
+ writeEphemeral5m: creation.ephemeral_5m_input_tokens,
515
+ }),
516
+ ...(creation?.ephemeral_1h_input_tokens !== undefined && {
517
+ writeEphemeral1h: creation.ephemeral_1h_input_tokens,
518
+ }),
505
519
  },
506
520
  };
507
521
  }
@@ -47,6 +47,7 @@ interface Notification {
47
47
  error?: { message?: string; additionalDetails?: string | null };
48
48
  willRetry?: boolean;
49
49
  message?: string;
50
+ toModel?: string;
50
51
  };
51
52
  }
52
53
 
@@ -55,6 +56,7 @@ export class CodexAppServerTransformer implements EventTransformer<unknown> {
55
56
  private readonly partByItem = new Map<string, Part>();
56
57
 
57
58
  private usage: TokenUsage = { ...DEFAULT_TOKEN_USAGE };
59
+ private readonly reportedModels = new Set<string>();
58
60
  private stopReason?: StopReason;
59
61
  private finishReason?: string;
60
62
  private error?: string;
@@ -66,6 +68,9 @@ export class CodexAppServerTransformer implements EventTransformer<unknown> {
66
68
  process(raw: unknown): AdapterEvent[] {
67
69
  const { method, params } = raw as Notification;
68
70
  switch (method) {
71
+ case "model/rerouted":
72
+ if (params.toModel) this.reportedModels.add(params.toModel);
73
+ return [];
69
74
  case "item/started":
70
75
  return params.item ? this.openItem(params.item) : [];
71
76
  case "item/completed":
@@ -105,6 +110,7 @@ export class CodexAppServerTransformer implements EventTransformer<unknown> {
105
110
  finish(): TransformResult {
106
111
  return {
107
112
  usage: this.usage,
113
+ ...(this.reportedModels.size && { reportedModels: [...this.reportedModels] }),
108
114
  stopReason: this.stopReason,
109
115
  finishReason: this.finishReason,
110
116
  error: this.error,
@@ -32,6 +32,8 @@ export type AdapterEvent =
32
32
  /** Terminal result of a turn, surfaced on `turn.ended`. */
33
33
  export interface TransformResult {
34
34
  usage: TokenUsage;
35
+ /** Models observed on the primary run, excluding subagents and configured defaults. */
36
+ reportedModels?: string[];
35
37
  /** Normalized terminal status, when the harness reported one. The engine
36
38
  * overrides with `cancelled`/`error` based on what it observed. */
37
39
  stopReason?: StopReason;
@@ -14,6 +14,7 @@ import type {
14
14
  RunRequest,
15
15
  StopReason,
16
16
  TokenUsage,
17
+ TurnExecution,
17
18
  } from "../../protocol/index.ts";
18
19
  import { DEFAULT_TOKEN_USAGE, generateUUIDv7 } from "../../protocol/index.ts";
19
20
  import type { AgentExecuteOptions, CancelResult, PermissionDecision } from "../agents/base.ts";
@@ -28,6 +29,7 @@ export interface RunSummary {
28
29
  sessionId: string;
29
30
  turnId: string;
30
31
  harness: AgentHarness;
32
+ execution: TurnExecution;
31
33
  /** Native session/thread id to persist for a future resume. */
32
34
  nativeSessionId?: string;
33
35
  /** When the turn requested a resume: whether the harness honored it. */
@@ -239,7 +241,12 @@ export class AgentRuntime {
239
241
  const { sessionId, turnId, input, config } = request;
240
242
  const agent = this.registry.getAgent(config.harness);
241
243
  const transformer = this.registry.getAdapter(config.harness)({ sessionId });
242
- const processor = new EventProcessor(sessionId, turnId, { model: config.model });
244
+ const execution: TurnExecution = {
245
+ harness: config.harness,
246
+ ...(config.model !== undefined && { model: config.model }),
247
+ ...(config.thinkingLevel !== undefined && { thinkingLevel: config.thinkingLevel }),
248
+ };
249
+ const processor = new EventProcessor(sessionId, turnId, execution);
243
250
 
244
251
  // Isolate sink failures: a misbehaving transport must never break the
245
252
  // turn's lifecycle bracketing (turn.started ... turn.ended).
@@ -374,7 +381,7 @@ export class AgentRuntime {
374
381
  await Promise.all([...settles, ...settleEmits]);
375
382
  };
376
383
 
377
- await emit({ type: "turn.started", turnId, sessionId, timestamp: Date.now() });
384
+ await emit({ type: "turn.started", turnId, sessionId, execution, timestamp: Date.now() });
378
385
 
379
386
  // The user echo (spec §7.2): the submitted input goes back onto the stream
380
387
  // as a complete user message (outputIndex 0) before any harness output, so
@@ -474,6 +481,10 @@ export class AgentRuntime {
474
481
  nativeSessionId,
475
482
  ...(resumed !== undefined && { resumed }),
476
483
  usage: result.usage ?? DEFAULT_TOKEN_USAGE,
484
+ execution: {
485
+ ...execution,
486
+ ...(result.reportedModels && { reportedModels: result.reportedModels }),
487
+ },
477
488
  stopReason,
478
489
  finishReason: result.finishReason,
479
490
  cost: result.cost,
@@ -4,7 +4,14 @@ import {
4
4
  echoMessageId,
5
5
  generateUUIDv7,
6
6
  } from "../../protocol/index.ts";
7
- import type { AgentInput, Delta, LifecycleEvent, Part, StopReason } from "../../protocol/index.ts";
7
+ import type {
8
+ AgentInput,
9
+ Delta,
10
+ LifecycleEvent,
11
+ Part,
12
+ StopReason,
13
+ TurnExecution,
14
+ } from "../../protocol/index.ts";
8
15
  import type { AdapterEvent, TransformResult } from "../agents/types.ts";
9
16
 
10
17
  /** Where a part lives on the wire — fixed at first emission for the whole turn. */
@@ -36,7 +43,7 @@ export class EventProcessor {
36
43
  constructor(
37
44
  private readonly sessionId: string,
38
45
  private readonly turnId: string,
39
- private readonly meta: { model?: string } = {},
46
+ private readonly execution?: TurnExecution,
40
47
  ) {}
41
48
 
42
49
  /**
@@ -125,6 +132,12 @@ export class EventProcessor {
125
132
  finishReason: result.finishReason,
126
133
  tokens: result.usage,
127
134
  cost: result.cost,
135
+ ...(this.execution && {
136
+ execution: {
137
+ ...this.execution,
138
+ ...(result.reportedModels && { reportedModels: result.reportedModels }),
139
+ },
140
+ }),
128
141
  error:
129
142
  result.error && !result.cancelled
130
143
  ? { category: classifyError(result.error), message: result.error }
@@ -160,7 +173,7 @@ export class EventProcessor {
160
173
  // user-authored. Gating here keeps it off every user message; every
161
174
  // assistant message (top-level AND parented sub-agent output) keeps it,
162
175
  // since `config.model` is the only per-turn model signal available.
163
- ...(role === "assistant" && this.meta.model && { model: this.meta.model }),
176
+ ...(role === "assistant" && this.execution?.model && { model: this.execution?.model }),
164
177
  timestamp: Date.now(),
165
178
  };
166
179
  }
@@ -0,0 +1,16 @@
1
+ import { z } from "zod";
2
+ import { AgentHarnessSchema } from "./harness.ts";
3
+ import { ThinkingLevelSchema } from "./thinking.ts";
4
+
5
+ /** Safe, per-turn execution metadata. Never serialize the full RunConfig here. */
6
+ export const TurnExecutionSchema = z.object({
7
+ harness: AgentHarnessSchema,
8
+ /** Model id/alias passed to the harness. Absent means the harness chose its default. */
9
+ model: z.string().optional(),
10
+ /** Requested setting, not a measurement of the model's internal reasoning. */
11
+ thinkingLevel: ThinkingLevelSchema.optional(),
12
+ /** Distinct models reported by the primary harness during this turn, in observation order.
13
+ * Excludes subagents. Absence means unreported, never a fallback to the requested alias. */
14
+ reportedModels: z.array(z.string()).optional(),
15
+ });
16
+ export type TurnExecution = z.infer<typeof TurnExecutionSchema>;
@@ -17,6 +17,7 @@ export * from "./harness.ts";
17
17
  export * from "./thinking.ts";
18
18
  export * from "./models.ts";
19
19
  export * from "./config.ts";
20
+ export * from "./execution.ts";
20
21
  export * from "./lifecycle.ts";
21
22
  export * from "./stop-reasons.ts";
22
23
  export * from "./seq-cursor.ts";
@@ -1,5 +1,6 @@
1
1
  import { z } from "zod";
2
2
  import { ErrorCategorySchema, ErrorInfoSchema } from "./errors.ts";
3
+ import { TurnExecutionSchema } from "./execution.ts";
3
4
  import { isKnownLifecycleEventType } from "./guards.ts";
4
5
  import { AgentHarnessSchema } from "./harness.ts";
5
6
  import { MetaSchema } from "./meta.ts";
@@ -168,6 +169,7 @@ export const TurnStartedEventSchema = z.object({
168
169
  type: z.literal("turn.started"),
169
170
  sessionId: z.string(),
170
171
  turnId: z.string(),
172
+ execution: TurnExecutionSchema.optional(),
171
173
  timestamp: EpochMsSchema,
172
174
  _meta: MetaSchema,
173
175
  });
@@ -177,6 +179,7 @@ export const TurnEndedEventSchema = z.object({
177
179
  type: z.literal("turn.ended"),
178
180
  sessionId: z.string(),
179
181
  turnId: z.string(),
182
+ execution: TurnExecutionSchema.optional(),
180
183
  /** Normalized terminal status — `cancelled` and `error` are first-class. */
181
184
  stopReason: StopReasonSchema,
182
185
  /** Raw provider finish string (`end_turn`, `completed`, …), always preserved when reported. */
@@ -2,6 +2,7 @@ import { z } from "zod";
2
2
  // The package ships raw .ts — keep imports exactly-used, or every consumer
3
3
  // compiling with noUnusedLocals inherits the noise as build errors.
4
4
  import { ErrorCategorySchema, ErrorInfoSchema } from "./errors.ts";
5
+ import { TurnExecutionSchema } from "./execution.ts";
5
6
  import { isUnknownEvent } from "./guards.ts";
6
7
  import { AgentHarnessSchema } from "./harness.ts";
7
8
  import {
@@ -115,6 +116,7 @@ export type TimelineEntry = z.infer<typeof TimelineEntrySchema>;
115
116
 
116
117
  export const ConversationTurnSchema = z.object({
117
118
  turnId: z.string(),
119
+ execution: TurnExecutionSchema.optional(),
118
120
  /** Derived view state, not wire vocabulary. */
119
121
  status: z.enum(["active", "ended"]),
120
122
  stopReason: StopReasonSchema.optional(),
@@ -496,6 +498,7 @@ function reduceWith(
496
498
  if (existing !== -1) return state; // replay overlap — the turn is known
497
499
  const turn: ConversationTurn = {
498
500
  turnId: event.turnId,
501
+ ...(event.execution !== undefined && { execution: event.execution }),
499
502
  status: "active",
500
503
  errors: [],
501
504
  startedAt: event.timestamp,
@@ -603,6 +606,7 @@ function reduceWith(
603
606
  ...base,
604
607
  status: "ended",
605
608
  stopReason: event.stopReason,
609
+ ...(event.execution !== undefined && { execution: event.execution }),
606
610
  ...(event.finishReason !== undefined && { finishReason: event.finishReason }),
607
611
  ...(event.tokens !== undefined && { tokens: event.tokens }),
608
612
  ...(event.cost !== undefined && { cost: event.cost }),