@intentface/latch-core 0.9.1 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/runtime.js CHANGED
@@ -5,6 +5,7 @@ import { summarizeForCompaction } from "./compaction.js";
5
5
  import { computeCost } from "./pricing.js";
6
6
  import { defaultPromptCachingPlan, markLastFunctionTool, mergeProviderOptions, toolsetHash, } from "./prompt-caching.js";
7
7
  import { LimitExceededError, exceededPolicy, remainingTokens, windowRef, } from "./limits.js";
8
+ import { compose, fireClientToolResults, insertStepParts, isNewSession, lastUserTextOf, renderTranscript, replaceLastUserText, wrapTools, } from "./extensions/index.js";
8
9
  /** Ids for assistant response messages — `msg_<random>`, matching storage ids. */
9
10
  const generateMessageId = createIdGenerator({ prefix: "msg", separator: "_" });
10
11
  /** Sub-thread chat ids start with `sub_` so the host can keep them out of the
@@ -575,7 +576,10 @@ export function createRuntime(config) {
575
576
  * server-side (resume).
576
577
  */
577
578
  async function buildTurn(args) {
578
- const { resolvedChatId, principal, modelId, cfg, run, lease, uiMessages, runtimeCtx } = args;
579
+ const { resolvedChatId, principal, modelId, cfg, run, lease, runtimeCtx } = args;
580
+ // Mutable: the start moments below may inject, replace or rewrite history
581
+ // before the model runs. From here on this is the turn's transcript.
582
+ let uiMessages = args.uiMessages;
579
583
  const admitted = args.admitted;
580
584
  const autoApprove = args.autoApprove ?? false;
581
585
  const blockGated = args.blockGated ?? false;
@@ -585,6 +589,49 @@ export function createRuntime(config) {
585
589
  // threaded a root down. Subagent turns spawned below inherit it, so a whole
586
590
  // delegation tree shares one telemetry session (see telemetryMeta).
587
591
  const rootChatId = args.rootChatId ?? resolvedChatId;
592
+ // --- the extension seam --------------------------------------------------
593
+ // Runtime-wide extensions first, then the agent's own. With none attached
594
+ // every moment below is a no-op and the turn is byte-identical to a runtime
595
+ // without the seam — `turn-snapshot.test.ts` holds it to that.
596
+ const hooks = compose([...(config.extensions ?? []), ...(cfg.extensions ?? [])]);
597
+ const xrun = {
598
+ runId: run.id,
599
+ principal: principal,
600
+ };
601
+ const xctx = (stepNumber, messages) => ({
602
+ ...xrun,
603
+ stepNumber,
604
+ messages,
605
+ });
606
+ // A command an extension answered without the model (see the `input` moment).
607
+ let handled;
608
+ // The start moments run once per conversation turn. A resume replays a
609
+ // history they already shaped, and a decision continues a turn they
610
+ // already opened — running them again would inject twice.
611
+ const startMoments = trigger !== "resume" && trigger !== "decision";
612
+ if (startMoments) {
613
+ if (isNewSession(uiMessages))
614
+ uiMessages = (await hooks.messagesIn("session_start", uiMessages, xrun));
615
+ uiMessages = (await hooks.messagesIn("before_agent_start", uiMessages, xrun));
616
+ const text = lastUserTextOf(uiMessages);
617
+ if (text != null) {
618
+ const e = await hooks.input(text, uiMessages, xrun);
619
+ if (e && "handled" in e && e.handled) {
620
+ const reply = e.reply;
621
+ handled =
622
+ typeof reply === "string"
623
+ ? [{ type: "text", text: reply }]
624
+ : Array.isArray(reply)
625
+ ? reply
626
+ : [];
627
+ if (e.replace)
628
+ uiMessages = e.replace;
629
+ }
630
+ else if (e && "transform" in e && e.transform != null) {
631
+ uiMessages = replaceLastUserText(uiMessages, e.transform);
632
+ }
633
+ }
634
+ }
588
635
  const abort = new AbortController();
589
636
  let heartbeat;
590
637
  const stopHeartbeat = () => {
@@ -648,7 +695,9 @@ export function createRuntime(config) {
648
695
  }
649
696
  const opened = [];
650
697
  try {
651
- for (const source of sources) {
698
+ // A command answered without the model needs no tools; don't open MCP
699
+ // connections for `/stats`.
700
+ for (const source of handled ? [] : sources) {
652
701
  opened.push(await source.open(principal));
653
702
  }
654
703
  }
@@ -746,6 +795,13 @@ export function createRuntime(config) {
746
795
  // Memory is optional and degradable — continue without it.
747
796
  }
748
797
  }
798
+ // instructions: an extension's appends land after the memory block; the
799
+ // date line stays the tail (see the agent construction). A `replace` may
800
+ // hand back system message(s) carrying providerOptions — the extension then
801
+ // owns that shape, and the caching plan's system breakpoint steps aside.
802
+ const composedInstructions = hooks.has("instructions")
803
+ ? await hooks.instructions(instructions, xctx(0, uiMessages))
804
+ : instructions;
749
805
  for (const name of cfg.disableTools ?? [])
750
806
  delete tools[name];
751
807
  // Per-agent tool description overrides — swap the resolved tool's
@@ -877,6 +933,27 @@ export function createRuntime(config) {
877
933
  });
878
934
  capabilityTools.push("spawn_agent");
879
935
  }
936
+ // register_tools: what extensions add are capabilities — always active, and
937
+ // gated by the agent's own policy like any other tool. Mutated in place so
938
+ // every later reference (toolsContext keys, the fingerprint) sees them.
939
+ if (hooks.has("register_tools")) {
940
+ const before = new Set(Object.keys(tools));
941
+ const registered = hooks.tools(tools, xrun);
942
+ for (const k of Object.keys(tools))
943
+ if (!(k in registered))
944
+ delete tools[k];
945
+ Object.assign(tools, registered);
946
+ for (const name of Object.keys(tools))
947
+ if (!before.has(name))
948
+ capabilityTools.push(name);
949
+ }
950
+ // Outputs a CLIENT produced (a browser tool's answer) are already in the
951
+ // transcript; fire tool_result for them now that the toolset can tell a
952
+ // client tool from a server one.
953
+ // A decision turn is exactly when a client's answer has just been recorded,
954
+ // so it fires here too; a resume replays outputs an earlier run handled.
955
+ if ((startMoments || trigger === "decision") && hooks.has("tool_result"))
956
+ uiMessages = (await fireClientToolResults(uiMessages, tools, hooks, xrun));
880
957
  // Subagent runs are autonomous — there's no human to authorize, so waive
881
958
  // approvals and drop the OAuth `connect_` gates (they'd pause forever).
882
959
  if (autoApprove) {
@@ -968,7 +1045,64 @@ export function createRuntime(config) {
968
1045
  ? ({ steps }) => steps.reduce((n, st) => n + (st.usage?.inputTokens ?? 0) + (st.usage?.outputTokens ?? 0), 0) >=
969
1046
  tokenBudget
970
1047
  : undefined;
971
- const stopWhen = budgetStop ? [stepCap ?? stepCountIs(20), budgetStop] : stepCap;
1048
+ const baseStopWhen = budgetStop ? [stepCap ?? stepCountIs(20), budgetStop] : stepCap;
1049
+ // --- step boundaries for the seam -------------------------------------------
1050
+ // A step ends twice from here: on the model side (`agent.stream`'s
1051
+ // onStepEnd — usage, finish reason, tool calls) and on the UI side (the
1052
+ // stream's onStepEnd — the assembled assistant message). `step_end` needs
1053
+ // both, so a step is SETTLED once both have arrived, and the next step's
1054
+ // prepareStep waits for the settlement: an extension that persists has
1055
+ // written before the model is called again. None of this runs when no
1056
+ // extension listens to a step moment.
1057
+ const stepHooks = hooks.has("step_start") || hooks.has("step_end");
1058
+ const slotOf = (m, k) => {
1059
+ let d = m.get(k);
1060
+ if (!d) {
1061
+ let resolve;
1062
+ const promise = new Promise((r) => (resolve = r));
1063
+ d = { promise, resolve };
1064
+ m.set(k, d);
1065
+ }
1066
+ return d;
1067
+ };
1068
+ const modelSide = new Map();
1069
+ const uiSide = new Map();
1070
+ const settledSteps = new Map();
1071
+ const appendedByStep = new Map();
1072
+ const stepStartedAt = new Map();
1073
+ let control0;
1074
+ let nextControl;
1075
+ let stoppedBeforeStart = false;
1076
+ let uiStepIndex = 0;
1077
+ let stepsSeen = 0;
1078
+ let turnWriter;
1079
+ let latestAssistant;
1080
+ // A continuation reuses the assistant message the SDK merges into; its
1081
+ // earlier step-start parts are not this run's steps.
1082
+ const anchor = uiMessages[uiMessages.length - 1];
1083
+ const anchorIsAssistant = anchor?.role === "assistant";
1084
+ const priorSteps = anchorIsAssistant
1085
+ ? anchor.parts.filter((p) => p.type === "step-start").length
1086
+ : 0;
1087
+ const transcriptNow = () => latestAssistant
1088
+ ? anchorIsAssistant
1089
+ ? [...uiMessages.slice(0, -1), latestAssistant]
1090
+ : [...uiMessages, latestAssistant]
1091
+ : uiMessages;
1092
+ // Assigned below, once the checkpoint chain exists; only ever called mid-stream.
1093
+ let settle = async () => { };
1094
+ const extensionStop = stepHooks
1095
+ ? async ({ steps }) => {
1096
+ const k = steps.length - 1;
1097
+ await settle(k);
1098
+ const c = await hooks.stepStart(xctx(steps.length, transcriptNow()));
1099
+ nextControl = c;
1100
+ return c.stop;
1101
+ }
1102
+ : undefined;
1103
+ const stopWhen = extensionStop
1104
+ ? [...(Array.isArray(baseStopWhen) ? baseStopWhen : [baseStopWhen ?? stepCountIs(20)]), extensionStop]
1105
+ : baseStopWhen;
972
1106
  // Reasoning/thinking effort → provider-specific options (the platform maps
973
1107
  // it per provider + gates to reasoning-capable models). Called even without
974
1108
  // an effort so the hook can set safe defaults (e.g. lift a provider's tiny
@@ -1001,6 +1135,14 @@ export function createRuntime(config) {
1001
1135
  // run so churn across a chat's turns (a cache invalidator) is queryable.
1002
1136
  const toolsHash = toolsetHash(activeTools ?? Object.keys(tools));
1003
1137
  const providerOptions = mergeProviderOptions(reasoning?.providerOptions, caching?.requestProviderOptions);
1138
+ // tool_call / tool_result: wrap in place, after the caching marker and the
1139
+ // test-run stubs, so every later reference sees the same names. Identity-
1140
+ // preserving when no extension listens.
1141
+ if (hooks.has("tool_call") || hooks.has("tool_result")) {
1142
+ const wrapped = wrapTools(tools, hooks, xrun);
1143
+ for (const k of Object.keys(wrapped))
1144
+ tools[k] = wrapped[k];
1145
+ }
1004
1146
  // Telemetry seam: per-run AI SDK telemetry options + trace correlation
1005
1147
  // identity, resolved once per turn (undefined → nothing traced).
1006
1148
  const telemetryMeta = {
@@ -1010,6 +1152,7 @@ export function createRuntime(config) {
1010
1152
  modelId,
1011
1153
  principal,
1012
1154
  startedAt: run.startedAt,
1155
+ agentVersion: cfg.configVersion,
1013
1156
  depth,
1014
1157
  trigger,
1015
1158
  parentRunId: args.parentRunId,
@@ -1032,24 +1175,46 @@ export function createRuntime(config) {
1032
1175
  // the instructions become a MARKED system message and the date line a
1033
1176
  // separate unmarked one, so the daily flip lands after the breakpoint
1034
1177
  // instead of invalidating it at midnight UTC.
1035
- instructions: caching?.systemProviderOptions
1036
- ? [
1037
- ...(instructions
1038
- ? [
1039
- {
1040
- role: "system",
1041
- content: instructions,
1042
- providerOptions: caching.systemProviderOptions,
1043
- },
1044
- ]
1045
- : []),
1046
- { role: "system", content: currentDateLine() },
1047
- ]
1048
- : [instructions, currentDateLine()].filter(Boolean).join("\n\n"),
1178
+ instructions: typeof composedInstructions === "object" && composedInstructions !== null
1179
+ ? // An extension replaced the system prompt with message(s) of its own,
1180
+ // providerOptions and all: keep them, with the date line as its own
1181
+ // trailing message so the daily flip stays past any breakpoint.
1182
+ [
1183
+ ...(Array.isArray(composedInstructions) ? composedInstructions : [composedInstructions]),
1184
+ { role: "system", content: currentDateLine() },
1185
+ ]
1186
+ : caching?.systemProviderOptions
1187
+ ? [
1188
+ ...(composedInstructions
1189
+ ? [
1190
+ {
1191
+ role: "system",
1192
+ content: composedInstructions,
1193
+ providerOptions: caching.systemProviderOptions,
1194
+ },
1195
+ ]
1196
+ : []),
1197
+ { role: "system", content: currentDateLine() },
1198
+ ]
1199
+ : [composedInstructions, currentDateLine()].filter(Boolean).join("\n\n"),
1049
1200
  tools,
1050
1201
  toolApproval,
1051
1202
  stopWhen,
1052
1203
  activeTools,
1204
+ ...(stepHooks
1205
+ ? {
1206
+ prepareStep: (async ({ stepNumber }) => {
1207
+ if (stepNumber > 0)
1208
+ await settle(stepNumber - 1);
1209
+ stepStartedAt.set(stepNumber, Date.now());
1210
+ const c = stepNumber === 0 ? control0 : nextControl;
1211
+ return {
1212
+ ...(c?.model ? { model: c.model } : {}),
1213
+ ...(c?.activeTools ? { activeTools: c.activeTools } : {}),
1214
+ };
1215
+ }),
1216
+ }
1217
+ : {}),
1053
1218
  ...(providerOptions ? { providerOptions } : {}),
1054
1219
  ...(reasoning?.maxOutputTokens ? { maxOutputTokens: reasoning.maxOutputTokens } : {}),
1055
1220
  ...(telemetryOptions ? { telemetry: telemetryOptions } : {}),
@@ -1076,6 +1241,80 @@ export function createRuntime(config) {
1076
1241
  // at most the write in flight and the one queued behind it. A failed write
1077
1242
  // logs and releases the chain — the next checkpoint still runs.
1078
1243
  let checkpoints = Promise.resolve();
1244
+ const checkpoint = (message) => {
1245
+ if (!dur.enabled || abort.signal.aborted)
1246
+ return;
1247
+ checkpoints = checkpoints
1248
+ .then(() => {
1249
+ // Re-checked at write time: the lease may have gone while this
1250
+ // checkpoint waited behind the previous one.
1251
+ if (abort.signal.aborted)
1252
+ return;
1253
+ return storage.appendMessages(resolvedChatId, [message]);
1254
+ })
1255
+ .catch((e) => console.warn(`[durability] step checkpoint failed for run ${run.id}:`, e));
1256
+ };
1257
+ if (stepHooks) {
1258
+ const needUi = hooks.has("step_end");
1259
+ settle = (k) => {
1260
+ const known = settledSteps.get(k);
1261
+ if (known)
1262
+ return known;
1263
+ const p = (async () => {
1264
+ const model = await slotOf(modelSide, k).promise;
1265
+ const response = needUi ? await slotOf(uiSide, k).promise : undefined;
1266
+ const raw = model.usage;
1267
+ // The SDK assembles the message from chunks; our parts were streamed
1268
+ // transient and never entered it. Fold every step's back in.
1269
+ const assistant = response
1270
+ ? insertStepParts(response, appendedByStep, priorSteps)
1271
+ : undefined;
1272
+ const messages = assistant
1273
+ ? anchorIsAssistant
1274
+ ? [...uiMessages.slice(0, -1), assistant]
1275
+ : [...uiMessages, assistant]
1276
+ : transcriptNow();
1277
+ const info = {
1278
+ ...xctx(k, messages),
1279
+ usage: {
1280
+ inputTokens: raw?.inputTokens,
1281
+ outputTokens: raw?.outputTokens,
1282
+ totalTokens: raw?.totalTokens,
1283
+ cacheReadTokens: raw?.inputTokenDetails?.cacheReadTokens,
1284
+ cacheWriteTokens: raw?.inputTokenDetails?.cacheWriteTokens,
1285
+ reasoningTokens: raw?.outputTokenDetails?.reasoningTokens,
1286
+ },
1287
+ finishReason: model.finishReason,
1288
+ model: model.model,
1289
+ durationMs: Date.now() - (stepStartedAt.get(k) ?? Date.now()),
1290
+ toolCalls: model.toolCalls,
1291
+ };
1292
+ if (needUi) {
1293
+ const appended = await hooks.stepEnd(messages, info);
1294
+ if (appended.length) {
1295
+ appendedByStep.set(k, appended);
1296
+ // The client copy goes through the view rules and is TRANSIENT, so
1297
+ // the SDK's own merge cannot persist a viewed copy over the truth,
1298
+ // which is folded into the checkpoint and the finish below.
1299
+ for (const part of appended) {
1300
+ if (!part.type.startsWith("data-"))
1301
+ continue;
1302
+ const view = hooks.view(part, false, xrun);
1303
+ if (view && "hide" in view && view.hide)
1304
+ continue;
1305
+ const shown = view && "replace" in view && view.replace ? view.replace : part;
1306
+ turnWriter?.write({ type: shown.type, data: shown.data, transient: true });
1307
+ }
1308
+ }
1309
+ latestAssistant = insertStepParts(response, appendedByStep, priorSteps);
1310
+ checkpoint(latestAssistant);
1311
+ }
1312
+ })();
1313
+ settledSteps.set(k, p);
1314
+ p.catch(() => { }); // surfaces where it is awaited: prepareStep, stopWhen, onFinish
1315
+ return p;
1316
+ };
1317
+ }
1079
1318
  // Post-turn telemetry: fired after persistence on every terminal path
1080
1319
  // (completed / awaiting_input / errored). Must never fail the turn.
1081
1320
  const reportRunFinished = async (status, messages, error) => {
@@ -1128,10 +1367,48 @@ export function createRuntime(config) {
1128
1367
  // turn would upsert onto the same row — see the dup-id regression test).
1129
1368
  generateId: generateMessageId,
1130
1369
  execute: async ({ writer }) => {
1131
- // (in-turn writer writes — e.g. a "compacting…" / compaction marker —
1132
- // would go here, before the model output is merged in)
1370
+ turnWriter = writer;
1371
+ if (handled) {
1372
+ // A command an extension answered without the model. Emitted as an
1373
+ // ordinary assistant turn so the conversation reads normally and the
1374
+ // finish hook persists it like any other.
1375
+ writer.write({
1376
+ type: "start",
1377
+ messageId: generateMessageId(),
1378
+ messageMetadata: { visibility: "user", model: modelId, createdAt: Date.now(), runId: run.id },
1379
+ });
1380
+ for (const p of handled) {
1381
+ if (p.type === "text") {
1382
+ const id = generateMessageId();
1383
+ writer.write({ type: "text-start", id });
1384
+ writer.write({ type: "text-delta", id, delta: p.text });
1385
+ writer.write({ type: "text-end", id });
1386
+ continue;
1387
+ }
1388
+ if (!p.type.startsWith("data-"))
1389
+ continue;
1390
+ const view = hooks.view(p, false, xrun);
1391
+ if (view && "hide" in view && view.hide)
1392
+ continue;
1393
+ const shown = view && "replace" in view && view.replace ? view.replace : p;
1394
+ writer.write({ type: shown.type, data: shown.data });
1395
+ }
1396
+ writer.write({ type: "finish" });
1397
+ return;
1398
+ }
1133
1399
  // Model projection: drops `sendToModel:false` messages (and data-* parts).
1134
- const modelMessages = await toModelMessages(uiMessages, { tools });
1400
+ // `context` lets an extension reshape what the model reads this turn.
1401
+ const projected = await toModelMessages(uiMessages, { tools });
1402
+ const modelMessages = hooks.has("context")
1403
+ ? await hooks.context(projected, xctx(0, uiMessages))
1404
+ : projected;
1405
+ // step_start for the first step — the SDK has no hook before its first call.
1406
+ control0 = stepHooks ? await hooks.stepStart(xctx(0, uiMessages)) : undefined;
1407
+ if (control0?.stop) {
1408
+ stoppedBeforeStart = true;
1409
+ writer.write({ type: "finish" });
1410
+ return;
1411
+ }
1135
1412
  // Per-session sandbox (one per chat) for agents that declared `sandbox`.
1136
1413
  // Exposed to sandbox tools as `options.experimental_sandbox`.
1137
1414
  const sandbox = wantSandbox
@@ -1195,6 +1472,18 @@ export function createRuntime(config) {
1195
1472
  toolResults: step.toolResults,
1196
1473
  }));
1197
1474
  }
1475
+ stepsSeen++;
1476
+ if (stepHooks && step.stepNumber !== undefined)
1477
+ slotOf(modelSide, step.stepNumber).resolve({
1478
+ usage: step.usage,
1479
+ finishReason: step.finishReason ?? "unknown",
1480
+ toolCalls: (step.toolCalls ?? []).map((c) => ({ toolName: c.toolName, toolCallId: c.toolCallId })),
1481
+ // The id the PROVIDER reports; pricing keys on that.
1482
+ model: {
1483
+ provider: step.model?.provider ?? "",
1484
+ modelId: step.response?.modelId ?? step.model?.modelId ?? modelId,
1485
+ },
1486
+ });
1198
1487
  },
1199
1488
  });
1200
1489
  const result = telemetryOn && config.telemetry?.withRunSpan
@@ -1235,24 +1524,61 @@ export function createRuntime(config) {
1235
1524
  // check inside the storage write itself — an adapter interface change,
1236
1525
  // tracked as a follow-up. With the heartbeat now aborting within one TTL
1237
1526
  // of losing contact, the exposure is bounded to ~leaseTtlMs.
1238
- onStepEnd: dur.enabled
1527
+ onStepEnd: dur.enabled || hooks.has("step_end")
1239
1528
  ? ({ responseMessage }) => {
1240
- if (abort.signal.aborted)
1529
+ if (hooks.has("step_end")) {
1530
+ // The step's parts are final only once `step_end` has appended
1531
+ // its own; `settle` writes the checkpoint after that.
1532
+ slotOf(uiSide, uiStepIndex).resolve(responseMessage);
1533
+ void settle(uiStepIndex++);
1241
1534
  return;
1242
- checkpoints = checkpoints
1243
- .then(() => {
1244
- // Re-checked at write time: the lease may have gone while this
1245
- // checkpoint waited behind the previous one.
1246
- if (abort.signal.aborted)
1247
- return;
1248
- return storage.appendMessages(resolvedChatId, [responseMessage]);
1249
- })
1250
- .catch((e) => console.warn(`[durability] step checkpoint failed for run ${run.id}:`, e));
1535
+ }
1536
+ checkpoint(responseMessage);
1251
1537
  }
1252
1538
  : undefined,
1253
- onFinish: async ({ messages, isAborted, }) => {
1539
+ onFinish: async ({ messages: finishedMessages, isAborted, }) => {
1254
1540
  stopHeartbeat();
1255
1541
  await closeSources();
1542
+ let messages = finishedMessages;
1543
+ if (stepHooks && failed === undefined && !isAborted) {
1544
+ // Every step's `step_end` has run before the record is settled. A
1545
+ // critical extension's throw on an EARLIER step surfaced through the
1546
+ // stream (prepareStep rethrows it); on the LAST step nothing awaited
1547
+ // it yet, so it lands here — and fails the run just the same.
1548
+ try {
1549
+ await Promise.all([...settledSteps.values()]);
1550
+ }
1551
+ catch (e) {
1552
+ failed = clientErrorMessage(e);
1553
+ }
1554
+ if (appendedByStep.size) {
1555
+ const i = messages.length - 1;
1556
+ const last = messages[i];
1557
+ if (last?.role === "assistant")
1558
+ messages = [...messages.slice(0, i), insertStepParts(last, appendedByStep, priorSteps)];
1559
+ }
1560
+ }
1561
+ if (stoppedBeforeStart || handled)
1562
+ // Nothing (or nothing but a command with no reply) was said: the SDK
1563
+ // still assembles an empty assistant message from the bare `finish`,
1564
+ // and there is no reason to store a blank turn.
1565
+ messages = messages.filter((m) => !(m.role === "assistant" && m.parts.length === 0));
1566
+ const agentEnd = (error) => hooks.has("agent_end")
1567
+ ? hooks.agentEnd({
1568
+ runId: run.id,
1569
+ principal: xrun.principal,
1570
+ messages,
1571
+ stepCount: stepsSeen,
1572
+ isAborted: isAborted === true,
1573
+ ...(error !== undefined ? { error } : {}),
1574
+ usage: {
1575
+ inputTokens,
1576
+ outputTokens,
1577
+ ...(cacheReadTokens ? { cacheReadTokens } : {}),
1578
+ ...(cacheWriteTokens ? { cacheWriteTokens } : {}),
1579
+ },
1580
+ })
1581
+ : Promise.resolve();
1256
1582
  // Cost for this turn (0 if no pricing configured). Stamp usage + cost
1257
1583
  // onto the assistant message — the message is complete; this is metadata.
1258
1584
  const cost = computeCost(config.pricing?.(modelId), {
@@ -1289,6 +1615,7 @@ export function createRuntime(config) {
1289
1615
  if (settled) {
1290
1616
  await settleLimits(admitted, { tokens: inputTokens + outputTokens, cost });
1291
1617
  await reportRunFinished("errored", messages, failed ?? "aborted");
1618
+ await agentEnd(failed ?? "aborted");
1292
1619
  }
1293
1620
  return;
1294
1621
  }
@@ -1333,13 +1660,17 @@ export function createRuntime(config) {
1333
1660
  // reaped and the run re-claimed elsewhere (`!applied`), that node will
1334
1661
  // finish it — a stale writer must not re-run the judge or emit a
1335
1662
  // duplicate "completed"/"awaiting_input" trace.
1336
- if (applied)
1663
+ if (applied) {
1337
1664
  await reportRunFinished(paused ? "awaiting_input" : "completed", messages);
1665
+ await agentEnd();
1666
+ }
1338
1667
  },
1339
1668
  onError: (error) => {
1340
1669
  stopHeartbeat();
1341
1670
  void closeSources();
1342
1671
  const message = clientErrorMessage(error);
1672
+ if (hooks.has("error"))
1673
+ void hooks.error({ error, stepNumber: stepsSeen, messages: transcriptNow() }, xrun);
1343
1674
  // Record only — the first error wins; later onError calls for the same
1344
1675
  // stream are noise. The SDK's onError must return the client string
1345
1676
  // synchronously and still runs onFinish afterwards, so that is where
@@ -1465,12 +1796,13 @@ export function createRuntime(config) {
1465
1796
  * persists nothing (the crash window between the two writes is the trade-off
1466
1797
  * an adapter without `openTurn` accepts).
1467
1798
  */
1468
- async function openTurn(principal, chatId, agentName, messages) {
1799
+ async function openTurn(principal, chatId, agentName, messages, agentVersion) {
1469
1800
  if (storage.openTurn) {
1470
1801
  const run = await storage.openTurn(principal, {
1471
1802
  chatId,
1472
1803
  agent: agentName,
1473
1804
  messages,
1805
+ agentVersion: agentVersion ?? null,
1474
1806
  ...(dur.enabled ? { lease: { owner: dur.instanceId, ttlMs: dur.leaseTtlMs } } : {}),
1475
1807
  });
1476
1808
  return {
@@ -1478,12 +1810,12 @@ export function createRuntime(config) {
1478
1810
  lease: dur.enabled ? { owner: dur.instanceId, fencingToken: run.fencingToken ?? 1 } : undefined,
1479
1811
  };
1480
1812
  }
1481
- const opened = await openRun(principal, chatId, agentName);
1813
+ const opened = await openRun(principal, chatId, agentName, agentVersion);
1482
1814
  if (messages.length > 0)
1483
1815
  await storage.appendMessages(chatId, messages);
1484
1816
  return opened;
1485
1817
  }
1486
- async function openRun(principal, chatId, agentName) {
1818
+ async function openRun(principal, chatId, agentName, agentVersion) {
1487
1819
  if (dur.enabled) {
1488
1820
  const run = await storage.claimRun({
1489
1821
  principal,
@@ -1491,11 +1823,17 @@ export function createRuntime(config) {
1491
1823
  agent: agentName,
1492
1824
  owner: dur.instanceId,
1493
1825
  ttlMs: dur.leaseTtlMs,
1826
+ agentVersion: agentVersion ?? null,
1494
1827
  });
1495
1828
  return { run, lease: { owner: dur.instanceId, fencingToken: run.fencingToken ?? 1 } };
1496
1829
  }
1497
1830
  return {
1498
- run: await storage.createRun(principal, { chatId, agent: agentName, kind: "turn" }),
1831
+ run: await storage.createRun(principal, {
1832
+ chatId,
1833
+ agent: agentName,
1834
+ kind: "turn",
1835
+ agentVersion: agentVersion ?? null,
1836
+ }),
1499
1837
  lease: undefined,
1500
1838
  };
1501
1839
  }
@@ -1530,7 +1868,7 @@ export function createRuntime(config) {
1530
1868
  // Run + message open together (atomically where the adapter can): a chat
1531
1869
  // with a live turn refuses (`ChatBusyError` → 409) and the message never
1532
1870
  // lands in history — the client retries it (with its turn reservation released).
1533
- const { run, lease } = await withReservation(admitted, () => openTurn(principal, resolvedChatId, agentName, [message]));
1871
+ const { run, lease } = await withReservation(admitted, () => openTurn(principal, resolvedChatId, agentName, [message], cfg.configVersion));
1534
1872
  return buildTurn({
1535
1873
  resolvedChatId,
1536
1874
  principal,
@@ -1607,6 +1945,9 @@ export function createRuntime(config) {
1607
1945
  modelId: ranAs(uiMessages, run.id) ?? "unknown",
1608
1946
  principal,
1609
1947
  startedAt: run.startedAt,
1948
+ // Likewise persisted, not re-resolved: reconciling a finished run must
1949
+ // report the version it ran on, not whatever is published now.
1950
+ agentVersion: run.agentVersion ?? undefined,
1610
1951
  depth: 0,
1611
1952
  trigger: "resume",
1612
1953
  rootChatId: run.chatId,
@@ -1800,7 +2141,7 @@ export function createRuntime(config) {
1800
2141
  }
1801
2142
  const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
1802
2143
  const admitted = await admitContinuation(principal, chat.agent, "decision");
1803
- const { run, lease } = await openRun(principal, chatId, chat.agent);
2144
+ const { run, lease } = await openRun(principal, chatId, chat.agent, cfg.configVersion);
1804
2145
  return buildTurn({
1805
2146
  resolvedChatId: chatId,
1806
2147
  principal,
@@ -1852,7 +2193,7 @@ export function createRuntime(config) {
1852
2193
  // a busy chat refuses before anything is recorded, so the retry finds the
1853
2194
  // call still unanswered and resumes it — persisting first would make the
1854
2195
  // retry a "duplicate submit" no-op and lose the answer.
1855
- const { run, lease } = await openTurn(principal, chatId, chat.agent, changed);
2196
+ const { run, lease } = await openTurn(principal, chatId, chat.agent, changed, cfg.configVersion);
1856
2197
  return buildTurn({
1857
2198
  resolvedChatId: chatId,
1858
2199
  principal,
@@ -1998,7 +2339,11 @@ export function createRuntime(config) {
1998
2339
  }
1999
2340
  async function loadHistory({ chatId, principal, }) {
2000
2341
  const messages = await storage.loadMessages(principal, chatId);
2001
- return toClientMessages(messages);
2342
+ // `render` rules apply on load as they do live. Runtime-wide extensions
2343
+ // only: an agent's own would need the chat's agent resolved here.
2344
+ return renderTranscript(config.extensions ?? [], toClientMessages(messages), {
2345
+ principal: principal,
2346
+ });
2002
2347
  }
2003
2348
  async function listChats({ principal, limit, kind, }) {
2004
2349
  // No default kind filter — passing nothing returns ALL chats (unchanged
@@ -2106,7 +2451,7 @@ export function createRuntime(config) {
2106
2451
  const admitted = await admitTurn(o.principal, o.agent, o.trigger);
2107
2452
  const { cfg, modelId, runtimeCtx, run, lease } = await withReservation(admitted, async () => {
2108
2453
  const resolved = await resolveAgentConfig(o.agent, o.principal, undefined, o.turnContext);
2109
- const opened = await openTurn(o.principal, chatId, o.agent, [o.message]);
2454
+ const opened = await openTurn(o.principal, chatId, o.agent, [o.message], resolved.cfg.configVersion);
2110
2455
  return { ...resolved, ...opened };
2111
2456
  });
2112
2457
  const onProgress = o.onProgress;
@@ -2192,7 +2537,7 @@ export function createRuntime(config) {
2192
2537
  return { answer: "" };
2193
2538
  const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(subChat.agent, principal);
2194
2539
  const admitted = await admitContinuation(principal, subChat.agent, "decision");
2195
- const { run, lease } = await openRun(principal, subChatId, subChat.agent);
2540
+ const { run, lease } = await openRun(principal, subChatId, subChat.agent, cfg.configVersion);
2196
2541
  const res = await buildTurn({
2197
2542
  resolvedChatId: subChatId,
2198
2543
  principal,