@intentface/latch-core 0.9.1 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/runtime.js CHANGED
@@ -5,6 +5,7 @@ import { summarizeForCompaction } from "./compaction.js";
5
5
  import { computeCost } from "./pricing.js";
6
6
  import { defaultPromptCachingPlan, markLastFunctionTool, mergeProviderOptions, toolsetHash, } from "./prompt-caching.js";
7
7
  import { LimitExceededError, exceededPolicy, remainingTokens, windowRef, } from "./limits.js";
8
+ import { compose, fireClientToolResults, insertStepParts, isNewSession, lastUserTextOf, renderTranscript, replaceLastUserText, wrapTools, } from "./extensions/index.js";
8
9
  /** Ids for assistant response messages — `msg_<random>`, matching storage ids. */
9
10
  const generateMessageId = createIdGenerator({ prefix: "msg", separator: "_" });
10
11
  /** Sub-thread chat ids start with `sub_` so the host can keep them out of the
@@ -575,7 +576,10 @@ export function createRuntime(config) {
575
576
  * server-side (resume).
576
577
  */
577
578
  async function buildTurn(args) {
578
- const { resolvedChatId, principal, modelId, cfg, run, lease, uiMessages, runtimeCtx } = args;
579
+ const { resolvedChatId, principal, modelId, cfg, run, lease, runtimeCtx } = args;
580
+ // Mutable: the start moments below may inject, replace or rewrite history
581
+ // before the model runs. From here on this is the turn's transcript.
582
+ let uiMessages = args.uiMessages;
579
583
  const admitted = args.admitted;
580
584
  const autoApprove = args.autoApprove ?? false;
581
585
  const blockGated = args.blockGated ?? false;
@@ -585,6 +589,49 @@ export function createRuntime(config) {
585
589
  // threaded a root down. Subagent turns spawned below inherit it, so a whole
586
590
  // delegation tree shares one telemetry session (see telemetryMeta).
587
591
  const rootChatId = args.rootChatId ?? resolvedChatId;
592
+ // --- the extension seam --------------------------------------------------
593
+ // Runtime-wide extensions first, then the agent's own. With none attached
594
+ // every moment below is a no-op and the turn is byte-identical to a runtime
595
+ // without the seam — `turn-snapshot.test.ts` holds it to that.
596
+ const hooks = compose([...(config.extensions ?? []), ...(cfg.extensions ?? [])]);
597
+ const xrun = {
598
+ runId: run.id,
599
+ principal: principal,
600
+ };
601
+ const xctx = (stepNumber, messages) => ({
602
+ ...xrun,
603
+ stepNumber,
604
+ messages,
605
+ });
606
+ // A command an extension answered without the model (see the `input` moment).
607
+ let handled;
608
+ // The start moments run once per conversation turn. A resume replays a
609
+ // history they already shaped, and a decision continues a turn they
610
+ // already opened — running them again would inject twice.
611
+ const startMoments = trigger !== "resume" && trigger !== "decision";
612
+ if (startMoments) {
613
+ if (isNewSession(uiMessages))
614
+ uiMessages = (await hooks.messagesIn("session_start", uiMessages, xrun));
615
+ uiMessages = (await hooks.messagesIn("before_agent_start", uiMessages, xrun));
616
+ const text = lastUserTextOf(uiMessages);
617
+ if (text != null) {
618
+ const e = await hooks.input(text, uiMessages, xrun);
619
+ if (e && "handled" in e && e.handled) {
620
+ const reply = e.reply;
621
+ handled =
622
+ typeof reply === "string"
623
+ ? [{ type: "text", text: reply }]
624
+ : Array.isArray(reply)
625
+ ? reply
626
+ : [];
627
+ if (e.replace)
628
+ uiMessages = e.replace;
629
+ }
630
+ else if (e && "transform" in e && e.transform != null) {
631
+ uiMessages = replaceLastUserText(uiMessages, e.transform);
632
+ }
633
+ }
634
+ }
588
635
  const abort = new AbortController();
589
636
  let heartbeat;
590
637
  const stopHeartbeat = () => {
@@ -648,7 +695,9 @@ export function createRuntime(config) {
648
695
  }
649
696
  const opened = [];
650
697
  try {
651
- for (const source of sources) {
698
+ // A command answered without the model needs no tools; don't open MCP
699
+ // connections for `/stats`.
700
+ for (const source of handled ? [] : sources) {
652
701
  opened.push(await source.open(principal));
653
702
  }
654
703
  }
@@ -746,6 +795,13 @@ export function createRuntime(config) {
746
795
  // Memory is optional and degradable — continue without it.
747
796
  }
748
797
  }
798
+ // instructions: an extension's appends land after the memory block; the
799
+ // date line stays the tail (see the agent construction). A `replace` may
800
+ // hand back system message(s) carrying providerOptions — the extension then
801
+ // owns that shape, and the caching plan's system breakpoint steps aside.
802
+ const composedInstructions = hooks.has("instructions")
803
+ ? await hooks.instructions(instructions, xctx(0, uiMessages))
804
+ : instructions;
749
805
  for (const name of cfg.disableTools ?? [])
750
806
  delete tools[name];
751
807
  // Per-agent tool description overrides — swap the resolved tool's
@@ -877,6 +933,27 @@ export function createRuntime(config) {
877
933
  });
878
934
  capabilityTools.push("spawn_agent");
879
935
  }
936
+ // register_tools: what extensions add are capabilities — always active, and
937
+ // gated by the agent's own policy like any other tool. Mutated in place so
938
+ // every later reference (toolsContext keys, the fingerprint) sees them.
939
+ if (hooks.has("register_tools")) {
940
+ const before = new Set(Object.keys(tools));
941
+ const registered = hooks.tools(tools, xrun);
942
+ for (const k of Object.keys(tools))
943
+ if (!(k in registered))
944
+ delete tools[k];
945
+ Object.assign(tools, registered);
946
+ for (const name of Object.keys(tools))
947
+ if (!before.has(name))
948
+ capabilityTools.push(name);
949
+ }
950
+ // Outputs a CLIENT produced (a browser tool's answer) are already in the
951
+ // transcript; fire tool_result for them now that the toolset can tell a
952
+ // client tool from a server one.
953
+ // A decision turn is exactly when a client's answer has just been recorded,
954
+ // so it fires here too; a resume replays outputs an earlier run handled.
955
+ if ((startMoments || trigger === "decision") && hooks.has("tool_result"))
956
+ uiMessages = (await fireClientToolResults(uiMessages, tools, hooks, xrun));
880
957
  // Subagent runs are autonomous — there's no human to authorize, so waive
881
958
  // approvals and drop the OAuth `connect_` gates (they'd pause forever).
882
959
  if (autoApprove) {
@@ -933,8 +1010,19 @@ export function createRuntime(config) {
933
1010
  // doesn't mention keep the default. Only for the object form (a function
934
1011
  // policy is left as-is).
935
1012
  const sourceApproval = Object.assign({}, harnessApproval, ...opened.map((o) => o.toolApproval ?? {}));
1013
+ // An unattended run waives the AGENT's own policy — that is what makes a
1014
+ // delegated or programmatic turn autonomous. It does NOT waive the gates a
1015
+ // TOOL SOURCE contributed: a connection's gates are the end user's choice
1016
+ // about that connector ("deletes here must ask me"), and a harness gate is
1017
+ // the host's about that tool. Dropping those was the one thing they were
1018
+ // set to prevent, so they survive and the turn parks instead — a first-class
1019
+ // outcome (`runDue` reports `parked`; a subagent's gate bubbles to its
1020
+ // parent). `blockGated` is the opt-out: it has already stubbed those tools,
1021
+ // so gating them too would park a trial run that is meant to complete.
936
1022
  const toolApproval = autoApprove
937
- ? undefined
1023
+ ? blockGated || Object.keys(sourceApproval).length === 0
1024
+ ? undefined
1025
+ : sourceApproval
938
1026
  : cfg.toolApproval && typeof cfg.toolApproval !== "function"
939
1027
  ? { ...sourceApproval, ...cfg.toolApproval }
940
1028
  : Object.keys(sourceApproval).length > 0 && !cfg.toolApproval
@@ -968,7 +1056,64 @@ export function createRuntime(config) {
968
1056
  ? ({ steps }) => steps.reduce((n, st) => n + (st.usage?.inputTokens ?? 0) + (st.usage?.outputTokens ?? 0), 0) >=
969
1057
  tokenBudget
970
1058
  : undefined;
971
- const stopWhen = budgetStop ? [stepCap ?? stepCountIs(20), budgetStop] : stepCap;
1059
+ const baseStopWhen = budgetStop ? [stepCap ?? stepCountIs(20), budgetStop] : stepCap;
1060
+ // --- step boundaries for the seam -------------------------------------------
1061
+ // A step ends twice from here: on the model side (`agent.stream`'s
1062
+ // onStepEnd — usage, finish reason, tool calls) and on the UI side (the
1063
+ // stream's onStepEnd — the assembled assistant message). `step_end` needs
1064
+ // both, so a step is SETTLED once both have arrived, and the next step's
1065
+ // prepareStep waits for the settlement: an extension that persists has
1066
+ // written before the model is called again. None of this runs when no
1067
+ // extension listens to a step moment.
1068
+ const stepHooks = hooks.has("step_start") || hooks.has("step_end");
1069
+ const slotOf = (m, k) => {
1070
+ let d = m.get(k);
1071
+ if (!d) {
1072
+ let resolve;
1073
+ const promise = new Promise((r) => (resolve = r));
1074
+ d = { promise, resolve };
1075
+ m.set(k, d);
1076
+ }
1077
+ return d;
1078
+ };
1079
+ const modelSide = new Map();
1080
+ const uiSide = new Map();
1081
+ const settledSteps = new Map();
1082
+ const appendedByStep = new Map();
1083
+ const stepStartedAt = new Map();
1084
+ let control0;
1085
+ let nextControl;
1086
+ let stoppedBeforeStart = false;
1087
+ let uiStepIndex = 0;
1088
+ let stepsSeen = 0;
1089
+ let turnWriter;
1090
+ let latestAssistant;
1091
+ // A continuation reuses the assistant message the SDK merges into; its
1092
+ // earlier step-start parts are not this run's steps.
1093
+ const anchor = uiMessages[uiMessages.length - 1];
1094
+ const anchorIsAssistant = anchor?.role === "assistant";
1095
+ const priorSteps = anchorIsAssistant
1096
+ ? anchor.parts.filter((p) => p.type === "step-start").length
1097
+ : 0;
1098
+ const transcriptNow = () => latestAssistant
1099
+ ? anchorIsAssistant
1100
+ ? [...uiMessages.slice(0, -1), latestAssistant]
1101
+ : [...uiMessages, latestAssistant]
1102
+ : uiMessages;
1103
+ // Assigned below, once the checkpoint chain exists; only ever called mid-stream.
1104
+ let settle = async () => { };
1105
+ const extensionStop = stepHooks
1106
+ ? async ({ steps }) => {
1107
+ const k = steps.length - 1;
1108
+ await settle(k);
1109
+ const c = await hooks.stepStart(xctx(steps.length, transcriptNow()));
1110
+ nextControl = c;
1111
+ return c.stop;
1112
+ }
1113
+ : undefined;
1114
+ const stopWhen = extensionStop
1115
+ ? [...(Array.isArray(baseStopWhen) ? baseStopWhen : [baseStopWhen ?? stepCountIs(20)]), extensionStop]
1116
+ : baseStopWhen;
972
1117
  // Reasoning/thinking effort → provider-specific options (the platform maps
973
1118
  // it per provider + gates to reasoning-capable models). Called even without
974
1119
  // an effort so the hook can set safe defaults (e.g. lift a provider's tiny
@@ -1001,6 +1146,14 @@ export function createRuntime(config) {
1001
1146
  // run so churn across a chat's turns (a cache invalidator) is queryable.
1002
1147
  const toolsHash = toolsetHash(activeTools ?? Object.keys(tools));
1003
1148
  const providerOptions = mergeProviderOptions(reasoning?.providerOptions, caching?.requestProviderOptions);
1149
+ // tool_call / tool_result: wrap in place, after the caching marker and the
1150
+ // test-run stubs, so every later reference sees the same names. Identity-
1151
+ // preserving when no extension listens.
1152
+ if (hooks.has("tool_call") || hooks.has("tool_result")) {
1153
+ const wrapped = wrapTools(tools, hooks, xrun);
1154
+ for (const k of Object.keys(wrapped))
1155
+ tools[k] = wrapped[k];
1156
+ }
1004
1157
  // Telemetry seam: per-run AI SDK telemetry options + trace correlation
1005
1158
  // identity, resolved once per turn (undefined → nothing traced).
1006
1159
  const telemetryMeta = {
@@ -1010,6 +1163,7 @@ export function createRuntime(config) {
1010
1163
  modelId,
1011
1164
  principal,
1012
1165
  startedAt: run.startedAt,
1166
+ agentVersion: cfg.configVersion,
1013
1167
  depth,
1014
1168
  trigger,
1015
1169
  parentRunId: args.parentRunId,
@@ -1032,24 +1186,46 @@ export function createRuntime(config) {
1032
1186
  // the instructions become a MARKED system message and the date line a
1033
1187
  // separate unmarked one, so the daily flip lands after the breakpoint
1034
1188
  // instead of invalidating it at midnight UTC.
1035
- instructions: caching?.systemProviderOptions
1036
- ? [
1037
- ...(instructions
1038
- ? [
1039
- {
1040
- role: "system",
1041
- content: instructions,
1042
- providerOptions: caching.systemProviderOptions,
1043
- },
1044
- ]
1045
- : []),
1046
- { role: "system", content: currentDateLine() },
1047
- ]
1048
- : [instructions, currentDateLine()].filter(Boolean).join("\n\n"),
1189
+ instructions: typeof composedInstructions === "object" && composedInstructions !== null
1190
+ ? // An extension replaced the system prompt with message(s) of its own,
1191
+ // providerOptions and all: keep them, with the date line as its own
1192
+ // trailing message so the daily flip stays past any breakpoint.
1193
+ [
1194
+ ...(Array.isArray(composedInstructions) ? composedInstructions : [composedInstructions]),
1195
+ { role: "system", content: currentDateLine() },
1196
+ ]
1197
+ : caching?.systemProviderOptions
1198
+ ? [
1199
+ ...(composedInstructions
1200
+ ? [
1201
+ {
1202
+ role: "system",
1203
+ content: composedInstructions,
1204
+ providerOptions: caching.systemProviderOptions,
1205
+ },
1206
+ ]
1207
+ : []),
1208
+ { role: "system", content: currentDateLine() },
1209
+ ]
1210
+ : [composedInstructions, currentDateLine()].filter(Boolean).join("\n\n"),
1049
1211
  tools,
1050
1212
  toolApproval,
1051
1213
  stopWhen,
1052
1214
  activeTools,
1215
+ ...(stepHooks
1216
+ ? {
1217
+ prepareStep: (async ({ stepNumber }) => {
1218
+ if (stepNumber > 0)
1219
+ await settle(stepNumber - 1);
1220
+ stepStartedAt.set(stepNumber, Date.now());
1221
+ const c = stepNumber === 0 ? control0 : nextControl;
1222
+ return {
1223
+ ...(c?.model ? { model: c.model } : {}),
1224
+ ...(c?.activeTools ? { activeTools: c.activeTools } : {}),
1225
+ };
1226
+ }),
1227
+ }
1228
+ : {}),
1053
1229
  ...(providerOptions ? { providerOptions } : {}),
1054
1230
  ...(reasoning?.maxOutputTokens ? { maxOutputTokens: reasoning.maxOutputTokens } : {}),
1055
1231
  ...(telemetryOptions ? { telemetry: telemetryOptions } : {}),
@@ -1076,6 +1252,80 @@ export function createRuntime(config) {
1076
1252
  // at most the write in flight and the one queued behind it. A failed write
1077
1253
  // logs and releases the chain — the next checkpoint still runs.
1078
1254
  let checkpoints = Promise.resolve();
1255
+ const checkpoint = (message) => {
1256
+ if (!dur.enabled || abort.signal.aborted)
1257
+ return;
1258
+ checkpoints = checkpoints
1259
+ .then(() => {
1260
+ // Re-checked at write time: the lease may have gone while this
1261
+ // checkpoint waited behind the previous one.
1262
+ if (abort.signal.aborted)
1263
+ return;
1264
+ return storage.appendMessages(resolvedChatId, [message]);
1265
+ })
1266
+ .catch((e) => console.warn(`[durability] step checkpoint failed for run ${run.id}:`, e));
1267
+ };
1268
+ if (stepHooks) {
1269
+ const needUi = hooks.has("step_end");
1270
+ settle = (k) => {
1271
+ const known = settledSteps.get(k);
1272
+ if (known)
1273
+ return known;
1274
+ const p = (async () => {
1275
+ const model = await slotOf(modelSide, k).promise;
1276
+ const response = needUi ? await slotOf(uiSide, k).promise : undefined;
1277
+ const raw = model.usage;
1278
+ // The SDK assembles the message from chunks; our parts were streamed
1279
+ // transient and never entered it. Fold every step's back in.
1280
+ const assistant = response
1281
+ ? insertStepParts(response, appendedByStep, priorSteps)
1282
+ : undefined;
1283
+ const messages = assistant
1284
+ ? anchorIsAssistant
1285
+ ? [...uiMessages.slice(0, -1), assistant]
1286
+ : [...uiMessages, assistant]
1287
+ : transcriptNow();
1288
+ const info = {
1289
+ ...xctx(k, messages),
1290
+ usage: {
1291
+ inputTokens: raw?.inputTokens,
1292
+ outputTokens: raw?.outputTokens,
1293
+ totalTokens: raw?.totalTokens,
1294
+ cacheReadTokens: raw?.inputTokenDetails?.cacheReadTokens,
1295
+ cacheWriteTokens: raw?.inputTokenDetails?.cacheWriteTokens,
1296
+ reasoningTokens: raw?.outputTokenDetails?.reasoningTokens,
1297
+ },
1298
+ finishReason: model.finishReason,
1299
+ model: model.model,
1300
+ durationMs: Date.now() - (stepStartedAt.get(k) ?? Date.now()),
1301
+ toolCalls: model.toolCalls,
1302
+ };
1303
+ if (needUi) {
1304
+ const appended = await hooks.stepEnd(messages, info);
1305
+ if (appended.length) {
1306
+ appendedByStep.set(k, appended);
1307
+ // The client copy goes through the view rules and is TRANSIENT, so
1308
+ // the SDK's own merge cannot persist a viewed copy over the truth,
1309
+ // which is folded into the checkpoint and the finish below.
1310
+ for (const part of appended) {
1311
+ if (!part.type.startsWith("data-"))
1312
+ continue;
1313
+ const view = hooks.view(part, false, xrun);
1314
+ if (view && "hide" in view && view.hide)
1315
+ continue;
1316
+ const shown = view && "replace" in view && view.replace ? view.replace : part;
1317
+ turnWriter?.write({ type: shown.type, data: shown.data, transient: true });
1318
+ }
1319
+ }
1320
+ latestAssistant = insertStepParts(response, appendedByStep, priorSteps);
1321
+ checkpoint(latestAssistant);
1322
+ }
1323
+ })();
1324
+ settledSteps.set(k, p);
1325
+ p.catch(() => { }); // surfaces where it is awaited: prepareStep, stopWhen, onFinish
1326
+ return p;
1327
+ };
1328
+ }
1079
1329
  // Post-turn telemetry: fired after persistence on every terminal path
1080
1330
  // (completed / awaiting_input / errored). Must never fail the turn.
1081
1331
  const reportRunFinished = async (status, messages, error) => {
@@ -1128,10 +1378,48 @@ export function createRuntime(config) {
1128
1378
  // turn would upsert onto the same row — see the dup-id regression test).
1129
1379
  generateId: generateMessageId,
1130
1380
  execute: async ({ writer }) => {
1131
- // (in-turn writer writes — e.g. a "compacting…" / compaction marker —
1132
- // would go here, before the model output is merged in)
1381
+ turnWriter = writer;
1382
+ if (handled) {
1383
+ // A command an extension answered without the model. Emitted as an
1384
+ // ordinary assistant turn so the conversation reads normally and the
1385
+ // finish hook persists it like any other.
1386
+ writer.write({
1387
+ type: "start",
1388
+ messageId: generateMessageId(),
1389
+ messageMetadata: { visibility: "user", model: modelId, createdAt: Date.now(), runId: run.id },
1390
+ });
1391
+ for (const p of handled) {
1392
+ if (p.type === "text") {
1393
+ const id = generateMessageId();
1394
+ writer.write({ type: "text-start", id });
1395
+ writer.write({ type: "text-delta", id, delta: p.text });
1396
+ writer.write({ type: "text-end", id });
1397
+ continue;
1398
+ }
1399
+ if (!p.type.startsWith("data-"))
1400
+ continue;
1401
+ const view = hooks.view(p, false, xrun);
1402
+ if (view && "hide" in view && view.hide)
1403
+ continue;
1404
+ const shown = view && "replace" in view && view.replace ? view.replace : p;
1405
+ writer.write({ type: shown.type, data: shown.data });
1406
+ }
1407
+ writer.write({ type: "finish" });
1408
+ return;
1409
+ }
1133
1410
  // Model projection: drops `sendToModel:false` messages (and data-* parts).
1134
- const modelMessages = await toModelMessages(uiMessages, { tools });
1411
+ // `context` lets an extension reshape what the model reads this turn.
1412
+ const projected = await toModelMessages(uiMessages, { tools });
1413
+ const modelMessages = hooks.has("context")
1414
+ ? await hooks.context(projected, xctx(0, uiMessages))
1415
+ : projected;
1416
+ // step_start for the first step — the SDK has no hook before its first call.
1417
+ control0 = stepHooks ? await hooks.stepStart(xctx(0, uiMessages)) : undefined;
1418
+ if (control0?.stop) {
1419
+ stoppedBeforeStart = true;
1420
+ writer.write({ type: "finish" });
1421
+ return;
1422
+ }
1135
1423
  // Per-session sandbox (one per chat) for agents that declared `sandbox`.
1136
1424
  // Exposed to sandbox tools as `options.experimental_sandbox`.
1137
1425
  const sandbox = wantSandbox
@@ -1195,6 +1483,18 @@ export function createRuntime(config) {
1195
1483
  toolResults: step.toolResults,
1196
1484
  }));
1197
1485
  }
1486
+ stepsSeen++;
1487
+ if (stepHooks && step.stepNumber !== undefined)
1488
+ slotOf(modelSide, step.stepNumber).resolve({
1489
+ usage: step.usage,
1490
+ finishReason: step.finishReason ?? "unknown",
1491
+ toolCalls: (step.toolCalls ?? []).map((c) => ({ toolName: c.toolName, toolCallId: c.toolCallId })),
1492
+ // The id the PROVIDER reports; pricing keys on that.
1493
+ model: {
1494
+ provider: step.model?.provider ?? "",
1495
+ modelId: step.response?.modelId ?? step.model?.modelId ?? modelId,
1496
+ },
1497
+ });
1198
1498
  },
1199
1499
  });
1200
1500
  const result = telemetryOn && config.telemetry?.withRunSpan
@@ -1235,24 +1535,61 @@ export function createRuntime(config) {
1235
1535
  // check inside the storage write itself — an adapter interface change,
1236
1536
  // tracked as a follow-up. With the heartbeat now aborting within one TTL
1237
1537
  // of losing contact, the exposure is bounded to ~leaseTtlMs.
1238
- onStepEnd: dur.enabled
1538
+ onStepEnd: dur.enabled || hooks.has("step_end")
1239
1539
  ? ({ responseMessage }) => {
1240
- if (abort.signal.aborted)
1540
+ if (hooks.has("step_end")) {
1541
+ // The step's parts are final only once `step_end` has appended
1542
+ // its own; `settle` writes the checkpoint after that.
1543
+ slotOf(uiSide, uiStepIndex).resolve(responseMessage);
1544
+ void settle(uiStepIndex++);
1241
1545
  return;
1242
- checkpoints = checkpoints
1243
- .then(() => {
1244
- // Re-checked at write time: the lease may have gone while this
1245
- // checkpoint waited behind the previous one.
1246
- if (abort.signal.aborted)
1247
- return;
1248
- return storage.appendMessages(resolvedChatId, [responseMessage]);
1249
- })
1250
- .catch((e) => console.warn(`[durability] step checkpoint failed for run ${run.id}:`, e));
1546
+ }
1547
+ checkpoint(responseMessage);
1251
1548
  }
1252
1549
  : undefined,
1253
- onFinish: async ({ messages, isAborted, }) => {
1550
+ onFinish: async ({ messages: finishedMessages, isAborted, }) => {
1254
1551
  stopHeartbeat();
1255
1552
  await closeSources();
1553
+ let messages = finishedMessages;
1554
+ if (stepHooks && failed === undefined && !isAborted) {
1555
+ // Every step's `step_end` has run before the record is settled. A
1556
+ // critical extension's throw on an EARLIER step surfaced through the
1557
+ // stream (prepareStep rethrows it); on the LAST step nothing awaited
1558
+ // it yet, so it lands here — and fails the run just the same.
1559
+ try {
1560
+ await Promise.all([...settledSteps.values()]);
1561
+ }
1562
+ catch (e) {
1563
+ failed = clientErrorMessage(e);
1564
+ }
1565
+ if (appendedByStep.size) {
1566
+ const i = messages.length - 1;
1567
+ const last = messages[i];
1568
+ if (last?.role === "assistant")
1569
+ messages = [...messages.slice(0, i), insertStepParts(last, appendedByStep, priorSteps)];
1570
+ }
1571
+ }
1572
+ if (stoppedBeforeStart || handled)
1573
+ // Nothing (or nothing but a command with no reply) was said: the SDK
1574
+ // still assembles an empty assistant message from the bare `finish`,
1575
+ // and there is no reason to store a blank turn.
1576
+ messages = messages.filter((m) => !(m.role === "assistant" && m.parts.length === 0));
1577
+ const agentEnd = (error) => hooks.has("agent_end")
1578
+ ? hooks.agentEnd({
1579
+ runId: run.id,
1580
+ principal: xrun.principal,
1581
+ messages,
1582
+ stepCount: stepsSeen,
1583
+ isAborted: isAborted === true,
1584
+ ...(error !== undefined ? { error } : {}),
1585
+ usage: {
1586
+ inputTokens,
1587
+ outputTokens,
1588
+ ...(cacheReadTokens ? { cacheReadTokens } : {}),
1589
+ ...(cacheWriteTokens ? { cacheWriteTokens } : {}),
1590
+ },
1591
+ })
1592
+ : Promise.resolve();
1256
1593
  // Cost for this turn (0 if no pricing configured). Stamp usage + cost
1257
1594
  // onto the assistant message — the message is complete; this is metadata.
1258
1595
  const cost = computeCost(config.pricing?.(modelId), {
@@ -1289,6 +1626,7 @@ export function createRuntime(config) {
1289
1626
  if (settled) {
1290
1627
  await settleLimits(admitted, { tokens: inputTokens + outputTokens, cost });
1291
1628
  await reportRunFinished("errored", messages, failed ?? "aborted");
1629
+ await agentEnd(failed ?? "aborted");
1292
1630
  }
1293
1631
  return;
1294
1632
  }
@@ -1333,13 +1671,17 @@ export function createRuntime(config) {
1333
1671
  // reaped and the run re-claimed elsewhere (`!applied`), that node will
1334
1672
  // finish it — a stale writer must not re-run the judge or emit a
1335
1673
  // duplicate "completed"/"awaiting_input" trace.
1336
- if (applied)
1674
+ if (applied) {
1337
1675
  await reportRunFinished(paused ? "awaiting_input" : "completed", messages);
1676
+ await agentEnd();
1677
+ }
1338
1678
  },
1339
1679
  onError: (error) => {
1340
1680
  stopHeartbeat();
1341
1681
  void closeSources();
1342
1682
  const message = clientErrorMessage(error);
1683
+ if (hooks.has("error"))
1684
+ void hooks.error({ error, stepNumber: stepsSeen, messages: transcriptNow() }, xrun);
1343
1685
  // Record only — the first error wins; later onError calls for the same
1344
1686
  // stream are noise. The SDK's onError must return the client string
1345
1687
  // synchronously and still runs onFinish afterwards, so that is where
@@ -1465,12 +1807,13 @@ export function createRuntime(config) {
1465
1807
  * persists nothing (the crash window between the two writes is the trade-off
1466
1808
  * an adapter without `openTurn` accepts).
1467
1809
  */
1468
- async function openTurn(principal, chatId, agentName, messages) {
1810
+ async function openTurn(principal, chatId, agentName, messages, agentVersion) {
1469
1811
  if (storage.openTurn) {
1470
1812
  const run = await storage.openTurn(principal, {
1471
1813
  chatId,
1472
1814
  agent: agentName,
1473
1815
  messages,
1816
+ agentVersion: agentVersion ?? null,
1474
1817
  ...(dur.enabled ? { lease: { owner: dur.instanceId, ttlMs: dur.leaseTtlMs } } : {}),
1475
1818
  });
1476
1819
  return {
@@ -1478,12 +1821,12 @@ export function createRuntime(config) {
1478
1821
  lease: dur.enabled ? { owner: dur.instanceId, fencingToken: run.fencingToken ?? 1 } : undefined,
1479
1822
  };
1480
1823
  }
1481
- const opened = await openRun(principal, chatId, agentName);
1824
+ const opened = await openRun(principal, chatId, agentName, agentVersion);
1482
1825
  if (messages.length > 0)
1483
1826
  await storage.appendMessages(chatId, messages);
1484
1827
  return opened;
1485
1828
  }
1486
- async function openRun(principal, chatId, agentName) {
1829
+ async function openRun(principal, chatId, agentName, agentVersion) {
1487
1830
  if (dur.enabled) {
1488
1831
  const run = await storage.claimRun({
1489
1832
  principal,
@@ -1491,11 +1834,17 @@ export function createRuntime(config) {
1491
1834
  agent: agentName,
1492
1835
  owner: dur.instanceId,
1493
1836
  ttlMs: dur.leaseTtlMs,
1837
+ agentVersion: agentVersion ?? null,
1494
1838
  });
1495
1839
  return { run, lease: { owner: dur.instanceId, fencingToken: run.fencingToken ?? 1 } };
1496
1840
  }
1497
1841
  return {
1498
- run: await storage.createRun(principal, { chatId, agent: agentName, kind: "turn" }),
1842
+ run: await storage.createRun(principal, {
1843
+ chatId,
1844
+ agent: agentName,
1845
+ kind: "turn",
1846
+ agentVersion: agentVersion ?? null,
1847
+ }),
1499
1848
  lease: undefined,
1500
1849
  };
1501
1850
  }
@@ -1530,7 +1879,7 @@ export function createRuntime(config) {
1530
1879
  // Run + message open together (atomically where the adapter can): a chat
1531
1880
  // with a live turn refuses (`ChatBusyError` → 409) and the message never
1532
1881
  // lands in history — the client retries it (with its turn reservation released).
1533
- const { run, lease } = await withReservation(admitted, () => openTurn(principal, resolvedChatId, agentName, [message]));
1882
+ const { run, lease } = await withReservation(admitted, () => openTurn(principal, resolvedChatId, agentName, [message], cfg.configVersion));
1534
1883
  return buildTurn({
1535
1884
  resolvedChatId,
1536
1885
  principal,
@@ -1607,6 +1956,9 @@ export function createRuntime(config) {
1607
1956
  modelId: ranAs(uiMessages, run.id) ?? "unknown",
1608
1957
  principal,
1609
1958
  startedAt: run.startedAt,
1959
+ // Likewise persisted, not re-resolved: reconciling a finished run must
1960
+ // report the version it ran on, not whatever is published now.
1961
+ agentVersion: run.agentVersion ?? undefined,
1610
1962
  depth: 0,
1611
1963
  trigger: "resume",
1612
1964
  rootChatId: run.chatId,
@@ -1800,7 +2152,7 @@ export function createRuntime(config) {
1800
2152
  }
1801
2153
  const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
1802
2154
  const admitted = await admitContinuation(principal, chat.agent, "decision");
1803
- const { run, lease } = await openRun(principal, chatId, chat.agent);
2155
+ const { run, lease } = await openRun(principal, chatId, chat.agent, cfg.configVersion);
1804
2156
  return buildTurn({
1805
2157
  resolvedChatId: chatId,
1806
2158
  principal,
@@ -1852,7 +2204,7 @@ export function createRuntime(config) {
1852
2204
  // a busy chat refuses before anything is recorded, so the retry finds the
1853
2205
  // call still unanswered and resumes it — persisting first would make the
1854
2206
  // retry a "duplicate submit" no-op and lose the answer.
1855
- const { run, lease } = await openTurn(principal, chatId, chat.agent, changed);
2207
+ const { run, lease } = await openTurn(principal, chatId, chat.agent, changed, cfg.configVersion);
1856
2208
  return buildTurn({
1857
2209
  resolvedChatId: chatId,
1858
2210
  principal,
@@ -1998,7 +2350,11 @@ export function createRuntime(config) {
1998
2350
  }
1999
2351
  async function loadHistory({ chatId, principal, }) {
2000
2352
  const messages = await storage.loadMessages(principal, chatId);
2001
- return toClientMessages(messages);
2353
+ // `render` rules apply on load as they do live. Runtime-wide extensions
2354
+ // only: an agent's own would need the chat's agent resolved here.
2355
+ return renderTranscript(config.extensions ?? [], toClientMessages(messages), {
2356
+ principal: principal,
2357
+ });
2002
2358
  }
2003
2359
  async function listChats({ principal, limit, kind, }) {
2004
2360
  // No default kind filter — passing nothing returns ALL chats (unchanged
@@ -2106,7 +2462,7 @@ export function createRuntime(config) {
2106
2462
  const admitted = await admitTurn(o.principal, o.agent, o.trigger);
2107
2463
  const { cfg, modelId, runtimeCtx, run, lease } = await withReservation(admitted, async () => {
2108
2464
  const resolved = await resolveAgentConfig(o.agent, o.principal, undefined, o.turnContext);
2109
- const opened = await openTurn(o.principal, chatId, o.agent, [o.message]);
2465
+ const opened = await openTurn(o.principal, chatId, o.agent, [o.message], resolved.cfg.configVersion);
2110
2466
  return { ...resolved, ...opened };
2111
2467
  });
2112
2468
  const onProgress = o.onProgress;
@@ -2192,7 +2548,7 @@ export function createRuntime(config) {
2192
2548
  return { answer: "" };
2193
2549
  const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(subChat.agent, principal);
2194
2550
  const admitted = await admitContinuation(principal, subChat.agent, "decision");
2195
- const { run, lease } = await openRun(principal, subChatId, subChat.agent);
2551
+ const { run, lease } = await openRun(principal, subChatId, subChat.agent, cfg.configVersion);
2196
2552
  const res = await buildTurn({
2197
2553
  resolvedChatId: subChatId,
2198
2554
  principal,