@intentface/latch-core 0.9.1 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent.d.ts +17 -0
- package/dist/agent.d.ts.map +1 -1
- package/dist/agent.js.map +1 -1
- package/dist/extensions/compose.d.ts +117 -0
- package/dist/extensions/compose.d.ts.map +1 -0
- package/dist/extensions/compose.js +249 -0
- package/dist/extensions/compose.js.map +1 -0
- package/dist/extensions/extension.d.ts +322 -0
- package/dist/extensions/extension.d.ts.map +1 -0
- package/dist/extensions/extension.js +171 -0
- package/dist/extensions/extension.js.map +1 -0
- package/dist/extensions/index.d.ts +11 -0
- package/dist/extensions/index.d.ts.map +1 -0
- package/dist/extensions/index.js +11 -0
- package/dist/extensions/index.js.map +1 -0
- package/dist/extensions/tools.d.ts +60 -0
- package/dist/extensions/tools.d.ts.map +1 -0
- package/dist/extensions/tools.js +229 -0
- package/dist/extensions/tools.js.map +1 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/dist/runtime.d.ts +10 -0
- package/dist/runtime.d.ts.map +1 -1
- package/dist/runtime.js +400 -44
- package/dist/runtime.js.map +1 -1
- package/dist/storage.d.ts +19 -0
- package/dist/storage.d.ts.map +1 -1
- package/dist/telemetry.d.ts +6 -0
- package/dist/telemetry.d.ts.map +1 -1
- package/package.json +3 -3
- package/dist/extensions.d.ts +0 -333
- package/dist/extensions.d.ts.map +0 -1
- package/dist/extensions.js +0 -569
- package/dist/extensions.js.map +0 -1
package/dist/runtime.js
CHANGED
|
@@ -5,6 +5,7 @@ import { summarizeForCompaction } from "./compaction.js";
|
|
|
5
5
|
import { computeCost } from "./pricing.js";
|
|
6
6
|
import { defaultPromptCachingPlan, markLastFunctionTool, mergeProviderOptions, toolsetHash, } from "./prompt-caching.js";
|
|
7
7
|
import { LimitExceededError, exceededPolicy, remainingTokens, windowRef, } from "./limits.js";
|
|
8
|
+
import { compose, fireClientToolResults, insertStepParts, isNewSession, lastUserTextOf, renderTranscript, replaceLastUserText, wrapTools, } from "./extensions/index.js";
|
|
8
9
|
/** Ids for assistant response messages — `msg_<random>`, matching storage ids. */
|
|
9
10
|
const generateMessageId = createIdGenerator({ prefix: "msg", separator: "_" });
|
|
10
11
|
/** Sub-thread chat ids start with `sub_` so the host can keep them out of the
|
|
@@ -575,7 +576,10 @@ export function createRuntime(config) {
|
|
|
575
576
|
* server-side (resume).
|
|
576
577
|
*/
|
|
577
578
|
async function buildTurn(args) {
|
|
578
|
-
const { resolvedChatId, principal, modelId, cfg, run, lease,
|
|
579
|
+
const { resolvedChatId, principal, modelId, cfg, run, lease, runtimeCtx } = args;
|
|
580
|
+
// Mutable: the start moments below may inject, replace or rewrite history
|
|
581
|
+
// before the model runs. From here on this is the turn's transcript.
|
|
582
|
+
let uiMessages = args.uiMessages;
|
|
579
583
|
const admitted = args.admitted;
|
|
580
584
|
const autoApprove = args.autoApprove ?? false;
|
|
581
585
|
const blockGated = args.blockGated ?? false;
|
|
@@ -585,6 +589,49 @@ export function createRuntime(config) {
|
|
|
585
589
|
// threaded a root down. Subagent turns spawned below inherit it, so a whole
|
|
586
590
|
// delegation tree shares one telemetry session (see telemetryMeta).
|
|
587
591
|
const rootChatId = args.rootChatId ?? resolvedChatId;
|
|
592
|
+
// --- the extension seam --------------------------------------------------
|
|
593
|
+
// Runtime-wide extensions first, then the agent's own. With none attached
|
|
594
|
+
// every moment below is a no-op and the turn is byte-identical to a runtime
|
|
595
|
+
// without the seam — `turn-snapshot.test.ts` holds it to that.
|
|
596
|
+
const hooks = compose([...(config.extensions ?? []), ...(cfg.extensions ?? [])]);
|
|
597
|
+
const xrun = {
|
|
598
|
+
runId: run.id,
|
|
599
|
+
principal: principal,
|
|
600
|
+
};
|
|
601
|
+
const xctx = (stepNumber, messages) => ({
|
|
602
|
+
...xrun,
|
|
603
|
+
stepNumber,
|
|
604
|
+
messages,
|
|
605
|
+
});
|
|
606
|
+
// A command an extension answered without the model (see the `input` moment).
|
|
607
|
+
let handled;
|
|
608
|
+
// The start moments run once per conversation turn. A resume replays a
|
|
609
|
+
// history they already shaped, and a decision continues a turn they
|
|
610
|
+
// already opened — running them again would inject twice.
|
|
611
|
+
const startMoments = trigger !== "resume" && trigger !== "decision";
|
|
612
|
+
if (startMoments) {
|
|
613
|
+
if (isNewSession(uiMessages))
|
|
614
|
+
uiMessages = (await hooks.messagesIn("session_start", uiMessages, xrun));
|
|
615
|
+
uiMessages = (await hooks.messagesIn("before_agent_start", uiMessages, xrun));
|
|
616
|
+
const text = lastUserTextOf(uiMessages);
|
|
617
|
+
if (text != null) {
|
|
618
|
+
const e = await hooks.input(text, uiMessages, xrun);
|
|
619
|
+
if (e && "handled" in e && e.handled) {
|
|
620
|
+
const reply = e.reply;
|
|
621
|
+
handled =
|
|
622
|
+
typeof reply === "string"
|
|
623
|
+
? [{ type: "text", text: reply }]
|
|
624
|
+
: Array.isArray(reply)
|
|
625
|
+
? reply
|
|
626
|
+
: [];
|
|
627
|
+
if (e.replace)
|
|
628
|
+
uiMessages = e.replace;
|
|
629
|
+
}
|
|
630
|
+
else if (e && "transform" in e && e.transform != null) {
|
|
631
|
+
uiMessages = replaceLastUserText(uiMessages, e.transform);
|
|
632
|
+
}
|
|
633
|
+
}
|
|
634
|
+
}
|
|
588
635
|
const abort = new AbortController();
|
|
589
636
|
let heartbeat;
|
|
590
637
|
const stopHeartbeat = () => {
|
|
@@ -648,7 +695,9 @@ export function createRuntime(config) {
|
|
|
648
695
|
}
|
|
649
696
|
const opened = [];
|
|
650
697
|
try {
|
|
651
|
-
|
|
698
|
+
// A command answered without the model needs no tools; don't open MCP
|
|
699
|
+
// connections for `/stats`.
|
|
700
|
+
for (const source of handled ? [] : sources) {
|
|
652
701
|
opened.push(await source.open(principal));
|
|
653
702
|
}
|
|
654
703
|
}
|
|
@@ -746,6 +795,13 @@ export function createRuntime(config) {
|
|
|
746
795
|
// Memory is optional and degradable — continue without it.
|
|
747
796
|
}
|
|
748
797
|
}
|
|
798
|
+
// instructions: an extension's appends land after the memory block; the
|
|
799
|
+
// date line stays the tail (see the agent construction). A `replace` may
|
|
800
|
+
// hand back system message(s) carrying providerOptions — the extension then
|
|
801
|
+
// owns that shape, and the caching plan's system breakpoint steps aside.
|
|
802
|
+
const composedInstructions = hooks.has("instructions")
|
|
803
|
+
? await hooks.instructions(instructions, xctx(0, uiMessages))
|
|
804
|
+
: instructions;
|
|
749
805
|
for (const name of cfg.disableTools ?? [])
|
|
750
806
|
delete tools[name];
|
|
751
807
|
// Per-agent tool description overrides — swap the resolved tool's
|
|
@@ -877,6 +933,27 @@ export function createRuntime(config) {
|
|
|
877
933
|
});
|
|
878
934
|
capabilityTools.push("spawn_agent");
|
|
879
935
|
}
|
|
936
|
+
// register_tools: what extensions add are capabilities — always active, and
|
|
937
|
+
// gated by the agent's own policy like any other tool. Mutated in place so
|
|
938
|
+
// every later reference (toolsContext keys, the fingerprint) sees them.
|
|
939
|
+
if (hooks.has("register_tools")) {
|
|
940
|
+
const before = new Set(Object.keys(tools));
|
|
941
|
+
const registered = hooks.tools(tools, xrun);
|
|
942
|
+
for (const k of Object.keys(tools))
|
|
943
|
+
if (!(k in registered))
|
|
944
|
+
delete tools[k];
|
|
945
|
+
Object.assign(tools, registered);
|
|
946
|
+
for (const name of Object.keys(tools))
|
|
947
|
+
if (!before.has(name))
|
|
948
|
+
capabilityTools.push(name);
|
|
949
|
+
}
|
|
950
|
+
// Outputs a CLIENT produced (a browser tool's answer) are already in the
|
|
951
|
+
// transcript; fire tool_result for them now that the toolset can tell a
|
|
952
|
+
// client tool from a server one.
|
|
953
|
+
// A decision turn is exactly when a client's answer has just been recorded,
|
|
954
|
+
// so it fires here too; a resume replays outputs an earlier run handled.
|
|
955
|
+
if ((startMoments || trigger === "decision") && hooks.has("tool_result"))
|
|
956
|
+
uiMessages = (await fireClientToolResults(uiMessages, tools, hooks, xrun));
|
|
880
957
|
// Subagent runs are autonomous — there's no human to authorize, so waive
|
|
881
958
|
// approvals and drop the OAuth `connect_` gates (they'd pause forever).
|
|
882
959
|
if (autoApprove) {
|
|
@@ -933,8 +1010,19 @@ export function createRuntime(config) {
|
|
|
933
1010
|
// doesn't mention keep the default. Only for the object form (a function
|
|
934
1011
|
// policy is left as-is).
|
|
935
1012
|
const sourceApproval = Object.assign({}, harnessApproval, ...opened.map((o) => o.toolApproval ?? {}));
|
|
1013
|
+
// An unattended run waives the AGENT's own policy — that is what makes a
|
|
1014
|
+
// delegated or programmatic turn autonomous. It does NOT waive the gates a
|
|
1015
|
+
// TOOL SOURCE contributed: a connection's gates are the end user's choice
|
|
1016
|
+
// about that connector ("deletes here must ask me"), and a harness gate is
|
|
1017
|
+
// the host's about that tool. Dropping those was the one thing they were
|
|
1018
|
+
// set to prevent, so they survive and the turn parks instead — a first-class
|
|
1019
|
+
// outcome (`runDue` reports `parked`; a subagent's gate bubbles to its
|
|
1020
|
+
// parent). `blockGated` is the opt-out: it has already stubbed those tools,
|
|
1021
|
+
// so gating them too would park a trial run that is meant to complete.
|
|
936
1022
|
const toolApproval = autoApprove
|
|
937
|
-
?
|
|
1023
|
+
? blockGated || Object.keys(sourceApproval).length === 0
|
|
1024
|
+
? undefined
|
|
1025
|
+
: sourceApproval
|
|
938
1026
|
: cfg.toolApproval && typeof cfg.toolApproval !== "function"
|
|
939
1027
|
? { ...sourceApproval, ...cfg.toolApproval }
|
|
940
1028
|
: Object.keys(sourceApproval).length > 0 && !cfg.toolApproval
|
|
@@ -968,7 +1056,64 @@ export function createRuntime(config) {
|
|
|
968
1056
|
? ({ steps }) => steps.reduce((n, st) => n + (st.usage?.inputTokens ?? 0) + (st.usage?.outputTokens ?? 0), 0) >=
|
|
969
1057
|
tokenBudget
|
|
970
1058
|
: undefined;
|
|
971
|
-
const
|
|
1059
|
+
const baseStopWhen = budgetStop ? [stepCap ?? stepCountIs(20), budgetStop] : stepCap;
|
|
1060
|
+
// --- step boundaries for the seam -------------------------------------------
|
|
1061
|
+
// A step ends twice from here: on the model side (`agent.stream`'s
|
|
1062
|
+
// onStepEnd — usage, finish reason, tool calls) and on the UI side (the
|
|
1063
|
+
// stream's onStepEnd — the assembled assistant message). `step_end` needs
|
|
1064
|
+
// both, so a step is SETTLED once both have arrived, and the next step's
|
|
1065
|
+
// prepareStep waits for the settlement: an extension that persists has
|
|
1066
|
+
// written before the model is called again. None of this runs when no
|
|
1067
|
+
// extension listens to a step moment.
|
|
1068
|
+
const stepHooks = hooks.has("step_start") || hooks.has("step_end");
|
|
1069
|
+
const slotOf = (m, k) => {
|
|
1070
|
+
let d = m.get(k);
|
|
1071
|
+
if (!d) {
|
|
1072
|
+
let resolve;
|
|
1073
|
+
const promise = new Promise((r) => (resolve = r));
|
|
1074
|
+
d = { promise, resolve };
|
|
1075
|
+
m.set(k, d);
|
|
1076
|
+
}
|
|
1077
|
+
return d;
|
|
1078
|
+
};
|
|
1079
|
+
const modelSide = new Map();
|
|
1080
|
+
const uiSide = new Map();
|
|
1081
|
+
const settledSteps = new Map();
|
|
1082
|
+
const appendedByStep = new Map();
|
|
1083
|
+
const stepStartedAt = new Map();
|
|
1084
|
+
let control0;
|
|
1085
|
+
let nextControl;
|
|
1086
|
+
let stoppedBeforeStart = false;
|
|
1087
|
+
let uiStepIndex = 0;
|
|
1088
|
+
let stepsSeen = 0;
|
|
1089
|
+
let turnWriter;
|
|
1090
|
+
let latestAssistant;
|
|
1091
|
+
// A continuation reuses the assistant message the SDK merges into; its
|
|
1092
|
+
// earlier step-start parts are not this run's steps.
|
|
1093
|
+
const anchor = uiMessages[uiMessages.length - 1];
|
|
1094
|
+
const anchorIsAssistant = anchor?.role === "assistant";
|
|
1095
|
+
const priorSteps = anchorIsAssistant
|
|
1096
|
+
? anchor.parts.filter((p) => p.type === "step-start").length
|
|
1097
|
+
: 0;
|
|
1098
|
+
const transcriptNow = () => latestAssistant
|
|
1099
|
+
? anchorIsAssistant
|
|
1100
|
+
? [...uiMessages.slice(0, -1), latestAssistant]
|
|
1101
|
+
: [...uiMessages, latestAssistant]
|
|
1102
|
+
: uiMessages;
|
|
1103
|
+
// Assigned below, once the checkpoint chain exists; only ever called mid-stream.
|
|
1104
|
+
let settle = async () => { };
|
|
1105
|
+
const extensionStop = stepHooks
|
|
1106
|
+
? async ({ steps }) => {
|
|
1107
|
+
const k = steps.length - 1;
|
|
1108
|
+
await settle(k);
|
|
1109
|
+
const c = await hooks.stepStart(xctx(steps.length, transcriptNow()));
|
|
1110
|
+
nextControl = c;
|
|
1111
|
+
return c.stop;
|
|
1112
|
+
}
|
|
1113
|
+
: undefined;
|
|
1114
|
+
const stopWhen = extensionStop
|
|
1115
|
+
? [...(Array.isArray(baseStopWhen) ? baseStopWhen : [baseStopWhen ?? stepCountIs(20)]), extensionStop]
|
|
1116
|
+
: baseStopWhen;
|
|
972
1117
|
// Reasoning/thinking effort → provider-specific options (the platform maps
|
|
973
1118
|
// it per provider + gates to reasoning-capable models). Called even without
|
|
974
1119
|
// an effort so the hook can set safe defaults (e.g. lift a provider's tiny
|
|
@@ -1001,6 +1146,14 @@ export function createRuntime(config) {
|
|
|
1001
1146
|
// run so churn across a chat's turns (a cache invalidator) is queryable.
|
|
1002
1147
|
const toolsHash = toolsetHash(activeTools ?? Object.keys(tools));
|
|
1003
1148
|
const providerOptions = mergeProviderOptions(reasoning?.providerOptions, caching?.requestProviderOptions);
|
|
1149
|
+
// tool_call / tool_result: wrap in place, after the caching marker and the
|
|
1150
|
+
// test-run stubs, so every later reference sees the same names. Identity-
|
|
1151
|
+
// preserving when no extension listens.
|
|
1152
|
+
if (hooks.has("tool_call") || hooks.has("tool_result")) {
|
|
1153
|
+
const wrapped = wrapTools(tools, hooks, xrun);
|
|
1154
|
+
for (const k of Object.keys(wrapped))
|
|
1155
|
+
tools[k] = wrapped[k];
|
|
1156
|
+
}
|
|
1004
1157
|
// Telemetry seam: per-run AI SDK telemetry options + trace correlation
|
|
1005
1158
|
// identity, resolved once per turn (undefined → nothing traced).
|
|
1006
1159
|
const telemetryMeta = {
|
|
@@ -1010,6 +1163,7 @@ export function createRuntime(config) {
|
|
|
1010
1163
|
modelId,
|
|
1011
1164
|
principal,
|
|
1012
1165
|
startedAt: run.startedAt,
|
|
1166
|
+
agentVersion: cfg.configVersion,
|
|
1013
1167
|
depth,
|
|
1014
1168
|
trigger,
|
|
1015
1169
|
parentRunId: args.parentRunId,
|
|
@@ -1032,24 +1186,46 @@ export function createRuntime(config) {
|
|
|
1032
1186
|
// the instructions become a MARKED system message and the date line a
|
|
1033
1187
|
// separate unmarked one, so the daily flip lands after the breakpoint
|
|
1034
1188
|
// instead of invalidating it at midnight UTC.
|
|
1035
|
-
instructions:
|
|
1036
|
-
?
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
|
|
1045
|
-
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1189
|
+
instructions: typeof composedInstructions === "object" && composedInstructions !== null
|
|
1190
|
+
? // An extension replaced the system prompt with message(s) of its own,
|
|
1191
|
+
// providerOptions and all: keep them, with the date line as its own
|
|
1192
|
+
// trailing message so the daily flip stays past any breakpoint.
|
|
1193
|
+
[
|
|
1194
|
+
...(Array.isArray(composedInstructions) ? composedInstructions : [composedInstructions]),
|
|
1195
|
+
{ role: "system", content: currentDateLine() },
|
|
1196
|
+
]
|
|
1197
|
+
: caching?.systemProviderOptions
|
|
1198
|
+
? [
|
|
1199
|
+
...(composedInstructions
|
|
1200
|
+
? [
|
|
1201
|
+
{
|
|
1202
|
+
role: "system",
|
|
1203
|
+
content: composedInstructions,
|
|
1204
|
+
providerOptions: caching.systemProviderOptions,
|
|
1205
|
+
},
|
|
1206
|
+
]
|
|
1207
|
+
: []),
|
|
1208
|
+
{ role: "system", content: currentDateLine() },
|
|
1209
|
+
]
|
|
1210
|
+
: [composedInstructions, currentDateLine()].filter(Boolean).join("\n\n"),
|
|
1049
1211
|
tools,
|
|
1050
1212
|
toolApproval,
|
|
1051
1213
|
stopWhen,
|
|
1052
1214
|
activeTools,
|
|
1215
|
+
...(stepHooks
|
|
1216
|
+
? {
|
|
1217
|
+
prepareStep: (async ({ stepNumber }) => {
|
|
1218
|
+
if (stepNumber > 0)
|
|
1219
|
+
await settle(stepNumber - 1);
|
|
1220
|
+
stepStartedAt.set(stepNumber, Date.now());
|
|
1221
|
+
const c = stepNumber === 0 ? control0 : nextControl;
|
|
1222
|
+
return {
|
|
1223
|
+
...(c?.model ? { model: c.model } : {}),
|
|
1224
|
+
...(c?.activeTools ? { activeTools: c.activeTools } : {}),
|
|
1225
|
+
};
|
|
1226
|
+
}),
|
|
1227
|
+
}
|
|
1228
|
+
: {}),
|
|
1053
1229
|
...(providerOptions ? { providerOptions } : {}),
|
|
1054
1230
|
...(reasoning?.maxOutputTokens ? { maxOutputTokens: reasoning.maxOutputTokens } : {}),
|
|
1055
1231
|
...(telemetryOptions ? { telemetry: telemetryOptions } : {}),
|
|
@@ -1076,6 +1252,80 @@ export function createRuntime(config) {
|
|
|
1076
1252
|
// at most the write in flight and the one queued behind it. A failed write
|
|
1077
1253
|
// logs and releases the chain — the next checkpoint still runs.
|
|
1078
1254
|
let checkpoints = Promise.resolve();
|
|
1255
|
+
const checkpoint = (message) => {
|
|
1256
|
+
if (!dur.enabled || abort.signal.aborted)
|
|
1257
|
+
return;
|
|
1258
|
+
checkpoints = checkpoints
|
|
1259
|
+
.then(() => {
|
|
1260
|
+
// Re-checked at write time: the lease may have gone while this
|
|
1261
|
+
// checkpoint waited behind the previous one.
|
|
1262
|
+
if (abort.signal.aborted)
|
|
1263
|
+
return;
|
|
1264
|
+
return storage.appendMessages(resolvedChatId, [message]);
|
|
1265
|
+
})
|
|
1266
|
+
.catch((e) => console.warn(`[durability] step checkpoint failed for run ${run.id}:`, e));
|
|
1267
|
+
};
|
|
1268
|
+
if (stepHooks) {
|
|
1269
|
+
const needUi = hooks.has("step_end");
|
|
1270
|
+
settle = (k) => {
|
|
1271
|
+
const known = settledSteps.get(k);
|
|
1272
|
+
if (known)
|
|
1273
|
+
return known;
|
|
1274
|
+
const p = (async () => {
|
|
1275
|
+
const model = await slotOf(modelSide, k).promise;
|
|
1276
|
+
const response = needUi ? await slotOf(uiSide, k).promise : undefined;
|
|
1277
|
+
const raw = model.usage;
|
|
1278
|
+
// The SDK assembles the message from chunks; our parts were streamed
|
|
1279
|
+
// transient and never entered it. Fold every step's back in.
|
|
1280
|
+
const assistant = response
|
|
1281
|
+
? insertStepParts(response, appendedByStep, priorSteps)
|
|
1282
|
+
: undefined;
|
|
1283
|
+
const messages = assistant
|
|
1284
|
+
? anchorIsAssistant
|
|
1285
|
+
? [...uiMessages.slice(0, -1), assistant]
|
|
1286
|
+
: [...uiMessages, assistant]
|
|
1287
|
+
: transcriptNow();
|
|
1288
|
+
const info = {
|
|
1289
|
+
...xctx(k, messages),
|
|
1290
|
+
usage: {
|
|
1291
|
+
inputTokens: raw?.inputTokens,
|
|
1292
|
+
outputTokens: raw?.outputTokens,
|
|
1293
|
+
totalTokens: raw?.totalTokens,
|
|
1294
|
+
cacheReadTokens: raw?.inputTokenDetails?.cacheReadTokens,
|
|
1295
|
+
cacheWriteTokens: raw?.inputTokenDetails?.cacheWriteTokens,
|
|
1296
|
+
reasoningTokens: raw?.outputTokenDetails?.reasoningTokens,
|
|
1297
|
+
},
|
|
1298
|
+
finishReason: model.finishReason,
|
|
1299
|
+
model: model.model,
|
|
1300
|
+
durationMs: Date.now() - (stepStartedAt.get(k) ?? Date.now()),
|
|
1301
|
+
toolCalls: model.toolCalls,
|
|
1302
|
+
};
|
|
1303
|
+
if (needUi) {
|
|
1304
|
+
const appended = await hooks.stepEnd(messages, info);
|
|
1305
|
+
if (appended.length) {
|
|
1306
|
+
appendedByStep.set(k, appended);
|
|
1307
|
+
// The client copy goes through the view rules and is TRANSIENT, so
|
|
1308
|
+
// the SDK's own merge cannot persist a viewed copy over the truth,
|
|
1309
|
+
// which is folded into the checkpoint and the finish below.
|
|
1310
|
+
for (const part of appended) {
|
|
1311
|
+
if (!part.type.startsWith("data-"))
|
|
1312
|
+
continue;
|
|
1313
|
+
const view = hooks.view(part, false, xrun);
|
|
1314
|
+
if (view && "hide" in view && view.hide)
|
|
1315
|
+
continue;
|
|
1316
|
+
const shown = view && "replace" in view && view.replace ? view.replace : part;
|
|
1317
|
+
turnWriter?.write({ type: shown.type, data: shown.data, transient: true });
|
|
1318
|
+
}
|
|
1319
|
+
}
|
|
1320
|
+
latestAssistant = insertStepParts(response, appendedByStep, priorSteps);
|
|
1321
|
+
checkpoint(latestAssistant);
|
|
1322
|
+
}
|
|
1323
|
+
})();
|
|
1324
|
+
settledSteps.set(k, p);
|
|
1325
|
+
p.catch(() => { }); // surfaces where it is awaited: prepareStep, stopWhen, onFinish
|
|
1326
|
+
return p;
|
|
1327
|
+
};
|
|
1328
|
+
}
|
|
1079
1329
|
// Post-turn telemetry: fired after persistence on every terminal path
|
|
1080
1330
|
// (completed / awaiting_input / errored). Must never fail the turn.
|
|
1081
1331
|
const reportRunFinished = async (status, messages, error) => {
|
|
@@ -1128,10 +1378,48 @@ export function createRuntime(config) {
|
|
|
1128
1378
|
// turn would upsert onto the same row — see the dup-id regression test).
|
|
1129
1379
|
generateId: generateMessageId,
|
|
1130
1380
|
execute: async ({ writer }) => {
|
|
1131
|
-
|
|
1132
|
-
|
|
1381
|
+
turnWriter = writer;
|
|
1382
|
+
if (handled) {
|
|
1383
|
+
// A command an extension answered without the model. Emitted as an
|
|
1384
|
+
// ordinary assistant turn so the conversation reads normally and the
|
|
1385
|
+
// finish hook persists it like any other.
|
|
1386
|
+
writer.write({
|
|
1387
|
+
type: "start",
|
|
1388
|
+
messageId: generateMessageId(),
|
|
1389
|
+
messageMetadata: { visibility: "user", model: modelId, createdAt: Date.now(), runId: run.id },
|
|
1390
|
+
});
|
|
1391
|
+
for (const p of handled) {
|
|
1392
|
+
if (p.type === "text") {
|
|
1393
|
+
const id = generateMessageId();
|
|
1394
|
+
writer.write({ type: "text-start", id });
|
|
1395
|
+
writer.write({ type: "text-delta", id, delta: p.text });
|
|
1396
|
+
writer.write({ type: "text-end", id });
|
|
1397
|
+
continue;
|
|
1398
|
+
}
|
|
1399
|
+
if (!p.type.startsWith("data-"))
|
|
1400
|
+
continue;
|
|
1401
|
+
const view = hooks.view(p, false, xrun);
|
|
1402
|
+
if (view && "hide" in view && view.hide)
|
|
1403
|
+
continue;
|
|
1404
|
+
const shown = view && "replace" in view && view.replace ? view.replace : p;
|
|
1405
|
+
writer.write({ type: shown.type, data: shown.data });
|
|
1406
|
+
}
|
|
1407
|
+
writer.write({ type: "finish" });
|
|
1408
|
+
return;
|
|
1409
|
+
}
|
|
1133
1410
|
// Model projection: drops `sendToModel:false` messages (and data-* parts).
|
|
1134
|
-
|
|
1411
|
+
// `context` lets an extension reshape what the model reads this turn.
|
|
1412
|
+
const projected = await toModelMessages(uiMessages, { tools });
|
|
1413
|
+
const modelMessages = hooks.has("context")
|
|
1414
|
+
? await hooks.context(projected, xctx(0, uiMessages))
|
|
1415
|
+
: projected;
|
|
1416
|
+
// step_start for the first step — the SDK has no hook before its first call.
|
|
1417
|
+
control0 = stepHooks ? await hooks.stepStart(xctx(0, uiMessages)) : undefined;
|
|
1418
|
+
if (control0?.stop) {
|
|
1419
|
+
stoppedBeforeStart = true;
|
|
1420
|
+
writer.write({ type: "finish" });
|
|
1421
|
+
return;
|
|
1422
|
+
}
|
|
1135
1423
|
// Per-session sandbox (one per chat) for agents that declared `sandbox`.
|
|
1136
1424
|
// Exposed to sandbox tools as `options.experimental_sandbox`.
|
|
1137
1425
|
const sandbox = wantSandbox
|
|
@@ -1195,6 +1483,18 @@ export function createRuntime(config) {
|
|
|
1195
1483
|
toolResults: step.toolResults,
|
|
1196
1484
|
}));
|
|
1197
1485
|
}
|
|
1486
|
+
stepsSeen++;
|
|
1487
|
+
if (stepHooks && step.stepNumber !== undefined)
|
|
1488
|
+
slotOf(modelSide, step.stepNumber).resolve({
|
|
1489
|
+
usage: step.usage,
|
|
1490
|
+
finishReason: step.finishReason ?? "unknown",
|
|
1491
|
+
toolCalls: (step.toolCalls ?? []).map((c) => ({ toolName: c.toolName, toolCallId: c.toolCallId })),
|
|
1492
|
+
// The id the PROVIDER reports; pricing keys on that.
|
|
1493
|
+
model: {
|
|
1494
|
+
provider: step.model?.provider ?? "",
|
|
1495
|
+
modelId: step.response?.modelId ?? step.model?.modelId ?? modelId,
|
|
1496
|
+
},
|
|
1497
|
+
});
|
|
1198
1498
|
},
|
|
1199
1499
|
});
|
|
1200
1500
|
const result = telemetryOn && config.telemetry?.withRunSpan
|
|
@@ -1235,24 +1535,61 @@ export function createRuntime(config) {
|
|
|
1235
1535
|
// check inside the storage write itself — an adapter interface change,
|
|
1236
1536
|
// tracked as a follow-up. With the heartbeat now aborting within one TTL
|
|
1237
1537
|
// of losing contact, the exposure is bounded to ~leaseTtlMs.
|
|
1238
|
-
onStepEnd: dur.enabled
|
|
1538
|
+
onStepEnd: dur.enabled || hooks.has("step_end")
|
|
1239
1539
|
? ({ responseMessage }) => {
|
|
1240
|
-
if (
|
|
1540
|
+
if (hooks.has("step_end")) {
|
|
1541
|
+
// The step's parts are final only once `step_end` has appended
|
|
1542
|
+
// its own; `settle` writes the checkpoint after that.
|
|
1543
|
+
slotOf(uiSide, uiStepIndex).resolve(responseMessage);
|
|
1544
|
+
void settle(uiStepIndex++);
|
|
1241
1545
|
return;
|
|
1242
|
-
|
|
1243
|
-
|
|
1244
|
-
// Re-checked at write time: the lease may have gone while this
|
|
1245
|
-
// checkpoint waited behind the previous one.
|
|
1246
|
-
if (abort.signal.aborted)
|
|
1247
|
-
return;
|
|
1248
|
-
return storage.appendMessages(resolvedChatId, [responseMessage]);
|
|
1249
|
-
})
|
|
1250
|
-
.catch((e) => console.warn(`[durability] step checkpoint failed for run ${run.id}:`, e));
|
|
1546
|
+
}
|
|
1547
|
+
checkpoint(responseMessage);
|
|
1251
1548
|
}
|
|
1252
1549
|
: undefined,
|
|
1253
|
-
onFinish: async ({ messages, isAborted, }) => {
|
|
1550
|
+
onFinish: async ({ messages: finishedMessages, isAborted, }) => {
|
|
1254
1551
|
stopHeartbeat();
|
|
1255
1552
|
await closeSources();
|
|
1553
|
+
let messages = finishedMessages;
|
|
1554
|
+
if (stepHooks && failed === undefined && !isAborted) {
|
|
1555
|
+
// Every step's `step_end` has run before the record is settled. A
|
|
1556
|
+
// critical extension's throw on an EARLIER step surfaced through the
|
|
1557
|
+
// stream (prepareStep rethrows it); on the LAST step nothing awaited
|
|
1558
|
+
// it yet, so it lands here — and fails the run just the same.
|
|
1559
|
+
try {
|
|
1560
|
+
await Promise.all([...settledSteps.values()]);
|
|
1561
|
+
}
|
|
1562
|
+
catch (e) {
|
|
1563
|
+
failed = clientErrorMessage(e);
|
|
1564
|
+
}
|
|
1565
|
+
if (appendedByStep.size) {
|
|
1566
|
+
const i = messages.length - 1;
|
|
1567
|
+
const last = messages[i];
|
|
1568
|
+
if (last?.role === "assistant")
|
|
1569
|
+
messages = [...messages.slice(0, i), insertStepParts(last, appendedByStep, priorSteps)];
|
|
1570
|
+
}
|
|
1571
|
+
}
|
|
1572
|
+
if (stoppedBeforeStart || handled)
|
|
1573
|
+
// Nothing (or nothing but a command with no reply) was said: the SDK
|
|
1574
|
+
// still assembles an empty assistant message from the bare `finish`,
|
|
1575
|
+
// and there is no reason to store a blank turn.
|
|
1576
|
+
messages = messages.filter((m) => !(m.role === "assistant" && m.parts.length === 0));
|
|
1577
|
+
const agentEnd = (error) => hooks.has("agent_end")
|
|
1578
|
+
? hooks.agentEnd({
|
|
1579
|
+
runId: run.id,
|
|
1580
|
+
principal: xrun.principal,
|
|
1581
|
+
messages,
|
|
1582
|
+
stepCount: stepsSeen,
|
|
1583
|
+
isAborted: isAborted === true,
|
|
1584
|
+
...(error !== undefined ? { error } : {}),
|
|
1585
|
+
usage: {
|
|
1586
|
+
inputTokens,
|
|
1587
|
+
outputTokens,
|
|
1588
|
+
...(cacheReadTokens ? { cacheReadTokens } : {}),
|
|
1589
|
+
...(cacheWriteTokens ? { cacheWriteTokens } : {}),
|
|
1590
|
+
},
|
|
1591
|
+
})
|
|
1592
|
+
: Promise.resolve();
|
|
1256
1593
|
// Cost for this turn (0 if no pricing configured). Stamp usage + cost
|
|
1257
1594
|
// onto the assistant message — the message is complete; this is metadata.
|
|
1258
1595
|
const cost = computeCost(config.pricing?.(modelId), {
|
|
@@ -1289,6 +1626,7 @@ export function createRuntime(config) {
|
|
|
1289
1626
|
if (settled) {
|
|
1290
1627
|
await settleLimits(admitted, { tokens: inputTokens + outputTokens, cost });
|
|
1291
1628
|
await reportRunFinished("errored", messages, failed ?? "aborted");
|
|
1629
|
+
await agentEnd(failed ?? "aborted");
|
|
1292
1630
|
}
|
|
1293
1631
|
return;
|
|
1294
1632
|
}
|
|
@@ -1333,13 +1671,17 @@ export function createRuntime(config) {
|
|
|
1333
1671
|
// reaped and the run re-claimed elsewhere (`!applied`), that node will
|
|
1334
1672
|
// finish it — a stale writer must not re-run the judge or emit a
|
|
1335
1673
|
// duplicate "completed"/"awaiting_input" trace.
|
|
1336
|
-
if (applied)
|
|
1674
|
+
if (applied) {
|
|
1337
1675
|
await reportRunFinished(paused ? "awaiting_input" : "completed", messages);
|
|
1676
|
+
await agentEnd();
|
|
1677
|
+
}
|
|
1338
1678
|
},
|
|
1339
1679
|
onError: (error) => {
|
|
1340
1680
|
stopHeartbeat();
|
|
1341
1681
|
void closeSources();
|
|
1342
1682
|
const message = clientErrorMessage(error);
|
|
1683
|
+
if (hooks.has("error"))
|
|
1684
|
+
void hooks.error({ error, stepNumber: stepsSeen, messages: transcriptNow() }, xrun);
|
|
1343
1685
|
// Record only — the first error wins; later onError calls for the same
|
|
1344
1686
|
// stream are noise. The SDK's onError must return the client string
|
|
1345
1687
|
// synchronously and still runs onFinish afterwards, so that is where
|
|
@@ -1465,12 +1807,13 @@ export function createRuntime(config) {
|
|
|
1465
1807
|
* persists nothing (the crash window between the two writes is the trade-off
|
|
1466
1808
|
* an adapter without `openTurn` accepts).
|
|
1467
1809
|
*/
|
|
1468
|
-
async function openTurn(principal, chatId, agentName, messages) {
|
|
1810
|
+
async function openTurn(principal, chatId, agentName, messages, agentVersion) {
|
|
1469
1811
|
if (storage.openTurn) {
|
|
1470
1812
|
const run = await storage.openTurn(principal, {
|
|
1471
1813
|
chatId,
|
|
1472
1814
|
agent: agentName,
|
|
1473
1815
|
messages,
|
|
1816
|
+
agentVersion: agentVersion ?? null,
|
|
1474
1817
|
...(dur.enabled ? { lease: { owner: dur.instanceId, ttlMs: dur.leaseTtlMs } } : {}),
|
|
1475
1818
|
});
|
|
1476
1819
|
return {
|
|
@@ -1478,12 +1821,12 @@ export function createRuntime(config) {
|
|
|
1478
1821
|
lease: dur.enabled ? { owner: dur.instanceId, fencingToken: run.fencingToken ?? 1 } : undefined,
|
|
1479
1822
|
};
|
|
1480
1823
|
}
|
|
1481
|
-
const opened = await openRun(principal, chatId, agentName);
|
|
1824
|
+
const opened = await openRun(principal, chatId, agentName, agentVersion);
|
|
1482
1825
|
if (messages.length > 0)
|
|
1483
1826
|
await storage.appendMessages(chatId, messages);
|
|
1484
1827
|
return opened;
|
|
1485
1828
|
}
|
|
1486
|
-
async function openRun(principal, chatId, agentName) {
|
|
1829
|
+
async function openRun(principal, chatId, agentName, agentVersion) {
|
|
1487
1830
|
if (dur.enabled) {
|
|
1488
1831
|
const run = await storage.claimRun({
|
|
1489
1832
|
principal,
|
|
@@ -1491,11 +1834,17 @@ export function createRuntime(config) {
|
|
|
1491
1834
|
agent: agentName,
|
|
1492
1835
|
owner: dur.instanceId,
|
|
1493
1836
|
ttlMs: dur.leaseTtlMs,
|
|
1837
|
+
agentVersion: agentVersion ?? null,
|
|
1494
1838
|
});
|
|
1495
1839
|
return { run, lease: { owner: dur.instanceId, fencingToken: run.fencingToken ?? 1 } };
|
|
1496
1840
|
}
|
|
1497
1841
|
return {
|
|
1498
|
-
run: await storage.createRun(principal, {
|
|
1842
|
+
run: await storage.createRun(principal, {
|
|
1843
|
+
chatId,
|
|
1844
|
+
agent: agentName,
|
|
1845
|
+
kind: "turn",
|
|
1846
|
+
agentVersion: agentVersion ?? null,
|
|
1847
|
+
}),
|
|
1499
1848
|
lease: undefined,
|
|
1500
1849
|
};
|
|
1501
1850
|
}
|
|
@@ -1530,7 +1879,7 @@ export function createRuntime(config) {
|
|
|
1530
1879
|
// Run + message open together (atomically where the adapter can): a chat
|
|
1531
1880
|
// with a live turn refuses (`ChatBusyError` → 409) and the message never
|
|
1532
1881
|
// lands in history — the client retries it (with its turn reservation released).
|
|
1533
|
-
const { run, lease } = await withReservation(admitted, () => openTurn(principal, resolvedChatId, agentName, [message]));
|
|
1882
|
+
const { run, lease } = await withReservation(admitted, () => openTurn(principal, resolvedChatId, agentName, [message], cfg.configVersion));
|
|
1534
1883
|
return buildTurn({
|
|
1535
1884
|
resolvedChatId,
|
|
1536
1885
|
principal,
|
|
@@ -1607,6 +1956,9 @@ export function createRuntime(config) {
|
|
|
1607
1956
|
modelId: ranAs(uiMessages, run.id) ?? "unknown",
|
|
1608
1957
|
principal,
|
|
1609
1958
|
startedAt: run.startedAt,
|
|
1959
|
+
// Likewise persisted, not re-resolved: reconciling a finished run must
|
|
1960
|
+
// report the version it ran on, not whatever is published now.
|
|
1961
|
+
agentVersion: run.agentVersion ?? undefined,
|
|
1610
1962
|
depth: 0,
|
|
1611
1963
|
trigger: "resume",
|
|
1612
1964
|
rootChatId: run.chatId,
|
|
@@ -1800,7 +2152,7 @@ export function createRuntime(config) {
|
|
|
1800
2152
|
}
|
|
1801
2153
|
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(chat.agent, principal, request, turnContext);
|
|
1802
2154
|
const admitted = await admitContinuation(principal, chat.agent, "decision");
|
|
1803
|
-
const { run, lease } = await openRun(principal, chatId, chat.agent);
|
|
2155
|
+
const { run, lease } = await openRun(principal, chatId, chat.agent, cfg.configVersion);
|
|
1804
2156
|
return buildTurn({
|
|
1805
2157
|
resolvedChatId: chatId,
|
|
1806
2158
|
principal,
|
|
@@ -1852,7 +2204,7 @@ export function createRuntime(config) {
|
|
|
1852
2204
|
// a busy chat refuses before anything is recorded, so the retry finds the
|
|
1853
2205
|
// call still unanswered and resumes it — persisting first would make the
|
|
1854
2206
|
// retry a "duplicate submit" no-op and lose the answer.
|
|
1855
|
-
const { run, lease } = await openTurn(principal, chatId, chat.agent, changed);
|
|
2207
|
+
const { run, lease } = await openTurn(principal, chatId, chat.agent, changed, cfg.configVersion);
|
|
1856
2208
|
return buildTurn({
|
|
1857
2209
|
resolvedChatId: chatId,
|
|
1858
2210
|
principal,
|
|
@@ -1998,7 +2350,11 @@ export function createRuntime(config) {
|
|
|
1998
2350
|
}
|
|
1999
2351
|
async function loadHistory({ chatId, principal, }) {
|
|
2000
2352
|
const messages = await storage.loadMessages(principal, chatId);
|
|
2001
|
-
|
|
2353
|
+
// `render` rules apply on load as they do live. Runtime-wide extensions
|
|
2354
|
+
// only: an agent's own would need the chat's agent resolved here.
|
|
2355
|
+
return renderTranscript(config.extensions ?? [], toClientMessages(messages), {
|
|
2356
|
+
principal: principal,
|
|
2357
|
+
});
|
|
2002
2358
|
}
|
|
2003
2359
|
async function listChats({ principal, limit, kind, }) {
|
|
2004
2360
|
// No default kind filter — passing nothing returns ALL chats (unchanged
|
|
@@ -2106,7 +2462,7 @@ export function createRuntime(config) {
|
|
|
2106
2462
|
const admitted = await admitTurn(o.principal, o.agent, o.trigger);
|
|
2107
2463
|
const { cfg, modelId, runtimeCtx, run, lease } = await withReservation(admitted, async () => {
|
|
2108
2464
|
const resolved = await resolveAgentConfig(o.agent, o.principal, undefined, o.turnContext);
|
|
2109
|
-
const opened = await openTurn(o.principal, chatId, o.agent, [o.message]);
|
|
2465
|
+
const opened = await openTurn(o.principal, chatId, o.agent, [o.message], resolved.cfg.configVersion);
|
|
2110
2466
|
return { ...resolved, ...opened };
|
|
2111
2467
|
});
|
|
2112
2468
|
const onProgress = o.onProgress;
|
|
@@ -2192,7 +2548,7 @@ export function createRuntime(config) {
|
|
|
2192
2548
|
return { answer: "" };
|
|
2193
2549
|
const { cfg, modelId, runtimeCtx } = await resolveAgentConfig(subChat.agent, principal);
|
|
2194
2550
|
const admitted = await admitContinuation(principal, subChat.agent, "decision");
|
|
2195
|
-
const { run, lease } = await openRun(principal, subChatId, subChat.agent);
|
|
2551
|
+
const { run, lease } = await openRun(principal, subChatId, subChat.agent, cfg.configVersion);
|
|
2196
2552
|
const res = await buildTurn({
|
|
2197
2553
|
resolvedChatId: subChatId,
|
|
2198
2554
|
principal,
|