@namzu/sdk 42.0.0 → 42.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +222 -0
- package/dist/manager/resident/outbox.d.ts +8 -8
- package/dist/manager/resident/store.d.ts +4 -4
- package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
- package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
- package/dist/runtime/query/cancelled-before-start.js +152 -0
- package/dist/runtime/query/cancelled-before-start.js.map +1 -0
- package/dist/runtime/query/checkpoint.d.ts +21 -0
- package/dist/runtime/query/checkpoint.d.ts.map +1 -1
- package/dist/runtime/query/checkpoint.js +23 -0
- package/dist/runtime/query/checkpoint.js.map +1 -1
- package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
- package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
- package/dist/runtime/query/executor/tool-call-admission.js +373 -0
- package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
- package/dist/runtime/query/executor.d.ts +70 -35
- package/dist/runtime/query/executor.d.ts.map +1 -1
- package/dist/runtime/query/executor.js +46 -380
- package/dist/runtime/query/executor.js.map +1 -1
- package/dist/runtime/query/finalize-run.d.ts +55 -0
- package/dist/runtime/query/finalize-run.d.ts.map +1 -0
- package/dist/runtime/query/finalize-run.js +113 -0
- package/dist/runtime/query/finalize-run.js.map +1 -0
- package/dist/runtime/query/index.d.ts +4 -9
- package/dist/runtime/query/index.d.ts.map +1 -1
- package/dist/runtime/query/index.js +238 -893
- package/dist/runtime/query/index.js.map +1 -1
- package/dist/runtime/query/iteration/index.d.ts +6 -161
- package/dist/runtime/query/iteration/index.d.ts.map +1 -1
- package/dist/runtime/query/iteration/index.js +23 -523
- package/dist/runtime/query/iteration/index.js.map +1 -1
- package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
- package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
- package/dist/runtime/query/iteration/outstanding-work.js +365 -0
- package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
- package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
- package/dist/runtime/query/iteration/phases/plan.js +13 -2
- package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
- package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
- package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
- package/dist/runtime/query/iteration/step-shaping.js +184 -0
- package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
- package/dist/runtime/query/prepare-run.d.ts +94 -0
- package/dist/runtime/query/prepare-run.d.ts.map +1 -0
- package/dist/runtime/query/prepare-run.js +589 -0
- package/dist/runtime/query/prepare-run.js.map +1 -0
- package/dist/runtime/query/release-run.d.ts +56 -0
- package/dist/runtime/query/release-run.d.ts.map +1 -0
- package/dist/runtime/query/release-run.js +101 -0
- package/dist/runtime/query/release-run.js.map +1 -0
- package/dist/runtime/query/resume-pending.d.ts +112 -1
- package/dist/runtime/query/resume-pending.d.ts.map +1 -1
- package/dist/runtime/query/resume-pending.js +133 -0
- package/dist/runtime/query/resume-pending.js.map +1 -1
- package/dist/store/evidence/compaction-archive.d.ts +2 -2
- package/dist/types/run/config.d.ts +12 -5
- package/dist/types/run/config.d.ts.map +1 -1
- package/package.json +4 -4
- package/src/runtime/query/cancelled-before-start.ts +189 -0
- package/src/runtime/query/checkpoint.ts +22 -0
- package/src/runtime/query/executor/tool-call-admission.ts +473 -0
- package/src/runtime/query/executor.ts +63 -442
- package/src/runtime/query/finalize-run.ts +192 -0
- package/src/runtime/query/index.ts +270 -1011
- package/src/runtime/query/iteration/index.ts +40 -586
- package/src/runtime/query/iteration/outstanding-work.ts +386 -0
- package/src/runtime/query/iteration/phases/plan.ts +18 -2
- package/src/runtime/query/iteration/step-shaping.ts +271 -0
- package/src/runtime/query/prepare-run.ts +718 -0
- package/src/runtime/query/release-run.ts +168 -0
- package/src/runtime/query/resume-pending.ts +158 -0
- package/src/types/run/config.ts +12 -5
|
@@ -1,9 +1,7 @@
|
|
|
1
1
|
import { join } from 'node:path';
|
|
2
2
|
import { AdvisorRegistry, AdvisoryContext, AdvisoryExecutor, TriggerEvaluator, assertBudgetEnforceable, } from '../../advisory/index.js';
|
|
3
|
-
import { drainQueuedMessages } from '../../agents/handle.js';
|
|
4
3
|
import { AuthorizationGate } from '../../authorization/gate.js';
|
|
5
|
-
import {
|
|
6
|
-
import { repairToolMessageHistory, toolHistoryRepairChanged, } from '../../compaction/dangling.js';
|
|
4
|
+
import { repairToolMessageHistory, toolHistoryRepairChanged } from '../../compaction/dangling.js';
|
|
7
5
|
import { extractFromUserMessage } from '../../compaction/extractor.js';
|
|
8
6
|
import { WorkingStateManager } from '../../compaction/manager.js';
|
|
9
7
|
import { serializeState as serializeWorkingState } from '../../compaction/serializer.js';
|
|
@@ -11,225 +9,44 @@ import { restoreWorkingState, snapshotWorkingState } from '../../compaction/wire
|
|
|
11
9
|
import { CompactionConfigSchema } from '../../config/runtime.js';
|
|
12
10
|
import { TOOL_OUTPUT_DIR_NAME } from '../../constants/tools/index.js';
|
|
13
11
|
import { EmergencySaveManager } from '../../manager/run/emergency.js';
|
|
14
|
-
import { resolveModelPricing } from '../../pricing/index.js';
|
|
15
12
|
import { PromptContributionRegistry } from '../../prompt/contributions.js';
|
|
16
13
|
import { resolveProviderCapabilities } from '../../provider/capabilities.js';
|
|
17
|
-
import {
|
|
18
|
-
import { withProviderFallback, } from '../../provider/fallback.js';
|
|
19
|
-
import { resolveStreamIdleTimeoutMs, withStreamIdleTimeout } from '../../provider/idle-timeout.js';
|
|
20
|
-
import { withProviderRetry } from '../../provider/retry.js';
|
|
14
|
+
import { withStreamIdleTimeout } from '../../provider/idle-timeout.js';
|
|
21
15
|
import { withTokenBudget } from '../../provider/token-budget.js';
|
|
22
|
-
import { resolveAttachments } from '../../store/attachment/index.js';
|
|
23
16
|
import { GENAI, NAMZU, agentRunSpanName, parentContext, serializeSpan, } from '../../telemetry/attributes.js';
|
|
24
|
-
import { recordRunDuration } from '../../telemetry/metrics.js';
|
|
25
17
|
import { getTracer } from '../../telemetry/runtime-accessors.js';
|
|
26
18
|
import { buildAdvisoryTools } from '../../tools/advisory/index.js';
|
|
27
19
|
import { SearchToolsTool } from '../../tools/builtins/search-tools.js';
|
|
28
20
|
import { STRUCTURED_OUTPUT_TOOL_NAME, createStructuredOutputTool, } from '../../tools/builtins/structuredOutput.js';
|
|
29
21
|
import { buildTaskTools } from '../../tools/task/index.js';
|
|
22
|
+
import { isTerminalStatus } from '../../types/common/index.js';
|
|
30
23
|
import { NamzuError } from '../../types/errors/index.js';
|
|
31
24
|
import { autoApproveHandler, } from '../../types/hitl/index.js';
|
|
32
25
|
import { createSystemMessage, } from '../../types/message/index.js';
|
|
33
|
-
import { cancelCauseOf } from '../../types/run/cancel-cause.js';
|
|
34
|
-
import { resolveRunEventReplay } from '../../types/run/event-cursor.js';
|
|
35
|
-
import { memoryCandidateFor } from '../../types/run/memory-promotion.js';
|
|
36
26
|
import { toErrorMessage } from '../../utils/error.js';
|
|
37
|
-
import { generateCheckpointId, generateRunId } from '../../utils/id.js';
|
|
38
27
|
import { errorAttributes } from '../../utils/log/exception.js';
|
|
39
28
|
import { AwaitedJobs } from '../jobs/awaited-jobs.js';
|
|
40
|
-
import {
|
|
41
|
-
import { CheckpointManager } from './checkpoint.js';
|
|
42
|
-
import {
|
|
43
|
-
import { EventTranslator } from './events.js';
|
|
29
|
+
import { catchUpFromCursor, settlePreStartCancellation } from './cancelled-before-start.js';
|
|
30
|
+
import { CheckpointManager, findPendingCheckpoint } from './checkpoint.js';
|
|
31
|
+
import { finalizeRun } from './finalize-run.js';
|
|
44
32
|
import { GuardCoordinator } from './guard.js';
|
|
45
|
-
import { runInputGuardrails
|
|
33
|
+
import { runInputGuardrails } from './guardrails.js';
|
|
46
34
|
import { IterationOrchestrator } from './iteration/index.js';
|
|
47
35
|
import { isCompactionMessage } from './iteration/phases/compaction.js';
|
|
48
36
|
import { isWorkingMemoryMessage } from './iteration/phases/working-memory.js';
|
|
49
37
|
import { applyLifecycleHookResults } from './plugin-hooks.js';
|
|
50
|
-
import {
|
|
38
|
+
import { prepareRun, projectStateBearingHistory, resolveProviderContextWindow, selectedResumeStates, } from './prepare-run.js';
|
|
51
39
|
import { PromptBuilder } from './prompt.js';
|
|
52
40
|
import { PendingAnswers, QuestionParkBinding } from './question-park.js';
|
|
41
|
+
import { releaseRunResources } from './release-run.js';
|
|
53
42
|
import { RepeatCallTracker } from './repeat-call.js';
|
|
54
|
-
import { resolveMaxRequestRichContentBytes } from './request-rich-content.js';
|
|
55
43
|
import { ResultAssembler } from './result.js';
|
|
56
|
-
import { applyPendingResume, interruptedToolCalls, planCrashResume, planPendingResume, recoverCompletedCalls, } from './resume-pending.js';
|
|
57
|
-
import { acquireSandbox
|
|
44
|
+
import { answersParkOf, applyPendingResume, interruptedToolCalls, planCrashResume, planPendingResume, recoverCompletedCalls, supersededByRecovery, } from './resume-pending.js';
|
|
45
|
+
import { acquireSandbox } from './sandbox-lifecycle.js';
|
|
58
46
|
import { SteeringBinding, isOperatorUserMessage } from './steering.js';
|
|
59
|
-
import { resolveQueryBudget } from './token-budget.js';
|
|
60
|
-
import { assertMaxToolCalls } from './tool-call-budget.js';
|
|
61
47
|
import { ToolGrantSet } from './tool-grants.js';
|
|
62
48
|
import { createToolPause } from './tool-pause.js';
|
|
63
49
|
import { ToolingBootstrap } from './tooling.js';
|
|
64
|
-
const selectedResumeStates = new WeakMap();
|
|
65
|
-
/**
|
|
66
|
-
* Refuse to price a run whose tokens two differently-priced members may produce.
|
|
67
|
-
*
|
|
68
|
-
* `RunPersistence` holds ONE {@link ModelPricing} table and applies it to every
|
|
69
|
-
* accumulation regardless of which model produced the tokens. Across a swap that
|
|
70
|
-
* makes `costInfo.totalCost` wrong by an unbounded margin, and silently — the
|
|
71
|
-
* number keeps the shape of an answer. `CostInfo` cannot express the truth
|
|
72
|
-
* either: it carries `inputCostPer1M` / `outputCostPer1M`, and there is no
|
|
73
|
-
* honest value for those once a total spans two rate cards.
|
|
74
|
-
*
|
|
75
|
-
* So the total is refused rather than blended. Naming what that costs is part
|
|
76
|
-
* of the refusal, because the caller loses `costLimitUsd` with it: the guard
|
|
77
|
-
* enforces that limit from this same accumulated total, and a limit enforced
|
|
78
|
-
* with the wrong rate card stops a run early or late by the same unbounded
|
|
79
|
-
* margin. A budget that is quietly wrong is worse than a budget that is
|
|
80
|
-
* declined.
|
|
81
|
-
*
|
|
82
|
-
* Reachable, not decorative: a host that passes `pricing` and declares a chain
|
|
83
|
-
* hits it on the first call. It costs `@namzu/cli` nothing, which passes no
|
|
84
|
-
* pricing at all — its `/cost` already reports that the provider gave no price.
|
|
85
|
-
*
|
|
86
|
-
* The way out is per-member pricing, which needs a `CostInfo` that can sum over
|
|
87
|
-
* heterogeneous rates. That is a public-type change and it is not this one.
|
|
88
|
-
*/
|
|
89
|
-
function assertCostIsAttributable(chain, pricing) {
|
|
90
|
-
if (pricing === undefined || chain.length < 2)
|
|
91
|
-
return;
|
|
92
|
-
throw new NamzuError({
|
|
93
|
-
code: 'invalid_config',
|
|
94
|
-
message: `A provider chain of ${chain.length} members was declared together with a single pricing table. ` +
|
|
95
|
-
'One table cannot price two members, so the run would report a total that is wrong by an unbounded ' +
|
|
96
|
-
'margin — and `runConfig.costLimitUsd` would be enforced against that same wrong total. ' +
|
|
97
|
-
'Either drop `pricing` (usage is still reported per model in the run) or declare one member.',
|
|
98
|
-
details: { chainLength: chain.length },
|
|
99
|
-
});
|
|
100
|
-
}
|
|
101
|
-
/**
|
|
102
|
-
* Refuse a budget that cannot be measured.
|
|
103
|
-
*
|
|
104
|
-
* `runConfig.costLimitUsd` is enforced against `costInfo.totalCost`, and that
|
|
105
|
-
* total only moves for tokens something has a rate for. A model no rate card
|
|
106
|
-
* covers therefore produced a limit that could never trip — a host that set a
|
|
107
|
-
* cost cap had no cost cap, and nothing said so. That was every run before the
|
|
108
|
-
* price catalogue existed, which is how it went unnoticed.
|
|
109
|
-
*
|
|
110
|
-
* Refusing at the front is the cheap half of the answer: it costs the caller
|
|
111
|
-
* nothing, fires before any spend, and names both ways out. The other half is
|
|
112
|
-
* the `cost_unmeasurable` stop, for the models this cannot see — a step naming
|
|
113
|
-
* its own, or a chain member declaring one.
|
|
114
|
-
*
|
|
115
|
-
* This is the same shape `advisory/budget.ts` already applies to
|
|
116
|
-
* `AdvisoryBudget.maxCostPerRun`, one layer down, and for the same reason. The
|
|
117
|
-
* run path simply never had it.
|
|
118
|
-
*/
|
|
119
|
-
function assertBudgetIsMeasurable(params) {
|
|
120
|
-
const limit = params.runConfig.costLimitUsd;
|
|
121
|
-
if (limit === undefined || limit <= 0)
|
|
122
|
-
return;
|
|
123
|
-
// A host-supplied table prices whatever it is pointed at, so a caller who
|
|
124
|
-
// brought one has answered the question themselves.
|
|
125
|
-
if (params.pricing !== undefined)
|
|
126
|
-
return;
|
|
127
|
-
const model = params.runConfig.model;
|
|
128
|
-
if (resolveModelPricing(params.provider.id, model) !== undefined)
|
|
129
|
-
return;
|
|
130
|
-
throw new NamzuError({
|
|
131
|
-
code: 'invalid_config',
|
|
132
|
-
message: `runConfig.costLimitUsd is set to ${limit}, but no rate is known for model "${model}" on ` +
|
|
133
|
-
`provider "${params.provider.id}". The limit is enforced against the run's accumulated ` +
|
|
134
|
-
'cost, and tokens with no rate never reach that total — so the budget would read as ' +
|
|
135
|
-
'satisfied for the whole run and stop nothing. Either pass `pricing` to declare the rate ' +
|
|
136
|
-
'yourself, add the model to packages/sdk/src/pricing/rates.source.json, or drop ' +
|
|
137
|
-
'`costLimitUsd` and bound the run with `tokenBudget`, which is measurable here.',
|
|
138
|
-
details: { model, providerId: params.provider.id, costLimitUsd: limit },
|
|
139
|
-
});
|
|
140
|
-
}
|
|
141
|
-
/**
|
|
142
|
-
* Ask the driver what this model's window is, and never let the answer
|
|
143
|
-
* cost the run.
|
|
144
|
-
*
|
|
145
|
-
* Three outcomes collapse to two here on purpose. No member and a resolved
|
|
146
|
-
* `undefined` both mean "no answer" — the distinction matters to a driver
|
|
147
|
-
* author, not to a caller about to fall through to the table. A rejection
|
|
148
|
-
* is the third, and it is logged rather than propagated: a run that would
|
|
149
|
-
* have worked on the table must not fail because a listing endpoint was
|
|
150
|
-
* down.
|
|
151
|
-
*/
|
|
152
|
-
async function resolveProviderContextWindow(provider, model, signal, timeoutMs, log) {
|
|
153
|
-
if (!provider.resolveContextWindow || !model)
|
|
154
|
-
return undefined;
|
|
155
|
-
if (signal?.aborted)
|
|
156
|
-
return undefined;
|
|
157
|
-
// The resolver is an optional optimisation that runs before RunContext
|
|
158
|
-
// owns its child controller. Give it a private deadline signal and fuse
|
|
159
|
-
// caller cancellation into that transport in the safe direction: neither
|
|
160
|
-
// outcome aborts the caller's controller. Passing a signal is necessary
|
|
161
|
-
// but not sufficient, because a third-party driver can accept it and still
|
|
162
|
-
// leave its promise pending; the race below makes fallback independent of
|
|
163
|
-
// driver cooperation. Promise.race keeps the losing provider promise
|
|
164
|
-
// observed, so a later rejection cannot become unhandled.
|
|
165
|
-
const deadline = new AbortController();
|
|
166
|
-
const resolverSignal = signal ? AbortSignal.any([signal, deadline.signal]) : deadline.signal;
|
|
167
|
-
const interrupted = Symbol('provider-context-window-interrupted');
|
|
168
|
-
let onAbort;
|
|
169
|
-
const interruption = new Promise((resolve) => {
|
|
170
|
-
onAbort = () => resolve(interrupted);
|
|
171
|
-
resolverSignal.addEventListener('abort', onAbort, { once: true });
|
|
172
|
-
});
|
|
173
|
-
// Direct QueryParams callers can supply a large run deadline. The clamp
|
|
174
|
-
// avoids Node's >2^31-1 one-millisecond timer coercion during metadata lookup.
|
|
175
|
-
// Metadata discovery remains optional and bounded even without a run deadline.
|
|
176
|
-
const deadlineMs = timeoutMs === 0 ? 5_000 : Math.min(Math.max(0, timeoutMs), 2_147_483_647);
|
|
177
|
-
const timer = setTimeout(() => {
|
|
178
|
-
deadline.abort(new Error(`Provider context-window lookup exceeded ${deadlineMs}ms`));
|
|
179
|
-
}, deadlineMs);
|
|
180
|
-
try {
|
|
181
|
-
const resolution = provider.resolveContextWindow(model, resolverSignal);
|
|
182
|
-
const reported = await Promise.race([resolution, interruption]);
|
|
183
|
-
if (reported === interrupted) {
|
|
184
|
-
if (deadline.signal.aborted) {
|
|
185
|
-
log.debug('Provider context-window lookup timed out; using the table', {
|
|
186
|
-
'namzu.model.id': model,
|
|
187
|
-
'namzu.runtime.timeout_ms': deadlineMs,
|
|
188
|
-
});
|
|
189
|
-
}
|
|
190
|
-
return undefined;
|
|
191
|
-
}
|
|
192
|
-
return typeof reported === 'number' && reported > 0 ? reported : undefined;
|
|
193
|
-
}
|
|
194
|
-
catch (err) {
|
|
195
|
-
log.debug('Provider could not report a context window; using the table', {
|
|
196
|
-
'namzu.model.id': model,
|
|
197
|
-
'namzu.error.message': toErrorMessage(err),
|
|
198
|
-
});
|
|
199
|
-
return undefined;
|
|
200
|
-
}
|
|
201
|
-
finally {
|
|
202
|
-
clearTimeout(timer);
|
|
203
|
-
if (onAbort)
|
|
204
|
-
resolverSignal.removeEventListener('abort', onAbort);
|
|
205
|
-
}
|
|
206
|
-
}
|
|
207
|
-
/**
|
|
208
|
-
* Project historical system messages exactly as a new run will persist them.
|
|
209
|
-
*
|
|
210
|
-
* Arbitrary historical prompt floors are rebuilt for this run and therefore
|
|
211
|
-
* never reach its provider-bound conversation. Repair must happen AFTER that
|
|
212
|
-
* removal: treating a soon-to-be-dropped system message as a tool-result
|
|
213
|
-
* boundary can replace an exact real result with an invented unknown outcome.
|
|
214
|
-
* The two state-bearing system forms survive; fresh inherited compaction is
|
|
215
|
-
* pinned until this run can prove it reconstructed equivalent state.
|
|
216
|
-
*/
|
|
217
|
-
function projectStateBearingHistory(messages, options) {
|
|
218
|
-
const projected = [];
|
|
219
|
-
for (const message of messages) {
|
|
220
|
-
if (message.role !== 'system') {
|
|
221
|
-
projected.push(message);
|
|
222
|
-
continue;
|
|
223
|
-
}
|
|
224
|
-
if (isCompactionMessage(message.content)) {
|
|
225
|
-
projected.push(options.pinCompaction ? { ...message, retain: true } : message);
|
|
226
|
-
}
|
|
227
|
-
else if (isWorkingMemoryMessage(message.content)) {
|
|
228
|
-
projected.push(message);
|
|
229
|
-
}
|
|
230
|
-
}
|
|
231
|
-
return collapseProjectInstructionSnapshots(projected);
|
|
232
|
-
}
|
|
233
50
|
/**
|
|
234
51
|
* Remove the incomplete turn still owned by a durable resume plan.
|
|
235
52
|
*
|
|
@@ -295,469 +112,10 @@ function withOwnedResumeOutcomes(messages, assistant, recovered) {
|
|
|
295
112
|
];
|
|
296
113
|
}
|
|
297
114
|
export async function* query(params) {
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
// before opening a budget or persisting a run without its owning identity.
|
|
301
|
-
const missingFields = ['sessionId', 'topicId', 'projectId', 'tenantId'].filter((field) => !params[field]);
|
|
302
|
-
if (missingFields.length > 0) {
|
|
303
|
-
throw new NamzuError({
|
|
304
|
-
code: 'invalid_config',
|
|
305
|
-
message: `query requires sessionId, topicId, projectId, and tenantId; missing: ${missingFields.join(', ')}.`,
|
|
306
|
-
details: { missingFields },
|
|
307
|
-
});
|
|
308
|
-
}
|
|
309
|
-
const selectedResumeState = selectedResumeStates.get(params);
|
|
310
|
-
selectedResumeStates.delete(params);
|
|
311
|
-
// Resolved at the DOOR, before a run id exists or a logger is built.
|
|
312
|
-
// A caller who set both spellings of a renamed field has a config bug,
|
|
313
|
-
// and refusing it here costs them nothing; refusing it at the read site
|
|
314
|
-
// deep in the loop turns the same bug into a mid-run failure, after a
|
|
315
|
-
// provider call has been paid for and a partial transcript written.
|
|
316
|
-
const promptCache = params.promptCache;
|
|
317
|
-
const taskScheduler = params.taskScheduler;
|
|
318
|
-
const streamIdleTimeoutMs = resolveStreamIdleTimeoutMs(params.runConfig.streamIdleTimeoutMs);
|
|
319
|
-
const maxRequestRichContentBytes = resolveMaxRequestRichContentBytes(params.runConfig.maxRequestRichContentBytes);
|
|
320
|
-
const sandboxTeardownTimeoutMs = resolveSandboxTeardownTimeoutMs(params.sandboxTeardownTimeoutMs);
|
|
321
|
-
// Persist the EFFECTIVE value, not only an override. A run replayed after a
|
|
322
|
-
// later release must be able to explain which liveness policy settled it;
|
|
323
|
-
// an absent field whose meaning follows the currently-installed default
|
|
324
|
-
// would rewrite that evidence at read time.
|
|
325
|
-
const runConfig = {
|
|
326
|
-
...params.runConfig,
|
|
327
|
-
streamIdleTimeoutMs,
|
|
328
|
-
maxRequestRichContentBytes,
|
|
329
|
-
};
|
|
330
|
-
// The run's one correlated logger, built before anything below needs
|
|
331
|
-
// one — the migration check, the retry/fallback wrappers and `ctx`
|
|
332
|
-
// itself all read this SAME object, so a retry warning and the run
|
|
333
|
-
// record it retried for carry the identical `namzu.run.id` instead of
|
|
334
|
-
// three separate `getRootLogger()` reads that happened to agree by
|
|
335
|
-
// accident. `runId` is resolved here, once, rather than left to
|
|
336
|
-
// `build`'s own `config.runId ?? generateRunId()` fallback —
|
|
337
|
-
// generating it twice would silently hand the log and the run two
|
|
338
|
-
// different ids.
|
|
339
|
-
const runId = params.runId ?? generateRunId();
|
|
340
|
-
const budget = await resolveQueryBudget(params, runId, selectedResumeState);
|
|
341
|
-
const log = RunContextFactory.buildLogger({
|
|
342
|
-
agentName: params.agentName,
|
|
343
|
-
runConfig,
|
|
344
|
-
runId,
|
|
345
|
-
parentRunId: params.parentRunId,
|
|
346
|
-
sessionId: params.sessionId,
|
|
347
|
-
topicId: params.topicId,
|
|
348
|
-
projectId: params.projectId,
|
|
349
|
-
tenantId: params.tenantId,
|
|
350
|
-
});
|
|
351
|
-
// Every model call in the run — the loop's turns, the forced-final
|
|
352
|
-
// summary, advisory and compaction side calls — goes through this one
|
|
353
|
-
// wrapped provider, so the retry policy cannot be bypassed by a code
|
|
354
|
-
// path that happens to hold the raw driver.
|
|
355
|
-
// The logger is passed on purpose: `withProviderRetry` guards every one
|
|
356
|
-
// of its warns behind `options.log`, and this is its only production
|
|
357
|
-
// call site — so without it the "failed, retrying" and "failed, giving
|
|
358
|
-
// up" lines were dead code and a backoff left no trace anywhere.
|
|
359
|
-
//
|
|
360
|
-
// With a chain declared, the same sentence holds two levels out. The idle
|
|
361
|
-
// watchdog is applied to each raw member, retry wraps that, and fallback
|
|
362
|
-
// wraps the members: `fallback(retry(idle(m0)), retry(idle(m1)), …)`. The
|
|
363
|
-
// idle layer cannot sit outside retry, because its timer would then count a
|
|
364
|
-
// legitimate backoff as provider silence. This order is not a
|
|
365
|
-
// preference. Assembled the other way round — which is what a host gets if
|
|
366
|
-
// it wraps its own chain and hands the result in, because this function
|
|
367
|
-
// would then wrap THAT in retry — an exhausted chain gets restarted from
|
|
368
|
-
// the head by the outer loop and a throttle on the last member is counted
|
|
369
|
-
// by two budgets. Building it here is what makes the order unspellable
|
|
370
|
-
// wrong.
|
|
371
|
-
const chain = [
|
|
372
|
-
{ provider: params.provider },
|
|
373
|
-
...(params.fallbackProviders ?? []),
|
|
374
|
-
];
|
|
375
|
-
assertCostIsAttributable(chain, params.pricing);
|
|
376
|
-
assertBudgetIsMeasurable(params);
|
|
377
|
-
const withRecovery = (provider) => {
|
|
378
|
-
const withIdleBound = withStreamIdleTimeout(provider, {
|
|
379
|
-
idleTimeoutMs: streamIdleTimeoutMs,
|
|
380
|
-
log,
|
|
381
|
-
});
|
|
382
|
-
const metered = withTokenBudget(withIdleBound, budget);
|
|
383
|
-
return params.retry === false
|
|
384
|
-
? metered
|
|
385
|
-
: withProviderRetry(metered, {
|
|
386
|
-
config: params.retry,
|
|
387
|
-
log,
|
|
388
|
-
canRetry: () => budget.remaining > 0,
|
|
389
|
-
});
|
|
390
|
-
};
|
|
391
|
-
// Who is serving right now, for the run RECORD rather than for the request.
|
|
392
|
-
//
|
|
393
|
-
// It starts at the head and moves only when the chain does, which is the
|
|
394
|
-
// whole of the truth because the cursor never rewinds. The run cannot read
|
|
395
|
-
// this off `resilientProvider`: that wrapper reports the head's `id` on
|
|
396
|
-
// purpose, so asking it produces the declaration back — the defect this
|
|
397
|
-
// record exists to fix.
|
|
398
|
-
const serving = {
|
|
399
|
-
current: { index: 0, providerId: params.provider.id },
|
|
400
|
-
};
|
|
401
|
-
const resilientProvider = withProviderFallback(chain.map((member) => ({
|
|
402
|
-
...member,
|
|
403
|
-
provider: withRecovery(member.provider),
|
|
404
|
-
})), {
|
|
405
|
-
log,
|
|
406
|
-
canFallback: () => budget.remaining > 0,
|
|
407
|
-
onSwap: (to) => {
|
|
408
|
-
serving.current = to;
|
|
409
|
-
// `ctx` is declared below and is initialized before anything can
|
|
410
|
-
// call the provider: this fires from inside a `chatStream`, and
|
|
411
|
-
// the first one is issued by the loop that `ctx` is built for.
|
|
412
|
-
ctx.runMgr.setServingProvider(to.providerId);
|
|
413
|
-
},
|
|
414
|
-
});
|
|
415
|
-
// Asked ONCE, here, before the loop exists. Both readers are synchronous
|
|
416
|
-
// and hot, so this can never move inside the iteration — and a driver
|
|
417
|
-
// that rejects, or one that hangs until the run is cancelled, must not
|
|
418
|
-
// take down a run the table could have served perfectly well. That is
|
|
419
|
-
// why the failure path is a swallow with a log rather than a throw: the
|
|
420
|
-
// window is an optimisation over a working default, not a prerequisite.
|
|
421
|
-
const providerContextWindow = await resolveProviderContextWindow(resilientProvider, runConfig.model, params.signal, runConfig.timeoutMs, log);
|
|
422
|
-
const modelContextWindows = new Map();
|
|
423
|
-
if (runConfig.model)
|
|
424
|
-
modelContextWindows.set(runConfig.model, providerContextWindow);
|
|
425
|
-
// The mode this conversation was left in, when the run config names none.
|
|
426
|
-
// Read once, before the loop exists, for the same reason the context
|
|
427
|
-
// window is: the executor's resolver is synchronous and hot.
|
|
428
|
-
//
|
|
429
|
-
// A store that throws is not a run failure — the run falls back to the
|
|
430
|
-
// config's answer, which is exactly what it did before this existed.
|
|
431
|
-
const topicState = params.topicStateStore
|
|
432
|
-
? await params.topicStateStore
|
|
433
|
-
.getState(params.topicId, params.tenantId)
|
|
434
|
-
.catch((err) => {
|
|
435
|
-
log.debug('Could not read the topic state; using the run config', {
|
|
436
|
-
'namzu.topic.id': params.topicId,
|
|
437
|
-
'namzu.error.message': toErrorMessage(err),
|
|
438
|
-
});
|
|
439
|
-
return null;
|
|
440
|
-
})
|
|
441
|
-
: null;
|
|
442
|
-
// Whatever a host left for "the next run", taken and cleared in one
|
|
443
|
-
// compare-and-set write. Prepended to the messages this run starts from,
|
|
444
|
-
// so it is in the FIRST request rather than arriving a turn late.
|
|
445
|
-
//
|
|
446
|
-
// Cleared as it is read: a queue read and cleared separately re-delivers
|
|
447
|
-
// on a crash between the two, and "start with this" arriving twice is a
|
|
448
|
-
// different instruction from the one that was left.
|
|
449
|
-
const queuedForThisRun = params.topicStateStore
|
|
450
|
-
? await drainQueuedMessages(params.topicStateStore, params.topicId, params.tenantId).catch((err) => {
|
|
451
|
-
log.debug('Could not drain the topic queue; starting without it', {
|
|
452
|
-
'namzu.topic.id': params.topicId,
|
|
453
|
-
'namzu.error.message': toErrorMessage(err),
|
|
454
|
-
});
|
|
455
|
-
return [];
|
|
456
|
-
})
|
|
457
|
-
: [];
|
|
458
|
-
// One effective list, used everywhere the run is seeded from. Three
|
|
459
|
-
// branches below push from it, and computing it at each would be three
|
|
460
|
-
// places to forget the queue.
|
|
461
|
-
//
|
|
462
|
-
// Stored attachments are resolved HERE, once, before the messages reach
|
|
463
|
-
// the run record. Resolving later — at the provider boundary — would put
|
|
464
|
-
// refs in the durable transcript and in every checkpoint, and a run
|
|
465
|
-
// resumed against a store that had since forgotten a ref would fail
|
|
466
|
-
// replaying its own history rather than at the moment somebody asked for
|
|
467
|
-
// the bytes. Every failure refuses: a message that silently lost its
|
|
468
|
-
// image is a model answering about a picture it never saw.
|
|
469
|
-
const seeded = queuedForThisRun.length > 0 ? [...queuedForThisRun, ...params.messages] : params.messages;
|
|
470
|
-
let resolvedInitialMessages;
|
|
471
|
-
let attachmentResolutionCancelled = false;
|
|
472
|
-
try {
|
|
473
|
-
resolvedInitialMessages = [
|
|
474
|
-
...(await resolveAttachments(seeded, params.attachmentStore, {
|
|
475
|
-
signal: params.signal,
|
|
476
|
-
timeoutMs: params.attachmentResolveTimeoutMs,
|
|
477
|
-
})),
|
|
478
|
-
];
|
|
479
|
-
params.signal?.throwIfAborted();
|
|
480
|
-
}
|
|
481
|
-
catch (error) {
|
|
482
|
-
// Attachment materialization precedes RunContext construction so stored
|
|
483
|
-
// bytes never enter a live run's checkpoints. Cancellation still belongs
|
|
484
|
-
// to that run: preserve the exact input refs, build the context below, and
|
|
485
|
-
// let its normal terminal path classify/persist a cancelled Run. Every
|
|
486
|
-
// other store failure remains a pre-run refusal.
|
|
487
|
-
if (!params.signal?.aborted || error !== params.signal.reason)
|
|
488
|
-
throw error;
|
|
489
|
-
resolvedInitialMessages = [...seeded];
|
|
490
|
-
attachmentResolutionCancelled = true;
|
|
491
|
-
}
|
|
492
|
-
if (!attachmentResolutionCancelled && params.projectInstructionContext?.prepareInitialSnapshot) {
|
|
493
|
-
const preparationSignal = params.signal ?? new AbortController().signal;
|
|
494
|
-
let snapshot;
|
|
495
|
-
try {
|
|
496
|
-
const prepared = await awaitProjectInstructionCallback(preparationSignal, () => params.projectInstructionContext?.prepareInitialSnapshot?.({
|
|
497
|
-
messages: [...resolvedInitialMessages],
|
|
498
|
-
signal: preparationSignal,
|
|
499
|
-
}));
|
|
500
|
-
// The callback promise can settle, remove its listener, and queue this
|
|
501
|
-
// continuation immediately before a queued abort. Publication is a
|
|
502
|
-
// separate authority boundary, so fence it too.
|
|
503
|
-
preparationSignal.throwIfAborted();
|
|
504
|
-
snapshot = prepared;
|
|
505
|
-
}
|
|
506
|
-
catch (error) {
|
|
507
|
-
// This callback runs before RunContext owns its child controller. A
|
|
508
|
-
// caller cancellation here still belongs to the run: publish no late
|
|
509
|
-
// snapshot and let the context below settle the normal cancelled Run.
|
|
510
|
-
// Compare the exact reason: a callback failure that won first must not
|
|
511
|
-
// be erased merely because cancellation arrived before this catch ran.
|
|
512
|
-
if (!preparationSignal.aborted || error !== preparationSignal.reason)
|
|
513
|
-
throw error;
|
|
514
|
-
}
|
|
515
|
-
if (snapshot !== undefined) {
|
|
516
|
-
resolvedInitialMessages = replaceProjectInstructionSnapshot(resolvedInitialMessages, snapshot, 'before-latest-user');
|
|
517
|
-
}
|
|
518
|
-
}
|
|
519
|
-
const pendingHistoryRepairs = [];
|
|
520
|
-
const projectedInitialMessages = collapseProjectInstructionSnapshots(params.resumeFromCheckpoint || params.continuationMode
|
|
521
|
-
? resolvedInitialMessages
|
|
522
|
-
: projectStateBearingHistory(resolvedInitialMessages, {
|
|
523
|
-
pinCompaction: true,
|
|
524
|
-
}));
|
|
525
|
-
const initialRepair = params.resumeFromCheckpoint
|
|
526
|
-
? { messages: projectedInitialMessages, report: undefined }
|
|
527
|
-
: repairToolMessageHistory(projectedInitialMessages);
|
|
528
|
-
const initialMessages = initialRepair.messages;
|
|
529
|
-
if (initialRepair.report && toolHistoryRepairChanged(initialRepair.report)) {
|
|
530
|
-
pendingHistoryRepairs.push({
|
|
531
|
-
source: 'fresh-history',
|
|
532
|
-
report: initialRepair.report,
|
|
533
|
-
});
|
|
534
|
-
log.warn('Repaired provider-invalid tool history before starting the run', {
|
|
535
|
-
[NAMZU.RUN_ID]: runId,
|
|
536
|
-
'namzu.history.source': 'fresh-history',
|
|
537
|
-
'namzu.history.duplicate_tool_results_removed': initialRepair.report.duplicateToolResultsRemoved,
|
|
538
|
-
'namzu.history.orphaned_tool_results_removed': initialRepair.report.orphanedToolResultsRemoved,
|
|
539
|
-
'namzu.history.synthetic_tool_results_inserted': initialRepair.report.syntheticToolResultsInserted,
|
|
540
|
-
});
|
|
541
|
-
}
|
|
542
|
-
const ctx = RunContextFactory.build({
|
|
543
|
-
budget,
|
|
544
|
-
...(topicState ? { topicPermissionMode: topicState.permissionMode } : {}),
|
|
545
|
-
...(params.permissionModeRef ? { permissionModeRef: params.permissionModeRef } : {}),
|
|
546
|
-
agentId: params.agentId,
|
|
547
|
-
agentName: params.agentName,
|
|
548
|
-
runConfig,
|
|
549
|
-
provider: resilientProvider,
|
|
550
|
-
workingDirectory: params.workingDirectory,
|
|
551
|
-
pricing: params.pricing,
|
|
552
|
-
enableActivityTracking: params.enableActivityTracking,
|
|
553
|
-
messages: initialMessages,
|
|
554
|
-
signal: params.signal,
|
|
555
|
-
sessionId: params.sessionId,
|
|
556
|
-
topicId: params.topicId,
|
|
557
|
-
projectId: params.projectId,
|
|
558
|
-
tenantId: params.tenantId,
|
|
559
|
-
pathBuilder: params.pathBuilder,
|
|
560
|
-
checkpointStore: params.checkpointStore,
|
|
561
|
-
runStore: params.runStore,
|
|
562
|
-
runId,
|
|
563
|
-
parentRunId: params.parentRunId,
|
|
564
|
-
depth: params.depth,
|
|
565
|
-
log,
|
|
566
|
-
});
|
|
567
|
-
// Built here because the plan-approval closure below captures it, and
|
|
568
|
-
// its `emit` resolves `eventTranslator` at CALL time — the translator is
|
|
569
|
-
// a `const` some lines further down.
|
|
570
|
-
//
|
|
571
|
-
// The HANDOUT is therefore deliberately NOT here. A host given the box
|
|
572
|
-
// at this point can call `set` synchronously, `emit` reaches
|
|
573
|
-
// `eventTranslator` inside its temporal dead zone, and the run dies
|
|
574
|
-
// before it starts. That is not hypothetical: it is what the first
|
|
575
|
-
// version of this did, and the test that hands out the box and
|
|
576
|
-
// immediately swaps the policy is the one that found it.
|
|
577
|
-
const approvalPolicy = createRunApprovalPolicy({
|
|
578
|
-
runId: ctx.runId,
|
|
579
|
-
initial: {
|
|
580
|
-
// By identity against the default, not by presence. `resumeHandler`
|
|
581
|
-
// is REQUIRED on `QueryParams` — `drainQuery` substitutes
|
|
582
|
-
// `autoApproveHandler` before calling here — so "is it set" is
|
|
583
|
-
// always yes and would name every run `host`, including the ones
|
|
584
|
-
// approving everything unattended. Identity is what actually
|
|
585
|
-
// separates the two.
|
|
586
|
-
name: params.approvalPolicyName ??
|
|
587
|
-
(params.resumeHandler === autoApproveHandler ? AUTO_APPROVE_POLICY_NAME : 'host'),
|
|
588
|
-
handler: params.resumeHandler,
|
|
589
|
-
},
|
|
590
|
-
emit: (event) => eventTranslator.emitEvent(event),
|
|
591
|
-
});
|
|
592
|
-
const planApprovalIds = new Map();
|
|
593
|
-
ctx.planManager.setApprovalHandler(async (request) => {
|
|
594
|
-
let checkpointId = planApprovalIds.get(request.planId);
|
|
595
|
-
if (!checkpointId) {
|
|
596
|
-
checkpointId = generateCheckpointId();
|
|
597
|
-
planApprovalIds.set(request.planId, checkpointId);
|
|
598
|
-
}
|
|
599
|
-
// `.current.handler`, never a captured `params.resumeHandler`. That
|
|
600
|
-
// capture is what made changing the policy mean ending the run.
|
|
601
|
-
const decision = await approvalPolicy.current.handler({
|
|
602
|
-
type: 'plan_approval',
|
|
603
|
-
runId: ctx.runId,
|
|
604
|
-
checkpointId,
|
|
605
|
-
plan: {
|
|
606
|
-
planId: request.planId,
|
|
607
|
-
title: request.title,
|
|
608
|
-
steps: request.steps.map((s, i) => ({
|
|
609
|
-
id: s.id,
|
|
610
|
-
description: s.description,
|
|
611
|
-
toolName: s.toolName,
|
|
612
|
-
agentId: s.agentId,
|
|
613
|
-
dependsOn: s.dependsOn,
|
|
614
|
-
order: s.order ?? i + 1,
|
|
615
|
-
})),
|
|
616
|
-
summary: request.summary,
|
|
617
|
-
},
|
|
618
|
-
});
|
|
619
|
-
if (decision.action === 'approve_plan') {
|
|
620
|
-
// Optional approve-with-edits channel: the host may attach
|
|
621
|
-
// feedback to an approval. `PlanApprovalResponse.feedback`
|
|
622
|
-
// already exists on the type; threading it through lets the
|
|
623
|
-
// coordinator's approve_plan tool surface the user's edits in
|
|
624
|
-
// the same tool_result that unblocks the park. Bare approvals
|
|
625
|
-
// stay byte-identical (`{ approved: true }`).
|
|
626
|
-
return decision.feedback
|
|
627
|
-
? { approved: true, feedback: decision.feedback }
|
|
628
|
-
: { approved: true };
|
|
629
|
-
}
|
|
630
|
-
if (decision.action === 'reject_plan') {
|
|
631
|
-
return { approved: false, feedback: decision.feedback };
|
|
632
|
-
}
|
|
633
|
-
return { approved: false, feedback: `Action: ${decision.action}` };
|
|
634
|
-
});
|
|
635
|
-
const eventTranslator = new EventTranslator(ctx.runMgr, undefined, ctx.log);
|
|
636
|
-
eventTranslator.wireActivityStore(ctx.activityStore, ctx.runId);
|
|
637
|
-
eventTranslator.wirePlanManager(ctx.planManager, ctx.runId);
|
|
638
|
-
eventTranslator.setGeneration(params.claimFence);
|
|
639
|
-
let interruptHooksStarted = false;
|
|
640
|
-
const executeUserInterruptHooks = async (terminalError) => {
|
|
641
|
-
if (interruptHooksStarted ||
|
|
642
|
-
!params.pluginManager ||
|
|
643
|
-
!isCallerAbortError(terminalError, ctx.abortController.signal) ||
|
|
644
|
-
params.parentRunId !== undefined ||
|
|
645
|
-
(params.depth ?? 0) !== 0 ||
|
|
646
|
-
cancelCauseOf(ctx.abortController.signal.reason) !== 'user') {
|
|
647
|
-
return;
|
|
648
|
-
}
|
|
649
|
-
interruptHooksStarted = true;
|
|
650
|
-
try {
|
|
651
|
-
// Deliberately omit the already-aborted run signal. The lifecycle
|
|
652
|
-
// manager still supplies each handler its own deadline signal, while
|
|
653
|
-
// `run_interrupt`'s observational fan-out prevents one result from
|
|
654
|
-
// suppressing the cleanup hooks that follow it.
|
|
655
|
-
await params.pluginManager.executeHooks('run_interrupt', { runId: ctx.runId, cancelCause: 'user' }, eventTranslator.emitEvent);
|
|
656
|
-
}
|
|
657
|
-
catch (error) {
|
|
658
|
-
// Cancellation is the terminal authority. A hook event sink or an
|
|
659
|
-
// unexpected manager failure is reported, but cannot turn Stop into a
|
|
660
|
-
// failed run or prevent the durable cancellation verdict.
|
|
661
|
-
ctx.log.error('Run interrupt hooks did not settle cleanly', {
|
|
662
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
663
|
-
...errorAttributes(error),
|
|
664
|
-
});
|
|
665
|
-
}
|
|
666
|
-
};
|
|
115
|
+
const prepared = await prepareRun(params);
|
|
116
|
+
const { runConfig, budget, log, ctx, resilientProvider, serving, providerContextWindow, modelContextWindows, approvalPolicy, eventTranslator, executeUserInterruptHooks, pendingHistoryRepairs, initialMessages, queuedForThisRun, selectedResumeState, attachmentResolutionCancelled, streamIdleTimeoutMs, sandboxTeardownTimeoutMs, promptCache, taskScheduler, } = prepared;
|
|
667
117
|
if (attachmentResolutionCancelled) {
|
|
668
|
-
|
|
669
|
-
// observes cancellation, do only the work required to leave an honest
|
|
670
|
-
// durable run: initialize the record, retain the unresolved references,
|
|
671
|
-
// and settle through the ordinary cancellation classifier. Prompt
|
|
672
|
-
// contributions/cache, host callbacks, tools, plugins, sandbox, guardrails,
|
|
673
|
-
// advisors, and providers are all authority-bearing work and stay out.
|
|
674
|
-
// The dedicated root interrupt notification is the sole plugin exception:
|
|
675
|
-
// it runs after cancellation under its own deadline and cannot regain model
|
|
676
|
-
// or tool authority.
|
|
677
|
-
if (params.resumeFromCheckpoint && !selectedResumeState) {
|
|
678
|
-
// The canonical resume surface hands query the checkpoint state it
|
|
679
|
-
// already selected. A raw resume query has no such snapshot; after
|
|
680
|
-
// cancellation, reading the store again could hang without a signal,
|
|
681
|
-
// while persisting without it would erase the existing transcript.
|
|
682
|
-
// Refuse before binding/persisting rather than choose either failure.
|
|
683
|
-
ctx.abortController.signal.throwIfAborted();
|
|
684
|
-
}
|
|
685
|
-
const cancelledPrompt = params.systemPrompt ?? '';
|
|
686
|
-
const cancelledAssembler = new ResultAssembler({
|
|
687
|
-
runMgr: ctx.runMgr,
|
|
688
|
-
planManager: ctx.planManager,
|
|
689
|
-
activityStore: ctx.activityStore,
|
|
690
|
-
log: ctx.log,
|
|
691
|
-
emitEvent: eventTranslator.emitEvent,
|
|
692
|
-
drainPending: () => eventTranslator.drainPending(),
|
|
693
|
-
signal: ctx.abortController.signal,
|
|
694
|
-
});
|
|
695
|
-
const rootSpan = getTracer().startSpan(agentRunSpanName(params.agentName), {}, parentContext(params.parentSpan ?? selectedResumeState?.traceContext));
|
|
696
|
-
rootSpan.setAttributes({
|
|
697
|
-
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
698
|
-
[GENAI.AGENT_NAME]: params.agentName,
|
|
699
|
-
[GENAI.AGENT_ID]: params.agentId,
|
|
700
|
-
[GENAI.REQUEST_MODEL]: runConfig.model,
|
|
701
|
-
[GENAI.SYSTEM]: params.provider.id,
|
|
702
|
-
});
|
|
703
|
-
try {
|
|
704
|
-
await ctx.runMgr.init();
|
|
705
|
-
if (selectedResumeState) {
|
|
706
|
-
ctx.runMgr.restoreUsage(selectedResumeState.tokenUsage, selectedResumeState.costInfo, selectedResumeState.currentIteration);
|
|
707
|
-
for (const message of selectedResumeState.messages)
|
|
708
|
-
ctx.runMgr.pushMessage(message);
|
|
709
|
-
for (const queued of queuedForThisRun)
|
|
710
|
-
ctx.runMgr.pushMessage(queued);
|
|
711
|
-
}
|
|
712
|
-
else if (params.continuationMode) {
|
|
713
|
-
for (const message of initialMessages)
|
|
714
|
-
ctx.runMgr.pushMessage(message);
|
|
715
|
-
}
|
|
716
|
-
else {
|
|
717
|
-
ctx.runMgr.pushMessage(createSystemMessage(cancelledPrompt, 'cache'));
|
|
718
|
-
for (const message of initialMessages)
|
|
719
|
-
ctx.runMgr.pushMessage(message);
|
|
720
|
-
}
|
|
721
|
-
if (params.eventCursor) {
|
|
722
|
-
yield* catchUpFromCursor(ctx.runMgr, params.eventCursor, params.onEventReplay, params.claimFence, (error) => {
|
|
723
|
-
ctx.log.warn('Replay observer failed after attachment cancellation', {
|
|
724
|
-
'exception.message': toErrorMessage(error),
|
|
725
|
-
});
|
|
726
|
-
});
|
|
727
|
-
}
|
|
728
|
-
if (selectedResumeState) {
|
|
729
|
-
await eventTranslator.emitEvent({
|
|
730
|
-
type: 'run_resuming',
|
|
731
|
-
runId: ctx.runId,
|
|
732
|
-
fromCheckpointId: selectedResumeState.checkpointId,
|
|
733
|
-
});
|
|
734
|
-
yield* eventTranslator.drainPending();
|
|
735
|
-
}
|
|
736
|
-
ctx.runMgr.markRunning();
|
|
737
|
-
await eventTranslator.emitEvent({
|
|
738
|
-
type: 'run_started',
|
|
739
|
-
runId: ctx.runId,
|
|
740
|
-
systemPrompt: cancelledPrompt,
|
|
741
|
-
});
|
|
742
|
-
yield* eventTranslator.drainPending();
|
|
743
|
-
ctx.abortController.signal.throwIfAborted();
|
|
744
|
-
}
|
|
745
|
-
catch (error) {
|
|
746
|
-
// Attachment resolution has already observed the caller's abort. A
|
|
747
|
-
// reconnect callback can still throw while replay is being reported,
|
|
748
|
-
// but it cannot replace that terminal cause or turn a cancelled run
|
|
749
|
-
// into an unpersisted rejection.
|
|
750
|
-
const terminalError = ctx.abortController.signal.aborted
|
|
751
|
-
? ctx.abortController.signal.reason
|
|
752
|
-
: error;
|
|
753
|
-
await executeUserInterruptHooks(terminalError);
|
|
754
|
-
yield* eventTranslator.drainPending();
|
|
755
|
-
yield* cancelledAssembler.handleError(terminalError, rootSpan);
|
|
756
|
-
}
|
|
757
|
-
finally {
|
|
758
|
-
rootSpan.end();
|
|
759
|
-
}
|
|
760
|
-
return await cancelledAssembler.finalize();
|
|
118
|
+
return yield* settlePreStartCancellation(params, prepared);
|
|
761
119
|
}
|
|
762
120
|
const unsubscribeTaskStore = params.taskStore
|
|
763
121
|
? eventTranslator.wireTaskStore(params.taskStore, ctx.runId)
|
|
@@ -1220,7 +578,11 @@ export async function* query(params) {
|
|
|
1220
578
|
pluginManager: params.pluginManager,
|
|
1221
579
|
});
|
|
1222
580
|
const tracer = getTracer();
|
|
1223
|
-
|
|
581
|
+
// Whether the run reached its settle. Read by the `finally` below, and
|
|
582
|
+
// the only thing that distinguishes a run that finished from one whose
|
|
583
|
+
// consumer walked away — see `settleAbandonedRun`.
|
|
584
|
+
let settled = false;
|
|
585
|
+
const runBody = (async function* () {
|
|
1224
586
|
// Parent explicitly when a caller supplied one. Without this every
|
|
1225
587
|
// run starts its OWN root trace, so a supervisor delegating to three
|
|
1226
588
|
// children produced four disconnected traces instead of one tree —
|
|
@@ -1324,6 +686,11 @@ export async function* query(params) {
|
|
|
1324
686
|
// Decided during checkpoint restore, executed after the sandbox
|
|
1325
687
|
// exists — the approved tools may well need it.
|
|
1326
688
|
let pendingResume = null;
|
|
689
|
+
/**
|
|
690
|
+
* The cadence park this resume answered, when the decision is one the
|
|
691
|
+
* ordinary continue path carries out. See the restore path below.
|
|
692
|
+
*/
|
|
693
|
+
let answeredParkId;
|
|
1327
694
|
/** Tool results recovered from the transcript; see the restore path. */
|
|
1328
695
|
let recoveredResults = new Map();
|
|
1329
696
|
let emergencyManager;
|
|
@@ -1451,6 +818,48 @@ export async function* query(params) {
|
|
|
1451
818
|
params.pendingDecision && projectedCheckpoint.pending
|
|
1452
819
|
? planPendingResume(projectedCheckpoint, params.pendingDecision, ctx.log)
|
|
1453
820
|
: null;
|
|
821
|
+
// The park this resume ANSWERS even though there is no plan to
|
|
822
|
+
// carry the decision out through.
|
|
823
|
+
//
|
|
824
|
+
// `planPendingResume` covers the two arms whose decision has to
|
|
825
|
+
// reach something — the calls a `tool_review` park is about, the
|
|
826
|
+
// tool a `user_question` park is inside. An `iteration_checkpoint`
|
|
827
|
+
// park has neither, so it returns no plan, and the unpark further
|
|
828
|
+
// down — which ran only when there was one — never fired for it.
|
|
829
|
+
// A run that parked on the cadence, was resumed with
|
|
830
|
+
// `{action: 'continue'}` and went on to finish its work therefore
|
|
831
|
+
// kept reporting an OUTSTANDING park to `findPendingCheckpoint`,
|
|
832
|
+
// so a second resume of the finished run was refused with
|
|
833
|
+
// `awaiting-decision` for a decision already taken, and because
|
|
834
|
+
// `prune` skips an unresolved park the row could no longer be
|
|
835
|
+
// collected by anything.
|
|
836
|
+
//
|
|
837
|
+
// The decision IS carried out here — continuing is exactly what
|
|
838
|
+
// the loop below does, and a plan verdict is the answer to the
|
|
839
|
+
// question the plan park asked — so the park is resolved at the
|
|
840
|
+
// same point and with the same meaning "resolved" carries
|
|
841
|
+
// everywhere else: the record stays, and only its pending state
|
|
842
|
+
// ends. A `pause` is deliberately not resolved: it holds the
|
|
843
|
+
// park rather than answering it, which is how the live path
|
|
844
|
+
// treats it too. `answersParkOf` is the whole map, park type to
|
|
845
|
+
// answering decision, so an arm cannot go missing by being
|
|
846
|
+
// absent from a condition again — which is how the plan arm
|
|
847
|
+
// leaked a finished run's park.
|
|
848
|
+
//
|
|
849
|
+
// Resolving it does not depend on the resumed process being able
|
|
850
|
+
// to act on it, and that is deliberate: the plan's own fate is a
|
|
851
|
+
// separate defect (nothing restores a plan on the resume path at
|
|
852
|
+
// all, so the new process has none to approve, execute or
|
|
853
|
+
// reject) and making the resolution wait for it would leave the
|
|
854
|
+
// row outstanding for exactly the runs that need it cleared.
|
|
855
|
+
const parked = projectedCheckpoint.pending;
|
|
856
|
+
answeredParkId =
|
|
857
|
+
params.pendingDecision &&
|
|
858
|
+
parked !== undefined &&
|
|
859
|
+
parked.resolvedAt === undefined &&
|
|
860
|
+
answersParkOf(parked.request.type, params.pendingDecision)
|
|
861
|
+
? projectedCheckpoint.id
|
|
862
|
+
: undefined;
|
|
1454
863
|
// Recover completed observations and explicitly unknown outcomes.
|
|
1455
864
|
// A recorded start is not proof that its external effect failed.
|
|
1456
865
|
const unanswered = interruptedToolCalls(projectedCheckpoint.messages);
|
|
@@ -1756,6 +1165,9 @@ export async function* query(params) {
|
|
|
1756
1165
|
});
|
|
1757
1166
|
}
|
|
1758
1167
|
yield* resultAssembler.completeRun(rootSpan);
|
|
1168
|
+
// The run HAS settled, so the outer `finally` must not read
|
|
1169
|
+
// this as an abandonment — it would persist a second time.
|
|
1170
|
+
settled = true;
|
|
1759
1171
|
return await resultAssembler.finalize();
|
|
1760
1172
|
}
|
|
1761
1173
|
sandbox = acquisition.sandbox;
|
|
@@ -1816,7 +1228,15 @@ export async function* query(params) {
|
|
|
1816
1228
|
ctx.runMgr.setStopReason('input_guardrail');
|
|
1817
1229
|
ctx.runMgr.setLastError(inputVerdict.reason ?? 'blocked by an input guardrail');
|
|
1818
1230
|
yield* resultAssembler.completeRun(rootSpan);
|
|
1819
|
-
|
|
1231
|
+
// Same two lines as the sandbox path above, and for the same
|
|
1232
|
+
// reasons — with one that path does not have. This return used to
|
|
1233
|
+
// hand back `getRun()` without persisting, so the terminal state
|
|
1234
|
+
// reached the disk only because the abandonment path found
|
|
1235
|
+
// `settled` false and settled it a second time. A branch that
|
|
1236
|
+
// exists for runs which did NOT settle must not be the reason a
|
|
1237
|
+
// settled one is written down.
|
|
1238
|
+
settled = true;
|
|
1239
|
+
return await resultAssembler.finalize();
|
|
1820
1240
|
}
|
|
1821
1241
|
// Honor the approval a human already gave, before the loop's
|
|
1822
1242
|
// first model call. The sandbox exists by now, so an approved
|
|
@@ -1838,130 +1258,55 @@ export async function* query(params) {
|
|
|
1838
1258
|
}
|
|
1839
1259
|
await applyPendingResume(pendingResume, ctx.runMgr, toolExecutor, recoveredResults);
|
|
1840
1260
|
yield* eventTranslator.drainPending();
|
|
1841
|
-
// The decision has now actually been carried out, so the park
|
|
1842
|
-
// is no longer outstanding. Without this the checkpoint keeps
|
|
1843
|
-
// reporting `pending` with no `resolvedAt`, and an approval
|
|
1844
|
-
// queue re-serves a destructive call that already ran — which
|
|
1845
|
-
// defeats the entire point of recording the park.
|
|
1846
|
-
const resolvedCheckpointId = pendingResume.checkpointId;
|
|
1847
|
-
if (params.pendingDecision) {
|
|
1848
|
-
await checkpointMgr
|
|
1849
|
-
.unpark(resolvedCheckpointId, params.pendingDecision)
|
|
1850
|
-
.catch((err) => {
|
|
1851
|
-
ctx.log.error('Applied a pending decision but failed to clear the park', {
|
|
1852
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
1853
|
-
'namzu.checkpoint.id': resolvedCheckpointId,
|
|
1854
|
-
'exception.message': err instanceof Error ? err.message : String(err),
|
|
1855
|
-
});
|
|
1856
|
-
return null;
|
|
1857
|
-
});
|
|
1858
|
-
}
|
|
1859
|
-
}
|
|
1860
|
-
yield* iterationOrchestrator.runLoop();
|
|
1861
|
-
if (params.pluginManager) {
|
|
1862
|
-
const hookResults = await params.pluginManager.executeHooks('run_end', { runId: ctx.runId, signal: ctx.abortController.signal }, eventTranslator.emitEvent);
|
|
1863
|
-
applyLifecycleHookResults('run_end', hookResults);
|
|
1864
|
-
yield* eventTranslator.drainPending();
|
|
1865
|
-
// A delegated run says so once more, by name, so a hook that
|
|
1866
|
-
// only cares when a subagent finishes need not read parent ids
|
|
1867
|
-
// off every run_end.
|
|
1868
|
-
if (params.parentRunId !== undefined) {
|
|
1869
|
-
const stopResults = await params.pluginManager.executeHooks('subagent_stop', {
|
|
1870
|
-
runId: ctx.runId,
|
|
1871
|
-
parentRunId: params.parentRunId,
|
|
1872
|
-
signal: ctx.abortController.signal,
|
|
1873
|
-
}, eventTranslator.emitEvent);
|
|
1874
|
-
applyLifecycleHookResults('subagent_stop', stopResults);
|
|
1875
|
-
yield* eventTranslator.drainPending();
|
|
1876
|
-
}
|
|
1877
1261
|
}
|
|
1878
|
-
//
|
|
1879
|
-
//
|
|
1880
|
-
|
|
1881
|
-
//
|
|
1882
|
-
//
|
|
1883
|
-
//
|
|
1884
|
-
//
|
|
1885
|
-
//
|
|
1886
|
-
|
|
1887
|
-
|
|
1888
|
-
|
|
1889
|
-
|
|
1890
|
-
|
|
1891
|
-
|
|
1892
|
-
|
|
1893
|
-
|
|
1894
|
-
|
|
1895
|
-
|
|
1896
|
-
|
|
1897
|
-
|
|
1898
|
-
|
|
1899
|
-
|
|
1900
|
-
|
|
1901
|
-
|
|
1902
|
-
|
|
1903
|
-
|
|
1904
|
-
|
|
1905
|
-
|
|
1906
|
-
|
|
1907
|
-
|
|
1908
|
-
|
|
1909
|
-
|
|
1910
|
-
|
|
1911
|
-
|
|
1912
|
-
await ctx.runMgr.recordAudit({
|
|
1913
|
-
what: { action: 'guardrail:output', resource: outputVerdict.name },
|
|
1914
|
-
outcome: 'refused',
|
|
1915
|
-
reason: outputVerdict.reason ?? 'blocked by an output guardrail',
|
|
1916
|
-
...(params.persona?.identity.role ? { persona: params.persona.identity.role } : {}),
|
|
1917
|
-
});
|
|
1918
|
-
ctx.runMgr.setStopReason('output_guardrail');
|
|
1919
|
-
ctx.runMgr.setLastError(outputVerdict.reason ?? 'blocked by an output guardrail');
|
|
1920
|
-
ctx.runMgr.setResult('');
|
|
1921
|
-
}
|
|
1922
|
-
else if (outputVerdict.rewritten !== undefined) {
|
|
1923
|
-
await eventTranslator.emitEvent({
|
|
1924
|
-
type: 'guardrail_triggered',
|
|
1925
|
-
runId: ctx.runId,
|
|
1926
|
-
stage: 'output',
|
|
1927
|
-
action: 'rewrite',
|
|
1928
|
-
...(outputVerdict.name ? { guardrail: outputVerdict.name } : {}),
|
|
1929
|
-
...(outputVerdict.reason ? { reason: outputVerdict.reason } : {}),
|
|
1262
|
+
// The decision has now actually been carried out, so the park it
|
|
1263
|
+
// answered is no longer outstanding. Without this the checkpoint
|
|
1264
|
+
// keeps reporting `pending` with no `resolvedAt`, and an approval
|
|
1265
|
+
// queue re-serves a call that already ran — or a question already
|
|
1266
|
+
// answered — which defeats the entire point of recording the park.
|
|
1267
|
+
//
|
|
1268
|
+
// Two arms reach this point, and being outside `if (pendingResume)`
|
|
1269
|
+
// is what the second one needs. One is a plan whose decision was
|
|
1270
|
+
// applied to a batch above. The other is the cadence arm, for which
|
|
1271
|
+
// `planPendingResume` rightly produces no plan because the loop
|
|
1272
|
+
// resuming IS its decision being carried out (`answeredParkId`, set
|
|
1273
|
+
// on the restore path). Resolving only the first left a finished run
|
|
1274
|
+
// reporting `awaiting-decision` forever.
|
|
1275
|
+
//
|
|
1276
|
+
// What is RECORDED depends on which of the two produced the plan. A
|
|
1277
|
+
// recovery plan means the batch was answered by the crash path
|
|
1278
|
+
// rather than by the decision — the calls the human was asked about
|
|
1279
|
+
// were closed with explicitly unknown outcomes — so the human's
|
|
1280
|
+
// answer must not be written down as what ended the park. The park is
|
|
1281
|
+
// still resolved: the question is moot, and leaving it outstanding
|
|
1282
|
+
// would have `findPendingCheckpoint` serve it as the newest
|
|
1283
|
+
// outstanding park, so a host resuming it would rewind this run to
|
|
1284
|
+
// the checkpoint the crash happened on and re-execute a batch the run
|
|
1285
|
+
// has long since moved past.
|
|
1286
|
+
const resolvedCheckpointId = pendingResume?.checkpointId ?? answeredParkId;
|
|
1287
|
+
const recordedDecision = pendingResume?.source === 'recovery' && params.pendingDecision
|
|
1288
|
+
? supersededByRecovery(params.pendingDecision)
|
|
1289
|
+
: params.pendingDecision;
|
|
1290
|
+
if (recordedDecision && resolvedCheckpointId) {
|
|
1291
|
+
await checkpointMgr.unpark(resolvedCheckpointId, recordedDecision).catch((err) => {
|
|
1292
|
+
ctx.log.error('Applied a pending decision but failed to clear the park', {
|
|
1293
|
+
[NAMZU.RUN_ID]: ctx.runId,
|
|
1294
|
+
'namzu.checkpoint.id': resolvedCheckpointId,
|
|
1295
|
+
'exception.message': err instanceof Error ? err.message : String(err),
|
|
1930
1296
|
});
|
|
1931
|
-
|
|
1932
|
-
ctx.runMgr.setResult(outputVerdict.rewritten);
|
|
1933
|
-
}
|
|
1934
|
-
}
|
|
1935
|
-
if (params.consolidateInto && workingStateManager) {
|
|
1936
|
-
const entry = consolidationEntry(workingStateManager.getState(), {
|
|
1937
|
-
runId: ctx.runId,
|
|
1938
|
-
at: Date.now(),
|
|
1297
|
+
return null;
|
|
1939
1298
|
});
|
|
1940
|
-
if (entry) {
|
|
1941
|
-
try {
|
|
1942
|
-
const { entry: saved } = await params.consolidateInto.create(entry);
|
|
1943
|
-
await eventTranslator.emitEvent({
|
|
1944
|
-
type: 'memory_consolidated',
|
|
1945
|
-
runId: ctx.runId,
|
|
1946
|
-
memoryId: saved.id,
|
|
1947
|
-
title: entry.title,
|
|
1948
|
-
decisions: workingStateManager.getState().decisions.length,
|
|
1949
|
-
discoveries: workingStateManager.getState().discoveries.length,
|
|
1950
|
-
failures: workingStateManager.getState().failures.length,
|
|
1951
|
-
});
|
|
1952
|
-
yield* eventTranslator.drainPending();
|
|
1953
|
-
}
|
|
1954
|
-
catch (error) {
|
|
1955
|
-
ctx.log.warn('consolidation into the memory store failed', {
|
|
1956
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
1957
|
-
'namzu.memory.error': toErrorMessage(error),
|
|
1958
|
-
});
|
|
1959
|
-
}
|
|
1960
|
-
}
|
|
1961
1299
|
}
|
|
1962
|
-
|
|
1963
|
-
|
|
1964
|
-
|
|
1300
|
+
yield* iterationOrchestrator.runLoop();
|
|
1301
|
+
yield* finalizeRun({
|
|
1302
|
+
ctx,
|
|
1303
|
+
params,
|
|
1304
|
+
eventTranslator,
|
|
1305
|
+
takeSteps: () => iterationOrchestrator.getSteps(),
|
|
1306
|
+
workingStateManager,
|
|
1307
|
+
resultAssembler,
|
|
1308
|
+
rootSpan,
|
|
1309
|
+
});
|
|
1965
1310
|
}
|
|
1966
1311
|
catch (err) {
|
|
1967
1312
|
// A failed run still spent its steps; report them.
|
|
@@ -1971,104 +1316,128 @@ export async function* query(params) {
|
|
|
1971
1316
|
yield* resultAssembler.handleError(err, rootSpan);
|
|
1972
1317
|
}
|
|
1973
1318
|
finally {
|
|
1974
|
-
|
|
1975
|
-
|
|
1976
|
-
|
|
1977
|
-
|
|
1978
|
-
|
|
1979
|
-
|
|
1980
|
-
|
|
1981
|
-
|
|
1982
|
-
|
|
1983
|
-
|
|
1984
|
-
|
|
1985
|
-
|
|
1986
|
-
|
|
1987
|
-
|
|
1988
|
-
|
|
1989
|
-
|
|
1990
|
-
|
|
1991
|
-
// the host's to stop, when the session ends.
|
|
1992
|
-
if (params.backgroundJobs && (params.backgroundJobOwner ?? ctx.runId) === ctx.runId) {
|
|
1993
|
-
try {
|
|
1994
|
-
const stopped = await params.backgroundJobs.killOwner(ctx.runId);
|
|
1995
|
-
if (stopped.length > 0) {
|
|
1996
|
-
ctx.log.info('Background jobs stopped with the run', {
|
|
1997
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
1998
|
-
'namzu.jobs.stopped': stopped.length,
|
|
1999
|
-
});
|
|
2000
|
-
}
|
|
2001
|
-
}
|
|
2002
|
-
catch (jobErr) {
|
|
2003
|
-
ctx.log.error('A background job did not stop cleanly', {
|
|
2004
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
2005
|
-
...errorAttributes(jobErr),
|
|
2006
|
-
});
|
|
2007
|
-
}
|
|
2008
|
-
}
|
|
2009
|
-
// Same reasoning for the question channel: the tools outlive the
|
|
2010
|
-
// run that bound them, so leaving it attached would have a later
|
|
2011
|
-
// run's question written into this run's checkpoint store.
|
|
2012
|
-
questionParks.unbind();
|
|
2013
|
-
// Offer what the run learned to whoever decides what is worth
|
|
2014
|
-
// keeping. In `finally` and awaited: a run that failed still
|
|
2015
|
-
// discovered things, and a fire-and-forget write would race the
|
|
2016
|
-
// process exiting on a one-shot CLI run. A throw here is
|
|
2017
|
-
// swallowed — a memory that failed to form must not retract an
|
|
2018
|
-
// answer that was already produced.
|
|
2019
|
-
const candidate = memoryCandidateFor(ctx.runId, workingStateManager);
|
|
2020
|
-
if (params.promoteMemory && candidate) {
|
|
2021
|
-
try {
|
|
2022
|
-
await params.promoteMemory(candidate);
|
|
2023
|
-
}
|
|
2024
|
-
catch (promoteErr) {
|
|
2025
|
-
ctx.log.error('Memory promotion threw — the run is unaffected', {
|
|
2026
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
2027
|
-
'exception.message': promoteErr instanceof Error ? promoteErr.message : String(promoteErr),
|
|
2028
|
-
});
|
|
2029
|
-
}
|
|
2030
|
-
}
|
|
2031
|
-
// --- Sandbox lifecycle: destroy after run ---
|
|
2032
|
-
if (sandbox) {
|
|
2033
|
-
const sandboxId = sandbox.id;
|
|
2034
|
-
const teardown = await teardownSandbox(sandbox, sandboxTeardownTimeoutMs);
|
|
2035
|
-
if (teardown.kind === 'destroyed') {
|
|
2036
|
-
await eventTranslator.emitEvent({
|
|
2037
|
-
type: 'sandbox_destroyed',
|
|
2038
|
-
runId: ctx.runId,
|
|
2039
|
-
sandboxId,
|
|
2040
|
-
});
|
|
2041
|
-
yield* eventTranslator.drainPending();
|
|
2042
|
-
ctx.log.info('Sandbox destroyed', { 'namzu.sandbox.id': sandboxId });
|
|
2043
|
-
}
|
|
2044
|
-
else {
|
|
2045
|
-
ctx.log.error('Sandbox destroy failed', {
|
|
2046
|
-
'namzu.sandbox.id': sandboxId,
|
|
2047
|
-
...errorAttributes(teardown.error),
|
|
2048
|
-
});
|
|
2049
|
-
}
|
|
2050
|
-
}
|
|
2051
|
-
unsubscribeTaskStore?.();
|
|
2052
|
-
// Keyed by HOW it settled, not just that it did: a run that was
|
|
2053
|
-
// cancelled and a run that hit its budget have very different
|
|
2054
|
-
// duration distributions, and averaging them together describes
|
|
2055
|
-
// neither.
|
|
2056
|
-
recordRunDuration(ctx.runMgr.getRun().status ?? 'unknown', Date.now() - runStartedAt);
|
|
2057
|
-
rootSpan.end();
|
|
1319
|
+
yield* releaseRunResources({
|
|
1320
|
+
ctx,
|
|
1321
|
+
eventTranslator,
|
|
1322
|
+
emergencyManager,
|
|
1323
|
+
unsubscribeJobExits,
|
|
1324
|
+
unsubscribeTaskStore,
|
|
1325
|
+
awaitedJobs,
|
|
1326
|
+
backgroundJobs: params.backgroundJobs,
|
|
1327
|
+
backgroundJobOwner: params.backgroundJobOwner,
|
|
1328
|
+
questionParks,
|
|
1329
|
+
workingStateManager,
|
|
1330
|
+
promoteMemory: params.promoteMemory,
|
|
1331
|
+
sandbox,
|
|
1332
|
+
sandboxTeardownTimeoutMs,
|
|
1333
|
+
runStartedAt,
|
|
1334
|
+
rootSpan,
|
|
1335
|
+
});
|
|
2058
1336
|
}
|
|
1337
|
+
// Reached only by a run that settled on its own terms. `finalize()` is
|
|
1338
|
+
// the only thing in this body that writes the durable half of the run,
|
|
1339
|
+
// and a `return` completion arriving from a consumer (`break` out of
|
|
1340
|
+
// `for await`, `gen.return()`) runs the `finally` above and stops short
|
|
1341
|
+
// of here. The flag is what tells the two apart, and this is one of
|
|
1342
|
+
// three sites that set it — the sandbox-acquisition and input-guardrail
|
|
1343
|
+
// returns settle early and set it there. Set before the await rather
|
|
1344
|
+
// than after, because a store that throws on the way out must not send
|
|
1345
|
+
// the abandonment path over the same broken ground.
|
|
1346
|
+
settled = true;
|
|
2059
1347
|
return await resultAssembler.finalize();
|
|
2060
1348
|
})();
|
|
1349
|
+
try {
|
|
1350
|
+
return yield* runBody;
|
|
1351
|
+
}
|
|
1352
|
+
finally {
|
|
1353
|
+
if (!settled)
|
|
1354
|
+
await settleAbandonedRun(ctx.runMgr, ctx.log);
|
|
1355
|
+
}
|
|
2061
1356
|
}
|
|
2062
1357
|
/**
|
|
2063
|
-
*
|
|
1358
|
+
* Write a terminal durable record for a run whose consumer walked away.
|
|
2064
1359
|
*
|
|
2065
|
-
*
|
|
2066
|
-
*
|
|
2067
|
-
*
|
|
2068
|
-
*
|
|
2069
|
-
*
|
|
2070
|
-
*
|
|
1360
|
+
* `for await (… ) break` and an explicit `gen.return()` both end the run
|
|
1361
|
+
* body early. Everything the run's `finally` owns still happens — background
|
|
1362
|
+
* jobs are killed, the sandbox is destroyed, the span ends, the duration is
|
|
1363
|
+
* recorded — and then the generator stops. `finalize()` never runs, so
|
|
1364
|
+
* `persist()` never runs, and the store keeps whatever `init()` wrote: a
|
|
1365
|
+
* non-terminal status for a run that no longer exists. `deriveRunStatus`
|
|
1366
|
+
* reads that record back as `queued`, work waiting to start, and a host
|
|
1367
|
+
* rebuilding its view from the store believes it.
|
|
1368
|
+
*
|
|
1369
|
+
* There is nothing to emit here and nothing to emit it to: the consumer
|
|
1370
|
+
* that would have received the events is the one that left. This is about
|
|
1371
|
+
* the durable record only.
|
|
1372
|
+
*
|
|
1373
|
+
* `cancelled` is the verdict, and it is chosen from the existing vocabulary
|
|
1374
|
+
* because it is the one that is true. The run did not complete — no result
|
|
1375
|
+
* was produced and no terminal event was ever delivered — and nothing
|
|
1376
|
+
* failed, so `failed` would name an error that never happened; a run whose
|
|
1377
|
+
* consumer stopped reading and whose processes were torn down under it is
|
|
1378
|
+
* the same fact `markCancelled` already records when a run abort tears one
|
|
1379
|
+
* down. It needs no new `RunExecutionStatus` and no new `StopReason`.
|
|
1380
|
+
*
|
|
1381
|
+
* A verdict the run already reached is left standing. A run that failed,
|
|
1382
|
+
* or was cancelled, before the consumer left still says so; what the
|
|
1383
|
+
* abandonment adds is that the record reaches the disk at all.
|
|
1384
|
+
*
|
|
1385
|
+
* Neither is a verdict written over a PARK. A park is a promise to a human
|
|
1386
|
+
* that outlives the consumer: the run is resumable and somebody is still owed
|
|
1387
|
+
* an answer, and `deriveRunStatus` reads a terminal status BEFORE it reads the
|
|
1388
|
+
* park — so recording `cancelled` turns `awaiting_hitl` into `cancelled` for a
|
|
1389
|
+
* run nobody answered for, while the unanswered question stays on the record
|
|
1390
|
+
* and the checkpoint it belongs to stays the place a resume starts from. The
|
|
1391
|
+
* durable state is asked rather than the in-memory one because the in-memory
|
|
1392
|
+
* one is the misleading half here: `handleHITLDecision` emits `run_paused` and
|
|
1393
|
+
* drains it BEFORE it calls `setStopReason('paused')`, so a consumer that
|
|
1394
|
+
* leaves on that event leaves a run whose status is `running` and whose stop
|
|
1395
|
+
* reason is unset at the exact instant its park is already durable.
|
|
1396
|
+
* `findPendingCheckpoint` is the same read an approval queue is built from,
|
|
1397
|
+
* expired parks included in its judgement: a park nobody answered in time is
|
|
1398
|
+
* not somebody still being asked.
|
|
1399
|
+
*
|
|
1400
|
+
* Never throws. It runs while an exception may already be unwinding, and a
|
|
1401
|
+
* store that cannot be written must not replace the run's real failure with
|
|
1402
|
+
* its own.
|
|
2071
1403
|
*/
|
|
1404
|
+
async function settleAbandonedRun(runMgr, log) {
|
|
1405
|
+
try {
|
|
1406
|
+
// A terminal verdict is written whatever the park says: `deriveRunStatus`
|
|
1407
|
+
// settles a run that finished, failed or was cancelled BEFORE it looks at
|
|
1408
|
+
// a park ("terminal beats parked"), so a settled run is not waiting for
|
|
1409
|
+
// anybody and the row it already wrote must reach the disk. This ordering
|
|
1410
|
+
// is also what keeps a stale park from suppressing the write.
|
|
1411
|
+
if (!isTerminalStatus(runMgr.status)) {
|
|
1412
|
+
const parked = await findPendingCheckpoint(runMgr.getCheckpointStore(), runMgr.getRunScope());
|
|
1413
|
+
if (parked) {
|
|
1414
|
+
// Left exactly as it stands: no verdict, no write. The park row is
|
|
1415
|
+
// this run's durable state, and `persist()` here would add a
|
|
1416
|
+
// second claim — `running`, for a process that is gone — beside it.
|
|
1417
|
+
log.info('Abandoned run left parked for a human to answer', {
|
|
1418
|
+
[NAMZU.RUN_ID]: runMgr.id,
|
|
1419
|
+
'namzu.checkpoint.id': parked.id,
|
|
1420
|
+
'namzu.runtime.park_type': parked.pending?.request.type,
|
|
1421
|
+
});
|
|
1422
|
+
return;
|
|
1423
|
+
}
|
|
1424
|
+
runMgr.markCancelled();
|
|
1425
|
+
}
|
|
1426
|
+
// Once: the `finally` that calls this runs once, and every site in the
|
|
1427
|
+
// run body that settles through `finalize()` sets `settled` before it
|
|
1428
|
+
// returns, so the two can never both write.
|
|
1429
|
+
await runMgr.persist();
|
|
1430
|
+
log.info('Abandoned run recorded as cancelled', {
|
|
1431
|
+
[NAMZU.RUN_ID]: runMgr.id,
|
|
1432
|
+
});
|
|
1433
|
+
}
|
|
1434
|
+
catch (err) {
|
|
1435
|
+
log.error('Failed to record the terminal state of an abandoned run', {
|
|
1436
|
+
[NAMZU.RUN_ID]: runMgr.id,
|
|
1437
|
+
'exception.message': err instanceof Error ? err.message : String(err),
|
|
1438
|
+
});
|
|
1439
|
+
}
|
|
1440
|
+
}
|
|
2072
1441
|
/** The text of the newest user turn, which is what a prompt hook is asked about. */
|
|
2073
1442
|
function lastUserPrompt(messages) {
|
|
2074
1443
|
for (let i = messages.length - 1; i >= 0; i--) {
|
|
@@ -2078,30 +1447,6 @@ function lastUserPrompt(messages) {
|
|
|
2078
1447
|
}
|
|
2079
1448
|
return '';
|
|
2080
1449
|
}
|
|
2081
|
-
async function* catchUpFromCursor(runMgr, cursor, onEventReplay, generation, onReplayObserverError) {
|
|
2082
|
-
const missed = await runMgr.getRunStore().readEvents({ sinceSeq: cursor.sinceSeq });
|
|
2083
|
-
const replay = resolveRunEventReplay(cursor, {
|
|
2084
|
-
lastSeq: runMgr.lastEventSeq,
|
|
2085
|
-
...(generation !== undefined ? { generation } : {}),
|
|
2086
|
-
}, missed);
|
|
2087
|
-
if (onEventReplay) {
|
|
2088
|
-
try {
|
|
2089
|
-
// A callback typed `void` may still be implemented with `async` in
|
|
2090
|
-
// TypeScript. Observe that runtime Promise so a late rejection cannot
|
|
2091
|
-
// become process-wide, but never await host code here: replay delivery
|
|
2092
|
-
// and an already-cancelled run must not inherit observer liveness.
|
|
2093
|
-
const settlement = onEventReplay(replay);
|
|
2094
|
-
void Promise.resolve(settlement).catch(onReplayObserverError);
|
|
2095
|
-
}
|
|
2096
|
-
catch (error) {
|
|
2097
|
-
onReplayObserverError(error);
|
|
2098
|
-
}
|
|
2099
|
-
}
|
|
2100
|
-
if (replay.status !== 'replayed')
|
|
2101
|
-
return;
|
|
2102
|
-
for (const event of replay.events)
|
|
2103
|
-
yield event;
|
|
2104
|
-
}
|
|
2105
1450
|
async function drainPreparedQuery(fullParams, listener) {
|
|
2106
1451
|
const gen = query(fullParams);
|
|
2107
1452
|
let result = await gen.next();
|