@namzu/sdk 42.0.0 → 42.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +222 -0
- package/dist/manager/resident/outbox.d.ts +8 -8
- package/dist/manager/resident/store.d.ts +4 -4
- package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
- package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
- package/dist/runtime/query/cancelled-before-start.js +152 -0
- package/dist/runtime/query/cancelled-before-start.js.map +1 -0
- package/dist/runtime/query/checkpoint.d.ts +21 -0
- package/dist/runtime/query/checkpoint.d.ts.map +1 -1
- package/dist/runtime/query/checkpoint.js +23 -0
- package/dist/runtime/query/checkpoint.js.map +1 -1
- package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
- package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
- package/dist/runtime/query/executor/tool-call-admission.js +373 -0
- package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
- package/dist/runtime/query/executor.d.ts +70 -35
- package/dist/runtime/query/executor.d.ts.map +1 -1
- package/dist/runtime/query/executor.js +46 -380
- package/dist/runtime/query/executor.js.map +1 -1
- package/dist/runtime/query/finalize-run.d.ts +55 -0
- package/dist/runtime/query/finalize-run.d.ts.map +1 -0
- package/dist/runtime/query/finalize-run.js +113 -0
- package/dist/runtime/query/finalize-run.js.map +1 -0
- package/dist/runtime/query/index.d.ts +4 -9
- package/dist/runtime/query/index.d.ts.map +1 -1
- package/dist/runtime/query/index.js +238 -893
- package/dist/runtime/query/index.js.map +1 -1
- package/dist/runtime/query/iteration/index.d.ts +6 -161
- package/dist/runtime/query/iteration/index.d.ts.map +1 -1
- package/dist/runtime/query/iteration/index.js +23 -523
- package/dist/runtime/query/iteration/index.js.map +1 -1
- package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
- package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
- package/dist/runtime/query/iteration/outstanding-work.js +365 -0
- package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
- package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
- package/dist/runtime/query/iteration/phases/plan.js +13 -2
- package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
- package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
- package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
- package/dist/runtime/query/iteration/step-shaping.js +184 -0
- package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
- package/dist/runtime/query/prepare-run.d.ts +94 -0
- package/dist/runtime/query/prepare-run.d.ts.map +1 -0
- package/dist/runtime/query/prepare-run.js +589 -0
- package/dist/runtime/query/prepare-run.js.map +1 -0
- package/dist/runtime/query/release-run.d.ts +56 -0
- package/dist/runtime/query/release-run.d.ts.map +1 -0
- package/dist/runtime/query/release-run.js +101 -0
- package/dist/runtime/query/release-run.js.map +1 -0
- package/dist/runtime/query/resume-pending.d.ts +112 -1
- package/dist/runtime/query/resume-pending.d.ts.map +1 -1
- package/dist/runtime/query/resume-pending.js +133 -0
- package/dist/runtime/query/resume-pending.js.map +1 -1
- package/dist/store/evidence/compaction-archive.d.ts +2 -2
- package/dist/types/run/config.d.ts +12 -5
- package/dist/types/run/config.d.ts.map +1 -1
- package/package.json +4 -4
- package/src/runtime/query/cancelled-before-start.ts +189 -0
- package/src/runtime/query/checkpoint.ts +22 -0
- package/src/runtime/query/executor/tool-call-admission.ts +473 -0
- package/src/runtime/query/executor.ts +63 -442
- package/src/runtime/query/finalize-run.ts +192 -0
- package/src/runtime/query/index.ts +270 -1011
- package/src/runtime/query/iteration/index.ts +40 -586
- package/src/runtime/query/iteration/outstanding-work.ts +386 -0
- package/src/runtime/query/iteration/phases/plan.ts +18 -2
- package/src/runtime/query/iteration/step-shaping.ts +271 -0
- package/src/runtime/query/prepare-run.ts +718 -0
- package/src/runtime/query/release-run.ts +168 -0
- package/src/runtime/query/resume-pending.ts +158 -0
- package/src/types/run/config.ts +12 -5
|
@@ -6,14 +6,8 @@ import {
|
|
|
6
6
|
TriggerEvaluator,
|
|
7
7
|
assertBudgetEnforceable,
|
|
8
8
|
} from '../../advisory/index.js'
|
|
9
|
-
import { drainQueuedMessages } from '../../agents/handle.js'
|
|
10
9
|
import { AuthorizationGate } from '../../authorization/gate.js'
|
|
11
|
-
import {
|
|
12
|
-
import {
|
|
13
|
-
type ToolHistoryRepairReport,
|
|
14
|
-
repairToolMessageHistory,
|
|
15
|
-
toolHistoryRepairChanged,
|
|
16
|
-
} from '../../compaction/dangling.js'
|
|
10
|
+
import { repairToolMessageHistory, toolHistoryRepairChanged } from '../../compaction/dangling.js'
|
|
17
11
|
import { extractFromUserMessage } from '../../compaction/extractor.js'
|
|
18
12
|
import { WorkingStateManager } from '../../compaction/manager.js'
|
|
19
13
|
import type { ContextReducer } from '../../compaction/reducer.js'
|
|
@@ -23,21 +17,14 @@ import { type CompactionConfig, CompactionConfigSchema } from '../../config/runt
|
|
|
23
17
|
import { TOOL_OUTPUT_DIR_NAME } from '../../constants/tools/index.js'
|
|
24
18
|
import { EmergencySaveManager } from '../../manager/run/emergency.js'
|
|
25
19
|
import type { RunPersistence } from '../../manager/run/persistence.js'
|
|
26
|
-
import { resolveModelPricing } from '../../pricing/index.js'
|
|
27
20
|
import { PromptContributionRegistry } from '../../prompt/contributions.js'
|
|
28
21
|
import { resolveProviderCapabilities } from '../../provider/capabilities.js'
|
|
29
|
-
import {
|
|
30
|
-
import {
|
|
31
|
-
|
|
32
|
-
type ServingMember,
|
|
33
|
-
withProviderFallback,
|
|
34
|
-
} from '../../provider/fallback.js'
|
|
35
|
-
import { resolveStreamIdleTimeoutMs, withStreamIdleTimeout } from '../../provider/idle-timeout.js'
|
|
36
|
-
import { type ProviderRetryConfig, withProviderRetry } from '../../provider/retry.js'
|
|
22
|
+
import type { ProviderChainMember } from '../../provider/fallback.js'
|
|
23
|
+
import { withStreamIdleTimeout } from '../../provider/idle-timeout.js'
|
|
24
|
+
import type { ProviderRetryConfig } from '../../provider/retry.js'
|
|
37
25
|
import { withTokenBudget } from '../../provider/token-budget.js'
|
|
38
26
|
import type { TokenBudget } from '../../run/token-budget.js'
|
|
39
27
|
import type { PathBuilder } from '../../session/workspace/path-builder.js'
|
|
40
|
-
import { resolveAttachments } from '../../store/attachment/index.js'
|
|
41
28
|
import {
|
|
42
29
|
GENAI,
|
|
43
30
|
NAMZU,
|
|
@@ -45,8 +32,6 @@ import {
|
|
|
45
32
|
parentContext,
|
|
46
33
|
serializeSpan,
|
|
47
34
|
} from '../../telemetry/attributes.js'
|
|
48
|
-
import type { SerializedSpanContext } from '../../telemetry/attributes.js'
|
|
49
|
-
import { recordRunDuration } from '../../telemetry/metrics.js'
|
|
50
35
|
import { getTracer } from '../../telemetry/runtime-accessors.js'
|
|
51
36
|
import { buildAdvisoryTools } from '../../tools/advisory/index.js'
|
|
52
37
|
import { SearchToolsTool } from '../../tools/builtins/search-tools.js'
|
|
@@ -60,6 +45,7 @@ import type { AgentRuntimeContext, RuntimeToolOverrides } from '../../types/agen
|
|
|
60
45
|
import type { AgentContextLevel } from '../../types/agent/factory.js'
|
|
61
46
|
import type { WorkingMemoryProvider } from '../../types/agent/working-memory.js'
|
|
62
47
|
import type { AuthorizationGateConfig } from '../../types/authorization/index.js'
|
|
48
|
+
import { isTerminalStatus } from '../../types/common/index.js'
|
|
63
49
|
import { NamzuError } from '../../types/errors/index.js'
|
|
64
50
|
import type { InputGuardrailSpec, OutputGuardrailSpec } from '../../types/guardrail/index.js'
|
|
65
51
|
import {
|
|
@@ -67,23 +53,20 @@ import {
|
|
|
67
53
|
type ResumeHandler,
|
|
68
54
|
autoApproveHandler,
|
|
69
55
|
} from '../../types/hitl/index.js'
|
|
70
|
-
import type { CheckpointId,
|
|
56
|
+
import type { CheckpointId, RunId, SessionId, TenantId } from '../../types/ids/index.js'
|
|
71
57
|
import type { InvocationState } from '../../types/invocation/index.js'
|
|
72
58
|
import type { MemoryStore } from '../../types/memory/index.js'
|
|
73
59
|
import {
|
|
74
60
|
type AssistantMessage,
|
|
75
61
|
type Message,
|
|
76
|
-
type UserMessage,
|
|
77
62
|
createSystemMessage,
|
|
78
63
|
} from '../../types/message/index.js'
|
|
79
64
|
import type { AgentPersona } from '../../types/persona/index.js'
|
|
80
65
|
import type { LLMProvider } from '../../types/provider/index.js'
|
|
81
66
|
import type { TaskRouterConfig } from '../../types/router/index.js'
|
|
82
67
|
import type { ReviewAnswer } from '../../types/run/answer-review.js'
|
|
83
|
-
import { cancelCauseOf } from '../../types/run/cancel-cause.js'
|
|
84
68
|
import type { CheckpointStore, FencingToken } from '../../types/run/checkpoint-store.js'
|
|
85
69
|
import type { RunEventCursor, RunEventReplay } from '../../types/run/event-cursor.js'
|
|
86
|
-
import { resolveRunEventReplay } from '../../types/run/event-cursor.js'
|
|
87
70
|
import type {
|
|
88
71
|
AgentRunConfig,
|
|
89
72
|
BeforeStep,
|
|
@@ -95,8 +78,6 @@ import type {
|
|
|
95
78
|
StopCondition,
|
|
96
79
|
} from '../../types/run/index.js'
|
|
97
80
|
import type { PromoteMemory } from '../../types/run/memory-promotion.js'
|
|
98
|
-
import { memoryCandidateFor } from '../../types/run/memory-promotion.js'
|
|
99
|
-
import type { RunState } from '../../types/run/state.js'
|
|
100
81
|
import type { RunStore } from '../../types/run/store.js'
|
|
101
82
|
import type { TokenBudgetStore } from '../../types/run/token-budget-store.js'
|
|
102
83
|
import type { Sandbox, SandboxProvider } from '../../types/sandbox/index.js'
|
|
@@ -109,50 +90,46 @@ import type { RepairToolCall } from '../../types/tool/repair.js'
|
|
|
109
90
|
import type { BackoffPolicy } from '../../utils/backoff.js'
|
|
110
91
|
import type { ModelPricing } from '../../utils/cost.js'
|
|
111
92
|
import { toErrorMessage } from '../../utils/error.js'
|
|
112
|
-
import { generateCheckpointId, generateRunId } from '../../utils/id.js'
|
|
113
93
|
import { errorAttributes } from '../../utils/log/exception.js'
|
|
114
94
|
import type { Logger } from '../../utils/logger.js'
|
|
115
95
|
import { AwaitedJobs } from '../jobs/awaited-jobs.js'
|
|
116
96
|
import type { BackgroundJobRegistry } from '../jobs/registry.js'
|
|
117
|
-
import {
|
|
118
|
-
import { CheckpointManager } from './checkpoint.js'
|
|
119
|
-
import {
|
|
120
|
-
import { EventTranslator } from './events.js'
|
|
97
|
+
import { catchUpFromCursor, settlePreStartCancellation } from './cancelled-before-start.js'
|
|
98
|
+
import { CheckpointManager, findPendingCheckpoint } from './checkpoint.js'
|
|
99
|
+
import { finalizeRun } from './finalize-run.js'
|
|
121
100
|
import { GuardCoordinator } from './guard.js'
|
|
122
|
-
import { runInputGuardrails
|
|
101
|
+
import { runInputGuardrails } from './guardrails.js'
|
|
123
102
|
import { IterationOrchestrator } from './iteration/index.js'
|
|
124
103
|
import { isCompactionMessage } from './iteration/phases/compaction.js'
|
|
125
104
|
import { isWorkingMemoryMessage } from './iteration/phases/working-memory.js'
|
|
126
105
|
import { applyLifecycleHookResults } from './plugin-hooks.js'
|
|
127
106
|
import {
|
|
128
|
-
type
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
107
|
+
type SelectedResumeState,
|
|
108
|
+
prepareRun,
|
|
109
|
+
projectStateBearingHistory,
|
|
110
|
+
resolveProviderContextWindow,
|
|
111
|
+
selectedResumeStates,
|
|
112
|
+
} from './prepare-run.js'
|
|
113
|
+
import type { ProjectInstructionContext } from './project-instructions.js'
|
|
133
114
|
import type { PromptCache } from './prompt-cache.js'
|
|
134
115
|
import { PromptBuilder } from './prompt.js'
|
|
135
116
|
import type { PromptSegments } from './prompt.js'
|
|
136
117
|
import { PendingAnswers, QuestionParkBinding } from './question-park.js'
|
|
118
|
+
import { releaseRunResources } from './release-run.js'
|
|
137
119
|
import { RepeatCallTracker } from './repeat-call.js'
|
|
138
|
-
import { resolveMaxRequestRichContentBytes } from './request-rich-content.js'
|
|
139
120
|
import { ResultAssembler } from './result.js'
|
|
140
121
|
import {
|
|
141
122
|
type PendingResumePlan,
|
|
123
|
+
answersParkOf,
|
|
142
124
|
applyPendingResume,
|
|
143
125
|
interruptedToolCalls,
|
|
144
126
|
planCrashResume,
|
|
145
127
|
planPendingResume,
|
|
146
128
|
recoverCompletedCalls,
|
|
129
|
+
supersededByRecovery,
|
|
147
130
|
} from './resume-pending.js'
|
|
148
|
-
import {
|
|
149
|
-
acquireSandbox,
|
|
150
|
-
resolveSandboxTeardownTimeoutMs,
|
|
151
|
-
teardownSandbox,
|
|
152
|
-
} from './sandbox-lifecycle.js'
|
|
131
|
+
import { acquireSandbox } from './sandbox-lifecycle.js'
|
|
153
132
|
import { SteeringBinding, type SteeringChannel, isOperatorUserMessage } from './steering.js'
|
|
154
|
-
import { resolveQueryBudget } from './token-budget.js'
|
|
155
|
-
import { assertMaxToolCalls } from './tool-call-budget.js'
|
|
156
133
|
import { ToolGrantSet } from './tool-grants.js'
|
|
157
134
|
import { createToolPause } from './tool-pause.js'
|
|
158
135
|
import { ToolingBootstrap } from './tooling.js'
|
|
@@ -832,196 +809,6 @@ export interface QueryParams {
|
|
|
832
809
|
strictCapabilities?: boolean
|
|
833
810
|
}
|
|
834
811
|
|
|
835
|
-
type SelectedResumeState = RunState & {
|
|
836
|
-
readonly checkpointId: CheckpointId
|
|
837
|
-
readonly traceContext?: SerializedSpanContext
|
|
838
|
-
}
|
|
839
|
-
const selectedResumeStates = new WeakMap<QueryParams, SelectedResumeState>()
|
|
840
|
-
|
|
841
|
-
/**
|
|
842
|
-
* Refuse to price a run whose tokens two differently-priced members may produce.
|
|
843
|
-
*
|
|
844
|
-
* `RunPersistence` holds ONE {@link ModelPricing} table and applies it to every
|
|
845
|
-
* accumulation regardless of which model produced the tokens. Across a swap that
|
|
846
|
-
* makes `costInfo.totalCost` wrong by an unbounded margin, and silently — the
|
|
847
|
-
* number keeps the shape of an answer. `CostInfo` cannot express the truth
|
|
848
|
-
* either: it carries `inputCostPer1M` / `outputCostPer1M`, and there is no
|
|
849
|
-
* honest value for those once a total spans two rate cards.
|
|
850
|
-
*
|
|
851
|
-
* So the total is refused rather than blended. Naming what that costs is part
|
|
852
|
-
* of the refusal, because the caller loses `costLimitUsd` with it: the guard
|
|
853
|
-
* enforces that limit from this same accumulated total, and a limit enforced
|
|
854
|
-
* with the wrong rate card stops a run early or late by the same unbounded
|
|
855
|
-
* margin. A budget that is quietly wrong is worse than a budget that is
|
|
856
|
-
* declined.
|
|
857
|
-
*
|
|
858
|
-
* Reachable, not decorative: a host that passes `pricing` and declares a chain
|
|
859
|
-
* hits it on the first call. It costs `@namzu/cli` nothing, which passes no
|
|
860
|
-
* pricing at all — its `/cost` already reports that the provider gave no price.
|
|
861
|
-
*
|
|
862
|
-
* The way out is per-member pricing, which needs a `CostInfo` that can sum over
|
|
863
|
-
* heterogeneous rates. That is a public-type change and it is not this one.
|
|
864
|
-
*/
|
|
865
|
-
function assertCostIsAttributable(
|
|
866
|
-
chain: readonly ProviderChainMember[],
|
|
867
|
-
pricing: ModelPricing | undefined,
|
|
868
|
-
): void {
|
|
869
|
-
if (pricing === undefined || chain.length < 2) return
|
|
870
|
-
throw new NamzuError({
|
|
871
|
-
code: 'invalid_config',
|
|
872
|
-
message:
|
|
873
|
-
`A provider chain of ${chain.length} members was declared together with a single pricing table. ` +
|
|
874
|
-
'One table cannot price two members, so the run would report a total that is wrong by an unbounded ' +
|
|
875
|
-
'margin — and `runConfig.costLimitUsd` would be enforced against that same wrong total. ' +
|
|
876
|
-
'Either drop `pricing` (usage is still reported per model in the run) or declare one member.',
|
|
877
|
-
details: { chainLength: chain.length },
|
|
878
|
-
})
|
|
879
|
-
}
|
|
880
|
-
|
|
881
|
-
/**
|
|
882
|
-
* Refuse a budget that cannot be measured.
|
|
883
|
-
*
|
|
884
|
-
* `runConfig.costLimitUsd` is enforced against `costInfo.totalCost`, and that
|
|
885
|
-
* total only moves for tokens something has a rate for. A model no rate card
|
|
886
|
-
* covers therefore produced a limit that could never trip — a host that set a
|
|
887
|
-
* cost cap had no cost cap, and nothing said so. That was every run before the
|
|
888
|
-
* price catalogue existed, which is how it went unnoticed.
|
|
889
|
-
*
|
|
890
|
-
* Refusing at the front is the cheap half of the answer: it costs the caller
|
|
891
|
-
* nothing, fires before any spend, and names both ways out. The other half is
|
|
892
|
-
* the `cost_unmeasurable` stop, for the models this cannot see — a step naming
|
|
893
|
-
* its own, or a chain member declaring one.
|
|
894
|
-
*
|
|
895
|
-
* This is the same shape `advisory/budget.ts` already applies to
|
|
896
|
-
* `AdvisoryBudget.maxCostPerRun`, one layer down, and for the same reason. The
|
|
897
|
-
* run path simply never had it.
|
|
898
|
-
*/
|
|
899
|
-
function assertBudgetIsMeasurable(params: QueryParams): void {
|
|
900
|
-
const limit = params.runConfig.costLimitUsd
|
|
901
|
-
if (limit === undefined || limit <= 0) return
|
|
902
|
-
// A host-supplied table prices whatever it is pointed at, so a caller who
|
|
903
|
-
// brought one has answered the question themselves.
|
|
904
|
-
if (params.pricing !== undefined) return
|
|
905
|
-
const model = params.runConfig.model
|
|
906
|
-
if (resolveModelPricing(params.provider.id, model) !== undefined) return
|
|
907
|
-
|
|
908
|
-
throw new NamzuError({
|
|
909
|
-
code: 'invalid_config',
|
|
910
|
-
message:
|
|
911
|
-
`runConfig.costLimitUsd is set to ${limit}, but no rate is known for model "${model}" on ` +
|
|
912
|
-
`provider "${params.provider.id}". The limit is enforced against the run's accumulated ` +
|
|
913
|
-
'cost, and tokens with no rate never reach that total — so the budget would read as ' +
|
|
914
|
-
'satisfied for the whole run and stop nothing. Either pass `pricing` to declare the rate ' +
|
|
915
|
-
'yourself, add the model to packages/sdk/src/pricing/rates.source.json, or drop ' +
|
|
916
|
-
'`costLimitUsd` and bound the run with `tokenBudget`, which is measurable here.',
|
|
917
|
-
details: { model, providerId: params.provider.id, costLimitUsd: limit },
|
|
918
|
-
})
|
|
919
|
-
}
|
|
920
|
-
|
|
921
|
-
/**
|
|
922
|
-
* Ask the driver what this model's window is, and never let the answer
|
|
923
|
-
* cost the run.
|
|
924
|
-
*
|
|
925
|
-
* Three outcomes collapse to two here on purpose. No member and a resolved
|
|
926
|
-
* `undefined` both mean "no answer" — the distinction matters to a driver
|
|
927
|
-
* author, not to a caller about to fall through to the table. A rejection
|
|
928
|
-
* is the third, and it is logged rather than propagated: a run that would
|
|
929
|
-
* have worked on the table must not fail because a listing endpoint was
|
|
930
|
-
* down.
|
|
931
|
-
*/
|
|
932
|
-
async function resolveProviderContextWindow(
|
|
933
|
-
provider: LLMProvider,
|
|
934
|
-
model: string | undefined,
|
|
935
|
-
signal: AbortSignal | undefined,
|
|
936
|
-
timeoutMs: number,
|
|
937
|
-
log: Logger,
|
|
938
|
-
): Promise<number | undefined> {
|
|
939
|
-
if (!provider.resolveContextWindow || !model) return undefined
|
|
940
|
-
if (signal?.aborted) return undefined
|
|
941
|
-
|
|
942
|
-
// The resolver is an optional optimisation that runs before RunContext
|
|
943
|
-
// owns its child controller. Give it a private deadline signal and fuse
|
|
944
|
-
// caller cancellation into that transport in the safe direction: neither
|
|
945
|
-
// outcome aborts the caller's controller. Passing a signal is necessary
|
|
946
|
-
// but not sufficient, because a third-party driver can accept it and still
|
|
947
|
-
// leave its promise pending; the race below makes fallback independent of
|
|
948
|
-
// driver cooperation. Promise.race keeps the losing provider promise
|
|
949
|
-
// observed, so a later rejection cannot become unhandled.
|
|
950
|
-
const deadline = new AbortController()
|
|
951
|
-
const resolverSignal = signal ? AbortSignal.any([signal, deadline.signal]) : deadline.signal
|
|
952
|
-
const interrupted = Symbol('provider-context-window-interrupted')
|
|
953
|
-
let onAbort: (() => void) | undefined
|
|
954
|
-
const interruption = new Promise<typeof interrupted>((resolve) => {
|
|
955
|
-
onAbort = () => resolve(interrupted)
|
|
956
|
-
resolverSignal.addEventListener('abort', onAbort, { once: true })
|
|
957
|
-
})
|
|
958
|
-
// Direct QueryParams callers can supply a large run deadline. The clamp
|
|
959
|
-
// avoids Node's >2^31-1 one-millisecond timer coercion during metadata lookup.
|
|
960
|
-
// Metadata discovery remains optional and bounded even without a run deadline.
|
|
961
|
-
const deadlineMs = timeoutMs === 0 ? 5_000 : Math.min(Math.max(0, timeoutMs), 2_147_483_647)
|
|
962
|
-
const timer = setTimeout(() => {
|
|
963
|
-
deadline.abort(new Error(`Provider context-window lookup exceeded ${deadlineMs}ms`))
|
|
964
|
-
}, deadlineMs)
|
|
965
|
-
|
|
966
|
-
try {
|
|
967
|
-
const resolution = provider.resolveContextWindow(model, resolverSignal)
|
|
968
|
-
const reported = await Promise.race([resolution, interruption])
|
|
969
|
-
if (reported === interrupted) {
|
|
970
|
-
if (deadline.signal.aborted) {
|
|
971
|
-
log.debug('Provider context-window lookup timed out; using the table', {
|
|
972
|
-
'namzu.model.id': model,
|
|
973
|
-
'namzu.runtime.timeout_ms': deadlineMs,
|
|
974
|
-
})
|
|
975
|
-
}
|
|
976
|
-
return undefined
|
|
977
|
-
}
|
|
978
|
-
return typeof reported === 'number' && reported > 0 ? reported : undefined
|
|
979
|
-
} catch (err) {
|
|
980
|
-
log.debug('Provider could not report a context window; using the table', {
|
|
981
|
-
'namzu.model.id': model,
|
|
982
|
-
'namzu.error.message': toErrorMessage(err),
|
|
983
|
-
})
|
|
984
|
-
return undefined
|
|
985
|
-
} finally {
|
|
986
|
-
clearTimeout(timer)
|
|
987
|
-
if (onAbort) resolverSignal.removeEventListener('abort', onAbort)
|
|
988
|
-
}
|
|
989
|
-
}
|
|
990
|
-
|
|
991
|
-
interface PendingHistoryRepairEvent {
|
|
992
|
-
readonly source: 'fresh-history' | 'abandoned-checkpoint'
|
|
993
|
-
readonly report: ToolHistoryRepairReport
|
|
994
|
-
}
|
|
995
|
-
|
|
996
|
-
/**
|
|
997
|
-
* Project historical system messages exactly as a new run will persist them.
|
|
998
|
-
*
|
|
999
|
-
* Arbitrary historical prompt floors are rebuilt for this run and therefore
|
|
1000
|
-
* never reach its provider-bound conversation. Repair must happen AFTER that
|
|
1001
|
-
* removal: treating a soon-to-be-dropped system message as a tool-result
|
|
1002
|
-
* boundary can replace an exact real result with an invented unknown outcome.
|
|
1003
|
-
* The two state-bearing system forms survive; fresh inherited compaction is
|
|
1004
|
-
* pinned until this run can prove it reconstructed equivalent state.
|
|
1005
|
-
*/
|
|
1006
|
-
function projectStateBearingHistory(
|
|
1007
|
-
messages: readonly Message[],
|
|
1008
|
-
options: { readonly pinCompaction: boolean },
|
|
1009
|
-
): Message[] {
|
|
1010
|
-
const projected: Message[] = []
|
|
1011
|
-
for (const message of messages) {
|
|
1012
|
-
if (message.role !== 'system') {
|
|
1013
|
-
projected.push(message)
|
|
1014
|
-
continue
|
|
1015
|
-
}
|
|
1016
|
-
if (isCompactionMessage(message.content)) {
|
|
1017
|
-
projected.push(options.pinCompaction ? { ...message, retain: true } : message)
|
|
1018
|
-
} else if (isWorkingMemoryMessage(message.content)) {
|
|
1019
|
-
projected.push(message)
|
|
1020
|
-
}
|
|
1021
|
-
}
|
|
1022
|
-
return collapseProjectInstructionSnapshots(projected)
|
|
1023
|
-
}
|
|
1024
|
-
|
|
1025
812
|
/**
|
|
1026
813
|
* Remove the incomplete turn still owned by a durable resume plan.
|
|
1027
814
|
*
|
|
@@ -1100,520 +887,32 @@ function withOwnedResumeOutcomes(
|
|
|
1100
887
|
}
|
|
1101
888
|
|
|
1102
889
|
export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run> {
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
// before opening a budget or persisting a run without its owning identity.
|
|
1106
|
-
const missingFields = (['sessionId', 'topicId', 'projectId', 'tenantId'] as const).filter(
|
|
1107
|
-
(field) => !params[field],
|
|
1108
|
-
)
|
|
1109
|
-
if (missingFields.length > 0) {
|
|
1110
|
-
throw new NamzuError({
|
|
1111
|
-
code: 'invalid_config',
|
|
1112
|
-
message: `query requires sessionId, topicId, projectId, and tenantId; missing: ${missingFields.join(', ')}.`,
|
|
1113
|
-
details: { missingFields },
|
|
1114
|
-
})
|
|
1115
|
-
}
|
|
1116
|
-
const selectedResumeState = selectedResumeStates.get(params)
|
|
1117
|
-
selectedResumeStates.delete(params)
|
|
1118
|
-
// Resolved at the DOOR, before a run id exists or a logger is built.
|
|
1119
|
-
// A caller who set both spellings of a renamed field has a config bug,
|
|
1120
|
-
// and refusing it here costs them nothing; refusing it at the read site
|
|
1121
|
-
// deep in the loop turns the same bug into a mid-run failure, after a
|
|
1122
|
-
// provider call has been paid for and a partial transcript written.
|
|
1123
|
-
const promptCache = params.promptCache
|
|
1124
|
-
const taskScheduler = params.taskScheduler
|
|
1125
|
-
const streamIdleTimeoutMs = resolveStreamIdleTimeoutMs(params.runConfig.streamIdleTimeoutMs)
|
|
1126
|
-
const maxRequestRichContentBytes = resolveMaxRequestRichContentBytes(
|
|
1127
|
-
params.runConfig.maxRequestRichContentBytes,
|
|
1128
|
-
)
|
|
1129
|
-
const sandboxTeardownTimeoutMs = resolveSandboxTeardownTimeoutMs(params.sandboxTeardownTimeoutMs)
|
|
1130
|
-
// Persist the EFFECTIVE value, not only an override. A run replayed after a
|
|
1131
|
-
// later release must be able to explain which liveness policy settled it;
|
|
1132
|
-
// an absent field whose meaning follows the currently-installed default
|
|
1133
|
-
// would rewrite that evidence at read time.
|
|
1134
|
-
const runConfig: AgentRunConfig = {
|
|
1135
|
-
...params.runConfig,
|
|
1136
|
-
streamIdleTimeoutMs,
|
|
1137
|
-
maxRequestRichContentBytes,
|
|
1138
|
-
}
|
|
1139
|
-
|
|
1140
|
-
// The run's one correlated logger, built before anything below needs
|
|
1141
|
-
// one — the migration check, the retry/fallback wrappers and `ctx`
|
|
1142
|
-
// itself all read this SAME object, so a retry warning and the run
|
|
1143
|
-
// record it retried for carry the identical `namzu.run.id` instead of
|
|
1144
|
-
// three separate `getRootLogger()` reads that happened to agree by
|
|
1145
|
-
// accident. `runId` is resolved here, once, rather than left to
|
|
1146
|
-
// `build`'s own `config.runId ?? generateRunId()` fallback —
|
|
1147
|
-
// generating it twice would silently hand the log and the run two
|
|
1148
|
-
// different ids.
|
|
1149
|
-
const runId = params.runId ?? generateRunId()
|
|
1150
|
-
const budget = await resolveQueryBudget(params, runId, selectedResumeState)
|
|
1151
|
-
const log = RunContextFactory.buildLogger({
|
|
1152
|
-
agentName: params.agentName,
|
|
890
|
+
const prepared = await prepareRun(params)
|
|
891
|
+
const {
|
|
1153
892
|
runConfig,
|
|
1154
|
-
runId,
|
|
1155
|
-
parentRunId: params.parentRunId,
|
|
1156
|
-
sessionId: params.sessionId,
|
|
1157
|
-
topicId: params.topicId,
|
|
1158
|
-
projectId: params.projectId,
|
|
1159
|
-
tenantId: params.tenantId,
|
|
1160
|
-
})
|
|
1161
|
-
|
|
1162
|
-
// Every model call in the run — the loop's turns, the forced-final
|
|
1163
|
-
// summary, advisory and compaction side calls — goes through this one
|
|
1164
|
-
// wrapped provider, so the retry policy cannot be bypassed by a code
|
|
1165
|
-
// path that happens to hold the raw driver.
|
|
1166
|
-
// The logger is passed on purpose: `withProviderRetry` guards every one
|
|
1167
|
-
// of its warns behind `options.log`, and this is its only production
|
|
1168
|
-
// call site — so without it the "failed, retrying" and "failed, giving
|
|
1169
|
-
// up" lines were dead code and a backoff left no trace anywhere.
|
|
1170
|
-
//
|
|
1171
|
-
// With a chain declared, the same sentence holds two levels out. The idle
|
|
1172
|
-
// watchdog is applied to each raw member, retry wraps that, and fallback
|
|
1173
|
-
// wraps the members: `fallback(retry(idle(m0)), retry(idle(m1)), …)`. The
|
|
1174
|
-
// idle layer cannot sit outside retry, because its timer would then count a
|
|
1175
|
-
// legitimate backoff as provider silence. This order is not a
|
|
1176
|
-
// preference. Assembled the other way round — which is what a host gets if
|
|
1177
|
-
// it wraps its own chain and hands the result in, because this function
|
|
1178
|
-
// would then wrap THAT in retry — an exhausted chain gets restarted from
|
|
1179
|
-
// the head by the outer loop and a throttle on the last member is counted
|
|
1180
|
-
// by two budgets. Building it here is what makes the order unspellable
|
|
1181
|
-
// wrong.
|
|
1182
|
-
const chain: readonly ProviderChainMember[] = [
|
|
1183
|
-
{ provider: params.provider },
|
|
1184
|
-
...(params.fallbackProviders ?? []),
|
|
1185
|
-
]
|
|
1186
|
-
assertCostIsAttributable(chain, params.pricing)
|
|
1187
|
-
assertBudgetIsMeasurable(params)
|
|
1188
|
-
const withRecovery = (provider: LLMProvider): LLMProvider => {
|
|
1189
|
-
const withIdleBound = withStreamIdleTimeout(provider, {
|
|
1190
|
-
idleTimeoutMs: streamIdleTimeoutMs,
|
|
1191
|
-
log,
|
|
1192
|
-
})
|
|
1193
|
-
const metered = withTokenBudget(withIdleBound, budget)
|
|
1194
|
-
return params.retry === false
|
|
1195
|
-
? metered
|
|
1196
|
-
: withProviderRetry(metered, {
|
|
1197
|
-
config: params.retry,
|
|
1198
|
-
log,
|
|
1199
|
-
canRetry: () => budget.remaining > 0,
|
|
1200
|
-
})
|
|
1201
|
-
}
|
|
1202
|
-
// Who is serving right now, for the run RECORD rather than for the request.
|
|
1203
|
-
//
|
|
1204
|
-
// It starts at the head and moves only when the chain does, which is the
|
|
1205
|
-
// whole of the truth because the cursor never rewinds. The run cannot read
|
|
1206
|
-
// this off `resilientProvider`: that wrapper reports the head's `id` on
|
|
1207
|
-
// purpose, so asking it produces the declaration back — the defect this
|
|
1208
|
-
// record exists to fix.
|
|
1209
|
-
const serving: { current: ServingMember } = {
|
|
1210
|
-
current: { index: 0, providerId: params.provider.id },
|
|
1211
|
-
}
|
|
1212
|
-
const resilientProvider = withProviderFallback(
|
|
1213
|
-
chain.map((member) => ({
|
|
1214
|
-
...member,
|
|
1215
|
-
provider: withRecovery(member.provider),
|
|
1216
|
-
})),
|
|
1217
|
-
{
|
|
1218
|
-
log,
|
|
1219
|
-
canFallback: () => budget.remaining > 0,
|
|
1220
|
-
onSwap: (to) => {
|
|
1221
|
-
serving.current = to
|
|
1222
|
-
// `ctx` is declared below and is initialized before anything can
|
|
1223
|
-
// call the provider: this fires from inside a `chatStream`, and
|
|
1224
|
-
// the first one is issued by the loop that `ctx` is built for.
|
|
1225
|
-
ctx.runMgr.setServingProvider(to.providerId)
|
|
1226
|
-
},
|
|
1227
|
-
},
|
|
1228
|
-
)
|
|
1229
|
-
|
|
1230
|
-
// Asked ONCE, here, before the loop exists. Both readers are synchronous
|
|
1231
|
-
// and hot, so this can never move inside the iteration — and a driver
|
|
1232
|
-
// that rejects, or one that hangs until the run is cancelled, must not
|
|
1233
|
-
// take down a run the table could have served perfectly well. That is
|
|
1234
|
-
// why the failure path is a swallow with a log rather than a throw: the
|
|
1235
|
-
// window is an optimisation over a working default, not a prerequisite.
|
|
1236
|
-
const providerContextWindow = await resolveProviderContextWindow(
|
|
1237
|
-
resilientProvider,
|
|
1238
|
-
runConfig.model,
|
|
1239
|
-
params.signal,
|
|
1240
|
-
runConfig.timeoutMs,
|
|
1241
|
-
log,
|
|
1242
|
-
)
|
|
1243
|
-
const modelContextWindows = new Map<string, number | undefined>()
|
|
1244
|
-
if (runConfig.model) modelContextWindows.set(runConfig.model, providerContextWindow)
|
|
1245
|
-
|
|
1246
|
-
// The mode this conversation was left in, when the run config names none.
|
|
1247
|
-
// Read once, before the loop exists, for the same reason the context
|
|
1248
|
-
// window is: the executor's resolver is synchronous and hot.
|
|
1249
|
-
//
|
|
1250
|
-
// A store that throws is not a run failure — the run falls back to the
|
|
1251
|
-
// config's answer, which is exactly what it did before this existed.
|
|
1252
|
-
const topicState = params.topicStateStore
|
|
1253
|
-
? await params.topicStateStore
|
|
1254
|
-
.getState(params.topicId, params.tenantId)
|
|
1255
|
-
.catch((err: unknown) => {
|
|
1256
|
-
log.debug('Could not read the topic state; using the run config', {
|
|
1257
|
-
'namzu.topic.id': params.topicId,
|
|
1258
|
-
'namzu.error.message': toErrorMessage(err),
|
|
1259
|
-
})
|
|
1260
|
-
return null
|
|
1261
|
-
})
|
|
1262
|
-
: null
|
|
1263
|
-
|
|
1264
|
-
// Whatever a host left for "the next run", taken and cleared in one
|
|
1265
|
-
// compare-and-set write. Prepended to the messages this run starts from,
|
|
1266
|
-
// so it is in the FIRST request rather than arriving a turn late.
|
|
1267
|
-
//
|
|
1268
|
-
// Cleared as it is read: a queue read and cleared separately re-delivers
|
|
1269
|
-
// on a crash between the two, and "start with this" arriving twice is a
|
|
1270
|
-
// different instruction from the one that was left.
|
|
1271
|
-
const queuedForThisRun: readonly Message[] = params.topicStateStore
|
|
1272
|
-
? await drainQueuedMessages(params.topicStateStore, params.topicId, params.tenantId).catch(
|
|
1273
|
-
(err: unknown) => {
|
|
1274
|
-
log.debug('Could not drain the topic queue; starting without it', {
|
|
1275
|
-
'namzu.topic.id': params.topicId,
|
|
1276
|
-
'namzu.error.message': toErrorMessage(err),
|
|
1277
|
-
})
|
|
1278
|
-
return []
|
|
1279
|
-
},
|
|
1280
|
-
)
|
|
1281
|
-
: []
|
|
1282
|
-
|
|
1283
|
-
// One effective list, used everywhere the run is seeded from. Three
|
|
1284
|
-
// branches below push from it, and computing it at each would be three
|
|
1285
|
-
// places to forget the queue.
|
|
1286
|
-
//
|
|
1287
|
-
// Stored attachments are resolved HERE, once, before the messages reach
|
|
1288
|
-
// the run record. Resolving later — at the provider boundary — would put
|
|
1289
|
-
// refs in the durable transcript and in every checkpoint, and a run
|
|
1290
|
-
// resumed against a store that had since forgotten a ref would fail
|
|
1291
|
-
// replaying its own history rather than at the moment somebody asked for
|
|
1292
|
-
// the bytes. Every failure refuses: a message that silently lost its
|
|
1293
|
-
// image is a model answering about a picture it never saw.
|
|
1294
|
-
const seeded: Message[] =
|
|
1295
|
-
queuedForThisRun.length > 0 ? [...queuedForThisRun, ...params.messages] : params.messages
|
|
1296
|
-
let resolvedInitialMessages: Message[]
|
|
1297
|
-
let attachmentResolutionCancelled = false
|
|
1298
|
-
try {
|
|
1299
|
-
resolvedInitialMessages = [
|
|
1300
|
-
...(await resolveAttachments(seeded, params.attachmentStore, {
|
|
1301
|
-
signal: params.signal,
|
|
1302
|
-
timeoutMs: params.attachmentResolveTimeoutMs,
|
|
1303
|
-
})),
|
|
1304
|
-
]
|
|
1305
|
-
params.signal?.throwIfAborted()
|
|
1306
|
-
} catch (error) {
|
|
1307
|
-
// Attachment materialization precedes RunContext construction so stored
|
|
1308
|
-
// bytes never enter a live run's checkpoints. Cancellation still belongs
|
|
1309
|
-
// to that run: preserve the exact input refs, build the context below, and
|
|
1310
|
-
// let its normal terminal path classify/persist a cancelled Run. Every
|
|
1311
|
-
// other store failure remains a pre-run refusal.
|
|
1312
|
-
if (!params.signal?.aborted || error !== params.signal.reason) throw error
|
|
1313
|
-
resolvedInitialMessages = [...seeded]
|
|
1314
|
-
attachmentResolutionCancelled = true
|
|
1315
|
-
}
|
|
1316
|
-
if (!attachmentResolutionCancelled && params.projectInstructionContext?.prepareInitialSnapshot) {
|
|
1317
|
-
const preparationSignal = params.signal ?? new AbortController().signal
|
|
1318
|
-
let snapshot: UserMessage | null | undefined
|
|
1319
|
-
try {
|
|
1320
|
-
const prepared = await awaitProjectInstructionCallback(preparationSignal, () =>
|
|
1321
|
-
params.projectInstructionContext?.prepareInitialSnapshot?.({
|
|
1322
|
-
messages: [...resolvedInitialMessages],
|
|
1323
|
-
signal: preparationSignal,
|
|
1324
|
-
}),
|
|
1325
|
-
)
|
|
1326
|
-
// The callback promise can settle, remove its listener, and queue this
|
|
1327
|
-
// continuation immediately before a queued abort. Publication is a
|
|
1328
|
-
// separate authority boundary, so fence it too.
|
|
1329
|
-
preparationSignal.throwIfAborted()
|
|
1330
|
-
snapshot = prepared
|
|
1331
|
-
} catch (error) {
|
|
1332
|
-
// This callback runs before RunContext owns its child controller. A
|
|
1333
|
-
// caller cancellation here still belongs to the run: publish no late
|
|
1334
|
-
// snapshot and let the context below settle the normal cancelled Run.
|
|
1335
|
-
// Compare the exact reason: a callback failure that won first must not
|
|
1336
|
-
// be erased merely because cancellation arrived before this catch ran.
|
|
1337
|
-
if (!preparationSignal.aborted || error !== preparationSignal.reason) throw error
|
|
1338
|
-
}
|
|
1339
|
-
if (snapshot !== undefined) {
|
|
1340
|
-
resolvedInitialMessages = replaceProjectInstructionSnapshot(
|
|
1341
|
-
resolvedInitialMessages,
|
|
1342
|
-
snapshot,
|
|
1343
|
-
'before-latest-user',
|
|
1344
|
-
)
|
|
1345
|
-
}
|
|
1346
|
-
}
|
|
1347
|
-
const pendingHistoryRepairs: PendingHistoryRepairEvent[] = []
|
|
1348
|
-
const projectedInitialMessages = collapseProjectInstructionSnapshots(
|
|
1349
|
-
params.resumeFromCheckpoint || params.continuationMode
|
|
1350
|
-
? resolvedInitialMessages
|
|
1351
|
-
: projectStateBearingHistory(resolvedInitialMessages, {
|
|
1352
|
-
pinCompaction: true,
|
|
1353
|
-
}),
|
|
1354
|
-
)
|
|
1355
|
-
const initialRepair = params.resumeFromCheckpoint
|
|
1356
|
-
? { messages: projectedInitialMessages, report: undefined }
|
|
1357
|
-
: repairToolMessageHistory(projectedInitialMessages)
|
|
1358
|
-
const initialMessages = initialRepair.messages
|
|
1359
|
-
if (initialRepair.report && toolHistoryRepairChanged(initialRepair.report)) {
|
|
1360
|
-
pendingHistoryRepairs.push({
|
|
1361
|
-
source: 'fresh-history',
|
|
1362
|
-
report: initialRepair.report,
|
|
1363
|
-
})
|
|
1364
|
-
log.warn('Repaired provider-invalid tool history before starting the run', {
|
|
1365
|
-
[NAMZU.RUN_ID]: runId,
|
|
1366
|
-
'namzu.history.source': 'fresh-history',
|
|
1367
|
-
'namzu.history.duplicate_tool_results_removed':
|
|
1368
|
-
initialRepair.report.duplicateToolResultsRemoved,
|
|
1369
|
-
'namzu.history.orphaned_tool_results_removed':
|
|
1370
|
-
initialRepair.report.orphanedToolResultsRemoved,
|
|
1371
|
-
'namzu.history.synthetic_tool_results_inserted':
|
|
1372
|
-
initialRepair.report.syntheticToolResultsInserted,
|
|
1373
|
-
})
|
|
1374
|
-
}
|
|
1375
|
-
|
|
1376
|
-
const ctx = RunContextFactory.build({
|
|
1377
893
|
budget,
|
|
1378
|
-
...(topicState ? { topicPermissionMode: topicState.permissionMode } : {}),
|
|
1379
|
-
...(params.permissionModeRef ? { permissionModeRef: params.permissionModeRef } : {}),
|
|
1380
|
-
agentId: params.agentId,
|
|
1381
|
-
agentName: params.agentName,
|
|
1382
|
-
runConfig,
|
|
1383
|
-
provider: resilientProvider,
|
|
1384
|
-
workingDirectory: params.workingDirectory,
|
|
1385
|
-
pricing: params.pricing,
|
|
1386
|
-
enableActivityTracking: params.enableActivityTracking,
|
|
1387
|
-
messages: initialMessages,
|
|
1388
|
-
signal: params.signal,
|
|
1389
|
-
sessionId: params.sessionId,
|
|
1390
|
-
topicId: params.topicId,
|
|
1391
|
-
projectId: params.projectId,
|
|
1392
|
-
tenantId: params.tenantId,
|
|
1393
|
-
pathBuilder: params.pathBuilder,
|
|
1394
|
-
checkpointStore: params.checkpointStore,
|
|
1395
|
-
runStore: params.runStore,
|
|
1396
|
-
runId,
|
|
1397
|
-
parentRunId: params.parentRunId,
|
|
1398
|
-
depth: params.depth,
|
|
1399
894
|
log,
|
|
1400
|
-
|
|
1401
|
-
|
|
1402
|
-
|
|
1403
|
-
|
|
1404
|
-
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
|
|
1416
|
-
|
|
1417
|
-
|
|
1418
|
-
// always yes and would name every run `host`, including the ones
|
|
1419
|
-
// approving everything unattended. Identity is what actually
|
|
1420
|
-
// separates the two.
|
|
1421
|
-
name:
|
|
1422
|
-
params.approvalPolicyName ??
|
|
1423
|
-
(params.resumeHandler === autoApproveHandler ? AUTO_APPROVE_POLICY_NAME : 'host'),
|
|
1424
|
-
handler: params.resumeHandler,
|
|
1425
|
-
},
|
|
1426
|
-
emit: (event) => eventTranslator.emitEvent(event),
|
|
1427
|
-
})
|
|
1428
|
-
|
|
1429
|
-
const planApprovalIds = new Map<PlanId, CheckpointId>()
|
|
1430
|
-
ctx.planManager.setApprovalHandler(async (request) => {
|
|
1431
|
-
let checkpointId = planApprovalIds.get(request.planId)
|
|
1432
|
-
if (!checkpointId) {
|
|
1433
|
-
checkpointId = generateCheckpointId()
|
|
1434
|
-
planApprovalIds.set(request.planId, checkpointId)
|
|
1435
|
-
}
|
|
1436
|
-
// `.current.handler`, never a captured `params.resumeHandler`. That
|
|
1437
|
-
// capture is what made changing the policy mean ending the run.
|
|
1438
|
-
const decision = await approvalPolicy.current.handler({
|
|
1439
|
-
type: 'plan_approval',
|
|
1440
|
-
runId: ctx.runId,
|
|
1441
|
-
checkpointId,
|
|
1442
|
-
plan: {
|
|
1443
|
-
planId: request.planId,
|
|
1444
|
-
title: request.title,
|
|
1445
|
-
steps: request.steps.map((s, i) => ({
|
|
1446
|
-
id: s.id,
|
|
1447
|
-
description: s.description,
|
|
1448
|
-
toolName: s.toolName,
|
|
1449
|
-
agentId: s.agentId,
|
|
1450
|
-
dependsOn: s.dependsOn,
|
|
1451
|
-
order: s.order ?? i + 1,
|
|
1452
|
-
})),
|
|
1453
|
-
summary: request.summary,
|
|
1454
|
-
},
|
|
1455
|
-
})
|
|
1456
|
-
|
|
1457
|
-
if (decision.action === 'approve_plan') {
|
|
1458
|
-
// Optional approve-with-edits channel: the host may attach
|
|
1459
|
-
// feedback to an approval. `PlanApprovalResponse.feedback`
|
|
1460
|
-
// already exists on the type; threading it through lets the
|
|
1461
|
-
// coordinator's approve_plan tool surface the user's edits in
|
|
1462
|
-
// the same tool_result that unblocks the park. Bare approvals
|
|
1463
|
-
// stay byte-identical (`{ approved: true }`).
|
|
1464
|
-
return decision.feedback
|
|
1465
|
-
? { approved: true, feedback: decision.feedback }
|
|
1466
|
-
: { approved: true }
|
|
1467
|
-
}
|
|
1468
|
-
if (decision.action === 'reject_plan') {
|
|
1469
|
-
return { approved: false, feedback: decision.feedback }
|
|
1470
|
-
}
|
|
1471
|
-
|
|
1472
|
-
return { approved: false, feedback: `Action: ${decision.action}` }
|
|
1473
|
-
})
|
|
1474
|
-
|
|
1475
|
-
const eventTranslator = new EventTranslator(ctx.runMgr, undefined, ctx.log)
|
|
1476
|
-
eventTranslator.wireActivityStore(ctx.activityStore, ctx.runId)
|
|
1477
|
-
eventTranslator.wirePlanManager(ctx.planManager, ctx.runId)
|
|
1478
|
-
eventTranslator.setGeneration(params.claimFence)
|
|
1479
|
-
let interruptHooksStarted = false
|
|
1480
|
-
const executeUserInterruptHooks = async (terminalError: unknown): Promise<void> => {
|
|
1481
|
-
if (
|
|
1482
|
-
interruptHooksStarted ||
|
|
1483
|
-
!params.pluginManager ||
|
|
1484
|
-
!isCallerAbortError(terminalError, ctx.abortController.signal) ||
|
|
1485
|
-
params.parentRunId !== undefined ||
|
|
1486
|
-
(params.depth ?? 0) !== 0 ||
|
|
1487
|
-
cancelCauseOf(ctx.abortController.signal.reason) !== 'user'
|
|
1488
|
-
) {
|
|
1489
|
-
return
|
|
1490
|
-
}
|
|
1491
|
-
|
|
1492
|
-
interruptHooksStarted = true
|
|
1493
|
-
try {
|
|
1494
|
-
// Deliberately omit the already-aborted run signal. The lifecycle
|
|
1495
|
-
// manager still supplies each handler its own deadline signal, while
|
|
1496
|
-
// `run_interrupt`'s observational fan-out prevents one result from
|
|
1497
|
-
// suppressing the cleanup hooks that follow it.
|
|
1498
|
-
await params.pluginManager.executeHooks(
|
|
1499
|
-
'run_interrupt',
|
|
1500
|
-
{ runId: ctx.runId, cancelCause: 'user' },
|
|
1501
|
-
eventTranslator.emitEvent,
|
|
1502
|
-
)
|
|
1503
|
-
} catch (error) {
|
|
1504
|
-
// Cancellation is the terminal authority. A hook event sink or an
|
|
1505
|
-
// unexpected manager failure is reported, but cannot turn Stop into a
|
|
1506
|
-
// failed run or prevent the durable cancellation verdict.
|
|
1507
|
-
ctx.log.error('Run interrupt hooks did not settle cleanly', {
|
|
1508
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
1509
|
-
...errorAttributes(error),
|
|
1510
|
-
})
|
|
1511
|
-
}
|
|
1512
|
-
}
|
|
895
|
+
ctx,
|
|
896
|
+
resilientProvider,
|
|
897
|
+
serving,
|
|
898
|
+
providerContextWindow,
|
|
899
|
+
modelContextWindows,
|
|
900
|
+
approvalPolicy,
|
|
901
|
+
eventTranslator,
|
|
902
|
+
executeUserInterruptHooks,
|
|
903
|
+
pendingHistoryRepairs,
|
|
904
|
+
initialMessages,
|
|
905
|
+
queuedForThisRun,
|
|
906
|
+
selectedResumeState,
|
|
907
|
+
attachmentResolutionCancelled,
|
|
908
|
+
streamIdleTimeoutMs,
|
|
909
|
+
sandboxTeardownTimeoutMs,
|
|
910
|
+
promptCache,
|
|
911
|
+
taskScheduler,
|
|
912
|
+
} = prepared
|
|
1513
913
|
|
|
1514
914
|
if (attachmentResolutionCancelled) {
|
|
1515
|
-
|
|
1516
|
-
// observes cancellation, do only the work required to leave an honest
|
|
1517
|
-
// durable run: initialize the record, retain the unresolved references,
|
|
1518
|
-
// and settle through the ordinary cancellation classifier. Prompt
|
|
1519
|
-
// contributions/cache, host callbacks, tools, plugins, sandbox, guardrails,
|
|
1520
|
-
// advisors, and providers are all authority-bearing work and stay out.
|
|
1521
|
-
// The dedicated root interrupt notification is the sole plugin exception:
|
|
1522
|
-
// it runs after cancellation under its own deadline and cannot regain model
|
|
1523
|
-
// or tool authority.
|
|
1524
|
-
if (params.resumeFromCheckpoint && !selectedResumeState) {
|
|
1525
|
-
// The canonical resume surface hands query the checkpoint state it
|
|
1526
|
-
// already selected. A raw resume query has no such snapshot; after
|
|
1527
|
-
// cancellation, reading the store again could hang without a signal,
|
|
1528
|
-
// while persisting without it would erase the existing transcript.
|
|
1529
|
-
// Refuse before binding/persisting rather than choose either failure.
|
|
1530
|
-
ctx.abortController.signal.throwIfAborted()
|
|
1531
|
-
}
|
|
1532
|
-
|
|
1533
|
-
const cancelledPrompt = params.systemPrompt ?? ''
|
|
1534
|
-
const cancelledAssembler = new ResultAssembler({
|
|
1535
|
-
runMgr: ctx.runMgr,
|
|
1536
|
-
planManager: ctx.planManager,
|
|
1537
|
-
activityStore: ctx.activityStore,
|
|
1538
|
-
log: ctx.log,
|
|
1539
|
-
emitEvent: eventTranslator.emitEvent,
|
|
1540
|
-
drainPending: () => eventTranslator.drainPending(),
|
|
1541
|
-
signal: ctx.abortController.signal,
|
|
1542
|
-
})
|
|
1543
|
-
const rootSpan = getTracer().startSpan(
|
|
1544
|
-
agentRunSpanName(params.agentName),
|
|
1545
|
-
{},
|
|
1546
|
-
parentContext(params.parentSpan ?? selectedResumeState?.traceContext),
|
|
1547
|
-
)
|
|
1548
|
-
rootSpan.setAttributes({
|
|
1549
|
-
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
1550
|
-
[GENAI.AGENT_NAME]: params.agentName,
|
|
1551
|
-
[GENAI.AGENT_ID]: params.agentId,
|
|
1552
|
-
[GENAI.REQUEST_MODEL]: runConfig.model,
|
|
1553
|
-
[GENAI.SYSTEM]: params.provider.id,
|
|
1554
|
-
})
|
|
1555
|
-
|
|
1556
|
-
try {
|
|
1557
|
-
await ctx.runMgr.init()
|
|
1558
|
-
if (selectedResumeState) {
|
|
1559
|
-
ctx.runMgr.restoreUsage(
|
|
1560
|
-
selectedResumeState.tokenUsage,
|
|
1561
|
-
selectedResumeState.costInfo,
|
|
1562
|
-
selectedResumeState.currentIteration,
|
|
1563
|
-
)
|
|
1564
|
-
for (const message of selectedResumeState.messages) ctx.runMgr.pushMessage(message)
|
|
1565
|
-
for (const queued of queuedForThisRun) ctx.runMgr.pushMessage(queued)
|
|
1566
|
-
} else if (params.continuationMode) {
|
|
1567
|
-
for (const message of initialMessages) ctx.runMgr.pushMessage(message)
|
|
1568
|
-
} else {
|
|
1569
|
-
ctx.runMgr.pushMessage(createSystemMessage(cancelledPrompt, 'cache'))
|
|
1570
|
-
for (const message of initialMessages) ctx.runMgr.pushMessage(message)
|
|
1571
|
-
}
|
|
1572
|
-
if (params.eventCursor) {
|
|
1573
|
-
yield* catchUpFromCursor(
|
|
1574
|
-
ctx.runMgr,
|
|
1575
|
-
params.eventCursor,
|
|
1576
|
-
params.onEventReplay,
|
|
1577
|
-
params.claimFence,
|
|
1578
|
-
(error) => {
|
|
1579
|
-
ctx.log.warn('Replay observer failed after attachment cancellation', {
|
|
1580
|
-
'exception.message': toErrorMessage(error),
|
|
1581
|
-
})
|
|
1582
|
-
},
|
|
1583
|
-
)
|
|
1584
|
-
}
|
|
1585
|
-
if (selectedResumeState) {
|
|
1586
|
-
await eventTranslator.emitEvent({
|
|
1587
|
-
type: 'run_resuming',
|
|
1588
|
-
runId: ctx.runId,
|
|
1589
|
-
fromCheckpointId: selectedResumeState.checkpointId,
|
|
1590
|
-
})
|
|
1591
|
-
yield* eventTranslator.drainPending()
|
|
1592
|
-
}
|
|
1593
|
-
ctx.runMgr.markRunning()
|
|
1594
|
-
await eventTranslator.emitEvent({
|
|
1595
|
-
type: 'run_started',
|
|
1596
|
-
runId: ctx.runId,
|
|
1597
|
-
systemPrompt: cancelledPrompt,
|
|
1598
|
-
})
|
|
1599
|
-
yield* eventTranslator.drainPending()
|
|
1600
|
-
ctx.abortController.signal.throwIfAborted()
|
|
1601
|
-
} catch (error) {
|
|
1602
|
-
// Attachment resolution has already observed the caller's abort. A
|
|
1603
|
-
// reconnect callback can still throw while replay is being reported,
|
|
1604
|
-
// but it cannot replace that terminal cause or turn a cancelled run
|
|
1605
|
-
// into an unpersisted rejection.
|
|
1606
|
-
const terminalError = ctx.abortController.signal.aborted
|
|
1607
|
-
? ctx.abortController.signal.reason
|
|
1608
|
-
: error
|
|
1609
|
-
await executeUserInterruptHooks(terminalError)
|
|
1610
|
-
yield* eventTranslator.drainPending()
|
|
1611
|
-
yield* cancelledAssembler.handleError(terminalError, rootSpan)
|
|
1612
|
-
} finally {
|
|
1613
|
-
rootSpan.end()
|
|
1614
|
-
}
|
|
1615
|
-
|
|
1616
|
-
return await cancelledAssembler.finalize()
|
|
915
|
+
return yield* settlePreStartCancellation(params, prepared)
|
|
1617
916
|
}
|
|
1618
917
|
|
|
1619
918
|
const unsubscribeTaskStore = params.taskStore
|
|
@@ -2142,7 +1441,12 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
2142
1441
|
|
|
2143
1442
|
const tracer = getTracer()
|
|
2144
1443
|
|
|
2145
|
-
|
|
1444
|
+
// Whether the run reached its settle. Read by the `finally` below, and
|
|
1445
|
+
// the only thing that distinguishes a run that finished from one whose
|
|
1446
|
+
// consumer walked away — see `settleAbandonedRun`.
|
|
1447
|
+
let settled = false
|
|
1448
|
+
|
|
1449
|
+
const runBody = (async function* (): AsyncGenerator<RunEvent, Run> {
|
|
2146
1450
|
// Parent explicitly when a caller supplied one. Without this every
|
|
2147
1451
|
// run starts its OWN root trace, so a supervisor delegating to three
|
|
2148
1452
|
// children produced four disconnected traces instead of one tree —
|
|
@@ -2252,6 +1556,11 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
2252
1556
|
// Decided during checkpoint restore, executed after the sandbox
|
|
2253
1557
|
// exists — the approved tools may well need it.
|
|
2254
1558
|
let pendingResume: PendingResumePlan | null = null
|
|
1559
|
+
/**
|
|
1560
|
+
* The cadence park this resume answered, when the decision is one the
|
|
1561
|
+
* ordinary continue path carries out. See the restore path below.
|
|
1562
|
+
*/
|
|
1563
|
+
let answeredParkId: CheckpointId | undefined
|
|
2255
1564
|
/** Tool results recovered from the transcript; see the restore path. */
|
|
2256
1565
|
let recoveredResults: ReadonlyMap<string, { result: string; isError: boolean }> = new Map()
|
|
2257
1566
|
let emergencyManager: EmergencySaveManager | undefined
|
|
@@ -2407,6 +1716,49 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
2407
1716
|
? planPendingResume(projectedCheckpoint, params.pendingDecision, ctx.log)
|
|
2408
1717
|
: null
|
|
2409
1718
|
|
|
1719
|
+
// The park this resume ANSWERS even though there is no plan to
|
|
1720
|
+
// carry the decision out through.
|
|
1721
|
+
//
|
|
1722
|
+
// `planPendingResume` covers the two arms whose decision has to
|
|
1723
|
+
// reach something — the calls a `tool_review` park is about, the
|
|
1724
|
+
// tool a `user_question` park is inside. An `iteration_checkpoint`
|
|
1725
|
+
// park has neither, so it returns no plan, and the unpark further
|
|
1726
|
+
// down — which ran only when there was one — never fired for it.
|
|
1727
|
+
// A run that parked on the cadence, was resumed with
|
|
1728
|
+
// `{action: 'continue'}` and went on to finish its work therefore
|
|
1729
|
+
// kept reporting an OUTSTANDING park to `findPendingCheckpoint`,
|
|
1730
|
+
// so a second resume of the finished run was refused with
|
|
1731
|
+
// `awaiting-decision` for a decision already taken, and because
|
|
1732
|
+
// `prune` skips an unresolved park the row could no longer be
|
|
1733
|
+
// collected by anything.
|
|
1734
|
+
//
|
|
1735
|
+
// The decision IS carried out here — continuing is exactly what
|
|
1736
|
+
// the loop below does, and a plan verdict is the answer to the
|
|
1737
|
+
// question the plan park asked — so the park is resolved at the
|
|
1738
|
+
// same point and with the same meaning "resolved" carries
|
|
1739
|
+
// everywhere else: the record stays, and only its pending state
|
|
1740
|
+
// ends. A `pause` is deliberately not resolved: it holds the
|
|
1741
|
+
// park rather than answering it, which is how the live path
|
|
1742
|
+
// treats it too. `answersParkOf` is the whole map, park type to
|
|
1743
|
+
// answering decision, so an arm cannot go missing by being
|
|
1744
|
+
// absent from a condition again — which is how the plan arm
|
|
1745
|
+
// leaked a finished run's park.
|
|
1746
|
+
//
|
|
1747
|
+
// Resolving it does not depend on the resumed process being able
|
|
1748
|
+
// to act on it, and that is deliberate: the plan's own fate is a
|
|
1749
|
+
// separate defect (nothing restores a plan on the resume path at
|
|
1750
|
+
// all, so the new process has none to approve, execute or
|
|
1751
|
+
// reject) and making the resolution wait for it would leave the
|
|
1752
|
+
// row outstanding for exactly the runs that need it cleared.
|
|
1753
|
+
const parked = projectedCheckpoint.pending
|
|
1754
|
+
answeredParkId =
|
|
1755
|
+
params.pendingDecision &&
|
|
1756
|
+
parked !== undefined &&
|
|
1757
|
+
parked.resolvedAt === undefined &&
|
|
1758
|
+
answersParkOf(parked.request.type, params.pendingDecision)
|
|
1759
|
+
? projectedCheckpoint.id
|
|
1760
|
+
: undefined
|
|
1761
|
+
|
|
2410
1762
|
// Recover completed observations and explicitly unknown outcomes.
|
|
2411
1763
|
// A recorded start is not proof that its external effect failed.
|
|
2412
1764
|
const unanswered = interruptedToolCalls(projectedCheckpoint.messages)
|
|
@@ -2734,6 +2086,9 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
2734
2086
|
})
|
|
2735
2087
|
}
|
|
2736
2088
|
yield* resultAssembler.completeRun(rootSpan)
|
|
2089
|
+
// The run HAS settled, so the outer `finally` must not read
|
|
2090
|
+
// this as an abandonment — it would persist a second time.
|
|
2091
|
+
settled = true
|
|
2737
2092
|
return await resultAssembler.finalize()
|
|
2738
2093
|
}
|
|
2739
2094
|
sandbox = acquisition.sandbox
|
|
@@ -2804,7 +2159,15 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
2804
2159
|
ctx.runMgr.setStopReason('input_guardrail')
|
|
2805
2160
|
ctx.runMgr.setLastError(inputVerdict.reason ?? 'blocked by an input guardrail')
|
|
2806
2161
|
yield* resultAssembler.completeRun(rootSpan)
|
|
2807
|
-
|
|
2162
|
+
// Same two lines as the sandbox path above, and for the same
|
|
2163
|
+
// reasons — with one that path does not have. This return used to
|
|
2164
|
+
// hand back `getRun()` without persisting, so the terminal state
|
|
2165
|
+
// reached the disk only because the abandonment path found
|
|
2166
|
+
// `settled` false and settled it a second time. A branch that
|
|
2167
|
+
// exists for runs which did NOT settle must not be the reason a
|
|
2168
|
+
// settled one is written down.
|
|
2169
|
+
settled = true
|
|
2170
|
+
return await resultAssembler.finalize()
|
|
2808
2171
|
}
|
|
2809
2172
|
|
|
2810
2173
|
// Honor the approval a human already gave, before the loop's
|
|
@@ -2830,149 +2193,59 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
2830
2193
|
|
|
2831
2194
|
await applyPendingResume(pendingResume, ctx.runMgr, toolExecutor, recoveredResults)
|
|
2832
2195
|
yield* eventTranslator.drainPending()
|
|
2833
|
-
|
|
2834
|
-
// The decision has now actually been carried out, so the park
|
|
2835
|
-
// is no longer outstanding. Without this the checkpoint keeps
|
|
2836
|
-
// reporting `pending` with no `resolvedAt`, and an approval
|
|
2837
|
-
// queue re-serves a destructive call that already ran — which
|
|
2838
|
-
// defeats the entire point of recording the park.
|
|
2839
|
-
const resolvedCheckpointId = pendingResume.checkpointId
|
|
2840
|
-
if (params.pendingDecision) {
|
|
2841
|
-
await checkpointMgr
|
|
2842
|
-
.unpark(resolvedCheckpointId, params.pendingDecision)
|
|
2843
|
-
.catch((err: unknown) => {
|
|
2844
|
-
ctx.log.error('Applied a pending decision but failed to clear the park', {
|
|
2845
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
2846
|
-
'namzu.checkpoint.id': resolvedCheckpointId,
|
|
2847
|
-
'exception.message': err instanceof Error ? err.message : String(err),
|
|
2848
|
-
})
|
|
2849
|
-
return null
|
|
2850
|
-
})
|
|
2851
|
-
}
|
|
2852
2196
|
}
|
|
2853
2197
|
|
|
2854
|
-
|
|
2855
|
-
|
|
2856
|
-
|
|
2857
|
-
|
|
2858
|
-
|
|
2859
|
-
|
|
2860
|
-
|
|
2861
|
-
|
|
2862
|
-
|
|
2863
|
-
|
|
2864
|
-
|
|
2865
|
-
|
|
2866
|
-
|
|
2867
|
-
|
|
2868
|
-
|
|
2869
|
-
|
|
2870
|
-
|
|
2871
|
-
|
|
2872
|
-
|
|
2873
|
-
|
|
2874
|
-
|
|
2875
|
-
|
|
2876
|
-
|
|
2877
|
-
|
|
2878
|
-
|
|
2879
|
-
|
|
2880
|
-
|
|
2881
|
-
|
|
2882
|
-
|
|
2883
|
-
|
|
2884
|
-
|
|
2885
|
-
|
|
2886
|
-
|
|
2887
|
-
|
|
2888
|
-
|
|
2889
|
-
// token to gate the stream itself would trade the streaming UX
|
|
2890
|
-
// for the guarantee, which is the host's call, not the SDK's.
|
|
2891
|
-
if (params.outputGuardrails && params.outputGuardrails.length > 0) {
|
|
2892
|
-
// Read what the run produced WITHOUT settling it. This used to
|
|
2893
|
-
// call `markCompleted()` just to materialize the text, which
|
|
2894
|
-
// force-marked a cancelled or paused run `completed` merely
|
|
2895
|
-
// because a guardrail was configured — the presence of a
|
|
2896
|
-
// safety check silently rewrote the run's own outcome.
|
|
2897
|
-
const produced = ctx.runMgr.materializeResult()
|
|
2898
|
-
const outputVerdict = await runOutputGuardrails(
|
|
2899
|
-
params.outputGuardrails,
|
|
2900
|
-
{ runId: ctx.runId, output: produced, messages: ctx.runMgr.messages },
|
|
2901
|
-
ctx.log,
|
|
2902
|
-
)
|
|
2903
|
-
|
|
2904
|
-
if (outputVerdict.blocked || outputVerdict.rewritten !== undefined) {
|
|
2905
|
-
ctx.runMgr.clearStructuredOutput()
|
|
2906
|
-
if (
|
|
2907
|
-
params.structuredOutput &&
|
|
2908
|
-
outputVerdict.rewritten !== undefined &&
|
|
2909
|
-
ctx.runMgr.stopReason === 'end_turn'
|
|
2910
|
-
)
|
|
2911
|
-
ctx.runMgr.setStopReason('output_guardrail')
|
|
2912
|
-
}
|
|
2913
|
-
|
|
2914
|
-
if (outputVerdict.blocked) {
|
|
2915
|
-
await eventTranslator.emitEvent({
|
|
2916
|
-
type: 'guardrail_triggered',
|
|
2917
|
-
runId: ctx.runId,
|
|
2918
|
-
stage: 'output',
|
|
2919
|
-
action: 'block',
|
|
2920
|
-
...(outputVerdict.name ? { guardrail: outputVerdict.name } : {}),
|
|
2921
|
-
...(outputVerdict.reason ? { reason: outputVerdict.reason } : {}),
|
|
2922
|
-
})
|
|
2923
|
-
yield* eventTranslator.drainPending()
|
|
2924
|
-
// Same reasoning as the input-guardrail branch above.
|
|
2925
|
-
await ctx.runMgr.recordAudit({
|
|
2926
|
-
what: { action: 'guardrail:output', resource: outputVerdict.name },
|
|
2927
|
-
outcome: 'refused',
|
|
2928
|
-
reason: outputVerdict.reason ?? 'blocked by an output guardrail',
|
|
2929
|
-
...(params.persona?.identity.role ? { persona: params.persona.identity.role } : {}),
|
|
2930
|
-
})
|
|
2931
|
-
ctx.runMgr.setStopReason('output_guardrail')
|
|
2932
|
-
ctx.runMgr.setLastError(outputVerdict.reason ?? 'blocked by an output guardrail')
|
|
2933
|
-
ctx.runMgr.setResult('')
|
|
2934
|
-
} else if (outputVerdict.rewritten !== undefined) {
|
|
2935
|
-
await eventTranslator.emitEvent({
|
|
2936
|
-
type: 'guardrail_triggered',
|
|
2937
|
-
runId: ctx.runId,
|
|
2938
|
-
stage: 'output',
|
|
2939
|
-
action: 'rewrite',
|
|
2940
|
-
...(outputVerdict.name ? { guardrail: outputVerdict.name } : {}),
|
|
2941
|
-
...(outputVerdict.reason ? { reason: outputVerdict.reason } : {}),
|
|
2198
|
+
// The decision has now actually been carried out, so the park it
|
|
2199
|
+
// answered is no longer outstanding. Without this the checkpoint
|
|
2200
|
+
// keeps reporting `pending` with no `resolvedAt`, and an approval
|
|
2201
|
+
// queue re-serves a call that already ran — or a question already
|
|
2202
|
+
// answered — which defeats the entire point of recording the park.
|
|
2203
|
+
//
|
|
2204
|
+
// Two arms reach this point, and being outside `if (pendingResume)`
|
|
2205
|
+
// is what the second one needs. One is a plan whose decision was
|
|
2206
|
+
// applied to a batch above. The other is the cadence arm, for which
|
|
2207
|
+
// `planPendingResume` rightly produces no plan because the loop
|
|
2208
|
+
// resuming IS its decision being carried out (`answeredParkId`, set
|
|
2209
|
+
// on the restore path). Resolving only the first left a finished run
|
|
2210
|
+
// reporting `awaiting-decision` forever.
|
|
2211
|
+
//
|
|
2212
|
+
// What is RECORDED depends on which of the two produced the plan. A
|
|
2213
|
+
// recovery plan means the batch was answered by the crash path
|
|
2214
|
+
// rather than by the decision — the calls the human was asked about
|
|
2215
|
+
// were closed with explicitly unknown outcomes — so the human's
|
|
2216
|
+
// answer must not be written down as what ended the park. The park is
|
|
2217
|
+
// still resolved: the question is moot, and leaving it outstanding
|
|
2218
|
+
// would have `findPendingCheckpoint` serve it as the newest
|
|
2219
|
+
// outstanding park, so a host resuming it would rewind this run to
|
|
2220
|
+
// the checkpoint the crash happened on and re-execute a batch the run
|
|
2221
|
+
// has long since moved past.
|
|
2222
|
+
const resolvedCheckpointId = pendingResume?.checkpointId ?? answeredParkId
|
|
2223
|
+
const recordedDecision =
|
|
2224
|
+
pendingResume?.source === 'recovery' && params.pendingDecision
|
|
2225
|
+
? supersededByRecovery(params.pendingDecision)
|
|
2226
|
+
: params.pendingDecision
|
|
2227
|
+
if (recordedDecision && resolvedCheckpointId) {
|
|
2228
|
+
await checkpointMgr.unpark(resolvedCheckpointId, recordedDecision).catch((err: unknown) => {
|
|
2229
|
+
ctx.log.error('Applied a pending decision but failed to clear the park', {
|
|
2230
|
+
[NAMZU.RUN_ID]: ctx.runId,
|
|
2231
|
+
'namzu.checkpoint.id': resolvedCheckpointId,
|
|
2232
|
+
'exception.message': err instanceof Error ? err.message : String(err),
|
|
2942
2233
|
})
|
|
2943
|
-
|
|
2944
|
-
ctx.runMgr.setResult(outputVerdict.rewritten)
|
|
2945
|
-
}
|
|
2946
|
-
}
|
|
2947
|
-
|
|
2948
|
-
if (params.consolidateInto && workingStateManager) {
|
|
2949
|
-
const entry = consolidationEntry(workingStateManager.getState(), {
|
|
2950
|
-
runId: ctx.runId,
|
|
2951
|
-
at: Date.now(),
|
|
2234
|
+
return null
|
|
2952
2235
|
})
|
|
2953
|
-
if (entry) {
|
|
2954
|
-
try {
|
|
2955
|
-
const { entry: saved } = await params.consolidateInto.create(entry)
|
|
2956
|
-
await eventTranslator.emitEvent({
|
|
2957
|
-
type: 'memory_consolidated',
|
|
2958
|
-
runId: ctx.runId,
|
|
2959
|
-
memoryId: saved.id,
|
|
2960
|
-
title: entry.title,
|
|
2961
|
-
decisions: workingStateManager.getState().decisions.length,
|
|
2962
|
-
discoveries: workingStateManager.getState().discoveries.length,
|
|
2963
|
-
failures: workingStateManager.getState().failures.length,
|
|
2964
|
-
})
|
|
2965
|
-
yield* eventTranslator.drainPending()
|
|
2966
|
-
} catch (error) {
|
|
2967
|
-
ctx.log.warn('consolidation into the memory store failed', {
|
|
2968
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
2969
|
-
'namzu.memory.error': toErrorMessage(error),
|
|
2970
|
-
})
|
|
2971
|
-
}
|
|
2972
|
-
}
|
|
2973
2236
|
}
|
|
2974
|
-
|
|
2975
|
-
yield*
|
|
2237
|
+
|
|
2238
|
+
yield* iterationOrchestrator.runLoop()
|
|
2239
|
+
|
|
2240
|
+
yield* finalizeRun({
|
|
2241
|
+
ctx,
|
|
2242
|
+
params,
|
|
2243
|
+
eventTranslator,
|
|
2244
|
+
takeSteps: () => iterationOrchestrator.getSteps(),
|
|
2245
|
+
workingStateManager,
|
|
2246
|
+
resultAssembler,
|
|
2247
|
+
rootSpan,
|
|
2248
|
+
})
|
|
2976
2249
|
} catch (err) {
|
|
2977
2250
|
// A failed run still spent its steps; report them.
|
|
2978
2251
|
ctx.runMgr.setSteps(iterationOrchestrator.getSteps())
|
|
@@ -2980,109 +2253,129 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
2980
2253
|
yield* eventTranslator.drainPending()
|
|
2981
2254
|
yield* resultAssembler.handleError(err, rootSpan)
|
|
2982
2255
|
} finally {
|
|
2983
|
-
|
|
2984
|
-
|
|
2985
|
-
|
|
2986
|
-
|
|
2987
|
-
|
|
2988
|
-
|
|
2989
|
-
|
|
2990
|
-
|
|
2991
|
-
|
|
2992
|
-
|
|
2993
|
-
|
|
2994
|
-
|
|
2995
|
-
|
|
2996
|
-
|
|
2997
|
-
|
|
2998
|
-
|
|
2999
|
-
|
|
3000
|
-
// Only jobs bound to this run. Jobs a host bound to its session are
|
|
3001
|
-
// the host's to stop, when the session ends.
|
|
3002
|
-
if (params.backgroundJobs && (params.backgroundJobOwner ?? ctx.runId) === ctx.runId) {
|
|
3003
|
-
try {
|
|
3004
|
-
const stopped = await params.backgroundJobs.killOwner(ctx.runId)
|
|
3005
|
-
if (stopped.length > 0) {
|
|
3006
|
-
ctx.log.info('Background jobs stopped with the run', {
|
|
3007
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
3008
|
-
'namzu.jobs.stopped': stopped.length,
|
|
3009
|
-
})
|
|
3010
|
-
}
|
|
3011
|
-
} catch (jobErr) {
|
|
3012
|
-
ctx.log.error('A background job did not stop cleanly', {
|
|
3013
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
3014
|
-
...errorAttributes(jobErr),
|
|
3015
|
-
})
|
|
3016
|
-
}
|
|
3017
|
-
}
|
|
3018
|
-
|
|
3019
|
-
// Same reasoning for the question channel: the tools outlive the
|
|
3020
|
-
// run that bound them, so leaving it attached would have a later
|
|
3021
|
-
// run's question written into this run's checkpoint store.
|
|
3022
|
-
questionParks.unbind()
|
|
3023
|
-
|
|
3024
|
-
// Offer what the run learned to whoever decides what is worth
|
|
3025
|
-
// keeping. In `finally` and awaited: a run that failed still
|
|
3026
|
-
// discovered things, and a fire-and-forget write would race the
|
|
3027
|
-
// process exiting on a one-shot CLI run. A throw here is
|
|
3028
|
-
// swallowed — a memory that failed to form must not retract an
|
|
3029
|
-
// answer that was already produced.
|
|
3030
|
-
const candidate = memoryCandidateFor(ctx.runId, workingStateManager)
|
|
3031
|
-
if (params.promoteMemory && candidate) {
|
|
3032
|
-
try {
|
|
3033
|
-
await params.promoteMemory(candidate)
|
|
3034
|
-
} catch (promoteErr) {
|
|
3035
|
-
ctx.log.error('Memory promotion threw — the run is unaffected', {
|
|
3036
|
-
[NAMZU.RUN_ID]: ctx.runId,
|
|
3037
|
-
'exception.message':
|
|
3038
|
-
promoteErr instanceof Error ? promoteErr.message : String(promoteErr),
|
|
3039
|
-
})
|
|
3040
|
-
}
|
|
3041
|
-
}
|
|
3042
|
-
|
|
3043
|
-
// --- Sandbox lifecycle: destroy after run ---
|
|
3044
|
-
if (sandbox) {
|
|
3045
|
-
const sandboxId = sandbox.id
|
|
3046
|
-
const teardown = await teardownSandbox(sandbox, sandboxTeardownTimeoutMs)
|
|
3047
|
-
if (teardown.kind === 'destroyed') {
|
|
3048
|
-
await eventTranslator.emitEvent({
|
|
3049
|
-
type: 'sandbox_destroyed',
|
|
3050
|
-
runId: ctx.runId,
|
|
3051
|
-
sandboxId,
|
|
3052
|
-
})
|
|
3053
|
-
yield* eventTranslator.drainPending()
|
|
3054
|
-
ctx.log.info('Sandbox destroyed', { 'namzu.sandbox.id': sandboxId })
|
|
3055
|
-
} else {
|
|
3056
|
-
ctx.log.error('Sandbox destroy failed', {
|
|
3057
|
-
'namzu.sandbox.id': sandboxId,
|
|
3058
|
-
...errorAttributes(teardown.error),
|
|
3059
|
-
})
|
|
3060
|
-
}
|
|
3061
|
-
}
|
|
3062
|
-
|
|
3063
|
-
unsubscribeTaskStore?.()
|
|
3064
|
-
// Keyed by HOW it settled, not just that it did: a run that was
|
|
3065
|
-
// cancelled and a run that hit its budget have very different
|
|
3066
|
-
// duration distributions, and averaging them together describes
|
|
3067
|
-
// neither.
|
|
3068
|
-
recordRunDuration(ctx.runMgr.getRun().status ?? 'unknown', Date.now() - runStartedAt)
|
|
3069
|
-
rootSpan.end()
|
|
2256
|
+
yield* releaseRunResources({
|
|
2257
|
+
ctx,
|
|
2258
|
+
eventTranslator,
|
|
2259
|
+
emergencyManager,
|
|
2260
|
+
unsubscribeJobExits,
|
|
2261
|
+
unsubscribeTaskStore,
|
|
2262
|
+
awaitedJobs,
|
|
2263
|
+
backgroundJobs: params.backgroundJobs,
|
|
2264
|
+
backgroundJobOwner: params.backgroundJobOwner,
|
|
2265
|
+
questionParks,
|
|
2266
|
+
workingStateManager,
|
|
2267
|
+
promoteMemory: params.promoteMemory,
|
|
2268
|
+
sandbox,
|
|
2269
|
+
sandboxTeardownTimeoutMs,
|
|
2270
|
+
runStartedAt,
|
|
2271
|
+
rootSpan,
|
|
2272
|
+
})
|
|
3070
2273
|
}
|
|
3071
2274
|
|
|
2275
|
+
// Reached only by a run that settled on its own terms. `finalize()` is
|
|
2276
|
+
// the only thing in this body that writes the durable half of the run,
|
|
2277
|
+
// and a `return` completion arriving from a consumer (`break` out of
|
|
2278
|
+
// `for await`, `gen.return()`) runs the `finally` above and stops short
|
|
2279
|
+
// of here. The flag is what tells the two apart, and this is one of
|
|
2280
|
+
// three sites that set it — the sandbox-acquisition and input-guardrail
|
|
2281
|
+
// returns settle early and set it there. Set before the await rather
|
|
2282
|
+
// than after, because a store that throws on the way out must not send
|
|
2283
|
+
// the abandonment path over the same broken ground.
|
|
2284
|
+
settled = true
|
|
3072
2285
|
return await resultAssembler.finalize()
|
|
3073
2286
|
})()
|
|
2287
|
+
|
|
2288
|
+
try {
|
|
2289
|
+
return yield* runBody
|
|
2290
|
+
} finally {
|
|
2291
|
+
if (!settled) await settleAbandonedRun(ctx.runMgr, ctx.log)
|
|
2292
|
+
}
|
|
3074
2293
|
}
|
|
3075
2294
|
|
|
3076
2295
|
/**
|
|
3077
|
-
*
|
|
2296
|
+
* Write a terminal durable record for a run whose consumer walked away.
|
|
2297
|
+
*
|
|
2298
|
+
* `for await (… ) break` and an explicit `gen.return()` both end the run
|
|
2299
|
+
* body early. Everything the run's `finally` owns still happens — background
|
|
2300
|
+
* jobs are killed, the sandbox is destroyed, the span ends, the duration is
|
|
2301
|
+
* recorded — and then the generator stops. `finalize()` never runs, so
|
|
2302
|
+
* `persist()` never runs, and the store keeps whatever `init()` wrote: a
|
|
2303
|
+
* non-terminal status for a run that no longer exists. `deriveRunStatus`
|
|
2304
|
+
* reads that record back as `queued`, work waiting to start, and a host
|
|
2305
|
+
* rebuilding its view from the store believes it.
|
|
2306
|
+
*
|
|
2307
|
+
* There is nothing to emit here and nothing to emit it to: the consumer
|
|
2308
|
+
* that would have received the events is the one that left. This is about
|
|
2309
|
+
* the durable record only.
|
|
2310
|
+
*
|
|
2311
|
+
* `cancelled` is the verdict, and it is chosen from the existing vocabulary
|
|
2312
|
+
* because it is the one that is true. The run did not complete — no result
|
|
2313
|
+
* was produced and no terminal event was ever delivered — and nothing
|
|
2314
|
+
* failed, so `failed` would name an error that never happened; a run whose
|
|
2315
|
+
* consumer stopped reading and whose processes were torn down under it is
|
|
2316
|
+
* the same fact `markCancelled` already records when a run abort tears one
|
|
2317
|
+
* down. It needs no new `RunExecutionStatus` and no new `StopReason`.
|
|
3078
2318
|
*
|
|
3079
|
-
*
|
|
3080
|
-
*
|
|
3081
|
-
*
|
|
3082
|
-
*
|
|
3083
|
-
*
|
|
3084
|
-
*
|
|
2319
|
+
* A verdict the run already reached is left standing. A run that failed,
|
|
2320
|
+
* or was cancelled, before the consumer left still says so; what the
|
|
2321
|
+
* abandonment adds is that the record reaches the disk at all.
|
|
2322
|
+
*
|
|
2323
|
+
* Neither is a verdict written over a PARK. A park is a promise to a human
|
|
2324
|
+
* that outlives the consumer: the run is resumable and somebody is still owed
|
|
2325
|
+
* an answer, and `deriveRunStatus` reads a terminal status BEFORE it reads the
|
|
2326
|
+
* park — so recording `cancelled` turns `awaiting_hitl` into `cancelled` for a
|
|
2327
|
+
* run nobody answered for, while the unanswered question stays on the record
|
|
2328
|
+
* and the checkpoint it belongs to stays the place a resume starts from. The
|
|
2329
|
+
* durable state is asked rather than the in-memory one because the in-memory
|
|
2330
|
+
* one is the misleading half here: `handleHITLDecision` emits `run_paused` and
|
|
2331
|
+
* drains it BEFORE it calls `setStopReason('paused')`, so a consumer that
|
|
2332
|
+
* leaves on that event leaves a run whose status is `running` and whose stop
|
|
2333
|
+
* reason is unset at the exact instant its park is already durable.
|
|
2334
|
+
* `findPendingCheckpoint` is the same read an approval queue is built from,
|
|
2335
|
+
* expired parks included in its judgement: a park nobody answered in time is
|
|
2336
|
+
* not somebody still being asked.
|
|
2337
|
+
*
|
|
2338
|
+
* Never throws. It runs while an exception may already be unwinding, and a
|
|
2339
|
+
* store that cannot be written must not replace the run's real failure with
|
|
2340
|
+
* its own.
|
|
3085
2341
|
*/
|
|
2342
|
+
async function settleAbandonedRun(runMgr: RunPersistence, log: Logger): Promise<void> {
|
|
2343
|
+
try {
|
|
2344
|
+
// A terminal verdict is written whatever the park says: `deriveRunStatus`
|
|
2345
|
+
// settles a run that finished, failed or was cancelled BEFORE it looks at
|
|
2346
|
+
// a park ("terminal beats parked"), so a settled run is not waiting for
|
|
2347
|
+
// anybody and the row it already wrote must reach the disk. This ordering
|
|
2348
|
+
// is also what keeps a stale park from suppressing the write.
|
|
2349
|
+
if (!isTerminalStatus(runMgr.status)) {
|
|
2350
|
+
const parked = await findPendingCheckpoint(runMgr.getCheckpointStore(), runMgr.getRunScope())
|
|
2351
|
+
if (parked) {
|
|
2352
|
+
// Left exactly as it stands: no verdict, no write. The park row is
|
|
2353
|
+
// this run's durable state, and `persist()` here would add a
|
|
2354
|
+
// second claim — `running`, for a process that is gone — beside it.
|
|
2355
|
+
log.info('Abandoned run left parked for a human to answer', {
|
|
2356
|
+
[NAMZU.RUN_ID]: runMgr.id,
|
|
2357
|
+
'namzu.checkpoint.id': parked.id,
|
|
2358
|
+
'namzu.runtime.park_type': parked.pending?.request.type,
|
|
2359
|
+
})
|
|
2360
|
+
return
|
|
2361
|
+
}
|
|
2362
|
+
runMgr.markCancelled()
|
|
2363
|
+
}
|
|
2364
|
+
// Once: the `finally` that calls this runs once, and every site in the
|
|
2365
|
+
// run body that settles through `finalize()` sets `settled` before it
|
|
2366
|
+
// returns, so the two can never both write.
|
|
2367
|
+
await runMgr.persist()
|
|
2368
|
+
log.info('Abandoned run recorded as cancelled', {
|
|
2369
|
+
[NAMZU.RUN_ID]: runMgr.id,
|
|
2370
|
+
})
|
|
2371
|
+
} catch (err) {
|
|
2372
|
+
log.error('Failed to record the terminal state of an abandoned run', {
|
|
2373
|
+
[NAMZU.RUN_ID]: runMgr.id,
|
|
2374
|
+
'exception.message': err instanceof Error ? err.message : String(err),
|
|
2375
|
+
})
|
|
2376
|
+
}
|
|
2377
|
+
}
|
|
2378
|
+
|
|
3086
2379
|
/** The text of the newest user turn, which is what a prompt hook is asked about. */
|
|
3087
2380
|
function lastUserPrompt(messages: readonly Message[]): string {
|
|
3088
2381
|
for (let i = messages.length - 1; i >= 0; i--) {
|
|
@@ -3092,40 +2385,6 @@ function lastUserPrompt(messages: readonly Message[]): string {
|
|
|
3092
2385
|
return ''
|
|
3093
2386
|
}
|
|
3094
2387
|
|
|
3095
|
-
async function* catchUpFromCursor(
|
|
3096
|
-
runMgr: RunPersistence,
|
|
3097
|
-
cursor: RunEventCursor,
|
|
3098
|
-
onEventReplay: ((replay: RunEventReplay) => void) | undefined,
|
|
3099
|
-
generation: FencingToken | undefined,
|
|
3100
|
-
onReplayObserverError: (error: unknown) => void,
|
|
3101
|
-
): AsyncGenerator<RunEvent, void> {
|
|
3102
|
-
const missed = await runMgr.getRunStore().readEvents({ sinceSeq: cursor.sinceSeq })
|
|
3103
|
-
const replay = resolveRunEventReplay(
|
|
3104
|
-
cursor,
|
|
3105
|
-
{
|
|
3106
|
-
lastSeq: runMgr.lastEventSeq,
|
|
3107
|
-
...(generation !== undefined ? { generation } : {}),
|
|
3108
|
-
},
|
|
3109
|
-
missed,
|
|
3110
|
-
)
|
|
3111
|
-
|
|
3112
|
-
if (onEventReplay) {
|
|
3113
|
-
try {
|
|
3114
|
-
// A callback typed `void` may still be implemented with `async` in
|
|
3115
|
-
// TypeScript. Observe that runtime Promise so a late rejection cannot
|
|
3116
|
-
// become process-wide, but never await host code here: replay delivery
|
|
3117
|
-
// and an already-cancelled run must not inherit observer liveness.
|
|
3118
|
-
const settlement = onEventReplay(replay)
|
|
3119
|
-
void Promise.resolve(settlement).catch(onReplayObserverError)
|
|
3120
|
-
} catch (error) {
|
|
3121
|
-
onReplayObserverError(error)
|
|
3122
|
-
}
|
|
3123
|
-
}
|
|
3124
|
-
|
|
3125
|
-
if (replay.status !== 'replayed') return
|
|
3126
|
-
for (const event of replay.events) yield event
|
|
3127
|
-
}
|
|
3128
|
-
|
|
3129
2388
|
type DrainQueryParams = Omit<QueryParams, 'resumeHandler'> & {
|
|
3130
2389
|
resumeHandler?: ResumeHandler
|
|
3131
2390
|
}
|