@namzu/sdk 42.0.0 → 42.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/CHANGELOG.md +222 -0
  2. package/dist/manager/resident/outbox.d.ts +8 -8
  3. package/dist/manager/resident/store.d.ts +4 -4
  4. package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
  5. package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
  6. package/dist/runtime/query/cancelled-before-start.js +152 -0
  7. package/dist/runtime/query/cancelled-before-start.js.map +1 -0
  8. package/dist/runtime/query/checkpoint.d.ts +21 -0
  9. package/dist/runtime/query/checkpoint.d.ts.map +1 -1
  10. package/dist/runtime/query/checkpoint.js +23 -0
  11. package/dist/runtime/query/checkpoint.js.map +1 -1
  12. package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
  13. package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
  14. package/dist/runtime/query/executor/tool-call-admission.js +373 -0
  15. package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
  16. package/dist/runtime/query/executor.d.ts +70 -35
  17. package/dist/runtime/query/executor.d.ts.map +1 -1
  18. package/dist/runtime/query/executor.js +46 -380
  19. package/dist/runtime/query/executor.js.map +1 -1
  20. package/dist/runtime/query/finalize-run.d.ts +55 -0
  21. package/dist/runtime/query/finalize-run.d.ts.map +1 -0
  22. package/dist/runtime/query/finalize-run.js +113 -0
  23. package/dist/runtime/query/finalize-run.js.map +1 -0
  24. package/dist/runtime/query/index.d.ts +4 -9
  25. package/dist/runtime/query/index.d.ts.map +1 -1
  26. package/dist/runtime/query/index.js +238 -893
  27. package/dist/runtime/query/index.js.map +1 -1
  28. package/dist/runtime/query/iteration/index.d.ts +6 -161
  29. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  30. package/dist/runtime/query/iteration/index.js +23 -523
  31. package/dist/runtime/query/iteration/index.js.map +1 -1
  32. package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
  33. package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
  34. package/dist/runtime/query/iteration/outstanding-work.js +365 -0
  35. package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
  36. package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
  37. package/dist/runtime/query/iteration/phases/plan.js +13 -2
  38. package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
  39. package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
  40. package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
  41. package/dist/runtime/query/iteration/step-shaping.js +184 -0
  42. package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
  43. package/dist/runtime/query/prepare-run.d.ts +94 -0
  44. package/dist/runtime/query/prepare-run.d.ts.map +1 -0
  45. package/dist/runtime/query/prepare-run.js +589 -0
  46. package/dist/runtime/query/prepare-run.js.map +1 -0
  47. package/dist/runtime/query/release-run.d.ts +56 -0
  48. package/dist/runtime/query/release-run.d.ts.map +1 -0
  49. package/dist/runtime/query/release-run.js +101 -0
  50. package/dist/runtime/query/release-run.js.map +1 -0
  51. package/dist/runtime/query/resume-pending.d.ts +112 -1
  52. package/dist/runtime/query/resume-pending.d.ts.map +1 -1
  53. package/dist/runtime/query/resume-pending.js +133 -0
  54. package/dist/runtime/query/resume-pending.js.map +1 -1
  55. package/dist/store/evidence/compaction-archive.d.ts +2 -2
  56. package/dist/types/run/config.d.ts +12 -5
  57. package/dist/types/run/config.d.ts.map +1 -1
  58. package/package.json +4 -4
  59. package/src/runtime/query/cancelled-before-start.ts +189 -0
  60. package/src/runtime/query/checkpoint.ts +22 -0
  61. package/src/runtime/query/executor/tool-call-admission.ts +473 -0
  62. package/src/runtime/query/executor.ts +63 -442
  63. package/src/runtime/query/finalize-run.ts +192 -0
  64. package/src/runtime/query/index.ts +270 -1011
  65. package/src/runtime/query/iteration/index.ts +40 -586
  66. package/src/runtime/query/iteration/outstanding-work.ts +386 -0
  67. package/src/runtime/query/iteration/phases/plan.ts +18 -2
  68. package/src/runtime/query/iteration/step-shaping.ts +271 -0
  69. package/src/runtime/query/prepare-run.ts +718 -0
  70. package/src/runtime/query/release-run.ts +168 -0
  71. package/src/runtime/query/resume-pending.ts +158 -0
  72. package/src/types/run/config.ts +12 -5
@@ -6,14 +6,8 @@ import {
6
6
  TriggerEvaluator,
7
7
  assertBudgetEnforceable,
8
8
  } from '../../advisory/index.js'
9
- import { drainQueuedMessages } from '../../agents/handle.js'
10
9
  import { AuthorizationGate } from '../../authorization/gate.js'
11
- import { consolidationEntry } from '../../compaction/consolidation.js'
12
- import {
13
- type ToolHistoryRepairReport,
14
- repairToolMessageHistory,
15
- toolHistoryRepairChanged,
16
- } from '../../compaction/dangling.js'
10
+ import { repairToolMessageHistory, toolHistoryRepairChanged } from '../../compaction/dangling.js'
17
11
  import { extractFromUserMessage } from '../../compaction/extractor.js'
18
12
  import { WorkingStateManager } from '../../compaction/manager.js'
19
13
  import type { ContextReducer } from '../../compaction/reducer.js'
@@ -23,21 +17,14 @@ import { type CompactionConfig, CompactionConfigSchema } from '../../config/runt
23
17
  import { TOOL_OUTPUT_DIR_NAME } from '../../constants/tools/index.js'
24
18
  import { EmergencySaveManager } from '../../manager/run/emergency.js'
25
19
  import type { RunPersistence } from '../../manager/run/persistence.js'
26
- import { resolveModelPricing } from '../../pricing/index.js'
27
20
  import { PromptContributionRegistry } from '../../prompt/contributions.js'
28
21
  import { resolveProviderCapabilities } from '../../provider/capabilities.js'
29
- import { isCallerAbortError } from '../../provider/errors.js'
30
- import {
31
- type ProviderChainMember,
32
- type ServingMember,
33
- withProviderFallback,
34
- } from '../../provider/fallback.js'
35
- import { resolveStreamIdleTimeoutMs, withStreamIdleTimeout } from '../../provider/idle-timeout.js'
36
- import { type ProviderRetryConfig, withProviderRetry } from '../../provider/retry.js'
22
+ import type { ProviderChainMember } from '../../provider/fallback.js'
23
+ import { withStreamIdleTimeout } from '../../provider/idle-timeout.js'
24
+ import type { ProviderRetryConfig } from '../../provider/retry.js'
37
25
  import { withTokenBudget } from '../../provider/token-budget.js'
38
26
  import type { TokenBudget } from '../../run/token-budget.js'
39
27
  import type { PathBuilder } from '../../session/workspace/path-builder.js'
40
- import { resolveAttachments } from '../../store/attachment/index.js'
41
28
  import {
42
29
  GENAI,
43
30
  NAMZU,
@@ -45,8 +32,6 @@ import {
45
32
  parentContext,
46
33
  serializeSpan,
47
34
  } from '../../telemetry/attributes.js'
48
- import type { SerializedSpanContext } from '../../telemetry/attributes.js'
49
- import { recordRunDuration } from '../../telemetry/metrics.js'
50
35
  import { getTracer } from '../../telemetry/runtime-accessors.js'
51
36
  import { buildAdvisoryTools } from '../../tools/advisory/index.js'
52
37
  import { SearchToolsTool } from '../../tools/builtins/search-tools.js'
@@ -60,6 +45,7 @@ import type { AgentRuntimeContext, RuntimeToolOverrides } from '../../types/agen
60
45
  import type { AgentContextLevel } from '../../types/agent/factory.js'
61
46
  import type { WorkingMemoryProvider } from '../../types/agent/working-memory.js'
62
47
  import type { AuthorizationGateConfig } from '../../types/authorization/index.js'
48
+ import { isTerminalStatus } from '../../types/common/index.js'
63
49
  import { NamzuError } from '../../types/errors/index.js'
64
50
  import type { InputGuardrailSpec, OutputGuardrailSpec } from '../../types/guardrail/index.js'
65
51
  import {
@@ -67,23 +53,20 @@ import {
67
53
  type ResumeHandler,
68
54
  autoApproveHandler,
69
55
  } from '../../types/hitl/index.js'
70
- import type { CheckpointId, PlanId, RunId, SessionId, TenantId } from '../../types/ids/index.js'
56
+ import type { CheckpointId, RunId, SessionId, TenantId } from '../../types/ids/index.js'
71
57
  import type { InvocationState } from '../../types/invocation/index.js'
72
58
  import type { MemoryStore } from '../../types/memory/index.js'
73
59
  import {
74
60
  type AssistantMessage,
75
61
  type Message,
76
- type UserMessage,
77
62
  createSystemMessage,
78
63
  } from '../../types/message/index.js'
79
64
  import type { AgentPersona } from '../../types/persona/index.js'
80
65
  import type { LLMProvider } from '../../types/provider/index.js'
81
66
  import type { TaskRouterConfig } from '../../types/router/index.js'
82
67
  import type { ReviewAnswer } from '../../types/run/answer-review.js'
83
- import { cancelCauseOf } from '../../types/run/cancel-cause.js'
84
68
  import type { CheckpointStore, FencingToken } from '../../types/run/checkpoint-store.js'
85
69
  import type { RunEventCursor, RunEventReplay } from '../../types/run/event-cursor.js'
86
- import { resolveRunEventReplay } from '../../types/run/event-cursor.js'
87
70
  import type {
88
71
  AgentRunConfig,
89
72
  BeforeStep,
@@ -95,8 +78,6 @@ import type {
95
78
  StopCondition,
96
79
  } from '../../types/run/index.js'
97
80
  import type { PromoteMemory } from '../../types/run/memory-promotion.js'
98
- import { memoryCandidateFor } from '../../types/run/memory-promotion.js'
99
- import type { RunState } from '../../types/run/state.js'
100
81
  import type { RunStore } from '../../types/run/store.js'
101
82
  import type { TokenBudgetStore } from '../../types/run/token-budget-store.js'
102
83
  import type { Sandbox, SandboxProvider } from '../../types/sandbox/index.js'
@@ -109,50 +90,46 @@ import type { RepairToolCall } from '../../types/tool/repair.js'
109
90
  import type { BackoffPolicy } from '../../utils/backoff.js'
110
91
  import type { ModelPricing } from '../../utils/cost.js'
111
92
  import { toErrorMessage } from '../../utils/error.js'
112
- import { generateCheckpointId, generateRunId } from '../../utils/id.js'
113
93
  import { errorAttributes } from '../../utils/log/exception.js'
114
94
  import type { Logger } from '../../utils/logger.js'
115
95
  import { AwaitedJobs } from '../jobs/awaited-jobs.js'
116
96
  import type { BackgroundJobRegistry } from '../jobs/registry.js'
117
- import { AUTO_APPROVE_POLICY_NAME, createRunApprovalPolicy } from './approval-policy.js'
118
- import { CheckpointManager } from './checkpoint.js'
119
- import { RunContextFactory } from './context.js'
120
- import { EventTranslator } from './events.js'
97
+ import { catchUpFromCursor, settlePreStartCancellation } from './cancelled-before-start.js'
98
+ import { CheckpointManager, findPendingCheckpoint } from './checkpoint.js'
99
+ import { finalizeRun } from './finalize-run.js'
121
100
  import { GuardCoordinator } from './guard.js'
122
- import { runInputGuardrails, runOutputGuardrails } from './guardrails.js'
101
+ import { runInputGuardrails } from './guardrails.js'
123
102
  import { IterationOrchestrator } from './iteration/index.js'
124
103
  import { isCompactionMessage } from './iteration/phases/compaction.js'
125
104
  import { isWorkingMemoryMessage } from './iteration/phases/working-memory.js'
126
105
  import { applyLifecycleHookResults } from './plugin-hooks.js'
127
106
  import {
128
- type ProjectInstructionContext,
129
- awaitProjectInstructionCallback,
130
- collapseProjectInstructionSnapshots,
131
- replaceProjectInstructionSnapshot,
132
- } from './project-instructions.js'
107
+ type SelectedResumeState,
108
+ prepareRun,
109
+ projectStateBearingHistory,
110
+ resolveProviderContextWindow,
111
+ selectedResumeStates,
112
+ } from './prepare-run.js'
113
+ import type { ProjectInstructionContext } from './project-instructions.js'
133
114
  import type { PromptCache } from './prompt-cache.js'
134
115
  import { PromptBuilder } from './prompt.js'
135
116
  import type { PromptSegments } from './prompt.js'
136
117
  import { PendingAnswers, QuestionParkBinding } from './question-park.js'
118
+ import { releaseRunResources } from './release-run.js'
137
119
  import { RepeatCallTracker } from './repeat-call.js'
138
- import { resolveMaxRequestRichContentBytes } from './request-rich-content.js'
139
120
  import { ResultAssembler } from './result.js'
140
121
  import {
141
122
  type PendingResumePlan,
123
+ answersParkOf,
142
124
  applyPendingResume,
143
125
  interruptedToolCalls,
144
126
  planCrashResume,
145
127
  planPendingResume,
146
128
  recoverCompletedCalls,
129
+ supersededByRecovery,
147
130
  } from './resume-pending.js'
148
- import {
149
- acquireSandbox,
150
- resolveSandboxTeardownTimeoutMs,
151
- teardownSandbox,
152
- } from './sandbox-lifecycle.js'
131
+ import { acquireSandbox } from './sandbox-lifecycle.js'
153
132
  import { SteeringBinding, type SteeringChannel, isOperatorUserMessage } from './steering.js'
154
- import { resolveQueryBudget } from './token-budget.js'
155
- import { assertMaxToolCalls } from './tool-call-budget.js'
156
133
  import { ToolGrantSet } from './tool-grants.js'
157
134
  import { createToolPause } from './tool-pause.js'
158
135
  import { ToolingBootstrap } from './tooling.js'
@@ -832,196 +809,6 @@ export interface QueryParams {
832
809
  strictCapabilities?: boolean
833
810
  }
834
811
 
835
- type SelectedResumeState = RunState & {
836
- readonly checkpointId: CheckpointId
837
- readonly traceContext?: SerializedSpanContext
838
- }
839
- const selectedResumeStates = new WeakMap<QueryParams, SelectedResumeState>()
840
-
841
- /**
842
- * Refuse to price a run whose tokens two differently-priced members may produce.
843
- *
844
- * `RunPersistence` holds ONE {@link ModelPricing} table and applies it to every
845
- * accumulation regardless of which model produced the tokens. Across a swap that
846
- * makes `costInfo.totalCost` wrong by an unbounded margin, and silently — the
847
- * number keeps the shape of an answer. `CostInfo` cannot express the truth
848
- * either: it carries `inputCostPer1M` / `outputCostPer1M`, and there is no
849
- * honest value for those once a total spans two rate cards.
850
- *
851
- * So the total is refused rather than blended. Naming what that costs is part
852
- * of the refusal, because the caller loses `costLimitUsd` with it: the guard
853
- * enforces that limit from this same accumulated total, and a limit enforced
854
- * with the wrong rate card stops a run early or late by the same unbounded
855
- * margin. A budget that is quietly wrong is worse than a budget that is
856
- * declined.
857
- *
858
- * Reachable, not decorative: a host that passes `pricing` and declares a chain
859
- * hits it on the first call. It costs `@namzu/cli` nothing, which passes no
860
- * pricing at all — its `/cost` already reports that the provider gave no price.
861
- *
862
- * The way out is per-member pricing, which needs a `CostInfo` that can sum over
863
- * heterogeneous rates. That is a public-type change and it is not this one.
864
- */
865
- function assertCostIsAttributable(
866
- chain: readonly ProviderChainMember[],
867
- pricing: ModelPricing | undefined,
868
- ): void {
869
- if (pricing === undefined || chain.length < 2) return
870
- throw new NamzuError({
871
- code: 'invalid_config',
872
- message:
873
- `A provider chain of ${chain.length} members was declared together with a single pricing table. ` +
874
- 'One table cannot price two members, so the run would report a total that is wrong by an unbounded ' +
875
- 'margin — and `runConfig.costLimitUsd` would be enforced against that same wrong total. ' +
876
- 'Either drop `pricing` (usage is still reported per model in the run) or declare one member.',
877
- details: { chainLength: chain.length },
878
- })
879
- }
880
-
881
- /**
882
- * Refuse a budget that cannot be measured.
883
- *
884
- * `runConfig.costLimitUsd` is enforced against `costInfo.totalCost`, and that
885
- * total only moves for tokens something has a rate for. A model no rate card
886
- * covers therefore produced a limit that could never trip — a host that set a
887
- * cost cap had no cost cap, and nothing said so. That was every run before the
888
- * price catalogue existed, which is how it went unnoticed.
889
- *
890
- * Refusing at the front is the cheap half of the answer: it costs the caller
891
- * nothing, fires before any spend, and names both ways out. The other half is
892
- * the `cost_unmeasurable` stop, for the models this cannot see — a step naming
893
- * its own, or a chain member declaring one.
894
- *
895
- * This is the same shape `advisory/budget.ts` already applies to
896
- * `AdvisoryBudget.maxCostPerRun`, one layer down, and for the same reason. The
897
- * run path simply never had it.
898
- */
899
- function assertBudgetIsMeasurable(params: QueryParams): void {
900
- const limit = params.runConfig.costLimitUsd
901
- if (limit === undefined || limit <= 0) return
902
- // A host-supplied table prices whatever it is pointed at, so a caller who
903
- // brought one has answered the question themselves.
904
- if (params.pricing !== undefined) return
905
- const model = params.runConfig.model
906
- if (resolveModelPricing(params.provider.id, model) !== undefined) return
907
-
908
- throw new NamzuError({
909
- code: 'invalid_config',
910
- message:
911
- `runConfig.costLimitUsd is set to ${limit}, but no rate is known for model "${model}" on ` +
912
- `provider "${params.provider.id}". The limit is enforced against the run's accumulated ` +
913
- 'cost, and tokens with no rate never reach that total — so the budget would read as ' +
914
- 'satisfied for the whole run and stop nothing. Either pass `pricing` to declare the rate ' +
915
- 'yourself, add the model to packages/sdk/src/pricing/rates.source.json, or drop ' +
916
- '`costLimitUsd` and bound the run with `tokenBudget`, which is measurable here.',
917
- details: { model, providerId: params.provider.id, costLimitUsd: limit },
918
- })
919
- }
920
-
921
- /**
922
- * Ask the driver what this model's window is, and never let the answer
923
- * cost the run.
924
- *
925
- * Three outcomes collapse to two here on purpose. No member and a resolved
926
- * `undefined` both mean "no answer" — the distinction matters to a driver
927
- * author, not to a caller about to fall through to the table. A rejection
928
- * is the third, and it is logged rather than propagated: a run that would
929
- * have worked on the table must not fail because a listing endpoint was
930
- * down.
931
- */
932
- async function resolveProviderContextWindow(
933
- provider: LLMProvider,
934
- model: string | undefined,
935
- signal: AbortSignal | undefined,
936
- timeoutMs: number,
937
- log: Logger,
938
- ): Promise<number | undefined> {
939
- if (!provider.resolveContextWindow || !model) return undefined
940
- if (signal?.aborted) return undefined
941
-
942
- // The resolver is an optional optimisation that runs before RunContext
943
- // owns its child controller. Give it a private deadline signal and fuse
944
- // caller cancellation into that transport in the safe direction: neither
945
- // outcome aborts the caller's controller. Passing a signal is necessary
946
- // but not sufficient, because a third-party driver can accept it and still
947
- // leave its promise pending; the race below makes fallback independent of
948
- // driver cooperation. Promise.race keeps the losing provider promise
949
- // observed, so a later rejection cannot become unhandled.
950
- const deadline = new AbortController()
951
- const resolverSignal = signal ? AbortSignal.any([signal, deadline.signal]) : deadline.signal
952
- const interrupted = Symbol('provider-context-window-interrupted')
953
- let onAbort: (() => void) | undefined
954
- const interruption = new Promise<typeof interrupted>((resolve) => {
955
- onAbort = () => resolve(interrupted)
956
- resolverSignal.addEventListener('abort', onAbort, { once: true })
957
- })
958
- // Direct QueryParams callers can supply a large run deadline. The clamp
959
- // avoids Node's >2^31-1 one-millisecond timer coercion during metadata lookup.
960
- // Metadata discovery remains optional and bounded even without a run deadline.
961
- const deadlineMs = timeoutMs === 0 ? 5_000 : Math.min(Math.max(0, timeoutMs), 2_147_483_647)
962
- const timer = setTimeout(() => {
963
- deadline.abort(new Error(`Provider context-window lookup exceeded ${deadlineMs}ms`))
964
- }, deadlineMs)
965
-
966
- try {
967
- const resolution = provider.resolveContextWindow(model, resolverSignal)
968
- const reported = await Promise.race([resolution, interruption])
969
- if (reported === interrupted) {
970
- if (deadline.signal.aborted) {
971
- log.debug('Provider context-window lookup timed out; using the table', {
972
- 'namzu.model.id': model,
973
- 'namzu.runtime.timeout_ms': deadlineMs,
974
- })
975
- }
976
- return undefined
977
- }
978
- return typeof reported === 'number' && reported > 0 ? reported : undefined
979
- } catch (err) {
980
- log.debug('Provider could not report a context window; using the table', {
981
- 'namzu.model.id': model,
982
- 'namzu.error.message': toErrorMessage(err),
983
- })
984
- return undefined
985
- } finally {
986
- clearTimeout(timer)
987
- if (onAbort) resolverSignal.removeEventListener('abort', onAbort)
988
- }
989
- }
990
-
991
- interface PendingHistoryRepairEvent {
992
- readonly source: 'fresh-history' | 'abandoned-checkpoint'
993
- readonly report: ToolHistoryRepairReport
994
- }
995
-
996
- /**
997
- * Project historical system messages exactly as a new run will persist them.
998
- *
999
- * Arbitrary historical prompt floors are rebuilt for this run and therefore
1000
- * never reach its provider-bound conversation. Repair must happen AFTER that
1001
- * removal: treating a soon-to-be-dropped system message as a tool-result
1002
- * boundary can replace an exact real result with an invented unknown outcome.
1003
- * The two state-bearing system forms survive; fresh inherited compaction is
1004
- * pinned until this run can prove it reconstructed equivalent state.
1005
- */
1006
- function projectStateBearingHistory(
1007
- messages: readonly Message[],
1008
- options: { readonly pinCompaction: boolean },
1009
- ): Message[] {
1010
- const projected: Message[] = []
1011
- for (const message of messages) {
1012
- if (message.role !== 'system') {
1013
- projected.push(message)
1014
- continue
1015
- }
1016
- if (isCompactionMessage(message.content)) {
1017
- projected.push(options.pinCompaction ? { ...message, retain: true } : message)
1018
- } else if (isWorkingMemoryMessage(message.content)) {
1019
- projected.push(message)
1020
- }
1021
- }
1022
- return collapseProjectInstructionSnapshots(projected)
1023
- }
1024
-
1025
812
  /**
1026
813
  * Remove the incomplete turn still owned by a durable resume plan.
1027
814
  *
@@ -1100,520 +887,32 @@ function withOwnedResumeOutcomes(
1100
887
  }
1101
888
 
1102
889
  export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run> {
1103
- assertMaxToolCalls(params.maxToolCalls)
1104
- // Required types do not protect JavaScript callers. Reject missing scope
1105
- // before opening a budget or persisting a run without its owning identity.
1106
- const missingFields = (['sessionId', 'topicId', 'projectId', 'tenantId'] as const).filter(
1107
- (field) => !params[field],
1108
- )
1109
- if (missingFields.length > 0) {
1110
- throw new NamzuError({
1111
- code: 'invalid_config',
1112
- message: `query requires sessionId, topicId, projectId, and tenantId; missing: ${missingFields.join(', ')}.`,
1113
- details: { missingFields },
1114
- })
1115
- }
1116
- const selectedResumeState = selectedResumeStates.get(params)
1117
- selectedResumeStates.delete(params)
1118
- // Resolved at the DOOR, before a run id exists or a logger is built.
1119
- // A caller who set both spellings of a renamed field has a config bug,
1120
- // and refusing it here costs them nothing; refusing it at the read site
1121
- // deep in the loop turns the same bug into a mid-run failure, after a
1122
- // provider call has been paid for and a partial transcript written.
1123
- const promptCache = params.promptCache
1124
- const taskScheduler = params.taskScheduler
1125
- const streamIdleTimeoutMs = resolveStreamIdleTimeoutMs(params.runConfig.streamIdleTimeoutMs)
1126
- const maxRequestRichContentBytes = resolveMaxRequestRichContentBytes(
1127
- params.runConfig.maxRequestRichContentBytes,
1128
- )
1129
- const sandboxTeardownTimeoutMs = resolveSandboxTeardownTimeoutMs(params.sandboxTeardownTimeoutMs)
1130
- // Persist the EFFECTIVE value, not only an override. A run replayed after a
1131
- // later release must be able to explain which liveness policy settled it;
1132
- // an absent field whose meaning follows the currently-installed default
1133
- // would rewrite that evidence at read time.
1134
- const runConfig: AgentRunConfig = {
1135
- ...params.runConfig,
1136
- streamIdleTimeoutMs,
1137
- maxRequestRichContentBytes,
1138
- }
1139
-
1140
- // The run's one correlated logger, built before anything below needs
1141
- // one — the migration check, the retry/fallback wrappers and `ctx`
1142
- // itself all read this SAME object, so a retry warning and the run
1143
- // record it retried for carry the identical `namzu.run.id` instead of
1144
- // three separate `getRootLogger()` reads that happened to agree by
1145
- // accident. `runId` is resolved here, once, rather than left to
1146
- // `build`'s own `config.runId ?? generateRunId()` fallback —
1147
- // generating it twice would silently hand the log and the run two
1148
- // different ids.
1149
- const runId = params.runId ?? generateRunId()
1150
- const budget = await resolveQueryBudget(params, runId, selectedResumeState)
1151
- const log = RunContextFactory.buildLogger({
1152
- agentName: params.agentName,
890
+ const prepared = await prepareRun(params)
891
+ const {
1153
892
  runConfig,
1154
- runId,
1155
- parentRunId: params.parentRunId,
1156
- sessionId: params.sessionId,
1157
- topicId: params.topicId,
1158
- projectId: params.projectId,
1159
- tenantId: params.tenantId,
1160
- })
1161
-
1162
- // Every model call in the run — the loop's turns, the forced-final
1163
- // summary, advisory and compaction side calls — goes through this one
1164
- // wrapped provider, so the retry policy cannot be bypassed by a code
1165
- // path that happens to hold the raw driver.
1166
- // The logger is passed on purpose: `withProviderRetry` guards every one
1167
- // of its warns behind `options.log`, and this is its only production
1168
- // call site — so without it the "failed, retrying" and "failed, giving
1169
- // up" lines were dead code and a backoff left no trace anywhere.
1170
- //
1171
- // With a chain declared, the same sentence holds two levels out. The idle
1172
- // watchdog is applied to each raw member, retry wraps that, and fallback
1173
- // wraps the members: `fallback(retry(idle(m0)), retry(idle(m1)), …)`. The
1174
- // idle layer cannot sit outside retry, because its timer would then count a
1175
- // legitimate backoff as provider silence. This order is not a
1176
- // preference. Assembled the other way round — which is what a host gets if
1177
- // it wraps its own chain and hands the result in, because this function
1178
- // would then wrap THAT in retry — an exhausted chain gets restarted from
1179
- // the head by the outer loop and a throttle on the last member is counted
1180
- // by two budgets. Building it here is what makes the order unspellable
1181
- // wrong.
1182
- const chain: readonly ProviderChainMember[] = [
1183
- { provider: params.provider },
1184
- ...(params.fallbackProviders ?? []),
1185
- ]
1186
- assertCostIsAttributable(chain, params.pricing)
1187
- assertBudgetIsMeasurable(params)
1188
- const withRecovery = (provider: LLMProvider): LLMProvider => {
1189
- const withIdleBound = withStreamIdleTimeout(provider, {
1190
- idleTimeoutMs: streamIdleTimeoutMs,
1191
- log,
1192
- })
1193
- const metered = withTokenBudget(withIdleBound, budget)
1194
- return params.retry === false
1195
- ? metered
1196
- : withProviderRetry(metered, {
1197
- config: params.retry,
1198
- log,
1199
- canRetry: () => budget.remaining > 0,
1200
- })
1201
- }
1202
- // Who is serving right now, for the run RECORD rather than for the request.
1203
- //
1204
- // It starts at the head and moves only when the chain does, which is the
1205
- // whole of the truth because the cursor never rewinds. The run cannot read
1206
- // this off `resilientProvider`: that wrapper reports the head's `id` on
1207
- // purpose, so asking it produces the declaration back — the defect this
1208
- // record exists to fix.
1209
- const serving: { current: ServingMember } = {
1210
- current: { index: 0, providerId: params.provider.id },
1211
- }
1212
- const resilientProvider = withProviderFallback(
1213
- chain.map((member) => ({
1214
- ...member,
1215
- provider: withRecovery(member.provider),
1216
- })),
1217
- {
1218
- log,
1219
- canFallback: () => budget.remaining > 0,
1220
- onSwap: (to) => {
1221
- serving.current = to
1222
- // `ctx` is declared below and is initialized before anything can
1223
- // call the provider: this fires from inside a `chatStream`, and
1224
- // the first one is issued by the loop that `ctx` is built for.
1225
- ctx.runMgr.setServingProvider(to.providerId)
1226
- },
1227
- },
1228
- )
1229
-
1230
- // Asked ONCE, here, before the loop exists. Both readers are synchronous
1231
- // and hot, so this can never move inside the iteration — and a driver
1232
- // that rejects, or one that hangs until the run is cancelled, must not
1233
- // take down a run the table could have served perfectly well. That is
1234
- // why the failure path is a swallow with a log rather than a throw: the
1235
- // window is an optimisation over a working default, not a prerequisite.
1236
- const providerContextWindow = await resolveProviderContextWindow(
1237
- resilientProvider,
1238
- runConfig.model,
1239
- params.signal,
1240
- runConfig.timeoutMs,
1241
- log,
1242
- )
1243
- const modelContextWindows = new Map<string, number | undefined>()
1244
- if (runConfig.model) modelContextWindows.set(runConfig.model, providerContextWindow)
1245
-
1246
- // The mode this conversation was left in, when the run config names none.
1247
- // Read once, before the loop exists, for the same reason the context
1248
- // window is: the executor's resolver is synchronous and hot.
1249
- //
1250
- // A store that throws is not a run failure — the run falls back to the
1251
- // config's answer, which is exactly what it did before this existed.
1252
- const topicState = params.topicStateStore
1253
- ? await params.topicStateStore
1254
- .getState(params.topicId, params.tenantId)
1255
- .catch((err: unknown) => {
1256
- log.debug('Could not read the topic state; using the run config', {
1257
- 'namzu.topic.id': params.topicId,
1258
- 'namzu.error.message': toErrorMessage(err),
1259
- })
1260
- return null
1261
- })
1262
- : null
1263
-
1264
- // Whatever a host left for "the next run", taken and cleared in one
1265
- // compare-and-set write. Prepended to the messages this run starts from,
1266
- // so it is in the FIRST request rather than arriving a turn late.
1267
- //
1268
- // Cleared as it is read: a queue read and cleared separately re-delivers
1269
- // on a crash between the two, and "start with this" arriving twice is a
1270
- // different instruction from the one that was left.
1271
- const queuedForThisRun: readonly Message[] = params.topicStateStore
1272
- ? await drainQueuedMessages(params.topicStateStore, params.topicId, params.tenantId).catch(
1273
- (err: unknown) => {
1274
- log.debug('Could not drain the topic queue; starting without it', {
1275
- 'namzu.topic.id': params.topicId,
1276
- 'namzu.error.message': toErrorMessage(err),
1277
- })
1278
- return []
1279
- },
1280
- )
1281
- : []
1282
-
1283
- // One effective list, used everywhere the run is seeded from. Three
1284
- // branches below push from it, and computing it at each would be three
1285
- // places to forget the queue.
1286
- //
1287
- // Stored attachments are resolved HERE, once, before the messages reach
1288
- // the run record. Resolving later — at the provider boundary — would put
1289
- // refs in the durable transcript and in every checkpoint, and a run
1290
- // resumed against a store that had since forgotten a ref would fail
1291
- // replaying its own history rather than at the moment somebody asked for
1292
- // the bytes. Every failure refuses: a message that silently lost its
1293
- // image is a model answering about a picture it never saw.
1294
- const seeded: Message[] =
1295
- queuedForThisRun.length > 0 ? [...queuedForThisRun, ...params.messages] : params.messages
1296
- let resolvedInitialMessages: Message[]
1297
- let attachmentResolutionCancelled = false
1298
- try {
1299
- resolvedInitialMessages = [
1300
- ...(await resolveAttachments(seeded, params.attachmentStore, {
1301
- signal: params.signal,
1302
- timeoutMs: params.attachmentResolveTimeoutMs,
1303
- })),
1304
- ]
1305
- params.signal?.throwIfAborted()
1306
- } catch (error) {
1307
- // Attachment materialization precedes RunContext construction so stored
1308
- // bytes never enter a live run's checkpoints. Cancellation still belongs
1309
- // to that run: preserve the exact input refs, build the context below, and
1310
- // let its normal terminal path classify/persist a cancelled Run. Every
1311
- // other store failure remains a pre-run refusal.
1312
- if (!params.signal?.aborted || error !== params.signal.reason) throw error
1313
- resolvedInitialMessages = [...seeded]
1314
- attachmentResolutionCancelled = true
1315
- }
1316
- if (!attachmentResolutionCancelled && params.projectInstructionContext?.prepareInitialSnapshot) {
1317
- const preparationSignal = params.signal ?? new AbortController().signal
1318
- let snapshot: UserMessage | null | undefined
1319
- try {
1320
- const prepared = await awaitProjectInstructionCallback(preparationSignal, () =>
1321
- params.projectInstructionContext?.prepareInitialSnapshot?.({
1322
- messages: [...resolvedInitialMessages],
1323
- signal: preparationSignal,
1324
- }),
1325
- )
1326
- // The callback promise can settle, remove its listener, and queue this
1327
- // continuation immediately before a queued abort. Publication is a
1328
- // separate authority boundary, so fence it too.
1329
- preparationSignal.throwIfAborted()
1330
- snapshot = prepared
1331
- } catch (error) {
1332
- // This callback runs before RunContext owns its child controller. A
1333
- // caller cancellation here still belongs to the run: publish no late
1334
- // snapshot and let the context below settle the normal cancelled Run.
1335
- // Compare the exact reason: a callback failure that won first must not
1336
- // be erased merely because cancellation arrived before this catch ran.
1337
- if (!preparationSignal.aborted || error !== preparationSignal.reason) throw error
1338
- }
1339
- if (snapshot !== undefined) {
1340
- resolvedInitialMessages = replaceProjectInstructionSnapshot(
1341
- resolvedInitialMessages,
1342
- snapshot,
1343
- 'before-latest-user',
1344
- )
1345
- }
1346
- }
1347
- const pendingHistoryRepairs: PendingHistoryRepairEvent[] = []
1348
- const projectedInitialMessages = collapseProjectInstructionSnapshots(
1349
- params.resumeFromCheckpoint || params.continuationMode
1350
- ? resolvedInitialMessages
1351
- : projectStateBearingHistory(resolvedInitialMessages, {
1352
- pinCompaction: true,
1353
- }),
1354
- )
1355
- const initialRepair = params.resumeFromCheckpoint
1356
- ? { messages: projectedInitialMessages, report: undefined }
1357
- : repairToolMessageHistory(projectedInitialMessages)
1358
- const initialMessages = initialRepair.messages
1359
- if (initialRepair.report && toolHistoryRepairChanged(initialRepair.report)) {
1360
- pendingHistoryRepairs.push({
1361
- source: 'fresh-history',
1362
- report: initialRepair.report,
1363
- })
1364
- log.warn('Repaired provider-invalid tool history before starting the run', {
1365
- [NAMZU.RUN_ID]: runId,
1366
- 'namzu.history.source': 'fresh-history',
1367
- 'namzu.history.duplicate_tool_results_removed':
1368
- initialRepair.report.duplicateToolResultsRemoved,
1369
- 'namzu.history.orphaned_tool_results_removed':
1370
- initialRepair.report.orphanedToolResultsRemoved,
1371
- 'namzu.history.synthetic_tool_results_inserted':
1372
- initialRepair.report.syntheticToolResultsInserted,
1373
- })
1374
- }
1375
-
1376
- const ctx = RunContextFactory.build({
1377
893
  budget,
1378
- ...(topicState ? { topicPermissionMode: topicState.permissionMode } : {}),
1379
- ...(params.permissionModeRef ? { permissionModeRef: params.permissionModeRef } : {}),
1380
- agentId: params.agentId,
1381
- agentName: params.agentName,
1382
- runConfig,
1383
- provider: resilientProvider,
1384
- workingDirectory: params.workingDirectory,
1385
- pricing: params.pricing,
1386
- enableActivityTracking: params.enableActivityTracking,
1387
- messages: initialMessages,
1388
- signal: params.signal,
1389
- sessionId: params.sessionId,
1390
- topicId: params.topicId,
1391
- projectId: params.projectId,
1392
- tenantId: params.tenantId,
1393
- pathBuilder: params.pathBuilder,
1394
- checkpointStore: params.checkpointStore,
1395
- runStore: params.runStore,
1396
- runId,
1397
- parentRunId: params.parentRunId,
1398
- depth: params.depth,
1399
894
  log,
1400
- })
1401
-
1402
- // Built here because the plan-approval closure below captures it, and
1403
- // its `emit` resolves `eventTranslator` at CALL time — the translator is
1404
- // a `const` some lines further down.
1405
- //
1406
- // The HANDOUT is therefore deliberately NOT here. A host given the box
1407
- // at this point can call `set` synchronously, `emit` reaches
1408
- // `eventTranslator` inside its temporal dead zone, and the run dies
1409
- // before it starts. That is not hypothetical: it is what the first
1410
- // version of this did, and the test that hands out the box and
1411
- // immediately swaps the policy is the one that found it.
1412
- const approvalPolicy = createRunApprovalPolicy({
1413
- runId: ctx.runId,
1414
- initial: {
1415
- // By identity against the default, not by presence. `resumeHandler`
1416
- // is REQUIRED on `QueryParams` — `drainQuery` substitutes
1417
- // `autoApproveHandler` before calling here — so "is it set" is
1418
- // always yes and would name every run `host`, including the ones
1419
- // approving everything unattended. Identity is what actually
1420
- // separates the two.
1421
- name:
1422
- params.approvalPolicyName ??
1423
- (params.resumeHandler === autoApproveHandler ? AUTO_APPROVE_POLICY_NAME : 'host'),
1424
- handler: params.resumeHandler,
1425
- },
1426
- emit: (event) => eventTranslator.emitEvent(event),
1427
- })
1428
-
1429
- const planApprovalIds = new Map<PlanId, CheckpointId>()
1430
- ctx.planManager.setApprovalHandler(async (request) => {
1431
- let checkpointId = planApprovalIds.get(request.planId)
1432
- if (!checkpointId) {
1433
- checkpointId = generateCheckpointId()
1434
- planApprovalIds.set(request.planId, checkpointId)
1435
- }
1436
- // `.current.handler`, never a captured `params.resumeHandler`. That
1437
- // capture is what made changing the policy mean ending the run.
1438
- const decision = await approvalPolicy.current.handler({
1439
- type: 'plan_approval',
1440
- runId: ctx.runId,
1441
- checkpointId,
1442
- plan: {
1443
- planId: request.planId,
1444
- title: request.title,
1445
- steps: request.steps.map((s, i) => ({
1446
- id: s.id,
1447
- description: s.description,
1448
- toolName: s.toolName,
1449
- agentId: s.agentId,
1450
- dependsOn: s.dependsOn,
1451
- order: s.order ?? i + 1,
1452
- })),
1453
- summary: request.summary,
1454
- },
1455
- })
1456
-
1457
- if (decision.action === 'approve_plan') {
1458
- // Optional approve-with-edits channel: the host may attach
1459
- // feedback to an approval. `PlanApprovalResponse.feedback`
1460
- // already exists on the type; threading it through lets the
1461
- // coordinator's approve_plan tool surface the user's edits in
1462
- // the same tool_result that unblocks the park. Bare approvals
1463
- // stay byte-identical (`{ approved: true }`).
1464
- return decision.feedback
1465
- ? { approved: true, feedback: decision.feedback }
1466
- : { approved: true }
1467
- }
1468
- if (decision.action === 'reject_plan') {
1469
- return { approved: false, feedback: decision.feedback }
1470
- }
1471
-
1472
- return { approved: false, feedback: `Action: ${decision.action}` }
1473
- })
1474
-
1475
- const eventTranslator = new EventTranslator(ctx.runMgr, undefined, ctx.log)
1476
- eventTranslator.wireActivityStore(ctx.activityStore, ctx.runId)
1477
- eventTranslator.wirePlanManager(ctx.planManager, ctx.runId)
1478
- eventTranslator.setGeneration(params.claimFence)
1479
- let interruptHooksStarted = false
1480
- const executeUserInterruptHooks = async (terminalError: unknown): Promise<void> => {
1481
- if (
1482
- interruptHooksStarted ||
1483
- !params.pluginManager ||
1484
- !isCallerAbortError(terminalError, ctx.abortController.signal) ||
1485
- params.parentRunId !== undefined ||
1486
- (params.depth ?? 0) !== 0 ||
1487
- cancelCauseOf(ctx.abortController.signal.reason) !== 'user'
1488
- ) {
1489
- return
1490
- }
1491
-
1492
- interruptHooksStarted = true
1493
- try {
1494
- // Deliberately omit the already-aborted run signal. The lifecycle
1495
- // manager still supplies each handler its own deadline signal, while
1496
- // `run_interrupt`'s observational fan-out prevents one result from
1497
- // suppressing the cleanup hooks that follow it.
1498
- await params.pluginManager.executeHooks(
1499
- 'run_interrupt',
1500
- { runId: ctx.runId, cancelCause: 'user' },
1501
- eventTranslator.emitEvent,
1502
- )
1503
- } catch (error) {
1504
- // Cancellation is the terminal authority. A hook event sink or an
1505
- // unexpected manager failure is reported, but cannot turn Stop into a
1506
- // failed run or prevent the durable cancellation verdict.
1507
- ctx.log.error('Run interrupt hooks did not settle cleanly', {
1508
- [NAMZU.RUN_ID]: ctx.runId,
1509
- ...errorAttributes(error),
1510
- })
1511
- }
1512
- }
895
+ ctx,
896
+ resilientProvider,
897
+ serving,
898
+ providerContextWindow,
899
+ modelContextWindows,
900
+ approvalPolicy,
901
+ eventTranslator,
902
+ executeUserInterruptHooks,
903
+ pendingHistoryRepairs,
904
+ initialMessages,
905
+ queuedForThisRun,
906
+ selectedResumeState,
907
+ attachmentResolutionCancelled,
908
+ streamIdleTimeoutMs,
909
+ sandboxTeardownTimeoutMs,
910
+ promptCache,
911
+ taskScheduler,
912
+ } = prepared
1513
913
 
1514
914
  if (attachmentResolutionCancelled) {
1515
- // Attachment materialization happens before RunContext exists. Once it
1516
- // observes cancellation, do only the work required to leave an honest
1517
- // durable run: initialize the record, retain the unresolved references,
1518
- // and settle through the ordinary cancellation classifier. Prompt
1519
- // contributions/cache, host callbacks, tools, plugins, sandbox, guardrails,
1520
- // advisors, and providers are all authority-bearing work and stay out.
1521
- // The dedicated root interrupt notification is the sole plugin exception:
1522
- // it runs after cancellation under its own deadline and cannot regain model
1523
- // or tool authority.
1524
- if (params.resumeFromCheckpoint && !selectedResumeState) {
1525
- // The canonical resume surface hands query the checkpoint state it
1526
- // already selected. A raw resume query has no such snapshot; after
1527
- // cancellation, reading the store again could hang without a signal,
1528
- // while persisting without it would erase the existing transcript.
1529
- // Refuse before binding/persisting rather than choose either failure.
1530
- ctx.abortController.signal.throwIfAborted()
1531
- }
1532
-
1533
- const cancelledPrompt = params.systemPrompt ?? ''
1534
- const cancelledAssembler = new ResultAssembler({
1535
- runMgr: ctx.runMgr,
1536
- planManager: ctx.planManager,
1537
- activityStore: ctx.activityStore,
1538
- log: ctx.log,
1539
- emitEvent: eventTranslator.emitEvent,
1540
- drainPending: () => eventTranslator.drainPending(),
1541
- signal: ctx.abortController.signal,
1542
- })
1543
- const rootSpan = getTracer().startSpan(
1544
- agentRunSpanName(params.agentName),
1545
- {},
1546
- parentContext(params.parentSpan ?? selectedResumeState?.traceContext),
1547
- )
1548
- rootSpan.setAttributes({
1549
- [NAMZU.RUN_ID]: ctx.runMgr.id,
1550
- [GENAI.AGENT_NAME]: params.agentName,
1551
- [GENAI.AGENT_ID]: params.agentId,
1552
- [GENAI.REQUEST_MODEL]: runConfig.model,
1553
- [GENAI.SYSTEM]: params.provider.id,
1554
- })
1555
-
1556
- try {
1557
- await ctx.runMgr.init()
1558
- if (selectedResumeState) {
1559
- ctx.runMgr.restoreUsage(
1560
- selectedResumeState.tokenUsage,
1561
- selectedResumeState.costInfo,
1562
- selectedResumeState.currentIteration,
1563
- )
1564
- for (const message of selectedResumeState.messages) ctx.runMgr.pushMessage(message)
1565
- for (const queued of queuedForThisRun) ctx.runMgr.pushMessage(queued)
1566
- } else if (params.continuationMode) {
1567
- for (const message of initialMessages) ctx.runMgr.pushMessage(message)
1568
- } else {
1569
- ctx.runMgr.pushMessage(createSystemMessage(cancelledPrompt, 'cache'))
1570
- for (const message of initialMessages) ctx.runMgr.pushMessage(message)
1571
- }
1572
- if (params.eventCursor) {
1573
- yield* catchUpFromCursor(
1574
- ctx.runMgr,
1575
- params.eventCursor,
1576
- params.onEventReplay,
1577
- params.claimFence,
1578
- (error) => {
1579
- ctx.log.warn('Replay observer failed after attachment cancellation', {
1580
- 'exception.message': toErrorMessage(error),
1581
- })
1582
- },
1583
- )
1584
- }
1585
- if (selectedResumeState) {
1586
- await eventTranslator.emitEvent({
1587
- type: 'run_resuming',
1588
- runId: ctx.runId,
1589
- fromCheckpointId: selectedResumeState.checkpointId,
1590
- })
1591
- yield* eventTranslator.drainPending()
1592
- }
1593
- ctx.runMgr.markRunning()
1594
- await eventTranslator.emitEvent({
1595
- type: 'run_started',
1596
- runId: ctx.runId,
1597
- systemPrompt: cancelledPrompt,
1598
- })
1599
- yield* eventTranslator.drainPending()
1600
- ctx.abortController.signal.throwIfAborted()
1601
- } catch (error) {
1602
- // Attachment resolution has already observed the caller's abort. A
1603
- // reconnect callback can still throw while replay is being reported,
1604
- // but it cannot replace that terminal cause or turn a cancelled run
1605
- // into an unpersisted rejection.
1606
- const terminalError = ctx.abortController.signal.aborted
1607
- ? ctx.abortController.signal.reason
1608
- : error
1609
- await executeUserInterruptHooks(terminalError)
1610
- yield* eventTranslator.drainPending()
1611
- yield* cancelledAssembler.handleError(terminalError, rootSpan)
1612
- } finally {
1613
- rootSpan.end()
1614
- }
1615
-
1616
- return await cancelledAssembler.finalize()
915
+ return yield* settlePreStartCancellation(params, prepared)
1617
916
  }
1618
917
 
1619
918
  const unsubscribeTaskStore = params.taskStore
@@ -2142,7 +1441,12 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2142
1441
 
2143
1442
  const tracer = getTracer()
2144
1443
 
2145
- return yield* (async function* (): AsyncGenerator<RunEvent, Run> {
1444
+ // Whether the run reached its settle. Read by the `finally` below, and
1445
+ // the only thing that distinguishes a run that finished from one whose
1446
+ // consumer walked away — see `settleAbandonedRun`.
1447
+ let settled = false
1448
+
1449
+ const runBody = (async function* (): AsyncGenerator<RunEvent, Run> {
2146
1450
  // Parent explicitly when a caller supplied one. Without this every
2147
1451
  // run starts its OWN root trace, so a supervisor delegating to three
2148
1452
  // children produced four disconnected traces instead of one tree —
@@ -2252,6 +1556,11 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2252
1556
  // Decided during checkpoint restore, executed after the sandbox
2253
1557
  // exists — the approved tools may well need it.
2254
1558
  let pendingResume: PendingResumePlan | null = null
1559
+ /**
1560
+ * The cadence park this resume answered, when the decision is one the
1561
+ * ordinary continue path carries out. See the restore path below.
1562
+ */
1563
+ let answeredParkId: CheckpointId | undefined
2255
1564
  /** Tool results recovered from the transcript; see the restore path. */
2256
1565
  let recoveredResults: ReadonlyMap<string, { result: string; isError: boolean }> = new Map()
2257
1566
  let emergencyManager: EmergencySaveManager | undefined
@@ -2407,6 +1716,49 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2407
1716
  ? planPendingResume(projectedCheckpoint, params.pendingDecision, ctx.log)
2408
1717
  : null
2409
1718
 
1719
+ // The park this resume ANSWERS even though there is no plan to
1720
+ // carry the decision out through.
1721
+ //
1722
+ // `planPendingResume` covers the two arms whose decision has to
1723
+ // reach something — the calls a `tool_review` park is about, the
1724
+ // tool a `user_question` park is inside. An `iteration_checkpoint`
1725
+ // park has neither, so it returns no plan, and the unpark further
1726
+ // down — which ran only when there was one — never fired for it.
1727
+ // A run that parked on the cadence, was resumed with
1728
+ // `{action: 'continue'}` and went on to finish its work therefore
1729
+ // kept reporting an OUTSTANDING park to `findPendingCheckpoint`,
1730
+ // so a second resume of the finished run was refused with
1731
+ // `awaiting-decision` for a decision already taken, and because
1732
+ // `prune` skips an unresolved park the row could no longer be
1733
+ // collected by anything.
1734
+ //
1735
+ // The decision IS carried out here — continuing is exactly what
1736
+ // the loop below does, and a plan verdict is the answer to the
1737
+ // question the plan park asked — so the park is resolved at the
1738
+ // same point and with the same meaning "resolved" carries
1739
+ // everywhere else: the record stays, and only its pending state
1740
+ // ends. A `pause` is deliberately not resolved: it holds the
1741
+ // park rather than answering it, which is how the live path
1742
+ // treats it too. `answersParkOf` is the whole map, park type to
1743
+ // answering decision, so an arm cannot go missing by being
1744
+ // absent from a condition again — which is how the plan arm
1745
+ // leaked a finished run's park.
1746
+ //
1747
+ // Resolving it does not depend on the resumed process being able
1748
+ // to act on it, and that is deliberate: the plan's own fate is a
1749
+ // separate defect (nothing restores a plan on the resume path at
1750
+ // all, so the new process has none to approve, execute or
1751
+ // reject) and making the resolution wait for it would leave the
1752
+ // row outstanding for exactly the runs that need it cleared.
1753
+ const parked = projectedCheckpoint.pending
1754
+ answeredParkId =
1755
+ params.pendingDecision &&
1756
+ parked !== undefined &&
1757
+ parked.resolvedAt === undefined &&
1758
+ answersParkOf(parked.request.type, params.pendingDecision)
1759
+ ? projectedCheckpoint.id
1760
+ : undefined
1761
+
2410
1762
  // Recover completed observations and explicitly unknown outcomes.
2411
1763
  // A recorded start is not proof that its external effect failed.
2412
1764
  const unanswered = interruptedToolCalls(projectedCheckpoint.messages)
@@ -2734,6 +2086,9 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2734
2086
  })
2735
2087
  }
2736
2088
  yield* resultAssembler.completeRun(rootSpan)
2089
+ // The run HAS settled, so the outer `finally` must not read
2090
+ // this as an abandonment — it would persist a second time.
2091
+ settled = true
2737
2092
  return await resultAssembler.finalize()
2738
2093
  }
2739
2094
  sandbox = acquisition.sandbox
@@ -2804,7 +2159,15 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2804
2159
  ctx.runMgr.setStopReason('input_guardrail')
2805
2160
  ctx.runMgr.setLastError(inputVerdict.reason ?? 'blocked by an input guardrail')
2806
2161
  yield* resultAssembler.completeRun(rootSpan)
2807
- return ctx.runMgr.getRun()
2162
+ // Same two lines as the sandbox path above, and for the same
2163
+ // reasons — with one that path does not have. This return used to
2164
+ // hand back `getRun()` without persisting, so the terminal state
2165
+ // reached the disk only because the abandonment path found
2166
+ // `settled` false and settled it a second time. A branch that
2167
+ // exists for runs which did NOT settle must not be the reason a
2168
+ // settled one is written down.
2169
+ settled = true
2170
+ return await resultAssembler.finalize()
2808
2171
  }
2809
2172
 
2810
2173
  // Honor the approval a human already gave, before the loop's
@@ -2830,149 +2193,59 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2830
2193
 
2831
2194
  await applyPendingResume(pendingResume, ctx.runMgr, toolExecutor, recoveredResults)
2832
2195
  yield* eventTranslator.drainPending()
2833
-
2834
- // The decision has now actually been carried out, so the park
2835
- // is no longer outstanding. Without this the checkpoint keeps
2836
- // reporting `pending` with no `resolvedAt`, and an approval
2837
- // queue re-serves a destructive call that already ran — which
2838
- // defeats the entire point of recording the park.
2839
- const resolvedCheckpointId = pendingResume.checkpointId
2840
- if (params.pendingDecision) {
2841
- await checkpointMgr
2842
- .unpark(resolvedCheckpointId, params.pendingDecision)
2843
- .catch((err: unknown) => {
2844
- ctx.log.error('Applied a pending decision but failed to clear the park', {
2845
- [NAMZU.RUN_ID]: ctx.runId,
2846
- 'namzu.checkpoint.id': resolvedCheckpointId,
2847
- 'exception.message': err instanceof Error ? err.message : String(err),
2848
- })
2849
- return null
2850
- })
2851
- }
2852
2196
  }
2853
2197
 
2854
- yield* iterationOrchestrator.runLoop()
2855
-
2856
- if (params.pluginManager) {
2857
- const hookResults = await params.pluginManager.executeHooks(
2858
- 'run_end',
2859
- { runId: ctx.runId, signal: ctx.abortController.signal },
2860
- eventTranslator.emitEvent,
2861
- )
2862
- applyLifecycleHookResults('run_end', hookResults)
2863
- yield* eventTranslator.drainPending()
2864
- // A delegated run says so once more, by name, so a hook that
2865
- // only cares when a subagent finishes need not read parent ids
2866
- // off every run_end.
2867
- if (params.parentRunId !== undefined) {
2868
- const stopResults = await params.pluginManager.executeHooks(
2869
- 'subagent_stop',
2870
- {
2871
- runId: ctx.runId,
2872
- parentRunId: params.parentRunId,
2873
- signal: ctx.abortController.signal,
2874
- },
2875
- eventTranslator.emitEvent,
2876
- )
2877
- applyLifecycleHookResults('subagent_stop', stopResults)
2878
- yield* eventTranslator.drainPending()
2879
- }
2880
- }
2881
-
2882
- // Hand the step record to the run before it settles, so the
2883
- // returned `Run` carries it.
2884
- ctx.runMgr.setSteps(iterationOrchestrator.getSteps())
2885
-
2886
- // Gates the FINAL result, not the stream — `text_delta` already
2887
- // reached the host as the model produced it. A rewrite is
2888
- // therefore a correction, and the event says so; buffering every
2889
- // token to gate the stream itself would trade the streaming UX
2890
- // for the guarantee, which is the host's call, not the SDK's.
2891
- if (params.outputGuardrails && params.outputGuardrails.length > 0) {
2892
- // Read what the run produced WITHOUT settling it. This used to
2893
- // call `markCompleted()` just to materialize the text, which
2894
- // force-marked a cancelled or paused run `completed` merely
2895
- // because a guardrail was configured — the presence of a
2896
- // safety check silently rewrote the run's own outcome.
2897
- const produced = ctx.runMgr.materializeResult()
2898
- const outputVerdict = await runOutputGuardrails(
2899
- params.outputGuardrails,
2900
- { runId: ctx.runId, output: produced, messages: ctx.runMgr.messages },
2901
- ctx.log,
2902
- )
2903
-
2904
- if (outputVerdict.blocked || outputVerdict.rewritten !== undefined) {
2905
- ctx.runMgr.clearStructuredOutput()
2906
- if (
2907
- params.structuredOutput &&
2908
- outputVerdict.rewritten !== undefined &&
2909
- ctx.runMgr.stopReason === 'end_turn'
2910
- )
2911
- ctx.runMgr.setStopReason('output_guardrail')
2912
- }
2913
-
2914
- if (outputVerdict.blocked) {
2915
- await eventTranslator.emitEvent({
2916
- type: 'guardrail_triggered',
2917
- runId: ctx.runId,
2918
- stage: 'output',
2919
- action: 'block',
2920
- ...(outputVerdict.name ? { guardrail: outputVerdict.name } : {}),
2921
- ...(outputVerdict.reason ? { reason: outputVerdict.reason } : {}),
2922
- })
2923
- yield* eventTranslator.drainPending()
2924
- // Same reasoning as the input-guardrail branch above.
2925
- await ctx.runMgr.recordAudit({
2926
- what: { action: 'guardrail:output', resource: outputVerdict.name },
2927
- outcome: 'refused',
2928
- reason: outputVerdict.reason ?? 'blocked by an output guardrail',
2929
- ...(params.persona?.identity.role ? { persona: params.persona.identity.role } : {}),
2930
- })
2931
- ctx.runMgr.setStopReason('output_guardrail')
2932
- ctx.runMgr.setLastError(outputVerdict.reason ?? 'blocked by an output guardrail')
2933
- ctx.runMgr.setResult('')
2934
- } else if (outputVerdict.rewritten !== undefined) {
2935
- await eventTranslator.emitEvent({
2936
- type: 'guardrail_triggered',
2937
- runId: ctx.runId,
2938
- stage: 'output',
2939
- action: 'rewrite',
2940
- ...(outputVerdict.name ? { guardrail: outputVerdict.name } : {}),
2941
- ...(outputVerdict.reason ? { reason: outputVerdict.reason } : {}),
2198
+ // The decision has now actually been carried out, so the park it
2199
+ // answered is no longer outstanding. Without this the checkpoint
2200
+ // keeps reporting `pending` with no `resolvedAt`, and an approval
2201
+ // queue re-serves a call that already ran — or a question already
2202
+ // answered — which defeats the entire point of recording the park.
2203
+ //
2204
+ // Two arms reach this point, and being outside `if (pendingResume)`
2205
+ // is what the second one needs. One is a plan whose decision was
2206
+ // applied to a batch above. The other is the cadence arm, for which
2207
+ // `planPendingResume` rightly produces no plan because the loop
2208
+ // resuming IS its decision being carried out (`answeredParkId`, set
2209
+ // on the restore path). Resolving only the first left a finished run
2210
+ // reporting `awaiting-decision` forever.
2211
+ //
2212
+ // What is RECORDED depends on which of the two produced the plan. A
2213
+ // recovery plan means the batch was answered by the crash path
2214
+ // rather than by the decision — the calls the human was asked about
2215
+ // were closed with explicitly unknown outcomes — so the human's
2216
+ // answer must not be written down as what ended the park. The park is
2217
+ // still resolved: the question is moot, and leaving it outstanding
2218
+ // would have `findPendingCheckpoint` serve it as the newest
2219
+ // outstanding park, so a host resuming it would rewind this run to
2220
+ // the checkpoint the crash happened on and re-execute a batch the run
2221
+ // has long since moved past.
2222
+ const resolvedCheckpointId = pendingResume?.checkpointId ?? answeredParkId
2223
+ const recordedDecision =
2224
+ pendingResume?.source === 'recovery' && params.pendingDecision
2225
+ ? supersededByRecovery(params.pendingDecision)
2226
+ : params.pendingDecision
2227
+ if (recordedDecision && resolvedCheckpointId) {
2228
+ await checkpointMgr.unpark(resolvedCheckpointId, recordedDecision).catch((err: unknown) => {
2229
+ ctx.log.error('Applied a pending decision but failed to clear the park', {
2230
+ [NAMZU.RUN_ID]: ctx.runId,
2231
+ 'namzu.checkpoint.id': resolvedCheckpointId,
2232
+ 'exception.message': err instanceof Error ? err.message : String(err),
2942
2233
  })
2943
- yield* eventTranslator.drainPending()
2944
- ctx.runMgr.setResult(outputVerdict.rewritten)
2945
- }
2946
- }
2947
-
2948
- if (params.consolidateInto && workingStateManager) {
2949
- const entry = consolidationEntry(workingStateManager.getState(), {
2950
- runId: ctx.runId,
2951
- at: Date.now(),
2234
+ return null
2952
2235
  })
2953
- if (entry) {
2954
- try {
2955
- const { entry: saved } = await params.consolidateInto.create(entry)
2956
- await eventTranslator.emitEvent({
2957
- type: 'memory_consolidated',
2958
- runId: ctx.runId,
2959
- memoryId: saved.id,
2960
- title: entry.title,
2961
- decisions: workingStateManager.getState().decisions.length,
2962
- discoveries: workingStateManager.getState().discoveries.length,
2963
- failures: workingStateManager.getState().failures.length,
2964
- })
2965
- yield* eventTranslator.drainPending()
2966
- } catch (error) {
2967
- ctx.log.warn('consolidation into the memory store failed', {
2968
- [NAMZU.RUN_ID]: ctx.runId,
2969
- 'namzu.memory.error': toErrorMessage(error),
2970
- })
2971
- }
2972
- }
2973
2236
  }
2974
- if (ctx.abortController.signal.aborted) ctx.runMgr.markCancelled()
2975
- yield* resultAssembler.completeRun(rootSpan)
2237
+
2238
+ yield* iterationOrchestrator.runLoop()
2239
+
2240
+ yield* finalizeRun({
2241
+ ctx,
2242
+ params,
2243
+ eventTranslator,
2244
+ takeSteps: () => iterationOrchestrator.getSteps(),
2245
+ workingStateManager,
2246
+ resultAssembler,
2247
+ rootSpan,
2248
+ })
2976
2249
  } catch (err) {
2977
2250
  // A failed run still spent its steps; report them.
2978
2251
  ctx.runMgr.setSteps(iterationOrchestrator.getSteps())
@@ -2980,109 +2253,129 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2980
2253
  yield* eventTranslator.drainPending()
2981
2254
  yield* resultAssembler.handleError(err, rootSpan)
2982
2255
  } finally {
2983
- // Release the process's termination path as soon as this run is
2984
- // done with it. Leaving the handlers installed would keep a
2985
- // WeakRef'd, settled run as the crash target for the rest of the
2986
- // process's life.
2987
- emergencyManager?.detach()
2988
-
2989
- // A background job outlives the tool call that started it — that
2990
- // is what it is for — so nothing but this stops it outliving the
2991
- // RUN. Scoped to this run's id: a shared registry serving several
2992
- // runs must not have one of them tear down another's work.
2993
- //
2994
- // Awaited, and its failure swallowed. A job that would not die is
2995
- // worth a log line, and is not worth retracting a run's answer.
2996
- unsubscribeJobExits?.()
2997
- // The wait-intent recorder listens on the same shared registry and
2998
- // leaks the same way if it is left attached.
2999
- awaitedJobs?.close()
3000
- // Only jobs bound to this run. Jobs a host bound to its session are
3001
- // the host's to stop, when the session ends.
3002
- if (params.backgroundJobs && (params.backgroundJobOwner ?? ctx.runId) === ctx.runId) {
3003
- try {
3004
- const stopped = await params.backgroundJobs.killOwner(ctx.runId)
3005
- if (stopped.length > 0) {
3006
- ctx.log.info('Background jobs stopped with the run', {
3007
- [NAMZU.RUN_ID]: ctx.runId,
3008
- 'namzu.jobs.stopped': stopped.length,
3009
- })
3010
- }
3011
- } catch (jobErr) {
3012
- ctx.log.error('A background job did not stop cleanly', {
3013
- [NAMZU.RUN_ID]: ctx.runId,
3014
- ...errorAttributes(jobErr),
3015
- })
3016
- }
3017
- }
3018
-
3019
- // Same reasoning for the question channel: the tools outlive the
3020
- // run that bound them, so leaving it attached would have a later
3021
- // run's question written into this run's checkpoint store.
3022
- questionParks.unbind()
3023
-
3024
- // Offer what the run learned to whoever decides what is worth
3025
- // keeping. In `finally` and awaited: a run that failed still
3026
- // discovered things, and a fire-and-forget write would race the
3027
- // process exiting on a one-shot CLI run. A throw here is
3028
- // swallowed — a memory that failed to form must not retract an
3029
- // answer that was already produced.
3030
- const candidate = memoryCandidateFor(ctx.runId, workingStateManager)
3031
- if (params.promoteMemory && candidate) {
3032
- try {
3033
- await params.promoteMemory(candidate)
3034
- } catch (promoteErr) {
3035
- ctx.log.error('Memory promotion threw — the run is unaffected', {
3036
- [NAMZU.RUN_ID]: ctx.runId,
3037
- 'exception.message':
3038
- promoteErr instanceof Error ? promoteErr.message : String(promoteErr),
3039
- })
3040
- }
3041
- }
3042
-
3043
- // --- Sandbox lifecycle: destroy after run ---
3044
- if (sandbox) {
3045
- const sandboxId = sandbox.id
3046
- const teardown = await teardownSandbox(sandbox, sandboxTeardownTimeoutMs)
3047
- if (teardown.kind === 'destroyed') {
3048
- await eventTranslator.emitEvent({
3049
- type: 'sandbox_destroyed',
3050
- runId: ctx.runId,
3051
- sandboxId,
3052
- })
3053
- yield* eventTranslator.drainPending()
3054
- ctx.log.info('Sandbox destroyed', { 'namzu.sandbox.id': sandboxId })
3055
- } else {
3056
- ctx.log.error('Sandbox destroy failed', {
3057
- 'namzu.sandbox.id': sandboxId,
3058
- ...errorAttributes(teardown.error),
3059
- })
3060
- }
3061
- }
3062
-
3063
- unsubscribeTaskStore?.()
3064
- // Keyed by HOW it settled, not just that it did: a run that was
3065
- // cancelled and a run that hit its budget have very different
3066
- // duration distributions, and averaging them together describes
3067
- // neither.
3068
- recordRunDuration(ctx.runMgr.getRun().status ?? 'unknown', Date.now() - runStartedAt)
3069
- rootSpan.end()
2256
+ yield* releaseRunResources({
2257
+ ctx,
2258
+ eventTranslator,
2259
+ emergencyManager,
2260
+ unsubscribeJobExits,
2261
+ unsubscribeTaskStore,
2262
+ awaitedJobs,
2263
+ backgroundJobs: params.backgroundJobs,
2264
+ backgroundJobOwner: params.backgroundJobOwner,
2265
+ questionParks,
2266
+ workingStateManager,
2267
+ promoteMemory: params.promoteMemory,
2268
+ sandbox,
2269
+ sandboxTeardownTimeoutMs,
2270
+ runStartedAt,
2271
+ rootSpan,
2272
+ })
3070
2273
  }
3071
2274
 
2275
+ // Reached only by a run that settled on its own terms. `finalize()` is
2276
+ // the only thing in this body that writes the durable half of the run,
2277
+ // and a `return` completion arriving from a consumer (`break` out of
2278
+ // `for await`, `gen.return()`) runs the `finally` above and stops short
2279
+ // of here. The flag is what tells the two apart, and this is one of
2280
+ // three sites that set it — the sandbox-acquisition and input-guardrail
2281
+ // returns settle early and set it there. Set before the await rather
2282
+ // than after, because a store that throws on the way out must not send
2283
+ // the abandonment path over the same broken ground.
2284
+ settled = true
3072
2285
  return await resultAssembler.finalize()
3073
2286
  })()
2287
+
2288
+ try {
2289
+ return yield* runBody
2290
+ } finally {
2291
+ if (!settled) await settleAbandonedRun(ctx.runMgr, ctx.log)
2292
+ }
3074
2293
  }
3075
2294
 
3076
2295
  /**
3077
- * Hand a returning consumer what it missed, or tell it why it cannot have it.
2296
+ * Write a terminal durable record for a run whose consumer walked away.
2297
+ *
2298
+ * `for await (… ) break` and an explicit `gen.return()` both end the run
2299
+ * body early. Everything the run's `finally` owns still happens — background
2300
+ * jobs are killed, the sandbox is destroyed, the span ends, the duration is
2301
+ * recorded — and then the generator stops. `finalize()` never runs, so
2302
+ * `persist()` never runs, and the store keeps whatever `init()` wrote: a
2303
+ * non-terminal status for a run that no longer exists. `deriveRunStatus`
2304
+ * reads that record back as `queued`, work waiting to start, and a host
2305
+ * rebuilding its view from the store believes it.
2306
+ *
2307
+ * There is nothing to emit here and nothing to emit it to: the consumer
2308
+ * that would have received the events is the one that left. This is about
2309
+ * the durable record only.
2310
+ *
2311
+ * `cancelled` is the verdict, and it is chosen from the existing vocabulary
2312
+ * because it is the one that is true. The run did not complete — no result
2313
+ * was produced and no terminal event was ever delivered — and nothing
2314
+ * failed, so `failed` would name an error that never happened; a run whose
2315
+ * consumer stopped reading and whose processes were torn down under it is
2316
+ * the same fact `markCancelled` already records when a run abort tears one
2317
+ * down. It needs no new `RunExecutionStatus` and no new `StopReason`.
3078
2318
  *
3079
- * Yields NOTHING on a refusal. A partial catch-up is the failure this exists to
3080
- * prevent: a consumer that receives some of the gap folds it into its state and
3081
- * cannot tell the state is wrong, where one that receives an explicit
3082
- * `unavailable` re-derives from the transcript and is right. The run continues
3083
- * either way — a stale cursor belongs to the client, and must not be able to
3084
- * stop the work.
2319
+ * A verdict the run already reached is left standing. A run that failed,
2320
+ * or was cancelled, before the consumer left still says so; what the
2321
+ * abandonment adds is that the record reaches the disk at all.
2322
+ *
2323
+ * Neither is a verdict written over a PARK. A park is a promise to a human
2324
+ * that outlives the consumer: the run is resumable and somebody is still owed
2325
+ * an answer, and `deriveRunStatus` reads a terminal status BEFORE it reads the
2326
+ * park — so recording `cancelled` turns `awaiting_hitl` into `cancelled` for a
2327
+ * run nobody answered for, while the unanswered question stays on the record
2328
+ * and the checkpoint it belongs to stays the place a resume starts from. The
2329
+ * durable state is asked rather than the in-memory one because the in-memory
2330
+ * one is the misleading half here: `handleHITLDecision` emits `run_paused` and
2331
+ * drains it BEFORE it calls `setStopReason('paused')`, so a consumer that
2332
+ * leaves on that event leaves a run whose status is `running` and whose stop
2333
+ * reason is unset at the exact instant its park is already durable.
2334
+ * `findPendingCheckpoint` is the same read an approval queue is built from,
2335
+ * expired parks included in its judgement: a park nobody answered in time is
2336
+ * not somebody still being asked.
2337
+ *
2338
+ * Never throws. It runs while an exception may already be unwinding, and a
2339
+ * store that cannot be written must not replace the run's real failure with
2340
+ * its own.
3085
2341
  */
2342
+ async function settleAbandonedRun(runMgr: RunPersistence, log: Logger): Promise<void> {
2343
+ try {
2344
+ // A terminal verdict is written whatever the park says: `deriveRunStatus`
2345
+ // settles a run that finished, failed or was cancelled BEFORE it looks at
2346
+ // a park ("terminal beats parked"), so a settled run is not waiting for
2347
+ // anybody and the row it already wrote must reach the disk. This ordering
2348
+ // is also what keeps a stale park from suppressing the write.
2349
+ if (!isTerminalStatus(runMgr.status)) {
2350
+ const parked = await findPendingCheckpoint(runMgr.getCheckpointStore(), runMgr.getRunScope())
2351
+ if (parked) {
2352
+ // Left exactly as it stands: no verdict, no write. The park row is
2353
+ // this run's durable state, and `persist()` here would add a
2354
+ // second claim — `running`, for a process that is gone — beside it.
2355
+ log.info('Abandoned run left parked for a human to answer', {
2356
+ [NAMZU.RUN_ID]: runMgr.id,
2357
+ 'namzu.checkpoint.id': parked.id,
2358
+ 'namzu.runtime.park_type': parked.pending?.request.type,
2359
+ })
2360
+ return
2361
+ }
2362
+ runMgr.markCancelled()
2363
+ }
2364
+ // Once: the `finally` that calls this runs once, and every site in the
2365
+ // run body that settles through `finalize()` sets `settled` before it
2366
+ // returns, so the two can never both write.
2367
+ await runMgr.persist()
2368
+ log.info('Abandoned run recorded as cancelled', {
2369
+ [NAMZU.RUN_ID]: runMgr.id,
2370
+ })
2371
+ } catch (err) {
2372
+ log.error('Failed to record the terminal state of an abandoned run', {
2373
+ [NAMZU.RUN_ID]: runMgr.id,
2374
+ 'exception.message': err instanceof Error ? err.message : String(err),
2375
+ })
2376
+ }
2377
+ }
2378
+
3086
2379
  /** The text of the newest user turn, which is what a prompt hook is asked about. */
3087
2380
  function lastUserPrompt(messages: readonly Message[]): string {
3088
2381
  for (let i = messages.length - 1; i >= 0; i--) {
@@ -3092,40 +2385,6 @@ function lastUserPrompt(messages: readonly Message[]): string {
3092
2385
  return ''
3093
2386
  }
3094
2387
 
3095
- async function* catchUpFromCursor(
3096
- runMgr: RunPersistence,
3097
- cursor: RunEventCursor,
3098
- onEventReplay: ((replay: RunEventReplay) => void) | undefined,
3099
- generation: FencingToken | undefined,
3100
- onReplayObserverError: (error: unknown) => void,
3101
- ): AsyncGenerator<RunEvent, void> {
3102
- const missed = await runMgr.getRunStore().readEvents({ sinceSeq: cursor.sinceSeq })
3103
- const replay = resolveRunEventReplay(
3104
- cursor,
3105
- {
3106
- lastSeq: runMgr.lastEventSeq,
3107
- ...(generation !== undefined ? { generation } : {}),
3108
- },
3109
- missed,
3110
- )
3111
-
3112
- if (onEventReplay) {
3113
- try {
3114
- // A callback typed `void` may still be implemented with `async` in
3115
- // TypeScript. Observe that runtime Promise so a late rejection cannot
3116
- // become process-wide, but never await host code here: replay delivery
3117
- // and an already-cancelled run must not inherit observer liveness.
3118
- const settlement = onEventReplay(replay)
3119
- void Promise.resolve(settlement).catch(onReplayObserverError)
3120
- } catch (error) {
3121
- onReplayObserverError(error)
3122
- }
3123
- }
3124
-
3125
- if (replay.status !== 'replayed') return
3126
- for (const event of replay.events) yield event
3127
- }
3128
-
3129
2388
  type DrainQueryParams = Omit<QueryParams, 'resumeHandler'> & {
3130
2389
  resumeHandler?: ResumeHandler
3131
2390
  }