@namzu/sdk 41.0.0 → 42.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. package/CHANGELOG.md +233 -0
  2. package/dist/agents/ReactiveAgent.d.ts.map +1 -1
  3. package/dist/agents/ReactiveAgent.js +3 -0
  4. package/dist/agents/ReactiveAgent.js.map +1 -1
  5. package/dist/agents/SupervisorAgent.d.ts.map +1 -1
  6. package/dist/agents/SupervisorAgent.js +11 -0
  7. package/dist/agents/SupervisorAgent.js.map +1 -1
  8. package/dist/agents/runAgent.d.ts +14 -0
  9. package/dist/agents/runAgent.d.ts.map +1 -1
  10. package/dist/agents/runAgent.js +3 -0
  11. package/dist/agents/runAgent.js.map +1 -1
  12. package/dist/manager/agent/lifecycle.d.ts.map +1 -1
  13. package/dist/manager/agent/lifecycle.js +20 -0
  14. package/dist/manager/agent/lifecycle.js.map +1 -1
  15. package/dist/manager/resident/outbox.d.ts +8 -8
  16. package/dist/manager/resident/store.d.ts +4 -4
  17. package/dist/public-runtime.d.ts +4 -1
  18. package/dist/public-runtime.d.ts.map +1 -1
  19. package/dist/public-runtime.js +13 -1
  20. package/dist/public-runtime.js.map +1 -1
  21. package/dist/public-tools.d.ts +1 -1
  22. package/dist/public-tools.d.ts.map +1 -1
  23. package/dist/public-tools.js +4 -2
  24. package/dist/public-tools.js.map +1 -1
  25. package/dist/registry/tool/execute.d.ts.map +1 -1
  26. package/dist/registry/tool/execute.js +10 -1
  27. package/dist/registry/tool/execute.js.map +1 -1
  28. package/dist/runtime/bidi/session.d.ts +11 -0
  29. package/dist/runtime/bidi/session.d.ts.map +1 -1
  30. package/dist/runtime/bidi/session.js +2 -0
  31. package/dist/runtime/bidi/session.js.map +1 -1
  32. package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
  33. package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
  34. package/dist/runtime/query/cancelled-before-start.js +152 -0
  35. package/dist/runtime/query/cancelled-before-start.js.map +1 -0
  36. package/dist/runtime/query/checkpoint.d.ts +21 -0
  37. package/dist/runtime/query/checkpoint.d.ts.map +1 -1
  38. package/dist/runtime/query/checkpoint.js +23 -0
  39. package/dist/runtime/query/checkpoint.js.map +1 -1
  40. package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
  41. package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
  42. package/dist/runtime/query/executor/tool-call-admission.js +373 -0
  43. package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
  44. package/dist/runtime/query/executor.d.ts +76 -35
  45. package/dist/runtime/query/executor.d.ts.map +1 -1
  46. package/dist/runtime/query/executor.js +52 -380
  47. package/dist/runtime/query/executor.js.map +1 -1
  48. package/dist/runtime/query/finalize-run.d.ts +55 -0
  49. package/dist/runtime/query/finalize-run.d.ts.map +1 -0
  50. package/dist/runtime/query/finalize-run.js +113 -0
  51. package/dist/runtime/query/finalize-run.js.map +1 -0
  52. package/dist/runtime/query/guardrail-presets.d.ts +187 -1
  53. package/dist/runtime/query/guardrail-presets.d.ts.map +1 -1
  54. package/dist/runtime/query/guardrail-presets.js +298 -0
  55. package/dist/runtime/query/guardrail-presets.js.map +1 -1
  56. package/dist/runtime/query/index.d.ts +18 -9
  57. package/dist/runtime/query/index.d.ts.map +1 -1
  58. package/dist/runtime/query/index.js +241 -893
  59. package/dist/runtime/query/index.js.map +1 -1
  60. package/dist/runtime/query/iteration/index.d.ts +6 -161
  61. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  62. package/dist/runtime/query/iteration/index.js +23 -523
  63. package/dist/runtime/query/iteration/index.js.map +1 -1
  64. package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
  65. package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
  66. package/dist/runtime/query/iteration/outstanding-work.js +365 -0
  67. package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
  68. package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
  69. package/dist/runtime/query/iteration/phases/plan.js +13 -2
  70. package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
  71. package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
  72. package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
  73. package/dist/runtime/query/iteration/step-shaping.js +184 -0
  74. package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
  75. package/dist/runtime/query/prepare-run.d.ts +94 -0
  76. package/dist/runtime/query/prepare-run.d.ts.map +1 -0
  77. package/dist/runtime/query/prepare-run.js +589 -0
  78. package/dist/runtime/query/prepare-run.js.map +1 -0
  79. package/dist/runtime/query/release-run.d.ts +56 -0
  80. package/dist/runtime/query/release-run.d.ts.map +1 -0
  81. package/dist/runtime/query/release-run.js +101 -0
  82. package/dist/runtime/query/release-run.js.map +1 -0
  83. package/dist/runtime/query/resume-pending.d.ts +112 -1
  84. package/dist/runtime/query/resume-pending.d.ts.map +1 -1
  85. package/dist/runtime/query/resume-pending.js +133 -0
  86. package/dist/runtime/query/resume-pending.js.map +1 -1
  87. package/dist/runtime/query/tooling.d.ts +2 -0
  88. package/dist/runtime/query/tooling.d.ts.map +1 -1
  89. package/dist/runtime/query/tooling.js +3 -0
  90. package/dist/runtime/query/tooling.js.map +1 -1
  91. package/dist/store/evidence/compaction-archive.d.ts +2 -2
  92. package/dist/tools/coordinator/agent.d.ts.map +1 -1
  93. package/dist/tools/coordinator/agent.js +17 -2
  94. package/dist/tools/coordinator/agent.js.map +1 -1
  95. package/dist/tools/coordinator/index.d.ts.map +1 -1
  96. package/dist/tools/coordinator/index.js +17 -3
  97. package/dist/tools/coordinator/index.js.map +1 -1
  98. package/dist/tools/untrusted-envelope.d.ts +35 -0
  99. package/dist/tools/untrusted-envelope.d.ts.map +1 -1
  100. package/dist/tools/untrusted-envelope.js +91 -3
  101. package/dist/tools/untrusted-envelope.js.map +1 -1
  102. package/dist/types/agent/base.d.ts +23 -0
  103. package/dist/types/agent/base.d.ts.map +1 -1
  104. package/dist/types/agent/task.d.ts +19 -0
  105. package/dist/types/agent/task.d.ts.map +1 -1
  106. package/dist/types/run/config.d.ts +12 -5
  107. package/dist/types/run/config.d.ts.map +1 -1
  108. package/dist/types/tool/index.d.ts +19 -0
  109. package/dist/types/tool/index.d.ts.map +1 -1
  110. package/dist/types/tool/index.js.map +1 -1
  111. package/package.json +1 -1
  112. package/src/agents/ReactiveAgent.ts +3 -0
  113. package/src/agents/SupervisorAgent.ts +11 -0
  114. package/src/agents/runAgent.ts +18 -0
  115. package/src/manager/agent/lifecycle.ts +22 -0
  116. package/src/public-runtime.ts +14 -0
  117. package/src/public-tools.ts +8 -2
  118. package/src/registry/tool/execute.ts +9 -1
  119. package/src/runtime/bidi/session.ts +13 -0
  120. package/src/runtime/query/cancelled-before-start.ts +189 -0
  121. package/src/runtime/query/checkpoint.ts +22 -0
  122. package/src/runtime/query/executor/tool-call-admission.ts +473 -0
  123. package/src/runtime/query/executor.ts +76 -442
  124. package/src/runtime/query/finalize-run.ts +192 -0
  125. package/src/runtime/query/guardrail-presets.ts +356 -0
  126. package/src/runtime/query/index.ts +287 -1011
  127. package/src/runtime/query/iteration/index.ts +40 -586
  128. package/src/runtime/query/iteration/outstanding-work.ts +386 -0
  129. package/src/runtime/query/iteration/phases/plan.ts +18 -2
  130. package/src/runtime/query/iteration/step-shaping.ts +271 -0
  131. package/src/runtime/query/prepare-run.ts +718 -0
  132. package/src/runtime/query/release-run.ts +168 -0
  133. package/src/runtime/query/resume-pending.ts +158 -0
  134. package/src/runtime/query/tooling.ts +5 -0
  135. package/src/tools/coordinator/agent.ts +17 -2
  136. package/src/tools/coordinator/index.ts +17 -3
  137. package/src/tools/untrusted-envelope.ts +94 -3
  138. package/src/types/agent/base.ts +24 -0
  139. package/src/types/agent/task.ts +20 -0
  140. package/src/types/run/config.ts +12 -5
  141. package/src/types/tool/index.ts +20 -0
@@ -6,14 +6,8 @@ import {
6
6
  TriggerEvaluator,
7
7
  assertBudgetEnforceable,
8
8
  } from '../../advisory/index.js'
9
- import { drainQueuedMessages } from '../../agents/handle.js'
10
9
  import { AuthorizationGate } from '../../authorization/gate.js'
11
- import { consolidationEntry } from '../../compaction/consolidation.js'
12
- import {
13
- type ToolHistoryRepairReport,
14
- repairToolMessageHistory,
15
- toolHistoryRepairChanged,
16
- } from '../../compaction/dangling.js'
10
+ import { repairToolMessageHistory, toolHistoryRepairChanged } from '../../compaction/dangling.js'
17
11
  import { extractFromUserMessage } from '../../compaction/extractor.js'
18
12
  import { WorkingStateManager } from '../../compaction/manager.js'
19
13
  import type { ContextReducer } from '../../compaction/reducer.js'
@@ -23,21 +17,14 @@ import { type CompactionConfig, CompactionConfigSchema } from '../../config/runt
23
17
  import { TOOL_OUTPUT_DIR_NAME } from '../../constants/tools/index.js'
24
18
  import { EmergencySaveManager } from '../../manager/run/emergency.js'
25
19
  import type { RunPersistence } from '../../manager/run/persistence.js'
26
- import { resolveModelPricing } from '../../pricing/index.js'
27
20
  import { PromptContributionRegistry } from '../../prompt/contributions.js'
28
21
  import { resolveProviderCapabilities } from '../../provider/capabilities.js'
29
- import { isCallerAbortError } from '../../provider/errors.js'
30
- import {
31
- type ProviderChainMember,
32
- type ServingMember,
33
- withProviderFallback,
34
- } from '../../provider/fallback.js'
35
- import { resolveStreamIdleTimeoutMs, withStreamIdleTimeout } from '../../provider/idle-timeout.js'
36
- import { type ProviderRetryConfig, withProviderRetry } from '../../provider/retry.js'
22
+ import type { ProviderChainMember } from '../../provider/fallback.js'
23
+ import { withStreamIdleTimeout } from '../../provider/idle-timeout.js'
24
+ import type { ProviderRetryConfig } from '../../provider/retry.js'
37
25
  import { withTokenBudget } from '../../provider/token-budget.js'
38
26
  import type { TokenBudget } from '../../run/token-budget.js'
39
27
  import type { PathBuilder } from '../../session/workspace/path-builder.js'
40
- import { resolveAttachments } from '../../store/attachment/index.js'
41
28
  import {
42
29
  GENAI,
43
30
  NAMZU,
@@ -45,8 +32,6 @@ import {
45
32
  parentContext,
46
33
  serializeSpan,
47
34
  } from '../../telemetry/attributes.js'
48
- import type { SerializedSpanContext } from '../../telemetry/attributes.js'
49
- import { recordRunDuration } from '../../telemetry/metrics.js'
50
35
  import { getTracer } from '../../telemetry/runtime-accessors.js'
51
36
  import { buildAdvisoryTools } from '../../tools/advisory/index.js'
52
37
  import { SearchToolsTool } from '../../tools/builtins/search-tools.js'
@@ -60,6 +45,7 @@ import type { AgentRuntimeContext, RuntimeToolOverrides } from '../../types/agen
60
45
  import type { AgentContextLevel } from '../../types/agent/factory.js'
61
46
  import type { WorkingMemoryProvider } from '../../types/agent/working-memory.js'
62
47
  import type { AuthorizationGateConfig } from '../../types/authorization/index.js'
48
+ import { isTerminalStatus } from '../../types/common/index.js'
63
49
  import { NamzuError } from '../../types/errors/index.js'
64
50
  import type { InputGuardrailSpec, OutputGuardrailSpec } from '../../types/guardrail/index.js'
65
51
  import {
@@ -67,23 +53,20 @@ import {
67
53
  type ResumeHandler,
68
54
  autoApproveHandler,
69
55
  } from '../../types/hitl/index.js'
70
- import type { CheckpointId, PlanId, RunId, SessionId, TenantId } from '../../types/ids/index.js'
56
+ import type { CheckpointId, RunId, SessionId, TenantId } from '../../types/ids/index.js'
71
57
  import type { InvocationState } from '../../types/invocation/index.js'
72
58
  import type { MemoryStore } from '../../types/memory/index.js'
73
59
  import {
74
60
  type AssistantMessage,
75
61
  type Message,
76
- type UserMessage,
77
62
  createSystemMessage,
78
63
  } from '../../types/message/index.js'
79
64
  import type { AgentPersona } from '../../types/persona/index.js'
80
65
  import type { LLMProvider } from '../../types/provider/index.js'
81
66
  import type { TaskRouterConfig } from '../../types/router/index.js'
82
67
  import type { ReviewAnswer } from '../../types/run/answer-review.js'
83
- import { cancelCauseOf } from '../../types/run/cancel-cause.js'
84
68
  import type { CheckpointStore, FencingToken } from '../../types/run/checkpoint-store.js'
85
69
  import type { RunEventCursor, RunEventReplay } from '../../types/run/event-cursor.js'
86
- import { resolveRunEventReplay } from '../../types/run/event-cursor.js'
87
70
  import type {
88
71
  AgentRunConfig,
89
72
  BeforeStep,
@@ -95,8 +78,6 @@ import type {
95
78
  StopCondition,
96
79
  } from '../../types/run/index.js'
97
80
  import type { PromoteMemory } from '../../types/run/memory-promotion.js'
98
- import { memoryCandidateFor } from '../../types/run/memory-promotion.js'
99
- import type { RunState } from '../../types/run/state.js'
100
81
  import type { RunStore } from '../../types/run/store.js'
101
82
  import type { TokenBudgetStore } from '../../types/run/token-budget-store.js'
102
83
  import type { Sandbox, SandboxProvider } from '../../types/sandbox/index.js'
@@ -109,50 +90,46 @@ import type { RepairToolCall } from '../../types/tool/repair.js'
109
90
  import type { BackoffPolicy } from '../../utils/backoff.js'
110
91
  import type { ModelPricing } from '../../utils/cost.js'
111
92
  import { toErrorMessage } from '../../utils/error.js'
112
- import { generateCheckpointId, generateRunId } from '../../utils/id.js'
113
93
  import { errorAttributes } from '../../utils/log/exception.js'
114
94
  import type { Logger } from '../../utils/logger.js'
115
95
  import { AwaitedJobs } from '../jobs/awaited-jobs.js'
116
96
  import type { BackgroundJobRegistry } from '../jobs/registry.js'
117
- import { AUTO_APPROVE_POLICY_NAME, createRunApprovalPolicy } from './approval-policy.js'
118
- import { CheckpointManager } from './checkpoint.js'
119
- import { RunContextFactory } from './context.js'
120
- import { EventTranslator } from './events.js'
97
+ import { catchUpFromCursor, settlePreStartCancellation } from './cancelled-before-start.js'
98
+ import { CheckpointManager, findPendingCheckpoint } from './checkpoint.js'
99
+ import { finalizeRun } from './finalize-run.js'
121
100
  import { GuardCoordinator } from './guard.js'
122
- import { runInputGuardrails, runOutputGuardrails } from './guardrails.js'
101
+ import { runInputGuardrails } from './guardrails.js'
123
102
  import { IterationOrchestrator } from './iteration/index.js'
124
103
  import { isCompactionMessage } from './iteration/phases/compaction.js'
125
104
  import { isWorkingMemoryMessage } from './iteration/phases/working-memory.js'
126
105
  import { applyLifecycleHookResults } from './plugin-hooks.js'
127
106
  import {
128
- type ProjectInstructionContext,
129
- awaitProjectInstructionCallback,
130
- collapseProjectInstructionSnapshots,
131
- replaceProjectInstructionSnapshot,
132
- } from './project-instructions.js'
107
+ type SelectedResumeState,
108
+ prepareRun,
109
+ projectStateBearingHistory,
110
+ resolveProviderContextWindow,
111
+ selectedResumeStates,
112
+ } from './prepare-run.js'
113
+ import type { ProjectInstructionContext } from './project-instructions.js'
133
114
  import type { PromptCache } from './prompt-cache.js'
134
115
  import { PromptBuilder } from './prompt.js'
135
116
  import type { PromptSegments } from './prompt.js'
136
117
  import { PendingAnswers, QuestionParkBinding } from './question-park.js'
118
+ import { releaseRunResources } from './release-run.js'
137
119
  import { RepeatCallTracker } from './repeat-call.js'
138
- import { resolveMaxRequestRichContentBytes } from './request-rich-content.js'
139
120
  import { ResultAssembler } from './result.js'
140
121
  import {
141
122
  type PendingResumePlan,
123
+ answersParkOf,
142
124
  applyPendingResume,
143
125
  interruptedToolCalls,
144
126
  planCrashResume,
145
127
  planPendingResume,
146
128
  recoverCompletedCalls,
129
+ supersededByRecovery,
147
130
  } from './resume-pending.js'
148
- import {
149
- acquireSandbox,
150
- resolveSandboxTeardownTimeoutMs,
151
- teardownSandbox,
152
- } from './sandbox-lifecycle.js'
131
+ import { acquireSandbox } from './sandbox-lifecycle.js'
153
132
  import { SteeringBinding, type SteeringChannel, isOperatorUserMessage } from './steering.js'
154
- import { resolveQueryBudget } from './token-budget.js'
155
- import { assertMaxToolCalls } from './tool-call-budget.js'
156
133
  import { ToolGrantSet } from './tool-grants.js'
157
134
  import { createToolPause } from './tool-pause.js'
158
135
  import { ToolingBootstrap } from './tooling.js'
@@ -363,6 +340,20 @@ export interface QueryParams {
363
340
  * agent decides the rest is worth re-reading. Set `0` to disable.
364
341
  */
365
342
  maxToolOutputChars?: number
343
+ /**
344
+ * Screens to run against every tool result, where the registry was not
345
+ * built with its own.
346
+ *
347
+ * This is the run's half of a boundary whose only other door is the
348
+ * registry constructor — and a registry is usually the HOST's, assembled
349
+ * before the run exists, so a run-config option is the only way a run
350
+ * screens a registry it did not build. A registry built WITH
351
+ * `resultGuardrails` states its own policy and wins, `[]` included.
352
+ *
353
+ * Absent installs {@link DEFAULT_TOOL_RESULT_GUARDRAILS}; an empty array
354
+ * installs none, which is how a caller turns the default off.
355
+ */
356
+ toolResultGuardrails?: readonly import('../../types/guardrail/index.js').ToolResultGuardrailSpec[]
366
357
  /**
367
358
  * Smaller preview for text that exceeded maxToolOutputChars, after its full
368
359
  * host output and integrity manifest have been saved. Unset/0 keeps the old
@@ -818,196 +809,6 @@ export interface QueryParams {
818
809
  strictCapabilities?: boolean
819
810
  }
820
811
 
821
- type SelectedResumeState = RunState & {
822
- readonly checkpointId: CheckpointId
823
- readonly traceContext?: SerializedSpanContext
824
- }
825
- const selectedResumeStates = new WeakMap<QueryParams, SelectedResumeState>()
826
-
827
- /**
828
- * Refuse to price a run whose tokens two differently-priced members may produce.
829
- *
830
- * `RunPersistence` holds ONE {@link ModelPricing} table and applies it to every
831
- * accumulation regardless of which model produced the tokens. Across a swap that
832
- * makes `costInfo.totalCost` wrong by an unbounded margin, and silently — the
833
- * number keeps the shape of an answer. `CostInfo` cannot express the truth
834
- * either: it carries `inputCostPer1M` / `outputCostPer1M`, and there is no
835
- * honest value for those once a total spans two rate cards.
836
- *
837
- * So the total is refused rather than blended. Naming what that costs is part
838
- * of the refusal, because the caller loses `costLimitUsd` with it: the guard
839
- * enforces that limit from this same accumulated total, and a limit enforced
840
- * with the wrong rate card stops a run early or late by the same unbounded
841
- * margin. A budget that is quietly wrong is worse than a budget that is
842
- * declined.
843
- *
844
- * Reachable, not decorative: a host that passes `pricing` and declares a chain
845
- * hits it on the first call. It costs `@namzu/cli` nothing, which passes no
846
- * pricing at all — its `/cost` already reports that the provider gave no price.
847
- *
848
- * The way out is per-member pricing, which needs a `CostInfo` that can sum over
849
- * heterogeneous rates. That is a public-type change and it is not this one.
850
- */
851
- function assertCostIsAttributable(
852
- chain: readonly ProviderChainMember[],
853
- pricing: ModelPricing | undefined,
854
- ): void {
855
- if (pricing === undefined || chain.length < 2) return
856
- throw new NamzuError({
857
- code: 'invalid_config',
858
- message:
859
- `A provider chain of ${chain.length} members was declared together with a single pricing table. ` +
860
- 'One table cannot price two members, so the run would report a total that is wrong by an unbounded ' +
861
- 'margin — and `runConfig.costLimitUsd` would be enforced against that same wrong total. ' +
862
- 'Either drop `pricing` (usage is still reported per model in the run) or declare one member.',
863
- details: { chainLength: chain.length },
864
- })
865
- }
866
-
867
- /**
868
- * Refuse a budget that cannot be measured.
869
- *
870
- * `runConfig.costLimitUsd` is enforced against `costInfo.totalCost`, and that
871
- * total only moves for tokens something has a rate for. A model no rate card
872
- * covers therefore produced a limit that could never trip — a host that set a
873
- * cost cap had no cost cap, and nothing said so. That was every run before the
874
- * price catalogue existed, which is how it went unnoticed.
875
- *
876
- * Refusing at the front is the cheap half of the answer: it costs the caller
877
- * nothing, fires before any spend, and names both ways out. The other half is
878
- * the `cost_unmeasurable` stop, for the models this cannot see — a step naming
879
- * its own, or a chain member declaring one.
880
- *
881
- * This is the same shape `advisory/budget.ts` already applies to
882
- * `AdvisoryBudget.maxCostPerRun`, one layer down, and for the same reason. The
883
- * run path simply never had it.
884
- */
885
- function assertBudgetIsMeasurable(params: QueryParams): void {
886
- const limit = params.runConfig.costLimitUsd
887
- if (limit === undefined || limit <= 0) return
888
- // A host-supplied table prices whatever it is pointed at, so a caller who
889
- // brought one has answered the question themselves.
890
- if (params.pricing !== undefined) return
891
- const model = params.runConfig.model
892
- if (resolveModelPricing(params.provider.id, model) !== undefined) return
893
-
894
- throw new NamzuError({
895
- code: 'invalid_config',
896
- message:
897
- `runConfig.costLimitUsd is set to ${limit}, but no rate is known for model "${model}" on ` +
898
- `provider "${params.provider.id}". The limit is enforced against the run's accumulated ` +
899
- 'cost, and tokens with no rate never reach that total — so the budget would read as ' +
900
- 'satisfied for the whole run and stop nothing. Either pass `pricing` to declare the rate ' +
901
- 'yourself, add the model to packages/sdk/src/pricing/rates.source.json, or drop ' +
902
- '`costLimitUsd` and bound the run with `tokenBudget`, which is measurable here.',
903
- details: { model, providerId: params.provider.id, costLimitUsd: limit },
904
- })
905
- }
906
-
907
- /**
908
- * Ask the driver what this model's window is, and never let the answer
909
- * cost the run.
910
- *
911
- * Three outcomes collapse to two here on purpose. No member and a resolved
912
- * `undefined` both mean "no answer" — the distinction matters to a driver
913
- * author, not to a caller about to fall through to the table. A rejection
914
- * is the third, and it is logged rather than propagated: a run that would
915
- * have worked on the table must not fail because a listing endpoint was
916
- * down.
917
- */
918
- async function resolveProviderContextWindow(
919
- provider: LLMProvider,
920
- model: string | undefined,
921
- signal: AbortSignal | undefined,
922
- timeoutMs: number,
923
- log: Logger,
924
- ): Promise<number | undefined> {
925
- if (!provider.resolveContextWindow || !model) return undefined
926
- if (signal?.aborted) return undefined
927
-
928
- // The resolver is an optional optimisation that runs before RunContext
929
- // owns its child controller. Give it a private deadline signal and fuse
930
- // caller cancellation into that transport in the safe direction: neither
931
- // outcome aborts the caller's controller. Passing a signal is necessary
932
- // but not sufficient, because a third-party driver can accept it and still
933
- // leave its promise pending; the race below makes fallback independent of
934
- // driver cooperation. Promise.race keeps the losing provider promise
935
- // observed, so a later rejection cannot become unhandled.
936
- const deadline = new AbortController()
937
- const resolverSignal = signal ? AbortSignal.any([signal, deadline.signal]) : deadline.signal
938
- const interrupted = Symbol('provider-context-window-interrupted')
939
- let onAbort: (() => void) | undefined
940
- const interruption = new Promise<typeof interrupted>((resolve) => {
941
- onAbort = () => resolve(interrupted)
942
- resolverSignal.addEventListener('abort', onAbort, { once: true })
943
- })
944
- // Direct QueryParams callers can supply a large run deadline. The clamp
945
- // avoids Node's >2^31-1 one-millisecond timer coercion during metadata lookup.
946
- // Metadata discovery remains optional and bounded even without a run deadline.
947
- const deadlineMs = timeoutMs === 0 ? 5_000 : Math.min(Math.max(0, timeoutMs), 2_147_483_647)
948
- const timer = setTimeout(() => {
949
- deadline.abort(new Error(`Provider context-window lookup exceeded ${deadlineMs}ms`))
950
- }, deadlineMs)
951
-
952
- try {
953
- const resolution = provider.resolveContextWindow(model, resolverSignal)
954
- const reported = await Promise.race([resolution, interruption])
955
- if (reported === interrupted) {
956
- if (deadline.signal.aborted) {
957
- log.debug('Provider context-window lookup timed out; using the table', {
958
- 'namzu.model.id': model,
959
- 'namzu.runtime.timeout_ms': deadlineMs,
960
- })
961
- }
962
- return undefined
963
- }
964
- return typeof reported === 'number' && reported > 0 ? reported : undefined
965
- } catch (err) {
966
- log.debug('Provider could not report a context window; using the table', {
967
- 'namzu.model.id': model,
968
- 'namzu.error.message': toErrorMessage(err),
969
- })
970
- return undefined
971
- } finally {
972
- clearTimeout(timer)
973
- if (onAbort) resolverSignal.removeEventListener('abort', onAbort)
974
- }
975
- }
976
-
977
- interface PendingHistoryRepairEvent {
978
- readonly source: 'fresh-history' | 'abandoned-checkpoint'
979
- readonly report: ToolHistoryRepairReport
980
- }
981
-
982
- /**
983
- * Project historical system messages exactly as a new run will persist them.
984
- *
985
- * Arbitrary historical prompt floors are rebuilt for this run and therefore
986
- * never reach its provider-bound conversation. Repair must happen AFTER that
987
- * removal: treating a soon-to-be-dropped system message as a tool-result
988
- * boundary can replace an exact real result with an invented unknown outcome.
989
- * The two state-bearing system forms survive; fresh inherited compaction is
990
- * pinned until this run can prove it reconstructed equivalent state.
991
- */
992
- function projectStateBearingHistory(
993
- messages: readonly Message[],
994
- options: { readonly pinCompaction: boolean },
995
- ): Message[] {
996
- const projected: Message[] = []
997
- for (const message of messages) {
998
- if (message.role !== 'system') {
999
- projected.push(message)
1000
- continue
1001
- }
1002
- if (isCompactionMessage(message.content)) {
1003
- projected.push(options.pinCompaction ? { ...message, retain: true } : message)
1004
- } else if (isWorkingMemoryMessage(message.content)) {
1005
- projected.push(message)
1006
- }
1007
- }
1008
- return collapseProjectInstructionSnapshots(projected)
1009
- }
1010
-
1011
812
  /**
1012
813
  * Remove the incomplete turn still owned by a durable resume plan.
1013
814
  *
@@ -1086,520 +887,32 @@ function withOwnedResumeOutcomes(
1086
887
  }
1087
888
 
1088
889
  export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run> {
1089
- assertMaxToolCalls(params.maxToolCalls)
1090
- // Required types do not protect JavaScript callers. Reject missing scope
1091
- // before opening a budget or persisting a run without its owning identity.
1092
- const missingFields = (['sessionId', 'topicId', 'projectId', 'tenantId'] as const).filter(
1093
- (field) => !params[field],
1094
- )
1095
- if (missingFields.length > 0) {
1096
- throw new NamzuError({
1097
- code: 'invalid_config',
1098
- message: `query requires sessionId, topicId, projectId, and tenantId; missing: ${missingFields.join(', ')}.`,
1099
- details: { missingFields },
1100
- })
1101
- }
1102
- const selectedResumeState = selectedResumeStates.get(params)
1103
- selectedResumeStates.delete(params)
1104
- // Resolved at the DOOR, before a run id exists or a logger is built.
1105
- // A caller who set both spellings of a renamed field has a config bug,
1106
- // and refusing it here costs them nothing; refusing it at the read site
1107
- // deep in the loop turns the same bug into a mid-run failure, after a
1108
- // provider call has been paid for and a partial transcript written.
1109
- const promptCache = params.promptCache
1110
- const taskScheduler = params.taskScheduler
1111
- const streamIdleTimeoutMs = resolveStreamIdleTimeoutMs(params.runConfig.streamIdleTimeoutMs)
1112
- const maxRequestRichContentBytes = resolveMaxRequestRichContentBytes(
1113
- params.runConfig.maxRequestRichContentBytes,
1114
- )
1115
- const sandboxTeardownTimeoutMs = resolveSandboxTeardownTimeoutMs(params.sandboxTeardownTimeoutMs)
1116
- // Persist the EFFECTIVE value, not only an override. A run replayed after a
1117
- // later release must be able to explain which liveness policy settled it;
1118
- // an absent field whose meaning follows the currently-installed default
1119
- // would rewrite that evidence at read time.
1120
- const runConfig: AgentRunConfig = {
1121
- ...params.runConfig,
1122
- streamIdleTimeoutMs,
1123
- maxRequestRichContentBytes,
1124
- }
1125
-
1126
- // The run's one correlated logger, built before anything below needs
1127
- // one — the migration check, the retry/fallback wrappers and `ctx`
1128
- // itself all read this SAME object, so a retry warning and the run
1129
- // record it retried for carry the identical `namzu.run.id` instead of
1130
- // three separate `getRootLogger()` reads that happened to agree by
1131
- // accident. `runId` is resolved here, once, rather than left to
1132
- // `build`'s own `config.runId ?? generateRunId()` fallback —
1133
- // generating it twice would silently hand the log and the run two
1134
- // different ids.
1135
- const runId = params.runId ?? generateRunId()
1136
- const budget = await resolveQueryBudget(params, runId, selectedResumeState)
1137
- const log = RunContextFactory.buildLogger({
1138
- agentName: params.agentName,
890
+ const prepared = await prepareRun(params)
891
+ const {
1139
892
  runConfig,
1140
- runId,
1141
- parentRunId: params.parentRunId,
1142
- sessionId: params.sessionId,
1143
- topicId: params.topicId,
1144
- projectId: params.projectId,
1145
- tenantId: params.tenantId,
1146
- })
1147
-
1148
- // Every model call in the run — the loop's turns, the forced-final
1149
- // summary, advisory and compaction side calls — goes through this one
1150
- // wrapped provider, so the retry policy cannot be bypassed by a code
1151
- // path that happens to hold the raw driver.
1152
- // The logger is passed on purpose: `withProviderRetry` guards every one
1153
- // of its warns behind `options.log`, and this is its only production
1154
- // call site — so without it the "failed, retrying" and "failed, giving
1155
- // up" lines were dead code and a backoff left no trace anywhere.
1156
- //
1157
- // With a chain declared, the same sentence holds two levels out. The idle
1158
- // watchdog is applied to each raw member, retry wraps that, and fallback
1159
- // wraps the members: `fallback(retry(idle(m0)), retry(idle(m1)), …)`. The
1160
- // idle layer cannot sit outside retry, because its timer would then count a
1161
- // legitimate backoff as provider silence. This order is not a
1162
- // preference. Assembled the other way round — which is what a host gets if
1163
- // it wraps its own chain and hands the result in, because this function
1164
- // would then wrap THAT in retry — an exhausted chain gets restarted from
1165
- // the head by the outer loop and a throttle on the last member is counted
1166
- // by two budgets. Building it here is what makes the order unspellable
1167
- // wrong.
1168
- const chain: readonly ProviderChainMember[] = [
1169
- { provider: params.provider },
1170
- ...(params.fallbackProviders ?? []),
1171
- ]
1172
- assertCostIsAttributable(chain, params.pricing)
1173
- assertBudgetIsMeasurable(params)
1174
- const withRecovery = (provider: LLMProvider): LLMProvider => {
1175
- const withIdleBound = withStreamIdleTimeout(provider, {
1176
- idleTimeoutMs: streamIdleTimeoutMs,
1177
- log,
1178
- })
1179
- const metered = withTokenBudget(withIdleBound, budget)
1180
- return params.retry === false
1181
- ? metered
1182
- : withProviderRetry(metered, {
1183
- config: params.retry,
1184
- log,
1185
- canRetry: () => budget.remaining > 0,
1186
- })
1187
- }
1188
- // Who is serving right now, for the run RECORD rather than for the request.
1189
- //
1190
- // It starts at the head and moves only when the chain does, which is the
1191
- // whole of the truth because the cursor never rewinds. The run cannot read
1192
- // this off `resilientProvider`: that wrapper reports the head's `id` on
1193
- // purpose, so asking it produces the declaration back — the defect this
1194
- // record exists to fix.
1195
- const serving: { current: ServingMember } = {
1196
- current: { index: 0, providerId: params.provider.id },
1197
- }
1198
- const resilientProvider = withProviderFallback(
1199
- chain.map((member) => ({
1200
- ...member,
1201
- provider: withRecovery(member.provider),
1202
- })),
1203
- {
1204
- log,
1205
- canFallback: () => budget.remaining > 0,
1206
- onSwap: (to) => {
1207
- serving.current = to
1208
- // `ctx` is declared below and is initialized before anything can
1209
- // call the provider: this fires from inside a `chatStream`, and
1210
- // the first one is issued by the loop that `ctx` is built for.
1211
- ctx.runMgr.setServingProvider(to.providerId)
1212
- },
1213
- },
1214
- )
1215
-
1216
- // Asked ONCE, here, before the loop exists. Both readers are synchronous
1217
- // and hot, so this can never move inside the iteration — and a driver
1218
- // that rejects, or one that hangs until the run is cancelled, must not
1219
- // take down a run the table could have served perfectly well. That is
1220
- // why the failure path is a swallow with a log rather than a throw: the
1221
- // window is an optimisation over a working default, not a prerequisite.
1222
- const providerContextWindow = await resolveProviderContextWindow(
1223
- resilientProvider,
1224
- runConfig.model,
1225
- params.signal,
1226
- runConfig.timeoutMs,
1227
- log,
1228
- )
1229
- const modelContextWindows = new Map<string, number | undefined>()
1230
- if (runConfig.model) modelContextWindows.set(runConfig.model, providerContextWindow)
1231
-
1232
- // The mode this conversation was left in, when the run config names none.
1233
- // Read once, before the loop exists, for the same reason the context
1234
- // window is: the executor's resolver is synchronous and hot.
1235
- //
1236
- // A store that throws is not a run failure — the run falls back to the
1237
- // config's answer, which is exactly what it did before this existed.
1238
- const topicState = params.topicStateStore
1239
- ? await params.topicStateStore
1240
- .getState(params.topicId, params.tenantId)
1241
- .catch((err: unknown) => {
1242
- log.debug('Could not read the topic state; using the run config', {
1243
- 'namzu.topic.id': params.topicId,
1244
- 'namzu.error.message': toErrorMessage(err),
1245
- })
1246
- return null
1247
- })
1248
- : null
1249
-
1250
- // Whatever a host left for "the next run", taken and cleared in one
1251
- // compare-and-set write. Prepended to the messages this run starts from,
1252
- // so it is in the FIRST request rather than arriving a turn late.
1253
- //
1254
- // Cleared as it is read: a queue read and cleared separately re-delivers
1255
- // on a crash between the two, and "start with this" arriving twice is a
1256
- // different instruction from the one that was left.
1257
- const queuedForThisRun: readonly Message[] = params.topicStateStore
1258
- ? await drainQueuedMessages(params.topicStateStore, params.topicId, params.tenantId).catch(
1259
- (err: unknown) => {
1260
- log.debug('Could not drain the topic queue; starting without it', {
1261
- 'namzu.topic.id': params.topicId,
1262
- 'namzu.error.message': toErrorMessage(err),
1263
- })
1264
- return []
1265
- },
1266
- )
1267
- : []
1268
-
1269
- // One effective list, used everywhere the run is seeded from. Three
1270
- // branches below push from it, and computing it at each would be three
1271
- // places to forget the queue.
1272
- //
1273
- // Stored attachments are resolved HERE, once, before the messages reach
1274
- // the run record. Resolving later — at the provider boundary — would put
1275
- // refs in the durable transcript and in every checkpoint, and a run
1276
- // resumed against a store that had since forgotten a ref would fail
1277
- // replaying its own history rather than at the moment somebody asked for
1278
- // the bytes. Every failure refuses: a message that silently lost its
1279
- // image is a model answering about a picture it never saw.
1280
- const seeded: Message[] =
1281
- queuedForThisRun.length > 0 ? [...queuedForThisRun, ...params.messages] : params.messages
1282
- let resolvedInitialMessages: Message[]
1283
- let attachmentResolutionCancelled = false
1284
- try {
1285
- resolvedInitialMessages = [
1286
- ...(await resolveAttachments(seeded, params.attachmentStore, {
1287
- signal: params.signal,
1288
- timeoutMs: params.attachmentResolveTimeoutMs,
1289
- })),
1290
- ]
1291
- params.signal?.throwIfAborted()
1292
- } catch (error) {
1293
- // Attachment materialization precedes RunContext construction so stored
1294
- // bytes never enter a live run's checkpoints. Cancellation still belongs
1295
- // to that run: preserve the exact input refs, build the context below, and
1296
- // let its normal terminal path classify/persist a cancelled Run. Every
1297
- // other store failure remains a pre-run refusal.
1298
- if (!params.signal?.aborted || error !== params.signal.reason) throw error
1299
- resolvedInitialMessages = [...seeded]
1300
- attachmentResolutionCancelled = true
1301
- }
1302
- if (!attachmentResolutionCancelled && params.projectInstructionContext?.prepareInitialSnapshot) {
1303
- const preparationSignal = params.signal ?? new AbortController().signal
1304
- let snapshot: UserMessage | null | undefined
1305
- try {
1306
- const prepared = await awaitProjectInstructionCallback(preparationSignal, () =>
1307
- params.projectInstructionContext?.prepareInitialSnapshot?.({
1308
- messages: [...resolvedInitialMessages],
1309
- signal: preparationSignal,
1310
- }),
1311
- )
1312
- // The callback promise can settle, remove its listener, and queue this
1313
- // continuation immediately before a queued abort. Publication is a
1314
- // separate authority boundary, so fence it too.
1315
- preparationSignal.throwIfAborted()
1316
- snapshot = prepared
1317
- } catch (error) {
1318
- // This callback runs before RunContext owns its child controller. A
1319
- // caller cancellation here still belongs to the run: publish no late
1320
- // snapshot and let the context below settle the normal cancelled Run.
1321
- // Compare the exact reason: a callback failure that won first must not
1322
- // be erased merely because cancellation arrived before this catch ran.
1323
- if (!preparationSignal.aborted || error !== preparationSignal.reason) throw error
1324
- }
1325
- if (snapshot !== undefined) {
1326
- resolvedInitialMessages = replaceProjectInstructionSnapshot(
1327
- resolvedInitialMessages,
1328
- snapshot,
1329
- 'before-latest-user',
1330
- )
1331
- }
1332
- }
1333
- const pendingHistoryRepairs: PendingHistoryRepairEvent[] = []
1334
- const projectedInitialMessages = collapseProjectInstructionSnapshots(
1335
- params.resumeFromCheckpoint || params.continuationMode
1336
- ? resolvedInitialMessages
1337
- : projectStateBearingHistory(resolvedInitialMessages, {
1338
- pinCompaction: true,
1339
- }),
1340
- )
1341
- const initialRepair = params.resumeFromCheckpoint
1342
- ? { messages: projectedInitialMessages, report: undefined }
1343
- : repairToolMessageHistory(projectedInitialMessages)
1344
- const initialMessages = initialRepair.messages
1345
- if (initialRepair.report && toolHistoryRepairChanged(initialRepair.report)) {
1346
- pendingHistoryRepairs.push({
1347
- source: 'fresh-history',
1348
- report: initialRepair.report,
1349
- })
1350
- log.warn('Repaired provider-invalid tool history before starting the run', {
1351
- [NAMZU.RUN_ID]: runId,
1352
- 'namzu.history.source': 'fresh-history',
1353
- 'namzu.history.duplicate_tool_results_removed':
1354
- initialRepair.report.duplicateToolResultsRemoved,
1355
- 'namzu.history.orphaned_tool_results_removed':
1356
- initialRepair.report.orphanedToolResultsRemoved,
1357
- 'namzu.history.synthetic_tool_results_inserted':
1358
- initialRepair.report.syntheticToolResultsInserted,
1359
- })
1360
- }
1361
-
1362
- const ctx = RunContextFactory.build({
1363
893
  budget,
1364
- ...(topicState ? { topicPermissionMode: topicState.permissionMode } : {}),
1365
- ...(params.permissionModeRef ? { permissionModeRef: params.permissionModeRef } : {}),
1366
- agentId: params.agentId,
1367
- agentName: params.agentName,
1368
- runConfig,
1369
- provider: resilientProvider,
1370
- workingDirectory: params.workingDirectory,
1371
- pricing: params.pricing,
1372
- enableActivityTracking: params.enableActivityTracking,
1373
- messages: initialMessages,
1374
- signal: params.signal,
1375
- sessionId: params.sessionId,
1376
- topicId: params.topicId,
1377
- projectId: params.projectId,
1378
- tenantId: params.tenantId,
1379
- pathBuilder: params.pathBuilder,
1380
- checkpointStore: params.checkpointStore,
1381
- runStore: params.runStore,
1382
- runId,
1383
- parentRunId: params.parentRunId,
1384
- depth: params.depth,
1385
894
  log,
1386
- })
1387
-
1388
- // Built here because the plan-approval closure below captures it, and
1389
- // its `emit` resolves `eventTranslator` at CALL time — the translator is
1390
- // a `const` some lines further down.
1391
- //
1392
- // The HANDOUT is therefore deliberately NOT here. A host given the box
1393
- // at this point can call `set` synchronously, `emit` reaches
1394
- // `eventTranslator` inside its temporal dead zone, and the run dies
1395
- // before it starts. That is not hypothetical: it is what the first
1396
- // version of this did, and the test that hands out the box and
1397
- // immediately swaps the policy is the one that found it.
1398
- const approvalPolicy = createRunApprovalPolicy({
1399
- runId: ctx.runId,
1400
- initial: {
1401
- // By identity against the default, not by presence. `resumeHandler`
1402
- // is REQUIRED on `QueryParams` — `drainQuery` substitutes
1403
- // `autoApproveHandler` before calling here — so "is it set" is
1404
- // always yes and would name every run `host`, including the ones
1405
- // approving everything unattended. Identity is what actually
1406
- // separates the two.
1407
- name:
1408
- params.approvalPolicyName ??
1409
- (params.resumeHandler === autoApproveHandler ? AUTO_APPROVE_POLICY_NAME : 'host'),
1410
- handler: params.resumeHandler,
1411
- },
1412
- emit: (event) => eventTranslator.emitEvent(event),
1413
- })
1414
-
1415
- const planApprovalIds = new Map<PlanId, CheckpointId>()
1416
- ctx.planManager.setApprovalHandler(async (request) => {
1417
- let checkpointId = planApprovalIds.get(request.planId)
1418
- if (!checkpointId) {
1419
- checkpointId = generateCheckpointId()
1420
- planApprovalIds.set(request.planId, checkpointId)
1421
- }
1422
- // `.current.handler`, never a captured `params.resumeHandler`. That
1423
- // capture is what made changing the policy mean ending the run.
1424
- const decision = await approvalPolicy.current.handler({
1425
- type: 'plan_approval',
1426
- runId: ctx.runId,
1427
- checkpointId,
1428
- plan: {
1429
- planId: request.planId,
1430
- title: request.title,
1431
- steps: request.steps.map((s, i) => ({
1432
- id: s.id,
1433
- description: s.description,
1434
- toolName: s.toolName,
1435
- agentId: s.agentId,
1436
- dependsOn: s.dependsOn,
1437
- order: s.order ?? i + 1,
1438
- })),
1439
- summary: request.summary,
1440
- },
1441
- })
1442
-
1443
- if (decision.action === 'approve_plan') {
1444
- // Optional approve-with-edits channel: the host may attach
1445
- // feedback to an approval. `PlanApprovalResponse.feedback`
1446
- // already exists on the type; threading it through lets the
1447
- // coordinator's approve_plan tool surface the user's edits in
1448
- // the same tool_result that unblocks the park. Bare approvals
1449
- // stay byte-identical (`{ approved: true }`).
1450
- return decision.feedback
1451
- ? { approved: true, feedback: decision.feedback }
1452
- : { approved: true }
1453
- }
1454
- if (decision.action === 'reject_plan') {
1455
- return { approved: false, feedback: decision.feedback }
1456
- }
1457
-
1458
- return { approved: false, feedback: `Action: ${decision.action}` }
1459
- })
1460
-
1461
- const eventTranslator = new EventTranslator(ctx.runMgr, undefined, ctx.log)
1462
- eventTranslator.wireActivityStore(ctx.activityStore, ctx.runId)
1463
- eventTranslator.wirePlanManager(ctx.planManager, ctx.runId)
1464
- eventTranslator.setGeneration(params.claimFence)
1465
- let interruptHooksStarted = false
1466
- const executeUserInterruptHooks = async (terminalError: unknown): Promise<void> => {
1467
- if (
1468
- interruptHooksStarted ||
1469
- !params.pluginManager ||
1470
- !isCallerAbortError(terminalError, ctx.abortController.signal) ||
1471
- params.parentRunId !== undefined ||
1472
- (params.depth ?? 0) !== 0 ||
1473
- cancelCauseOf(ctx.abortController.signal.reason) !== 'user'
1474
- ) {
1475
- return
1476
- }
1477
-
1478
- interruptHooksStarted = true
1479
- try {
1480
- // Deliberately omit the already-aborted run signal. The lifecycle
1481
- // manager still supplies each handler its own deadline signal, while
1482
- // `run_interrupt`'s observational fan-out prevents one result from
1483
- // suppressing the cleanup hooks that follow it.
1484
- await params.pluginManager.executeHooks(
1485
- 'run_interrupt',
1486
- { runId: ctx.runId, cancelCause: 'user' },
1487
- eventTranslator.emitEvent,
1488
- )
1489
- } catch (error) {
1490
- // Cancellation is the terminal authority. A hook event sink or an
1491
- // unexpected manager failure is reported, but cannot turn Stop into a
1492
- // failed run or prevent the durable cancellation verdict.
1493
- ctx.log.error('Run interrupt hooks did not settle cleanly', {
1494
- [NAMZU.RUN_ID]: ctx.runId,
1495
- ...errorAttributes(error),
1496
- })
1497
- }
1498
- }
895
+ ctx,
896
+ resilientProvider,
897
+ serving,
898
+ providerContextWindow,
899
+ modelContextWindows,
900
+ approvalPolicy,
901
+ eventTranslator,
902
+ executeUserInterruptHooks,
903
+ pendingHistoryRepairs,
904
+ initialMessages,
905
+ queuedForThisRun,
906
+ selectedResumeState,
907
+ attachmentResolutionCancelled,
908
+ streamIdleTimeoutMs,
909
+ sandboxTeardownTimeoutMs,
910
+ promptCache,
911
+ taskScheduler,
912
+ } = prepared
1499
913
 
1500
914
  if (attachmentResolutionCancelled) {
1501
- // Attachment materialization happens before RunContext exists. Once it
1502
- // observes cancellation, do only the work required to leave an honest
1503
- // durable run: initialize the record, retain the unresolved references,
1504
- // and settle through the ordinary cancellation classifier. Prompt
1505
- // contributions/cache, host callbacks, tools, plugins, sandbox, guardrails,
1506
- // advisors, and providers are all authority-bearing work and stay out.
1507
- // The dedicated root interrupt notification is the sole plugin exception:
1508
- // it runs after cancellation under its own deadline and cannot regain model
1509
- // or tool authority.
1510
- if (params.resumeFromCheckpoint && !selectedResumeState) {
1511
- // The canonical resume surface hands query the checkpoint state it
1512
- // already selected. A raw resume query has no such snapshot; after
1513
- // cancellation, reading the store again could hang without a signal,
1514
- // while persisting without it would erase the existing transcript.
1515
- // Refuse before binding/persisting rather than choose either failure.
1516
- ctx.abortController.signal.throwIfAborted()
1517
- }
1518
-
1519
- const cancelledPrompt = params.systemPrompt ?? ''
1520
- const cancelledAssembler = new ResultAssembler({
1521
- runMgr: ctx.runMgr,
1522
- planManager: ctx.planManager,
1523
- activityStore: ctx.activityStore,
1524
- log: ctx.log,
1525
- emitEvent: eventTranslator.emitEvent,
1526
- drainPending: () => eventTranslator.drainPending(),
1527
- signal: ctx.abortController.signal,
1528
- })
1529
- const rootSpan = getTracer().startSpan(
1530
- agentRunSpanName(params.agentName),
1531
- {},
1532
- parentContext(params.parentSpan ?? selectedResumeState?.traceContext),
1533
- )
1534
- rootSpan.setAttributes({
1535
- [NAMZU.RUN_ID]: ctx.runMgr.id,
1536
- [GENAI.AGENT_NAME]: params.agentName,
1537
- [GENAI.AGENT_ID]: params.agentId,
1538
- [GENAI.REQUEST_MODEL]: runConfig.model,
1539
- [GENAI.SYSTEM]: params.provider.id,
1540
- })
1541
-
1542
- try {
1543
- await ctx.runMgr.init()
1544
- if (selectedResumeState) {
1545
- ctx.runMgr.restoreUsage(
1546
- selectedResumeState.tokenUsage,
1547
- selectedResumeState.costInfo,
1548
- selectedResumeState.currentIteration,
1549
- )
1550
- for (const message of selectedResumeState.messages) ctx.runMgr.pushMessage(message)
1551
- for (const queued of queuedForThisRun) ctx.runMgr.pushMessage(queued)
1552
- } else if (params.continuationMode) {
1553
- for (const message of initialMessages) ctx.runMgr.pushMessage(message)
1554
- } else {
1555
- ctx.runMgr.pushMessage(createSystemMessage(cancelledPrompt, 'cache'))
1556
- for (const message of initialMessages) ctx.runMgr.pushMessage(message)
1557
- }
1558
- if (params.eventCursor) {
1559
- yield* catchUpFromCursor(
1560
- ctx.runMgr,
1561
- params.eventCursor,
1562
- params.onEventReplay,
1563
- params.claimFence,
1564
- (error) => {
1565
- ctx.log.warn('Replay observer failed after attachment cancellation', {
1566
- 'exception.message': toErrorMessage(error),
1567
- })
1568
- },
1569
- )
1570
- }
1571
- if (selectedResumeState) {
1572
- await eventTranslator.emitEvent({
1573
- type: 'run_resuming',
1574
- runId: ctx.runId,
1575
- fromCheckpointId: selectedResumeState.checkpointId,
1576
- })
1577
- yield* eventTranslator.drainPending()
1578
- }
1579
- ctx.runMgr.markRunning()
1580
- await eventTranslator.emitEvent({
1581
- type: 'run_started',
1582
- runId: ctx.runId,
1583
- systemPrompt: cancelledPrompt,
1584
- })
1585
- yield* eventTranslator.drainPending()
1586
- ctx.abortController.signal.throwIfAborted()
1587
- } catch (error) {
1588
- // Attachment resolution has already observed the caller's abort. A
1589
- // reconnect callback can still throw while replay is being reported,
1590
- // but it cannot replace that terminal cause or turn a cancelled run
1591
- // into an unpersisted rejection.
1592
- const terminalError = ctx.abortController.signal.aborted
1593
- ? ctx.abortController.signal.reason
1594
- : error
1595
- await executeUserInterruptHooks(terminalError)
1596
- yield* eventTranslator.drainPending()
1597
- yield* cancelledAssembler.handleError(terminalError, rootSpan)
1598
- } finally {
1599
- rootSpan.end()
1600
- }
1601
-
1602
- return await cancelledAssembler.finalize()
915
+ return yield* settlePreStartCancellation(params, prepared)
1603
916
  }
1604
917
 
1605
918
  const unsubscribeTaskStore = params.taskStore
@@ -1841,6 +1154,9 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
1841
1154
  ...(params.maxToolOutputChars !== undefined
1842
1155
  ? { maxToolOutputChars: params.maxToolOutputChars }
1843
1156
  : {}),
1157
+ ...(params.toolResultGuardrails !== undefined
1158
+ ? { toolResultGuardrails: params.toolResultGuardrails }
1159
+ : {}),
1844
1160
  ...(params.retainedToolPreviewChars !== undefined
1845
1161
  ? { retainedToolPreviewChars: params.retainedToolPreviewChars }
1846
1162
  : {}),
@@ -2125,7 +1441,12 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2125
1441
 
2126
1442
  const tracer = getTracer()
2127
1443
 
2128
- return yield* (async function* (): AsyncGenerator<RunEvent, Run> {
1444
+ // Whether the run reached its settle. Read by the `finally` below, and
1445
+ // the only thing that distinguishes a run that finished from one whose
1446
+ // consumer walked away — see `settleAbandonedRun`.
1447
+ let settled = false
1448
+
1449
+ const runBody = (async function* (): AsyncGenerator<RunEvent, Run> {
2129
1450
  // Parent explicitly when a caller supplied one. Without this every
2130
1451
  // run starts its OWN root trace, so a supervisor delegating to three
2131
1452
  // children produced four disconnected traces instead of one tree —
@@ -2235,6 +1556,11 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2235
1556
  // Decided during checkpoint restore, executed after the sandbox
2236
1557
  // exists — the approved tools may well need it.
2237
1558
  let pendingResume: PendingResumePlan | null = null
1559
+ /**
1560
+ * The cadence park this resume answered, when the decision is one the
1561
+ * ordinary continue path carries out. See the restore path below.
1562
+ */
1563
+ let answeredParkId: CheckpointId | undefined
2238
1564
  /** Tool results recovered from the transcript; see the restore path. */
2239
1565
  let recoveredResults: ReadonlyMap<string, { result: string; isError: boolean }> = new Map()
2240
1566
  let emergencyManager: EmergencySaveManager | undefined
@@ -2390,6 +1716,49 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2390
1716
  ? planPendingResume(projectedCheckpoint, params.pendingDecision, ctx.log)
2391
1717
  : null
2392
1718
 
1719
+ // The park this resume ANSWERS even though there is no plan to
1720
+ // carry the decision out through.
1721
+ //
1722
+ // `planPendingResume` covers the two arms whose decision has to
1723
+ // reach something — the calls a `tool_review` park is about, the
1724
+ // tool a `user_question` park is inside. An `iteration_checkpoint`
1725
+ // park has neither, so it returns no plan, and the unpark further
1726
+ // down — which ran only when there was one — never fired for it.
1727
+ // A run that parked on the cadence, was resumed with
1728
+ // `{action: 'continue'}` and went on to finish its work therefore
1729
+ // kept reporting an OUTSTANDING park to `findPendingCheckpoint`,
1730
+ // so a second resume of the finished run was refused with
1731
+ // `awaiting-decision` for a decision already taken, and because
1732
+ // `prune` skips an unresolved park the row could no longer be
1733
+ // collected by anything.
1734
+ //
1735
+ // The decision IS carried out here — continuing is exactly what
1736
+ // the loop below does, and a plan verdict is the answer to the
1737
+ // question the plan park asked — so the park is resolved at the
1738
+ // same point and with the same meaning "resolved" carries
1739
+ // everywhere else: the record stays, and only its pending state
1740
+ // ends. A `pause` is deliberately not resolved: it holds the
1741
+ // park rather than answering it, which is how the live path
1742
+ // treats it too. `answersParkOf` is the whole map, park type to
1743
+ // answering decision, so an arm cannot go missing by being
1744
+ // absent from a condition again — which is how the plan arm
1745
+ // leaked a finished run's park.
1746
+ //
1747
+ // Resolving it does not depend on the resumed process being able
1748
+ // to act on it, and that is deliberate: the plan's own fate is a
1749
+ // separate defect (nothing restores a plan on the resume path at
1750
+ // all, so the new process has none to approve, execute or
1751
+ // reject) and making the resolution wait for it would leave the
1752
+ // row outstanding for exactly the runs that need it cleared.
1753
+ const parked = projectedCheckpoint.pending
1754
+ answeredParkId =
1755
+ params.pendingDecision &&
1756
+ parked !== undefined &&
1757
+ parked.resolvedAt === undefined &&
1758
+ answersParkOf(parked.request.type, params.pendingDecision)
1759
+ ? projectedCheckpoint.id
1760
+ : undefined
1761
+
2393
1762
  // Recover completed observations and explicitly unknown outcomes.
2394
1763
  // A recorded start is not proof that its external effect failed.
2395
1764
  const unanswered = interruptedToolCalls(projectedCheckpoint.messages)
@@ -2717,6 +2086,9 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2717
2086
  })
2718
2087
  }
2719
2088
  yield* resultAssembler.completeRun(rootSpan)
2089
+ // The run HAS settled, so the outer `finally` must not read
2090
+ // this as an abandonment — it would persist a second time.
2091
+ settled = true
2720
2092
  return await resultAssembler.finalize()
2721
2093
  }
2722
2094
  sandbox = acquisition.sandbox
@@ -2787,7 +2159,15 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2787
2159
  ctx.runMgr.setStopReason('input_guardrail')
2788
2160
  ctx.runMgr.setLastError(inputVerdict.reason ?? 'blocked by an input guardrail')
2789
2161
  yield* resultAssembler.completeRun(rootSpan)
2790
- return ctx.runMgr.getRun()
2162
+ // Same two lines as the sandbox path above, and for the same
2163
+ // reasons — with one that path does not have. This return used to
2164
+ // hand back `getRun()` without persisting, so the terminal state
2165
+ // reached the disk only because the abandonment path found
2166
+ // `settled` false and settled it a second time. A branch that
2167
+ // exists for runs which did NOT settle must not be the reason a
2168
+ // settled one is written down.
2169
+ settled = true
2170
+ return await resultAssembler.finalize()
2791
2171
  }
2792
2172
 
2793
2173
  // Honor the approval a human already gave, before the loop's
@@ -2813,149 +2193,59 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2813
2193
 
2814
2194
  await applyPendingResume(pendingResume, ctx.runMgr, toolExecutor, recoveredResults)
2815
2195
  yield* eventTranslator.drainPending()
2816
-
2817
- // The decision has now actually been carried out, so the park
2818
- // is no longer outstanding. Without this the checkpoint keeps
2819
- // reporting `pending` with no `resolvedAt`, and an approval
2820
- // queue re-serves a destructive call that already ran — which
2821
- // defeats the entire point of recording the park.
2822
- const resolvedCheckpointId = pendingResume.checkpointId
2823
- if (params.pendingDecision) {
2824
- await checkpointMgr
2825
- .unpark(resolvedCheckpointId, params.pendingDecision)
2826
- .catch((err: unknown) => {
2827
- ctx.log.error('Applied a pending decision but failed to clear the park', {
2828
- [NAMZU.RUN_ID]: ctx.runId,
2829
- 'namzu.checkpoint.id': resolvedCheckpointId,
2830
- 'exception.message': err instanceof Error ? err.message : String(err),
2831
- })
2832
- return null
2833
- })
2834
- }
2835
2196
  }
2836
2197
 
2837
- yield* iterationOrchestrator.runLoop()
2838
-
2839
- if (params.pluginManager) {
2840
- const hookResults = await params.pluginManager.executeHooks(
2841
- 'run_end',
2842
- { runId: ctx.runId, signal: ctx.abortController.signal },
2843
- eventTranslator.emitEvent,
2844
- )
2845
- applyLifecycleHookResults('run_end', hookResults)
2846
- yield* eventTranslator.drainPending()
2847
- // A delegated run says so once more, by name, so a hook that
2848
- // only cares when a subagent finishes need not read parent ids
2849
- // off every run_end.
2850
- if (params.parentRunId !== undefined) {
2851
- const stopResults = await params.pluginManager.executeHooks(
2852
- 'subagent_stop',
2853
- {
2854
- runId: ctx.runId,
2855
- parentRunId: params.parentRunId,
2856
- signal: ctx.abortController.signal,
2857
- },
2858
- eventTranslator.emitEvent,
2859
- )
2860
- applyLifecycleHookResults('subagent_stop', stopResults)
2861
- yield* eventTranslator.drainPending()
2862
- }
2863
- }
2864
-
2865
- // Hand the step record to the run before it settles, so the
2866
- // returned `Run` carries it.
2867
- ctx.runMgr.setSteps(iterationOrchestrator.getSteps())
2868
-
2869
- // Gates the FINAL result, not the stream — `text_delta` already
2870
- // reached the host as the model produced it. A rewrite is
2871
- // therefore a correction, and the event says so; buffering every
2872
- // token to gate the stream itself would trade the streaming UX
2873
- // for the guarantee, which is the host's call, not the SDK's.
2874
- if (params.outputGuardrails && params.outputGuardrails.length > 0) {
2875
- // Read what the run produced WITHOUT settling it. This used to
2876
- // call `markCompleted()` just to materialize the text, which
2877
- // force-marked a cancelled or paused run `completed` merely
2878
- // because a guardrail was configured — the presence of a
2879
- // safety check silently rewrote the run's own outcome.
2880
- const produced = ctx.runMgr.materializeResult()
2881
- const outputVerdict = await runOutputGuardrails(
2882
- params.outputGuardrails,
2883
- { runId: ctx.runId, output: produced, messages: ctx.runMgr.messages },
2884
- ctx.log,
2885
- )
2886
-
2887
- if (outputVerdict.blocked || outputVerdict.rewritten !== undefined) {
2888
- ctx.runMgr.clearStructuredOutput()
2889
- if (
2890
- params.structuredOutput &&
2891
- outputVerdict.rewritten !== undefined &&
2892
- ctx.runMgr.stopReason === 'end_turn'
2893
- )
2894
- ctx.runMgr.setStopReason('output_guardrail')
2895
- }
2896
-
2897
- if (outputVerdict.blocked) {
2898
- await eventTranslator.emitEvent({
2899
- type: 'guardrail_triggered',
2900
- runId: ctx.runId,
2901
- stage: 'output',
2902
- action: 'block',
2903
- ...(outputVerdict.name ? { guardrail: outputVerdict.name } : {}),
2904
- ...(outputVerdict.reason ? { reason: outputVerdict.reason } : {}),
2905
- })
2906
- yield* eventTranslator.drainPending()
2907
- // Same reasoning as the input-guardrail branch above.
2908
- await ctx.runMgr.recordAudit({
2909
- what: { action: 'guardrail:output', resource: outputVerdict.name },
2910
- outcome: 'refused',
2911
- reason: outputVerdict.reason ?? 'blocked by an output guardrail',
2912
- ...(params.persona?.identity.role ? { persona: params.persona.identity.role } : {}),
2913
- })
2914
- ctx.runMgr.setStopReason('output_guardrail')
2915
- ctx.runMgr.setLastError(outputVerdict.reason ?? 'blocked by an output guardrail')
2916
- ctx.runMgr.setResult('')
2917
- } else if (outputVerdict.rewritten !== undefined) {
2918
- await eventTranslator.emitEvent({
2919
- type: 'guardrail_triggered',
2920
- runId: ctx.runId,
2921
- stage: 'output',
2922
- action: 'rewrite',
2923
- ...(outputVerdict.name ? { guardrail: outputVerdict.name } : {}),
2924
- ...(outputVerdict.reason ? { reason: outputVerdict.reason } : {}),
2198
+ // The decision has now actually been carried out, so the park it
2199
+ // answered is no longer outstanding. Without this the checkpoint
2200
+ // keeps reporting `pending` with no `resolvedAt`, and an approval
2201
+ // queue re-serves a call that already ran — or a question already
2202
+ // answered — which defeats the entire point of recording the park.
2203
+ //
2204
+ // Two arms reach this point, and being outside `if (pendingResume)`
2205
+ // is what the second one needs. One is a plan whose decision was
2206
+ // applied to a batch above. The other is the cadence arm, for which
2207
+ // `planPendingResume` rightly produces no plan because the loop
2208
+ // resuming IS its decision being carried out (`answeredParkId`, set
2209
+ // on the restore path). Resolving only the first left a finished run
2210
+ // reporting `awaiting-decision` forever.
2211
+ //
2212
+ // What is RECORDED depends on which of the two produced the plan. A
2213
+ // recovery plan means the batch was answered by the crash path
2214
+ // rather than by the decision — the calls the human was asked about
2215
+ // were closed with explicitly unknown outcomes — so the human's
2216
+ // answer must not be written down as what ended the park. The park is
2217
+ // still resolved: the question is moot, and leaving it outstanding
2218
+ // would have `findPendingCheckpoint` serve it as the newest
2219
+ // outstanding park, so a host resuming it would rewind this run to
2220
+ // the checkpoint the crash happened on and re-execute a batch the run
2221
+ // has long since moved past.
2222
+ const resolvedCheckpointId = pendingResume?.checkpointId ?? answeredParkId
2223
+ const recordedDecision =
2224
+ pendingResume?.source === 'recovery' && params.pendingDecision
2225
+ ? supersededByRecovery(params.pendingDecision)
2226
+ : params.pendingDecision
2227
+ if (recordedDecision && resolvedCheckpointId) {
2228
+ await checkpointMgr.unpark(resolvedCheckpointId, recordedDecision).catch((err: unknown) => {
2229
+ ctx.log.error('Applied a pending decision but failed to clear the park', {
2230
+ [NAMZU.RUN_ID]: ctx.runId,
2231
+ 'namzu.checkpoint.id': resolvedCheckpointId,
2232
+ 'exception.message': err instanceof Error ? err.message : String(err),
2925
2233
  })
2926
- yield* eventTranslator.drainPending()
2927
- ctx.runMgr.setResult(outputVerdict.rewritten)
2928
- }
2929
- }
2930
-
2931
- if (params.consolidateInto && workingStateManager) {
2932
- const entry = consolidationEntry(workingStateManager.getState(), {
2933
- runId: ctx.runId,
2934
- at: Date.now(),
2234
+ return null
2935
2235
  })
2936
- if (entry) {
2937
- try {
2938
- const { entry: saved } = await params.consolidateInto.create(entry)
2939
- await eventTranslator.emitEvent({
2940
- type: 'memory_consolidated',
2941
- runId: ctx.runId,
2942
- memoryId: saved.id,
2943
- title: entry.title,
2944
- decisions: workingStateManager.getState().decisions.length,
2945
- discoveries: workingStateManager.getState().discoveries.length,
2946
- failures: workingStateManager.getState().failures.length,
2947
- })
2948
- yield* eventTranslator.drainPending()
2949
- } catch (error) {
2950
- ctx.log.warn('consolidation into the memory store failed', {
2951
- [NAMZU.RUN_ID]: ctx.runId,
2952
- 'namzu.memory.error': toErrorMessage(error),
2953
- })
2954
- }
2955
- }
2956
2236
  }
2957
- if (ctx.abortController.signal.aborted) ctx.runMgr.markCancelled()
2958
- yield* resultAssembler.completeRun(rootSpan)
2237
+
2238
+ yield* iterationOrchestrator.runLoop()
2239
+
2240
+ yield* finalizeRun({
2241
+ ctx,
2242
+ params,
2243
+ eventTranslator,
2244
+ takeSteps: () => iterationOrchestrator.getSteps(),
2245
+ workingStateManager,
2246
+ resultAssembler,
2247
+ rootSpan,
2248
+ })
2959
2249
  } catch (err) {
2960
2250
  // A failed run still spent its steps; report them.
2961
2251
  ctx.runMgr.setSteps(iterationOrchestrator.getSteps())
@@ -2963,109 +2253,129 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
2963
2253
  yield* eventTranslator.drainPending()
2964
2254
  yield* resultAssembler.handleError(err, rootSpan)
2965
2255
  } finally {
2966
- // Release the process's termination path as soon as this run is
2967
- // done with it. Leaving the handlers installed would keep a
2968
- // WeakRef'd, settled run as the crash target for the rest of the
2969
- // process's life.
2970
- emergencyManager?.detach()
2971
-
2972
- // A background job outlives the tool call that started it — that
2973
- // is what it is for — so nothing but this stops it outliving the
2974
- // RUN. Scoped to this run's id: a shared registry serving several
2975
- // runs must not have one of them tear down another's work.
2976
- //
2977
- // Awaited, and its failure swallowed. A job that would not die is
2978
- // worth a log line, and is not worth retracting a run's answer.
2979
- unsubscribeJobExits?.()
2980
- // The wait-intent recorder listens on the same shared registry and
2981
- // leaks the same way if it is left attached.
2982
- awaitedJobs?.close()
2983
- // Only jobs bound to this run. Jobs a host bound to its session are
2984
- // the host's to stop, when the session ends.
2985
- if (params.backgroundJobs && (params.backgroundJobOwner ?? ctx.runId) === ctx.runId) {
2986
- try {
2987
- const stopped = await params.backgroundJobs.killOwner(ctx.runId)
2988
- if (stopped.length > 0) {
2989
- ctx.log.info('Background jobs stopped with the run', {
2990
- [NAMZU.RUN_ID]: ctx.runId,
2991
- 'namzu.jobs.stopped': stopped.length,
2992
- })
2993
- }
2994
- } catch (jobErr) {
2995
- ctx.log.error('A background job did not stop cleanly', {
2996
- [NAMZU.RUN_ID]: ctx.runId,
2997
- ...errorAttributes(jobErr),
2998
- })
2999
- }
3000
- }
3001
-
3002
- // Same reasoning for the question channel: the tools outlive the
3003
- // run that bound them, so leaving it attached would have a later
3004
- // run's question written into this run's checkpoint store.
3005
- questionParks.unbind()
3006
-
3007
- // Offer what the run learned to whoever decides what is worth
3008
- // keeping. In `finally` and awaited: a run that failed still
3009
- // discovered things, and a fire-and-forget write would race the
3010
- // process exiting on a one-shot CLI run. A throw here is
3011
- // swallowed — a memory that failed to form must not retract an
3012
- // answer that was already produced.
3013
- const candidate = memoryCandidateFor(ctx.runId, workingStateManager)
3014
- if (params.promoteMemory && candidate) {
3015
- try {
3016
- await params.promoteMemory(candidate)
3017
- } catch (promoteErr) {
3018
- ctx.log.error('Memory promotion threw — the run is unaffected', {
3019
- [NAMZU.RUN_ID]: ctx.runId,
3020
- 'exception.message':
3021
- promoteErr instanceof Error ? promoteErr.message : String(promoteErr),
3022
- })
3023
- }
3024
- }
3025
-
3026
- // --- Sandbox lifecycle: destroy after run ---
3027
- if (sandbox) {
3028
- const sandboxId = sandbox.id
3029
- const teardown = await teardownSandbox(sandbox, sandboxTeardownTimeoutMs)
3030
- if (teardown.kind === 'destroyed') {
3031
- await eventTranslator.emitEvent({
3032
- type: 'sandbox_destroyed',
3033
- runId: ctx.runId,
3034
- sandboxId,
3035
- })
3036
- yield* eventTranslator.drainPending()
3037
- ctx.log.info('Sandbox destroyed', { 'namzu.sandbox.id': sandboxId })
3038
- } else {
3039
- ctx.log.error('Sandbox destroy failed', {
3040
- 'namzu.sandbox.id': sandboxId,
3041
- ...errorAttributes(teardown.error),
3042
- })
3043
- }
3044
- }
3045
-
3046
- unsubscribeTaskStore?.()
3047
- // Keyed by HOW it settled, not just that it did: a run that was
3048
- // cancelled and a run that hit its budget have very different
3049
- // duration distributions, and averaging them together describes
3050
- // neither.
3051
- recordRunDuration(ctx.runMgr.getRun().status ?? 'unknown', Date.now() - runStartedAt)
3052
- rootSpan.end()
2256
+ yield* releaseRunResources({
2257
+ ctx,
2258
+ eventTranslator,
2259
+ emergencyManager,
2260
+ unsubscribeJobExits,
2261
+ unsubscribeTaskStore,
2262
+ awaitedJobs,
2263
+ backgroundJobs: params.backgroundJobs,
2264
+ backgroundJobOwner: params.backgroundJobOwner,
2265
+ questionParks,
2266
+ workingStateManager,
2267
+ promoteMemory: params.promoteMemory,
2268
+ sandbox,
2269
+ sandboxTeardownTimeoutMs,
2270
+ runStartedAt,
2271
+ rootSpan,
2272
+ })
3053
2273
  }
3054
2274
 
2275
+ // Reached only by a run that settled on its own terms. `finalize()` is
2276
+ // the only thing in this body that writes the durable half of the run,
2277
+ // and a `return` completion arriving from a consumer (`break` out of
2278
+ // `for await`, `gen.return()`) runs the `finally` above and stops short
2279
+ // of here. The flag is what tells the two apart, and this is one of
2280
+ // three sites that set it — the sandbox-acquisition and input-guardrail
2281
+ // returns settle early and set it there. Set before the await rather
2282
+ // than after, because a store that throws on the way out must not send
2283
+ // the abandonment path over the same broken ground.
2284
+ settled = true
3055
2285
  return await resultAssembler.finalize()
3056
2286
  })()
2287
+
2288
+ try {
2289
+ return yield* runBody
2290
+ } finally {
2291
+ if (!settled) await settleAbandonedRun(ctx.runMgr, ctx.log)
2292
+ }
3057
2293
  }
3058
2294
 
3059
2295
  /**
3060
- * Hand a returning consumer what it missed, or tell it why it cannot have it.
2296
+ * Write a terminal durable record for a run whose consumer walked away.
2297
+ *
2298
+ * `for await (… ) break` and an explicit `gen.return()` both end the run
2299
+ * body early. Everything the run's `finally` owns still happens — background
2300
+ * jobs are killed, the sandbox is destroyed, the span ends, the duration is
2301
+ * recorded — and then the generator stops. `finalize()` never runs, so
2302
+ * `persist()` never runs, and the store keeps whatever `init()` wrote: a
2303
+ * non-terminal status for a run that no longer exists. `deriveRunStatus`
2304
+ * reads that record back as `queued`, work waiting to start, and a host
2305
+ * rebuilding its view from the store believes it.
2306
+ *
2307
+ * There is nothing to emit here and nothing to emit it to: the consumer
2308
+ * that would have received the events is the one that left. This is about
2309
+ * the durable record only.
2310
+ *
2311
+ * `cancelled` is the verdict, and it is chosen from the existing vocabulary
2312
+ * because it is the one that is true. The run did not complete — no result
2313
+ * was produced and no terminal event was ever delivered — and nothing
2314
+ * failed, so `failed` would name an error that never happened; a run whose
2315
+ * consumer stopped reading and whose processes were torn down under it is
2316
+ * the same fact `markCancelled` already records when a run abort tears one
2317
+ * down. It needs no new `RunExecutionStatus` and no new `StopReason`.
3061
2318
  *
3062
- * Yields NOTHING on a refusal. A partial catch-up is the failure this exists to
3063
- * prevent: a consumer that receives some of the gap folds it into its state and
3064
- * cannot tell the state is wrong, where one that receives an explicit
3065
- * `unavailable` re-derives from the transcript and is right. The run continues
3066
- * either way — a stale cursor belongs to the client, and must not be able to
3067
- * stop the work.
2319
+ * A verdict the run already reached is left standing. A run that failed,
2320
+ * or was cancelled, before the consumer left still says so; what the
2321
+ * abandonment adds is that the record reaches the disk at all.
2322
+ *
2323
+ * Neither is a verdict written over a PARK. A park is a promise to a human
2324
+ * that outlives the consumer: the run is resumable and somebody is still owed
2325
+ * an answer, and `deriveRunStatus` reads a terminal status BEFORE it reads the
2326
+ * park — so recording `cancelled` turns `awaiting_hitl` into `cancelled` for a
2327
+ * run nobody answered for, while the unanswered question stays on the record
2328
+ * and the checkpoint it belongs to stays the place a resume starts from. The
2329
+ * durable state is asked rather than the in-memory one because the in-memory
2330
+ * one is the misleading half here: `handleHITLDecision` emits `run_paused` and
2331
+ * drains it BEFORE it calls `setStopReason('paused')`, so a consumer that
2332
+ * leaves on that event leaves a run whose status is `running` and whose stop
2333
+ * reason is unset at the exact instant its park is already durable.
2334
+ * `findPendingCheckpoint` is the same read an approval queue is built from,
2335
+ * expired parks included in its judgement: a park nobody answered in time is
2336
+ * not somebody still being asked.
2337
+ *
2338
+ * Never throws. It runs while an exception may already be unwinding, and a
2339
+ * store that cannot be written must not replace the run's real failure with
2340
+ * its own.
3068
2341
  */
2342
+ async function settleAbandonedRun(runMgr: RunPersistence, log: Logger): Promise<void> {
2343
+ try {
2344
+ // A terminal verdict is written whatever the park says: `deriveRunStatus`
2345
+ // settles a run that finished, failed or was cancelled BEFORE it looks at
2346
+ // a park ("terminal beats parked"), so a settled run is not waiting for
2347
+ // anybody and the row it already wrote must reach the disk. This ordering
2348
+ // is also what keeps a stale park from suppressing the write.
2349
+ if (!isTerminalStatus(runMgr.status)) {
2350
+ const parked = await findPendingCheckpoint(runMgr.getCheckpointStore(), runMgr.getRunScope())
2351
+ if (parked) {
2352
+ // Left exactly as it stands: no verdict, no write. The park row is
2353
+ // this run's durable state, and `persist()` here would add a
2354
+ // second claim — `running`, for a process that is gone — beside it.
2355
+ log.info('Abandoned run left parked for a human to answer', {
2356
+ [NAMZU.RUN_ID]: runMgr.id,
2357
+ 'namzu.checkpoint.id': parked.id,
2358
+ 'namzu.runtime.park_type': parked.pending?.request.type,
2359
+ })
2360
+ return
2361
+ }
2362
+ runMgr.markCancelled()
2363
+ }
2364
+ // Once: the `finally` that calls this runs once, and every site in the
2365
+ // run body that settles through `finalize()` sets `settled` before it
2366
+ // returns, so the two can never both write.
2367
+ await runMgr.persist()
2368
+ log.info('Abandoned run recorded as cancelled', {
2369
+ [NAMZU.RUN_ID]: runMgr.id,
2370
+ })
2371
+ } catch (err) {
2372
+ log.error('Failed to record the terminal state of an abandoned run', {
2373
+ [NAMZU.RUN_ID]: runMgr.id,
2374
+ 'exception.message': err instanceof Error ? err.message : String(err),
2375
+ })
2376
+ }
2377
+ }
2378
+
3069
2379
  /** The text of the newest user turn, which is what a prompt hook is asked about. */
3070
2380
  function lastUserPrompt(messages: readonly Message[]): string {
3071
2381
  for (let i = messages.length - 1; i >= 0; i--) {
@@ -3075,40 +2385,6 @@ function lastUserPrompt(messages: readonly Message[]): string {
3075
2385
  return ''
3076
2386
  }
3077
2387
 
3078
- async function* catchUpFromCursor(
3079
- runMgr: RunPersistence,
3080
- cursor: RunEventCursor,
3081
- onEventReplay: ((replay: RunEventReplay) => void) | undefined,
3082
- generation: FencingToken | undefined,
3083
- onReplayObserverError: (error: unknown) => void,
3084
- ): AsyncGenerator<RunEvent, void> {
3085
- const missed = await runMgr.getRunStore().readEvents({ sinceSeq: cursor.sinceSeq })
3086
- const replay = resolveRunEventReplay(
3087
- cursor,
3088
- {
3089
- lastSeq: runMgr.lastEventSeq,
3090
- ...(generation !== undefined ? { generation } : {}),
3091
- },
3092
- missed,
3093
- )
3094
-
3095
- if (onEventReplay) {
3096
- try {
3097
- // A callback typed `void` may still be implemented with `async` in
3098
- // TypeScript. Observe that runtime Promise so a late rejection cannot
3099
- // become process-wide, but never await host code here: replay delivery
3100
- // and an already-cancelled run must not inherit observer liveness.
3101
- const settlement = onEventReplay(replay)
3102
- void Promise.resolve(settlement).catch(onReplayObserverError)
3103
- } catch (error) {
3104
- onReplayObserverError(error)
3105
- }
3106
- }
3107
-
3108
- if (replay.status !== 'replayed') return
3109
- for (const event of replay.events) yield event
3110
- }
3111
-
3112
2388
  type DrainQueryParams = Omit<QueryParams, 'resumeHandler'> & {
3113
2389
  resumeHandler?: ResumeHandler
3114
2390
  }