@namzu/sdk 41.0.0 → 42.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. package/CHANGELOG.md +233 -0
  2. package/dist/agents/ReactiveAgent.d.ts.map +1 -1
  3. package/dist/agents/ReactiveAgent.js +3 -0
  4. package/dist/agents/ReactiveAgent.js.map +1 -1
  5. package/dist/agents/SupervisorAgent.d.ts.map +1 -1
  6. package/dist/agents/SupervisorAgent.js +11 -0
  7. package/dist/agents/SupervisorAgent.js.map +1 -1
  8. package/dist/agents/runAgent.d.ts +14 -0
  9. package/dist/agents/runAgent.d.ts.map +1 -1
  10. package/dist/agents/runAgent.js +3 -0
  11. package/dist/agents/runAgent.js.map +1 -1
  12. package/dist/manager/agent/lifecycle.d.ts.map +1 -1
  13. package/dist/manager/agent/lifecycle.js +20 -0
  14. package/dist/manager/agent/lifecycle.js.map +1 -1
  15. package/dist/manager/resident/outbox.d.ts +8 -8
  16. package/dist/manager/resident/store.d.ts +4 -4
  17. package/dist/public-runtime.d.ts +4 -1
  18. package/dist/public-runtime.d.ts.map +1 -1
  19. package/dist/public-runtime.js +13 -1
  20. package/dist/public-runtime.js.map +1 -1
  21. package/dist/public-tools.d.ts +1 -1
  22. package/dist/public-tools.d.ts.map +1 -1
  23. package/dist/public-tools.js +4 -2
  24. package/dist/public-tools.js.map +1 -1
  25. package/dist/registry/tool/execute.d.ts.map +1 -1
  26. package/dist/registry/tool/execute.js +10 -1
  27. package/dist/registry/tool/execute.js.map +1 -1
  28. package/dist/runtime/bidi/session.d.ts +11 -0
  29. package/dist/runtime/bidi/session.d.ts.map +1 -1
  30. package/dist/runtime/bidi/session.js +2 -0
  31. package/dist/runtime/bidi/session.js.map +1 -1
  32. package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
  33. package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
  34. package/dist/runtime/query/cancelled-before-start.js +152 -0
  35. package/dist/runtime/query/cancelled-before-start.js.map +1 -0
  36. package/dist/runtime/query/checkpoint.d.ts +21 -0
  37. package/dist/runtime/query/checkpoint.d.ts.map +1 -1
  38. package/dist/runtime/query/checkpoint.js +23 -0
  39. package/dist/runtime/query/checkpoint.js.map +1 -1
  40. package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
  41. package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
  42. package/dist/runtime/query/executor/tool-call-admission.js +373 -0
  43. package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
  44. package/dist/runtime/query/executor.d.ts +76 -35
  45. package/dist/runtime/query/executor.d.ts.map +1 -1
  46. package/dist/runtime/query/executor.js +52 -380
  47. package/dist/runtime/query/executor.js.map +1 -1
  48. package/dist/runtime/query/finalize-run.d.ts +55 -0
  49. package/dist/runtime/query/finalize-run.d.ts.map +1 -0
  50. package/dist/runtime/query/finalize-run.js +113 -0
  51. package/dist/runtime/query/finalize-run.js.map +1 -0
  52. package/dist/runtime/query/guardrail-presets.d.ts +187 -1
  53. package/dist/runtime/query/guardrail-presets.d.ts.map +1 -1
  54. package/dist/runtime/query/guardrail-presets.js +298 -0
  55. package/dist/runtime/query/guardrail-presets.js.map +1 -1
  56. package/dist/runtime/query/index.d.ts +18 -9
  57. package/dist/runtime/query/index.d.ts.map +1 -1
  58. package/dist/runtime/query/index.js +241 -893
  59. package/dist/runtime/query/index.js.map +1 -1
  60. package/dist/runtime/query/iteration/index.d.ts +6 -161
  61. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  62. package/dist/runtime/query/iteration/index.js +23 -523
  63. package/dist/runtime/query/iteration/index.js.map +1 -1
  64. package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
  65. package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
  66. package/dist/runtime/query/iteration/outstanding-work.js +365 -0
  67. package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
  68. package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
  69. package/dist/runtime/query/iteration/phases/plan.js +13 -2
  70. package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
  71. package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
  72. package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
  73. package/dist/runtime/query/iteration/step-shaping.js +184 -0
  74. package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
  75. package/dist/runtime/query/prepare-run.d.ts +94 -0
  76. package/dist/runtime/query/prepare-run.d.ts.map +1 -0
  77. package/dist/runtime/query/prepare-run.js +589 -0
  78. package/dist/runtime/query/prepare-run.js.map +1 -0
  79. package/dist/runtime/query/release-run.d.ts +56 -0
  80. package/dist/runtime/query/release-run.d.ts.map +1 -0
  81. package/dist/runtime/query/release-run.js +101 -0
  82. package/dist/runtime/query/release-run.js.map +1 -0
  83. package/dist/runtime/query/resume-pending.d.ts +112 -1
  84. package/dist/runtime/query/resume-pending.d.ts.map +1 -1
  85. package/dist/runtime/query/resume-pending.js +133 -0
  86. package/dist/runtime/query/resume-pending.js.map +1 -1
  87. package/dist/runtime/query/tooling.d.ts +2 -0
  88. package/dist/runtime/query/tooling.d.ts.map +1 -1
  89. package/dist/runtime/query/tooling.js +3 -0
  90. package/dist/runtime/query/tooling.js.map +1 -1
  91. package/dist/store/evidence/compaction-archive.d.ts +2 -2
  92. package/dist/tools/coordinator/agent.d.ts.map +1 -1
  93. package/dist/tools/coordinator/agent.js +17 -2
  94. package/dist/tools/coordinator/agent.js.map +1 -1
  95. package/dist/tools/coordinator/index.d.ts.map +1 -1
  96. package/dist/tools/coordinator/index.js +17 -3
  97. package/dist/tools/coordinator/index.js.map +1 -1
  98. package/dist/tools/untrusted-envelope.d.ts +35 -0
  99. package/dist/tools/untrusted-envelope.d.ts.map +1 -1
  100. package/dist/tools/untrusted-envelope.js +91 -3
  101. package/dist/tools/untrusted-envelope.js.map +1 -1
  102. package/dist/types/agent/base.d.ts +23 -0
  103. package/dist/types/agent/base.d.ts.map +1 -1
  104. package/dist/types/agent/task.d.ts +19 -0
  105. package/dist/types/agent/task.d.ts.map +1 -1
  106. package/dist/types/run/config.d.ts +12 -5
  107. package/dist/types/run/config.d.ts.map +1 -1
  108. package/dist/types/tool/index.d.ts +19 -0
  109. package/dist/types/tool/index.d.ts.map +1 -1
  110. package/dist/types/tool/index.js.map +1 -1
  111. package/package.json +1 -1
  112. package/src/agents/ReactiveAgent.ts +3 -0
  113. package/src/agents/SupervisorAgent.ts +11 -0
  114. package/src/agents/runAgent.ts +18 -0
  115. package/src/manager/agent/lifecycle.ts +22 -0
  116. package/src/public-runtime.ts +14 -0
  117. package/src/public-tools.ts +8 -2
  118. package/src/registry/tool/execute.ts +9 -1
  119. package/src/runtime/bidi/session.ts +13 -0
  120. package/src/runtime/query/cancelled-before-start.ts +189 -0
  121. package/src/runtime/query/checkpoint.ts +22 -0
  122. package/src/runtime/query/executor/tool-call-admission.ts +473 -0
  123. package/src/runtime/query/executor.ts +76 -442
  124. package/src/runtime/query/finalize-run.ts +192 -0
  125. package/src/runtime/query/guardrail-presets.ts +356 -0
  126. package/src/runtime/query/index.ts +287 -1011
  127. package/src/runtime/query/iteration/index.ts +40 -586
  128. package/src/runtime/query/iteration/outstanding-work.ts +386 -0
  129. package/src/runtime/query/iteration/phases/plan.ts +18 -2
  130. package/src/runtime/query/iteration/step-shaping.ts +271 -0
  131. package/src/runtime/query/prepare-run.ts +718 -0
  132. package/src/runtime/query/release-run.ts +168 -0
  133. package/src/runtime/query/resume-pending.ts +158 -0
  134. package/src/runtime/query/tooling.ts +5 -0
  135. package/src/tools/coordinator/agent.ts +17 -2
  136. package/src/tools/coordinator/index.ts +17 -3
  137. package/src/tools/untrusted-envelope.ts +94 -3
  138. package/src/types/agent/base.ts +24 -0
  139. package/src/types/agent/task.ts +20 -0
  140. package/src/types/run/config.ts +12 -5
  141. package/src/types/tool/index.ts +20 -0
@@ -0,0 +1,386 @@
1
+ import { NAMZU } from '../../../constants/telemetry/index.js'
2
+ import { formatCompletionNotification } from '../../../scheduler/completion-inbox.js'
3
+ import { DELEGATION_TIMEOUT_MS } from '../../../tools/coordinator/index.js'
4
+ import { createRuntimeContextMessage } from '../../../types/message/index.js'
5
+ import type { RunEvent } from '../../../types/run/index.js'
6
+ import { readPositiveIntEnv } from '../../../utils/env.js'
7
+ import { formatJobNote } from '../steering.js'
8
+ import type { IterationContext } from './phases/index.js'
9
+
10
+ /**
11
+ * The run's settle points: holding open for work that has not finished, and
12
+ * delivering what arrived.
13
+ *
14
+ * Two kinds of work qualify — a delegated task the `CompletionInbox` is still
15
+ * expecting, and a background job the model told `wait_for_job` it is waiting
16
+ * on — and both are raced together, because a run has one settle point and one
17
+ * grace period to spend at it.
18
+ *
19
+ * Everything here reads the iteration context rather than capturing it, and
20
+ * the one thing it cannot read off that context — how this run drains its
21
+ * inbound queue — arrives as an explicit `deliverInbound` input, because the
22
+ * recording of operator intent that goes with it belongs to the orchestrator.
23
+ * `holdForOutstandingWork` is a generator and is reached with `yield*`: its
24
+ * one event must land at the position it landed at when it lived on the class.
25
+ */
26
+
27
+ /**
28
+ * The share of a run's REMAINING time a settle-hold may take.
29
+ *
30
+ * The rule is borrowed from `AGENT_MANAGER_DEFAULTS.maxBudgetFraction`, which
31
+ * gives a spawned child at most half of what its parent has left: one
32
+ * sub-activity may take a share of the remainder, never the remainder. The
33
+ * value is written out here rather than imported, because that field is a
34
+ * host-tunable knob about TOKEN allocation and coupling the two would let a
35
+ * host lowering one silently change the other.
36
+ *
37
+ * Half, specifically, because the hold is not the last thing the run does.
38
+ * Its whole purpose is to put a worker's result where the model can read it,
39
+ * and reading it costs a turn. A hold that spent everything remaining would
40
+ * deliver a notification into a run with no turn left to act on it — the same
41
+ * "the result exists and the model is never told" failure this mechanism was
42
+ * built to close, wearing a different costume.
43
+ */
44
+ const SETTLE_GRACE_FRACTION = 0.5
45
+
46
+ /**
47
+ * How long a finishing run waits for a background worker it launched.
48
+ *
49
+ * Derived from the run rather than fixed, because a constant is wrong in both
50
+ * directions at once. The 120 seconds this replaces held a run configured for
51
+ * a twenty-second timeout open for 120,267 ms — six times its own budget, and
52
+ * unreachable by the guard, which only checks between iterations — while on an
53
+ * hour-long run it abandoned workers measured at 4m21s, 5m58s and 8m04s, all
54
+ * of them well inside the hour the delegation tools themselves declare.
55
+ *
56
+ * **Bounded by construction, and against the right boundary.** The input is
57
+ * time-to-FINALIZE, not time-to-deadline (see
58
+ * `GuardCoordinator.remainingBeforeFinalizeMs`). Measuring to the deadline was
59
+ * the first attempt and it was wrong in a way that looked safe: a hold cannot
60
+ * outlive the deadline either way, but half of the time-to-deadline started
61
+ * just under the warning threshold ends at 95% of the budget — so the slice
62
+ * that exists for the run to produce a closing answer is half spent waiting
63
+ * for the result that answer was supposed to use. Against the finalize point
64
+ * the hold cannot reach the reserve at all, which is what makes the guard's
65
+ * inability to interrupt a hold a non-issue rather than a smaller issue.
66
+ *
67
+ * **The floor of zero is a decision, not a clamp artefact.** A run with no
68
+ * time left before it must start finishing has no turn in which to read a
69
+ * notification, so waiting could only delay a stop that is already due.
70
+ * Nothing is lost by it: `CompletionInbox.waitForArrival` returns before it
71
+ * looks at its timer when a completion is already in hand, so a zero grace
72
+ * still delivers everything that has arrived. No minimum is invented on top,
73
+ * because zero is exactly what a run past the threshold should wait — and
74
+ * reading the remainder at hold time rather than trusting `forceFinalize`,
75
+ * which is sampled at the top of the iteration, is what makes a long iteration
76
+ * that crossed the line in between compute it.
77
+ *
78
+ * **The ceiling is the longest anything in this subsystem waits for a
79
+ * delegated worker.** It binds only for a host whose run timeout exceeds
80
+ * roughly two and a quarter hours; below that the fraction is smaller.
81
+ */
82
+ export function settleGraceMs(remainingBeforeFinalizeMs: number): number {
83
+ return Math.min(
84
+ Math.floor(remainingBeforeFinalizeMs * SETTLE_GRACE_FRACTION),
85
+ DELEGATION_TIMEOUT_MS,
86
+ )
87
+ }
88
+
89
+ /**
90
+ * The ceiling on the job half of that grace, in milliseconds.
91
+ *
92
+ * `DELEGATION_TIMEOUT_MS` is the wrong ceiling for a shell job, and the gap
93
+ * only opens where it matters most: a run with no `timeoutMs` — the CLI's
94
+ * shipping default, `No run deadline by default` — has infinite time before
95
+ * it must start finishing, so `settleGraceMs` returns the ceiling flat. For a
96
+ * delegated task that is sound, because the hour is the longest the task
97
+ * itself may live: the hold cannot outlast the work. A background job has no
98
+ * such bound. `tail -f`, a watcher and a dev server all outlive any hold, so
99
+ * the same arithmetic parks an interactive session for an hour on a job that
100
+ * was never going to exit.
101
+ *
102
+ * So the job leg gets its own bound, and it is sized to what the wait buys
103
+ * rather than to how long a job may live: a turn in which to use the exit.
104
+ * A model that already waited its `wait_for_job` bound out and saw nothing is
105
+ * not usually two minutes from an exit, and the run ending is not the news
106
+ * being lost — with no run in flight the session announces the exit itself
107
+ * (`docs/cli/background-jobs.md`, *Learning that it ended*), which is the
108
+ * cheaper of the two places to hear it.
109
+ */
110
+ const DEFAULT_JOB_HOLD_MAX_MS = 2 * 60 * 1000
111
+
112
+ /**
113
+ * The same share of the run, under {@link DEFAULT_JOB_HOLD_MAX_MS}.
114
+ *
115
+ * `NAMZU_JOB_HOLD_MAX_MS` overrides the ceiling for a host that wants a
116
+ * longer or shorter park, the way `NAMZU_JOB_WAIT_TIMEOUT_MS` overrides
117
+ * `wait_for_job`'s own bound — and it is the same parse, so a value that is
118
+ * not a positive whole number of milliseconds leaves the default standing
119
+ * rather than holding a run for `NaN`. Called here rather than at module
120
+ * load, because a host that sets it after import is not ignored.
121
+ */
122
+ export function awaitedJobGraceMs(remainingBeforeFinalizeMs: number): number {
123
+ const ceiling = readPositiveIntEnv('NAMZU_JOB_HOLD_MAX_MS', DEFAULT_JOB_HOLD_MAX_MS)
124
+ return Math.min(settleGraceMs(remainingBeforeFinalizeMs), ceiling)
125
+ }
126
+
127
+ /**
128
+ * Hold the run open for work that has not finished, and deliver it.
129
+ *
130
+ * Returns whether a completion, a job exit or an operator message entered
131
+ * the transcript — the caller continues on `true`, so the model gets a turn
132
+ * to respond. That turn is the entire justification for waiting, which
133
+ * is why only the exits that can still take one call this.
134
+ *
135
+ * Two kinds of work qualify and they are raced together, because a run has
136
+ * one settle point and one grace period to spend at it:
137
+ *
138
+ * - a delegated task the `CompletionInbox` is still expecting;
139
+ * - a background job the model told `wait_for_job` it is waiting on.
140
+ *
141
+ * The job half is deliberately narrow. Intent comes from the wait and from
142
+ * nothing else — a dev server the model started and never waited on is
143
+ * running because somebody wanted it running, and a hold for it would add
144
+ * the grace period to the end of every turn for the rest of the session.
145
+ *
146
+ * Each leg is opened only when it has something pending: both
147
+ * `waitForArrival` implementations resolve immediately when their own side
148
+ * is idle, so racing an idle one would end the hold before it began.
149
+ *
150
+ * Bounded by `settleGraceMs` and by `maxIterations`, so work that never
151
+ * finishes cannot keep the run open. On a run with a deadline the grace is
152
+ * a share of what is LEFT of it rather than a fresh allowance, so a
153
+ * `wait_for_job` call that already spent minutes has shortened this hold
154
+ * by the same minutes. On a run without one — the CLI's default — there is
155
+ * no remainder to take a share of, and the job leg's own ceiling
156
+ * (`awaitedJobGraceMs`) is what keeps a timed-out wait from being followed
157
+ * by an hour of silence.
158
+ */
159
+ export async function* holdForOutstandingWork(
160
+ ctx: IterationContext,
161
+ iterationNum: number,
162
+ hasToolCalls: boolean,
163
+ deliverInbound: () => number,
164
+ ): AsyncGenerator<RunEvent, boolean> {
165
+ const inbox = ctx.completionInbox?.hasPendingWork ? ctx.completionInbox : undefined
166
+ const jobs = ctx.awaitedJobs?.hasPendingWork ? ctx.awaitedJobs : undefined
167
+ if (!inbox && !jobs) return false
168
+
169
+ // Read HERE rather than from `forceFinalize`, which was sampled at the
170
+ // top of the iteration: one that has since crossed the finalize point
171
+ // must not open a wait against a reserve it has already entered.
172
+ const remainingMs = ctx.guard.remainingBeforeFinalizeMs()
173
+ // One deadline for the race, and it is the LONGEST ceiling any pending
174
+ // leg justifies. A leg resolving on its own timer ends the whole race,
175
+ // so handing the job leg its shorter ceiling while a task was also
176
+ // outstanding would cut the task's hold down to the job's — a run
177
+ // walking away from a worker it had time for, because a job happened
178
+ // to be running. A job therefore never shortens a wait, and it never
179
+ // lengthens one either: where a task is outstanding too, that is how
180
+ // long this run was waiting anyway.
181
+ const graceMs = inbox ? settleGraceMs(remainingMs) : awaitedJobGraceMs(remainingMs)
182
+ ctx.log.info('Holding the run open for outstanding work', {
183
+ [NAMZU.RUN_ID]: ctx.runMgr.id,
184
+ [NAMZU.ITERATION]: iterationNum,
185
+ 'namzu.runtime.grace_ms': graceMs,
186
+ 'namzu.runtime.awaited_jobs': jobs?.outstandingJobIds ?? [],
187
+ })
188
+ // User input releases this wait without cancelling any child. Both waits
189
+ // share a disposable signal so the losing arrival listener cannot leak.
190
+ const waiting = new AbortController()
191
+ const runSignal = ctx.abortController.signal
192
+ const cancelWait = () => waiting.abort(runSignal.reason)
193
+ runSignal.addEventListener('abort', cancelWait, { once: true })
194
+ if (runSignal.aborted) cancelWait()
195
+ try {
196
+ await Promise.race([
197
+ ...(inbox ? [inbox.waitForArrival(graceMs, waiting.signal)] : []),
198
+ ...(jobs ? [jobs.waitForArrival(graceMs, waiting.signal)] : []),
199
+ ...(ctx.waitForInbound ? [ctx.waitForInbound(waiting.signal)] : []),
200
+ ])
201
+ } catch (error) {
202
+ if (!runSignal.aborted) throw error
203
+ } finally {
204
+ waiting.abort()
205
+ runSignal.removeEventListener('abort', cancelWait)
206
+ }
207
+ runSignal.throwIfAborted()
208
+
209
+ const arrived = ctx.completionInbox?.drain() ?? []
210
+ if (arrived.length > 0) {
211
+ ctx.runMgr.pushMessage(
212
+ createRuntimeContextMessage(formatCompletionNotification(arrived), 'task-completion'),
213
+ )
214
+ }
215
+ const exited = deliverAwaitedJobExits(ctx)
216
+ const inbound = deliverInbound()
217
+ if (arrived.length === 0 && !exited && inbound === 0) return false
218
+ await ctx.emitEvent({
219
+ type: 'iteration_completed',
220
+ runId: ctx.runMgr.id,
221
+ iteration: iterationNum,
222
+ hasToolCalls,
223
+ })
224
+ yield* ctx.drainPending()
225
+ return true
226
+ }
227
+
228
+ /**
229
+ * Put the job exits this hold was waiting for in front of the model.
230
+ *
231
+ * Through `jobNotices`, which is the channel a job exit already travels on
232
+ * — `attachNotice` rides it out on the next tool result — rather than a
233
+ * second one built for this path. A turn that called no tools has no such
234
+ * result, so the queued text becomes a `runtime-context` message instead,
235
+ * exactly as `deliverInbound` does for steering that found no tool result
236
+ * to attach to.
237
+ *
238
+ * That drain is also what keeps one exit from being delivered twice: the
239
+ * channel hands its text over once, so an exit already attached to a tool
240
+ * result earlier in the turn leaves nothing here — and the record of it
241
+ * went with that delivery, so this returns `false` rather than buying a
242
+ * turn to re-read what the model has read.
243
+ *
244
+ * `takeDelivery` is what pairs the two. Taking the exits first and then
245
+ * finding no notice would discard them, which is the one way this path
246
+ * can lose an exit outright; neither is taken unless both are there.
247
+ *
248
+ * The channel is not per-job, so the text taken here can include a notice
249
+ * for a job nobody awaited that ended while the hold was open. Delivering
250
+ * it is right — it is unread either way, and the alternative is stranding
251
+ * it — but it is not a reason to WAIT, which is why what opens this hold
252
+ * is `AwaitedJobs`, and the two are asked separately.
253
+ */
254
+ export function deliverAwaitedJobExits(ctx: IterationContext): boolean {
255
+ const delivered = ctx.awaitedJobs?.takeDelivery(() => ctx.jobNotices?.drain())
256
+ if (!delivered) return false
257
+
258
+ ctx.log.info('Delivering a background job exit the run held open for', {
259
+ [NAMZU.RUN_ID]: ctx.runMgr.id,
260
+ 'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
261
+ })
262
+ ctx.runMgr.pushMessage(createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'))
263
+ return true
264
+ }
265
+
266
+ /**
267
+ * Account for outstanding work on the way out: deliver what arrived, and
268
+ * say what did not.
269
+ *
270
+ * A run that ends with a worker outstanding must not leave the impression
271
+ * that the worker's result was delivered. There are exactly two honest
272
+ * outcomes and this does both:
273
+ *
274
+ * - **What has already arrived is delivered.** It makes no false claim,
275
+ * and dropping it is pure loss — the message rides out on
276
+ * `Run.messages`, so a host reads it and the next turn of a continued
277
+ * thread starts with it. This does NOT wait: a hold buys the model a
278
+ * turn in which to USE a result, and on an exit whose answer is already
279
+ * decided there is no such turn, so waiting would delay a settled answer
280
+ * to append text this run will not read. The bounded hold stays where it
281
+ * was, on the exits that do have a turn left.
282
+ * - **What is still running is NAMED, not cancelled.** Giving up on a wait
283
+ * is a statement about the waiter, not about the work — the rule
284
+ * `wait-with-idle-bound.ts` already states for the same subsystem — and
285
+ * "the parent answered early" is a weaker warrant for killing a child
286
+ * than "the clock ran out", not a stronger one. Killing a worker that
287
+ * may be mid-write is a policy only the host can judge, and it has
288
+ * `cancel_task` and the run controller to judge it with.
289
+ */
290
+ export function settleOutstandingWork(ctx: IterationContext): void {
291
+ deliverArrivedCompletions(ctx)
292
+ deliverArrivedJobExits(ctx)
293
+ recordAbandonedWork(ctx)
294
+ }
295
+
296
+ /** Work this run walked away from. See {@link settleOutstandingWork}. */
297
+ export function recordAbandonedWork(ctx: IterationContext): void {
298
+ const abandoned = ctx.completionInbox?.outstandingTaskIds ?? []
299
+ if (abandoned.length > 0) {
300
+ ctx.log.warn('Run ended with delegated work still running', {
301
+ [NAMZU.RUN_ID]: ctx.runMgr.id,
302
+ 'namzu.runtime.tasks': abandoned,
303
+ })
304
+ ctx.runMgr.setAbandonedTaskIds(abandoned)
305
+ }
306
+
307
+ // The same statement for a job the model was waiting on when the grace
308
+ // ran out. Only awaited ones: a job nobody waited for was never work
309
+ // this run was holding, so naming it would report an abandonment that
310
+ // did not happen.
311
+ const abandonedJobs = ctx.awaitedJobs?.outstandingJobIds ?? []
312
+ if (abandonedJobs.length === 0) return
313
+
314
+ ctx.log.warn('Run ended with an awaited background job still running', {
315
+ [NAMZU.RUN_ID]: ctx.runMgr.id,
316
+ 'namzu.runtime.jobs': abandonedJobs,
317
+ })
318
+ ctx.runMgr.setAbandonedJobIds(abandonedJobs)
319
+ }
320
+
321
+ export function deliverArrivedCompletions(ctx: IterationContext): void {
322
+ const unheard = ctx.completionInbox?.drain() ?? []
323
+ if (unheard.length === 0) return
324
+
325
+ // Fix the run's answer BEFORE appending anything after it.
326
+ //
327
+ // `RunPersistence.resolveResult` walks the message tail backwards and
328
+ // stops at the first non-assistant message, and it runs at
329
+ // `markCompleted` — which is AFTER this. So a notification appended
330
+ // after the final assistant turn makes the run's own answer
331
+ // unreachable. Measured, on a run whose model had just said "THIS IS
332
+ // THE RUN ANSWER.": `run.result` came back `undefined`. That trades a
333
+ // lost worker result for a lost RUN result, which is strictly worse
334
+ // than the defect this delivery exists to fix.
335
+ //
336
+ // Materialising resolves it while the tail is still the assistant's;
337
+ // pinning it means the later re-resolution cannot undo the fix. Only
338
+ // when there is something to pin: on the cancelled and thrown paths
339
+ // there may be no answer, and pinning an empty string there would
340
+ // suppress whatever the error path assembles.
341
+ const answer = ctx.runMgr.materializeResult()
342
+ if (answer.length > 0) ctx.runMgr.setResult(answer)
343
+
344
+ ctx.log.info('Delivering task completions the run would have settled over', {
345
+ [NAMZU.RUN_ID]: ctx.runMgr.id,
346
+ 'namzu.runtime.tasks': unheard.map((h) => h.taskId),
347
+ })
348
+ ctx.runMgr.pushMessage(
349
+ createRuntimeContextMessage(formatCompletionNotification(unheard), 'task-completion'),
350
+ )
351
+ }
352
+
353
+ /**
354
+ * The job half of {@link deliverArrivedCompletions}: an exit that arrived
355
+ * too late to earn a turn is still delivered on the way out.
356
+ *
357
+ * The window this closes is one tick wide and it is nobody else's. An
358
+ * awaited job that exits between the hold's grace expiring and the run
359
+ * settling was never delivered — the hold had already looked — and is no
360
+ * longer named either, because the exit took it off the outstanding list
361
+ * on its way past, so `abandonedJobIds` would be lying to claim it. The
362
+ * host's own listener is no help: the CLI queues an exit for the next
363
+ * turn only when no run is in flight, and this one is still in flight.
364
+ * Delivered here it reaches `Run.messages`, so the transcript has it and
365
+ * a continued thread opens with it.
366
+ *
367
+ * Before `recordAbandonedWork`, which then reports only what is still
368
+ * running, and after `deliverArrivedCompletions`, so the two appended
369
+ * messages land in the order the work finished in.
370
+ */
371
+ export function deliverArrivedJobExits(ctx: IterationContext): void {
372
+ const delivered = ctx.awaitedJobs?.takeDelivery(() => ctx.jobNotices?.drain())
373
+ if (!delivered) return
374
+
375
+ // Fix the run's answer BEFORE appending anything after it — the same
376
+ // `resolveResult` tail walk `deliverArrivedCompletions` explains just
377
+ // above, and the same guard against pinning an empty one.
378
+ const answer = ctx.runMgr.materializeResult()
379
+ if (answer.length > 0) ctx.runMgr.setResult(answer)
380
+
381
+ ctx.log.info('Delivering a background job exit the run would have settled over', {
382
+ [NAMZU.RUN_ID]: ctx.runMgr.id,
383
+ 'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
384
+ })
385
+ ctx.runMgr.pushMessage(createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'))
386
+ }
@@ -1,6 +1,11 @@
1
1
  import type { HITLDecisionRequest } from '../../../../types/hitl/index.js'
2
2
  import type { RunEvent } from '../../../../types/run/index.js'
3
- import { type IterationContext, type PhaseSignal, handleHITLDecision } from './context.js'
3
+ import {
4
+ type IterationContext,
5
+ type PhaseSignal,
6
+ awaitDecisionOrAbort,
7
+ handleHITLDecision,
8
+ } from './context.js'
4
9
 
5
10
  export async function* runPlanGate(ctx: IterationContext): AsyncGenerator<RunEvent, PhaseSignal> {
6
11
  if (!ctx.planManager.active || ctx.planManager.active.status !== 'ready') {
@@ -40,8 +45,19 @@ export async function* runPlanGate(ctx: IterationContext): AsyncGenerator<RunEve
40
45
  // Record the park BEFORE awaiting it. A process that dies while a human
41
46
  // is reading the plan otherwise leaves nothing behind saying the plan
42
47
  // was ever put up for approval.
48
+ //
49
+ // The park stays eager — `awaitDecisionDurably` records only after
50
+ // `PARK_RECORD_DELAY_MS`, which is the right trade for a gate that runs
51
+ // on every iteration and the wrong one for a gate that runs once and is
52
+ // read by a human. Only the AWAIT below is raced.
43
53
  await ctx.checkpointMgr.park(planCheckpoint, request)
44
- const planDecision = await ctx.resumeHandler(request)
54
+ // Raced against the run's abort signal, like every other park. A bare
55
+ // `await ctx.resumeHandler(request)` here meant a Stop did nothing until
56
+ // the host answered: `runPlanGate` runs in the iteration loop rather than
57
+ // inside a tool call, so nothing downstream bounded the wait. A Stop now
58
+ // resolves the park as `abort`, which `handleHITLDecision` turns into
59
+ // `setStopReason('cancelled') + markCancelled + stop`.
60
+ const planDecision = await awaitDecisionOrAbort(ctx, request)
45
61
  await ctx.checkpointMgr.unpark(planCheckpoint.id, planDecision)
46
62
 
47
63
  return yield* handleHITLDecision(ctx, planDecision, planCheckpoint.id, 'plan_gate')
@@ -0,0 +1,271 @@
1
+ import { resolveContextWindow } from '../../../compaction/context-window.js'
2
+ import { estimateMessageTokens } from '../../../compaction/token-estimate.js'
3
+ import { NAMZU } from '../../../constants/telemetry/index.js'
4
+ import { renderSkillsSection } from '../../../persona/assembler.js'
5
+ import { PreparationContextError } from '../../../run/preparation-context-error.js'
6
+ import {
7
+ type Message,
8
+ type UserMessage,
9
+ createRuntimeContextMessage,
10
+ createSystemMessage,
11
+ } from '../../../types/message/index.js'
12
+ import type { ToolChoice } from '../../../types/provider/chat.js'
13
+ import type {
14
+ PrepareStepContext,
15
+ PrepareStepResult,
16
+ StepResult,
17
+ StepVeto,
18
+ } from '../../../types/run/index.js'
19
+ import type { Skill } from '../../../types/skills/index.js'
20
+ import { toErrorMessage } from '../../../utils/error.js'
21
+ import { createCallbackInference } from '../callback-inference.js'
22
+ import { measureContext } from './phases/compaction.js'
23
+ import type { IterationContext } from './phases/index.js'
24
+
25
+ /**
26
+ * How the next request is shaped: admission, preparation, and the context
27
+ * budget handed to both.
28
+ *
29
+ * Three of these read state the loop replaces or grows as it runs, so they
30
+ * arrive as accessors rather than as values — `latestUserMessage` is replaced
31
+ * on every operator turn and `steps` gains a member per step, and a captured
32
+ * copy of either would describe an earlier return. `ctx` is the run's own
33
+ * context object, passed through rather than re-derived, because
34
+ * `selectContextModel` WRITES two of its fields (`contextModel` and
35
+ * `activeProviderContextWindow`) for the compaction pass to read.
36
+ */
37
+ export interface StepShaping {
38
+ readonly ctx: IterationContext
39
+ readonly latestUserMessage: () => UserMessage | undefined
40
+ readonly steps: () => readonly StepResult[]
41
+ }
42
+
43
+ export function stepContextMessage(content: string) {
44
+ return createRuntimeContextMessage(
45
+ `Current step context (runtime-generated; not a new user request):\n${content}`,
46
+ 'step-context',
47
+ )
48
+ }
49
+
50
+ /** Derived after request projection; never accumulates in canonical history or replaces operator intent. */
51
+ export function appendWorkContext(
52
+ shaping: StepShaping,
53
+ messages: Message[],
54
+ stepNumber: number,
55
+ prepared: PrepareStepResult,
56
+ ): void {
57
+ const { ctx } = shaping
58
+
59
+ const contributions = [
60
+ ctx.completionInbox?.describeOwnedWork(),
61
+ ctx.toolExecutor.describeFileEvidence(messages),
62
+ ].filter((content): content is string => Boolean(content))
63
+ if (contributions.length === 0) return
64
+ let room = stepContext(shaping, stepNumber, prepared).contextBudget?.remainingTokens ?? 0
65
+ // Leave room for the actual task; admit whole contributions, never dangling partial references.
66
+ if (room < 1_500) return
67
+ for (const content of contributions) {
68
+ if (!content || content.length > 8_000) continue
69
+ const message = stepContextMessage(content)
70
+ const tokens = estimateMessageTokens(message)
71
+ if (tokens > Math.min(2_000, room - 1_000)) continue
72
+ messages.push(message)
73
+ room -= tokens
74
+ }
75
+ }
76
+
77
+ export function stepContext(
78
+ shaping: StepShaping,
79
+ stepNumber: number,
80
+ prepared: PrepareStepResult,
81
+ ): PrepareStepContext {
82
+ const { ctx, latestUserMessage, steps } = shaping
83
+
84
+ const model = prepared.model ?? ctx.runConfig.model
85
+ const window = resolveContextWindow(
86
+ ctx.compactionConfig?.contextWindowTokens,
87
+ model,
88
+ model === ctx.runConfig.model
89
+ ? ctx.providerContextWindow
90
+ : model === ctx.contextModel
91
+ ? ctx.activeProviderContextWindow
92
+ : undefined,
93
+ )
94
+ const skills = prepared.skills ? renderSkillsSection([...prepared.skills]) : null
95
+ const preamble = [prepared.system, skills].filter(Boolean).join('\n\n')
96
+ const preparedTokens =
97
+ (preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0) +
98
+ (prepared.context ? estimateMessageTokens(stepContextMessage(prepared.context)) : 0)
99
+ const responseReserve = Math.min(
100
+ prepared.maxResponseTokens ?? ctx.runConfig.maxResponseTokens ?? Math.floor(window.tokens / 4),
101
+ Math.floor(window.tokens / 4),
102
+ )
103
+ return {
104
+ runId: ctx.runMgr.id,
105
+ stepNumber,
106
+ messages: ctx.runMgr.messages,
107
+ ...(ctx.captureRunEvidence ? { captureRunEvidence: ctx.captureRunEvidence } : {}),
108
+ ...(latestUserMessage() ? { latestUserMessage: latestUserMessage() } : {}),
109
+ signal: ctx.abortController.signal,
110
+ contextBudget: {
111
+ windowTokens: window.tokens,
112
+ remainingTokens: Math.max(
113
+ 0,
114
+ Math.floor(window.tokens - measureContext(ctx).tokens - preparedTokens - responseReserve),
115
+ ),
116
+ },
117
+ steps: steps(),
118
+ prepared,
119
+ }
120
+ }
121
+
122
+ /** Refuse the next call on a veto or hook error; do not skip a failed admission check. */
123
+ export async function beforeStep(
124
+ shaping: StepShaping,
125
+ stepNumber: number,
126
+ ): Promise<StepVeto | undefined> {
127
+ const { ctx } = shaping
128
+
129
+ const configured = ctx.beforeStep
130
+ if (!configured) return undefined
131
+ try {
132
+ return (await configured(stepContext(shaping, stepNumber, {}))) ?? undefined
133
+ } catch (err) {
134
+ return { reason: `beforeStep threw: ${toErrorMessage(err)}` }
135
+ }
136
+ }
137
+
138
+ /** Shape the next request. A failed tuning stage is skipped; admission belongs to beforeStep. */
139
+ export async function prepareStep(
140
+ shaping: StepShaping,
141
+ stepNumber: number,
142
+ ): Promise<{
143
+ allowedTools?: string[]
144
+ toolChoice?: ToolChoice
145
+ model?: string
146
+ system?: string
147
+ context?: string
148
+ skills?: readonly Skill[]
149
+ temperature?: number
150
+ maxResponseTokens?: number
151
+ }> {
152
+ const { ctx } = shaping
153
+
154
+ const configured = ctx.prepareStep
155
+ if (!configured) return {}
156
+ const stages = Array.isArray(configured) ? configured : [configured]
157
+
158
+ // Folded in DECLARATION order, each stage seeing what the ones
159
+ // before it decided. A later stage overriding a field is last-writer
160
+ // wins — visibly, because the order is a line in the host's code
161
+ // rather than an accident of install history.
162
+ let result: PrepareStepResult = {}
163
+ for (const stage of stages) {
164
+ const inference = createCallbackInference(
165
+ ctx,
166
+ result.model ?? ctx.runConfig.model,
167
+ 'preparation',
168
+ )
169
+ try {
170
+ const decided = await stage({
171
+ ...stepContext(shaping, stepNumber, result),
172
+ generateText: inference.generateText,
173
+ })
174
+ if (decided) result = { ...result, ...decided }
175
+ await selectContextModel(shaping, result.model ?? ctx.runConfig.model)
176
+ } catch (err) {
177
+ // Skipped, and the rest still run: one broken concern must
178
+ // not silently disable the others it was declared beside.
179
+ ctx.log.error('a prepareStep stage threw — skipping it', {
180
+ [NAMZU.RUN_ID]: ctx.runMgr.id,
181
+ 'namzu.runtime.step_number': stepNumber,
182
+ 'exception.message': toErrorMessage(err),
183
+ })
184
+ // An SDK stage may report availability and validated fallback evidence
185
+ // without exposing its error. Preserve prior decisions and the context budget;
186
+ // ordinary exceptions still contribute nothing to the model request.
187
+ if (err instanceof PreparationContextError && !ctx.abortController.signal.aborted) {
188
+ const room = stepContext(shaping, stepNumber, result).contextBudget?.remainingTokens ?? 0
189
+ if (
190
+ typeof err.context === 'string' &&
191
+ err.context.length > 0 &&
192
+ err.context.length + (result.context ? 2 : 0) <= Math.min(12_000, Math.floor(room))
193
+ )
194
+ result = {
195
+ ...result,
196
+ context: [result.context, err.context].filter(Boolean).join('\n\n'),
197
+ }
198
+ }
199
+ } finally {
200
+ inference.close()
201
+ }
202
+ }
203
+
204
+ const prepared: {
205
+ allowedTools?: string[]
206
+ toolChoice?: ToolChoice
207
+ model?: string
208
+ system?: string
209
+ context?: string
210
+ skills?: readonly Skill[]
211
+ temperature?: number
212
+ maxResponseTokens?: number
213
+ } = {}
214
+
215
+ if (result.activeTools) {
216
+ const known = result.activeTools.filter((name: string) => ctx.tools.has(name))
217
+ const unknown = result.activeTools.filter((name: string) => !ctx.tools.has(name))
218
+ if (unknown.length > 0) {
219
+ // The all-unknown case gets its own sentence because it has its
220
+ // own consequence. Some names dropped narrows the step; ALL of
221
+ // them dropped leaves it able to call nothing — which is the
222
+ // honest reading of "only these tools" when none of them exist,
223
+ // and is not what a reader of "ignoring them" would expect.
224
+ //
225
+ // Widening back to the run's list would be worse: it grants
226
+ // exactly the tools the caller asked to exclude, on the grounds
227
+ // that their own list failed. A step that can call nothing is
228
+ // constrained; a step that can call everything is a control
229
+ // that stopped applying.
230
+ const message =
231
+ known.length === 0
232
+ ? 'prepareStep named only tools that are not registered — this step can call nothing'
233
+ : 'prepareStep named tools that are not registered — ignoring them'
234
+ ctx.log.warn(message, {
235
+ [NAMZU.RUN_ID]: ctx.runMgr.id,
236
+ 'namzu.runtime.step_number': stepNumber,
237
+ 'namzu.runtime.unknown': unknown,
238
+ 'namzu.runtime.remaining': known.length,
239
+ })
240
+ }
241
+ prepared.allowedTools = known
242
+ }
243
+ if (result.toolChoice !== undefined) prepared.toolChoice = result.toolChoice
244
+ if (result.model !== undefined) prepared.model = result.model
245
+ if (result.system !== undefined) prepared.system = result.system
246
+ if (result.context !== undefined) prepared.context = result.context
247
+ if (result.skills !== undefined) prepared.skills = result.skills
248
+ if (result.temperature !== undefined) prepared.temperature = result.temperature
249
+ if (result.maxResponseTokens !== undefined) {
250
+ prepared.maxResponseTokens = result.maxResponseTokens
251
+ }
252
+
253
+ return prepared
254
+ }
255
+
256
+ export async function selectContextModel(
257
+ shaping: StepShaping,
258
+ model: string | undefined,
259
+ ): Promise<void> {
260
+ const { ctx } = shaping
261
+
262
+ if (model !== (ctx.contextModel ?? ctx.runConfig.model)) {
263
+ // A measurement from another tokenizer cannot price the new request.
264
+ ctx.runMgr.clearLastPromptTokens()
265
+ }
266
+ ctx.contextModel = model
267
+ ctx.activeProviderContextWindow =
268
+ model && model !== ctx.runConfig.model && !ctx.compactionConfig?.contextWindowTokens
269
+ ? await ctx.resolveModelContextWindow?.(model)
270
+ : undefined
271
+ }