@namzu/sdk 42.0.0 → 42.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/CHANGELOG.md +174 -0
  2. package/dist/manager/resident/outbox.d.ts +8 -8
  3. package/dist/manager/resident/store.d.ts +4 -4
  4. package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
  5. package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
  6. package/dist/runtime/query/cancelled-before-start.js +152 -0
  7. package/dist/runtime/query/cancelled-before-start.js.map +1 -0
  8. package/dist/runtime/query/checkpoint.d.ts +21 -0
  9. package/dist/runtime/query/checkpoint.d.ts.map +1 -1
  10. package/dist/runtime/query/checkpoint.js +23 -0
  11. package/dist/runtime/query/checkpoint.js.map +1 -1
  12. package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
  13. package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
  14. package/dist/runtime/query/executor/tool-call-admission.js +373 -0
  15. package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
  16. package/dist/runtime/query/executor.d.ts +70 -35
  17. package/dist/runtime/query/executor.d.ts.map +1 -1
  18. package/dist/runtime/query/executor.js +46 -380
  19. package/dist/runtime/query/executor.js.map +1 -1
  20. package/dist/runtime/query/finalize-run.d.ts +55 -0
  21. package/dist/runtime/query/finalize-run.d.ts.map +1 -0
  22. package/dist/runtime/query/finalize-run.js +113 -0
  23. package/dist/runtime/query/finalize-run.js.map +1 -0
  24. package/dist/runtime/query/index.d.ts +4 -9
  25. package/dist/runtime/query/index.d.ts.map +1 -1
  26. package/dist/runtime/query/index.js +238 -893
  27. package/dist/runtime/query/index.js.map +1 -1
  28. package/dist/runtime/query/iteration/index.d.ts +6 -161
  29. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  30. package/dist/runtime/query/iteration/index.js +23 -523
  31. package/dist/runtime/query/iteration/index.js.map +1 -1
  32. package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
  33. package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
  34. package/dist/runtime/query/iteration/outstanding-work.js +365 -0
  35. package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
  36. package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
  37. package/dist/runtime/query/iteration/phases/plan.js +13 -2
  38. package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
  39. package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
  40. package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
  41. package/dist/runtime/query/iteration/step-shaping.js +184 -0
  42. package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
  43. package/dist/runtime/query/prepare-run.d.ts +94 -0
  44. package/dist/runtime/query/prepare-run.d.ts.map +1 -0
  45. package/dist/runtime/query/prepare-run.js +589 -0
  46. package/dist/runtime/query/prepare-run.js.map +1 -0
  47. package/dist/runtime/query/release-run.d.ts +56 -0
  48. package/dist/runtime/query/release-run.d.ts.map +1 -0
  49. package/dist/runtime/query/release-run.js +101 -0
  50. package/dist/runtime/query/release-run.js.map +1 -0
  51. package/dist/runtime/query/resume-pending.d.ts +112 -1
  52. package/dist/runtime/query/resume-pending.d.ts.map +1 -1
  53. package/dist/runtime/query/resume-pending.js +133 -0
  54. package/dist/runtime/query/resume-pending.js.map +1 -1
  55. package/dist/store/evidence/compaction-archive.d.ts +2 -2
  56. package/dist/types/run/config.d.ts +12 -5
  57. package/dist/types/run/config.d.ts.map +1 -1
  58. package/package.json +1 -1
  59. package/src/runtime/query/cancelled-before-start.ts +189 -0
  60. package/src/runtime/query/checkpoint.ts +22 -0
  61. package/src/runtime/query/executor/tool-call-admission.ts +473 -0
  62. package/src/runtime/query/executor.ts +63 -442
  63. package/src/runtime/query/finalize-run.ts +192 -0
  64. package/src/runtime/query/index.ts +270 -1011
  65. package/src/runtime/query/iteration/index.ts +40 -586
  66. package/src/runtime/query/iteration/outstanding-work.ts +386 -0
  67. package/src/runtime/query/iteration/phases/plan.ts +18 -2
  68. package/src/runtime/query/iteration/step-shaping.ts +271 -0
  69. package/src/runtime/query/prepare-run.ts +718 -0
  70. package/src/runtime/query/release-run.ts +168 -0
  71. package/src/runtime/query/resume-pending.ts +158 -0
  72. package/src/types/run/config.ts +12 -5
@@ -1,10 +1,8 @@
1
1
  import { type Span, SpanStatusCode } from '@opentelemetry/api'
2
- import { resolveContextWindow } from '../../../compaction/context-window.js'
3
2
  import {
4
3
  extractFromAssistantMessage,
5
4
  extractFromUserMessage,
6
5
  } from '../../../compaction/extractor.js'
7
- import { estimateMessageTokens } from '../../../compaction/token-estimate.js'
8
6
  import { AUTO_CONTINUATION_USER_MESSAGE } from '../../../constants/continuation.js'
9
7
  import {
10
8
  DEFAULT_STRUCTURED_OUTPUT_RETRIES,
@@ -14,7 +12,6 @@ import { renderSkillsSection } from '../../../persona/assembler.js'
14
12
  import { resolveProviderCapabilities } from '../../../provider/capabilities.js'
15
13
  import { collectChatCompletion } from '../../../provider/collect-chat-completion.js'
16
14
  import { renderToolSchema } from '../../../registry/tool/schema.js'
17
- import { PreparationContextError } from '../../../run/preparation-context-error.js'
18
15
  import { formatCompletionNotification } from '../../../scheduler/completion-inbox.js'
19
16
  import {
20
17
  GENAI,
@@ -24,7 +21,6 @@ import {
24
21
  } from '../../../telemetry/attributes.js'
25
22
  import { getTracer } from '../../../telemetry/runtime-accessors.js'
26
23
  import { STRUCTURED_OUTPUT_TOOL_NAME } from '../../../tools/builtins/structuredOutput.js'
27
- import { DELEGATION_TIMEOUT_MS } from '../../../tools/coordinator/index.js'
28
24
  import type { CostInfo, TokenUsage } from '../../../types/common/index.js'
29
25
  import { NamzuError } from '../../../types/errors/index.js'
30
26
  import type { MessageId } from '../../../types/ids/index.js'
@@ -35,23 +31,17 @@ import {
35
31
  createRuntimeContextMessage,
36
32
  createSystemMessage,
37
33
  } from '../../../types/message/index.js'
38
- import type { ToolChoice } from '../../../types/provider/chat.js'
39
34
  import { classifyProviderError } from '../../../types/provider/errors.js'
40
35
  import type { ChatCompletionResponse } from '../../../types/provider/index.js'
41
36
  import type { AnswerReview, AnswerReviewContext } from '../../../types/run/answer-review.js'
42
37
  import type {
43
- PrepareStepContext,
44
- PrepareStepResult,
45
38
  RunEvent,
46
39
  StepFailure,
47
40
  StepProvenance,
48
41
  StepResult,
49
- StepVeto,
50
42
  StopReason,
51
43
  } from '../../../types/run/index.js'
52
- import type { Skill } from '../../../types/skills/index.js'
53
44
  import type { LLMToolSchema, ToolRegistryContract } from '../../../types/tool/index.js'
54
- import { readPositiveIntEnv } from '../../../utils/env.js'
55
45
  import { toErrorMessage } from '../../../utils/error.js'
56
46
  import { stableDigest } from '../../../utils/hash.js'
57
47
  import { generateMessageId } from '../../../utils/id.js'
@@ -70,8 +60,9 @@ import {
70
60
  markProviderRejectedImage,
71
61
  projectRequestRichContent,
72
62
  } from '../request-rich-content.js'
73
- import { formatJobNote, formatSteeringNote, isOperatorUserMessage } from '../steering.js'
63
+ import { formatSteeringNote, isOperatorUserMessage } from '../steering.js'
74
64
  import { parseNativeCandidate } from './native-output.js'
65
+ import { holdForOutstandingWork, settleOutstandingWork } from './outstanding-work.js'
75
66
  import { runAdvisoryPhase } from './phases/advisory.js'
76
67
  import { runIterationCheckpoint } from './phases/checkpoint.js'
77
68
  import {
@@ -85,6 +76,14 @@ import { runPlanGate } from './phases/plan.js'
85
76
  import { runToolReview } from './phases/tool-review.js'
86
77
  import { refreshWorkingMemory } from './phases/working-memory.js'
87
78
  import { streamWithProviderRejectedImageRecovery } from './provider-rejected-image.js'
79
+ import {
80
+ type StepShaping,
81
+ appendWorkContext,
82
+ beforeStep,
83
+ prepareStep,
84
+ selectContextModel,
85
+ stepContextMessage,
86
+ } from './step-shaping.js'
88
87
  import { streamProviderTurn } from './stream-turn.js'
89
88
 
90
89
  type ReviewRequest = Pick<AnswerReviewContext, 'requestMessages' | 'latestUserMessage'>
@@ -120,105 +119,7 @@ const DEFAULT_ANSWER_REVIEW_LIMIT = 3
120
119
  const CLOSING_RESPONSE_GUIDANCE =
121
120
  'Give a concise response using only what the available evidence supports. Attribute unverified statements to their source instead of presenting them as observed facts. If evidence is missing or conflicting, state what cannot be established. Do not claim unfinished work is complete. Do not request any more tool calls.'
122
121
 
123
- /**
124
- * The share of a run's REMAINING time a settle-hold may take.
125
- *
126
- * The rule is borrowed from `AGENT_MANAGER_DEFAULTS.maxBudgetFraction`, which
127
- * gives a spawned child at most half of what its parent has left: one
128
- * sub-activity may take a share of the remainder, never the remainder. The
129
- * value is written out here rather than imported, because that field is a
130
- * host-tunable knob about TOKEN allocation and coupling the two would let a
131
- * host lowering one silently change the other.
132
- *
133
- * Half, specifically, because the hold is not the last thing the run does.
134
- * Its whole purpose is to put a worker's result where the model can read it,
135
- * and reading it costs a turn. A hold that spent everything remaining would
136
- * deliver a notification into a run with no turn left to act on it — the same
137
- * "the result exists and the model is never told" failure this mechanism was
138
- * built to close, wearing a different costume.
139
- */
140
- const SETTLE_GRACE_FRACTION = 0.5
141
-
142
- /**
143
- * How long a finishing run waits for a background worker it launched.
144
- *
145
- * Derived from the run rather than fixed, because a constant is wrong in both
146
- * directions at once. The 120 seconds this replaces held a run configured for
147
- * a twenty-second timeout open for 120,267 ms — six times its own budget, and
148
- * unreachable by the guard, which only checks between iterations — while on an
149
- * hour-long run it abandoned workers measured at 4m21s, 5m58s and 8m04s, all
150
- * of them well inside the hour the delegation tools themselves declare.
151
- *
152
- * **Bounded by construction, and against the right boundary.** The input is
153
- * time-to-FINALIZE, not time-to-deadline (see
154
- * `GuardCoordinator.remainingBeforeFinalizeMs`). Measuring to the deadline was
155
- * the first attempt and it was wrong in a way that looked safe: a hold cannot
156
- * outlive the deadline either way, but half of the time-to-deadline started
157
- * just under the warning threshold ends at 95% of the budget — so the slice
158
- * that exists for the run to produce a closing answer is half spent waiting
159
- * for the result that answer was supposed to use. Against the finalize point
160
- * the hold cannot reach the reserve at all, which is what makes the guard's
161
- * inability to interrupt a hold a non-issue rather than a smaller issue.
162
- *
163
- * **The floor of zero is a decision, not a clamp artefact.** A run with no
164
- * time left before it must start finishing has no turn in which to read a
165
- * notification, so waiting could only delay a stop that is already due.
166
- * Nothing is lost by it: `CompletionInbox.waitForArrival` returns before it
167
- * looks at its timer when a completion is already in hand, so a zero grace
168
- * still delivers everything that has arrived. No minimum is invented on top,
169
- * because zero is exactly what a run past the threshold should wait — and
170
- * reading the remainder at hold time rather than trusting `forceFinalize`,
171
- * which is sampled at the top of the iteration, is what makes a long iteration
172
- * that crossed the line in between compute it.
173
- *
174
- * **The ceiling is the longest anything in this subsystem waits for a
175
- * delegated worker.** It binds only for a host whose run timeout exceeds
176
- * roughly two and a quarter hours; below that the fraction is smaller.
177
- */
178
- export function settleGraceMs(remainingBeforeFinalizeMs: number): number {
179
- return Math.min(
180
- Math.floor(remainingBeforeFinalizeMs * SETTLE_GRACE_FRACTION),
181
- DELEGATION_TIMEOUT_MS,
182
- )
183
- }
184
-
185
- /**
186
- * The ceiling on the job half of that grace, in milliseconds.
187
- *
188
- * `DELEGATION_TIMEOUT_MS` is the wrong ceiling for a shell job, and the gap
189
- * only opens where it matters most: a run with no `timeoutMs` — the CLI's
190
- * shipping default, `No run deadline by default` — has infinite time before
191
- * it must start finishing, so `settleGraceMs` returns the ceiling flat. For a
192
- * delegated task that is sound, because the hour is the longest the task
193
- * itself may live: the hold cannot outlast the work. A background job has no
194
- * such bound. `tail -f`, a watcher and a dev server all outlive any hold, so
195
- * the same arithmetic parks an interactive session for an hour on a job that
196
- * was never going to exit.
197
- *
198
- * So the job leg gets its own bound, and it is sized to what the wait buys
199
- * rather than to how long a job may live: a turn in which to use the exit.
200
- * A model that already waited its `wait_for_job` bound out and saw nothing is
201
- * not usually two minutes from an exit, and the run ending is not the news
202
- * being lost — with no run in flight the session announces the exit itself
203
- * (`docs/cli/background-jobs.md`, *Learning that it ended*), which is the
204
- * cheaper of the two places to hear it.
205
- */
206
- const DEFAULT_JOB_HOLD_MAX_MS = 2 * 60 * 1000
207
-
208
- /**
209
- * The same share of the run, under {@link DEFAULT_JOB_HOLD_MAX_MS}.
210
- *
211
- * `NAMZU_JOB_HOLD_MAX_MS` overrides the ceiling for a host that wants a
212
- * longer or shorter park, the way `NAMZU_JOB_WAIT_TIMEOUT_MS` overrides
213
- * `wait_for_job`'s own bound — and it is the same parse, so a value that is
214
- * not a positive whole number of milliseconds leaves the default standing
215
- * rather than holding a run for `NaN`. Called here rather than at module
216
- * load, because a host that sets it after import is not ignored.
217
- */
218
- export function awaitedJobGraceMs(remainingBeforeFinalizeMs: number): number {
219
- const ceiling = readPositiveIntEnv('NAMZU_JOB_HOLD_MAX_MS', DEFAULT_JOB_HOLD_MAX_MS)
220
- return Math.min(settleGraceMs(remainingBeforeFinalizeMs), ceiling)
221
- }
122
+ export { awaitedJobGraceMs, settleGraceMs } from './outstanding-work.js'
222
123
 
223
124
  export class IterationOrchestrator {
224
125
  private ctx: IterationContext
@@ -503,7 +404,7 @@ export class IterationOrchestrator {
503
404
  // and so can only speak after the step it disliked has already
504
405
  // run and been paid for; this is the seam a host with a live
505
406
  // rate limit or a revoked tenant actually needs.
506
- const veto = await this.beforeStep(runMgr.currentIteration + 1)
407
+ const veto = await beforeStep(this.stepShaping(), runMgr.currentIteration + 1)
507
408
  // The hook may settle because its run signal was aborted. Stop
508
409
  // before interpreting that settlement as a policy refusal or
509
410
  // counting an iteration that will never reach the provider.
@@ -634,12 +535,12 @@ export class IterationOrchestrator {
634
535
  // whether to keep going; this decides HOW. No-op when the host
635
536
  // supplied no hook.
636
537
  const contextModelBeforePreparation = this.ctx.contextModel ?? model
637
- const step = await this.prepareStep(iterationNum)
538
+ const step = await prepareStep(this.stepShaping(), iterationNum)
638
539
  // Preparation inference belongs to the run, not the main-model step.
639
540
  usageBefore = { ...runMgr.tokenUsage }
640
541
  costBefore = { ...runMgr.costInfo }
641
542
  stepModel = step.model ?? model
642
- await this.selectContextModel(stepModel)
543
+ await selectContextModel(this.stepShaping(), stepModel)
643
544
  // Preserve post-compaction preparation/recall semantics. A changed
644
545
  // model needs a second check against its own window; never replay
645
546
  // host preparation effects merely to rebuild its request guidance.
@@ -723,12 +624,12 @@ export class IterationOrchestrator {
723
624
  const requestHistory = stepPreamble
724
625
  ? [...baseMessages, createSystemMessage(stepPreamble)]
725
626
  : [...baseMessages]
726
- if (step.context) requestHistory.push(this.stepContextMessage(step.context))
627
+ if (step.context) requestHistory.push(stepContextMessage(step.context))
727
628
  const messages = projectRequestRichContent(
728
629
  this.projectObservations(requestHistory),
729
630
  this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES,
730
631
  )
731
- this.appendWorkContext(messages, iterationNum, step)
632
+ appendWorkContext(this.stepShaping(), messages, iterationNum, step)
732
633
  await this.reportUnsupportedToolResults(messages)
733
634
  yield* this.ctx.drainPending()
734
635
 
@@ -1206,7 +1107,9 @@ export class IterationOrchestrator {
1206
1107
  }
1207
1108
  if (outcome === 'accepted') {
1208
1109
  if (!forceFinalize) {
1209
- const changed = yield* this.holdForOutstandingWork(iterationNum, false)
1110
+ const changed = yield* holdForOutstandingWork(this.ctx, iterationNum, false, () =>
1111
+ this.deliverInbound(),
1112
+ )
1210
1113
  const inbound = this.deliverInbound()
1211
1114
  if (changed || inbound > 0) continue
1212
1115
  }
@@ -1376,7 +1279,12 @@ export class IterationOrchestrator {
1376
1279
  // Settling here would throw away the very thing the launch
1377
1280
  // existed to produce: the supervisor said "launched", the
1378
1281
  // worker had not finished, and the run closed over it.
1379
- if (!forceFinalize && (yield* this.holdForOutstandingWork(iterationNum, false))) {
1282
+ if (
1283
+ !forceFinalize &&
1284
+ (yield* holdForOutstandingWork(this.ctx, iterationNum, false, () =>
1285
+ this.deliverInbound(),
1286
+ ))
1287
+ ) {
1380
1288
  continue
1381
1289
  }
1382
1290
 
@@ -1602,7 +1510,11 @@ export class IterationOrchestrator {
1602
1510
  // `maxIterations` and the run's own deadline bound all of
1603
1511
  // it regardless, and a leg with nothing pending never
1604
1512
  // opens a hold at all.
1605
- if (yield* this.holdForOutstandingWork(iterationNum, true)) {
1513
+ if (
1514
+ yield* holdForOutstandingWork(this.ctx, iterationNum, true, () =>
1515
+ this.deliverInbound(),
1516
+ )
1517
+ ) {
1606
1518
  // Remember WHY the next turn exists, so the turn that
1607
1519
  // ends the run can name the host's decision instead of
1608
1520
  // reporting the shape of the last message.
@@ -1825,481 +1737,23 @@ export class IterationOrchestrator {
1825
1737
  }
1826
1738
  }
1827
1739
  } finally {
1828
- this.settleOutstandingWork()
1829
- }
1830
- }
1831
-
1832
- /**
1833
- * Hold the run open for work that has not finished, and deliver it.
1834
- *
1835
- * Returns whether a completion, a job exit or an operator message entered
1836
- * the transcript — the caller continues on `true`, so the model gets a turn
1837
- * to respond. That turn is the entire justification for waiting, which
1838
- * is why only the exits that can still take one call this.
1839
- *
1840
- * Two kinds of work qualify and they are raced together, because a run has
1841
- * one settle point and one grace period to spend at it:
1842
- *
1843
- * - a delegated task the `CompletionInbox` is still expecting;
1844
- * - a background job the model told `wait_for_job` it is waiting on.
1845
- *
1846
- * The job half is deliberately narrow. Intent comes from the wait and from
1847
- * nothing else — a dev server the model started and never waited on is
1848
- * running because somebody wanted it running, and a hold for it would add
1849
- * the grace period to the end of every turn for the rest of the session.
1850
- *
1851
- * Each leg is opened only when it has something pending: both
1852
- * `waitForArrival` implementations resolve immediately when their own side
1853
- * is idle, so racing an idle one would end the hold before it began.
1854
- *
1855
- * Bounded by `settleGraceMs` and by `maxIterations`, so work that never
1856
- * finishes cannot keep the run open. On a run with a deadline the grace is
1857
- * a share of what is LEFT of it rather than a fresh allowance, so a
1858
- * `wait_for_job` call that already spent minutes has shortened this hold
1859
- * by the same minutes. On a run without one — the CLI's default — there is
1860
- * no remainder to take a share of, and the job leg's own ceiling
1861
- * (`awaitedJobGraceMs`) is what keeps a timed-out wait from being followed
1862
- * by an hour of silence.
1863
- */
1864
- private async *holdForOutstandingWork(
1865
- iterationNum: number,
1866
- hasToolCalls: boolean,
1867
- ): AsyncGenerator<RunEvent, boolean> {
1868
- const inbox = this.ctx.completionInbox?.hasPendingWork ? this.ctx.completionInbox : undefined
1869
- const jobs = this.ctx.awaitedJobs?.hasPendingWork ? this.ctx.awaitedJobs : undefined
1870
- if (!inbox && !jobs) return false
1871
-
1872
- // Read HERE rather than from `forceFinalize`, which was sampled at the
1873
- // top of the iteration: one that has since crossed the finalize point
1874
- // must not open a wait against a reserve it has already entered.
1875
- const remainingMs = this.ctx.guard.remainingBeforeFinalizeMs()
1876
- // One deadline for the race, and it is the LONGEST ceiling any pending
1877
- // leg justifies. A leg resolving on its own timer ends the whole race,
1878
- // so handing the job leg its shorter ceiling while a task was also
1879
- // outstanding would cut the task's hold down to the job's — a run
1880
- // walking away from a worker it had time for, because a job happened
1881
- // to be running. A job therefore never shortens a wait, and it never
1882
- // lengthens one either: where a task is outstanding too, that is how
1883
- // long this run was waiting anyway.
1884
- const graceMs = inbox ? settleGraceMs(remainingMs) : awaitedJobGraceMs(remainingMs)
1885
- this.ctx.log.info('Holding the run open for outstanding work', {
1886
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1887
- [NAMZU.ITERATION]: iterationNum,
1888
- 'namzu.runtime.grace_ms': graceMs,
1889
- 'namzu.runtime.awaited_jobs': jobs?.outstandingJobIds ?? [],
1890
- })
1891
- // User input releases this wait without cancelling any child. Both waits
1892
- // share a disposable signal so the losing arrival listener cannot leak.
1893
- const waiting = new AbortController()
1894
- const runSignal = this.ctx.abortController.signal
1895
- const cancelWait = () => waiting.abort(runSignal.reason)
1896
- runSignal.addEventListener('abort', cancelWait, { once: true })
1897
- if (runSignal.aborted) cancelWait()
1898
- try {
1899
- await Promise.race([
1900
- ...(inbox ? [inbox.waitForArrival(graceMs, waiting.signal)] : []),
1901
- ...(jobs ? [jobs.waitForArrival(graceMs, waiting.signal)] : []),
1902
- ...(this.ctx.waitForInbound ? [this.ctx.waitForInbound(waiting.signal)] : []),
1903
- ])
1904
- } catch (error) {
1905
- if (!runSignal.aborted) throw error
1906
- } finally {
1907
- waiting.abort()
1908
- runSignal.removeEventListener('abort', cancelWait)
1909
- }
1910
- runSignal.throwIfAborted()
1911
-
1912
- const arrived = this.ctx.completionInbox?.drain() ?? []
1913
- if (arrived.length > 0) {
1914
- this.ctx.runMgr.pushMessage(
1915
- createRuntimeContextMessage(formatCompletionNotification(arrived), 'task-completion'),
1916
- )
1917
- }
1918
- const exited = this.deliverAwaitedJobExits()
1919
- const inbound = this.deliverInbound()
1920
- if (arrived.length === 0 && !exited && inbound === 0) return false
1921
- await this.ctx.emitEvent({
1922
- type: 'iteration_completed',
1923
- runId: this.ctx.runMgr.id,
1924
- iteration: iterationNum,
1925
- hasToolCalls,
1926
- })
1927
- yield* this.ctx.drainPending()
1928
- return true
1929
- }
1930
-
1931
- /**
1932
- * Put the job exits this hold was waiting for in front of the model.
1933
- *
1934
- * Through `jobNotices`, which is the channel a job exit already travels on
1935
- * — `attachNotice` rides it out on the next tool result — rather than a
1936
- * second one built for this path. A turn that called no tools has no such
1937
- * result, so the queued text becomes a `runtime-context` message instead,
1938
- * exactly as `deliverInbound` does for steering that found no tool result
1939
- * to attach to.
1940
- *
1941
- * That drain is also what keeps one exit from being delivered twice: the
1942
- * channel hands its text over once, so an exit already attached to a tool
1943
- * result earlier in the turn leaves nothing here — and the record of it
1944
- * went with that delivery, so this returns `false` rather than buying a
1945
- * turn to re-read what the model has read.
1946
- *
1947
- * `takeDelivery` is what pairs the two. Taking the exits first and then
1948
- * finding no notice would discard them, which is the one way this path
1949
- * can lose an exit outright; neither is taken unless both are there.
1950
- *
1951
- * The channel is not per-job, so the text taken here can include a notice
1952
- * for a job nobody awaited that ended while the hold was open. Delivering
1953
- * it is right — it is unread either way, and the alternative is stranding
1954
- * it — but it is not a reason to WAIT, which is why what opens this hold
1955
- * is `AwaitedJobs`, and the two are asked separately.
1956
- */
1957
- private deliverAwaitedJobExits(): boolean {
1958
- const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain())
1959
- if (!delivered) return false
1960
-
1961
- this.ctx.log.info('Delivering a background job exit the run held open for', {
1962
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1963
- 'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
1964
- })
1965
- this.ctx.runMgr.pushMessage(
1966
- createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'),
1967
- )
1968
- return true
1969
- }
1970
-
1971
- /**
1972
- * Account for outstanding work on the way out: deliver what arrived, and
1973
- * say what did not.
1974
- *
1975
- * A run that ends with a worker outstanding must not leave the impression
1976
- * that the worker's result was delivered. There are exactly two honest
1977
- * outcomes and this does both:
1978
- *
1979
- * - **What has already arrived is delivered.** It makes no false claim,
1980
- * and dropping it is pure loss — the message rides out on
1981
- * `Run.messages`, so a host reads it and the next turn of a continued
1982
- * thread starts with it. This does NOT wait: a hold buys the model a
1983
- * turn in which to USE a result, and on an exit whose answer is already
1984
- * decided there is no such turn, so waiting would delay a settled answer
1985
- * to append text this run will not read. The bounded hold stays where it
1986
- * was, on the exits that do have a turn left.
1987
- * - **What is still running is NAMED, not cancelled.** Giving up on a wait
1988
- * is a statement about the waiter, not about the work — the rule
1989
- * `wait-with-idle-bound.ts` already states for the same subsystem — and
1990
- * "the parent answered early" is a weaker warrant for killing a child
1991
- * than "the clock ran out", not a stronger one. Killing a worker that
1992
- * may be mid-write is a policy only the host can judge, and it has
1993
- * `cancel_task` and the run controller to judge it with.
1994
- */
1995
- private settleOutstandingWork(): void {
1996
- this.deliverArrivedCompletions()
1997
- this.deliverArrivedJobExits()
1998
- this.recordAbandonedWork()
1999
- }
2000
-
2001
- /** Work this run walked away from. See {@link settleOutstandingWork}. */
2002
- private recordAbandonedWork(): void {
2003
- const abandoned = this.ctx.completionInbox?.outstandingTaskIds ?? []
2004
- if (abandoned.length > 0) {
2005
- this.ctx.log.warn('Run ended with delegated work still running', {
2006
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2007
- 'namzu.runtime.tasks': abandoned,
2008
- })
2009
- this.ctx.runMgr.setAbandonedTaskIds(abandoned)
1740
+ settleOutstandingWork(this.ctx)
2010
1741
  }
2011
-
2012
- // The same statement for a job the model was waiting on when the grace
2013
- // ran out. Only awaited ones: a job nobody waited for was never work
2014
- // this run was holding, so naming it would report an abandonment that
2015
- // did not happen.
2016
- const abandonedJobs = this.ctx.awaitedJobs?.outstandingJobIds ?? []
2017
- if (abandonedJobs.length === 0) return
2018
-
2019
- this.ctx.log.warn('Run ended with an awaited background job still running', {
2020
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2021
- 'namzu.runtime.jobs': abandonedJobs,
2022
- })
2023
- this.ctx.runMgr.setAbandonedJobIds(abandonedJobs)
2024
- }
2025
-
2026
- private deliverArrivedCompletions(): void {
2027
- const unheard = this.ctx.completionInbox?.drain() ?? []
2028
- if (unheard.length === 0) return
2029
-
2030
- // Fix the run's answer BEFORE appending anything after it.
2031
- //
2032
- // `RunPersistence.resolveResult` walks the message tail backwards and
2033
- // stops at the first non-assistant message, and it runs at
2034
- // `markCompleted` — which is AFTER this. So a notification appended
2035
- // after the final assistant turn makes the run's own answer
2036
- // unreachable. Measured, on a run whose model had just said "THIS IS
2037
- // THE RUN ANSWER.": `run.result` came back `undefined`. That trades a
2038
- // lost worker result for a lost RUN result, which is strictly worse
2039
- // than the defect this delivery exists to fix.
2040
- //
2041
- // Materialising resolves it while the tail is still the assistant's;
2042
- // pinning it means the later re-resolution cannot undo the fix. Only
2043
- // when there is something to pin: on the cancelled and thrown paths
2044
- // there may be no answer, and pinning an empty string there would
2045
- // suppress whatever the error path assembles.
2046
- const answer = this.ctx.runMgr.materializeResult()
2047
- if (answer.length > 0) this.ctx.runMgr.setResult(answer)
2048
-
2049
- this.ctx.log.info('Delivering task completions the run would have settled over', {
2050
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2051
- 'namzu.runtime.tasks': unheard.map((h) => h.taskId),
2052
- })
2053
- this.ctx.runMgr.pushMessage(
2054
- createRuntimeContextMessage(formatCompletionNotification(unheard), 'task-completion'),
2055
- )
2056
1742
  }
2057
1743
 
2058
1744
  /**
2059
- * The job half of {@link deliverArrivedCompletions}: an exit that arrived
2060
- * too late to earn a turn is still delivered on the way out.
1745
+ * The context and the two live reads the step-shaping helpers share.
2061
1746
  *
2062
- * The window this closes is one tick wide and it is nobody else's. An
2063
- * awaited job that exits between the hold's grace expiring and the run
2064
- * settling was never delivered — the hold had already looked — and is no
2065
- * longer named either, because the exit took it off the outstanding list
2066
- * on its way past, so `abandonedJobIds` would be lying to claim it. The
2067
- * host's own listener is no help: the CLI queues an exit for the next
2068
- * turn only when no run is in flight, and this one is still in flight.
2069
- * Delivered here it reaches `Run.messages`, so the transcript has it and
2070
- * a continued thread opens with it.
2071
- *
2072
- * Before `recordAbandonedWork`, which then reports only what is still
2073
- * running, and after `deliverArrivedCompletions`, so the two appended
2074
- * messages land in the order the work finished in.
1747
+ * Built per call rather than held: `latestUserMessage` is replaced on
1748
+ * every operator turn and `steps` grows by one per step, so a captured
1749
+ * value would describe an earlier return.
2075
1750
  */
2076
- private deliverArrivedJobExits(): void {
2077
- const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain())
2078
- if (!delivered) return
2079
-
2080
- // Fix the run's answer BEFORE appending anything after it — the same
2081
- // `resolveResult` tail walk `deliverArrivedCompletions` explains just
2082
- // above, and the same guard against pinning an empty one.
2083
- const answer = this.ctx.runMgr.materializeResult()
2084
- if (answer.length > 0) this.ctx.runMgr.setResult(answer)
2085
-
2086
- this.ctx.log.info('Delivering a background job exit the run would have settled over', {
2087
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2088
- 'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
2089
- })
2090
- this.ctx.runMgr.pushMessage(
2091
- createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'),
2092
- )
2093
- }
2094
-
2095
- private stepContextMessage(content: string) {
2096
- return createRuntimeContextMessage(
2097
- `Current step context (runtime-generated; not a new user request):\n${content}`,
2098
- 'step-context',
2099
- )
2100
- }
2101
-
2102
- /** Derived after request projection; never accumulates in canonical history or replaces operator intent. */
2103
- private appendWorkContext(
2104
- messages: Message[],
2105
- stepNumber: number,
2106
- prepared: PrepareStepResult,
2107
- ): void {
2108
- const contributions = [
2109
- this.ctx.completionInbox?.describeOwnedWork(),
2110
- this.ctx.toolExecutor.describeFileEvidence(messages),
2111
- ].filter((content): content is string => Boolean(content))
2112
- if (contributions.length === 0) return
2113
- let room = this.stepContext(stepNumber, prepared).contextBudget?.remainingTokens ?? 0
2114
- // Leave room for the actual task; admit whole contributions, never dangling partial references.
2115
- if (room < 1_500) return
2116
- for (const content of contributions) {
2117
- if (!content || content.length > 8_000) continue
2118
- const message = this.stepContextMessage(content)
2119
- const tokens = estimateMessageTokens(message)
2120
- if (tokens > Math.min(2_000, room - 1_000)) continue
2121
- messages.push(message)
2122
- room -= tokens
2123
- }
2124
- }
2125
-
2126
- private stepContext(stepNumber: number, prepared: PrepareStepResult): PrepareStepContext {
2127
- const model = prepared.model ?? this.ctx.runConfig.model
2128
- const window = resolveContextWindow(
2129
- this.ctx.compactionConfig?.contextWindowTokens,
2130
- model,
2131
- model === this.ctx.runConfig.model
2132
- ? this.ctx.providerContextWindow
2133
- : model === this.ctx.contextModel
2134
- ? this.ctx.activeProviderContextWindow
2135
- : undefined,
2136
- )
2137
- const skills = prepared.skills ? renderSkillsSection([...prepared.skills]) : null
2138
- const preamble = [prepared.system, skills].filter(Boolean).join('\n\n')
2139
- const preparedTokens =
2140
- (preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0) +
2141
- (prepared.context ? estimateMessageTokens(this.stepContextMessage(prepared.context)) : 0)
2142
- const responseReserve = Math.min(
2143
- prepared.maxResponseTokens ??
2144
- this.ctx.runConfig.maxResponseTokens ??
2145
- Math.floor(window.tokens / 4),
2146
- Math.floor(window.tokens / 4),
2147
- )
1751
+ private stepShaping(): StepShaping {
2148
1752
  return {
2149
- runId: this.ctx.runMgr.id,
2150
- stepNumber,
2151
- messages: this.ctx.runMgr.messages,
2152
- ...(this.ctx.captureRunEvidence ? { captureRunEvidence: this.ctx.captureRunEvidence } : {}),
2153
- ...(this.latestUserMessage ? { latestUserMessage: this.latestUserMessage } : {}),
2154
- signal: this.ctx.abortController.signal,
2155
- contextBudget: {
2156
- windowTokens: window.tokens,
2157
- remainingTokens: Math.max(
2158
- 0,
2159
- Math.floor(
2160
- window.tokens - measureContext(this.ctx).tokens - preparedTokens - responseReserve,
2161
- ),
2162
- ),
2163
- },
2164
- steps: this.steps,
2165
- prepared,
2166
- }
2167
- }
2168
-
2169
- /** Refuse the next call on a veto or hook error; do not skip a failed admission check. */
2170
- private async beforeStep(stepNumber: number): Promise<StepVeto | undefined> {
2171
- const configured = this.ctx.beforeStep
2172
- if (!configured) return undefined
2173
- try {
2174
- return (await configured(this.stepContext(stepNumber, {}))) ?? undefined
2175
- } catch (err) {
2176
- return { reason: `beforeStep threw: ${toErrorMessage(err)}` }
2177
- }
2178
- }
2179
-
2180
- /** Shape the next request. A failed tuning stage is skipped; admission belongs to beforeStep. */
2181
- private async prepareStep(stepNumber: number): Promise<{
2182
- allowedTools?: string[]
2183
- toolChoice?: ToolChoice
2184
- model?: string
2185
- system?: string
2186
- context?: string
2187
- skills?: readonly Skill[]
2188
- temperature?: number
2189
- maxResponseTokens?: number
2190
- }> {
2191
- const configured = this.ctx.prepareStep
2192
- if (!configured) return {}
2193
- const stages = Array.isArray(configured) ? configured : [configured]
2194
-
2195
- // Folded in DECLARATION order, each stage seeing what the ones
2196
- // before it decided. A later stage overriding a field is last-writer
2197
- // wins — visibly, because the order is a line in the host's code
2198
- // rather than an accident of install history.
2199
- let result: PrepareStepResult = {}
2200
- for (const stage of stages) {
2201
- const inference = createCallbackInference(
2202
- this.ctx,
2203
- result.model ?? this.ctx.runConfig.model,
2204
- 'preparation',
2205
- )
2206
- try {
2207
- const decided = await stage({
2208
- ...this.stepContext(stepNumber, result),
2209
- generateText: inference.generateText,
2210
- })
2211
- if (decided) result = { ...result, ...decided }
2212
- await this.selectContextModel(result.model ?? this.ctx.runConfig.model)
2213
- } catch (err) {
2214
- // Skipped, and the rest still run: one broken concern must
2215
- // not silently disable the others it was declared beside.
2216
- this.ctx.log.error('a prepareStep stage threw — skipping it', {
2217
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2218
- 'namzu.runtime.step_number': stepNumber,
2219
- 'exception.message': toErrorMessage(err),
2220
- })
2221
- // An SDK stage may report availability and validated fallback evidence
2222
- // without exposing its error. Preserve prior decisions and the context budget;
2223
- // ordinary exceptions still contribute nothing to the model request.
2224
- if (err instanceof PreparationContextError && !this.ctx.abortController.signal.aborted) {
2225
- const room = this.stepContext(stepNumber, result).contextBudget?.remainingTokens ?? 0
2226
- if (
2227
- typeof err.context === 'string' &&
2228
- err.context.length > 0 &&
2229
- err.context.length + (result.context ? 2 : 0) <= Math.min(12_000, Math.floor(room))
2230
- )
2231
- result = {
2232
- ...result,
2233
- context: [result.context, err.context].filter(Boolean).join('\n\n'),
2234
- }
2235
- }
2236
- } finally {
2237
- inference.close()
2238
- }
2239
- }
2240
-
2241
- const prepared: {
2242
- allowedTools?: string[]
2243
- toolChoice?: ToolChoice
2244
- model?: string
2245
- system?: string
2246
- context?: string
2247
- skills?: readonly Skill[]
2248
- temperature?: number
2249
- maxResponseTokens?: number
2250
- } = {}
2251
-
2252
- if (result.activeTools) {
2253
- const known = result.activeTools.filter((name: string) => this.ctx.tools.has(name))
2254
- const unknown = result.activeTools.filter((name: string) => !this.ctx.tools.has(name))
2255
- if (unknown.length > 0) {
2256
- // The all-unknown case gets its own sentence because it has its
2257
- // own consequence. Some names dropped narrows the step; ALL of
2258
- // them dropped leaves it able to call nothing — which is the
2259
- // honest reading of "only these tools" when none of them exist,
2260
- // and is not what a reader of "ignoring them" would expect.
2261
- //
2262
- // Widening back to the run's list would be worse: it grants
2263
- // exactly the tools the caller asked to exclude, on the grounds
2264
- // that their own list failed. A step that can call nothing is
2265
- // constrained; a step that can call everything is a control
2266
- // that stopped applying.
2267
- const message =
2268
- known.length === 0
2269
- ? 'prepareStep named only tools that are not registered — this step can call nothing'
2270
- : 'prepareStep named tools that are not registered — ignoring them'
2271
- this.ctx.log.warn(message, {
2272
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
2273
- 'namzu.runtime.step_number': stepNumber,
2274
- 'namzu.runtime.unknown': unknown,
2275
- 'namzu.runtime.remaining': known.length,
2276
- })
2277
- }
2278
- prepared.allowedTools = known
2279
- }
2280
- if (result.toolChoice !== undefined) prepared.toolChoice = result.toolChoice
2281
- if (result.model !== undefined) prepared.model = result.model
2282
- if (result.system !== undefined) prepared.system = result.system
2283
- if (result.context !== undefined) prepared.context = result.context
2284
- if (result.skills !== undefined) prepared.skills = result.skills
2285
- if (result.temperature !== undefined) prepared.temperature = result.temperature
2286
- if (result.maxResponseTokens !== undefined) {
2287
- prepared.maxResponseTokens = result.maxResponseTokens
2288
- }
2289
-
2290
- return prepared
2291
- }
2292
-
2293
- private async selectContextModel(model: string | undefined): Promise<void> {
2294
- if (model !== (this.ctx.contextModel ?? this.ctx.runConfig.model)) {
2295
- // A measurement from another tokenizer cannot price the new request.
2296
- this.ctx.runMgr.clearLastPromptTokens()
1753
+ ctx: this.ctx,
1754
+ latestUserMessage: () => this.latestUserMessage,
1755
+ steps: () => this.steps,
2297
1756
  }
2298
- this.ctx.contextModel = model
2299
- this.ctx.activeProviderContextWindow =
2300
- model && model !== this.ctx.runConfig.model && !this.ctx.compactionConfig?.contextWindowTokens
2301
- ? await this.ctx.resolveModelContextWindow?.(model)
2302
- : undefined
2303
1757
  }
2304
1758
 
2305
1759
  /** Steps completed so far, exposed on the returned `Run`. */
@@ -2805,7 +2259,7 @@ export class IterationOrchestrator {
2805
2259
  this.projectObservations(finalHistory),
2806
2260
  this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES,
2807
2261
  )
2808
- this.appendWorkContext(finalMessages, this.steps.length + 1, { model })
2262
+ appendWorkContext(this.stepShaping(), finalMessages, this.steps.length + 1, { model })
2809
2263
  await this.reportUnsupportedToolResults(finalMessages)
2810
2264
 
2811
2265
  // Same cache discipline as the forced-final iteration: keep the