@namzu/sdk 42.0.0 → 42.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/CHANGELOG.md +174 -0
  2. package/dist/manager/resident/outbox.d.ts +8 -8
  3. package/dist/manager/resident/store.d.ts +4 -4
  4. package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
  5. package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
  6. package/dist/runtime/query/cancelled-before-start.js +152 -0
  7. package/dist/runtime/query/cancelled-before-start.js.map +1 -0
  8. package/dist/runtime/query/checkpoint.d.ts +21 -0
  9. package/dist/runtime/query/checkpoint.d.ts.map +1 -1
  10. package/dist/runtime/query/checkpoint.js +23 -0
  11. package/dist/runtime/query/checkpoint.js.map +1 -1
  12. package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
  13. package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
  14. package/dist/runtime/query/executor/tool-call-admission.js +373 -0
  15. package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
  16. package/dist/runtime/query/executor.d.ts +70 -35
  17. package/dist/runtime/query/executor.d.ts.map +1 -1
  18. package/dist/runtime/query/executor.js +46 -380
  19. package/dist/runtime/query/executor.js.map +1 -1
  20. package/dist/runtime/query/finalize-run.d.ts +55 -0
  21. package/dist/runtime/query/finalize-run.d.ts.map +1 -0
  22. package/dist/runtime/query/finalize-run.js +113 -0
  23. package/dist/runtime/query/finalize-run.js.map +1 -0
  24. package/dist/runtime/query/index.d.ts +4 -9
  25. package/dist/runtime/query/index.d.ts.map +1 -1
  26. package/dist/runtime/query/index.js +238 -893
  27. package/dist/runtime/query/index.js.map +1 -1
  28. package/dist/runtime/query/iteration/index.d.ts +6 -161
  29. package/dist/runtime/query/iteration/index.d.ts.map +1 -1
  30. package/dist/runtime/query/iteration/index.js +23 -523
  31. package/dist/runtime/query/iteration/index.js.map +1 -1
  32. package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
  33. package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
  34. package/dist/runtime/query/iteration/outstanding-work.js +365 -0
  35. package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
  36. package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
  37. package/dist/runtime/query/iteration/phases/plan.js +13 -2
  38. package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
  39. package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
  40. package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
  41. package/dist/runtime/query/iteration/step-shaping.js +184 -0
  42. package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
  43. package/dist/runtime/query/prepare-run.d.ts +94 -0
  44. package/dist/runtime/query/prepare-run.d.ts.map +1 -0
  45. package/dist/runtime/query/prepare-run.js +589 -0
  46. package/dist/runtime/query/prepare-run.js.map +1 -0
  47. package/dist/runtime/query/release-run.d.ts +56 -0
  48. package/dist/runtime/query/release-run.d.ts.map +1 -0
  49. package/dist/runtime/query/release-run.js +101 -0
  50. package/dist/runtime/query/release-run.js.map +1 -0
  51. package/dist/runtime/query/resume-pending.d.ts +112 -1
  52. package/dist/runtime/query/resume-pending.d.ts.map +1 -1
  53. package/dist/runtime/query/resume-pending.js +133 -0
  54. package/dist/runtime/query/resume-pending.js.map +1 -1
  55. package/dist/store/evidence/compaction-archive.d.ts +2 -2
  56. package/dist/types/run/config.d.ts +12 -5
  57. package/dist/types/run/config.d.ts.map +1 -1
  58. package/package.json +1 -1
  59. package/src/runtime/query/cancelled-before-start.ts +189 -0
  60. package/src/runtime/query/checkpoint.ts +22 -0
  61. package/src/runtime/query/executor/tool-call-admission.ts +473 -0
  62. package/src/runtime/query/executor.ts +63 -442
  63. package/src/runtime/query/finalize-run.ts +192 -0
  64. package/src/runtime/query/index.ts +270 -1011
  65. package/src/runtime/query/iteration/index.ts +40 -586
  66. package/src/runtime/query/iteration/outstanding-work.ts +386 -0
  67. package/src/runtime/query/iteration/phases/plan.ts +18 -2
  68. package/src/runtime/query/iteration/step-shaping.ts +271 -0
  69. package/src/runtime/query/prepare-run.ts +718 -0
  70. package/src/runtime/query/release-run.ts +168 -0
  71. package/src/runtime/query/resume-pending.ts +158 -0
  72. package/src/types/run/config.ts +12 -5
@@ -1,23 +1,18 @@
1
1
  import { SpanStatusCode } from '@opentelemetry/api';
2
- import { resolveContextWindow } from '../../../compaction/context-window.js';
3
2
  import { extractFromAssistantMessage, extractFromUserMessage, } from '../../../compaction/extractor.js';
4
- import { estimateMessageTokens } from '../../../compaction/token-estimate.js';
5
3
  import { AUTO_CONTINUATION_USER_MESSAGE } from '../../../constants/continuation.js';
6
4
  import { DEFAULT_STRUCTURED_OUTPUT_RETRIES, STRUCTURED_OUTPUT_REPROMPT, } from '../../../constants/tools/index.js';
7
5
  import { renderSkillsSection } from '../../../persona/assembler.js';
8
6
  import { resolveProviderCapabilities } from '../../../provider/capabilities.js';
9
7
  import { collectChatCompletion } from '../../../provider/collect-chat-completion.js';
10
8
  import { renderToolSchema } from '../../../registry/tool/schema.js';
11
- import { PreparationContextError } from '../../../run/preparation-context-error.js';
12
9
  import { formatCompletionNotification } from '../../../scheduler/completion-inbox.js';
13
10
  import { GENAI, NAMZU, agentIterationSpanName, parentContext, } from '../../../telemetry/attributes.js';
14
11
  import { getTracer } from '../../../telemetry/runtime-accessors.js';
15
12
  import { STRUCTURED_OUTPUT_TOOL_NAME } from '../../../tools/builtins/structuredOutput.js';
16
- import { DELEGATION_TIMEOUT_MS } from '../../../tools/coordinator/index.js';
17
13
  import { NamzuError } from '../../../types/errors/index.js';
18
14
  import { createAssistantMessage, createRuntimeContextMessage, createSystemMessage, } from '../../../types/message/index.js';
19
15
  import { classifyProviderError } from '../../../types/provider/errors.js';
20
- import { readPositiveIntEnv } from '../../../utils/env.js';
21
16
  import { toErrorMessage } from '../../../utils/error.js';
22
17
  import { stableDigest } from '../../../utils/hash.js';
23
18
  import { generateMessageId } from '../../../utils/id.js';
@@ -26,8 +21,9 @@ import { projectObservationContext } from '../observation-context.js';
26
21
  import { applyLifecycleHookResults } from '../plugin-hooks.js';
27
22
  import { diffRequestContext, snapshotRequestContext, } from '../request-context.js';
28
23
  import { DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES, markProviderRejectedImage, projectRequestRichContent, } from '../request-rich-content.js';
29
- import { formatJobNote, formatSteeringNote, isOperatorUserMessage } from '../steering.js';
24
+ import { formatSteeringNote, isOperatorUserMessage } from '../steering.js';
30
25
  import { parseNativeCandidate } from './native-output.js';
26
+ import { holdForOutstandingWork, settleOutstandingWork } from './outstanding-work.js';
31
27
  import { runAdvisoryPhase } from './phases/advisory.js';
32
28
  import { runIterationCheckpoint } from './phases/checkpoint.js';
33
29
  import { activeContextWindow, measureContext, relieveOverflow, runCompactionCheck, } from './phases/compaction.js';
@@ -35,6 +31,7 @@ import { runPlanGate } from './phases/plan.js';
35
31
  import { runToolReview } from './phases/tool-review.js';
36
32
  import { refreshWorkingMemory } from './phases/working-memory.js';
37
33
  import { streamWithProviderRejectedImageRecovery } from './provider-rejected-image.js';
34
+ import { appendWorkContext, beforeStep, prepareStep, selectContextModel, stepContextMessage, } from './step-shaping.js';
38
35
  import { streamProviderTurn } from './stream-turn.js';
39
36
  /** A host reviewer is not the model transport, even when its cause is an HTTP failure. */
40
37
  class AnswerReviewFailure extends NamzuError {
@@ -60,99 +57,7 @@ const DEFAULT_ANSWER_REVIEW_LIMIT = 3;
60
57
  // Ending a run changes the available actions, not the strength of its evidence.
61
58
  // Use the same standard for warning closure and empty-completion recovery.
62
59
  const CLOSING_RESPONSE_GUIDANCE = 'Give a concise response using only what the available evidence supports. Attribute unverified statements to their source instead of presenting them as observed facts. If evidence is missing or conflicting, state what cannot be established. Do not claim unfinished work is complete. Do not request any more tool calls.';
63
- /**
64
- * The share of a run's REMAINING time a settle-hold may take.
65
- *
66
- * The rule is borrowed from `AGENT_MANAGER_DEFAULTS.maxBudgetFraction`, which
67
- * gives a spawned child at most half of what its parent has left: one
68
- * sub-activity may take a share of the remainder, never the remainder. The
69
- * value is written out here rather than imported, because that field is a
70
- * host-tunable knob about TOKEN allocation and coupling the two would let a
71
- * host lowering one silently change the other.
72
- *
73
- * Half, specifically, because the hold is not the last thing the run does.
74
- * Its whole purpose is to put a worker's result where the model can read it,
75
- * and reading it costs a turn. A hold that spent everything remaining would
76
- * deliver a notification into a run with no turn left to act on it — the same
77
- * "the result exists and the model is never told" failure this mechanism was
78
- * built to close, wearing a different costume.
79
- */
80
- const SETTLE_GRACE_FRACTION = 0.5;
81
- /**
82
- * How long a finishing run waits for a background worker it launched.
83
- *
84
- * Derived from the run rather than fixed, because a constant is wrong in both
85
- * directions at once. The 120 seconds this replaces held a run configured for
86
- * a twenty-second timeout open for 120,267 ms — six times its own budget, and
87
- * unreachable by the guard, which only checks between iterations — while on an
88
- * hour-long run it abandoned workers measured at 4m21s, 5m58s and 8m04s, all
89
- * of them well inside the hour the delegation tools themselves declare.
90
- *
91
- * **Bounded by construction, and against the right boundary.** The input is
92
- * time-to-FINALIZE, not time-to-deadline (see
93
- * `GuardCoordinator.remainingBeforeFinalizeMs`). Measuring to the deadline was
94
- * the first attempt and it was wrong in a way that looked safe: a hold cannot
95
- * outlive the deadline either way, but half of the time-to-deadline started
96
- * just under the warning threshold ends at 95% of the budget — so the slice
97
- * that exists for the run to produce a closing answer is half spent waiting
98
- * for the result that answer was supposed to use. Against the finalize point
99
- * the hold cannot reach the reserve at all, which is what makes the guard's
100
- * inability to interrupt a hold a non-issue rather than a smaller issue.
101
- *
102
- * **The floor of zero is a decision, not a clamp artefact.** A run with no
103
- * time left before it must start finishing has no turn in which to read a
104
- * notification, so waiting could only delay a stop that is already due.
105
- * Nothing is lost by it: `CompletionInbox.waitForArrival` returns before it
106
- * looks at its timer when a completion is already in hand, so a zero grace
107
- * still delivers everything that has arrived. No minimum is invented on top,
108
- * because zero is exactly what a run past the threshold should wait — and
109
- * reading the remainder at hold time rather than trusting `forceFinalize`,
110
- * which is sampled at the top of the iteration, is what makes a long iteration
111
- * that crossed the line in between compute it.
112
- *
113
- * **The ceiling is the longest anything in this subsystem waits for a
114
- * delegated worker.** It binds only for a host whose run timeout exceeds
115
- * roughly two and a quarter hours; below that the fraction is smaller.
116
- */
117
- export function settleGraceMs(remainingBeforeFinalizeMs) {
118
- return Math.min(Math.floor(remainingBeforeFinalizeMs * SETTLE_GRACE_FRACTION), DELEGATION_TIMEOUT_MS);
119
- }
120
- /**
121
- * The ceiling on the job half of that grace, in milliseconds.
122
- *
123
- * `DELEGATION_TIMEOUT_MS` is the wrong ceiling for a shell job, and the gap
124
- * only opens where it matters most: a run with no `timeoutMs` — the CLI's
125
- * shipping default, `No run deadline by default` — has infinite time before
126
- * it must start finishing, so `settleGraceMs` returns the ceiling flat. For a
127
- * delegated task that is sound, because the hour is the longest the task
128
- * itself may live: the hold cannot outlast the work. A background job has no
129
- * such bound. `tail -f`, a watcher and a dev server all outlive any hold, so
130
- * the same arithmetic parks an interactive session for an hour on a job that
131
- * was never going to exit.
132
- *
133
- * So the job leg gets its own bound, and it is sized to what the wait buys
134
- * rather than to how long a job may live: a turn in which to use the exit.
135
- * A model that already waited its `wait_for_job` bound out and saw nothing is
136
- * not usually two minutes from an exit, and the run ending is not the news
137
- * being lost — with no run in flight the session announces the exit itself
138
- * (`docs/cli/background-jobs.md`, *Learning that it ended*), which is the
139
- * cheaper of the two places to hear it.
140
- */
141
- const DEFAULT_JOB_HOLD_MAX_MS = 2 * 60 * 1000;
142
- /**
143
- * The same share of the run, under {@link DEFAULT_JOB_HOLD_MAX_MS}.
144
- *
145
- * `NAMZU_JOB_HOLD_MAX_MS` overrides the ceiling for a host that wants a
146
- * longer or shorter park, the way `NAMZU_JOB_WAIT_TIMEOUT_MS` overrides
147
- * `wait_for_job`'s own bound — and it is the same parse, so a value that is
148
- * not a positive whole number of milliseconds leaves the default standing
149
- * rather than holding a run for `NaN`. Called here rather than at module
150
- * load, because a host that sets it after import is not ignored.
151
- */
152
- export function awaitedJobGraceMs(remainingBeforeFinalizeMs) {
153
- const ceiling = readPositiveIntEnv('NAMZU_JOB_HOLD_MAX_MS', DEFAULT_JOB_HOLD_MAX_MS);
154
- return Math.min(settleGraceMs(remainingBeforeFinalizeMs), ceiling);
155
- }
60
+ export { awaitedJobGraceMs, settleGraceMs } from './outstanding-work.js';
156
61
  export class IterationOrchestrator {
157
62
  ctx;
158
63
  advisoryTurn;
@@ -405,7 +310,7 @@ export class IterationOrchestrator {
405
310
  // and so can only speak after the step it disliked has already
406
311
  // run and been paid for; this is the seam a host with a live
407
312
  // rate limit or a revoked tenant actually needs.
408
- const veto = await this.beforeStep(runMgr.currentIteration + 1);
313
+ const veto = await beforeStep(this.stepShaping(), runMgr.currentIteration + 1);
409
314
  // The hook may settle because its run signal was aborted. Stop
410
315
  // before interpreting that settlement as a policy refusal or
411
316
  // counting an iteration that will never reach the provider.
@@ -517,12 +422,12 @@ export class IterationOrchestrator {
517
422
  // whether to keep going; this decides HOW. No-op when the host
518
423
  // supplied no hook.
519
424
  const contextModelBeforePreparation = this.ctx.contextModel ?? model;
520
- const step = await this.prepareStep(iterationNum);
425
+ const step = await prepareStep(this.stepShaping(), iterationNum);
521
426
  // Preparation inference belongs to the run, not the main-model step.
522
427
  usageBefore = { ...runMgr.tokenUsage };
523
428
  costBefore = { ...runMgr.costInfo };
524
429
  stepModel = step.model ?? model;
525
- await this.selectContextModel(stepModel);
430
+ await selectContextModel(this.stepShaping(), stepModel);
526
431
  // Preserve post-compaction preparation/recall semantics. A changed
527
432
  // model needs a second check against its own window; never replay
528
433
  // host preparation effects merely to rebuild its request guidance.
@@ -597,9 +502,9 @@ export class IterationOrchestrator {
597
502
  ? [...baseMessages, createSystemMessage(stepPreamble)]
598
503
  : [...baseMessages];
599
504
  if (step.context)
600
- requestHistory.push(this.stepContextMessage(step.context));
505
+ requestHistory.push(stepContextMessage(step.context));
601
506
  const messages = projectRequestRichContent(this.projectObservations(requestHistory), this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES);
602
- this.appendWorkContext(messages, iterationNum, step);
507
+ appendWorkContext(this.stepShaping(), messages, iterationNum, step);
603
508
  await this.reportUnsupportedToolResults(messages);
604
509
  yield* this.ctx.drainPending();
605
510
  // What the model is about to be ASKED, recorded when it
@@ -1001,7 +906,7 @@ export class IterationOrchestrator {
1001
906
  }
1002
907
  if (outcome === 'accepted') {
1003
908
  if (!forceFinalize) {
1004
- const changed = yield* this.holdForOutstandingWork(iterationNum, false);
909
+ const changed = yield* holdForOutstandingWork(this.ctx, iterationNum, false, () => this.deliverInbound());
1005
910
  const inbound = this.deliverInbound();
1006
911
  if (changed || inbound > 0)
1007
912
  continue;
@@ -1156,7 +1061,8 @@ export class IterationOrchestrator {
1156
1061
  // Settling here would throw away the very thing the launch
1157
1062
  // existed to produce: the supervisor said "launched", the
1158
1063
  // worker had not finished, and the run closed over it.
1159
- if (!forceFinalize && (yield* this.holdForOutstandingWork(iterationNum, false))) {
1064
+ if (!forceFinalize &&
1065
+ (yield* holdForOutstandingWork(this.ctx, iterationNum, false, () => this.deliverInbound()))) {
1160
1066
  continue;
1161
1067
  }
1162
1068
  // Anything queued while this turn ran, on the path where
@@ -1364,7 +1270,7 @@ export class IterationOrchestrator {
1364
1270
  // `maxIterations` and the run's own deadline bound all of
1365
1271
  // it regardless, and a leg with nothing pending never
1366
1272
  // opens a hold at all.
1367
- if (yield* this.holdForOutstandingWork(iterationNum, true)) {
1273
+ if (yield* holdForOutstandingWork(this.ctx, iterationNum, true, () => this.deliverInbound())) {
1368
1274
  // Remember WHY the next turn exists, so the turn that
1369
1275
  // ends the run can name the host's decision instead of
1370
1276
  // reporting the shape of the last message.
@@ -1571,429 +1477,23 @@ export class IterationOrchestrator {
1571
1477
  }
1572
1478
  }
1573
1479
  finally {
1574
- this.settleOutstandingWork();
1480
+ settleOutstandingWork(this.ctx);
1575
1481
  }
1576
1482
  }
1577
1483
  /**
1578
- * Hold the run open for work that has not finished, and deliver it.
1579
- *
1580
- * Returns whether a completion, a job exit or an operator message entered
1581
- * the transcript — the caller continues on `true`, so the model gets a turn
1582
- * to respond. That turn is the entire justification for waiting, which
1583
- * is why only the exits that can still take one call this.
1584
- *
1585
- * Two kinds of work qualify and they are raced together, because a run has
1586
- * one settle point and one grace period to spend at it:
1484
+ * The context and the two live reads the step-shaping helpers share.
1587
1485
  *
1588
- * - a delegated task the `CompletionInbox` is still expecting;
1589
- * - a background job the model told `wait_for_job` it is waiting on.
1590
- *
1591
- * The job half is deliberately narrow. Intent comes from the wait and from
1592
- * nothing else — a dev server the model started and never waited on is
1593
- * running because somebody wanted it running, and a hold for it would add
1594
- * the grace period to the end of every turn for the rest of the session.
1595
- *
1596
- * Each leg is opened only when it has something pending: both
1597
- * `waitForArrival` implementations resolve immediately when their own side
1598
- * is idle, so racing an idle one would end the hold before it began.
1599
- *
1600
- * Bounded by `settleGraceMs` and by `maxIterations`, so work that never
1601
- * finishes cannot keep the run open. On a run with a deadline the grace is
1602
- * a share of what is LEFT of it rather than a fresh allowance, so a
1603
- * `wait_for_job` call that already spent minutes has shortened this hold
1604
- * by the same minutes. On a run without one — the CLI's default — there is
1605
- * no remainder to take a share of, and the job leg's own ceiling
1606
- * (`awaitedJobGraceMs`) is what keeps a timed-out wait from being followed
1607
- * by an hour of silence.
1486
+ * Built per call rather than held: `latestUserMessage` is replaced on
1487
+ * every operator turn and `steps` grows by one per step, so a captured
1488
+ * value would describe an earlier return.
1608
1489
  */
1609
- async *holdForOutstandingWork(iterationNum, hasToolCalls) {
1610
- const inbox = this.ctx.completionInbox?.hasPendingWork ? this.ctx.completionInbox : undefined;
1611
- const jobs = this.ctx.awaitedJobs?.hasPendingWork ? this.ctx.awaitedJobs : undefined;
1612
- if (!inbox && !jobs)
1613
- return false;
1614
- // Read HERE rather than from `forceFinalize`, which was sampled at the
1615
- // top of the iteration: one that has since crossed the finalize point
1616
- // must not open a wait against a reserve it has already entered.
1617
- const remainingMs = this.ctx.guard.remainingBeforeFinalizeMs();
1618
- // One deadline for the race, and it is the LONGEST ceiling any pending
1619
- // leg justifies. A leg resolving on its own timer ends the whole race,
1620
- // so handing the job leg its shorter ceiling while a task was also
1621
- // outstanding would cut the task's hold down to the job's — a run
1622
- // walking away from a worker it had time for, because a job happened
1623
- // to be running. A job therefore never shortens a wait, and it never
1624
- // lengthens one either: where a task is outstanding too, that is how
1625
- // long this run was waiting anyway.
1626
- const graceMs = inbox ? settleGraceMs(remainingMs) : awaitedJobGraceMs(remainingMs);
1627
- this.ctx.log.info('Holding the run open for outstanding work', {
1628
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1629
- [NAMZU.ITERATION]: iterationNum,
1630
- 'namzu.runtime.grace_ms': graceMs,
1631
- 'namzu.runtime.awaited_jobs': jobs?.outstandingJobIds ?? [],
1632
- });
1633
- // User input releases this wait without cancelling any child. Both waits
1634
- // share a disposable signal so the losing arrival listener cannot leak.
1635
- const waiting = new AbortController();
1636
- const runSignal = this.ctx.abortController.signal;
1637
- const cancelWait = () => waiting.abort(runSignal.reason);
1638
- runSignal.addEventListener('abort', cancelWait, { once: true });
1639
- if (runSignal.aborted)
1640
- cancelWait();
1641
- try {
1642
- await Promise.race([
1643
- ...(inbox ? [inbox.waitForArrival(graceMs, waiting.signal)] : []),
1644
- ...(jobs ? [jobs.waitForArrival(graceMs, waiting.signal)] : []),
1645
- ...(this.ctx.waitForInbound ? [this.ctx.waitForInbound(waiting.signal)] : []),
1646
- ]);
1647
- }
1648
- catch (error) {
1649
- if (!runSignal.aborted)
1650
- throw error;
1651
- }
1652
- finally {
1653
- waiting.abort();
1654
- runSignal.removeEventListener('abort', cancelWait);
1655
- }
1656
- runSignal.throwIfAborted();
1657
- const arrived = this.ctx.completionInbox?.drain() ?? [];
1658
- if (arrived.length > 0) {
1659
- this.ctx.runMgr.pushMessage(createRuntimeContextMessage(formatCompletionNotification(arrived), 'task-completion'));
1660
- }
1661
- const exited = this.deliverAwaitedJobExits();
1662
- const inbound = this.deliverInbound();
1663
- if (arrived.length === 0 && !exited && inbound === 0)
1664
- return false;
1665
- await this.ctx.emitEvent({
1666
- type: 'iteration_completed',
1667
- runId: this.ctx.runMgr.id,
1668
- iteration: iterationNum,
1669
- hasToolCalls,
1670
- });
1671
- yield* this.ctx.drainPending();
1672
- return true;
1673
- }
1674
- /**
1675
- * Put the job exits this hold was waiting for in front of the model.
1676
- *
1677
- * Through `jobNotices`, which is the channel a job exit already travels on
1678
- * — `attachNotice` rides it out on the next tool result — rather than a
1679
- * second one built for this path. A turn that called no tools has no such
1680
- * result, so the queued text becomes a `runtime-context` message instead,
1681
- * exactly as `deliverInbound` does for steering that found no tool result
1682
- * to attach to.
1683
- *
1684
- * That drain is also what keeps one exit from being delivered twice: the
1685
- * channel hands its text over once, so an exit already attached to a tool
1686
- * result earlier in the turn leaves nothing here — and the record of it
1687
- * went with that delivery, so this returns `false` rather than buying a
1688
- * turn to re-read what the model has read.
1689
- *
1690
- * `takeDelivery` is what pairs the two. Taking the exits first and then
1691
- * finding no notice would discard them, which is the one way this path
1692
- * can lose an exit outright; neither is taken unless both are there.
1693
- *
1694
- * The channel is not per-job, so the text taken here can include a notice
1695
- * for a job nobody awaited that ended while the hold was open. Delivering
1696
- * it is right — it is unread either way, and the alternative is stranding
1697
- * it — but it is not a reason to WAIT, which is why what opens this hold
1698
- * is `AwaitedJobs`, and the two are asked separately.
1699
- */
1700
- deliverAwaitedJobExits() {
1701
- const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain());
1702
- if (!delivered)
1703
- return false;
1704
- this.ctx.log.info('Delivering a background job exit the run held open for', {
1705
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1706
- 'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
1707
- });
1708
- this.ctx.runMgr.pushMessage(createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'));
1709
- return true;
1710
- }
1711
- /**
1712
- * Account for outstanding work on the way out: deliver what arrived, and
1713
- * say what did not.
1714
- *
1715
- * A run that ends with a worker outstanding must not leave the impression
1716
- * that the worker's result was delivered. There are exactly two honest
1717
- * outcomes and this does both:
1718
- *
1719
- * - **What has already arrived is delivered.** It makes no false claim,
1720
- * and dropping it is pure loss — the message rides out on
1721
- * `Run.messages`, so a host reads it and the next turn of a continued
1722
- * thread starts with it. This does NOT wait: a hold buys the model a
1723
- * turn in which to USE a result, and on an exit whose answer is already
1724
- * decided there is no such turn, so waiting would delay a settled answer
1725
- * to append text this run will not read. The bounded hold stays where it
1726
- * was, on the exits that do have a turn left.
1727
- * - **What is still running is NAMED, not cancelled.** Giving up on a wait
1728
- * is a statement about the waiter, not about the work — the rule
1729
- * `wait-with-idle-bound.ts` already states for the same subsystem — and
1730
- * "the parent answered early" is a weaker warrant for killing a child
1731
- * than "the clock ran out", not a stronger one. Killing a worker that
1732
- * may be mid-write is a policy only the host can judge, and it has
1733
- * `cancel_task` and the run controller to judge it with.
1734
- */
1735
- settleOutstandingWork() {
1736
- this.deliverArrivedCompletions();
1737
- this.deliverArrivedJobExits();
1738
- this.recordAbandonedWork();
1739
- }
1740
- /** Work this run walked away from. See {@link settleOutstandingWork}. */
1741
- recordAbandonedWork() {
1742
- const abandoned = this.ctx.completionInbox?.outstandingTaskIds ?? [];
1743
- if (abandoned.length > 0) {
1744
- this.ctx.log.warn('Run ended with delegated work still running', {
1745
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1746
- 'namzu.runtime.tasks': abandoned,
1747
- });
1748
- this.ctx.runMgr.setAbandonedTaskIds(abandoned);
1749
- }
1750
- // The same statement for a job the model was waiting on when the grace
1751
- // ran out. Only awaited ones: a job nobody waited for was never work
1752
- // this run was holding, so naming it would report an abandonment that
1753
- // did not happen.
1754
- const abandonedJobs = this.ctx.awaitedJobs?.outstandingJobIds ?? [];
1755
- if (abandonedJobs.length === 0)
1756
- return;
1757
- this.ctx.log.warn('Run ended with an awaited background job still running', {
1758
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1759
- 'namzu.runtime.jobs': abandonedJobs,
1760
- });
1761
- this.ctx.runMgr.setAbandonedJobIds(abandonedJobs);
1762
- }
1763
- deliverArrivedCompletions() {
1764
- const unheard = this.ctx.completionInbox?.drain() ?? [];
1765
- if (unheard.length === 0)
1766
- return;
1767
- // Fix the run's answer BEFORE appending anything after it.
1768
- //
1769
- // `RunPersistence.resolveResult` walks the message tail backwards and
1770
- // stops at the first non-assistant message, and it runs at
1771
- // `markCompleted` — which is AFTER this. So a notification appended
1772
- // after the final assistant turn makes the run's own answer
1773
- // unreachable. Measured, on a run whose model had just said "THIS IS
1774
- // THE RUN ANSWER.": `run.result` came back `undefined`. That trades a
1775
- // lost worker result for a lost RUN result, which is strictly worse
1776
- // than the defect this delivery exists to fix.
1777
- //
1778
- // Materialising resolves it while the tail is still the assistant's;
1779
- // pinning it means the later re-resolution cannot undo the fix. Only
1780
- // when there is something to pin: on the cancelled and thrown paths
1781
- // there may be no answer, and pinning an empty string there would
1782
- // suppress whatever the error path assembles.
1783
- const answer = this.ctx.runMgr.materializeResult();
1784
- if (answer.length > 0)
1785
- this.ctx.runMgr.setResult(answer);
1786
- this.ctx.log.info('Delivering task completions the run would have settled over', {
1787
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1788
- 'namzu.runtime.tasks': unheard.map((h) => h.taskId),
1789
- });
1790
- this.ctx.runMgr.pushMessage(createRuntimeContextMessage(formatCompletionNotification(unheard), 'task-completion'));
1791
- }
1792
- /**
1793
- * The job half of {@link deliverArrivedCompletions}: an exit that arrived
1794
- * too late to earn a turn is still delivered on the way out.
1795
- *
1796
- * The window this closes is one tick wide and it is nobody else's. An
1797
- * awaited job that exits between the hold's grace expiring and the run
1798
- * settling was never delivered — the hold had already looked — and is no
1799
- * longer named either, because the exit took it off the outstanding list
1800
- * on its way past, so `abandonedJobIds` would be lying to claim it. The
1801
- * host's own listener is no help: the CLI queues an exit for the next
1802
- * turn only when no run is in flight, and this one is still in flight.
1803
- * Delivered here it reaches `Run.messages`, so the transcript has it and
1804
- * a continued thread opens with it.
1805
- *
1806
- * Before `recordAbandonedWork`, which then reports only what is still
1807
- * running, and after `deliverArrivedCompletions`, so the two appended
1808
- * messages land in the order the work finished in.
1809
- */
1810
- deliverArrivedJobExits() {
1811
- const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain());
1812
- if (!delivered)
1813
- return;
1814
- // Fix the run's answer BEFORE appending anything after it — the same
1815
- // `resolveResult` tail walk `deliverArrivedCompletions` explains just
1816
- // above, and the same guard against pinning an empty one.
1817
- const answer = this.ctx.runMgr.materializeResult();
1818
- if (answer.length > 0)
1819
- this.ctx.runMgr.setResult(answer);
1820
- this.ctx.log.info('Delivering a background job exit the run would have settled over', {
1821
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1822
- 'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
1823
- });
1824
- this.ctx.runMgr.pushMessage(createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'));
1825
- }
1826
- stepContextMessage(content) {
1827
- return createRuntimeContextMessage(`Current step context (runtime-generated; not a new user request):\n${content}`, 'step-context');
1828
- }
1829
- /** Derived after request projection; never accumulates in canonical history or replaces operator intent. */
1830
- appendWorkContext(messages, stepNumber, prepared) {
1831
- const contributions = [
1832
- this.ctx.completionInbox?.describeOwnedWork(),
1833
- this.ctx.toolExecutor.describeFileEvidence(messages),
1834
- ].filter((content) => Boolean(content));
1835
- if (contributions.length === 0)
1836
- return;
1837
- let room = this.stepContext(stepNumber, prepared).contextBudget?.remainingTokens ?? 0;
1838
- // Leave room for the actual task; admit whole contributions, never dangling partial references.
1839
- if (room < 1_500)
1840
- return;
1841
- for (const content of contributions) {
1842
- if (!content || content.length > 8_000)
1843
- continue;
1844
- const message = this.stepContextMessage(content);
1845
- const tokens = estimateMessageTokens(message);
1846
- if (tokens > Math.min(2_000, room - 1_000))
1847
- continue;
1848
- messages.push(message);
1849
- room -= tokens;
1850
- }
1851
- }
1852
- stepContext(stepNumber, prepared) {
1853
- const model = prepared.model ?? this.ctx.runConfig.model;
1854
- const window = resolveContextWindow(this.ctx.compactionConfig?.contextWindowTokens, model, model === this.ctx.runConfig.model
1855
- ? this.ctx.providerContextWindow
1856
- : model === this.ctx.contextModel
1857
- ? this.ctx.activeProviderContextWindow
1858
- : undefined);
1859
- const skills = prepared.skills ? renderSkillsSection([...prepared.skills]) : null;
1860
- const preamble = [prepared.system, skills].filter(Boolean).join('\n\n');
1861
- const preparedTokens = (preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0) +
1862
- (prepared.context ? estimateMessageTokens(this.stepContextMessage(prepared.context)) : 0);
1863
- const responseReserve = Math.min(prepared.maxResponseTokens ??
1864
- this.ctx.runConfig.maxResponseTokens ??
1865
- Math.floor(window.tokens / 4), Math.floor(window.tokens / 4));
1490
+ stepShaping() {
1866
1491
  return {
1867
- runId: this.ctx.runMgr.id,
1868
- stepNumber,
1869
- messages: this.ctx.runMgr.messages,
1870
- ...(this.ctx.captureRunEvidence ? { captureRunEvidence: this.ctx.captureRunEvidence } : {}),
1871
- ...(this.latestUserMessage ? { latestUserMessage: this.latestUserMessage } : {}),
1872
- signal: this.ctx.abortController.signal,
1873
- contextBudget: {
1874
- windowTokens: window.tokens,
1875
- remainingTokens: Math.max(0, Math.floor(window.tokens - measureContext(this.ctx).tokens - preparedTokens - responseReserve)),
1876
- },
1877
- steps: this.steps,
1878
- prepared,
1492
+ ctx: this.ctx,
1493
+ latestUserMessage: () => this.latestUserMessage,
1494
+ steps: () => this.steps,
1879
1495
  };
1880
1496
  }
1881
- /** Refuse the next call on a veto or hook error; do not skip a failed admission check. */
1882
- async beforeStep(stepNumber) {
1883
- const configured = this.ctx.beforeStep;
1884
- if (!configured)
1885
- return undefined;
1886
- try {
1887
- return (await configured(this.stepContext(stepNumber, {}))) ?? undefined;
1888
- }
1889
- catch (err) {
1890
- return { reason: `beforeStep threw: ${toErrorMessage(err)}` };
1891
- }
1892
- }
1893
- /** Shape the next request. A failed tuning stage is skipped; admission belongs to beforeStep. */
1894
- async prepareStep(stepNumber) {
1895
- const configured = this.ctx.prepareStep;
1896
- if (!configured)
1897
- return {};
1898
- const stages = Array.isArray(configured) ? configured : [configured];
1899
- // Folded in DECLARATION order, each stage seeing what the ones
1900
- // before it decided. A later stage overriding a field is last-writer
1901
- // wins — visibly, because the order is a line in the host's code
1902
- // rather than an accident of install history.
1903
- let result = {};
1904
- for (const stage of stages) {
1905
- const inference = createCallbackInference(this.ctx, result.model ?? this.ctx.runConfig.model, 'preparation');
1906
- try {
1907
- const decided = await stage({
1908
- ...this.stepContext(stepNumber, result),
1909
- generateText: inference.generateText,
1910
- });
1911
- if (decided)
1912
- result = { ...result, ...decided };
1913
- await this.selectContextModel(result.model ?? this.ctx.runConfig.model);
1914
- }
1915
- catch (err) {
1916
- // Skipped, and the rest still run: one broken concern must
1917
- // not silently disable the others it was declared beside.
1918
- this.ctx.log.error('a prepareStep stage threw — skipping it', {
1919
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1920
- 'namzu.runtime.step_number': stepNumber,
1921
- 'exception.message': toErrorMessage(err),
1922
- });
1923
- // An SDK stage may report availability and validated fallback evidence
1924
- // without exposing its error. Preserve prior decisions and the context budget;
1925
- // ordinary exceptions still contribute nothing to the model request.
1926
- if (err instanceof PreparationContextError && !this.ctx.abortController.signal.aborted) {
1927
- const room = this.stepContext(stepNumber, result).contextBudget?.remainingTokens ?? 0;
1928
- if (typeof err.context === 'string' &&
1929
- err.context.length > 0 &&
1930
- err.context.length + (result.context ? 2 : 0) <= Math.min(12_000, Math.floor(room)))
1931
- result = {
1932
- ...result,
1933
- context: [result.context, err.context].filter(Boolean).join('\n\n'),
1934
- };
1935
- }
1936
- }
1937
- finally {
1938
- inference.close();
1939
- }
1940
- }
1941
- const prepared = {};
1942
- if (result.activeTools) {
1943
- const known = result.activeTools.filter((name) => this.ctx.tools.has(name));
1944
- const unknown = result.activeTools.filter((name) => !this.ctx.tools.has(name));
1945
- if (unknown.length > 0) {
1946
- // The all-unknown case gets its own sentence because it has its
1947
- // own consequence. Some names dropped narrows the step; ALL of
1948
- // them dropped leaves it able to call nothing — which is the
1949
- // honest reading of "only these tools" when none of them exist,
1950
- // and is not what a reader of "ignoring them" would expect.
1951
- //
1952
- // Widening back to the run's list would be worse: it grants
1953
- // exactly the tools the caller asked to exclude, on the grounds
1954
- // that their own list failed. A step that can call nothing is
1955
- // constrained; a step that can call everything is a control
1956
- // that stopped applying.
1957
- const message = known.length === 0
1958
- ? 'prepareStep named only tools that are not registered — this step can call nothing'
1959
- : 'prepareStep named tools that are not registered — ignoring them';
1960
- this.ctx.log.warn(message, {
1961
- [NAMZU.RUN_ID]: this.ctx.runMgr.id,
1962
- 'namzu.runtime.step_number': stepNumber,
1963
- 'namzu.runtime.unknown': unknown,
1964
- 'namzu.runtime.remaining': known.length,
1965
- });
1966
- }
1967
- prepared.allowedTools = known;
1968
- }
1969
- if (result.toolChoice !== undefined)
1970
- prepared.toolChoice = result.toolChoice;
1971
- if (result.model !== undefined)
1972
- prepared.model = result.model;
1973
- if (result.system !== undefined)
1974
- prepared.system = result.system;
1975
- if (result.context !== undefined)
1976
- prepared.context = result.context;
1977
- if (result.skills !== undefined)
1978
- prepared.skills = result.skills;
1979
- if (result.temperature !== undefined)
1980
- prepared.temperature = result.temperature;
1981
- if (result.maxResponseTokens !== undefined) {
1982
- prepared.maxResponseTokens = result.maxResponseTokens;
1983
- }
1984
- return prepared;
1985
- }
1986
- async selectContextModel(model) {
1987
- if (model !== (this.ctx.contextModel ?? this.ctx.runConfig.model)) {
1988
- // A measurement from another tokenizer cannot price the new request.
1989
- this.ctx.runMgr.clearLastPromptTokens();
1990
- }
1991
- this.ctx.contextModel = model;
1992
- this.ctx.activeProviderContextWindow =
1993
- model && model !== this.ctx.runConfig.model && !this.ctx.compactionConfig?.contextWindowTokens
1994
- ? await this.ctx.resolveModelContextWindow?.(model)
1995
- : undefined;
1996
- }
1997
1497
  /** Steps completed so far, exposed on the returned `Run`. */
1998
1498
  steps = [];
1999
1499
  getSteps() {
@@ -2433,7 +1933,7 @@ export class IterationOrchestrator {
2433
1933
  createRuntimeContextMessage(`[SYSTEM] Run is ending due to ${reason}. ${CLOSING_RESPONSE_GUIDANCE}`, 'limit-finalization'),
2434
1934
  ];
2435
1935
  const finalMessages = projectRequestRichContent(this.projectObservations(finalHistory), this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES);
2436
- this.appendWorkContext(finalMessages, this.steps.length + 1, { model });
1936
+ appendWorkContext(this.stepShaping(), finalMessages, this.steps.length + 1, { model });
2437
1937
  await this.reportUnsupportedToolResults(finalMessages);
2438
1938
  // Same cache discipline as the forced-final iteration: keep the
2439
1939
  // tools param identical to prior iterations (cache prefix intact,