@namzu/sdk 42.0.0 → 42.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +174 -0
- package/dist/manager/resident/outbox.d.ts +8 -8
- package/dist/manager/resident/store.d.ts +4 -4
- package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
- package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
- package/dist/runtime/query/cancelled-before-start.js +152 -0
- package/dist/runtime/query/cancelled-before-start.js.map +1 -0
- package/dist/runtime/query/checkpoint.d.ts +21 -0
- package/dist/runtime/query/checkpoint.d.ts.map +1 -1
- package/dist/runtime/query/checkpoint.js +23 -0
- package/dist/runtime/query/checkpoint.js.map +1 -1
- package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
- package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
- package/dist/runtime/query/executor/tool-call-admission.js +373 -0
- package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
- package/dist/runtime/query/executor.d.ts +70 -35
- package/dist/runtime/query/executor.d.ts.map +1 -1
- package/dist/runtime/query/executor.js +46 -380
- package/dist/runtime/query/executor.js.map +1 -1
- package/dist/runtime/query/finalize-run.d.ts +55 -0
- package/dist/runtime/query/finalize-run.d.ts.map +1 -0
- package/dist/runtime/query/finalize-run.js +113 -0
- package/dist/runtime/query/finalize-run.js.map +1 -0
- package/dist/runtime/query/index.d.ts +4 -9
- package/dist/runtime/query/index.d.ts.map +1 -1
- package/dist/runtime/query/index.js +238 -893
- package/dist/runtime/query/index.js.map +1 -1
- package/dist/runtime/query/iteration/index.d.ts +6 -161
- package/dist/runtime/query/iteration/index.d.ts.map +1 -1
- package/dist/runtime/query/iteration/index.js +23 -523
- package/dist/runtime/query/iteration/index.js.map +1 -1
- package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
- package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
- package/dist/runtime/query/iteration/outstanding-work.js +365 -0
- package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
- package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
- package/dist/runtime/query/iteration/phases/plan.js +13 -2
- package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
- package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
- package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
- package/dist/runtime/query/iteration/step-shaping.js +184 -0
- package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
- package/dist/runtime/query/prepare-run.d.ts +94 -0
- package/dist/runtime/query/prepare-run.d.ts.map +1 -0
- package/dist/runtime/query/prepare-run.js +589 -0
- package/dist/runtime/query/prepare-run.js.map +1 -0
- package/dist/runtime/query/release-run.d.ts +56 -0
- package/dist/runtime/query/release-run.d.ts.map +1 -0
- package/dist/runtime/query/release-run.js +101 -0
- package/dist/runtime/query/release-run.js.map +1 -0
- package/dist/runtime/query/resume-pending.d.ts +112 -1
- package/dist/runtime/query/resume-pending.d.ts.map +1 -1
- package/dist/runtime/query/resume-pending.js +133 -0
- package/dist/runtime/query/resume-pending.js.map +1 -1
- package/dist/store/evidence/compaction-archive.d.ts +2 -2
- package/dist/types/run/config.d.ts +12 -5
- package/dist/types/run/config.d.ts.map +1 -1
- package/package.json +1 -1
- package/src/runtime/query/cancelled-before-start.ts +189 -0
- package/src/runtime/query/checkpoint.ts +22 -0
- package/src/runtime/query/executor/tool-call-admission.ts +473 -0
- package/src/runtime/query/executor.ts +63 -442
- package/src/runtime/query/finalize-run.ts +192 -0
- package/src/runtime/query/index.ts +270 -1011
- package/src/runtime/query/iteration/index.ts +40 -586
- package/src/runtime/query/iteration/outstanding-work.ts +386 -0
- package/src/runtime/query/iteration/phases/plan.ts +18 -2
- package/src/runtime/query/iteration/step-shaping.ts +271 -0
- package/src/runtime/query/prepare-run.ts +718 -0
- package/src/runtime/query/release-run.ts +168 -0
- package/src/runtime/query/resume-pending.ts +158 -0
- package/src/types/run/config.ts +12 -5
|
@@ -1,10 +1,8 @@
|
|
|
1
1
|
import { type Span, SpanStatusCode } from '@opentelemetry/api'
|
|
2
|
-
import { resolveContextWindow } from '../../../compaction/context-window.js'
|
|
3
2
|
import {
|
|
4
3
|
extractFromAssistantMessage,
|
|
5
4
|
extractFromUserMessage,
|
|
6
5
|
} from '../../../compaction/extractor.js'
|
|
7
|
-
import { estimateMessageTokens } from '../../../compaction/token-estimate.js'
|
|
8
6
|
import { AUTO_CONTINUATION_USER_MESSAGE } from '../../../constants/continuation.js'
|
|
9
7
|
import {
|
|
10
8
|
DEFAULT_STRUCTURED_OUTPUT_RETRIES,
|
|
@@ -14,7 +12,6 @@ import { renderSkillsSection } from '../../../persona/assembler.js'
|
|
|
14
12
|
import { resolveProviderCapabilities } from '../../../provider/capabilities.js'
|
|
15
13
|
import { collectChatCompletion } from '../../../provider/collect-chat-completion.js'
|
|
16
14
|
import { renderToolSchema } from '../../../registry/tool/schema.js'
|
|
17
|
-
import { PreparationContextError } from '../../../run/preparation-context-error.js'
|
|
18
15
|
import { formatCompletionNotification } from '../../../scheduler/completion-inbox.js'
|
|
19
16
|
import {
|
|
20
17
|
GENAI,
|
|
@@ -24,7 +21,6 @@ import {
|
|
|
24
21
|
} from '../../../telemetry/attributes.js'
|
|
25
22
|
import { getTracer } from '../../../telemetry/runtime-accessors.js'
|
|
26
23
|
import { STRUCTURED_OUTPUT_TOOL_NAME } from '../../../tools/builtins/structuredOutput.js'
|
|
27
|
-
import { DELEGATION_TIMEOUT_MS } from '../../../tools/coordinator/index.js'
|
|
28
24
|
import type { CostInfo, TokenUsage } from '../../../types/common/index.js'
|
|
29
25
|
import { NamzuError } from '../../../types/errors/index.js'
|
|
30
26
|
import type { MessageId } from '../../../types/ids/index.js'
|
|
@@ -35,23 +31,17 @@ import {
|
|
|
35
31
|
createRuntimeContextMessage,
|
|
36
32
|
createSystemMessage,
|
|
37
33
|
} from '../../../types/message/index.js'
|
|
38
|
-
import type { ToolChoice } from '../../../types/provider/chat.js'
|
|
39
34
|
import { classifyProviderError } from '../../../types/provider/errors.js'
|
|
40
35
|
import type { ChatCompletionResponse } from '../../../types/provider/index.js'
|
|
41
36
|
import type { AnswerReview, AnswerReviewContext } from '../../../types/run/answer-review.js'
|
|
42
37
|
import type {
|
|
43
|
-
PrepareStepContext,
|
|
44
|
-
PrepareStepResult,
|
|
45
38
|
RunEvent,
|
|
46
39
|
StepFailure,
|
|
47
40
|
StepProvenance,
|
|
48
41
|
StepResult,
|
|
49
|
-
StepVeto,
|
|
50
42
|
StopReason,
|
|
51
43
|
} from '../../../types/run/index.js'
|
|
52
|
-
import type { Skill } from '../../../types/skills/index.js'
|
|
53
44
|
import type { LLMToolSchema, ToolRegistryContract } from '../../../types/tool/index.js'
|
|
54
|
-
import { readPositiveIntEnv } from '../../../utils/env.js'
|
|
55
45
|
import { toErrorMessage } from '../../../utils/error.js'
|
|
56
46
|
import { stableDigest } from '../../../utils/hash.js'
|
|
57
47
|
import { generateMessageId } from '../../../utils/id.js'
|
|
@@ -70,8 +60,9 @@ import {
|
|
|
70
60
|
markProviderRejectedImage,
|
|
71
61
|
projectRequestRichContent,
|
|
72
62
|
} from '../request-rich-content.js'
|
|
73
|
-
import {
|
|
63
|
+
import { formatSteeringNote, isOperatorUserMessage } from '../steering.js'
|
|
74
64
|
import { parseNativeCandidate } from './native-output.js'
|
|
65
|
+
import { holdForOutstandingWork, settleOutstandingWork } from './outstanding-work.js'
|
|
75
66
|
import { runAdvisoryPhase } from './phases/advisory.js'
|
|
76
67
|
import { runIterationCheckpoint } from './phases/checkpoint.js'
|
|
77
68
|
import {
|
|
@@ -85,6 +76,14 @@ import { runPlanGate } from './phases/plan.js'
|
|
|
85
76
|
import { runToolReview } from './phases/tool-review.js'
|
|
86
77
|
import { refreshWorkingMemory } from './phases/working-memory.js'
|
|
87
78
|
import { streamWithProviderRejectedImageRecovery } from './provider-rejected-image.js'
|
|
79
|
+
import {
|
|
80
|
+
type StepShaping,
|
|
81
|
+
appendWorkContext,
|
|
82
|
+
beforeStep,
|
|
83
|
+
prepareStep,
|
|
84
|
+
selectContextModel,
|
|
85
|
+
stepContextMessage,
|
|
86
|
+
} from './step-shaping.js'
|
|
88
87
|
import { streamProviderTurn } from './stream-turn.js'
|
|
89
88
|
|
|
90
89
|
type ReviewRequest = Pick<AnswerReviewContext, 'requestMessages' | 'latestUserMessage'>
|
|
@@ -120,105 +119,7 @@ const DEFAULT_ANSWER_REVIEW_LIMIT = 3
|
|
|
120
119
|
const CLOSING_RESPONSE_GUIDANCE =
|
|
121
120
|
'Give a concise response using only what the available evidence supports. Attribute unverified statements to their source instead of presenting them as observed facts. If evidence is missing or conflicting, state what cannot be established. Do not claim unfinished work is complete. Do not request any more tool calls.'
|
|
122
121
|
|
|
123
|
-
|
|
124
|
-
* The share of a run's REMAINING time a settle-hold may take.
|
|
125
|
-
*
|
|
126
|
-
* The rule is borrowed from `AGENT_MANAGER_DEFAULTS.maxBudgetFraction`, which
|
|
127
|
-
* gives a spawned child at most half of what its parent has left: one
|
|
128
|
-
* sub-activity may take a share of the remainder, never the remainder. The
|
|
129
|
-
* value is written out here rather than imported, because that field is a
|
|
130
|
-
* host-tunable knob about TOKEN allocation and coupling the two would let a
|
|
131
|
-
* host lowering one silently change the other.
|
|
132
|
-
*
|
|
133
|
-
* Half, specifically, because the hold is not the last thing the run does.
|
|
134
|
-
* Its whole purpose is to put a worker's result where the model can read it,
|
|
135
|
-
* and reading it costs a turn. A hold that spent everything remaining would
|
|
136
|
-
* deliver a notification into a run with no turn left to act on it — the same
|
|
137
|
-
* "the result exists and the model is never told" failure this mechanism was
|
|
138
|
-
* built to close, wearing a different costume.
|
|
139
|
-
*/
|
|
140
|
-
const SETTLE_GRACE_FRACTION = 0.5
|
|
141
|
-
|
|
142
|
-
/**
|
|
143
|
-
* How long a finishing run waits for a background worker it launched.
|
|
144
|
-
*
|
|
145
|
-
* Derived from the run rather than fixed, because a constant is wrong in both
|
|
146
|
-
* directions at once. The 120 seconds this replaces held a run configured for
|
|
147
|
-
* a twenty-second timeout open for 120,267 ms — six times its own budget, and
|
|
148
|
-
* unreachable by the guard, which only checks between iterations — while on an
|
|
149
|
-
* hour-long run it abandoned workers measured at 4m21s, 5m58s and 8m04s, all
|
|
150
|
-
* of them well inside the hour the delegation tools themselves declare.
|
|
151
|
-
*
|
|
152
|
-
* **Bounded by construction, and against the right boundary.** The input is
|
|
153
|
-
* time-to-FINALIZE, not time-to-deadline (see
|
|
154
|
-
* `GuardCoordinator.remainingBeforeFinalizeMs`). Measuring to the deadline was
|
|
155
|
-
* the first attempt and it was wrong in a way that looked safe: a hold cannot
|
|
156
|
-
* outlive the deadline either way, but half of the time-to-deadline started
|
|
157
|
-
* just under the warning threshold ends at 95% of the budget — so the slice
|
|
158
|
-
* that exists for the run to produce a closing answer is half spent waiting
|
|
159
|
-
* for the result that answer was supposed to use. Against the finalize point
|
|
160
|
-
* the hold cannot reach the reserve at all, which is what makes the guard's
|
|
161
|
-
* inability to interrupt a hold a non-issue rather than a smaller issue.
|
|
162
|
-
*
|
|
163
|
-
* **The floor of zero is a decision, not a clamp artefact.** A run with no
|
|
164
|
-
* time left before it must start finishing has no turn in which to read a
|
|
165
|
-
* notification, so waiting could only delay a stop that is already due.
|
|
166
|
-
* Nothing is lost by it: `CompletionInbox.waitForArrival` returns before it
|
|
167
|
-
* looks at its timer when a completion is already in hand, so a zero grace
|
|
168
|
-
* still delivers everything that has arrived. No minimum is invented on top,
|
|
169
|
-
* because zero is exactly what a run past the threshold should wait — and
|
|
170
|
-
* reading the remainder at hold time rather than trusting `forceFinalize`,
|
|
171
|
-
* which is sampled at the top of the iteration, is what makes a long iteration
|
|
172
|
-
* that crossed the line in between compute it.
|
|
173
|
-
*
|
|
174
|
-
* **The ceiling is the longest anything in this subsystem waits for a
|
|
175
|
-
* delegated worker.** It binds only for a host whose run timeout exceeds
|
|
176
|
-
* roughly two and a quarter hours; below that the fraction is smaller.
|
|
177
|
-
*/
|
|
178
|
-
export function settleGraceMs(remainingBeforeFinalizeMs: number): number {
|
|
179
|
-
return Math.min(
|
|
180
|
-
Math.floor(remainingBeforeFinalizeMs * SETTLE_GRACE_FRACTION),
|
|
181
|
-
DELEGATION_TIMEOUT_MS,
|
|
182
|
-
)
|
|
183
|
-
}
|
|
184
|
-
|
|
185
|
-
/**
|
|
186
|
-
* The ceiling on the job half of that grace, in milliseconds.
|
|
187
|
-
*
|
|
188
|
-
* `DELEGATION_TIMEOUT_MS` is the wrong ceiling for a shell job, and the gap
|
|
189
|
-
* only opens where it matters most: a run with no `timeoutMs` — the CLI's
|
|
190
|
-
* shipping default, `No run deadline by default` — has infinite time before
|
|
191
|
-
* it must start finishing, so `settleGraceMs` returns the ceiling flat. For a
|
|
192
|
-
* delegated task that is sound, because the hour is the longest the task
|
|
193
|
-
* itself may live: the hold cannot outlast the work. A background job has no
|
|
194
|
-
* such bound. `tail -f`, a watcher and a dev server all outlive any hold, so
|
|
195
|
-
* the same arithmetic parks an interactive session for an hour on a job that
|
|
196
|
-
* was never going to exit.
|
|
197
|
-
*
|
|
198
|
-
* So the job leg gets its own bound, and it is sized to what the wait buys
|
|
199
|
-
* rather than to how long a job may live: a turn in which to use the exit.
|
|
200
|
-
* A model that already waited its `wait_for_job` bound out and saw nothing is
|
|
201
|
-
* not usually two minutes from an exit, and the run ending is not the news
|
|
202
|
-
* being lost — with no run in flight the session announces the exit itself
|
|
203
|
-
* (`docs/cli/background-jobs.md`, *Learning that it ended*), which is the
|
|
204
|
-
* cheaper of the two places to hear it.
|
|
205
|
-
*/
|
|
206
|
-
const DEFAULT_JOB_HOLD_MAX_MS = 2 * 60 * 1000
|
|
207
|
-
|
|
208
|
-
/**
|
|
209
|
-
* The same share of the run, under {@link DEFAULT_JOB_HOLD_MAX_MS}.
|
|
210
|
-
*
|
|
211
|
-
* `NAMZU_JOB_HOLD_MAX_MS` overrides the ceiling for a host that wants a
|
|
212
|
-
* longer or shorter park, the way `NAMZU_JOB_WAIT_TIMEOUT_MS` overrides
|
|
213
|
-
* `wait_for_job`'s own bound — and it is the same parse, so a value that is
|
|
214
|
-
* not a positive whole number of milliseconds leaves the default standing
|
|
215
|
-
* rather than holding a run for `NaN`. Called here rather than at module
|
|
216
|
-
* load, because a host that sets it after import is not ignored.
|
|
217
|
-
*/
|
|
218
|
-
export function awaitedJobGraceMs(remainingBeforeFinalizeMs: number): number {
|
|
219
|
-
const ceiling = readPositiveIntEnv('NAMZU_JOB_HOLD_MAX_MS', DEFAULT_JOB_HOLD_MAX_MS)
|
|
220
|
-
return Math.min(settleGraceMs(remainingBeforeFinalizeMs), ceiling)
|
|
221
|
-
}
|
|
122
|
+
export { awaitedJobGraceMs, settleGraceMs } from './outstanding-work.js'
|
|
222
123
|
|
|
223
124
|
export class IterationOrchestrator {
|
|
224
125
|
private ctx: IterationContext
|
|
@@ -503,7 +404,7 @@ export class IterationOrchestrator {
|
|
|
503
404
|
// and so can only speak after the step it disliked has already
|
|
504
405
|
// run and been paid for; this is the seam a host with a live
|
|
505
406
|
// rate limit or a revoked tenant actually needs.
|
|
506
|
-
const veto = await this.
|
|
407
|
+
const veto = await beforeStep(this.stepShaping(), runMgr.currentIteration + 1)
|
|
507
408
|
// The hook may settle because its run signal was aborted. Stop
|
|
508
409
|
// before interpreting that settlement as a policy refusal or
|
|
509
410
|
// counting an iteration that will never reach the provider.
|
|
@@ -634,12 +535,12 @@ export class IterationOrchestrator {
|
|
|
634
535
|
// whether to keep going; this decides HOW. No-op when the host
|
|
635
536
|
// supplied no hook.
|
|
636
537
|
const contextModelBeforePreparation = this.ctx.contextModel ?? model
|
|
637
|
-
const step = await this.
|
|
538
|
+
const step = await prepareStep(this.stepShaping(), iterationNum)
|
|
638
539
|
// Preparation inference belongs to the run, not the main-model step.
|
|
639
540
|
usageBefore = { ...runMgr.tokenUsage }
|
|
640
541
|
costBefore = { ...runMgr.costInfo }
|
|
641
542
|
stepModel = step.model ?? model
|
|
642
|
-
await this.
|
|
543
|
+
await selectContextModel(this.stepShaping(), stepModel)
|
|
643
544
|
// Preserve post-compaction preparation/recall semantics. A changed
|
|
644
545
|
// model needs a second check against its own window; never replay
|
|
645
546
|
// host preparation effects merely to rebuild its request guidance.
|
|
@@ -723,12 +624,12 @@ export class IterationOrchestrator {
|
|
|
723
624
|
const requestHistory = stepPreamble
|
|
724
625
|
? [...baseMessages, createSystemMessage(stepPreamble)]
|
|
725
626
|
: [...baseMessages]
|
|
726
|
-
if (step.context) requestHistory.push(
|
|
627
|
+
if (step.context) requestHistory.push(stepContextMessage(step.context))
|
|
727
628
|
const messages = projectRequestRichContent(
|
|
728
629
|
this.projectObservations(requestHistory),
|
|
729
630
|
this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES,
|
|
730
631
|
)
|
|
731
|
-
this.
|
|
632
|
+
appendWorkContext(this.stepShaping(), messages, iterationNum, step)
|
|
732
633
|
await this.reportUnsupportedToolResults(messages)
|
|
733
634
|
yield* this.ctx.drainPending()
|
|
734
635
|
|
|
@@ -1206,7 +1107,9 @@ export class IterationOrchestrator {
|
|
|
1206
1107
|
}
|
|
1207
1108
|
if (outcome === 'accepted') {
|
|
1208
1109
|
if (!forceFinalize) {
|
|
1209
|
-
const changed = yield* this.
|
|
1110
|
+
const changed = yield* holdForOutstandingWork(this.ctx, iterationNum, false, () =>
|
|
1111
|
+
this.deliverInbound(),
|
|
1112
|
+
)
|
|
1210
1113
|
const inbound = this.deliverInbound()
|
|
1211
1114
|
if (changed || inbound > 0) continue
|
|
1212
1115
|
}
|
|
@@ -1376,7 +1279,12 @@ export class IterationOrchestrator {
|
|
|
1376
1279
|
// Settling here would throw away the very thing the launch
|
|
1377
1280
|
// existed to produce: the supervisor said "launched", the
|
|
1378
1281
|
// worker had not finished, and the run closed over it.
|
|
1379
|
-
if (
|
|
1282
|
+
if (
|
|
1283
|
+
!forceFinalize &&
|
|
1284
|
+
(yield* holdForOutstandingWork(this.ctx, iterationNum, false, () =>
|
|
1285
|
+
this.deliverInbound(),
|
|
1286
|
+
))
|
|
1287
|
+
) {
|
|
1380
1288
|
continue
|
|
1381
1289
|
}
|
|
1382
1290
|
|
|
@@ -1602,7 +1510,11 @@ export class IterationOrchestrator {
|
|
|
1602
1510
|
// `maxIterations` and the run's own deadline bound all of
|
|
1603
1511
|
// it regardless, and a leg with nothing pending never
|
|
1604
1512
|
// opens a hold at all.
|
|
1605
|
-
if (
|
|
1513
|
+
if (
|
|
1514
|
+
yield* holdForOutstandingWork(this.ctx, iterationNum, true, () =>
|
|
1515
|
+
this.deliverInbound(),
|
|
1516
|
+
)
|
|
1517
|
+
) {
|
|
1606
1518
|
// Remember WHY the next turn exists, so the turn that
|
|
1607
1519
|
// ends the run can name the host's decision instead of
|
|
1608
1520
|
// reporting the shape of the last message.
|
|
@@ -1825,481 +1737,23 @@ export class IterationOrchestrator {
|
|
|
1825
1737
|
}
|
|
1826
1738
|
}
|
|
1827
1739
|
} finally {
|
|
1828
|
-
this.
|
|
1829
|
-
}
|
|
1830
|
-
}
|
|
1831
|
-
|
|
1832
|
-
/**
|
|
1833
|
-
* Hold the run open for work that has not finished, and deliver it.
|
|
1834
|
-
*
|
|
1835
|
-
* Returns whether a completion, a job exit or an operator message entered
|
|
1836
|
-
* the transcript — the caller continues on `true`, so the model gets a turn
|
|
1837
|
-
* to respond. That turn is the entire justification for waiting, which
|
|
1838
|
-
* is why only the exits that can still take one call this.
|
|
1839
|
-
*
|
|
1840
|
-
* Two kinds of work qualify and they are raced together, because a run has
|
|
1841
|
-
* one settle point and one grace period to spend at it:
|
|
1842
|
-
*
|
|
1843
|
-
* - a delegated task the `CompletionInbox` is still expecting;
|
|
1844
|
-
* - a background job the model told `wait_for_job` it is waiting on.
|
|
1845
|
-
*
|
|
1846
|
-
* The job half is deliberately narrow. Intent comes from the wait and from
|
|
1847
|
-
* nothing else — a dev server the model started and never waited on is
|
|
1848
|
-
* running because somebody wanted it running, and a hold for it would add
|
|
1849
|
-
* the grace period to the end of every turn for the rest of the session.
|
|
1850
|
-
*
|
|
1851
|
-
* Each leg is opened only when it has something pending: both
|
|
1852
|
-
* `waitForArrival` implementations resolve immediately when their own side
|
|
1853
|
-
* is idle, so racing an idle one would end the hold before it began.
|
|
1854
|
-
*
|
|
1855
|
-
* Bounded by `settleGraceMs` and by `maxIterations`, so work that never
|
|
1856
|
-
* finishes cannot keep the run open. On a run with a deadline the grace is
|
|
1857
|
-
* a share of what is LEFT of it rather than a fresh allowance, so a
|
|
1858
|
-
* `wait_for_job` call that already spent minutes has shortened this hold
|
|
1859
|
-
* by the same minutes. On a run without one — the CLI's default — there is
|
|
1860
|
-
* no remainder to take a share of, and the job leg's own ceiling
|
|
1861
|
-
* (`awaitedJobGraceMs`) is what keeps a timed-out wait from being followed
|
|
1862
|
-
* by an hour of silence.
|
|
1863
|
-
*/
|
|
1864
|
-
private async *holdForOutstandingWork(
|
|
1865
|
-
iterationNum: number,
|
|
1866
|
-
hasToolCalls: boolean,
|
|
1867
|
-
): AsyncGenerator<RunEvent, boolean> {
|
|
1868
|
-
const inbox = this.ctx.completionInbox?.hasPendingWork ? this.ctx.completionInbox : undefined
|
|
1869
|
-
const jobs = this.ctx.awaitedJobs?.hasPendingWork ? this.ctx.awaitedJobs : undefined
|
|
1870
|
-
if (!inbox && !jobs) return false
|
|
1871
|
-
|
|
1872
|
-
// Read HERE rather than from `forceFinalize`, which was sampled at the
|
|
1873
|
-
// top of the iteration: one that has since crossed the finalize point
|
|
1874
|
-
// must not open a wait against a reserve it has already entered.
|
|
1875
|
-
const remainingMs = this.ctx.guard.remainingBeforeFinalizeMs()
|
|
1876
|
-
// One deadline for the race, and it is the LONGEST ceiling any pending
|
|
1877
|
-
// leg justifies. A leg resolving on its own timer ends the whole race,
|
|
1878
|
-
// so handing the job leg its shorter ceiling while a task was also
|
|
1879
|
-
// outstanding would cut the task's hold down to the job's — a run
|
|
1880
|
-
// walking away from a worker it had time for, because a job happened
|
|
1881
|
-
// to be running. A job therefore never shortens a wait, and it never
|
|
1882
|
-
// lengthens one either: where a task is outstanding too, that is how
|
|
1883
|
-
// long this run was waiting anyway.
|
|
1884
|
-
const graceMs = inbox ? settleGraceMs(remainingMs) : awaitedJobGraceMs(remainingMs)
|
|
1885
|
-
this.ctx.log.info('Holding the run open for outstanding work', {
|
|
1886
|
-
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
1887
|
-
[NAMZU.ITERATION]: iterationNum,
|
|
1888
|
-
'namzu.runtime.grace_ms': graceMs,
|
|
1889
|
-
'namzu.runtime.awaited_jobs': jobs?.outstandingJobIds ?? [],
|
|
1890
|
-
})
|
|
1891
|
-
// User input releases this wait without cancelling any child. Both waits
|
|
1892
|
-
// share a disposable signal so the losing arrival listener cannot leak.
|
|
1893
|
-
const waiting = new AbortController()
|
|
1894
|
-
const runSignal = this.ctx.abortController.signal
|
|
1895
|
-
const cancelWait = () => waiting.abort(runSignal.reason)
|
|
1896
|
-
runSignal.addEventListener('abort', cancelWait, { once: true })
|
|
1897
|
-
if (runSignal.aborted) cancelWait()
|
|
1898
|
-
try {
|
|
1899
|
-
await Promise.race([
|
|
1900
|
-
...(inbox ? [inbox.waitForArrival(graceMs, waiting.signal)] : []),
|
|
1901
|
-
...(jobs ? [jobs.waitForArrival(graceMs, waiting.signal)] : []),
|
|
1902
|
-
...(this.ctx.waitForInbound ? [this.ctx.waitForInbound(waiting.signal)] : []),
|
|
1903
|
-
])
|
|
1904
|
-
} catch (error) {
|
|
1905
|
-
if (!runSignal.aborted) throw error
|
|
1906
|
-
} finally {
|
|
1907
|
-
waiting.abort()
|
|
1908
|
-
runSignal.removeEventListener('abort', cancelWait)
|
|
1909
|
-
}
|
|
1910
|
-
runSignal.throwIfAborted()
|
|
1911
|
-
|
|
1912
|
-
const arrived = this.ctx.completionInbox?.drain() ?? []
|
|
1913
|
-
if (arrived.length > 0) {
|
|
1914
|
-
this.ctx.runMgr.pushMessage(
|
|
1915
|
-
createRuntimeContextMessage(formatCompletionNotification(arrived), 'task-completion'),
|
|
1916
|
-
)
|
|
1917
|
-
}
|
|
1918
|
-
const exited = this.deliverAwaitedJobExits()
|
|
1919
|
-
const inbound = this.deliverInbound()
|
|
1920
|
-
if (arrived.length === 0 && !exited && inbound === 0) return false
|
|
1921
|
-
await this.ctx.emitEvent({
|
|
1922
|
-
type: 'iteration_completed',
|
|
1923
|
-
runId: this.ctx.runMgr.id,
|
|
1924
|
-
iteration: iterationNum,
|
|
1925
|
-
hasToolCalls,
|
|
1926
|
-
})
|
|
1927
|
-
yield* this.ctx.drainPending()
|
|
1928
|
-
return true
|
|
1929
|
-
}
|
|
1930
|
-
|
|
1931
|
-
/**
|
|
1932
|
-
* Put the job exits this hold was waiting for in front of the model.
|
|
1933
|
-
*
|
|
1934
|
-
* Through `jobNotices`, which is the channel a job exit already travels on
|
|
1935
|
-
* — `attachNotice` rides it out on the next tool result — rather than a
|
|
1936
|
-
* second one built for this path. A turn that called no tools has no such
|
|
1937
|
-
* result, so the queued text becomes a `runtime-context` message instead,
|
|
1938
|
-
* exactly as `deliverInbound` does for steering that found no tool result
|
|
1939
|
-
* to attach to.
|
|
1940
|
-
*
|
|
1941
|
-
* That drain is also what keeps one exit from being delivered twice: the
|
|
1942
|
-
* channel hands its text over once, so an exit already attached to a tool
|
|
1943
|
-
* result earlier in the turn leaves nothing here — and the record of it
|
|
1944
|
-
* went with that delivery, so this returns `false` rather than buying a
|
|
1945
|
-
* turn to re-read what the model has read.
|
|
1946
|
-
*
|
|
1947
|
-
* `takeDelivery` is what pairs the two. Taking the exits first and then
|
|
1948
|
-
* finding no notice would discard them, which is the one way this path
|
|
1949
|
-
* can lose an exit outright; neither is taken unless both are there.
|
|
1950
|
-
*
|
|
1951
|
-
* The channel is not per-job, so the text taken here can include a notice
|
|
1952
|
-
* for a job nobody awaited that ended while the hold was open. Delivering
|
|
1953
|
-
* it is right — it is unread either way, and the alternative is stranding
|
|
1954
|
-
* it — but it is not a reason to WAIT, which is why what opens this hold
|
|
1955
|
-
* is `AwaitedJobs`, and the two are asked separately.
|
|
1956
|
-
*/
|
|
1957
|
-
private deliverAwaitedJobExits(): boolean {
|
|
1958
|
-
const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain())
|
|
1959
|
-
if (!delivered) return false
|
|
1960
|
-
|
|
1961
|
-
this.ctx.log.info('Delivering a background job exit the run held open for', {
|
|
1962
|
-
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
1963
|
-
'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
|
|
1964
|
-
})
|
|
1965
|
-
this.ctx.runMgr.pushMessage(
|
|
1966
|
-
createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'),
|
|
1967
|
-
)
|
|
1968
|
-
return true
|
|
1969
|
-
}
|
|
1970
|
-
|
|
1971
|
-
/**
|
|
1972
|
-
* Account for outstanding work on the way out: deliver what arrived, and
|
|
1973
|
-
* say what did not.
|
|
1974
|
-
*
|
|
1975
|
-
* A run that ends with a worker outstanding must not leave the impression
|
|
1976
|
-
* that the worker's result was delivered. There are exactly two honest
|
|
1977
|
-
* outcomes and this does both:
|
|
1978
|
-
*
|
|
1979
|
-
* - **What has already arrived is delivered.** It makes no false claim,
|
|
1980
|
-
* and dropping it is pure loss — the message rides out on
|
|
1981
|
-
* `Run.messages`, so a host reads it and the next turn of a continued
|
|
1982
|
-
* thread starts with it. This does NOT wait: a hold buys the model a
|
|
1983
|
-
* turn in which to USE a result, and on an exit whose answer is already
|
|
1984
|
-
* decided there is no such turn, so waiting would delay a settled answer
|
|
1985
|
-
* to append text this run will not read. The bounded hold stays where it
|
|
1986
|
-
* was, on the exits that do have a turn left.
|
|
1987
|
-
* - **What is still running is NAMED, not cancelled.** Giving up on a wait
|
|
1988
|
-
* is a statement about the waiter, not about the work — the rule
|
|
1989
|
-
* `wait-with-idle-bound.ts` already states for the same subsystem — and
|
|
1990
|
-
* "the parent answered early" is a weaker warrant for killing a child
|
|
1991
|
-
* than "the clock ran out", not a stronger one. Killing a worker that
|
|
1992
|
-
* may be mid-write is a policy only the host can judge, and it has
|
|
1993
|
-
* `cancel_task` and the run controller to judge it with.
|
|
1994
|
-
*/
|
|
1995
|
-
private settleOutstandingWork(): void {
|
|
1996
|
-
this.deliverArrivedCompletions()
|
|
1997
|
-
this.deliverArrivedJobExits()
|
|
1998
|
-
this.recordAbandonedWork()
|
|
1999
|
-
}
|
|
2000
|
-
|
|
2001
|
-
/** Work this run walked away from. See {@link settleOutstandingWork}. */
|
|
2002
|
-
private recordAbandonedWork(): void {
|
|
2003
|
-
const abandoned = this.ctx.completionInbox?.outstandingTaskIds ?? []
|
|
2004
|
-
if (abandoned.length > 0) {
|
|
2005
|
-
this.ctx.log.warn('Run ended with delegated work still running', {
|
|
2006
|
-
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
2007
|
-
'namzu.runtime.tasks': abandoned,
|
|
2008
|
-
})
|
|
2009
|
-
this.ctx.runMgr.setAbandonedTaskIds(abandoned)
|
|
1740
|
+
settleOutstandingWork(this.ctx)
|
|
2010
1741
|
}
|
|
2011
|
-
|
|
2012
|
-
// The same statement for a job the model was waiting on when the grace
|
|
2013
|
-
// ran out. Only awaited ones: a job nobody waited for was never work
|
|
2014
|
-
// this run was holding, so naming it would report an abandonment that
|
|
2015
|
-
// did not happen.
|
|
2016
|
-
const abandonedJobs = this.ctx.awaitedJobs?.outstandingJobIds ?? []
|
|
2017
|
-
if (abandonedJobs.length === 0) return
|
|
2018
|
-
|
|
2019
|
-
this.ctx.log.warn('Run ended with an awaited background job still running', {
|
|
2020
|
-
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
2021
|
-
'namzu.runtime.jobs': abandonedJobs,
|
|
2022
|
-
})
|
|
2023
|
-
this.ctx.runMgr.setAbandonedJobIds(abandonedJobs)
|
|
2024
|
-
}
|
|
2025
|
-
|
|
2026
|
-
private deliverArrivedCompletions(): void {
|
|
2027
|
-
const unheard = this.ctx.completionInbox?.drain() ?? []
|
|
2028
|
-
if (unheard.length === 0) return
|
|
2029
|
-
|
|
2030
|
-
// Fix the run's answer BEFORE appending anything after it.
|
|
2031
|
-
//
|
|
2032
|
-
// `RunPersistence.resolveResult` walks the message tail backwards and
|
|
2033
|
-
// stops at the first non-assistant message, and it runs at
|
|
2034
|
-
// `markCompleted` — which is AFTER this. So a notification appended
|
|
2035
|
-
// after the final assistant turn makes the run's own answer
|
|
2036
|
-
// unreachable. Measured, on a run whose model had just said "THIS IS
|
|
2037
|
-
// THE RUN ANSWER.": `run.result` came back `undefined`. That trades a
|
|
2038
|
-
// lost worker result for a lost RUN result, which is strictly worse
|
|
2039
|
-
// than the defect this delivery exists to fix.
|
|
2040
|
-
//
|
|
2041
|
-
// Materialising resolves it while the tail is still the assistant's;
|
|
2042
|
-
// pinning it means the later re-resolution cannot undo the fix. Only
|
|
2043
|
-
// when there is something to pin: on the cancelled and thrown paths
|
|
2044
|
-
// there may be no answer, and pinning an empty string there would
|
|
2045
|
-
// suppress whatever the error path assembles.
|
|
2046
|
-
const answer = this.ctx.runMgr.materializeResult()
|
|
2047
|
-
if (answer.length > 0) this.ctx.runMgr.setResult(answer)
|
|
2048
|
-
|
|
2049
|
-
this.ctx.log.info('Delivering task completions the run would have settled over', {
|
|
2050
|
-
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
2051
|
-
'namzu.runtime.tasks': unheard.map((h) => h.taskId),
|
|
2052
|
-
})
|
|
2053
|
-
this.ctx.runMgr.pushMessage(
|
|
2054
|
-
createRuntimeContextMessage(formatCompletionNotification(unheard), 'task-completion'),
|
|
2055
|
-
)
|
|
2056
1742
|
}
|
|
2057
1743
|
|
|
2058
1744
|
/**
|
|
2059
|
-
* The
|
|
2060
|
-
* too late to earn a turn is still delivered on the way out.
|
|
1745
|
+
* The context and the two live reads the step-shaping helpers share.
|
|
2061
1746
|
*
|
|
2062
|
-
*
|
|
2063
|
-
*
|
|
2064
|
-
*
|
|
2065
|
-
* longer named either, because the exit took it off the outstanding list
|
|
2066
|
-
* on its way past, so `abandonedJobIds` would be lying to claim it. The
|
|
2067
|
-
* host's own listener is no help: the CLI queues an exit for the next
|
|
2068
|
-
* turn only when no run is in flight, and this one is still in flight.
|
|
2069
|
-
* Delivered here it reaches `Run.messages`, so the transcript has it and
|
|
2070
|
-
* a continued thread opens with it.
|
|
2071
|
-
*
|
|
2072
|
-
* Before `recordAbandonedWork`, which then reports only what is still
|
|
2073
|
-
* running, and after `deliverArrivedCompletions`, so the two appended
|
|
2074
|
-
* messages land in the order the work finished in.
|
|
1747
|
+
* Built per call rather than held: `latestUserMessage` is replaced on
|
|
1748
|
+
* every operator turn and `steps` grows by one per step, so a captured
|
|
1749
|
+
* value would describe an earlier return.
|
|
2075
1750
|
*/
|
|
2076
|
-
private
|
|
2077
|
-
const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain())
|
|
2078
|
-
if (!delivered) return
|
|
2079
|
-
|
|
2080
|
-
// Fix the run's answer BEFORE appending anything after it — the same
|
|
2081
|
-
// `resolveResult` tail walk `deliverArrivedCompletions` explains just
|
|
2082
|
-
// above, and the same guard against pinning an empty one.
|
|
2083
|
-
const answer = this.ctx.runMgr.materializeResult()
|
|
2084
|
-
if (answer.length > 0) this.ctx.runMgr.setResult(answer)
|
|
2085
|
-
|
|
2086
|
-
this.ctx.log.info('Delivering a background job exit the run would have settled over', {
|
|
2087
|
-
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
2088
|
-
'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
|
|
2089
|
-
})
|
|
2090
|
-
this.ctx.runMgr.pushMessage(
|
|
2091
|
-
createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'),
|
|
2092
|
-
)
|
|
2093
|
-
}
|
|
2094
|
-
|
|
2095
|
-
private stepContextMessage(content: string) {
|
|
2096
|
-
return createRuntimeContextMessage(
|
|
2097
|
-
`Current step context (runtime-generated; not a new user request):\n${content}`,
|
|
2098
|
-
'step-context',
|
|
2099
|
-
)
|
|
2100
|
-
}
|
|
2101
|
-
|
|
2102
|
-
/** Derived after request projection; never accumulates in canonical history or replaces operator intent. */
|
|
2103
|
-
private appendWorkContext(
|
|
2104
|
-
messages: Message[],
|
|
2105
|
-
stepNumber: number,
|
|
2106
|
-
prepared: PrepareStepResult,
|
|
2107
|
-
): void {
|
|
2108
|
-
const contributions = [
|
|
2109
|
-
this.ctx.completionInbox?.describeOwnedWork(),
|
|
2110
|
-
this.ctx.toolExecutor.describeFileEvidence(messages),
|
|
2111
|
-
].filter((content): content is string => Boolean(content))
|
|
2112
|
-
if (contributions.length === 0) return
|
|
2113
|
-
let room = this.stepContext(stepNumber, prepared).contextBudget?.remainingTokens ?? 0
|
|
2114
|
-
// Leave room for the actual task; admit whole contributions, never dangling partial references.
|
|
2115
|
-
if (room < 1_500) return
|
|
2116
|
-
for (const content of contributions) {
|
|
2117
|
-
if (!content || content.length > 8_000) continue
|
|
2118
|
-
const message = this.stepContextMessage(content)
|
|
2119
|
-
const tokens = estimateMessageTokens(message)
|
|
2120
|
-
if (tokens > Math.min(2_000, room - 1_000)) continue
|
|
2121
|
-
messages.push(message)
|
|
2122
|
-
room -= tokens
|
|
2123
|
-
}
|
|
2124
|
-
}
|
|
2125
|
-
|
|
2126
|
-
private stepContext(stepNumber: number, prepared: PrepareStepResult): PrepareStepContext {
|
|
2127
|
-
const model = prepared.model ?? this.ctx.runConfig.model
|
|
2128
|
-
const window = resolveContextWindow(
|
|
2129
|
-
this.ctx.compactionConfig?.contextWindowTokens,
|
|
2130
|
-
model,
|
|
2131
|
-
model === this.ctx.runConfig.model
|
|
2132
|
-
? this.ctx.providerContextWindow
|
|
2133
|
-
: model === this.ctx.contextModel
|
|
2134
|
-
? this.ctx.activeProviderContextWindow
|
|
2135
|
-
: undefined,
|
|
2136
|
-
)
|
|
2137
|
-
const skills = prepared.skills ? renderSkillsSection([...prepared.skills]) : null
|
|
2138
|
-
const preamble = [prepared.system, skills].filter(Boolean).join('\n\n')
|
|
2139
|
-
const preparedTokens =
|
|
2140
|
-
(preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0) +
|
|
2141
|
-
(prepared.context ? estimateMessageTokens(this.stepContextMessage(prepared.context)) : 0)
|
|
2142
|
-
const responseReserve = Math.min(
|
|
2143
|
-
prepared.maxResponseTokens ??
|
|
2144
|
-
this.ctx.runConfig.maxResponseTokens ??
|
|
2145
|
-
Math.floor(window.tokens / 4),
|
|
2146
|
-
Math.floor(window.tokens / 4),
|
|
2147
|
-
)
|
|
1751
|
+
private stepShaping(): StepShaping {
|
|
2148
1752
|
return {
|
|
2149
|
-
|
|
2150
|
-
|
|
2151
|
-
|
|
2152
|
-
...(this.ctx.captureRunEvidence ? { captureRunEvidence: this.ctx.captureRunEvidence } : {}),
|
|
2153
|
-
...(this.latestUserMessage ? { latestUserMessage: this.latestUserMessage } : {}),
|
|
2154
|
-
signal: this.ctx.abortController.signal,
|
|
2155
|
-
contextBudget: {
|
|
2156
|
-
windowTokens: window.tokens,
|
|
2157
|
-
remainingTokens: Math.max(
|
|
2158
|
-
0,
|
|
2159
|
-
Math.floor(
|
|
2160
|
-
window.tokens - measureContext(this.ctx).tokens - preparedTokens - responseReserve,
|
|
2161
|
-
),
|
|
2162
|
-
),
|
|
2163
|
-
},
|
|
2164
|
-
steps: this.steps,
|
|
2165
|
-
prepared,
|
|
2166
|
-
}
|
|
2167
|
-
}
|
|
2168
|
-
|
|
2169
|
-
/** Refuse the next call on a veto or hook error; do not skip a failed admission check. */
|
|
2170
|
-
private async beforeStep(stepNumber: number): Promise<StepVeto | undefined> {
|
|
2171
|
-
const configured = this.ctx.beforeStep
|
|
2172
|
-
if (!configured) return undefined
|
|
2173
|
-
try {
|
|
2174
|
-
return (await configured(this.stepContext(stepNumber, {}))) ?? undefined
|
|
2175
|
-
} catch (err) {
|
|
2176
|
-
return { reason: `beforeStep threw: ${toErrorMessage(err)}` }
|
|
2177
|
-
}
|
|
2178
|
-
}
|
|
2179
|
-
|
|
2180
|
-
/** Shape the next request. A failed tuning stage is skipped; admission belongs to beforeStep. */
|
|
2181
|
-
private async prepareStep(stepNumber: number): Promise<{
|
|
2182
|
-
allowedTools?: string[]
|
|
2183
|
-
toolChoice?: ToolChoice
|
|
2184
|
-
model?: string
|
|
2185
|
-
system?: string
|
|
2186
|
-
context?: string
|
|
2187
|
-
skills?: readonly Skill[]
|
|
2188
|
-
temperature?: number
|
|
2189
|
-
maxResponseTokens?: number
|
|
2190
|
-
}> {
|
|
2191
|
-
const configured = this.ctx.prepareStep
|
|
2192
|
-
if (!configured) return {}
|
|
2193
|
-
const stages = Array.isArray(configured) ? configured : [configured]
|
|
2194
|
-
|
|
2195
|
-
// Folded in DECLARATION order, each stage seeing what the ones
|
|
2196
|
-
// before it decided. A later stage overriding a field is last-writer
|
|
2197
|
-
// wins — visibly, because the order is a line in the host's code
|
|
2198
|
-
// rather than an accident of install history.
|
|
2199
|
-
let result: PrepareStepResult = {}
|
|
2200
|
-
for (const stage of stages) {
|
|
2201
|
-
const inference = createCallbackInference(
|
|
2202
|
-
this.ctx,
|
|
2203
|
-
result.model ?? this.ctx.runConfig.model,
|
|
2204
|
-
'preparation',
|
|
2205
|
-
)
|
|
2206
|
-
try {
|
|
2207
|
-
const decided = await stage({
|
|
2208
|
-
...this.stepContext(stepNumber, result),
|
|
2209
|
-
generateText: inference.generateText,
|
|
2210
|
-
})
|
|
2211
|
-
if (decided) result = { ...result, ...decided }
|
|
2212
|
-
await this.selectContextModel(result.model ?? this.ctx.runConfig.model)
|
|
2213
|
-
} catch (err) {
|
|
2214
|
-
// Skipped, and the rest still run: one broken concern must
|
|
2215
|
-
// not silently disable the others it was declared beside.
|
|
2216
|
-
this.ctx.log.error('a prepareStep stage threw — skipping it', {
|
|
2217
|
-
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
2218
|
-
'namzu.runtime.step_number': stepNumber,
|
|
2219
|
-
'exception.message': toErrorMessage(err),
|
|
2220
|
-
})
|
|
2221
|
-
// An SDK stage may report availability and validated fallback evidence
|
|
2222
|
-
// without exposing its error. Preserve prior decisions and the context budget;
|
|
2223
|
-
// ordinary exceptions still contribute nothing to the model request.
|
|
2224
|
-
if (err instanceof PreparationContextError && !this.ctx.abortController.signal.aborted) {
|
|
2225
|
-
const room = this.stepContext(stepNumber, result).contextBudget?.remainingTokens ?? 0
|
|
2226
|
-
if (
|
|
2227
|
-
typeof err.context === 'string' &&
|
|
2228
|
-
err.context.length > 0 &&
|
|
2229
|
-
err.context.length + (result.context ? 2 : 0) <= Math.min(12_000, Math.floor(room))
|
|
2230
|
-
)
|
|
2231
|
-
result = {
|
|
2232
|
-
...result,
|
|
2233
|
-
context: [result.context, err.context].filter(Boolean).join('\n\n'),
|
|
2234
|
-
}
|
|
2235
|
-
}
|
|
2236
|
-
} finally {
|
|
2237
|
-
inference.close()
|
|
2238
|
-
}
|
|
2239
|
-
}
|
|
2240
|
-
|
|
2241
|
-
const prepared: {
|
|
2242
|
-
allowedTools?: string[]
|
|
2243
|
-
toolChoice?: ToolChoice
|
|
2244
|
-
model?: string
|
|
2245
|
-
system?: string
|
|
2246
|
-
context?: string
|
|
2247
|
-
skills?: readonly Skill[]
|
|
2248
|
-
temperature?: number
|
|
2249
|
-
maxResponseTokens?: number
|
|
2250
|
-
} = {}
|
|
2251
|
-
|
|
2252
|
-
if (result.activeTools) {
|
|
2253
|
-
const known = result.activeTools.filter((name: string) => this.ctx.tools.has(name))
|
|
2254
|
-
const unknown = result.activeTools.filter((name: string) => !this.ctx.tools.has(name))
|
|
2255
|
-
if (unknown.length > 0) {
|
|
2256
|
-
// The all-unknown case gets its own sentence because it has its
|
|
2257
|
-
// own consequence. Some names dropped narrows the step; ALL of
|
|
2258
|
-
// them dropped leaves it able to call nothing — which is the
|
|
2259
|
-
// honest reading of "only these tools" when none of them exist,
|
|
2260
|
-
// and is not what a reader of "ignoring them" would expect.
|
|
2261
|
-
//
|
|
2262
|
-
// Widening back to the run's list would be worse: it grants
|
|
2263
|
-
// exactly the tools the caller asked to exclude, on the grounds
|
|
2264
|
-
// that their own list failed. A step that can call nothing is
|
|
2265
|
-
// constrained; a step that can call everything is a control
|
|
2266
|
-
// that stopped applying.
|
|
2267
|
-
const message =
|
|
2268
|
-
known.length === 0
|
|
2269
|
-
? 'prepareStep named only tools that are not registered — this step can call nothing'
|
|
2270
|
-
: 'prepareStep named tools that are not registered — ignoring them'
|
|
2271
|
-
this.ctx.log.warn(message, {
|
|
2272
|
-
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
2273
|
-
'namzu.runtime.step_number': stepNumber,
|
|
2274
|
-
'namzu.runtime.unknown': unknown,
|
|
2275
|
-
'namzu.runtime.remaining': known.length,
|
|
2276
|
-
})
|
|
2277
|
-
}
|
|
2278
|
-
prepared.allowedTools = known
|
|
2279
|
-
}
|
|
2280
|
-
if (result.toolChoice !== undefined) prepared.toolChoice = result.toolChoice
|
|
2281
|
-
if (result.model !== undefined) prepared.model = result.model
|
|
2282
|
-
if (result.system !== undefined) prepared.system = result.system
|
|
2283
|
-
if (result.context !== undefined) prepared.context = result.context
|
|
2284
|
-
if (result.skills !== undefined) prepared.skills = result.skills
|
|
2285
|
-
if (result.temperature !== undefined) prepared.temperature = result.temperature
|
|
2286
|
-
if (result.maxResponseTokens !== undefined) {
|
|
2287
|
-
prepared.maxResponseTokens = result.maxResponseTokens
|
|
2288
|
-
}
|
|
2289
|
-
|
|
2290
|
-
return prepared
|
|
2291
|
-
}
|
|
2292
|
-
|
|
2293
|
-
private async selectContextModel(model: string | undefined): Promise<void> {
|
|
2294
|
-
if (model !== (this.ctx.contextModel ?? this.ctx.runConfig.model)) {
|
|
2295
|
-
// A measurement from another tokenizer cannot price the new request.
|
|
2296
|
-
this.ctx.runMgr.clearLastPromptTokens()
|
|
1753
|
+
ctx: this.ctx,
|
|
1754
|
+
latestUserMessage: () => this.latestUserMessage,
|
|
1755
|
+
steps: () => this.steps,
|
|
2297
1756
|
}
|
|
2298
|
-
this.ctx.contextModel = model
|
|
2299
|
-
this.ctx.activeProviderContextWindow =
|
|
2300
|
-
model && model !== this.ctx.runConfig.model && !this.ctx.compactionConfig?.contextWindowTokens
|
|
2301
|
-
? await this.ctx.resolveModelContextWindow?.(model)
|
|
2302
|
-
: undefined
|
|
2303
1757
|
}
|
|
2304
1758
|
|
|
2305
1759
|
/** Steps completed so far, exposed on the returned `Run`. */
|
|
@@ -2805,7 +2259,7 @@ export class IterationOrchestrator {
|
|
|
2805
2259
|
this.projectObservations(finalHistory),
|
|
2806
2260
|
this.ctx.runConfig.maxRequestRichContentBytes ?? DEFAULT_MAX_REQUEST_RICH_CONTENT_BYTES,
|
|
2807
2261
|
)
|
|
2808
|
-
this.
|
|
2262
|
+
appendWorkContext(this.stepShaping(), finalMessages, this.steps.length + 1, { model })
|
|
2809
2263
|
await this.reportUnsupportedToolResults(finalMessages)
|
|
2810
2264
|
|
|
2811
2265
|
// Same cache discipline as the forced-final iteration: keep the
|