@namzu/sdk 42.0.0 → 42.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +174 -0
- package/dist/manager/resident/outbox.d.ts +8 -8
- package/dist/manager/resident/store.d.ts +4 -4
- package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
- package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
- package/dist/runtime/query/cancelled-before-start.js +152 -0
- package/dist/runtime/query/cancelled-before-start.js.map +1 -0
- package/dist/runtime/query/checkpoint.d.ts +21 -0
- package/dist/runtime/query/checkpoint.d.ts.map +1 -1
- package/dist/runtime/query/checkpoint.js +23 -0
- package/dist/runtime/query/checkpoint.js.map +1 -1
- package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
- package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
- package/dist/runtime/query/executor/tool-call-admission.js +373 -0
- package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
- package/dist/runtime/query/executor.d.ts +70 -35
- package/dist/runtime/query/executor.d.ts.map +1 -1
- package/dist/runtime/query/executor.js +46 -380
- package/dist/runtime/query/executor.js.map +1 -1
- package/dist/runtime/query/finalize-run.d.ts +55 -0
- package/dist/runtime/query/finalize-run.d.ts.map +1 -0
- package/dist/runtime/query/finalize-run.js +113 -0
- package/dist/runtime/query/finalize-run.js.map +1 -0
- package/dist/runtime/query/index.d.ts +4 -9
- package/dist/runtime/query/index.d.ts.map +1 -1
- package/dist/runtime/query/index.js +238 -893
- package/dist/runtime/query/index.js.map +1 -1
- package/dist/runtime/query/iteration/index.d.ts +6 -161
- package/dist/runtime/query/iteration/index.d.ts.map +1 -1
- package/dist/runtime/query/iteration/index.js +23 -523
- package/dist/runtime/query/iteration/index.js.map +1 -1
- package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
- package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
- package/dist/runtime/query/iteration/outstanding-work.js +365 -0
- package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
- package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
- package/dist/runtime/query/iteration/phases/plan.js +13 -2
- package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
- package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
- package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
- package/dist/runtime/query/iteration/step-shaping.js +184 -0
- package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
- package/dist/runtime/query/prepare-run.d.ts +94 -0
- package/dist/runtime/query/prepare-run.d.ts.map +1 -0
- package/dist/runtime/query/prepare-run.js +589 -0
- package/dist/runtime/query/prepare-run.js.map +1 -0
- package/dist/runtime/query/release-run.d.ts +56 -0
- package/dist/runtime/query/release-run.d.ts.map +1 -0
- package/dist/runtime/query/release-run.js +101 -0
- package/dist/runtime/query/release-run.js.map +1 -0
- package/dist/runtime/query/resume-pending.d.ts +112 -1
- package/dist/runtime/query/resume-pending.d.ts.map +1 -1
- package/dist/runtime/query/resume-pending.js +133 -0
- package/dist/runtime/query/resume-pending.js.map +1 -1
- package/dist/store/evidence/compaction-archive.d.ts +2 -2
- package/dist/types/run/config.d.ts +12 -5
- package/dist/types/run/config.d.ts.map +1 -1
- package/package.json +1 -1
- package/src/runtime/query/cancelled-before-start.ts +189 -0
- package/src/runtime/query/checkpoint.ts +22 -0
- package/src/runtime/query/executor/tool-call-admission.ts +473 -0
- package/src/runtime/query/executor.ts +63 -442
- package/src/runtime/query/finalize-run.ts +192 -0
- package/src/runtime/query/index.ts +270 -1011
- package/src/runtime/query/iteration/index.ts +40 -586
- package/src/runtime/query/iteration/outstanding-work.ts +386 -0
- package/src/runtime/query/iteration/phases/plan.ts +18 -2
- package/src/runtime/query/iteration/step-shaping.ts +271 -0
- package/src/runtime/query/prepare-run.ts +718 -0
- package/src/runtime/query/release-run.ts +168 -0
- package/src/runtime/query/resume-pending.ts +158 -0
- package/src/types/run/config.ts +12 -5
|
@@ -0,0 +1,365 @@
|
|
|
1
|
+
import { NAMZU } from '../../../constants/telemetry/index.js';
|
|
2
|
+
import { formatCompletionNotification } from '../../../scheduler/completion-inbox.js';
|
|
3
|
+
import { DELEGATION_TIMEOUT_MS } from '../../../tools/coordinator/index.js';
|
|
4
|
+
import { createRuntimeContextMessage } from '../../../types/message/index.js';
|
|
5
|
+
import { readPositiveIntEnv } from '../../../utils/env.js';
|
|
6
|
+
import { formatJobNote } from '../steering.js';
|
|
7
|
+
/**
|
|
8
|
+
* The run's settle points: holding open for work that has not finished, and
|
|
9
|
+
* delivering what arrived.
|
|
10
|
+
*
|
|
11
|
+
* Two kinds of work qualify — a delegated task the `CompletionInbox` is still
|
|
12
|
+
* expecting, and a background job the model told `wait_for_job` it is waiting
|
|
13
|
+
* on — and both are raced together, because a run has one settle point and one
|
|
14
|
+
* grace period to spend at it.
|
|
15
|
+
*
|
|
16
|
+
* Everything here reads the iteration context rather than capturing it, and
|
|
17
|
+
* the one thing it cannot read off that context — how this run drains its
|
|
18
|
+
* inbound queue — arrives as an explicit `deliverInbound` input, because the
|
|
19
|
+
* recording of operator intent that goes with it belongs to the orchestrator.
|
|
20
|
+
* `holdForOutstandingWork` is a generator and is reached with `yield*`: its
|
|
21
|
+
* one event must land at the position it landed at when it lived on the class.
|
|
22
|
+
*/
|
|
23
|
+
/**
|
|
24
|
+
* The share of a run's REMAINING time a settle-hold may take.
|
|
25
|
+
*
|
|
26
|
+
* The rule is borrowed from `AGENT_MANAGER_DEFAULTS.maxBudgetFraction`, which
|
|
27
|
+
* gives a spawned child at most half of what its parent has left: one
|
|
28
|
+
* sub-activity may take a share of the remainder, never the remainder. The
|
|
29
|
+
* value is written out here rather than imported, because that field is a
|
|
30
|
+
* host-tunable knob about TOKEN allocation and coupling the two would let a
|
|
31
|
+
* host lowering one silently change the other.
|
|
32
|
+
*
|
|
33
|
+
* Half, specifically, because the hold is not the last thing the run does.
|
|
34
|
+
* Its whole purpose is to put a worker's result where the model can read it,
|
|
35
|
+
* and reading it costs a turn. A hold that spent everything remaining would
|
|
36
|
+
* deliver a notification into a run with no turn left to act on it — the same
|
|
37
|
+
* "the result exists and the model is never told" failure this mechanism was
|
|
38
|
+
* built to close, wearing a different costume.
|
|
39
|
+
*/
|
|
40
|
+
const SETTLE_GRACE_FRACTION = 0.5;
|
|
41
|
+
/**
|
|
42
|
+
* How long a finishing run waits for a background worker it launched.
|
|
43
|
+
*
|
|
44
|
+
* Derived from the run rather than fixed, because a constant is wrong in both
|
|
45
|
+
* directions at once. The 120 seconds this replaces held a run configured for
|
|
46
|
+
* a twenty-second timeout open for 120,267 ms — six times its own budget, and
|
|
47
|
+
* unreachable by the guard, which only checks between iterations — while on an
|
|
48
|
+
* hour-long run it abandoned workers measured at 4m21s, 5m58s and 8m04s, all
|
|
49
|
+
* of them well inside the hour the delegation tools themselves declare.
|
|
50
|
+
*
|
|
51
|
+
* **Bounded by construction, and against the right boundary.** The input is
|
|
52
|
+
* time-to-FINALIZE, not time-to-deadline (see
|
|
53
|
+
* `GuardCoordinator.remainingBeforeFinalizeMs`). Measuring to the deadline was
|
|
54
|
+
* the first attempt and it was wrong in a way that looked safe: a hold cannot
|
|
55
|
+
* outlive the deadline either way, but half of the time-to-deadline started
|
|
56
|
+
* just under the warning threshold ends at 95% of the budget — so the slice
|
|
57
|
+
* that exists for the run to produce a closing answer is half spent waiting
|
|
58
|
+
* for the result that answer was supposed to use. Against the finalize point
|
|
59
|
+
* the hold cannot reach the reserve at all, which is what makes the guard's
|
|
60
|
+
* inability to interrupt a hold a non-issue rather than a smaller issue.
|
|
61
|
+
*
|
|
62
|
+
* **The floor of zero is a decision, not a clamp artefact.** A run with no
|
|
63
|
+
* time left before it must start finishing has no turn in which to read a
|
|
64
|
+
* notification, so waiting could only delay a stop that is already due.
|
|
65
|
+
* Nothing is lost by it: `CompletionInbox.waitForArrival` returns before it
|
|
66
|
+
* looks at its timer when a completion is already in hand, so a zero grace
|
|
67
|
+
* still delivers everything that has arrived. No minimum is invented on top,
|
|
68
|
+
* because zero is exactly what a run past the threshold should wait — and
|
|
69
|
+
* reading the remainder at hold time rather than trusting `forceFinalize`,
|
|
70
|
+
* which is sampled at the top of the iteration, is what makes a long iteration
|
|
71
|
+
* that crossed the line in between compute it.
|
|
72
|
+
*
|
|
73
|
+
* **The ceiling is the longest anything in this subsystem waits for a
|
|
74
|
+
* delegated worker.** It binds only for a host whose run timeout exceeds
|
|
75
|
+
* roughly two and a quarter hours; below that the fraction is smaller.
|
|
76
|
+
*/
|
|
77
|
+
export function settleGraceMs(remainingBeforeFinalizeMs) {
|
|
78
|
+
return Math.min(Math.floor(remainingBeforeFinalizeMs * SETTLE_GRACE_FRACTION), DELEGATION_TIMEOUT_MS);
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* The ceiling on the job half of that grace, in milliseconds.
|
|
82
|
+
*
|
|
83
|
+
* `DELEGATION_TIMEOUT_MS` is the wrong ceiling for a shell job, and the gap
|
|
84
|
+
* only opens where it matters most: a run with no `timeoutMs` — the CLI's
|
|
85
|
+
* shipping default, `No run deadline by default` — has infinite time before
|
|
86
|
+
* it must start finishing, so `settleGraceMs` returns the ceiling flat. For a
|
|
87
|
+
* delegated task that is sound, because the hour is the longest the task
|
|
88
|
+
* itself may live: the hold cannot outlast the work. A background job has no
|
|
89
|
+
* such bound. `tail -f`, a watcher and a dev server all outlive any hold, so
|
|
90
|
+
* the same arithmetic parks an interactive session for an hour on a job that
|
|
91
|
+
* was never going to exit.
|
|
92
|
+
*
|
|
93
|
+
* So the job leg gets its own bound, and it is sized to what the wait buys
|
|
94
|
+
* rather than to how long a job may live: a turn in which to use the exit.
|
|
95
|
+
* A model that already waited its `wait_for_job` bound out and saw nothing is
|
|
96
|
+
* not usually two minutes from an exit, and the run ending is not the news
|
|
97
|
+
* being lost — with no run in flight the session announces the exit itself
|
|
98
|
+
* (`docs/cli/background-jobs.md`, *Learning that it ended*), which is the
|
|
99
|
+
* cheaper of the two places to hear it.
|
|
100
|
+
*/
|
|
101
|
+
const DEFAULT_JOB_HOLD_MAX_MS = 2 * 60 * 1000;
|
|
102
|
+
/**
|
|
103
|
+
* The same share of the run, under {@link DEFAULT_JOB_HOLD_MAX_MS}.
|
|
104
|
+
*
|
|
105
|
+
* `NAMZU_JOB_HOLD_MAX_MS` overrides the ceiling for a host that wants a
|
|
106
|
+
* longer or shorter park, the way `NAMZU_JOB_WAIT_TIMEOUT_MS` overrides
|
|
107
|
+
* `wait_for_job`'s own bound — and it is the same parse, so a value that is
|
|
108
|
+
* not a positive whole number of milliseconds leaves the default standing
|
|
109
|
+
* rather than holding a run for `NaN`. Called here rather than at module
|
|
110
|
+
* load, because a host that sets it after import is not ignored.
|
|
111
|
+
*/
|
|
112
|
+
export function awaitedJobGraceMs(remainingBeforeFinalizeMs) {
|
|
113
|
+
const ceiling = readPositiveIntEnv('NAMZU_JOB_HOLD_MAX_MS', DEFAULT_JOB_HOLD_MAX_MS);
|
|
114
|
+
return Math.min(settleGraceMs(remainingBeforeFinalizeMs), ceiling);
|
|
115
|
+
}
|
|
116
|
+
/**
|
|
117
|
+
* Hold the run open for work that has not finished, and deliver it.
|
|
118
|
+
*
|
|
119
|
+
* Returns whether a completion, a job exit or an operator message entered
|
|
120
|
+
* the transcript — the caller continues on `true`, so the model gets a turn
|
|
121
|
+
* to respond. That turn is the entire justification for waiting, which
|
|
122
|
+
* is why only the exits that can still take one call this.
|
|
123
|
+
*
|
|
124
|
+
* Two kinds of work qualify and they are raced together, because a run has
|
|
125
|
+
* one settle point and one grace period to spend at it:
|
|
126
|
+
*
|
|
127
|
+
* - a delegated task the `CompletionInbox` is still expecting;
|
|
128
|
+
* - a background job the model told `wait_for_job` it is waiting on.
|
|
129
|
+
*
|
|
130
|
+
* The job half is deliberately narrow. Intent comes from the wait and from
|
|
131
|
+
* nothing else — a dev server the model started and never waited on is
|
|
132
|
+
* running because somebody wanted it running, and a hold for it would add
|
|
133
|
+
* the grace period to the end of every turn for the rest of the session.
|
|
134
|
+
*
|
|
135
|
+
* Each leg is opened only when it has something pending: both
|
|
136
|
+
* `waitForArrival` implementations resolve immediately when their own side
|
|
137
|
+
* is idle, so racing an idle one would end the hold before it began.
|
|
138
|
+
*
|
|
139
|
+
* Bounded by `settleGraceMs` and by `maxIterations`, so work that never
|
|
140
|
+
* finishes cannot keep the run open. On a run with a deadline the grace is
|
|
141
|
+
* a share of what is LEFT of it rather than a fresh allowance, so a
|
|
142
|
+
* `wait_for_job` call that already spent minutes has shortened this hold
|
|
143
|
+
* by the same minutes. On a run without one — the CLI's default — there is
|
|
144
|
+
* no remainder to take a share of, and the job leg's own ceiling
|
|
145
|
+
* (`awaitedJobGraceMs`) is what keeps a timed-out wait from being followed
|
|
146
|
+
* by an hour of silence.
|
|
147
|
+
*/
|
|
148
|
+
export async function* holdForOutstandingWork(ctx, iterationNum, hasToolCalls, deliverInbound) {
|
|
149
|
+
const inbox = ctx.completionInbox?.hasPendingWork ? ctx.completionInbox : undefined;
|
|
150
|
+
const jobs = ctx.awaitedJobs?.hasPendingWork ? ctx.awaitedJobs : undefined;
|
|
151
|
+
if (!inbox && !jobs)
|
|
152
|
+
return false;
|
|
153
|
+
// Read HERE rather than from `forceFinalize`, which was sampled at the
|
|
154
|
+
// top of the iteration: one that has since crossed the finalize point
|
|
155
|
+
// must not open a wait against a reserve it has already entered.
|
|
156
|
+
const remainingMs = ctx.guard.remainingBeforeFinalizeMs();
|
|
157
|
+
// One deadline for the race, and it is the LONGEST ceiling any pending
|
|
158
|
+
// leg justifies. A leg resolving on its own timer ends the whole race,
|
|
159
|
+
// so handing the job leg its shorter ceiling while a task was also
|
|
160
|
+
// outstanding would cut the task's hold down to the job's — a run
|
|
161
|
+
// walking away from a worker it had time for, because a job happened
|
|
162
|
+
// to be running. A job therefore never shortens a wait, and it never
|
|
163
|
+
// lengthens one either: where a task is outstanding too, that is how
|
|
164
|
+
// long this run was waiting anyway.
|
|
165
|
+
const graceMs = inbox ? settleGraceMs(remainingMs) : awaitedJobGraceMs(remainingMs);
|
|
166
|
+
ctx.log.info('Holding the run open for outstanding work', {
|
|
167
|
+
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
168
|
+
[NAMZU.ITERATION]: iterationNum,
|
|
169
|
+
'namzu.runtime.grace_ms': graceMs,
|
|
170
|
+
'namzu.runtime.awaited_jobs': jobs?.outstandingJobIds ?? [],
|
|
171
|
+
});
|
|
172
|
+
// User input releases this wait without cancelling any child. Both waits
|
|
173
|
+
// share a disposable signal so the losing arrival listener cannot leak.
|
|
174
|
+
const waiting = new AbortController();
|
|
175
|
+
const runSignal = ctx.abortController.signal;
|
|
176
|
+
const cancelWait = () => waiting.abort(runSignal.reason);
|
|
177
|
+
runSignal.addEventListener('abort', cancelWait, { once: true });
|
|
178
|
+
if (runSignal.aborted)
|
|
179
|
+
cancelWait();
|
|
180
|
+
try {
|
|
181
|
+
await Promise.race([
|
|
182
|
+
...(inbox ? [inbox.waitForArrival(graceMs, waiting.signal)] : []),
|
|
183
|
+
...(jobs ? [jobs.waitForArrival(graceMs, waiting.signal)] : []),
|
|
184
|
+
...(ctx.waitForInbound ? [ctx.waitForInbound(waiting.signal)] : []),
|
|
185
|
+
]);
|
|
186
|
+
}
|
|
187
|
+
catch (error) {
|
|
188
|
+
if (!runSignal.aborted)
|
|
189
|
+
throw error;
|
|
190
|
+
}
|
|
191
|
+
finally {
|
|
192
|
+
waiting.abort();
|
|
193
|
+
runSignal.removeEventListener('abort', cancelWait);
|
|
194
|
+
}
|
|
195
|
+
runSignal.throwIfAborted();
|
|
196
|
+
const arrived = ctx.completionInbox?.drain() ?? [];
|
|
197
|
+
if (arrived.length > 0) {
|
|
198
|
+
ctx.runMgr.pushMessage(createRuntimeContextMessage(formatCompletionNotification(arrived), 'task-completion'));
|
|
199
|
+
}
|
|
200
|
+
const exited = deliverAwaitedJobExits(ctx);
|
|
201
|
+
const inbound = deliverInbound();
|
|
202
|
+
if (arrived.length === 0 && !exited && inbound === 0)
|
|
203
|
+
return false;
|
|
204
|
+
await ctx.emitEvent({
|
|
205
|
+
type: 'iteration_completed',
|
|
206
|
+
runId: ctx.runMgr.id,
|
|
207
|
+
iteration: iterationNum,
|
|
208
|
+
hasToolCalls,
|
|
209
|
+
});
|
|
210
|
+
yield* ctx.drainPending();
|
|
211
|
+
return true;
|
|
212
|
+
}
|
|
213
|
+
/**
|
|
214
|
+
* Put the job exits this hold was waiting for in front of the model.
|
|
215
|
+
*
|
|
216
|
+
* Through `jobNotices`, which is the channel a job exit already travels on
|
|
217
|
+
* — `attachNotice` rides it out on the next tool result — rather than a
|
|
218
|
+
* second one built for this path. A turn that called no tools has no such
|
|
219
|
+
* result, so the queued text becomes a `runtime-context` message instead,
|
|
220
|
+
* exactly as `deliverInbound` does for steering that found no tool result
|
|
221
|
+
* to attach to.
|
|
222
|
+
*
|
|
223
|
+
* That drain is also what keeps one exit from being delivered twice: the
|
|
224
|
+
* channel hands its text over once, so an exit already attached to a tool
|
|
225
|
+
* result earlier in the turn leaves nothing here — and the record of it
|
|
226
|
+
* went with that delivery, so this returns `false` rather than buying a
|
|
227
|
+
* turn to re-read what the model has read.
|
|
228
|
+
*
|
|
229
|
+
* `takeDelivery` is what pairs the two. Taking the exits first and then
|
|
230
|
+
* finding no notice would discard them, which is the one way this path
|
|
231
|
+
* can lose an exit outright; neither is taken unless both are there.
|
|
232
|
+
*
|
|
233
|
+
* The channel is not per-job, so the text taken here can include a notice
|
|
234
|
+
* for a job nobody awaited that ended while the hold was open. Delivering
|
|
235
|
+
* it is right — it is unread either way, and the alternative is stranding
|
|
236
|
+
* it — but it is not a reason to WAIT, which is why what opens this hold
|
|
237
|
+
* is `AwaitedJobs`, and the two are asked separately.
|
|
238
|
+
*/
|
|
239
|
+
export function deliverAwaitedJobExits(ctx) {
|
|
240
|
+
const delivered = ctx.awaitedJobs?.takeDelivery(() => ctx.jobNotices?.drain());
|
|
241
|
+
if (!delivered)
|
|
242
|
+
return false;
|
|
243
|
+
ctx.log.info('Delivering a background job exit the run held open for', {
|
|
244
|
+
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
245
|
+
'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
|
|
246
|
+
});
|
|
247
|
+
ctx.runMgr.pushMessage(createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'));
|
|
248
|
+
return true;
|
|
249
|
+
}
|
|
250
|
+
/**
|
|
251
|
+
* Account for outstanding work on the way out: deliver what arrived, and
|
|
252
|
+
* say what did not.
|
|
253
|
+
*
|
|
254
|
+
* A run that ends with a worker outstanding must not leave the impression
|
|
255
|
+
* that the worker's result was delivered. There are exactly two honest
|
|
256
|
+
* outcomes and this does both:
|
|
257
|
+
*
|
|
258
|
+
* - **What has already arrived is delivered.** It makes no false claim,
|
|
259
|
+
* and dropping it is pure loss — the message rides out on
|
|
260
|
+
* `Run.messages`, so a host reads it and the next turn of a continued
|
|
261
|
+
* thread starts with it. This does NOT wait: a hold buys the model a
|
|
262
|
+
* turn in which to USE a result, and on an exit whose answer is already
|
|
263
|
+
* decided there is no such turn, so waiting would delay a settled answer
|
|
264
|
+
* to append text this run will not read. The bounded hold stays where it
|
|
265
|
+
* was, on the exits that do have a turn left.
|
|
266
|
+
* - **What is still running is NAMED, not cancelled.** Giving up on a wait
|
|
267
|
+
* is a statement about the waiter, not about the work — the rule
|
|
268
|
+
* `wait-with-idle-bound.ts` already states for the same subsystem — and
|
|
269
|
+
* "the parent answered early" is a weaker warrant for killing a child
|
|
270
|
+
* than "the clock ran out", not a stronger one. Killing a worker that
|
|
271
|
+
* may be mid-write is a policy only the host can judge, and it has
|
|
272
|
+
* `cancel_task` and the run controller to judge it with.
|
|
273
|
+
*/
|
|
274
|
+
export function settleOutstandingWork(ctx) {
|
|
275
|
+
deliverArrivedCompletions(ctx);
|
|
276
|
+
deliverArrivedJobExits(ctx);
|
|
277
|
+
recordAbandonedWork(ctx);
|
|
278
|
+
}
|
|
279
|
+
/** Work this run walked away from. See {@link settleOutstandingWork}. */
|
|
280
|
+
export function recordAbandonedWork(ctx) {
|
|
281
|
+
const abandoned = ctx.completionInbox?.outstandingTaskIds ?? [];
|
|
282
|
+
if (abandoned.length > 0) {
|
|
283
|
+
ctx.log.warn('Run ended with delegated work still running', {
|
|
284
|
+
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
285
|
+
'namzu.runtime.tasks': abandoned,
|
|
286
|
+
});
|
|
287
|
+
ctx.runMgr.setAbandonedTaskIds(abandoned);
|
|
288
|
+
}
|
|
289
|
+
// The same statement for a job the model was waiting on when the grace
|
|
290
|
+
// ran out. Only awaited ones: a job nobody waited for was never work
|
|
291
|
+
// this run was holding, so naming it would report an abandonment that
|
|
292
|
+
// did not happen.
|
|
293
|
+
const abandonedJobs = ctx.awaitedJobs?.outstandingJobIds ?? [];
|
|
294
|
+
if (abandonedJobs.length === 0)
|
|
295
|
+
return;
|
|
296
|
+
ctx.log.warn('Run ended with an awaited background job still running', {
|
|
297
|
+
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
298
|
+
'namzu.runtime.jobs': abandonedJobs,
|
|
299
|
+
});
|
|
300
|
+
ctx.runMgr.setAbandonedJobIds(abandonedJobs);
|
|
301
|
+
}
|
|
302
|
+
export function deliverArrivedCompletions(ctx) {
|
|
303
|
+
const unheard = ctx.completionInbox?.drain() ?? [];
|
|
304
|
+
if (unheard.length === 0)
|
|
305
|
+
return;
|
|
306
|
+
// Fix the run's answer BEFORE appending anything after it.
|
|
307
|
+
//
|
|
308
|
+
// `RunPersistence.resolveResult` walks the message tail backwards and
|
|
309
|
+
// stops at the first non-assistant message, and it runs at
|
|
310
|
+
// `markCompleted` — which is AFTER this. So a notification appended
|
|
311
|
+
// after the final assistant turn makes the run's own answer
|
|
312
|
+
// unreachable. Measured, on a run whose model had just said "THIS IS
|
|
313
|
+
// THE RUN ANSWER.": `run.result` came back `undefined`. That trades a
|
|
314
|
+
// lost worker result for a lost RUN result, which is strictly worse
|
|
315
|
+
// than the defect this delivery exists to fix.
|
|
316
|
+
//
|
|
317
|
+
// Materialising resolves it while the tail is still the assistant's;
|
|
318
|
+
// pinning it means the later re-resolution cannot undo the fix. Only
|
|
319
|
+
// when there is something to pin: on the cancelled and thrown paths
|
|
320
|
+
// there may be no answer, and pinning an empty string there would
|
|
321
|
+
// suppress whatever the error path assembles.
|
|
322
|
+
const answer = ctx.runMgr.materializeResult();
|
|
323
|
+
if (answer.length > 0)
|
|
324
|
+
ctx.runMgr.setResult(answer);
|
|
325
|
+
ctx.log.info('Delivering task completions the run would have settled over', {
|
|
326
|
+
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
327
|
+
'namzu.runtime.tasks': unheard.map((h) => h.taskId),
|
|
328
|
+
});
|
|
329
|
+
ctx.runMgr.pushMessage(createRuntimeContextMessage(formatCompletionNotification(unheard), 'task-completion'));
|
|
330
|
+
}
|
|
331
|
+
/**
|
|
332
|
+
* The job half of {@link deliverArrivedCompletions}: an exit that arrived
|
|
333
|
+
* too late to earn a turn is still delivered on the way out.
|
|
334
|
+
*
|
|
335
|
+
* The window this closes is one tick wide and it is nobody else's. An
|
|
336
|
+
* awaited job that exits between the hold's grace expiring and the run
|
|
337
|
+
* settling was never delivered — the hold had already looked — and is no
|
|
338
|
+
* longer named either, because the exit took it off the outstanding list
|
|
339
|
+
* on its way past, so `abandonedJobIds` would be lying to claim it. The
|
|
340
|
+
* host's own listener is no help: the CLI queues an exit for the next
|
|
341
|
+
* turn only when no run is in flight, and this one is still in flight.
|
|
342
|
+
* Delivered here it reaches `Run.messages`, so the transcript has it and
|
|
343
|
+
* a continued thread opens with it.
|
|
344
|
+
*
|
|
345
|
+
* Before `recordAbandonedWork`, which then reports only what is still
|
|
346
|
+
* running, and after `deliverArrivedCompletions`, so the two appended
|
|
347
|
+
* messages land in the order the work finished in.
|
|
348
|
+
*/
|
|
349
|
+
export function deliverArrivedJobExits(ctx) {
|
|
350
|
+
const delivered = ctx.awaitedJobs?.takeDelivery(() => ctx.jobNotices?.drain());
|
|
351
|
+
if (!delivered)
|
|
352
|
+
return;
|
|
353
|
+
// Fix the run's answer BEFORE appending anything after it — the same
|
|
354
|
+
// `resolveResult` tail walk `deliverArrivedCompletions` explains just
|
|
355
|
+
// above, and the same guard against pinning an empty one.
|
|
356
|
+
const answer = ctx.runMgr.materializeResult();
|
|
357
|
+
if (answer.length > 0)
|
|
358
|
+
ctx.runMgr.setResult(answer);
|
|
359
|
+
ctx.log.info('Delivering a background job exit the run would have settled over', {
|
|
360
|
+
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
361
|
+
'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
|
|
362
|
+
});
|
|
363
|
+
ctx.runMgr.pushMessage(createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'));
|
|
364
|
+
}
|
|
365
|
+
//# sourceMappingURL=outstanding-work.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"outstanding-work.js","sourceRoot":"","sources":["../../../../src/runtime/query/iteration/outstanding-work.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,KAAK,EAAE,MAAM,uCAAuC,CAAA;AAC7D,OAAO,EAAE,4BAA4B,EAAE,MAAM,wCAAwC,CAAA;AACrF,OAAO,EAAE,qBAAqB,EAAE,MAAM,qCAAqC,CAAA;AAC3E,OAAO,EAAE,2BAA2B,EAAE,MAAM,iCAAiC,CAAA;AAE7E,OAAO,EAAE,kBAAkB,EAAE,MAAM,uBAAuB,CAAA;AAC1D,OAAO,EAAE,aAAa,EAAE,MAAM,gBAAgB,CAAA;AAG9C;;;;;;;;;;;;;;;GAeG;AAEH;;;;;;;;;;;;;;;;GAgBG;AACH,MAAM,qBAAqB,GAAG,GAAG,CAAA;AAEjC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAmCG;AACH,MAAM,UAAU,aAAa,CAAC,yBAAiC;IAC9D,OAAO,IAAI,CAAC,GAAG,CACd,IAAI,CAAC,KAAK,CAAC,yBAAyB,GAAG,qBAAqB,CAAC,EAC7D,qBAAqB,CACrB,CAAA;AACF,CAAC;AAED;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,MAAM,uBAAuB,GAAG,CAAC,GAAG,EAAE,GAAG,IAAI,CAAA;AAE7C;;;;;;;;;GASG;AACH,MAAM,UAAU,iBAAiB,CAAC,yBAAiC;IAClE,MAAM,OAAO,GAAG,kBAAkB,CAAC,uBAAuB,EAAE,uBAAuB,CAAC,CAAA;IACpF,OAAO,IAAI,CAAC,GAAG,CAAC,aAAa,CAAC,yBAAyB,CAAC,EAAE,OAAO,CAAC,CAAA;AACnE,CAAC;AAED;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA+BG;AACH,MAAM,CAAC,KAAK,SAAS,CAAC,CAAC,sBAAsB,CAC5C,GAAqB,EACrB,YAAoB,EACpB,YAAqB,EACrB,cAA4B;IAE5B,MAAM,KAAK,GAAG,GAAG,CAAC,eAAe,EAAE,cAAc,CAAC,CAAC,CAAC,GAAG,CAAC,eAAe,CAAC,CAAC,CAAC,SAAS,CAAA;IACnF,MAAM,IAAI,GAAG,GAAG,CAAC,WAAW,EAAE,cAAc,CAAC,CAAC,CAAC,GAAG,CAAC,WAAW,CAAC,CAAC,CAAC,SAAS,CAAA;IAC1E,IAAI,CAAC,KAAK,IAAI,CAAC,IAAI;QAAE,OAAO,KAAK,CAAA;IAEjC,uEAAuE;IACvE,sEAAsE;IACtE,iEAAiE;IACjE,MAAM,WAAW,GAAG,GAAG,CAAC,KAAK,CAAC,yBAAyB,EAAE,CAAA;IACzD,uEAAuE;IACvE,uEAAuE;IACvE,mEAAmE;IACnE,kEAAkE;IAClE,qEAAqE;IACrE,qEAAqE;IACrE,qEAAqE;IACrE,oCAAoC;IACpC,MAAM,OAAO,GAAG,KAAK,CAAC,CAAC,CAAC,aAAa,CAAC,WAAW,CAAC,CAAC,CAAC,CAAC,iBAAiB,CAAC,WAAW,CAAC,CAAA;IACnF,GAAG,CAAC,GAAG,CAAC,IAAI,CAAC,2CAA2C,EAAE;QACzD,CAAC,KAAK,CAAC,MAAM,CAAC,EAAE,GAAG,CAAC,MAAM,CAAC,EAAE;QAC7B,CAAC,KAAK,CAAC,SAAS,CAAC,EAAE,YAAY;QAC/B,wBAAwB,EAAE,OAAO;QACjC,4BAA4B,EAAE,IAAI,EAAE,iBAAiB,IAAI,EAAE;KAC3D,CAAC,CAAA;IACF,yEAAyE;IACzE,wEAAwE;IACxE,MAAM,OAAO,GAAG,IAAI,eAAe,EAAE,CAAA;IACrC,MAAM,SAAS,GAAG,GAAG,CAAC,eAAe,CAAC,MAAM,CAAA;IAC5C,MAAM,UAAU,GAAG,GAAG,EAAE,CAAC,OAAO,CAAC,KAAK,CAAC,SAAS,CAAC,MAAM,CAAC,CAAA;IACxD,SAAS,CAAC,gBAAgB,CAAC,OAAO,EAAE,UAAU,EAAE,EAAE,IAAI,EAAE,IAAI,EAAE,CAAC,CAAA;IAC/D,IAAI,SAAS,CAAC,OAAO;QAAE,UAAU,EAAE,CAAA;IACnC,IAAI,CAAC;QACJ,MAAM,OAAO,CAAC,IAAI,CAAC;YAClB,GAAG,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,KAAK,CAAC,cAAc,CAAC,OAAO,EAAE,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC;YACjE,GAAG,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,cAAc,CAAC,OAAO,EAAE,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC;YAC/D,GAAG,CAAC,GAAG,CAAC,cAAc,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,cAAc,CAAC,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC;SACnE,CAAC,CAAA;IACH,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QAChB,IAAI,CAAC,SAAS,CAAC,OAAO;YAAE,MAAM,KAAK,CAAA;IACpC,CAAC;YAAS,CAAC;QACV,OAAO,CAAC,KAAK,EAAE,CAAA;QACf,SAAS,CAAC,mBAAmB,CAAC,OAAO,EAAE,UAAU,CAAC,CAAA;IACnD,CAAC;IACD,SAAS,CAAC,cAAc,EAAE,CAAA;IAE1B,MAAM,OAAO,GAAG,GAAG,CAAC,eAAe,EAAE,KAAK,EAAE,IAAI,EAAE,CAAA;IAClD,IAAI,OAAO,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACxB,GAAG,CAAC,MAAM,CAAC,WAAW,CACrB,2BAA2B,CAAC,4BAA4B,CAAC,OAAO,CAAC,EAAE,iBAAiB,CAAC,CACrF,CAAA;IACF,CAAC;IACD,MAAM,MAAM,GAAG,sBAAsB,CAAC,GAAG,CAAC,CAAA;IAC1C,MAAM,OAAO,GAAG,cAAc,EAAE,CAAA;IAChC,IAAI,OAAO,CAAC,MAAM,KAAK,CAAC,IAAI,CAAC,MAAM,IAAI,OAAO,KAAK,CAAC;QAAE,OAAO,KAAK,CAAA;IAClE,MAAM,GAAG,CAAC,SAAS,CAAC;QACnB,IAAI,EAAE,qBAAqB;QAC3B,KAAK,EAAE,GAAG,CAAC,MAAM,CAAC,EAAE;QACpB,SAAS,EAAE,YAAY;QACvB,YAAY;KACZ,CAAC,CAAA;IACF,KAAK,CAAC,CAAC,GAAG,CAAC,YAAY,EAAE,CAAA;IACzB,OAAO,IAAI,CAAA;AACZ,CAAC;AAED;;;;;;;;;;;;;;;;;;;;;;;;;GAyBG;AACH,MAAM,UAAU,sBAAsB,CAAC,GAAqB;IAC3D,MAAM,SAAS,GAAG,GAAG,CAAC,WAAW,EAAE,YAAY,CAAC,GAAG,EAAE,CAAC,GAAG,CAAC,UAAU,EAAE,KAAK,EAAE,CAAC,CAAA;IAC9E,IAAI,CAAC,SAAS;QAAE,OAAO,KAAK,CAAA;IAE5B,GAAG,CAAC,GAAG,CAAC,IAAI,CAAC,wDAAwD,EAAE;QACtE,CAAC,KAAK,CAAC,MAAM,CAAC,EAAE,GAAG,CAAC,MAAM,CAAC,EAAE;QAC7B,oBAAoB,EAAE,SAAS,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,EAAE,CAAC;KAC1D,CAAC,CAAA;IACF,GAAG,CAAC,MAAM,CAAC,WAAW,CAAC,2BAA2B,CAAC,aAAa,CAAC,SAAS,CAAC,IAAI,CAAC,EAAE,UAAU,CAAC,CAAC,CAAA;IAC9F,OAAO,IAAI,CAAA;AACZ,CAAC;AAED;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,MAAM,UAAU,qBAAqB,CAAC,GAAqB;IAC1D,yBAAyB,CAAC,GAAG,CAAC,CAAA;IAC9B,sBAAsB,CAAC,GAAG,CAAC,CAAA;IAC3B,mBAAmB,CAAC,GAAG,CAAC,CAAA;AACzB,CAAC;AAED,yEAAyE;AACzE,MAAM,UAAU,mBAAmB,CAAC,GAAqB;IACxD,MAAM,SAAS,GAAG,GAAG,CAAC,eAAe,EAAE,kBAAkB,IAAI,EAAE,CAAA;IAC/D,IAAI,SAAS,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAC1B,GAAG,CAAC,GAAG,CAAC,IAAI,CAAC,6CAA6C,EAAE;YAC3D,CAAC,KAAK,CAAC,MAAM,CAAC,EAAE,GAAG,CAAC,MAAM,CAAC,EAAE;YAC7B,qBAAqB,EAAE,SAAS;SAChC,CAAC,CAAA;QACF,GAAG,CAAC,MAAM,CAAC,mBAAmB,CAAC,SAAS,CAAC,CAAA;IAC1C,CAAC;IAED,uEAAuE;IACvE,qEAAqE;IACrE,sEAAsE;IACtE,kBAAkB;IAClB,MAAM,aAAa,GAAG,GAAG,CAAC,WAAW,EAAE,iBAAiB,IAAI,EAAE,CAAA;IAC9D,IAAI,aAAa,CAAC,MAAM,KAAK,CAAC;QAAE,OAAM;IAEtC,GAAG,CAAC,GAAG,CAAC,IAAI,CAAC,wDAAwD,EAAE;QACtE,CAAC,KAAK,CAAC,MAAM,CAAC,EAAE,GAAG,CAAC,MAAM,CAAC,EAAE;QAC7B,oBAAoB,EAAE,aAAa;KACnC,CAAC,CAAA;IACF,GAAG,CAAC,MAAM,CAAC,kBAAkB,CAAC,aAAa,CAAC,CAAA;AAC7C,CAAC;AAED,MAAM,UAAU,yBAAyB,CAAC,GAAqB;IAC9D,MAAM,OAAO,GAAG,GAAG,CAAC,eAAe,EAAE,KAAK,EAAE,IAAI,EAAE,CAAA;IAClD,IAAI,OAAO,CAAC,MAAM,KAAK,CAAC;QAAE,OAAM;IAEhC,2DAA2D;IAC3D,EAAE;IACF,sEAAsE;IACtE,2DAA2D;IAC3D,oEAAoE;IACpE,4DAA4D;IAC5D,qEAAqE;IACrE,sEAAsE;IACtE,oEAAoE;IACpE,+CAA+C;IAC/C,EAAE;IACF,qEAAqE;IACrE,qEAAqE;IACrE,oEAAoE;IACpE,kEAAkE;IAClE,8CAA8C;IAC9C,MAAM,MAAM,GAAG,GAAG,CAAC,MAAM,CAAC,iBAAiB,EAAE,CAAA;IAC7C,IAAI,MAAM,CAAC,MAAM,GAAG,CAAC;QAAE,GAAG,CAAC,MAAM,CAAC,SAAS,CAAC,MAAM,CAAC,CAAA;IAEnD,GAAG,CAAC,GAAG,CAAC,IAAI,CAAC,6DAA6D,EAAE;QAC3E,CAAC,KAAK,CAAC,MAAM,CAAC,EAAE,GAAG,CAAC,MAAM,CAAC,EAAE;QAC7B,qBAAqB,EAAE,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,MAAM,CAAC;KACnD,CAAC,CAAA;IACF,GAAG,CAAC,MAAM,CAAC,WAAW,CACrB,2BAA2B,CAAC,4BAA4B,CAAC,OAAO,CAAC,EAAE,iBAAiB,CAAC,CACrF,CAAA;AACF,CAAC;AAED;;;;;;;;;;;;;;;;;GAiBG;AACH,MAAM,UAAU,sBAAsB,CAAC,GAAqB;IAC3D,MAAM,SAAS,GAAG,GAAG,CAAC,WAAW,EAAE,YAAY,CAAC,GAAG,EAAE,CAAC,GAAG,CAAC,UAAU,EAAE,KAAK,EAAE,CAAC,CAAA;IAC9E,IAAI,CAAC,SAAS;QAAE,OAAM;IAEtB,qEAAqE;IACrE,sEAAsE;IACtE,0DAA0D;IAC1D,MAAM,MAAM,GAAG,GAAG,CAAC,MAAM,CAAC,iBAAiB,EAAE,CAAA;IAC7C,IAAI,MAAM,CAAC,MAAM,GAAG,CAAC;QAAE,GAAG,CAAC,MAAM,CAAC,SAAS,CAAC,MAAM,CAAC,CAAA;IAEnD,GAAG,CAAC,GAAG,CAAC,IAAI,CAAC,kEAAkE,EAAE;QAChF,CAAC,KAAK,CAAC,MAAM,CAAC,EAAE,GAAG,CAAC,MAAM,CAAC,EAAE;QAC7B,oBAAoB,EAAE,SAAS,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,GAAG,EAAE,EAAE,CAAC,GAAG,CAAC,EAAE,CAAC;KAC1D,CAAC,CAAA;IACF,GAAG,CAAC,MAAM,CAAC,WAAW,CAAC,2BAA2B,CAAC,aAAa,CAAC,SAAS,CAAC,IAAI,CAAC,EAAE,UAAU,CAAC,CAAC,CAAA;AAC/F,CAAC"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"plan.d.ts","sourceRoot":"","sources":["../../../../../src/runtime/query/iteration/phases/plan.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,QAAQ,EAAE,MAAM,gCAAgC,CAAA;AAC9D,OAAO,
|
|
1
|
+
{"version":3,"file":"plan.d.ts","sourceRoot":"","sources":["../../../../../src/runtime/query/iteration/phases/plan.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,QAAQ,EAAE,MAAM,gCAAgC,CAAA;AAC9D,OAAO,EACN,KAAK,gBAAgB,EACrB,KAAK,WAAW,EAGhB,MAAM,cAAc,CAAA;AAErB,wBAAuB,WAAW,CAAC,GAAG,EAAE,gBAAgB,GAAG,cAAc,CAAC,QAAQ,EAAE,WAAW,CAAC,CAsD/F"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { handleHITLDecision } from './context.js';
|
|
1
|
+
import { awaitDecisionOrAbort, handleHITLDecision, } from './context.js';
|
|
2
2
|
export async function* runPlanGate(ctx) {
|
|
3
3
|
if (!ctx.planManager.active || ctx.planManager.active.status !== 'ready') {
|
|
4
4
|
return 'continue';
|
|
@@ -33,8 +33,19 @@ export async function* runPlanGate(ctx) {
|
|
|
33
33
|
// Record the park BEFORE awaiting it. A process that dies while a human
|
|
34
34
|
// is reading the plan otherwise leaves nothing behind saying the plan
|
|
35
35
|
// was ever put up for approval.
|
|
36
|
+
//
|
|
37
|
+
// The park stays eager — `awaitDecisionDurably` records only after
|
|
38
|
+
// `PARK_RECORD_DELAY_MS`, which is the right trade for a gate that runs
|
|
39
|
+
// on every iteration and the wrong one for a gate that runs once and is
|
|
40
|
+
// read by a human. Only the AWAIT below is raced.
|
|
36
41
|
await ctx.checkpointMgr.park(planCheckpoint, request);
|
|
37
|
-
|
|
42
|
+
// Raced against the run's abort signal, like every other park. A bare
|
|
43
|
+
// `await ctx.resumeHandler(request)` here meant a Stop did nothing until
|
|
44
|
+
// the host answered: `runPlanGate` runs in the iteration loop rather than
|
|
45
|
+
// inside a tool call, so nothing downstream bounded the wait. A Stop now
|
|
46
|
+
// resolves the park as `abort`, which `handleHITLDecision` turns into
|
|
47
|
+
// `setStopReason('cancelled') + markCancelled + stop`.
|
|
48
|
+
const planDecision = await awaitDecisionOrAbort(ctx, request);
|
|
38
49
|
await ctx.checkpointMgr.unpark(planCheckpoint.id, planDecision);
|
|
39
50
|
return yield* handleHITLDecision(ctx, planDecision, planCheckpoint.id, 'plan_gate');
|
|
40
51
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"plan.js","sourceRoot":"","sources":["../../../../../src/runtime/query/iteration/phases/plan.ts"],"names":[],"mappings":"AAEA,OAAO,
|
|
1
|
+
{"version":3,"file":"plan.js","sourceRoot":"","sources":["../../../../../src/runtime/query/iteration/phases/plan.ts"],"names":[],"mappings":"AAEA,OAAO,EAGN,oBAAoB,EACpB,kBAAkB,GAClB,MAAM,cAAc,CAAA;AAErB,MAAM,CAAC,KAAK,SAAS,CAAC,CAAC,WAAW,CAAC,GAAqB;IACvD,IAAI,CAAC,GAAG,CAAC,WAAW,CAAC,MAAM,IAAI,GAAG,CAAC,WAAW,CAAC,MAAM,CAAC,MAAM,KAAK,OAAO,EAAE,CAAC;QAC1E,OAAO,UAAU,CAAA;IAClB,CAAC;IAED,MAAM,cAAc,GAAG,MAAM,GAAG,CAAC,aAAa,CAAC,MAAM,CAAC,GAAG,CAAC,MAAM,EAAE,CAAC,CAAC,CAAA;IAEpE,MAAM,GAAG,CAAC,SAAS,CAAC;QACnB,IAAI,EAAE,oBAAoB;QAC1B,KAAK,EAAE,GAAG,CAAC,MAAM,CAAC,EAAE;QACpB,YAAY,EAAE,cAAc,CAAC,EAAE;QAC/B,SAAS,EAAE,CAAC;KACZ,CAAC,CAAA;IACF,KAAK,CAAC,CAAC,GAAG,CAAC,YAAY,EAAE,CAAA;IAEzB,MAAM,IAAI,GAAG,GAAG,CAAC,WAAW,CAAC,MAAM,CAAA;IACnC,MAAM,OAAO,GAAwB;QACpC,IAAI,EAAE,eAAe;QACrB,KAAK,EAAE,GAAG,CAAC,MAAM,CAAC,EAAE;QACpB,YAAY,EAAE,cAAc,CAAC,EAAE;QAC/B,IAAI,EAAE;YACL,MAAM,EAAE,IAAI,CAAC,EAAE;YACf,KAAK,EAAE,IAAI,CAAC,KAAK;YACjB,KAAK,EAAE,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC;gBAC7B,EAAE,EAAE,CAAC,CAAC,EAAE;gBACR,WAAW,EAAE,CAAC,CAAC,WAAW;gBAC1B,QAAQ,EAAE,CAAC,CAAC,QAAQ;gBACpB,OAAO,EAAE,CAAC,CAAC,OAAO;gBAClB,SAAS,EAAE,CAAC,CAAC,SAAS;gBACtB,KAAK,EAAE,CAAC,CAAC,KAAK;aACd,CAAC,CAAC;YACH,OAAO,EAAE,IAAI,CAAC,OAAO;SACrB;KACD,CAAA;IAED,wEAAwE;IACxE,sEAAsE;IACtE,gCAAgC;IAChC,EAAE;IACF,mEAAmE;IACnE,wEAAwE;IACxE,wEAAwE;IACxE,kDAAkD;IAClD,MAAM,GAAG,CAAC,aAAa,CAAC,IAAI,CAAC,cAAc,EAAE,OAAO,CAAC,CAAA;IACrD,sEAAsE;IACtE,yEAAyE;IACzE,0EAA0E;IAC1E,yEAAyE;IACzE,sEAAsE;IACtE,uDAAuD;IACvD,MAAM,YAAY,GAAG,MAAM,oBAAoB,CAAC,GAAG,EAAE,OAAO,CAAC,CAAA;IAC7D,MAAM,GAAG,CAAC,aAAa,CAAC,MAAM,CAAC,cAAc,CAAC,EAAE,EAAE,YAAY,CAAC,CAAA;IAE/D,OAAO,KAAK,CAAC,CAAC,kBAAkB,CAAC,GAAG,EAAE,YAAY,EAAE,cAAc,CAAC,EAAE,EAAE,WAAW,CAAC,CAAA;AACpF,CAAC"}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import { type Message, type UserMessage } from '../../../types/message/index.js';
|
|
2
|
+
import type { ToolChoice } from '../../../types/provider/chat.js';
|
|
3
|
+
import type { PrepareStepContext, PrepareStepResult, StepResult, StepVeto } from '../../../types/run/index.js';
|
|
4
|
+
import type { Skill } from '../../../types/skills/index.js';
|
|
5
|
+
import type { IterationContext } from './phases/index.js';
|
|
6
|
+
/**
|
|
7
|
+
* How the next request is shaped: admission, preparation, and the context
|
|
8
|
+
* budget handed to both.
|
|
9
|
+
*
|
|
10
|
+
* Three of these read state the loop replaces or grows as it runs, so they
|
|
11
|
+
* arrive as accessors rather than as values — `latestUserMessage` is replaced
|
|
12
|
+
* on every operator turn and `steps` gains a member per step, and a captured
|
|
13
|
+
* copy of either would describe an earlier return. `ctx` is the run's own
|
|
14
|
+
* context object, passed through rather than re-derived, because
|
|
15
|
+
* `selectContextModel` WRITES two of its fields (`contextModel` and
|
|
16
|
+
* `activeProviderContextWindow`) for the compaction pass to read.
|
|
17
|
+
*/
|
|
18
|
+
export interface StepShaping {
|
|
19
|
+
readonly ctx: IterationContext;
|
|
20
|
+
readonly latestUserMessage: () => UserMessage | undefined;
|
|
21
|
+
readonly steps: () => readonly StepResult[];
|
|
22
|
+
}
|
|
23
|
+
export declare function stepContextMessage(content: string): UserMessage;
|
|
24
|
+
/** Derived after request projection; never accumulates in canonical history or replaces operator intent. */
|
|
25
|
+
export declare function appendWorkContext(shaping: StepShaping, messages: Message[], stepNumber: number, prepared: PrepareStepResult): void;
|
|
26
|
+
export declare function stepContext(shaping: StepShaping, stepNumber: number, prepared: PrepareStepResult): PrepareStepContext;
|
|
27
|
+
/** Refuse the next call on a veto or hook error; do not skip a failed admission check. */
|
|
28
|
+
export declare function beforeStep(shaping: StepShaping, stepNumber: number): Promise<StepVeto | undefined>;
|
|
29
|
+
/** Shape the next request. A failed tuning stage is skipped; admission belongs to beforeStep. */
|
|
30
|
+
export declare function prepareStep(shaping: StepShaping, stepNumber: number): Promise<{
|
|
31
|
+
allowedTools?: string[];
|
|
32
|
+
toolChoice?: ToolChoice;
|
|
33
|
+
model?: string;
|
|
34
|
+
system?: string;
|
|
35
|
+
context?: string;
|
|
36
|
+
skills?: readonly Skill[];
|
|
37
|
+
temperature?: number;
|
|
38
|
+
maxResponseTokens?: number;
|
|
39
|
+
}>;
|
|
40
|
+
export declare function selectContextModel(shaping: StepShaping, model: string | undefined): Promise<void>;
|
|
41
|
+
//# sourceMappingURL=step-shaping.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"step-shaping.d.ts","sourceRoot":"","sources":["../../../../src/runtime/query/iteration/step-shaping.ts"],"names":[],"mappings":"AAKA,OAAO,EACN,KAAK,OAAO,EACZ,KAAK,WAAW,EAGhB,MAAM,iCAAiC,CAAA;AACxC,OAAO,KAAK,EAAE,UAAU,EAAE,MAAM,iCAAiC,CAAA;AACjE,OAAO,KAAK,EACX,kBAAkB,EAClB,iBAAiB,EACjB,UAAU,EACV,QAAQ,EACR,MAAM,6BAA6B,CAAA;AACpC,OAAO,KAAK,EAAE,KAAK,EAAE,MAAM,gCAAgC,CAAA;AAI3D,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,mBAAmB,CAAA;AAEzD;;;;;;;;;;;GAWG;AACH,MAAM,WAAW,WAAW;IAC3B,QAAQ,CAAC,GAAG,EAAE,gBAAgB,CAAA;IAC9B,QAAQ,CAAC,iBAAiB,EAAE,MAAM,WAAW,GAAG,SAAS,CAAA;IACzD,QAAQ,CAAC,KAAK,EAAE,MAAM,SAAS,UAAU,EAAE,CAAA;CAC3C;AAED,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,MAAM,eAKjD;AAED,4GAA4G;AAC5G,wBAAgB,iBAAiB,CAChC,OAAO,EAAE,WAAW,EACpB,QAAQ,EAAE,OAAO,EAAE,EACnB,UAAU,EAAE,MAAM,EAClB,QAAQ,EAAE,iBAAiB,GACzB,IAAI,CAmBN;AAED,wBAAgB,WAAW,CAC1B,OAAO,EAAE,WAAW,EACpB,UAAU,EAAE,MAAM,EAClB,QAAQ,EAAE,iBAAiB,GACzB,kBAAkB,CAuCpB;AAED,0FAA0F;AAC1F,wBAAsB,UAAU,CAC/B,OAAO,EAAE,WAAW,EACpB,UAAU,EAAE,MAAM,GAChB,OAAO,CAAC,QAAQ,GAAG,SAAS,CAAC,CAU/B;AAED,iGAAiG;AACjG,wBAAsB,WAAW,CAChC,OAAO,EAAE,WAAW,EACpB,UAAU,EAAE,MAAM,GAChB,OAAO,CAAC;IACV,YAAY,CAAC,EAAE,MAAM,EAAE,CAAA;IACvB,UAAU,CAAC,EAAE,UAAU,CAAA;IACvB,KAAK,CAAC,EAAE,MAAM,CAAA;IACd,MAAM,CAAC,EAAE,MAAM,CAAA;IACf,OAAO,CAAC,EAAE,MAAM,CAAA;IAChB,MAAM,CAAC,EAAE,SAAS,KAAK,EAAE,CAAA;IACzB,WAAW,CAAC,EAAE,MAAM,CAAA;IACpB,iBAAiB,CAAC,EAAE,MAAM,CAAA;CAC1B,CAAC,CAuGD;AAED,wBAAsB,kBAAkB,CACvC,OAAO,EAAE,WAAW,EACpB,KAAK,EAAE,MAAM,GAAG,SAAS,GACvB,OAAO,CAAC,IAAI,CAAC,CAYf"}
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
import { resolveContextWindow } from '../../../compaction/context-window.js';
|
|
2
|
+
import { estimateMessageTokens } from '../../../compaction/token-estimate.js';
|
|
3
|
+
import { NAMZU } from '../../../constants/telemetry/index.js';
|
|
4
|
+
import { renderSkillsSection } from '../../../persona/assembler.js';
|
|
5
|
+
import { PreparationContextError } from '../../../run/preparation-context-error.js';
|
|
6
|
+
import { createRuntimeContextMessage, createSystemMessage, } from '../../../types/message/index.js';
|
|
7
|
+
import { toErrorMessage } from '../../../utils/error.js';
|
|
8
|
+
import { createCallbackInference } from '../callback-inference.js';
|
|
9
|
+
import { measureContext } from './phases/compaction.js';
|
|
10
|
+
export function stepContextMessage(content) {
|
|
11
|
+
return createRuntimeContextMessage(`Current step context (runtime-generated; not a new user request):\n${content}`, 'step-context');
|
|
12
|
+
}
|
|
13
|
+
/** Derived after request projection; never accumulates in canonical history or replaces operator intent. */
|
|
14
|
+
export function appendWorkContext(shaping, messages, stepNumber, prepared) {
|
|
15
|
+
const { ctx } = shaping;
|
|
16
|
+
const contributions = [
|
|
17
|
+
ctx.completionInbox?.describeOwnedWork(),
|
|
18
|
+
ctx.toolExecutor.describeFileEvidence(messages),
|
|
19
|
+
].filter((content) => Boolean(content));
|
|
20
|
+
if (contributions.length === 0)
|
|
21
|
+
return;
|
|
22
|
+
let room = stepContext(shaping, stepNumber, prepared).contextBudget?.remainingTokens ?? 0;
|
|
23
|
+
// Leave room for the actual task; admit whole contributions, never dangling partial references.
|
|
24
|
+
if (room < 1_500)
|
|
25
|
+
return;
|
|
26
|
+
for (const content of contributions) {
|
|
27
|
+
if (!content || content.length > 8_000)
|
|
28
|
+
continue;
|
|
29
|
+
const message = stepContextMessage(content);
|
|
30
|
+
const tokens = estimateMessageTokens(message);
|
|
31
|
+
if (tokens > Math.min(2_000, room - 1_000))
|
|
32
|
+
continue;
|
|
33
|
+
messages.push(message);
|
|
34
|
+
room -= tokens;
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
export function stepContext(shaping, stepNumber, prepared) {
|
|
38
|
+
const { ctx, latestUserMessage, steps } = shaping;
|
|
39
|
+
const model = prepared.model ?? ctx.runConfig.model;
|
|
40
|
+
const window = resolveContextWindow(ctx.compactionConfig?.contextWindowTokens, model, model === ctx.runConfig.model
|
|
41
|
+
? ctx.providerContextWindow
|
|
42
|
+
: model === ctx.contextModel
|
|
43
|
+
? ctx.activeProviderContextWindow
|
|
44
|
+
: undefined);
|
|
45
|
+
const skills = prepared.skills ? renderSkillsSection([...prepared.skills]) : null;
|
|
46
|
+
const preamble = [prepared.system, skills].filter(Boolean).join('\n\n');
|
|
47
|
+
const preparedTokens = (preamble ? estimateMessageTokens(createSystemMessage(preamble)) : 0) +
|
|
48
|
+
(prepared.context ? estimateMessageTokens(stepContextMessage(prepared.context)) : 0);
|
|
49
|
+
const responseReserve = Math.min(prepared.maxResponseTokens ?? ctx.runConfig.maxResponseTokens ?? Math.floor(window.tokens / 4), Math.floor(window.tokens / 4));
|
|
50
|
+
return {
|
|
51
|
+
runId: ctx.runMgr.id,
|
|
52
|
+
stepNumber,
|
|
53
|
+
messages: ctx.runMgr.messages,
|
|
54
|
+
...(ctx.captureRunEvidence ? { captureRunEvidence: ctx.captureRunEvidence } : {}),
|
|
55
|
+
...(latestUserMessage() ? { latestUserMessage: latestUserMessage() } : {}),
|
|
56
|
+
signal: ctx.abortController.signal,
|
|
57
|
+
contextBudget: {
|
|
58
|
+
windowTokens: window.tokens,
|
|
59
|
+
remainingTokens: Math.max(0, Math.floor(window.tokens - measureContext(ctx).tokens - preparedTokens - responseReserve)),
|
|
60
|
+
},
|
|
61
|
+
steps: steps(),
|
|
62
|
+
prepared,
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
/** Refuse the next call on a veto or hook error; do not skip a failed admission check. */
|
|
66
|
+
export async function beforeStep(shaping, stepNumber) {
|
|
67
|
+
const { ctx } = shaping;
|
|
68
|
+
const configured = ctx.beforeStep;
|
|
69
|
+
if (!configured)
|
|
70
|
+
return undefined;
|
|
71
|
+
try {
|
|
72
|
+
return (await configured(stepContext(shaping, stepNumber, {}))) ?? undefined;
|
|
73
|
+
}
|
|
74
|
+
catch (err) {
|
|
75
|
+
return { reason: `beforeStep threw: ${toErrorMessage(err)}` };
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
/** Shape the next request. A failed tuning stage is skipped; admission belongs to beforeStep. */
|
|
79
|
+
export async function prepareStep(shaping, stepNumber) {
|
|
80
|
+
const { ctx } = shaping;
|
|
81
|
+
const configured = ctx.prepareStep;
|
|
82
|
+
if (!configured)
|
|
83
|
+
return {};
|
|
84
|
+
const stages = Array.isArray(configured) ? configured : [configured];
|
|
85
|
+
// Folded in DECLARATION order, each stage seeing what the ones
|
|
86
|
+
// before it decided. A later stage overriding a field is last-writer
|
|
87
|
+
// wins — visibly, because the order is a line in the host's code
|
|
88
|
+
// rather than an accident of install history.
|
|
89
|
+
let result = {};
|
|
90
|
+
for (const stage of stages) {
|
|
91
|
+
const inference = createCallbackInference(ctx, result.model ?? ctx.runConfig.model, 'preparation');
|
|
92
|
+
try {
|
|
93
|
+
const decided = await stage({
|
|
94
|
+
...stepContext(shaping, stepNumber, result),
|
|
95
|
+
generateText: inference.generateText,
|
|
96
|
+
});
|
|
97
|
+
if (decided)
|
|
98
|
+
result = { ...result, ...decided };
|
|
99
|
+
await selectContextModel(shaping, result.model ?? ctx.runConfig.model);
|
|
100
|
+
}
|
|
101
|
+
catch (err) {
|
|
102
|
+
// Skipped, and the rest still run: one broken concern must
|
|
103
|
+
// not silently disable the others it was declared beside.
|
|
104
|
+
ctx.log.error('a prepareStep stage threw — skipping it', {
|
|
105
|
+
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
106
|
+
'namzu.runtime.step_number': stepNumber,
|
|
107
|
+
'exception.message': toErrorMessage(err),
|
|
108
|
+
});
|
|
109
|
+
// An SDK stage may report availability and validated fallback evidence
|
|
110
|
+
// without exposing its error. Preserve prior decisions and the context budget;
|
|
111
|
+
// ordinary exceptions still contribute nothing to the model request.
|
|
112
|
+
if (err instanceof PreparationContextError && !ctx.abortController.signal.aborted) {
|
|
113
|
+
const room = stepContext(shaping, stepNumber, result).contextBudget?.remainingTokens ?? 0;
|
|
114
|
+
if (typeof err.context === 'string' &&
|
|
115
|
+
err.context.length > 0 &&
|
|
116
|
+
err.context.length + (result.context ? 2 : 0) <= Math.min(12_000, Math.floor(room)))
|
|
117
|
+
result = {
|
|
118
|
+
...result,
|
|
119
|
+
context: [result.context, err.context].filter(Boolean).join('\n\n'),
|
|
120
|
+
};
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
finally {
|
|
124
|
+
inference.close();
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
const prepared = {};
|
|
128
|
+
if (result.activeTools) {
|
|
129
|
+
const known = result.activeTools.filter((name) => ctx.tools.has(name));
|
|
130
|
+
const unknown = result.activeTools.filter((name) => !ctx.tools.has(name));
|
|
131
|
+
if (unknown.length > 0) {
|
|
132
|
+
// The all-unknown case gets its own sentence because it has its
|
|
133
|
+
// own consequence. Some names dropped narrows the step; ALL of
|
|
134
|
+
// them dropped leaves it able to call nothing — which is the
|
|
135
|
+
// honest reading of "only these tools" when none of them exist,
|
|
136
|
+
// and is not what a reader of "ignoring them" would expect.
|
|
137
|
+
//
|
|
138
|
+
// Widening back to the run's list would be worse: it grants
|
|
139
|
+
// exactly the tools the caller asked to exclude, on the grounds
|
|
140
|
+
// that their own list failed. A step that can call nothing is
|
|
141
|
+
// constrained; a step that can call everything is a control
|
|
142
|
+
// that stopped applying.
|
|
143
|
+
const message = known.length === 0
|
|
144
|
+
? 'prepareStep named only tools that are not registered — this step can call nothing'
|
|
145
|
+
: 'prepareStep named tools that are not registered — ignoring them';
|
|
146
|
+
ctx.log.warn(message, {
|
|
147
|
+
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
148
|
+
'namzu.runtime.step_number': stepNumber,
|
|
149
|
+
'namzu.runtime.unknown': unknown,
|
|
150
|
+
'namzu.runtime.remaining': known.length,
|
|
151
|
+
});
|
|
152
|
+
}
|
|
153
|
+
prepared.allowedTools = known;
|
|
154
|
+
}
|
|
155
|
+
if (result.toolChoice !== undefined)
|
|
156
|
+
prepared.toolChoice = result.toolChoice;
|
|
157
|
+
if (result.model !== undefined)
|
|
158
|
+
prepared.model = result.model;
|
|
159
|
+
if (result.system !== undefined)
|
|
160
|
+
prepared.system = result.system;
|
|
161
|
+
if (result.context !== undefined)
|
|
162
|
+
prepared.context = result.context;
|
|
163
|
+
if (result.skills !== undefined)
|
|
164
|
+
prepared.skills = result.skills;
|
|
165
|
+
if (result.temperature !== undefined)
|
|
166
|
+
prepared.temperature = result.temperature;
|
|
167
|
+
if (result.maxResponseTokens !== undefined) {
|
|
168
|
+
prepared.maxResponseTokens = result.maxResponseTokens;
|
|
169
|
+
}
|
|
170
|
+
return prepared;
|
|
171
|
+
}
|
|
172
|
+
export async function selectContextModel(shaping, model) {
|
|
173
|
+
const { ctx } = shaping;
|
|
174
|
+
if (model !== (ctx.contextModel ?? ctx.runConfig.model)) {
|
|
175
|
+
// A measurement from another tokenizer cannot price the new request.
|
|
176
|
+
ctx.runMgr.clearLastPromptTokens();
|
|
177
|
+
}
|
|
178
|
+
ctx.contextModel = model;
|
|
179
|
+
ctx.activeProviderContextWindow =
|
|
180
|
+
model && model !== ctx.runConfig.model && !ctx.compactionConfig?.contextWindowTokens
|
|
181
|
+
? await ctx.resolveModelContextWindow?.(model)
|
|
182
|
+
: undefined;
|
|
183
|
+
}
|
|
184
|
+
//# sourceMappingURL=step-shaping.js.map
|