@namzu/sdk 39.0.0 → 40.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +151 -0
- package/dist/connector/mcp/adapter.d.ts.map +1 -1
- package/dist/connector/mcp/adapter.js +20 -6
- package/dist/connector/mcp/adapter.js.map +1 -1
- package/dist/manager/run/persistence.d.ts +8 -0
- package/dist/manager/run/persistence.d.ts.map +1 -1
- package/dist/manager/run/persistence.js +12 -0
- package/dist/manager/run/persistence.js.map +1 -1
- package/dist/public-runtime.d.ts +3 -1
- package/dist/public-runtime.d.ts.map +1 -1
- package/dist/public-runtime.js +5 -1
- package/dist/public-runtime.js.map +1 -1
- package/dist/public-tools.d.ts +11 -0
- package/dist/public-tools.d.ts.map +1 -1
- package/dist/public-tools.js +14 -0
- package/dist/public-tools.js.map +1 -1
- package/dist/registry/tool/execute.d.ts.map +1 -1
- package/dist/registry/tool/execute.js +2 -3
- package/dist/registry/tool/execute.js.map +1 -1
- package/dist/registry/tool/portable.d.ts +65 -0
- package/dist/registry/tool/portable.d.ts.map +1 -0
- package/dist/registry/tool/portable.js +244 -0
- package/dist/registry/tool/portable.js.map +1 -0
- package/dist/registry/tool/schema.d.ts +32 -5
- package/dist/registry/tool/schema.d.ts.map +1 -1
- package/dist/registry/tool/schema.js +35 -9
- package/dist/registry/tool/schema.js.map +1 -1
- package/dist/registry/toolset/catalog.js +8 -8
- package/dist/registry/toolset/catalog.js.map +1 -1
- package/dist/runtime/jobs/awaited-jobs.d.ts +215 -0
- package/dist/runtime/jobs/awaited-jobs.d.ts.map +1 -0
- package/dist/runtime/jobs/awaited-jobs.js +259 -0
- package/dist/runtime/jobs/awaited-jobs.js.map +1 -0
- package/dist/runtime/jobs/registry.d.ts +33 -2
- package/dist/runtime/jobs/registry.d.ts.map +1 -1
- package/dist/runtime/jobs/registry.js +37 -0
- package/dist/runtime/jobs/registry.js.map +1 -1
- package/dist/runtime/query/executor.d.ts +28 -0
- package/dist/runtime/query/executor.d.ts.map +1 -1
- package/dist/runtime/query/executor.js +39 -1
- package/dist/runtime/query/executor.js.map +1 -1
- package/dist/runtime/query/file-evidence-context.d.ts.map +1 -1
- package/dist/runtime/query/file-evidence-context.js +159 -43
- package/dist/runtime/query/file-evidence-context.js.map +1 -1
- package/dist/runtime/query/file-evidence-replay.d.ts +260 -0
- package/dist/runtime/query/file-evidence-replay.d.ts.map +1 -0
- package/dist/runtime/query/file-evidence-replay.js +647 -0
- package/dist/runtime/query/file-evidence-replay.js.map +1 -0
- package/dist/runtime/query/file-evidence-seed.d.ts +50 -0
- package/dist/runtime/query/file-evidence-seed.d.ts.map +1 -0
- package/dist/runtime/query/file-evidence-seed.js +100 -0
- package/dist/runtime/query/file-evidence-seed.js.map +1 -0
- package/dist/runtime/query/index.d.ts.map +1 -1
- package/dist/runtime/query/index.js +94 -2
- package/dist/runtime/query/index.js.map +1 -1
- package/dist/runtime/query/iteration/index.d.ts +87 -9
- package/dist/runtime/query/iteration/index.d.ts.map +1 -1
- package/dist/runtime/query/iteration/index.js +193 -28
- package/dist/runtime/query/iteration/index.js.map +1 -1
- package/dist/runtime/query/iteration/phases/context.d.ts +10 -0
- package/dist/runtime/query/iteration/phases/context.d.ts.map +1 -1
- package/dist/runtime/query/iteration/phases/context.js.map +1 -1
- package/dist/runtime/query/iteration/phases/tool-review.d.ts.map +1 -1
- package/dist/runtime/query/iteration/phases/tool-review.js +5 -1
- package/dist/runtime/query/iteration/phases/tool-review.js.map +1 -1
- package/dist/runtime/query/plugin-hooks.d.ts +14 -0
- package/dist/runtime/query/plugin-hooks.d.ts.map +1 -1
- package/dist/runtime/query/plugin-hooks.js +18 -0
- package/dist/runtime/query/plugin-hooks.js.map +1 -1
- package/dist/runtime/query/repeat-call.d.ts +17 -4
- package/dist/runtime/query/repeat-call.d.ts.map +1 -1
- package/dist/runtime/query/repeat-call.js +26 -19
- package/dist/runtime/query/repeat-call.js.map +1 -1
- package/dist/runtime/query/steering.d.ts +11 -1
- package/dist/runtime/query/steering.d.ts.map +1 -1
- package/dist/runtime/query/steering.js +12 -1
- package/dist/runtime/query/steering.js.map +1 -1
- package/dist/runtime/query/tooling.d.ts +2 -0
- package/dist/runtime/query/tooling.d.ts.map +1 -1
- package/dist/runtime/query/tooling.js +1 -0
- package/dist/runtime/query/tooling.js.map +1 -1
- package/dist/scheduler/completion-inbox.d.ts +48 -2
- package/dist/scheduler/completion-inbox.d.ts.map +1 -1
- package/dist/scheduler/completion-inbox.js +102 -10
- package/dist/scheduler/completion-inbox.js.map +1 -1
- package/dist/tools/builtins/bash.d.ts.map +1 -1
- package/dist/tools/builtins/bash.js +4 -10
- package/dist/tools/builtins/bash.js.map +1 -1
- package/dist/tools/builtins/edit-apply.d.ts +126 -0
- package/dist/tools/builtins/edit-apply.d.ts.map +1 -0
- package/dist/tools/builtins/edit-apply.js +360 -0
- package/dist/tools/builtins/edit-apply.js.map +1 -0
- package/dist/tools/builtins/edit.d.ts +143 -1
- package/dist/tools/builtins/edit.d.ts.map +1 -1
- package/dist/tools/builtins/edit.js +37 -219
- package/dist/tools/builtins/edit.js.map +1 -1
- package/dist/tools/builtins/index.d.ts +1 -0
- package/dist/tools/builtins/index.d.ts.map +1 -1
- package/dist/tools/builtins/index.js +9 -3
- package/dist/tools/builtins/index.js.map +1 -1
- package/dist/tools/builtins/job.js +1 -1
- package/dist/tools/builtins/job.js.map +1 -1
- package/dist/tools/builtins/read-file.d.ts +2 -2
- package/dist/tools/builtins/read-file.d.ts.map +1 -1
- package/dist/tools/builtins/read-file.js +50 -65
- package/dist/tools/builtins/read-file.js.map +1 -1
- package/dist/tools/builtins/read-render.d.ts +56 -0
- package/dist/tools/builtins/read-render.d.ts.map +1 -0
- package/dist/tools/builtins/read-render.js +73 -0
- package/dist/tools/builtins/read-render.js.map +1 -0
- package/dist/tools/builtins/wait-for-job-bounds.d.ts +67 -0
- package/dist/tools/builtins/wait-for-job-bounds.d.ts.map +1 -0
- package/dist/tools/builtins/wait-for-job-bounds.js +108 -0
- package/dist/tools/builtins/wait-for-job-bounds.js.map +1 -0
- package/dist/tools/builtins/wait-for-job.d.ts +6 -0
- package/dist/tools/builtins/wait-for-job.d.ts.map +1 -0
- package/dist/tools/builtins/wait-for-job.js +162 -0
- package/dist/tools/builtins/wait-for-job.js.map +1 -0
- package/dist/tools/builtins/write-file.js +5 -0
- package/dist/tools/builtins/write-file.js.map +1 -1
- package/dist/tools/coordinator/index.d.ts.map +1 -1
- package/dist/tools/coordinator/index.js +1 -7
- package/dist/tools/coordinator/index.js.map +1 -1
- package/dist/tools/file-read-tracker.d.ts.map +1 -1
- package/dist/tools/file-read-tracker.js +88 -10
- package/dist/tools/file-read-tracker.js.map +1 -1
- package/dist/types/message/index.d.ts +1 -1
- package/dist/types/message/index.d.ts.map +1 -1
- package/dist/types/message/index.js +2 -0
- package/dist/types/message/index.js.map +1 -1
- package/dist/types/run/entity.d.ts +13 -0
- package/dist/types/run/entity.d.ts.map +1 -1
- package/dist/types/sandbox/index.d.ts +15 -14
- package/dist/types/sandbox/index.d.ts.map +1 -1
- package/dist/types/sandbox/index.js.map +1 -1
- package/dist/types/tool/index.d.ts +109 -0
- package/dist/types/tool/index.d.ts.map +1 -1
- package/dist/types/tool/index.js.map +1 -1
- package/dist/utils/env.d.ts +19 -0
- package/dist/utils/env.d.ts.map +1 -0
- package/dist/utils/env.js +25 -0
- package/dist/utils/env.js.map +1 -0
- package/package.json +1 -1
- package/src/connector/mcp/adapter.ts +20 -6
- package/src/manager/run/persistence.ts +12 -0
- package/src/public-runtime.ts +9 -1
- package/src/public-tools.ts +18 -0
- package/src/registry/tool/execute.ts +2 -4
- package/src/registry/tool/portable.ts +264 -0
- package/src/registry/tool/schema.ts +38 -8
- package/src/registry/toolset/catalog.ts +8 -9
- package/src/runtime/jobs/awaited-jobs.ts +271 -0
- package/src/runtime/jobs/registry.ts +50 -0
- package/src/runtime/query/executor.ts +49 -1
- package/src/runtime/query/file-evidence-context.ts +190 -46
- package/src/runtime/query/file-evidence-replay.ts +776 -0
- package/src/runtime/query/file-evidence-seed.ts +126 -0
- package/src/runtime/query/index.ts +104 -2
- package/src/runtime/query/iteration/index.ts +202 -28
- package/src/runtime/query/iteration/phases/context.ts +10 -0
- package/src/runtime/query/iteration/phases/tool-review.ts +4 -0
- package/src/runtime/query/plugin-hooks.ts +20 -0
- package/src/runtime/query/repeat-call.ts +28 -18
- package/src/runtime/query/steering.ts +11 -0
- package/src/runtime/query/tooling.ts +3 -0
- package/src/scheduler/completion-inbox.ts +105 -9
- package/src/tools/builtins/bash.ts +4 -10
- package/src/tools/builtins/edit-apply.ts +456 -0
- package/src/tools/builtins/edit.ts +39 -270
- package/src/tools/builtins/index.ts +9 -3
- package/src/tools/builtins/job.ts +1 -1
- package/src/tools/builtins/read-file.ts +56 -77
- package/src/tools/builtins/read-render.ts +104 -0
- package/src/tools/builtins/wait-for-job-bounds.ts +179 -0
- package/src/tools/builtins/wait-for-job.ts +184 -0
- package/src/tools/builtins/write-file.ts +5 -0
- package/src/tools/coordinator/index.ts +1 -7
- package/src/tools/file-read-tracker.ts +85 -7
- package/src/types/message/index.ts +2 -0
- package/src/types/run/entity.ts +14 -0
- package/src/types/sandbox/index.ts +15 -14
- package/src/types/tool/index.ts +104 -0
- package/src/utils/env.ts +23 -0
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
import { resolveWithinAnyReal, toolRoots } from '../../tools/paths.js'
|
|
2
|
+
import type { Message } from '../../types/message/index.js'
|
|
3
|
+
import type { FileReadTracker } from '../../types/tool/index.js'
|
|
4
|
+
import {
|
|
5
|
+
type FileKeyResolver,
|
|
6
|
+
type LedgerReplayReport,
|
|
7
|
+
type PathAttributions,
|
|
8
|
+
collectObservedPaths,
|
|
9
|
+
createPathAttributions,
|
|
10
|
+
replayObservationLedger,
|
|
11
|
+
} from './file-evidence-replay.js'
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Seeding an observation ledger from history, in the key space the tools use.
|
|
15
|
+
*
|
|
16
|
+
* The replay itself is pure and stays that way. What lives here is the one
|
|
17
|
+
* thing it cannot do for itself: work out which ledger key each path in the
|
|
18
|
+
* history belongs to. On a host that is not `resolve(cwd, path)` — `write`,
|
|
19
|
+
* `edit` and `read` all key on `resolveWithinAnyReal`, which canonicalizes
|
|
20
|
+
* every symlink on the way, because a ledger entry has to identify a FILE and
|
|
21
|
+
* two spellings of one file must not become two entries.
|
|
22
|
+
*
|
|
23
|
+
* Seeding lexically instead is worse than useless. The projection looks up
|
|
24
|
+
* lexically too, so it would match a fingerprint the seed had just written into
|
|
25
|
+
* a key space nothing else touches — while every drift refusal, made by a tool,
|
|
26
|
+
* lands on the canonical key and never withdraws it. The runtime would go on
|
|
27
|
+
* telling the model a body it can no longer vouch for, with no mechanism left
|
|
28
|
+
* to take it back. So the resolution happens here, with the tools' own
|
|
29
|
+
* function, and a path that will not resolve gets no key at all.
|
|
30
|
+
*/
|
|
31
|
+
const NOTHING: LedgerReplayReport = { pathsWitnessed: 0, pathsSeen: 0, unitsReplayed: 0 }
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Distinct paths one seeding will canonicalize before it replays anything.
|
|
35
|
+
*
|
|
36
|
+
* A ceiling rather than a budget because the work is uniform: each path is a
|
|
37
|
+
* handful of `realpath` calls, once per conversation. Counted on distinct
|
|
38
|
+
* SPELLINGS, the ones only `read` names included — a read has to be keyed too
|
|
39
|
+
* or it cannot contradict a claim — so two spellings of one file count twice.
|
|
40
|
+
* A history naming more paths than this is not a conversation whose ledger is
|
|
41
|
+
* worth guessing at, and it seeds nothing rather than resolving a prefix and
|
|
42
|
+
* abandoning the rest — a partially keyed walk is a walk that cannot say what
|
|
43
|
+
* a mutation replaced. That is a total loss for the conversation, not a partial
|
|
44
|
+
* one: it starts from the empty ledger a resume has always started from, and
|
|
45
|
+
* its first mutation of each path re-establishes it.
|
|
46
|
+
*/
|
|
47
|
+
export const MAX_RESOLVED_PATHS = 1024
|
|
48
|
+
|
|
49
|
+
/** What a caller needs to key a path the way its tools will. */
|
|
50
|
+
export interface ObservationSeedContext {
|
|
51
|
+
readonly workingDirectory: string
|
|
52
|
+
/** See `ToolContext.additionalDirectories`; part of the tools' resolution. */
|
|
53
|
+
readonly additionalDirectories?: readonly string[]
|
|
54
|
+
/** True when this run's tools address a sandbox, whose keys are paths as written. */
|
|
55
|
+
readonly sandboxed?: boolean
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Rebuild `tracker` from a conversation's own `messages`.
|
|
60
|
+
*
|
|
61
|
+
* For a host that keeps one tracker per conversation and has just restored one:
|
|
62
|
+
* the ledger is process memory, so a resumed conversation starts with nothing
|
|
63
|
+
* in it and re-reads files whose whole body is already in the transcript. Call
|
|
64
|
+
* it once, before the conversation's first request. Reading files is confined
|
|
65
|
+
* to canonicalizing the paths in the history; no file's CONTENT is read, and
|
|
66
|
+
* every body restored is one the visible calls reconstruct exactly.
|
|
67
|
+
*
|
|
68
|
+
* Content-backed observations, and only those. A path the walk cannot rebuild
|
|
69
|
+
* is left out of the ledger rather than entered without a fingerprint, so
|
|
70
|
+
* `write`'s read-before-overwrite refusal stands over it exactly as it does
|
|
71
|
+
* against the empty ledger a resume gets today. Three things seed nothing at
|
|
72
|
+
* all, and the report says so: a history naming more than
|
|
73
|
+
* {@link MAX_RESOLVED_PATHS} distinct path spellings, one whose ids are
|
|
74
|
+
* ambiguous, and one holding a mutation no path can be recovered from —
|
|
75
|
+
* whatever the transcript says came back to that mutation, since a key is what
|
|
76
|
+
* withdrawing one path rather than the whole pass takes. A call whose arguments
|
|
77
|
+
* are too long to read as JSON at all is one of that last kind; the replay
|
|
78
|
+
* states the ceiling.
|
|
79
|
+
*/
|
|
80
|
+
export async function seedObservationLedger(
|
|
81
|
+
messages: readonly Message[],
|
|
82
|
+
tracker: FileReadTracker,
|
|
83
|
+
context: ObservationSeedContext,
|
|
84
|
+
): Promise<LedgerReplayReport> {
|
|
85
|
+
// One attribution cache across both halves. The path collection and the
|
|
86
|
+
// walk ask the same question of the same calls — which file did this touch
|
|
87
|
+
// — and answering it means reading a string that, for a `write`, is a whole
|
|
88
|
+
// file body. Asking once per seeding rather than once per half is why the
|
|
89
|
+
// replay's own ceiling on that read can be as generous as it is.
|
|
90
|
+
const attributions = createPathAttributions()
|
|
91
|
+
const keyOf = await resolveObservedFileKeys(messages, context, attributions)
|
|
92
|
+
return keyOf ? replayObservationLedger(messages, tracker, keyOf, attributions) : NOTHING
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* The tools' key for every path this history names, or `undefined` when the
|
|
97
|
+
* pass should not run at all.
|
|
98
|
+
*
|
|
99
|
+
* `undefined` for a history with nothing to reconstruct and for one past the
|
|
100
|
+
* ceiling. A path the resolver refuses — it escapes the roots this run may
|
|
101
|
+
* reach, or the working directory itself is gone — is simply left unkeyed;
|
|
102
|
+
* the replay then treats the mutation that named it as unattributable, which
|
|
103
|
+
* is the fail-closed answer and the same one it gives a call it cannot parse.
|
|
104
|
+
*/
|
|
105
|
+
async function resolveObservedFileKeys(
|
|
106
|
+
messages: readonly Message[],
|
|
107
|
+
context: ObservationSeedContext,
|
|
108
|
+
attributions: PathAttributions,
|
|
109
|
+
): Promise<FileKeyResolver | undefined> {
|
|
110
|
+
const paths = collectObservedPaths(messages, attributions)
|
|
111
|
+
if (paths.length === 0 || paths.length > MAX_RESOLVED_PATHS) return undefined
|
|
112
|
+
// A sandbox has its own root and its own resolver, and the tools key on the
|
|
113
|
+
// path as written there. Canonicalizing it against the host filesystem would
|
|
114
|
+
// ask about the wrong machine.
|
|
115
|
+
if (context.sandboxed) return (path: string) => path
|
|
116
|
+
const roots = toolRoots(context)
|
|
117
|
+
const keys = new Map<string, string>()
|
|
118
|
+
for (const path of paths) {
|
|
119
|
+
try {
|
|
120
|
+
keys.set(path, await resolveWithinAnyReal(roots, path))
|
|
121
|
+
} catch {
|
|
122
|
+
// Left unkeyed on purpose; see above.
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
return (path: string) => keys.get(path)
|
|
126
|
+
}
|
|
@@ -112,6 +112,7 @@ import { toErrorMessage } from '../../utils/error.js'
|
|
|
112
112
|
import { generateCheckpointId, generateRunId } from '../../utils/id.js'
|
|
113
113
|
import { errorAttributes } from '../../utils/log/exception.js'
|
|
114
114
|
import type { Logger } from '../../utils/logger.js'
|
|
115
|
+
import { AwaitedJobs } from '../jobs/awaited-jobs.js'
|
|
115
116
|
import type { BackgroundJobRegistry } from '../jobs/registry.js'
|
|
116
117
|
import { AUTO_APPROVE_POLICY_NAME, createRunApprovalPolicy } from './approval-policy.js'
|
|
117
118
|
import { CheckpointManager } from './checkpoint.js'
|
|
@@ -1034,6 +1035,56 @@ function withoutOwnedResumeTurn(
|
|
|
1034
1035
|
)
|
|
1035
1036
|
}
|
|
1036
1037
|
|
|
1038
|
+
/**
|
|
1039
|
+
* The history plus the part of the owned resume turn that already RAN, for the
|
|
1040
|
+
* observation ledger to be seeded from.
|
|
1041
|
+
*
|
|
1042
|
+
* `withoutOwnedResumeTurn` takes that turn out so generic repair cannot answer
|
|
1043
|
+
* it, and the plan re-appends it much later — after the sandbox exists, after
|
|
1044
|
+
* the input guardrails, immediately before the loop. Seeding from the list the
|
|
1045
|
+
* model finally sees would therefore have to happen after `applyPendingResume`,
|
|
1046
|
+
* and that is the wrong seam for a reason that is not about ordering: the plan
|
|
1047
|
+
* does not merely re-append the turn, it EXECUTES the calls in it that never
|
|
1048
|
+
* started. Those tools read the ledger this seeding builds, so a seed placed
|
|
1049
|
+
* after them would refuse the very write the resume exists to carry out — no
|
|
1050
|
+
* `hasRead` for a file the conversation had read three turns earlier.
|
|
1051
|
+
*
|
|
1052
|
+
* So the turn is folded in here instead, and only as far as it actually got.
|
|
1053
|
+
* A call `recoverCompletedCalls` found an outcome for is one that ran: its
|
|
1054
|
+
* receipt goes in beside it, and the walk reads it exactly as it reads any
|
|
1055
|
+
* other — a completed `write` restores the body it put there, and the
|
|
1056
|
+
* unknown-outcome result the recovery writes for an interrupted one withdraws
|
|
1057
|
+
* the path instead. A call ABSENT from that map is absent because a complete
|
|
1058
|
+
* scan proved it has no recorded start, so the file it names is untouched and
|
|
1059
|
+
* the claim history established for it still stands; leaving it out is what
|
|
1060
|
+
* lets it execute in a moment. Without any of this the seed never saw the
|
|
1061
|
+
* turn at all, and an executed write inside it left the pre-write body standing
|
|
1062
|
+
* as a claim until the next mutation's drift check happened to catch it.
|
|
1063
|
+
*/
|
|
1064
|
+
function withOwnedResumeOutcomes(
|
|
1065
|
+
messages: readonly Message[],
|
|
1066
|
+
assistant: AssistantMessage,
|
|
1067
|
+
recovered: ReadonlyMap<string, { result: string; isError: boolean }>,
|
|
1068
|
+
): Message[] {
|
|
1069
|
+
const ran = (assistant.toolCalls ?? []).flatMap((call) => {
|
|
1070
|
+
const outcome = recovered.get(call.id)
|
|
1071
|
+
return outcome ? [{ call, outcome }] : []
|
|
1072
|
+
})
|
|
1073
|
+
if (ran.length === 0) return [...messages]
|
|
1074
|
+
return [
|
|
1075
|
+
...messages,
|
|
1076
|
+
{ ...assistant, toolCalls: ran.map(({ call }) => call) },
|
|
1077
|
+
...ran.map(
|
|
1078
|
+
({ call, outcome }): Message => ({
|
|
1079
|
+
role: 'tool',
|
|
1080
|
+
toolCallId: call.id,
|
|
1081
|
+
content: outcome.result,
|
|
1082
|
+
isError: outcome.isError,
|
|
1083
|
+
}),
|
|
1084
|
+
),
|
|
1085
|
+
]
|
|
1086
|
+
}
|
|
1087
|
+
|
|
1037
1088
|
export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run> {
|
|
1038
1089
|
assertMaxToolCalls(params.maxToolCalls)
|
|
1039
1090
|
// Required types do not protect JavaScript callers. Reject missing scope
|
|
@@ -1734,6 +1785,23 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
1734
1785
|
return source
|
|
1735
1786
|
}
|
|
1736
1787
|
|
|
1788
|
+
// Whose jobs this run speaks for: its own by default, the session's when
|
|
1789
|
+
// the host said so. Resolved before the tools are built, because the
|
|
1790
|
+
// wait-intent recorder below is bound into them.
|
|
1791
|
+
const jobOwner = params.backgroundJobOwner ?? ctx.runId
|
|
1792
|
+
// What the model reads when a job ends. Built up here, ahead of the
|
|
1793
|
+
// subscription that fills it below, because the wait-intent recorder needs
|
|
1794
|
+
// to ask whether its text has been read yet.
|
|
1795
|
+
const jobNotices = params.backgroundJobs ? new SteeringBinding() : undefined
|
|
1796
|
+
// Jobs the model told `wait_for_job` it is waiting on, which is the only
|
|
1797
|
+
// thing that can hold this run open for a job. Built only where there is a
|
|
1798
|
+
// registry, so a host with no background mode carries no recorder and the
|
|
1799
|
+
// bound ref has no `markAwaited` to offer.
|
|
1800
|
+
const awaitedJobs = params.backgroundJobs
|
|
1801
|
+
? new AwaitedJobs(params.backgroundJobs, jobOwner, () => jobNotices?.pending ?? false)
|
|
1802
|
+
: undefined
|
|
1803
|
+
awaitedJobs?.attach()
|
|
1804
|
+
|
|
1737
1805
|
const toolExecutor = ToolingBootstrap.init(
|
|
1738
1806
|
{
|
|
1739
1807
|
tools: params.tools,
|
|
@@ -1750,6 +1818,7 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
1750
1818
|
pluginManager: params.pluginManager,
|
|
1751
1819
|
...(params.backgroundJobs ? { backgroundJobs: params.backgroundJobs } : {}),
|
|
1752
1820
|
...(params.backgroundJobOwner ? { backgroundJobOwner: params.backgroundJobOwner } : {}),
|
|
1821
|
+
...(awaitedJobs ? { onJobAwaited: (id: string) => awaitedJobs.expect(id) } : {}),
|
|
1753
1822
|
// The `skill` tool's registry. Threaded from the run rather than
|
|
1754
1823
|
// held by the tool, because a tool that reached for a module-level
|
|
1755
1824
|
// registry would answer about whatever the last run configured.
|
|
@@ -1809,8 +1878,6 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
1809
1878
|
// result, and the host as an event — without either polling. Subscribed
|
|
1810
1879
|
// for the owner the run's jobs are bound to, so a session-owned job that
|
|
1811
1880
|
// ends during this run is reported here too.
|
|
1812
|
-
const jobOwner = params.backgroundJobOwner ?? ctx.runId
|
|
1813
|
-
const jobNotices = params.backgroundJobs ? new SteeringBinding() : undefined
|
|
1814
1881
|
const unsubscribeJobExits = params.backgroundJobs?.onExit((job) => {
|
|
1815
1882
|
if (job.owner !== jobOwner) return
|
|
1816
1883
|
const outcome =
|
|
@@ -2010,6 +2077,7 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
2010
2077
|
promptContributions,
|
|
2011
2078
|
...(params.steering ? { steering: params.steering } : {}),
|
|
2012
2079
|
...(jobNotices ? { jobNotices } : {}),
|
|
2080
|
+
...(awaitedJobs ? { awaitedJobs } : {}),
|
|
2013
2081
|
checkpointMgr,
|
|
2014
2082
|
planManager: ctx.planManager,
|
|
2015
2083
|
taskGateway: taskScheduler,
|
|
@@ -2368,6 +2436,37 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
2368
2436
|
})
|
|
2369
2437
|
}
|
|
2370
2438
|
|
|
2439
|
+
// The ledger is process state and a resumed run starts with an empty
|
|
2440
|
+
// one, so until something reads a file again this run knows nothing
|
|
2441
|
+
// about files the conversation already wrote in full — and re-reads
|
|
2442
|
+
// them. Rebuilt from the REPAIRED history, which is what the model
|
|
2443
|
+
// is about to be shown, rather than from the checkpoint's own
|
|
2444
|
+
// messages — plus whatever of the owned resume turn actually ran,
|
|
2445
|
+
// which the repaired list does not carry; see
|
|
2446
|
+
// `withOwnedResumeOutcomes`. Reads no file's content: every
|
|
2447
|
+
// fingerprint recovered here is still checked against the real one
|
|
2448
|
+
// at mutation time.
|
|
2449
|
+
try {
|
|
2450
|
+
await toolExecutor.seedFileObservations(
|
|
2451
|
+
pendingResume
|
|
2452
|
+
? withOwnedResumeOutcomes(restoredMessages, pendingResume.assistant, recoveredResults)
|
|
2453
|
+
: restoredMessages,
|
|
2454
|
+
params.sandboxProvider !== undefined,
|
|
2455
|
+
)
|
|
2456
|
+
} catch (err: unknown) {
|
|
2457
|
+
// A ledger that could not be rebuilt is the empty one every resume
|
|
2458
|
+
// used to get, so the run continues without its witnesses and the
|
|
2459
|
+
// model reads what it needs. Failing the resume over it would trade
|
|
2460
|
+
// a conversation that works for one that does not, to protect an
|
|
2461
|
+
// optimisation. Said out loud all the same, because a seeding that
|
|
2462
|
+
// failed and one that found nothing are otherwise the same silence.
|
|
2463
|
+
ctx.log.debug('Could not rebuild the file observation ledger on resume', {
|
|
2464
|
+
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
2465
|
+
'namzu.checkpoint.id': checkpoint.id,
|
|
2466
|
+
'exception.message': err instanceof Error ? err.message : String(err),
|
|
2467
|
+
})
|
|
2468
|
+
}
|
|
2469
|
+
|
|
2371
2470
|
for (const msg of restoredMessages) {
|
|
2372
2471
|
if (msg.role === 'system') {
|
|
2373
2472
|
// Re-push the FRESH static/dynamic floor (done above) but PRESERVE
|
|
@@ -2878,6 +2977,9 @@ export async function* query(params: QueryParams): AsyncGenerator<RunEvent, Run>
|
|
|
2878
2977
|
// Awaited, and its failure swallowed. A job that would not die is
|
|
2879
2978
|
// worth a log line, and is not worth retracting a run's answer.
|
|
2880
2979
|
unsubscribeJobExits?.()
|
|
2980
|
+
// The wait-intent recorder listens on the same shared registry and
|
|
2981
|
+
// leaks the same way if it is left attached.
|
|
2982
|
+
awaitedJobs?.close()
|
|
2881
2983
|
// Only jobs bound to this run. Jobs a host bound to its session are
|
|
2882
2984
|
// the host's to stop, when the session ends.
|
|
2883
2985
|
if (params.backgroundJobs && (params.backgroundJobOwner ?? ctx.runId) === ctx.runId) {
|
|
@@ -51,6 +51,7 @@ import type {
|
|
|
51
51
|
} from '../../../types/run/index.js'
|
|
52
52
|
import type { Skill } from '../../../types/skills/index.js'
|
|
53
53
|
import type { LLMToolSchema, ToolRegistryContract } from '../../../types/tool/index.js'
|
|
54
|
+
import { readPositiveIntEnv } from '../../../utils/env.js'
|
|
54
55
|
import { toErrorMessage } from '../../../utils/error.js'
|
|
55
56
|
import { stableDigest } from '../../../utils/hash.js'
|
|
56
57
|
import { generateMessageId } from '../../../utils/id.js'
|
|
@@ -69,7 +70,7 @@ import {
|
|
|
69
70
|
markProviderRejectedImage,
|
|
70
71
|
projectRequestRichContent,
|
|
71
72
|
} from '../request-rich-content.js'
|
|
72
|
-
import { formatSteeringNote, isOperatorUserMessage } from '../steering.js'
|
|
73
|
+
import { formatJobNote, formatSteeringNote, isOperatorUserMessage } from '../steering.js'
|
|
73
74
|
import { parseNativeCandidate } from './native-output.js'
|
|
74
75
|
import { runAdvisoryPhase } from './phases/advisory.js'
|
|
75
76
|
import { runIterationCheckpoint } from './phases/checkpoint.js'
|
|
@@ -181,6 +182,44 @@ export function settleGraceMs(remainingBeforeFinalizeMs: number): number {
|
|
|
181
182
|
)
|
|
182
183
|
}
|
|
183
184
|
|
|
185
|
+
/**
|
|
186
|
+
* The ceiling on the job half of that grace, in milliseconds.
|
|
187
|
+
*
|
|
188
|
+
* `DELEGATION_TIMEOUT_MS` is the wrong ceiling for a shell job, and the gap
|
|
189
|
+
* only opens where it matters most: a run with no `timeoutMs` — the CLI's
|
|
190
|
+
* shipping default, `No run deadline by default` — has infinite time before
|
|
191
|
+
* it must start finishing, so `settleGraceMs` returns the ceiling flat. For a
|
|
192
|
+
* delegated task that is sound, because the hour is the longest the task
|
|
193
|
+
* itself may live: the hold cannot outlast the work. A background job has no
|
|
194
|
+
* such bound. `tail -f`, a watcher and a dev server all outlive any hold, so
|
|
195
|
+
* the same arithmetic parks an interactive session for an hour on a job that
|
|
196
|
+
* was never going to exit.
|
|
197
|
+
*
|
|
198
|
+
* So the job leg gets its own bound, and it is sized to what the wait buys
|
|
199
|
+
* rather than to how long a job may live: a turn in which to use the exit.
|
|
200
|
+
* A model that already waited its `wait_for_job` bound out and saw nothing is
|
|
201
|
+
* not usually two minutes from an exit, and the run ending is not the news
|
|
202
|
+
* being lost — with no run in flight the session announces the exit itself
|
|
203
|
+
* (`docs/cli/background-jobs.md`, *Learning that it ended*), which is the
|
|
204
|
+
* cheaper of the two places to hear it.
|
|
205
|
+
*/
|
|
206
|
+
const DEFAULT_JOB_HOLD_MAX_MS = 2 * 60 * 1000
|
|
207
|
+
|
|
208
|
+
/**
|
|
209
|
+
* The same share of the run, under {@link DEFAULT_JOB_HOLD_MAX_MS}.
|
|
210
|
+
*
|
|
211
|
+
* `NAMZU_JOB_HOLD_MAX_MS` overrides the ceiling for a host that wants a
|
|
212
|
+
* longer or shorter park, the way `NAMZU_JOB_WAIT_TIMEOUT_MS` overrides
|
|
213
|
+
* `wait_for_job`'s own bound — and it is the same parse, so a value that is
|
|
214
|
+
* not a positive whole number of milliseconds leaves the default standing
|
|
215
|
+
* rather than holding a run for `NaN`. Called here rather than at module
|
|
216
|
+
* load, because a host that sets it after import is not ignored.
|
|
217
|
+
*/
|
|
218
|
+
export function awaitedJobGraceMs(remainingBeforeFinalizeMs: number): number {
|
|
219
|
+
const ceiling = readPositiveIntEnv('NAMZU_JOB_HOLD_MAX_MS', DEFAULT_JOB_HOLD_MAX_MS)
|
|
220
|
+
return Math.min(settleGraceMs(remainingBeforeFinalizeMs), ceiling)
|
|
221
|
+
}
|
|
222
|
+
|
|
184
223
|
export class IterationOrchestrator {
|
|
185
224
|
private ctx: IterationContext
|
|
186
225
|
private advisoryTurn:
|
|
@@ -1537,8 +1576,10 @@ export class IterationOrchestrator {
|
|
|
1537
1576
|
// returned — which is what makes a terminal submit_answer tool
|
|
1538
1577
|
// usable without discarding its output.
|
|
1539
1578
|
if (await this.shouldStop()) {
|
|
1540
|
-
// Outstanding
|
|
1541
|
-
//
|
|
1579
|
+
// Outstanding work outranks the host's stop predicate —
|
|
1580
|
+
// a delegated task the completion inbox is expecting, or
|
|
1581
|
+
// a background job the model told `wait_for_job` it is
|
|
1582
|
+
// waiting on.
|
|
1542
1583
|
//
|
|
1543
1584
|
// This is a precedence rule chosen here, not something
|
|
1544
1585
|
// `stopWhen` implies — a stop predicate is a programmable
|
|
@@ -1547,13 +1588,20 @@ export class IterationOrchestrator {
|
|
|
1547
1588
|
// tool or a captured structured output. Those decide the
|
|
1548
1589
|
// result, so no turn follows and a hold would buy nothing.
|
|
1549
1590
|
// This one only says "stop", and stopping one turn later
|
|
1550
|
-
// with the
|
|
1551
|
-
//
|
|
1591
|
+
// with the result in hand is a better reading of the
|
|
1592
|
+
// host's intent than stopping now and discarding it.
|
|
1552
1593
|
//
|
|
1553
|
-
// Bounded
|
|
1554
|
-
//
|
|
1555
|
-
//
|
|
1556
|
-
//
|
|
1594
|
+
// Bounded by what is left to deliver, not by a count.
|
|
1595
|
+
// Each delivery consumes what it delivered — the inbox is
|
|
1596
|
+
// drained, and a job exit's notice is taken with the
|
|
1597
|
+
// record of the exits it accounts for — so the predicate
|
|
1598
|
+
// is asked again next turn against whatever is still
|
|
1599
|
+
// outstanding. One task deferred it once; two awaited
|
|
1600
|
+
// jobs exiting a minute apart defer it twice, each time
|
|
1601
|
+
// for a turn the model spends on news it has not read.
|
|
1602
|
+
// `maxIterations` and the run's own deadline bound all of
|
|
1603
|
+
// it regardless, and a leg with nothing pending never
|
|
1604
|
+
// opens a hold at all.
|
|
1557
1605
|
if (yield* this.holdForOutstandingWork(iterationNum, true)) {
|
|
1558
1606
|
// Remember WHY the next turn exists, so the turn that
|
|
1559
1607
|
// ends the run can name the host's decision instead of
|
|
@@ -1782,30 +1830,63 @@ export class IterationOrchestrator {
|
|
|
1782
1830
|
}
|
|
1783
1831
|
|
|
1784
1832
|
/**
|
|
1785
|
-
* Hold the run open for
|
|
1833
|
+
* Hold the run open for work that has not finished, and deliver it.
|
|
1786
1834
|
*
|
|
1787
|
-
* Returns whether a completion or operator message entered
|
|
1788
|
-
* the caller continues on `true`, so the model gets a turn
|
|
1789
|
-
* That turn is the entire justification for waiting, which
|
|
1835
|
+
* Returns whether a completion, a job exit or an operator message entered
|
|
1836
|
+
* the transcript — the caller continues on `true`, so the model gets a turn
|
|
1837
|
+
* to respond. That turn is the entire justification for waiting, which
|
|
1790
1838
|
* is why only the exits that can still take one call this.
|
|
1791
1839
|
*
|
|
1792
|
-
*
|
|
1793
|
-
*
|
|
1840
|
+
* Two kinds of work qualify and they are raced together, because a run has
|
|
1841
|
+
* one settle point and one grace period to spend at it:
|
|
1842
|
+
*
|
|
1843
|
+
* - a delegated task the `CompletionInbox` is still expecting;
|
|
1844
|
+
* - a background job the model told `wait_for_job` it is waiting on.
|
|
1845
|
+
*
|
|
1846
|
+
* The job half is deliberately narrow. Intent comes from the wait and from
|
|
1847
|
+
* nothing else — a dev server the model started and never waited on is
|
|
1848
|
+
* running because somebody wanted it running, and a hold for it would add
|
|
1849
|
+
* the grace period to the end of every turn for the rest of the session.
|
|
1850
|
+
*
|
|
1851
|
+
* Each leg is opened only when it has something pending: both
|
|
1852
|
+
* `waitForArrival` implementations resolve immediately when their own side
|
|
1853
|
+
* is idle, so racing an idle one would end the hold before it began.
|
|
1854
|
+
*
|
|
1855
|
+
* Bounded by `settleGraceMs` and by `maxIterations`, so work that never
|
|
1856
|
+
* finishes cannot keep the run open. On a run with a deadline the grace is
|
|
1857
|
+
* a share of what is LEFT of it rather than a fresh allowance, so a
|
|
1858
|
+
* `wait_for_job` call that already spent minutes has shortened this hold
|
|
1859
|
+
* by the same minutes. On a run without one — the CLI's default — there is
|
|
1860
|
+
* no remainder to take a share of, and the job leg's own ceiling
|
|
1861
|
+
* (`awaitedJobGraceMs`) is what keeps a timed-out wait from being followed
|
|
1862
|
+
* by an hour of silence.
|
|
1794
1863
|
*/
|
|
1795
1864
|
private async *holdForOutstandingWork(
|
|
1796
1865
|
iterationNum: number,
|
|
1797
1866
|
hasToolCalls: boolean,
|
|
1798
1867
|
): AsyncGenerator<RunEvent, boolean> {
|
|
1799
|
-
|
|
1868
|
+
const inbox = this.ctx.completionInbox?.hasPendingWork ? this.ctx.completionInbox : undefined
|
|
1869
|
+
const jobs = this.ctx.awaitedJobs?.hasPendingWork ? this.ctx.awaitedJobs : undefined
|
|
1870
|
+
if (!inbox && !jobs) return false
|
|
1800
1871
|
|
|
1801
1872
|
// Read HERE rather than from `forceFinalize`, which was sampled at the
|
|
1802
1873
|
// top of the iteration: one that has since crossed the finalize point
|
|
1803
1874
|
// must not open a wait against a reserve it has already entered.
|
|
1804
|
-
const
|
|
1805
|
-
|
|
1875
|
+
const remainingMs = this.ctx.guard.remainingBeforeFinalizeMs()
|
|
1876
|
+
// One deadline for the race, and it is the LONGEST ceiling any pending
|
|
1877
|
+
// leg justifies. A leg resolving on its own timer ends the whole race,
|
|
1878
|
+
// so handing the job leg its shorter ceiling while a task was also
|
|
1879
|
+
// outstanding would cut the task's hold down to the job's — a run
|
|
1880
|
+
// walking away from a worker it had time for, because a job happened
|
|
1881
|
+
// to be running. A job therefore never shortens a wait, and it never
|
|
1882
|
+
// lengthens one either: where a task is outstanding too, that is how
|
|
1883
|
+
// long this run was waiting anyway.
|
|
1884
|
+
const graceMs = inbox ? settleGraceMs(remainingMs) : awaitedJobGraceMs(remainingMs)
|
|
1885
|
+
this.ctx.log.info('Holding the run open for outstanding work', {
|
|
1806
1886
|
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
1807
1887
|
[NAMZU.ITERATION]: iterationNum,
|
|
1808
1888
|
'namzu.runtime.grace_ms': graceMs,
|
|
1889
|
+
'namzu.runtime.awaited_jobs': jobs?.outstandingJobIds ?? [],
|
|
1809
1890
|
})
|
|
1810
1891
|
// User input releases this wait without cancelling any child. Both waits
|
|
1811
1892
|
// share a disposable signal so the losing arrival listener cannot leak.
|
|
@@ -1816,7 +1897,8 @@ export class IterationOrchestrator {
|
|
|
1816
1897
|
if (runSignal.aborted) cancelWait()
|
|
1817
1898
|
try {
|
|
1818
1899
|
await Promise.race([
|
|
1819
|
-
|
|
1900
|
+
...(inbox ? [inbox.waitForArrival(graceMs, waiting.signal)] : []),
|
|
1901
|
+
...(jobs ? [jobs.waitForArrival(graceMs, waiting.signal)] : []),
|
|
1820
1902
|
...(this.ctx.waitForInbound ? [this.ctx.waitForInbound(waiting.signal)] : []),
|
|
1821
1903
|
])
|
|
1822
1904
|
} catch (error) {
|
|
@@ -1827,14 +1909,15 @@ export class IterationOrchestrator {
|
|
|
1827
1909
|
}
|
|
1828
1910
|
runSignal.throwIfAborted()
|
|
1829
1911
|
|
|
1830
|
-
const arrived = this.ctx.completionInbox
|
|
1912
|
+
const arrived = this.ctx.completionInbox?.drain() ?? []
|
|
1831
1913
|
if (arrived.length > 0) {
|
|
1832
1914
|
this.ctx.runMgr.pushMessage(
|
|
1833
1915
|
createRuntimeContextMessage(formatCompletionNotification(arrived), 'task-completion'),
|
|
1834
1916
|
)
|
|
1835
1917
|
}
|
|
1918
|
+
const exited = this.deliverAwaitedJobExits()
|
|
1836
1919
|
const inbound = this.deliverInbound()
|
|
1837
|
-
if (arrived.length === 0 && inbound === 0) return false
|
|
1920
|
+
if (arrived.length === 0 && !exited && inbound === 0) return false
|
|
1838
1921
|
await this.ctx.emitEvent({
|
|
1839
1922
|
type: 'iteration_completed',
|
|
1840
1923
|
runId: this.ctx.runMgr.id,
|
|
@@ -1846,8 +1929,48 @@ export class IterationOrchestrator {
|
|
|
1846
1929
|
}
|
|
1847
1930
|
|
|
1848
1931
|
/**
|
|
1849
|
-
*
|
|
1850
|
-
*
|
|
1932
|
+
* Put the job exits this hold was waiting for in front of the model.
|
|
1933
|
+
*
|
|
1934
|
+
* Through `jobNotices`, which is the channel a job exit already travels on
|
|
1935
|
+
* — `attachNotice` rides it out on the next tool result — rather than a
|
|
1936
|
+
* second one built for this path. A turn that called no tools has no such
|
|
1937
|
+
* result, so the queued text becomes a `runtime-context` message instead,
|
|
1938
|
+
* exactly as `deliverInbound` does for steering that found no tool result
|
|
1939
|
+
* to attach to.
|
|
1940
|
+
*
|
|
1941
|
+
* That drain is also what keeps one exit from being delivered twice: the
|
|
1942
|
+
* channel hands its text over once, so an exit already attached to a tool
|
|
1943
|
+
* result earlier in the turn leaves nothing here — and the record of it
|
|
1944
|
+
* went with that delivery, so this returns `false` rather than buying a
|
|
1945
|
+
* turn to re-read what the model has read.
|
|
1946
|
+
*
|
|
1947
|
+
* `takeDelivery` is what pairs the two. Taking the exits first and then
|
|
1948
|
+
* finding no notice would discard them, which is the one way this path
|
|
1949
|
+
* can lose an exit outright; neither is taken unless both are there.
|
|
1950
|
+
*
|
|
1951
|
+
* The channel is not per-job, so the text taken here can include a notice
|
|
1952
|
+
* for a job nobody awaited that ended while the hold was open. Delivering
|
|
1953
|
+
* it is right — it is unread either way, and the alternative is stranding
|
|
1954
|
+
* it — but it is not a reason to WAIT, which is why what opens this hold
|
|
1955
|
+
* is `AwaitedJobs`, and the two are asked separately.
|
|
1956
|
+
*/
|
|
1957
|
+
private deliverAwaitedJobExits(): boolean {
|
|
1958
|
+
const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain())
|
|
1959
|
+
if (!delivered) return false
|
|
1960
|
+
|
|
1961
|
+
this.ctx.log.info('Delivering a background job exit the run held open for', {
|
|
1962
|
+
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
1963
|
+
'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
|
|
1964
|
+
})
|
|
1965
|
+
this.ctx.runMgr.pushMessage(
|
|
1966
|
+
createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'),
|
|
1967
|
+
)
|
|
1968
|
+
return true
|
|
1969
|
+
}
|
|
1970
|
+
|
|
1971
|
+
/**
|
|
1972
|
+
* Account for outstanding work on the way out: deliver what arrived, and
|
|
1973
|
+
* say what did not.
|
|
1851
1974
|
*
|
|
1852
1975
|
* A run that ends with a worker outstanding must not leave the impression
|
|
1853
1976
|
* that the worker's result was delivered. There are exactly two honest
|
|
@@ -1871,19 +1994,33 @@ export class IterationOrchestrator {
|
|
|
1871
1994
|
*/
|
|
1872
1995
|
private settleOutstandingWork(): void {
|
|
1873
1996
|
this.deliverArrivedCompletions()
|
|
1997
|
+
this.deliverArrivedJobExits()
|
|
1874
1998
|
this.recordAbandonedWork()
|
|
1875
1999
|
}
|
|
1876
2000
|
|
|
1877
|
-
/**
|
|
2001
|
+
/** Work this run walked away from. See {@link settleOutstandingWork}. */
|
|
1878
2002
|
private recordAbandonedWork(): void {
|
|
1879
2003
|
const abandoned = this.ctx.completionInbox?.outstandingTaskIds ?? []
|
|
1880
|
-
if (abandoned.length
|
|
2004
|
+
if (abandoned.length > 0) {
|
|
2005
|
+
this.ctx.log.warn('Run ended with delegated work still running', {
|
|
2006
|
+
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
2007
|
+
'namzu.runtime.tasks': abandoned,
|
|
2008
|
+
})
|
|
2009
|
+
this.ctx.runMgr.setAbandonedTaskIds(abandoned)
|
|
2010
|
+
}
|
|
2011
|
+
|
|
2012
|
+
// The same statement for a job the model was waiting on when the grace
|
|
2013
|
+
// ran out. Only awaited ones: a job nobody waited for was never work
|
|
2014
|
+
// this run was holding, so naming it would report an abandonment that
|
|
2015
|
+
// did not happen.
|
|
2016
|
+
const abandonedJobs = this.ctx.awaitedJobs?.outstandingJobIds ?? []
|
|
2017
|
+
if (abandonedJobs.length === 0) return
|
|
1881
2018
|
|
|
1882
|
-
this.ctx.log.warn('Run ended with
|
|
2019
|
+
this.ctx.log.warn('Run ended with an awaited background job still running', {
|
|
1883
2020
|
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
1884
|
-
'namzu.runtime.
|
|
2021
|
+
'namzu.runtime.jobs': abandonedJobs,
|
|
1885
2022
|
})
|
|
1886
|
-
this.ctx.runMgr.
|
|
2023
|
+
this.ctx.runMgr.setAbandonedJobIds(abandonedJobs)
|
|
1887
2024
|
}
|
|
1888
2025
|
|
|
1889
2026
|
private deliverArrivedCompletions(): void {
|
|
@@ -1918,6 +2055,43 @@ export class IterationOrchestrator {
|
|
|
1918
2055
|
)
|
|
1919
2056
|
}
|
|
1920
2057
|
|
|
2058
|
+
/**
|
|
2059
|
+
* The job half of {@link deliverArrivedCompletions}: an exit that arrived
|
|
2060
|
+
* too late to earn a turn is still delivered on the way out.
|
|
2061
|
+
*
|
|
2062
|
+
* The window this closes is one tick wide and it is nobody else's. An
|
|
2063
|
+
* awaited job that exits between the hold's grace expiring and the run
|
|
2064
|
+
* settling was never delivered — the hold had already looked — and is no
|
|
2065
|
+
* longer named either, because the exit took it off the outstanding list
|
|
2066
|
+
* on its way past, so `abandonedJobIds` would be lying to claim it. The
|
|
2067
|
+
* host's own listener is no help: the CLI queues an exit for the next
|
|
2068
|
+
* turn only when no run is in flight, and this one is still in flight.
|
|
2069
|
+
* Delivered here it reaches `Run.messages`, so the transcript has it and
|
|
2070
|
+
* a continued thread opens with it.
|
|
2071
|
+
*
|
|
2072
|
+
* Before `recordAbandonedWork`, which then reports only what is still
|
|
2073
|
+
* running, and after `deliverArrivedCompletions`, so the two appended
|
|
2074
|
+
* messages land in the order the work finished in.
|
|
2075
|
+
*/
|
|
2076
|
+
private deliverArrivedJobExits(): void {
|
|
2077
|
+
const delivered = this.ctx.awaitedJobs?.takeDelivery(() => this.ctx.jobNotices?.drain())
|
|
2078
|
+
if (!delivered) return
|
|
2079
|
+
|
|
2080
|
+
// Fix the run's answer BEFORE appending anything after it — the same
|
|
2081
|
+
// `resolveResult` tail walk `deliverArrivedCompletions` explains just
|
|
2082
|
+
// above, and the same guard against pinning an empty one.
|
|
2083
|
+
const answer = this.ctx.runMgr.materializeResult()
|
|
2084
|
+
if (answer.length > 0) this.ctx.runMgr.setResult(answer)
|
|
2085
|
+
|
|
2086
|
+
this.ctx.log.info('Delivering a background job exit the run would have settled over', {
|
|
2087
|
+
[NAMZU.RUN_ID]: this.ctx.runMgr.id,
|
|
2088
|
+
'namzu.runtime.jobs': delivered.exits.map((job) => job.id),
|
|
2089
|
+
})
|
|
2090
|
+
this.ctx.runMgr.pushMessage(
|
|
2091
|
+
createRuntimeContextMessage(formatJobNote(delivered.text), 'job-exit'),
|
|
2092
|
+
)
|
|
2093
|
+
}
|
|
2094
|
+
|
|
1921
2095
|
private stepContextMessage(content: string) {
|
|
1922
2096
|
return createRuntimeContextMessage(
|
|
1923
2097
|
`Current step context (runtime-generated; not a new user request):\n${content}`,
|
|
@@ -35,6 +35,7 @@ import type { StructuredOutputConfig } from '../../../../types/structured-output
|
|
|
35
35
|
import type { TaskStore } from '../../../../types/task/index.js'
|
|
36
36
|
import type { ToolRegistryContract } from '../../../../types/tool/index.js'
|
|
37
37
|
import type { Logger } from '../../../../utils/logger.js'
|
|
38
|
+
import type { AwaitedJobs } from '../../../jobs/awaited-jobs.js'
|
|
38
39
|
import type { CheckpointManager } from '../../checkpoint.js'
|
|
39
40
|
import type { EmitEvent } from '../../events.js'
|
|
40
41
|
import type { ToolExecutor } from '../../executor.js'
|
|
@@ -136,6 +137,15 @@ export interface IterationContext {
|
|
|
136
137
|
readonly onSteeringDelivered?: (text: string) => void
|
|
137
138
|
/** Exit notices for the run's background jobs, drained into the next tool result. */
|
|
138
139
|
readonly jobNotices?: SteeringChannel
|
|
140
|
+
/**
|
|
141
|
+
* Background jobs the model said it is waiting on, which is the only kind
|
|
142
|
+
* the loop holds a finishing run open for.
|
|
143
|
+
*
|
|
144
|
+
* Absent means the loop behaves exactly as it did before this existed: a
|
|
145
|
+
* job's exit still reaches the model as a notice on the next tool result,
|
|
146
|
+
* and a run whose model stopped calling tools settles without waiting.
|
|
147
|
+
*/
|
|
148
|
+
readonly awaitedJobs?: AwaitedJobs
|
|
139
149
|
readonly checkpointMgr: CheckpointManager
|
|
140
150
|
readonly planManager: PlanManager
|
|
141
151
|
|
|
@@ -150,6 +150,10 @@ export async function* runToolReview(
|
|
|
150
150
|
attachSteering(batch.messages, ctx.steering, ctx.onSteeringDelivered),
|
|
151
151
|
ctx.jobNotices,
|
|
152
152
|
formatJobNote,
|
|
153
|
+
// The exits that text accounts for have now been read, so the
|
|
154
|
+
// record of them stops being pending work. Left standing, it
|
|
155
|
+
// buys the model a turn the next time any job queues a notice.
|
|
156
|
+
() => ctx.awaitedJobs?.noticesDelivered(),
|
|
153
157
|
),
|
|
154
158
|
notices,
|
|
155
159
|
)) {
|