@namzu/sdk 42.0.0 → 42.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +174 -0
- package/dist/manager/resident/outbox.d.ts +8 -8
- package/dist/manager/resident/store.d.ts +4 -4
- package/dist/runtime/query/cancelled-before-start.d.ts +34 -0
- package/dist/runtime/query/cancelled-before-start.d.ts.map +1 -0
- package/dist/runtime/query/cancelled-before-start.js +152 -0
- package/dist/runtime/query/cancelled-before-start.js.map +1 -0
- package/dist/runtime/query/checkpoint.d.ts +21 -0
- package/dist/runtime/query/checkpoint.d.ts.map +1 -1
- package/dist/runtime/query/checkpoint.js +23 -0
- package/dist/runtime/query/checkpoint.js.map +1 -1
- package/dist/runtime/query/executor/tool-call-admission.d.ts +57 -0
- package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -0
- package/dist/runtime/query/executor/tool-call-admission.js +373 -0
- package/dist/runtime/query/executor/tool-call-admission.js.map +1 -0
- package/dist/runtime/query/executor.d.ts +70 -35
- package/dist/runtime/query/executor.d.ts.map +1 -1
- package/dist/runtime/query/executor.js +46 -380
- package/dist/runtime/query/executor.js.map +1 -1
- package/dist/runtime/query/finalize-run.d.ts +55 -0
- package/dist/runtime/query/finalize-run.d.ts.map +1 -0
- package/dist/runtime/query/finalize-run.js +113 -0
- package/dist/runtime/query/finalize-run.js.map +1 -0
- package/dist/runtime/query/index.d.ts +4 -9
- package/dist/runtime/query/index.d.ts.map +1 -1
- package/dist/runtime/query/index.js +238 -893
- package/dist/runtime/query/index.js.map +1 -1
- package/dist/runtime/query/iteration/index.d.ts +6 -161
- package/dist/runtime/query/iteration/index.d.ts.map +1 -1
- package/dist/runtime/query/iteration/index.js +23 -523
- package/dist/runtime/query/iteration/index.js.map +1 -1
- package/dist/runtime/query/iteration/outstanding-work.d.ts +158 -0
- package/dist/runtime/query/iteration/outstanding-work.d.ts.map +1 -0
- package/dist/runtime/query/iteration/outstanding-work.js +365 -0
- package/dist/runtime/query/iteration/outstanding-work.js.map +1 -0
- package/dist/runtime/query/iteration/phases/plan.d.ts.map +1 -1
- package/dist/runtime/query/iteration/phases/plan.js +13 -2
- package/dist/runtime/query/iteration/phases/plan.js.map +1 -1
- package/dist/runtime/query/iteration/step-shaping.d.ts +41 -0
- package/dist/runtime/query/iteration/step-shaping.d.ts.map +1 -0
- package/dist/runtime/query/iteration/step-shaping.js +184 -0
- package/dist/runtime/query/iteration/step-shaping.js.map +1 -0
- package/dist/runtime/query/prepare-run.d.ts +94 -0
- package/dist/runtime/query/prepare-run.d.ts.map +1 -0
- package/dist/runtime/query/prepare-run.js +589 -0
- package/dist/runtime/query/prepare-run.js.map +1 -0
- package/dist/runtime/query/release-run.d.ts +56 -0
- package/dist/runtime/query/release-run.d.ts.map +1 -0
- package/dist/runtime/query/release-run.js +101 -0
- package/dist/runtime/query/release-run.js.map +1 -0
- package/dist/runtime/query/resume-pending.d.ts +112 -1
- package/dist/runtime/query/resume-pending.d.ts.map +1 -1
- package/dist/runtime/query/resume-pending.js +133 -0
- package/dist/runtime/query/resume-pending.js.map +1 -1
- package/dist/store/evidence/compaction-archive.d.ts +2 -2
- package/dist/types/run/config.d.ts +12 -5
- package/dist/types/run/config.d.ts.map +1 -1
- package/package.json +1 -1
- package/src/runtime/query/cancelled-before-start.ts +189 -0
- package/src/runtime/query/checkpoint.ts +22 -0
- package/src/runtime/query/executor/tool-call-admission.ts +473 -0
- package/src/runtime/query/executor.ts +63 -442
- package/src/runtime/query/finalize-run.ts +192 -0
- package/src/runtime/query/index.ts +270 -1011
- package/src/runtime/query/iteration/index.ts +40 -586
- package/src/runtime/query/iteration/outstanding-work.ts +386 -0
- package/src/runtime/query/iteration/phases/plan.ts +18 -2
- package/src/runtime/query/iteration/step-shaping.ts +271 -0
- package/src/runtime/query/prepare-run.ts +718 -0
- package/src/runtime/query/release-run.ts +168 -0
- package/src/runtime/query/resume-pending.ts +158 -0
- package/src/types/run/config.ts +12 -5
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
import type { RunPersistence } from '../../manager/run/persistence.js'
|
|
2
|
+
import { GENAI, NAMZU, agentRunSpanName, parentContext } from '../../telemetry/attributes.js'
|
|
3
|
+
import { getTracer } from '../../telemetry/runtime-accessors.js'
|
|
4
|
+
import { createSystemMessage } from '../../types/message/index.js'
|
|
5
|
+
import type { FencingToken } from '../../types/run/checkpoint-store.js'
|
|
6
|
+
import type { RunEventCursor, RunEventReplay } from '../../types/run/event-cursor.js'
|
|
7
|
+
import { resolveRunEventReplay } from '../../types/run/event-cursor.js'
|
|
8
|
+
import type { Run, RunEvent } from '../../types/run/index.js'
|
|
9
|
+
import { toErrorMessage } from '../../utils/error.js'
|
|
10
|
+
import type { QueryParams } from './index.js'
|
|
11
|
+
import type { PreparedRun } from './prepare-run.js'
|
|
12
|
+
import { ResultAssembler } from './result.js'
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* The run that was cancelled before it started.
|
|
16
|
+
*
|
|
17
|
+
* Attachment materialization happens before `RunContext` exists, and a
|
|
18
|
+
* cancellation observed there still belongs to a run: it must be recorded,
|
|
19
|
+
* classified and reported like any other, without any of the authority-bearing
|
|
20
|
+
* work — prompt contributions, host callbacks, tools, plugins, sandbox,
|
|
21
|
+
* guardrails, advisors, providers — that a run which is allowed to proceed
|
|
22
|
+
* would do next.
|
|
23
|
+
*
|
|
24
|
+
* An `async function*` rather than a plain async function, and reached with
|
|
25
|
+
* `yield*`, because this path still emits and drains: the events it produces
|
|
26
|
+
* must occupy exactly the positions in the stream they occupied when the code
|
|
27
|
+
* lived inline, and only delegation preserves every yield point.
|
|
28
|
+
*/
|
|
29
|
+
export async function* settlePreStartCancellation(
|
|
30
|
+
params: QueryParams,
|
|
31
|
+
prepared: PreparedRun,
|
|
32
|
+
): AsyncGenerator<RunEvent, Run> {
|
|
33
|
+
const {
|
|
34
|
+
ctx,
|
|
35
|
+
runConfig,
|
|
36
|
+
eventTranslator,
|
|
37
|
+
executeUserInterruptHooks,
|
|
38
|
+
selectedResumeState,
|
|
39
|
+
queuedForThisRun,
|
|
40
|
+
initialMessages,
|
|
41
|
+
} = prepared
|
|
42
|
+
|
|
43
|
+
// Attachment materialization happens before RunContext exists. Once it
|
|
44
|
+
// observes cancellation, do only the work required to leave an honest
|
|
45
|
+
// durable run: initialize the record, retain the unresolved references,
|
|
46
|
+
// and settle through the ordinary cancellation classifier. Prompt
|
|
47
|
+
// contributions/cache, host callbacks, tools, plugins, sandbox, guardrails,
|
|
48
|
+
// advisors, and providers are all authority-bearing work and stay out.
|
|
49
|
+
// The dedicated root interrupt notification is the sole plugin exception:
|
|
50
|
+
// it runs after cancellation under its own deadline and cannot regain model
|
|
51
|
+
// or tool authority.
|
|
52
|
+
if (params.resumeFromCheckpoint && !selectedResumeState) {
|
|
53
|
+
// The canonical resume surface hands query the checkpoint state it
|
|
54
|
+
// already selected. A raw resume query has no such snapshot; after
|
|
55
|
+
// cancellation, reading the store again could hang without a signal,
|
|
56
|
+
// while persisting without it would erase the existing transcript.
|
|
57
|
+
// Refuse before binding/persisting rather than choose either failure.
|
|
58
|
+
ctx.abortController.signal.throwIfAborted()
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const cancelledPrompt = params.systemPrompt ?? ''
|
|
62
|
+
const cancelledAssembler = new ResultAssembler({
|
|
63
|
+
runMgr: ctx.runMgr,
|
|
64
|
+
planManager: ctx.planManager,
|
|
65
|
+
activityStore: ctx.activityStore,
|
|
66
|
+
log: ctx.log,
|
|
67
|
+
emitEvent: eventTranslator.emitEvent,
|
|
68
|
+
drainPending: () => eventTranslator.drainPending(),
|
|
69
|
+
signal: ctx.abortController.signal,
|
|
70
|
+
})
|
|
71
|
+
const rootSpan = getTracer().startSpan(
|
|
72
|
+
agentRunSpanName(params.agentName),
|
|
73
|
+
{},
|
|
74
|
+
parentContext(params.parentSpan ?? selectedResumeState?.traceContext),
|
|
75
|
+
)
|
|
76
|
+
rootSpan.setAttributes({
|
|
77
|
+
[NAMZU.RUN_ID]: ctx.runMgr.id,
|
|
78
|
+
[GENAI.AGENT_NAME]: params.agentName,
|
|
79
|
+
[GENAI.AGENT_ID]: params.agentId,
|
|
80
|
+
[GENAI.REQUEST_MODEL]: runConfig.model,
|
|
81
|
+
[GENAI.SYSTEM]: params.provider.id,
|
|
82
|
+
})
|
|
83
|
+
|
|
84
|
+
try {
|
|
85
|
+
await ctx.runMgr.init()
|
|
86
|
+
if (selectedResumeState) {
|
|
87
|
+
ctx.runMgr.restoreUsage(
|
|
88
|
+
selectedResumeState.tokenUsage,
|
|
89
|
+
selectedResumeState.costInfo,
|
|
90
|
+
selectedResumeState.currentIteration,
|
|
91
|
+
)
|
|
92
|
+
for (const message of selectedResumeState.messages) ctx.runMgr.pushMessage(message)
|
|
93
|
+
for (const queued of queuedForThisRun) ctx.runMgr.pushMessage(queued)
|
|
94
|
+
} else if (params.continuationMode) {
|
|
95
|
+
for (const message of initialMessages) ctx.runMgr.pushMessage(message)
|
|
96
|
+
} else {
|
|
97
|
+
ctx.runMgr.pushMessage(createSystemMessage(cancelledPrompt, 'cache'))
|
|
98
|
+
for (const message of initialMessages) ctx.runMgr.pushMessage(message)
|
|
99
|
+
}
|
|
100
|
+
if (params.eventCursor) {
|
|
101
|
+
yield* catchUpFromCursor(
|
|
102
|
+
ctx.runMgr,
|
|
103
|
+
params.eventCursor,
|
|
104
|
+
params.onEventReplay,
|
|
105
|
+
params.claimFence,
|
|
106
|
+
(error) => {
|
|
107
|
+
ctx.log.warn('Replay observer failed after attachment cancellation', {
|
|
108
|
+
'exception.message': toErrorMessage(error),
|
|
109
|
+
})
|
|
110
|
+
},
|
|
111
|
+
)
|
|
112
|
+
}
|
|
113
|
+
if (selectedResumeState) {
|
|
114
|
+
await eventTranslator.emitEvent({
|
|
115
|
+
type: 'run_resuming',
|
|
116
|
+
runId: ctx.runId,
|
|
117
|
+
fromCheckpointId: selectedResumeState.checkpointId,
|
|
118
|
+
})
|
|
119
|
+
yield* eventTranslator.drainPending()
|
|
120
|
+
}
|
|
121
|
+
ctx.runMgr.markRunning()
|
|
122
|
+
await eventTranslator.emitEvent({
|
|
123
|
+
type: 'run_started',
|
|
124
|
+
runId: ctx.runId,
|
|
125
|
+
systemPrompt: cancelledPrompt,
|
|
126
|
+
})
|
|
127
|
+
yield* eventTranslator.drainPending()
|
|
128
|
+
ctx.abortController.signal.throwIfAborted()
|
|
129
|
+
} catch (error) {
|
|
130
|
+
// Attachment resolution has already observed the caller's abort. A
|
|
131
|
+
// reconnect callback can still throw while replay is being reported,
|
|
132
|
+
// but it cannot replace that terminal cause or turn a cancelled run
|
|
133
|
+
// into an unpersisted rejection.
|
|
134
|
+
const terminalError = ctx.abortController.signal.aborted
|
|
135
|
+
? ctx.abortController.signal.reason
|
|
136
|
+
: error
|
|
137
|
+
await executeUserInterruptHooks(terminalError)
|
|
138
|
+
yield* eventTranslator.drainPending()
|
|
139
|
+
yield* cancelledAssembler.handleError(terminalError, rootSpan)
|
|
140
|
+
} finally {
|
|
141
|
+
rootSpan.end()
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
return await cancelledAssembler.finalize()
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* Hand a returning consumer what it missed, or tell it why it cannot have it.
|
|
149
|
+
*
|
|
150
|
+
* Yields NOTHING on a refusal. A partial catch-up is the failure this exists to
|
|
151
|
+
* prevent: a consumer that receives some of the gap folds it into its state and
|
|
152
|
+
* cannot tell the state is wrong, where one that receives an explicit
|
|
153
|
+
* `unavailable` re-derives from the transcript and is right. The run continues
|
|
154
|
+
* either way — a stale cursor belongs to the client, and must not be able to
|
|
155
|
+
* stop the work.
|
|
156
|
+
*/
|
|
157
|
+
export async function* catchUpFromCursor(
|
|
158
|
+
runMgr: RunPersistence,
|
|
159
|
+
cursor: RunEventCursor,
|
|
160
|
+
onEventReplay: ((replay: RunEventReplay) => void) | undefined,
|
|
161
|
+
generation: FencingToken | undefined,
|
|
162
|
+
onReplayObserverError: (error: unknown) => void,
|
|
163
|
+
): AsyncGenerator<RunEvent, void> {
|
|
164
|
+
const missed = await runMgr.getRunStore().readEvents({ sinceSeq: cursor.sinceSeq })
|
|
165
|
+
const replay = resolveRunEventReplay(
|
|
166
|
+
cursor,
|
|
167
|
+
{
|
|
168
|
+
lastSeq: runMgr.lastEventSeq,
|
|
169
|
+
...(generation !== undefined ? { generation } : {}),
|
|
170
|
+
},
|
|
171
|
+
missed,
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
if (onEventReplay) {
|
|
175
|
+
try {
|
|
176
|
+
// A callback typed `void` may still be implemented with `async` in
|
|
177
|
+
// TypeScript. Observe that runtime Promise so a late rejection cannot
|
|
178
|
+
// become process-wide, but never await host code here: replay delivery
|
|
179
|
+
// and an already-cancelled run must not inherit observer liveness.
|
|
180
|
+
const settlement = onEventReplay(replay)
|
|
181
|
+
void Promise.resolve(settlement).catch(onReplayObserverError)
|
|
182
|
+
} catch (error) {
|
|
183
|
+
onReplayObserverError(error)
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
if (replay.status !== 'replayed') return
|
|
188
|
+
for (const event of replay.events) yield event
|
|
189
|
+
}
|
|
@@ -584,6 +584,27 @@ export class CheckpointManager {
|
|
|
584
584
|
return checkpoints.map(toCheckpointListEntry)
|
|
585
585
|
}
|
|
586
586
|
|
|
587
|
+
/**
|
|
588
|
+
* Collect old checkpoints until `keepLast` newer ones remain.
|
|
589
|
+
*
|
|
590
|
+
* Growth control, and growth control stops at a park. A checkpoint with
|
|
591
|
+
* an unresolved `pending` is the durable fact that a human was asked
|
|
592
|
+
* something: it is what `findPendingCheckpoint` serves to an approval
|
|
593
|
+
* queue and what `listExpiredParks` enumerates for a sweep. Collecting
|
|
594
|
+
* one deletes the only record of a question somebody may be in the middle
|
|
595
|
+
* of answering, and their answer then lands nowhere — the store refuses
|
|
596
|
+
* the unpark and the run it belonged to resumes without it.
|
|
597
|
+
*
|
|
598
|
+
* So the candidates are still the oldest `all.length - keepLast`, which
|
|
599
|
+
* is what keeps the newest `keepLast` — the run's resume point — out of
|
|
600
|
+
* reach, and a park among them is skipped rather than counted. Pruning
|
|
601
|
+
* therefore holds a few more rows while a park is outstanding. That is
|
|
602
|
+
* the right side to err on: parks resolve, and the next prune collects
|
|
603
|
+
* them. Expired ones are skipped too — the host's sweep is
|
|
604
|
+
* {@link expire}, which resolves the park by running out of time rather
|
|
605
|
+
* than by deleting the evidence, and `prune` racing it would take the
|
|
606
|
+
* question away before it was read.
|
|
607
|
+
*/
|
|
587
608
|
async prune(keepLast: number): Promise<void> {
|
|
588
609
|
const all = await this.list()
|
|
589
610
|
if (all.length <= keepLast) return
|
|
@@ -591,6 +612,7 @@ export class CheckpointManager {
|
|
|
591
612
|
const toDelete = all.sort((a, b) => a.createdAt - b.createdAt).slice(0, all.length - keepLast)
|
|
592
613
|
|
|
593
614
|
for (const cp of toDelete) {
|
|
615
|
+
if (cp.pending && cp.pending.resolvedAt === undefined) continue
|
|
594
616
|
await this.store.deleteCheckpoint(this.scope, cp.id)
|
|
595
617
|
}
|
|
596
618
|
}
|
|
@@ -0,0 +1,473 @@
|
|
|
1
|
+
import { GENAI, NAMZU } from '../../../constants/telemetry/index.js'
|
|
2
|
+
import { renderToolSchema } from '../../../registry/tool/schema.js'
|
|
3
|
+
import type { ToolCall } from '../../../types/message/index.js'
|
|
4
|
+
import type { PluginHookResult } from '../../../types/plugin/index.js'
|
|
5
|
+
import type { ToolCallRepair, ToolCallRepairReason } from '../../../types/tool/repair.js'
|
|
6
|
+
import { toErrorMessage } from '../../../utils/error.js'
|
|
7
|
+
import type { Logger } from '../../../utils/logger.js'
|
|
8
|
+
import type {
|
|
9
|
+
EmitEvent,
|
|
10
|
+
PreToolHookOutcome,
|
|
11
|
+
PreparedDirectCall,
|
|
12
|
+
ToolExecutorConfig,
|
|
13
|
+
} from '../executor.js'
|
|
14
|
+
import { skippedToolResultText } from '../plugin-hooks.js'
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* One provider tool call, from raw to admitted.
|
|
18
|
+
*
|
|
19
|
+
* The executor's job is to run tools; this is the gate every call passes
|
|
20
|
+
* before one runs. A call arrives as the model streamed it — possibly not
|
|
21
|
+
* valid JSON, possibly cut off mid-argument, possibly naming a tool that does
|
|
22
|
+
* not exist — and leaves as a prepared, plugin-approved, authorized call, or
|
|
23
|
+
* as a synthetic failure that answers the model with what was wrong. Nothing
|
|
24
|
+
* here dispatches a tool, and nothing here writes a result: that is the
|
|
25
|
+
* executor's half.
|
|
26
|
+
*
|
|
27
|
+
* The family reads exactly three things off the executor it serves, and they
|
|
28
|
+
* arrive as one value rather than as a captured reference: the tool registry
|
|
29
|
+
* config, the event sink and the logger. `config` in particular is read
|
|
30
|
+
* per call and never held — `ToolExecutor.setSandbox` REPLACES it, so a host
|
|
31
|
+
* captured once would hand the next admission a stale sandbox.
|
|
32
|
+
*/
|
|
33
|
+
export interface ToolAdmissionHost {
|
|
34
|
+
readonly config: ToolExecutorConfig
|
|
35
|
+
readonly emitEvent: EmitEvent
|
|
36
|
+
readonly log: Logger
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export async function runPreToolHook(
|
|
40
|
+
host: ToolAdmissionHost,
|
|
41
|
+
toolName: string,
|
|
42
|
+
input: unknown,
|
|
43
|
+
signal: AbortSignal = host.config.abortSignal,
|
|
44
|
+
): Promise<PreToolHookOutcome> {
|
|
45
|
+
if (!host.config.pluginManager) return { kind: 'continue', input, modified: false }
|
|
46
|
+
const results = await host.config.pluginManager.executeHooks(
|
|
47
|
+
'pre_tool_use',
|
|
48
|
+
{
|
|
49
|
+
runId: host.config.runId,
|
|
50
|
+
toolName,
|
|
51
|
+
toolInput: input,
|
|
52
|
+
signal,
|
|
53
|
+
},
|
|
54
|
+
host.emitEvent,
|
|
55
|
+
)
|
|
56
|
+
return interpretPreToolResults(toolName, input, results)
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export async function prepareDirectCall(
|
|
60
|
+
host: ToolAdmissionHost,
|
|
61
|
+
toolCall: ToolCall,
|
|
62
|
+
): Promise<PreparedDirectCall> {
|
|
63
|
+
let toolName = toolCall.function.name
|
|
64
|
+
const truncationRepair =
|
|
65
|
+
toolCall.metadata?.inputTruncated === true
|
|
66
|
+
? await repairTruncatedCall(host, toolCall, toolName)
|
|
67
|
+
: null
|
|
68
|
+
if (toolCall.metadata?.inputTruncated === true && !truncationRepair) {
|
|
69
|
+
return {
|
|
70
|
+
kind: 'synthetic',
|
|
71
|
+
toolCall,
|
|
72
|
+
toolName,
|
|
73
|
+
input: {},
|
|
74
|
+
message: truncatedToolInputMessage(toolName),
|
|
75
|
+
isError: true,
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const prepare = host.config.tools.prepareExecution
|
|
80
|
+
const executePrepared = host.config.tools.executePrepared
|
|
81
|
+
if (typeof prepare !== 'function' || typeof executePrepared !== 'function') {
|
|
82
|
+
const resolved = await resolveCall(
|
|
83
|
+
host,
|
|
84
|
+
truncationRepair
|
|
85
|
+
? {
|
|
86
|
+
...toolCall,
|
|
87
|
+
function: {
|
|
88
|
+
...toolCall.function,
|
|
89
|
+
name: truncationRepair.toolName ?? toolName,
|
|
90
|
+
arguments: truncationRepair.arguments,
|
|
91
|
+
},
|
|
92
|
+
metadata: {},
|
|
93
|
+
}
|
|
94
|
+
: toolCall,
|
|
95
|
+
)
|
|
96
|
+
toolName = resolved.toolName
|
|
97
|
+
if (!resolved.ok) {
|
|
98
|
+
return {
|
|
99
|
+
kind: 'synthetic',
|
|
100
|
+
toolCall,
|
|
101
|
+
toolName,
|
|
102
|
+
input: {},
|
|
103
|
+
message: resolved.message,
|
|
104
|
+
isError: true,
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
const preOutcome = await runPreToolHook(host, toolName, resolved.input)
|
|
108
|
+
if (preOutcome.kind === 'skip' || preOutcome.kind === 'error') {
|
|
109
|
+
return {
|
|
110
|
+
kind: 'synthetic',
|
|
111
|
+
toolCall,
|
|
112
|
+
toolName,
|
|
113
|
+
input: preOutcome.input,
|
|
114
|
+
message: preOutcome.output,
|
|
115
|
+
isError: preOutcome.kind === 'error',
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
if (!host.config.authorizationGate) {
|
|
119
|
+
return {
|
|
120
|
+
kind: 'legacy',
|
|
121
|
+
toolCall,
|
|
122
|
+
toolName,
|
|
123
|
+
input: preOutcome.input,
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
return {
|
|
127
|
+
kind: 'synthetic',
|
|
128
|
+
toolCall,
|
|
129
|
+
toolName,
|
|
130
|
+
input: preOutcome.input,
|
|
131
|
+
message: `Tool "${toolName}" was not executed because its registry cannot bind authorization to one prepared input.`,
|
|
132
|
+
isError: true,
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
let raw = truncationRepair?.arguments ?? toolCall.function.arguments
|
|
137
|
+
toolName = truncationRepair?.toolName ?? toolName
|
|
138
|
+
let repairUsed = truncationRepair !== null
|
|
139
|
+
let preparation: ReturnType<typeof prepare>
|
|
140
|
+
for (;;) {
|
|
141
|
+
let parsed: unknown
|
|
142
|
+
try {
|
|
143
|
+
parsed = parseArguments(raw)
|
|
144
|
+
} catch {
|
|
145
|
+
const message = `Error: Invalid JSON in tool arguments for "${toolName}"`
|
|
146
|
+
const repair =
|
|
147
|
+
!repairUsed && host.config.repairToolCall
|
|
148
|
+
? await requestRepair(host, toolCall, toolName, {
|
|
149
|
+
reason: 'invalid_json',
|
|
150
|
+
message,
|
|
151
|
+
})
|
|
152
|
+
: null
|
|
153
|
+
if (repair) {
|
|
154
|
+
repairUsed = true
|
|
155
|
+
toolName = repair.toolName ?? toolName
|
|
156
|
+
raw = repair.arguments
|
|
157
|
+
continue
|
|
158
|
+
}
|
|
159
|
+
return { kind: 'synthetic', toolCall, toolName, input: {}, message, isError: true }
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
try {
|
|
163
|
+
preparation = prepare.call(host.config.tools, toolName, parsed)
|
|
164
|
+
} catch (err) {
|
|
165
|
+
const message = `Error: Unknown or unavailable tool "${toolName}": ${toErrorMessage(err)}`
|
|
166
|
+
const repair =
|
|
167
|
+
!repairUsed && host.config.repairToolCall
|
|
168
|
+
? await requestRepair(host, toolCall, toolName, {
|
|
169
|
+
reason: 'unknown_tool',
|
|
170
|
+
message,
|
|
171
|
+
})
|
|
172
|
+
: null
|
|
173
|
+
if (repair) {
|
|
174
|
+
repairUsed = true
|
|
175
|
+
toolName = repair.toolName ?? toolName
|
|
176
|
+
raw = repair.arguments
|
|
177
|
+
continue
|
|
178
|
+
}
|
|
179
|
+
return { kind: 'synthetic', toolCall, toolName, input: parsed, message, isError: true }
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
if (preparation.success) break
|
|
183
|
+
const message = formatFailedToolOutput(preparation.result.output, preparation.result.error)
|
|
184
|
+
const repair =
|
|
185
|
+
!repairUsed && host.config.repairToolCall
|
|
186
|
+
? await requestRepair(host, toolCall, toolName, {
|
|
187
|
+
reason: 'schema_validation',
|
|
188
|
+
message,
|
|
189
|
+
})
|
|
190
|
+
: null
|
|
191
|
+
if (repair) {
|
|
192
|
+
repairUsed = true
|
|
193
|
+
toolName = repair.toolName ?? toolName
|
|
194
|
+
raw = repair.arguments
|
|
195
|
+
continue
|
|
196
|
+
}
|
|
197
|
+
return {
|
|
198
|
+
kind: 'synthetic',
|
|
199
|
+
toolCall,
|
|
200
|
+
toolName,
|
|
201
|
+
input: parsed,
|
|
202
|
+
message,
|
|
203
|
+
isError: true,
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
const preOutcome = await runPreToolHook(host, toolName, preparation.prepared.input)
|
|
208
|
+
if (preOutcome.kind === 'skip' || preOutcome.kind === 'error') {
|
|
209
|
+
return {
|
|
210
|
+
kind: 'synthetic',
|
|
211
|
+
toolCall,
|
|
212
|
+
toolName,
|
|
213
|
+
input: preOutcome.input,
|
|
214
|
+
message: preOutcome.output,
|
|
215
|
+
isError: preOutcome.kind === 'error',
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
if (preOutcome.modified) {
|
|
220
|
+
const modified = prepare.call(host.config.tools, toolName, preOutcome.input)
|
|
221
|
+
if (!modified.success) {
|
|
222
|
+
return {
|
|
223
|
+
kind: 'synthetic',
|
|
224
|
+
toolCall,
|
|
225
|
+
toolName,
|
|
226
|
+
input: preOutcome.input,
|
|
227
|
+
message: formatFailedToolOutput(modified.result.output, modified.result.error),
|
|
228
|
+
isError: true,
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
preparation = modified
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
return {
|
|
235
|
+
kind: 'ready',
|
|
236
|
+
toolCall,
|
|
237
|
+
toolName,
|
|
238
|
+
input: preparation.prepared.input,
|
|
239
|
+
prepared: preparation.prepared,
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
function interpretPreToolResults(
|
|
244
|
+
toolName: string,
|
|
245
|
+
initialInput: unknown,
|
|
246
|
+
results: readonly PluginHookResult[],
|
|
247
|
+
): PreToolHookOutcome {
|
|
248
|
+
let currentInput = initialInput
|
|
249
|
+
let modified = false
|
|
250
|
+
for (const result of results) {
|
|
251
|
+
switch (result.action) {
|
|
252
|
+
case 'continue':
|
|
253
|
+
continue
|
|
254
|
+
case 'modify':
|
|
255
|
+
currentInput = result.input
|
|
256
|
+
modified = true
|
|
257
|
+
continue
|
|
258
|
+
case 'skip':
|
|
259
|
+
return {
|
|
260
|
+
kind: 'skip',
|
|
261
|
+
input: currentInput,
|
|
262
|
+
output: skippedToolResultText(toolName, result.reason),
|
|
263
|
+
}
|
|
264
|
+
case 'error':
|
|
265
|
+
return {
|
|
266
|
+
kind: 'error',
|
|
267
|
+
input: currentInput,
|
|
268
|
+
output: `Error: ${result.message}`,
|
|
269
|
+
}
|
|
270
|
+
case 'retry':
|
|
271
|
+
case 'annotate':
|
|
272
|
+
// There is no result to replace yet. Rejecting loudly beats
|
|
273
|
+
// silently ignoring it: a hook author who returned this here
|
|
274
|
+
// meant to redact something and would otherwise watch the secret
|
|
275
|
+
// go through.
|
|
276
|
+
case 'replace':
|
|
277
|
+
throw new Error(
|
|
278
|
+
`Plugin hook pre_tool_use returned unsupported action '${result.action}' for tool ${toolName}`,
|
|
279
|
+
)
|
|
280
|
+
default: {
|
|
281
|
+
const _exhaustive: never = result
|
|
282
|
+
throw new Error(`Unknown PluginHookResult: ${JSON.stringify(_exhaustive)}`)
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
return { kind: 'continue', input: currentInput, modified }
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
/**
|
|
290
|
+
* Turn the call the model issued into a name and a parsed input, giving
|
|
291
|
+
* a configured repairer one chance to fix it first.
|
|
292
|
+
*
|
|
293
|
+
* Exactly one chance: a repairer that produces a call which is still
|
|
294
|
+
* broken will not do better on a second look, and an unbounded loop
|
|
295
|
+
* here is a hang rather than a degradation.
|
|
296
|
+
*
|
|
297
|
+
* `invalid_json` is the ONLY failure that stops the call here, and it
|
|
298
|
+
* stopped it before this function existed too. `unknown_tool` and
|
|
299
|
+
* `schema_validation` merely OFFER the repair and otherwise fall
|
|
300
|
+
* through to the registry, which reports both with better messages —
|
|
301
|
+
* its schema error already ships a "Required: <field>: <type>" hint the
|
|
302
|
+
* model can self-correct from. So with no repairer configured this is
|
|
303
|
+
* behaviorally identical to the bare `JSON.parse` it replaced.
|
|
304
|
+
*/
|
|
305
|
+
export async function resolveCall(
|
|
306
|
+
host: ToolAdmissionHost,
|
|
307
|
+
toolCall: ToolCall,
|
|
308
|
+
): Promise<
|
|
309
|
+
{ ok: true; toolName: string; input: unknown } | { ok: false; toolName: string; message: string }
|
|
310
|
+
> {
|
|
311
|
+
let toolName = toolCall.function.name
|
|
312
|
+
let raw = toolCall.function.arguments
|
|
313
|
+
|
|
314
|
+
for (let attempt = 0; ; attempt++) {
|
|
315
|
+
const failure = inspectCall(host, toolName, raw)
|
|
316
|
+
if (!failure) return { ok: true, toolName, input: parseArguments(raw) }
|
|
317
|
+
|
|
318
|
+
const repair =
|
|
319
|
+
attempt === 0 && host.config.repairToolCall
|
|
320
|
+
? await requestRepair(host, toolCall, toolName, failure)
|
|
321
|
+
: null
|
|
322
|
+
|
|
323
|
+
if (!repair) {
|
|
324
|
+
if (failure.reason === 'invalid_json') {
|
|
325
|
+
return { ok: false, toolName, message: failure.message }
|
|
326
|
+
}
|
|
327
|
+
return { ok: true, toolName, input: parseArguments(raw) }
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
host.log.info('Repaired a malformed tool call', {
|
|
331
|
+
[NAMZU.RUN_ID]: host.config.runId,
|
|
332
|
+
[GENAI.TOOL_NAME]: toolName,
|
|
333
|
+
'namzu.runtime.reason': failure.reason,
|
|
334
|
+
...(repair.toolName && repair.toolName !== toolName
|
|
335
|
+
? { 'namzu.runtime.repaired_to': repair.toolName }
|
|
336
|
+
: {}),
|
|
337
|
+
})
|
|
338
|
+
toolName = repair.toolName ?? toolName
|
|
339
|
+
raw = repair.arguments
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
export async function repairTruncatedCall(
|
|
344
|
+
host: ToolAdmissionHost,
|
|
345
|
+
toolCall: ToolCall,
|
|
346
|
+
toolName: string,
|
|
347
|
+
): Promise<ToolCallRepair | null> {
|
|
348
|
+
if (!host.config.repairToolCall) return null
|
|
349
|
+
|
|
350
|
+
// Present the PARTIAL buffer, not the normalized `"{}"` — a repairer
|
|
351
|
+
// handed an empty object has nothing to work from.
|
|
352
|
+
const partial = toolCall.metadata?.partialArguments ?? ''
|
|
353
|
+
const repair = await requestRepair(
|
|
354
|
+
host,
|
|
355
|
+
{ ...toolCall, function: { ...toolCall.function, arguments: partial } },
|
|
356
|
+
toolName,
|
|
357
|
+
{ reason: 'invalid_json', message: truncatedToolInputMessage(toolName) },
|
|
358
|
+
)
|
|
359
|
+
if (repair) {
|
|
360
|
+
host.log.info('Repaired a tool call whose input stream was truncated', {
|
|
361
|
+
[NAMZU.RUN_ID]: host.config.runId,
|
|
362
|
+
[GENAI.TOOL_NAME]: toolName,
|
|
363
|
+
'namzu.runtime.partial_length': partial.length,
|
|
364
|
+
})
|
|
365
|
+
}
|
|
366
|
+
return repair
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
/**
|
|
370
|
+
* What is wrong with this call, or `null` if nothing is.
|
|
371
|
+
*
|
|
372
|
+
* JSON is checked before the tool is looked up: an unparseable argument
|
|
373
|
+
* string is broken regardless of which tool it was aimed at, and it is
|
|
374
|
+
* the one problem the executor itself has to answer.
|
|
375
|
+
*/
|
|
376
|
+
function inspectCall(
|
|
377
|
+
host: ToolAdmissionHost,
|
|
378
|
+
toolName: string,
|
|
379
|
+
raw: string,
|
|
380
|
+
): { reason: ToolCallRepairReason; message: string } | null {
|
|
381
|
+
let parsed: unknown
|
|
382
|
+
try {
|
|
383
|
+
parsed = parseArguments(raw)
|
|
384
|
+
} catch {
|
|
385
|
+
return {
|
|
386
|
+
reason: 'invalid_json',
|
|
387
|
+
message: `Error: Invalid JSON in tool arguments for "${toolName}"`,
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
const tool = host.config.tools.get?.(toolName)
|
|
392
|
+
if (!tool) {
|
|
393
|
+
// Either the model named a tool that does not exist, or this
|
|
394
|
+
// registry does not implement `get`. Both are the registry's to
|
|
395
|
+
// answer; a repairer still gets offered the `unknown_tool` case.
|
|
396
|
+
return {
|
|
397
|
+
reason: 'unknown_tool',
|
|
398
|
+
message: `Error: Unknown tool "${toolName}"`,
|
|
399
|
+
}
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
// A registry that hands back a tool with no schema has nothing to
|
|
403
|
+
// validate against; that is not a repairable condition, just an
|
|
404
|
+
// unvalidatable one.
|
|
405
|
+
const validation = tool.inputSchema?.safeParse(parsed)
|
|
406
|
+
if (validation && !validation.success) {
|
|
407
|
+
return {
|
|
408
|
+
reason: 'schema_validation',
|
|
409
|
+
message: `Error: Invalid arguments for "${toolName}": ${validation.error.issues
|
|
410
|
+
.map((issue) => `${issue.path.join('.') || '(root)'}: ${issue.message}`)
|
|
411
|
+
.join('; ')}`,
|
|
412
|
+
}
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
return null
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
async function requestRepair(
|
|
419
|
+
host: ToolAdmissionHost,
|
|
420
|
+
toolCall: ToolCall,
|
|
421
|
+
toolName: string,
|
|
422
|
+
failure: { reason: ToolCallRepairReason; message: string },
|
|
423
|
+
): Promise<ToolCallRepair | null> {
|
|
424
|
+
const repairToolCall = host.config.repairToolCall
|
|
425
|
+
if (!repairToolCall) return null
|
|
426
|
+
|
|
427
|
+
const tool = host.config.tools.get(toolName)
|
|
428
|
+
try {
|
|
429
|
+
return await repairToolCall({
|
|
430
|
+
toolCall,
|
|
431
|
+
reason: failure.reason,
|
|
432
|
+
message: failure.message,
|
|
433
|
+
...(tool
|
|
434
|
+
? {
|
|
435
|
+
tool,
|
|
436
|
+
jsonSchema: tool.modelInputSchema ?? renderToolSchema(tool.inputSchema),
|
|
437
|
+
}
|
|
438
|
+
: {}),
|
|
439
|
+
availableTools: host.config.tools.listNames(),
|
|
440
|
+
})
|
|
441
|
+
} catch (err) {
|
|
442
|
+
// A broken repairer must not turn a recoverable tool error into a
|
|
443
|
+
// failed run: the original error is still a perfectly good answer
|
|
444
|
+
// to give the model.
|
|
445
|
+
host.log.error('repairToolCall threw — falling back to the original error', {
|
|
446
|
+
[NAMZU.RUN_ID]: host.config.runId,
|
|
447
|
+
[GENAI.TOOL_NAME]: toolName,
|
|
448
|
+
'exception.message': toErrorMessage(err),
|
|
449
|
+
})
|
|
450
|
+
return null
|
|
451
|
+
}
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
/**
|
|
455
|
+
* An empty arguments string means "no arguments", not "malformed" — the
|
|
456
|
+
* shape a no-parameter tool arrives in.
|
|
457
|
+
*/
|
|
458
|
+
function parseArguments(raw: string): unknown {
|
|
459
|
+
return JSON.parse(raw || '{}')
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
export function formatFailedToolOutput(
|
|
463
|
+
output: string | undefined,
|
|
464
|
+
error: string | undefined,
|
|
465
|
+
): string {
|
|
466
|
+
const errorText = `Error: ${error ?? 'Tool execution failed'}`
|
|
467
|
+
if (!output || output.trim().length === 0) return errorText
|
|
468
|
+
return `${output}\n\n${errorText}`
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
export function truncatedToolInputMessage(toolName: string): string {
|
|
472
|
+
return `Error: Tool "${toolName}" call was cut off while the model was streaming JSON arguments. The tool was NOT executed. Retry with a much shorter input. Self-budget content/new_string under 12000 characters before calling file tools. For long files, create a short opening with write and a deterministic marker, then advance that marker with bounded exact edit calls; for delegated work, pass a shared workspace filename/reference instead of embedding the content in the tool call.`
|
|
473
|
+
}
|