@mcp-abap-adt/llm-agent-server-libs 20.0.0 → 20.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/builders/controller-skill-pipeline-builder.d.ts +57 -0
- package/dist/builders/controller-skill-pipeline-builder.d.ts.map +1 -0
- package/dist/builders/controller-skill-pipeline-builder.js +175 -0
- package/dist/builders/controller-skill-pipeline-builder.js.map +1 -0
- package/dist/factories/controller-factory.d.ts +18 -1
- package/dist/factories/controller-factory.d.ts.map +1 -1
- package/dist/factories/controller-factory.js +12 -1
- package/dist/factories/controller-factory.js.map +1 -1
- package/dist/generated/version.d.ts +1 -1
- package/dist/generated/version.js +1 -1
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/dist/mcp/compose-auxiliary.d.ts +35 -0
- package/dist/mcp/compose-auxiliary.d.ts.map +1 -0
- package/dist/mcp/compose-auxiliary.js +64 -0
- package/dist/mcp/compose-auxiliary.js.map +1 -0
- package/dist/pipelines/controller.d.ts.map +1 -1
- package/dist/pipelines/controller.js +87 -5
- package/dist/pipelines/controller.js.map +1 -1
- package/dist/pipelines/coordinator-resolvers.d.ts +68 -0
- package/dist/pipelines/coordinator-resolvers.d.ts.map +1 -0
- package/dist/pipelines/coordinator-resolvers.js +97 -0
- package/dist/pipelines/coordinator-resolvers.js.map +1 -0
- package/dist/pipelines/parsers.d.ts +1 -1
- package/dist/pipelines/parsers.d.ts.map +1 -1
- package/dist/pipelines/parsers.js +5 -4
- package/dist/pipelines/parsers.js.map +1 -1
- package/dist/pipelines/register-skill-sources.d.ts.map +1 -1
- package/dist/pipelines/register-skill-sources.js +5 -0
- package/dist/pipelines/register-skill-sources.js.map +1 -1
- package/dist/smart-agent/build-stepper-root.d.ts +3 -2
- package/dist/smart-agent/build-stepper-root.d.ts.map +1 -1
- package/dist/smart-agent/build-stepper-root.js +2 -1
- package/dist/smart-agent/build-stepper-root.js.map +1 -1
- package/dist/smart-agent/config-reload-watcher.d.ts +30 -0
- package/dist/smart-agent/config-reload-watcher.d.ts.map +1 -0
- package/dist/smart-agent/config-reload-watcher.js +101 -0
- package/dist/smart-agent/config-reload-watcher.js.map +1 -0
- package/dist/smart-agent/config-validator.d.ts +21 -0
- package/dist/smart-agent/config-validator.d.ts.map +1 -0
- package/dist/smart-agent/config-validator.js +196 -0
- package/dist/smart-agent/config-validator.js.map +1 -0
- package/dist/smart-agent/config.d.ts +16 -254
- package/dist/smart-agent/config.d.ts.map +1 -1
- package/dist/smart-agent/config.js +17 -904
- package/dist/smart-agent/config.js.map +1 -1
- package/dist/smart-agent/controller/board.d.ts +6 -3
- package/dist/smart-agent/controller/board.d.ts.map +1 -1
- package/dist/smart-agent/controller/board.js +21 -0
- package/dist/smart-agent/controller/board.js.map +1 -1
- package/dist/smart-agent/controller/controller-coordinator-handler.d.ts +24 -50
- package/dist/smart-agent/controller/controller-coordinator-handler.d.ts.map +1 -1
- package/dist/smart-agent/controller/controller-coordinator-handler.js +542 -676
- package/dist/smart-agent/controller/controller-coordinator-handler.js.map +1 -1
- package/dist/smart-agent/controller/default-step-execution-control.d.ts +7 -0
- package/dist/smart-agent/controller/default-step-execution-control.d.ts.map +1 -0
- package/dist/smart-agent/controller/default-step-execution-control.js +36 -0
- package/dist/smart-agent/controller/default-step-execution-control.js.map +1 -0
- package/dist/smart-agent/controller/finalizer.d.ts +5 -0
- package/dist/smart-agent/controller/finalizer.d.ts.map +1 -1
- package/dist/smart-agent/controller/finalizer.js +12 -5
- package/dist/smart-agent/controller/finalizer.js.map +1 -1
- package/dist/smart-agent/controller/noop-run-execution-control.d.ts +6 -0
- package/dist/smart-agent/controller/noop-run-execution-control.d.ts.map +1 -0
- package/dist/smart-agent/controller/noop-run-execution-control.js +14 -0
- package/dist/smart-agent/controller/noop-run-execution-control.js.map +1 -0
- package/dist/smart-agent/controller/parser.d.ts +11 -0
- package/dist/smart-agent/controller/parser.d.ts.map +1 -0
- package/dist/smart-agent/controller/parser.js +77 -0
- package/dist/smart-agent/controller/parser.js.map +1 -0
- package/dist/smart-agent/controller/planner.js +1 -1
- package/dist/smart-agent/controller/planner.js.map +1 -1
- package/dist/smart-agent/controller/recall.d.ts +48 -0
- package/dist/smart-agent/controller/recall.d.ts.map +1 -0
- package/dist/smart-agent/controller/recall.js +196 -0
- package/dist/smart-agent/controller/recall.js.map +1 -0
- package/dist/smart-agent/controller/reviewer.d.ts.map +1 -1
- package/dist/smart-agent/controller/reviewer.js +1 -1
- package/dist/smart-agent/controller/reviewer.js.map +1 -1
- package/dist/smart-agent/controller/subagent-client.d.ts +2 -2
- package/dist/smart-agent/controller/subagent-client.d.ts.map +1 -1
- package/dist/smart-agent/controller/subagent-client.js +2 -2
- package/dist/smart-agent/controller/subagent-client.js.map +1 -1
- package/dist/smart-agent/controller/types.d.ts +16 -4
- package/dist/smart-agent/controller/types.d.ts.map +1 -1
- package/dist/smart-agent/controller/types.js.map +1 -1
- package/dist/smart-agent/controller/usage-logging.d.ts +16 -0
- package/dist/smart-agent/controller/usage-logging.d.ts.map +1 -0
- package/dist/smart-agent/controller/usage-logging.js +39 -0
- package/dist/smart-agent/controller/usage-logging.js.map +1 -0
- package/dist/smart-agent/http/adapter-route-handler.d.ts +13 -0
- package/dist/smart-agent/http/adapter-route-handler.d.ts.map +1 -0
- package/dist/smart-agent/http/adapter-route-handler.js +76 -0
- package/dist/smart-agent/http/adapter-route-handler.js.map +1 -0
- package/dist/smart-agent/http/chat-route-handler.d.ts +15 -0
- package/dist/smart-agent/http/chat-route-handler.d.ts.map +1 -0
- package/dist/smart-agent/http/chat-route-handler.js +327 -0
- package/dist/smart-agent/http/chat-route-handler.js.map +1 -0
- package/dist/smart-agent/http/config-route-handler.d.ts +21 -0
- package/dist/smart-agent/http/config-route-handler.d.ts.map +1 -0
- package/dist/smart-agent/http/config-route-handler.js +149 -0
- package/dist/smart-agent/http/config-route-handler.js.map +1 -0
- package/dist/smart-agent/http/health-route-handler.d.ts +12 -0
- package/dist/smart-agent/http/health-route-handler.d.ts.map +1 -0
- package/dist/smart-agent/http/health-route-handler.js +19 -0
- package/dist/smart-agent/http/health-route-handler.js.map +1 -0
- package/dist/smart-agent/http/models-route-handler.d.ts +17 -0
- package/dist/smart-agent/http/models-route-handler.d.ts.map +1 -0
- package/dist/smart-agent/http/models-route-handler.js +65 -0
- package/dist/smart-agent/http/models-route-handler.js.map +1 -0
- package/dist/smart-agent/http/response-helpers.d.ts +31 -0
- package/dist/smart-agent/http/response-helpers.d.ts.map +1 -0
- package/dist/smart-agent/http/response-helpers.js +59 -0
- package/dist/smart-agent/http/response-helpers.js.map +1 -0
- package/dist/smart-agent/http/route-table.d.ts +44 -0
- package/dist/smart-agent/http/route-table.d.ts.map +1 -0
- package/dist/smart-agent/http/route-table.js +35 -0
- package/dist/smart-agent/http/route-table.js.map +1 -0
- package/dist/smart-agent/http/session-cookie.d.ts +10 -0
- package/dist/smart-agent/http/session-cookie.d.ts.map +1 -0
- package/dist/smart-agent/http/session-cookie.js +16 -0
- package/dist/smart-agent/http/session-cookie.js.map +1 -0
- package/dist/smart-agent/http/sessions-route-handler.d.ts +8 -0
- package/dist/smart-agent/http/sessions-route-handler.d.ts.map +1 -0
- package/dist/smart-agent/http/sessions-route-handler.js +54 -0
- package/dist/smart-agent/http/sessions-route-handler.js.map +1 -0
- package/dist/smart-agent/http/usage-route-handler.d.ts +12 -0
- package/dist/smart-agent/http/usage-route-handler.d.ts.map +1 -0
- package/dist/smart-agent/http/usage-route-handler.js +28 -0
- package/dist/smart-agent/http/usage-route-handler.js.map +1 -0
- package/dist/smart-agent/knowledge/make-knowledge-backend.d.ts +13 -0
- package/dist/smart-agent/knowledge/make-knowledge-backend.d.ts.map +1 -0
- package/dist/smart-agent/knowledge/make-knowledge-backend.js +18 -0
- package/dist/smart-agent/knowledge/make-knowledge-backend.js.map +1 -0
- package/dist/smart-agent/llm/role-llm-resolver.d.ts +30 -0
- package/dist/smart-agent/llm/role-llm-resolver.d.ts.map +1 -0
- package/dist/smart-agent/llm/role-llm-resolver.js +44 -0
- package/dist/smart-agent/llm/role-llm-resolver.js.map +1 -0
- package/dist/smart-agent/llm-config-map.d.ts +38 -0
- package/dist/smart-agent/llm-config-map.d.ts.map +1 -0
- package/dist/smart-agent/llm-config-map.js +75 -0
- package/dist/smart-agent/llm-config-map.js.map +1 -0
- package/dist/smart-agent/mcp-readiness-monitor.d.ts +38 -0
- package/dist/smart-agent/mcp-readiness-monitor.d.ts.map +1 -0
- package/dist/smart-agent/mcp-readiness-monitor.js +81 -0
- package/dist/smart-agent/mcp-readiness-monitor.js.map +1 -0
- package/dist/smart-agent/mcp-readiness-registry.d.ts +37 -0
- package/dist/smart-agent/mcp-readiness-registry.d.ts.map +1 -0
- package/dist/smart-agent/mcp-readiness-registry.js +55 -0
- package/dist/smart-agent/mcp-readiness-registry.js.map +1 -0
- package/dist/smart-agent/resolve-config-sections.d.ts +15 -0
- package/dist/smart-agent/resolve-config-sections.d.ts.map +1 -0
- package/dist/smart-agent/resolve-config-sections.js +212 -0
- package/dist/smart-agent/resolve-config-sections.js.map +1 -0
- package/dist/smart-agent/session-lifecycle/index.d.ts +117 -0
- package/dist/smart-agent/session-lifecycle/index.d.ts.map +1 -0
- package/dist/smart-agent/session-lifecycle/index.js +152 -0
- package/dist/smart-agent/session-lifecycle/index.js.map +1 -0
- package/dist/smart-agent/skill-plugins-config.d.ts +9 -1
- package/dist/smart-agent/skill-plugins-config.d.ts.map +1 -1
- package/dist/smart-agent/skill-plugins-config.js +27 -1
- package/dist/smart-agent/skill-plugins-config.js.map +1 -1
- package/dist/smart-agent/skill-plugins-host-factory.d.ts +14 -2
- package/dist/smart-agent/skill-plugins-host-factory.d.ts.map +1 -1
- package/dist/smart-agent/skill-plugins-host-factory.js +16 -4
- package/dist/smart-agent/skill-plugins-host-factory.js.map +1 -1
- package/dist/smart-agent/smart-server.d.ts +169 -222
- package/dist/smart-agent/smart-server.d.ts.map +1 -1
- package/dist/smart-agent/smart-server.js +569 -1387
- package/dist/smart-agent/smart-server.js.map +1 -1
- package/dist/smart-agent/stepper-config.d.ts +138 -0
- package/dist/smart-agent/stepper-config.d.ts.map +1 -0
- package/dist/smart-agent/stepper-config.js +216 -0
- package/dist/smart-agent/stepper-config.js.map +1 -0
- package/dist/smart-agent/tools-rag-handle.d.ts +10 -0
- package/dist/smart-agent/tools-rag-handle.d.ts.map +1 -0
- package/dist/smart-agent/tools-rag-handle.js +76 -0
- package/dist/smart-agent/tools-rag-handle.js.map +1 -0
- package/dist/smart-agent/workers/worker-registry.d.ts +159 -0
- package/dist/smart-agent/workers/worker-registry.d.ts.map +1 -0
- package/dist/smart-agent/workers/worker-registry.js +203 -0
- package/dist/smart-agent/workers/worker-registry.js.map +1 -0
- package/dist/smart-agent/yaml-loader.d.ts +7 -0
- package/dist/smart-agent/yaml-loader.d.ts.map +1 -0
- package/dist/smart-agent/yaml-loader.js +146 -0
- package/dist/smart-agent/yaml-loader.js.map +1 -0
- package/package.json +7 -7
|
@@ -1,55 +1,34 @@
|
|
|
1
|
-
import { externalToolCallId, } from '@mcp-abap-adt/llm-agent';
|
|
2
|
-
import { summaryToUsage, } from '@mcp-abap-adt/llm-agent-libs';
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
1
|
+
import { externalToolCallId, McpError, } from '@mcp-abap-adt/llm-agent';
|
|
2
|
+
import { LegacyAccumulateContextStrategy, LegacyTranscriptContextStrategy, summaryToUsage, } from '@mcp-abap-adt/llm-agent-libs';
|
|
3
|
+
import { writePlanDecision } from './artifacts.js';
|
|
4
|
+
import { BoardOverBudgetError, renderLiveBoard, } from './board.js';
|
|
5
|
+
import { DefaultStepExecutionControl } from './default-step-execution-control.js';
|
|
6
6
|
import { writeArtifact } from './memorizer.js';
|
|
7
7
|
import { resolveByPrecedence } from './outcome.js';
|
|
8
8
|
import { makeControllerPlanner } from './planner.js';
|
|
9
9
|
import { appendHint } from './prompts.js';
|
|
10
|
+
import { buildRecallBlock, collectApproved, RECALL_ARTIFACT_TYPES, RECALL_EVIDENCE_CHARS, RECALL_K_STEP, RECALL_MAX_CHARS_STEP, relevantExtract, runScopedRecall, } from './recall.js';
|
|
10
11
|
import { classifyRequest, readTerminal, writeTerminal } from './run-scope.js';
|
|
11
12
|
import { hydrateBundle, persistBundle, resetRun } from './session-bundle.js';
|
|
12
13
|
import { establishTargetState } from './target-state.js';
|
|
13
|
-
import {
|
|
14
|
+
import { makeLogUsage } from './usage-logging.js';
|
|
14
15
|
// ---------------------------------------------------------------------------
|
|
15
16
|
// Debug logging — gated behind DEBUG_CONTROLLER (e.g. DEBUG_CONTROLLER=1).
|
|
16
17
|
// Surfaces the steps the planner delegates and per-role/total token usage to
|
|
17
18
|
// stderr, for tuning step granularity and watching token spend. Off by default.
|
|
19
|
+
// (Also in usage-logging.ts — intentional small duplication; no 3rd copy exists.)
|
|
18
20
|
// ---------------------------------------------------------------------------
|
|
19
21
|
function dlog(msg) {
|
|
20
22
|
if (process.env.DEBUG_CONTROLLER)
|
|
21
23
|
console.error(`[controller] ${msg}`);
|
|
22
24
|
}
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
export function makeLogUsage(requestLogger, requestId, models) {
|
|
31
|
-
return (role, u) => {
|
|
32
|
-
if (!u)
|
|
33
|
-
return;
|
|
34
|
-
const model = role === 'finalizer'
|
|
35
|
-
? (models.finalizer ?? models.planner)
|
|
36
|
-
: role === 'reviewer'
|
|
37
|
-
? (models.reviewer ?? models.planner)
|
|
38
|
-
: role === 'embedding'
|
|
39
|
-
? 'embedder'
|
|
40
|
-
: (models[role] ?? 'unknown');
|
|
41
|
-
requestLogger.logLlmCall({
|
|
42
|
-
component: role,
|
|
43
|
-
model,
|
|
44
|
-
promptTokens: u.promptTokens ?? 0,
|
|
45
|
-
completionTokens: u.completionTokens ?? 0,
|
|
46
|
-
totalTokens: u.totalTokens ?? 0,
|
|
47
|
-
durationMs: 0,
|
|
48
|
-
requestId,
|
|
49
|
-
});
|
|
50
|
-
dlog(`tokens ${role}: prompt=${u.promptTokens} completion=${u.completionTokens} total=${u.totalTokens}`);
|
|
51
|
-
};
|
|
52
|
-
}
|
|
25
|
+
// ---------------------------------------------------------------------------
|
|
26
|
+
// Re-exported for import-path stability (helpers moved to sibling modules).
|
|
27
|
+
// ---------------------------------------------------------------------------
|
|
28
|
+
export { renderLiveBoard } from './board.js';
|
|
29
|
+
export { parseNextStep } from './parser.js';
|
|
30
|
+
export { relevantExtract, runScopedRecall } from './recall.js';
|
|
31
|
+
export { makeLogUsage } from './usage-logging.js';
|
|
53
32
|
// ---------------------------------------------------------------------------
|
|
54
33
|
// Handler
|
|
55
34
|
// ---------------------------------------------------------------------------
|
|
@@ -585,183 +564,222 @@ export class ControllerCoordinatorHandler {
|
|
|
585
564
|
const cfg = deps.config.budgets;
|
|
586
565
|
const maxToolCalls = cfg.maxToolCalls ?? 10;
|
|
587
566
|
const inFlight = bundle.inFlightStep; // set by the caller (block A or B)
|
|
588
|
-
//
|
|
589
|
-
//
|
|
590
|
-
//
|
|
591
|
-
//
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
567
|
+
// Per-step execution control: a wall-clock time budget (perStepTimeoutMs) +
|
|
568
|
+
// the prospective maxToolCalls gate, consumer-swappable via deps. Its signal
|
|
569
|
+
// is merged with the caller's cancel signal (below) and fed into both the
|
|
570
|
+
// executor.send and callMcp so a non-converging / hung step is CUT rather than
|
|
571
|
+
// livelocking. Absent perStepTimeoutMs → time never fires → count-only bound.
|
|
572
|
+
const stepControl = deps.stepExecutionControl ?? new DefaultStepExecutionControl();
|
|
573
|
+
const budget = stepControl.beginStep({
|
|
574
|
+
stepName: step.name,
|
|
575
|
+
seq: inFlight?.seq ?? 0,
|
|
576
|
+
attempt: inFlight?.attempt ?? 0,
|
|
577
|
+
budgets: { maxToolCalls, perStepTimeoutMs: cfg.perStepTimeoutMs },
|
|
578
|
+
});
|
|
579
|
+
// The budget owns a wall-clock timer that is NOT unref'd — it MUST be disposed
|
|
580
|
+
// on every step exit. Open the try IMMEDIATELY after beginStep so the
|
|
581
|
+
// budget-dependent pre-loop (recall / evidence embed / selectTools /
|
|
582
|
+
// strategy.record) is inside the SAME try…finally; a throw from any of those
|
|
583
|
+
// documented-fallible awaits (e.g. an embedder 429) would otherwise leak the
|
|
584
|
+
// timer and later fire controller.abort() on an orphaned signal.
|
|
585
|
+
try {
|
|
586
|
+
const stepStartedAt = Date.now();
|
|
587
|
+
// Merge the caller's request/cancel signal with the step budget: an inner call
|
|
588
|
+
// is cancelled by EITHER. The step-timeout DISCRIMINATOR remains
|
|
589
|
+
// budget.signal.aborted SPECIFICALLY, so a pure caller-cancel is NOT mis-mapped
|
|
590
|
+
// to a step-timeout control-failure.
|
|
591
|
+
const callSignal = ctx.options?.signal
|
|
592
|
+
? AbortSignal.any([ctx.options.signal, budget.signal])
|
|
593
|
+
: budget.signal;
|
|
594
|
+
// Typed reason code (StepControlDecision.reason) → human note. Preserves
|
|
595
|
+
// today's exact wording so existing suites stay byte-identical.
|
|
596
|
+
const noteFor = (r) => r === 'maxToolCalls'
|
|
597
|
+
? 'tool-call budget exhausted (maxToolCalls)'
|
|
598
|
+
: r === 'step-timeout'
|
|
599
|
+
? 'step time budget exhausted (step-timeout)'
|
|
600
|
+
: r;
|
|
601
|
+
let roundNo = 0;
|
|
602
|
+
const state = () => ({
|
|
603
|
+
round: roundNo,
|
|
604
|
+
toolCallCount: inFlight?.toolCallCount ?? 0,
|
|
605
|
+
elapsedMs: Date.now() - stepStartedAt,
|
|
606
|
+
});
|
|
607
|
+
// Persist the step outcome ATOMICALLY: record lastOutcome (durable, so a
|
|
608
|
+
// resume after a failed step replans instead of repeating it) AND advance the
|
|
609
|
+
// planner cursor (onCommit) in the SAME persistBundle that records the step
|
|
610
|
+
// result — never in a separate write, so a crash cannot replay a completed step.
|
|
611
|
+
const settle = async (outcome) => {
|
|
612
|
+
bundle.lastOutcome = outcome;
|
|
613
|
+
onCommit?.(outcome);
|
|
614
|
+
if (outcome === 'advanced' || outcome === 'partial') {
|
|
615
|
+
bundle.nextSeq = (bundle.nextSeq ?? 0) + 1;
|
|
616
|
+
bundle.inFlightStep = undefined;
|
|
617
|
+
bundle.runPhase = 'planning';
|
|
618
|
+
}
|
|
619
|
+
else {
|
|
620
|
+
// 'failed' — keep the same seq, mark awaiting-replan in the SAME persist so
|
|
621
|
+
// recovery routes by durable phase.
|
|
622
|
+
if (bundle.inFlightStep)
|
|
623
|
+
bundle.inFlightStep.phase = 'awaiting-replan';
|
|
624
|
+
bundle.runPhase = 'executing';
|
|
625
|
+
}
|
|
626
|
+
await persistBundle(deps.backend, sessionId, bundle);
|
|
627
|
+
return outcome;
|
|
628
|
+
};
|
|
629
|
+
// The IMMUTABLE per-round prefix: system + step user message + the step-result
|
|
630
|
+
// recall block. Re-emitted verbatim every round via strategy.form(); the dynamic
|
|
631
|
+
// tool rounds are owned by the injected context strategy, NOT accumulated here.
|
|
632
|
+
const staticPrefix = [
|
|
633
|
+
{
|
|
634
|
+
role: 'system',
|
|
635
|
+
content: appendHint(EXECUTOR_SYSTEM, deps.config.subagents.executor?.hint),
|
|
636
|
+
},
|
|
637
|
+
{
|
|
638
|
+
role: 'user',
|
|
639
|
+
content: `Goal: ${bundle.goal}\nStep: ${step.name}\nInstructions: ${step.instructions}`,
|
|
640
|
+
},
|
|
641
|
+
];
|
|
642
|
+
// Episodic recall: pull prior STEP-RESULT artifacts relevant to this step from
|
|
643
|
+
// session-memory and inject them as static context. The session-memory rag shares
|
|
644
|
+
// the bundle backend, so restrict to 'step-result' (excludes the
|
|
645
|
+
// 'controller-bundle' infrastructure record). Bounded by k and length. The
|
|
646
|
+
// per-round MCP context is now the context strategy's job (its form() supplies
|
|
647
|
+
// the mcp-result rounds — the Window keeps its own buffer, RagRecall recalls),
|
|
648
|
+
// so it is NOT part of the handler-built static prefix.
|
|
649
|
+
const recallText = step.instructions || step.name;
|
|
650
|
+
const maxAttempts = cfg.maxStepAttempts ?? 5;
|
|
651
|
+
const recalledSteps = await runScopedRecall(rag, recallText, RECALL_K_STEP, bundle.runId, RECALL_K_STEP * (maxAttempts + 1), ['step-result'], ctx.options);
|
|
652
|
+
const stepBlock = buildRecallBlock(recalledSteps, RECALL_MAX_CHARS_STEP);
|
|
653
|
+
if (stepBlock) {
|
|
654
|
+
staticPrefix.push({ role: 'user', content: stepBlock });
|
|
655
|
+
}
|
|
656
|
+
// Per-step tool-loop context strategy (record/form). Absent factory →
|
|
657
|
+
// LegacyAccumulateContextStrategy (byte-identical to the historical growing
|
|
658
|
+
// transcript).
|
|
659
|
+
const makeStrategy = () => (deps.toolLoopContextStrategyFactory ??
|
|
660
|
+
(() => new LegacyAccumulateContextStrategy()))({
|
|
661
|
+
run: { rag, runId: bundle.runId, meta, stepName: step.name },
|
|
662
|
+
});
|
|
663
|
+
// Resume / migration selection (Task 12). A step that suspended under the new
|
|
664
|
+
// design carries a serialized `contextStrategyState` → RESTORE it so the
|
|
665
|
+
// pre-suspend rounds (including any INTERNAL tool rounds before an external
|
|
666
|
+
// suspend) come back exactly as the executor last saw them. A PRE-RELEASE
|
|
667
|
+
// in-flight step carries only a raw `transcript` (no snapshot) → migrate it
|
|
668
|
+
// verbatim via the migration-only LegacyTranscriptContextStrategy (one release).
|
|
669
|
+
// Otherwise a fresh step.
|
|
670
|
+
let strategy;
|
|
671
|
+
if (inFlight?.contextStrategyState !== undefined) {
|
|
672
|
+
const state = inFlight.contextStrategyState;
|
|
673
|
+
// A migrated step persisted a LegacyTranscript snapshot ({rawMessages,
|
|
674
|
+
// newRounds}). Restore it through the SAME strategy type so its raw history
|
|
675
|
+
// + post-migration rounds survive a SECOND resume; a normal snapshot
|
|
676
|
+
// restores via the injected/default strategy. Discriminate on shape.
|
|
677
|
+
strategy =
|
|
678
|
+
state.rawMessages !== undefined
|
|
679
|
+
? new LegacyTranscriptContextStrategy({ rawMessages: [] })
|
|
680
|
+
: makeStrategy();
|
|
681
|
+
strategy.restore(state);
|
|
682
|
+
}
|
|
683
|
+
else if (inFlight?.transcript?.length) {
|
|
684
|
+
strategy = new LegacyTranscriptContextStrategy({
|
|
685
|
+
rawMessages: inFlight.transcript,
|
|
686
|
+
});
|
|
687
|
+
// Adopted verbatim (rawMessages is copied) → clear the durable transcript so
|
|
688
|
+
// it only ever carries UN-recorded external-continuation pairs from here on.
|
|
689
|
+
// The next suspend snapshots this strategy → subsequent resumes take the
|
|
690
|
+
// restore branch above, so the raw history is never re-injected (no double).
|
|
691
|
+
inFlight.transcript.length = 0;
|
|
599
692
|
}
|
|
600
693
|
else {
|
|
601
|
-
|
|
602
|
-
// recovery routes by durable phase.
|
|
603
|
-
if (bundle.inFlightStep)
|
|
604
|
-
bundle.inFlightStep.phase = 'awaiting-replan';
|
|
605
|
-
bundle.runPhase = 'executing';
|
|
694
|
+
strategy = makeStrategy(); // fresh step
|
|
606
695
|
}
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
{
|
|
612
|
-
role: 'system',
|
|
613
|
-
content: appendHint(EXECUTOR_SYSTEM, deps.config.subagents.executor?.hint),
|
|
614
|
-
},
|
|
615
|
-
{
|
|
616
|
-
role: 'user',
|
|
617
|
-
content: `Goal: ${bundle.goal}\nStep: ${step.name}\nInstructions: ${step.instructions}`,
|
|
618
|
-
},
|
|
619
|
-
];
|
|
620
|
-
// Episodic recall: pull prior artifacts relevant to this step from
|
|
621
|
-
// session-memory and inject them as context. The session-memory rag shares
|
|
622
|
-
// the bundle backend, so restrict to artifact types (excludes the
|
|
623
|
-
// 'controller-bundle' infrastructure record). Bounded by k and length.
|
|
624
|
-
const recallText = step.instructions || step.name;
|
|
625
|
-
const maxAttempts = cfg.maxStepAttempts ?? 5;
|
|
626
|
-
const maxTool = cfg.maxToolCalls ?? 10;
|
|
627
|
-
// Per-kind run-scoped recall with GUARANTEED over-fetch bounds: step-result
|
|
628
|
-
// retries are bounded by maxStepAttempts (k×(maxStepAttempts+1)); mcp-results
|
|
629
|
-
// are NOT deduped on re-fetch here and toolCallCount RESETS per attempt, so one
|
|
630
|
-
// run can emit up to maxSteps × maxStepAttempts × maxToolCalls of them — over-
|
|
631
|
-
// fetch that full run bound so every distinct identityKey is seen before the cap.
|
|
632
|
-
const mcpBound = cfg.maxSteps * maxAttempts * maxTool;
|
|
633
|
-
const recalledSteps = await runScopedRecall(rag, recallText, RECALL_K_STEP, bundle.runId, RECALL_K_STEP * (maxAttempts + 1), ['step-result'], ctx.options);
|
|
634
|
-
const recalledMcp = await runScopedRecall(rag, recallText, RECALL_K_MCP, bundle.runId, mcpBound, ['mcp-result'], ctx.options);
|
|
635
|
-
// SEPARATE character budgets per kind: a single huge step-result cannot consume
|
|
636
|
-
// the whole budget and starve the MCP context (and vice-versa).
|
|
637
|
-
const stepBlock = buildRecallBlock(recalledSteps, RECALL_MAX_CHARS_STEP);
|
|
638
|
-
const mcpBlock = buildRecallBlock(recalledMcp, RECALL_MAX_CHARS_MCP);
|
|
639
|
-
const recallBlock = [stepBlock, mcpBlock].filter(Boolean).join('\n\n');
|
|
640
|
-
if (recallBlock) {
|
|
641
|
-
messages.push({ role: 'user', content: recallBlock });
|
|
642
|
-
}
|
|
643
|
-
// Durable transcript = static prefix (system/user/recall) + the dynamic
|
|
644
|
-
// executor/tool turns. On a resume/continuation the dynamic tail is rebuilt
|
|
645
|
-
// from inFlightStep.transcript so the executor sees the FULL exchange it had
|
|
646
|
-
// (prior tool rounds + the injected external result), not just a fragment.
|
|
647
|
-
const staticLen = messages.length;
|
|
648
|
-
if (inFlight && inFlight.transcript.length > 0) {
|
|
649
|
-
messages.push(...inFlight.transcript);
|
|
650
|
-
}
|
|
651
|
-
// Persist the dynamic tail after every executor/tool exchange so a suspend or
|
|
652
|
-
// crash never rebuilds with a shorter conversation than the executor saw.
|
|
653
|
-
const syncTranscript = async () => {
|
|
696
|
+
// Durable, bounded control-message tail (retries only). Aliased IN PLACE so
|
|
697
|
+
// push/prune mutate the persisted field; a legacy call with no inFlightStep gets
|
|
698
|
+
// an ephemeral local (no durable tail to persist).
|
|
699
|
+
let controlTail;
|
|
654
700
|
if (inFlight) {
|
|
655
|
-
inFlight.
|
|
656
|
-
|
|
701
|
+
inFlight.controlTail = inFlight.controlTail ?? [];
|
|
702
|
+
controlTail = inFlight.controlTail;
|
|
657
703
|
}
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
679
|
-
}
|
|
680
|
-
// Tools offered to the executor = the INTERNAL (MCP) tools semantically
|
|
681
|
-
// relevant to THIS step (top-K from toolsRag) PLUS the per-request external
|
|
682
|
-
// (consumer-supplied) tools. The executor decides which to call; internal
|
|
683
|
-
// calls route through `callMcp`, external calls round-trip via `isExternalTool`.
|
|
684
|
-
const relevant = await deps.selectTools(step.instructions || step.name, TOOL_SELECT_K, ctx.options);
|
|
685
|
-
const offeredTools = [...relevant, ...(ctx.externalTools ?? [])];
|
|
686
|
-
// The executor may ONLY call a tool that was offered to it: an internal tool
|
|
687
|
-
// selected for this step, or a per-request external tool. Any other name
|
|
688
|
-
// (hallucinated / stale / not in the top-K) is rejected — never executed —
|
|
689
|
-
// so the semantic exposure boundary actually bounds what runs.
|
|
690
|
-
const offeredInternalNames = new Set(relevant.map((t) => t.name));
|
|
691
|
-
let retries = 0;
|
|
692
|
-
// (D) Persist a 'failed' step-result artifact for controller-level failures
|
|
693
|
-
// (reviewer unverifiable, executor error exhausted, maxToolCalls, unavailable
|
|
694
|
-
// tool) so the board can project the step's terminal state from artifacts alone.
|
|
695
|
-
const writeControlFailure = async (reason) => {
|
|
696
|
-
const seq = bundle.inFlightStep?.seq ?? bundle.nextSeq ?? 0;
|
|
697
|
-
const attempt = bundle.inFlightStep?.attempt ?? 0;
|
|
698
|
-
bundle.writeOrdinal = (bundle.writeOrdinal ?? 0) + 1;
|
|
699
|
-
await writeArtifact(rag, {
|
|
700
|
-
...meta,
|
|
701
|
-
artifactType: 'step-result',
|
|
702
|
-
task: step.name,
|
|
703
|
-
runId: bundle.runId,
|
|
704
|
-
seq,
|
|
705
|
-
attempt,
|
|
706
|
-
status: 'failed',
|
|
707
|
-
note: reason,
|
|
708
|
-
remainder: '',
|
|
709
|
-
stepId: step.stepId,
|
|
710
|
-
digest: reason.slice(0, cfg.maxDigestChars ?? 500),
|
|
711
|
-
writeOrdinal: bundle.writeOrdinal,
|
|
712
|
-
content: '',
|
|
713
|
-
}, ctx.options);
|
|
714
|
-
};
|
|
715
|
-
// Inner loop handles tool routing / error retries until the executor
|
|
716
|
-
// produces content for this step (or the step suspends on an external tool).
|
|
717
|
-
while (true) {
|
|
718
|
-
const res = await deps.executor.send(messages, offeredTools);
|
|
719
|
-
logUsage?.('executor', res.usage);
|
|
720
|
-
if (res.kind === 'content') {
|
|
721
|
-
// Hold the executor's result; the reviewer (NOT the executor) decides the
|
|
722
|
-
// outcome. Default reviewer (no deps.reviewer) approves as 'ok' (legacy).
|
|
723
|
-
let review = deps.reviewer
|
|
724
|
-
? await deps.reviewer.review(step, evidence, res.content, {
|
|
725
|
-
hint: deps.config.subagents.reviewer?.hint,
|
|
726
|
-
logUsage,
|
|
727
|
-
maxDigestChars: cfg.maxDigestChars ?? 500,
|
|
728
|
-
})
|
|
729
|
-
: {
|
|
730
|
-
kind: 'outcome',
|
|
731
|
-
outcome: {
|
|
732
|
-
status: 'ok',
|
|
733
|
-
approved: res.content,
|
|
734
|
-
remainder: '',
|
|
735
|
-
note: '',
|
|
736
|
-
digest: res.content.slice(0, cfg.maxDigestChars ?? 500),
|
|
737
|
-
},
|
|
738
|
-
};
|
|
739
|
-
// Judge failure (provider error / malformed / contradictory ok-with-empty)
|
|
740
|
-
// is NOT a step failure: re-ask within maxReviewRetries, then ABORT (the
|
|
741
|
-
// outcome is unverifiable). Never mapped to settle('failed')/replan.
|
|
742
|
-
let reviewRetries = 0;
|
|
743
|
-
while (review.kind === 'judge-failure') {
|
|
744
|
-
reviewRetries++;
|
|
745
|
-
if (reviewRetries > (cfg.maxReviewRetries ?? 2)) {
|
|
746
|
-
// The reviewer could not produce a usable verdict within the retry
|
|
747
|
-
// budget (provider error / unparsable). DEGRADE to a failed step so the
|
|
748
|
-
// planner replans, rather than aborting the whole run — the terminal
|
|
749
|
-
// backstop is maxStepAttempts/maxSteps, not a single unverifiable verdict.
|
|
750
|
-
bundle.budgets.stepsUsed++;
|
|
751
|
-
await writeControlFailure(`reviewer unverifiable after ${cfg.maxReviewRetries ?? 2} retries: ${review.reason}`);
|
|
752
|
-
bundle.plannerPrivate += `\n[seq ${bundle.inFlightStep?.seq ?? bundle.nextSeq ?? 0} ${step.name} failed] reviewer unverifiable after ${cfg.maxReviewRetries ?? 2} retries: ${review.reason}`;
|
|
753
|
-
return settle('failed');
|
|
704
|
+
else {
|
|
705
|
+
controlTail = [];
|
|
706
|
+
}
|
|
707
|
+
// External CONTINUATION bridge: the resume preamble APPENDS the freshly
|
|
708
|
+
// resolved external assistant/tool pair(s) to inFlight.transcript. Record them
|
|
709
|
+
// as rounds ON TOP of the restored strategy so the executor continues from its
|
|
710
|
+
// own tool call, then CLEAR the transcript — they now live in the strategy, so
|
|
711
|
+
// leaving them would double-record on the next external round-trip. (The
|
|
712
|
+
// LegacyTranscript migration above already adopted AND cleared its transcript,
|
|
713
|
+
// so here the transcript holds ONLY the just-injected external pair — never the
|
|
714
|
+
// migrated raw history; the two never double-inject the same rounds.)
|
|
715
|
+
if (inFlight && inFlight.transcript.length > 0) {
|
|
716
|
+
const t = inFlight.transcript;
|
|
717
|
+
let i = 0;
|
|
718
|
+
while (i < t.length) {
|
|
719
|
+
const assistant = t[i];
|
|
720
|
+
i++;
|
|
721
|
+
const results = [];
|
|
722
|
+
while (i < t.length && t[i]?.role === 'tool') {
|
|
723
|
+
results.push(t[i]);
|
|
724
|
+
i++;
|
|
754
725
|
}
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
logUsage,
|
|
758
|
-
maxDigestChars: cfg.maxDigestChars ?? 500,
|
|
759
|
-
});
|
|
726
|
+
if (assistant)
|
|
727
|
+
await strategy.record({ assistant, results }, ctx.options);
|
|
760
728
|
}
|
|
761
|
-
|
|
729
|
+
inFlight.transcript.length = 0;
|
|
730
|
+
}
|
|
731
|
+
// Snapshot the strategy state + persist the bundle after every executor/tool
|
|
732
|
+
// exchange so a suspend or crash resumes with the same context the executor saw.
|
|
733
|
+
const persistExchange = async () => {
|
|
734
|
+
if (inFlight) {
|
|
735
|
+
inFlight.contextStrategyState = strategy.snapshot();
|
|
736
|
+
await persistBundle(deps.backend, sessionId, bundle);
|
|
737
|
+
}
|
|
738
|
+
};
|
|
739
|
+
// Per-reference evidence: one recall per requires[] reference. A non-empty
|
|
740
|
+
// top-K does NOT prove the dependency is present — semantic recall returns the
|
|
741
|
+
// NEAREST artifact even at low relevance — so we hand the reviewer the TOP
|
|
742
|
+
// artifact's relevant fragment (Evidence.topArtifact) and let IT (the judging
|
|
743
|
+
// role) decide whether the ref is actually satisfied. `hit` is a coarse
|
|
744
|
+
// any-candidate flag. Gathered SEQUENTIALLY (NOT Promise.all): each
|
|
745
|
+
// relevantExtract is itself bounded-sequential, so the outer sequential loop
|
|
746
|
+
// keeps at most ONE embed request in flight at a time (rate-limit-safe).
|
|
747
|
+
const refs = step.requires && step.requires.length > 0
|
|
748
|
+
? step.requires
|
|
749
|
+
: [recallText];
|
|
750
|
+
const evBound = RECALL_K_STEP * (maxAttempts + 1) +
|
|
751
|
+
cfg.maxSteps * maxAttempts * (cfg.maxToolCalls ?? 10);
|
|
752
|
+
const evidence = [];
|
|
753
|
+
for (const ref of refs) {
|
|
754
|
+
const hits = await runScopedRecall(rag, ref, 1, bundle.runId, evBound, RECALL_ARTIFACT_TYPES, ctx.options);
|
|
755
|
+
const topArtifact = hits[0]
|
|
756
|
+
? await relevantExtract(hits[0].content, ref, RECALL_EVIDENCE_CHARS,
|
|
757
|
+
// biome-ignore lint/style/noNonNullAssertion: distance strategies require an embedder; the factory enforces it (Task 17).
|
|
758
|
+
deps.embedder, ctx.options)
|
|
759
|
+
: undefined;
|
|
760
|
+
evidence.push({ ref, hit: hits.length > 0, topArtifact });
|
|
761
|
+
}
|
|
762
|
+
// Tools offered to the executor = the INTERNAL (MCP) tools semantically
|
|
763
|
+
// relevant to THIS step (top-K from toolsRag) PLUS the per-request external
|
|
764
|
+
// (consumer-supplied) tools. The executor decides which to call; internal
|
|
765
|
+
// calls route through `callMcp`, external calls round-trip via `isExternalTool`.
|
|
766
|
+
const relevant = await deps.selectTools(step.instructions || step.name, TOOL_SELECT_K, ctx.options);
|
|
767
|
+
const offeredTools = [
|
|
768
|
+
...relevant,
|
|
769
|
+
...(ctx.externalTools ?? []),
|
|
770
|
+
];
|
|
771
|
+
// The executor may ONLY call a tool that was offered to it: an internal tool
|
|
772
|
+
// selected for this step, or a per-request external tool. Any other name
|
|
773
|
+
// (hallucinated / stale / not in the top-K) is rejected — never executed —
|
|
774
|
+
// so the semantic exposure boundary actually bounds what runs.
|
|
775
|
+
const offeredInternalNames = new Set(relevant.map((t) => t.name));
|
|
776
|
+
let retries = 0;
|
|
777
|
+
// (D) Persist a 'failed' step-result artifact for controller-level failures
|
|
778
|
+
// (reviewer unverifiable, executor error exhausted, maxToolCalls, unavailable
|
|
779
|
+
// tool) so the board can project the step's terminal state from artifacts alone.
|
|
780
|
+
const writeControlFailure = async (reason) => {
|
|
762
781
|
const seq = bundle.inFlightStep?.seq ?? bundle.nextSeq ?? 0;
|
|
763
782
|
const attempt = bundle.inFlightStep?.attempt ?? 0;
|
|
764
|
-
// ONE post-review write carrying the COMPLETE Outcome + identity.
|
|
765
783
|
bundle.writeOrdinal = (bundle.writeOrdinal ?? 0) + 1;
|
|
766
784
|
await writeArtifact(rag, {
|
|
767
785
|
...meta,
|
|
@@ -770,174 +788,334 @@ export class ControllerCoordinatorHandler {
|
|
|
770
788
|
runId: bundle.runId,
|
|
771
789
|
seq,
|
|
772
790
|
attempt,
|
|
773
|
-
status:
|
|
774
|
-
note:
|
|
775
|
-
remainder:
|
|
791
|
+
status: 'failed',
|
|
792
|
+
note: reason,
|
|
793
|
+
remainder: '',
|
|
776
794
|
stepId: step.stepId,
|
|
777
|
-
digest:
|
|
795
|
+
digest: reason.slice(0, cfg.maxDigestChars ?? 500),
|
|
778
796
|
writeOrdinal: bundle.writeOrdinal,
|
|
779
|
-
content:
|
|
797
|
+
content: '',
|
|
780
798
|
}, ctx.options);
|
|
799
|
+
};
|
|
800
|
+
// A controller-level (non-reviewer) cut: persist a 'failed' step-result +
|
|
801
|
+
// planner note (preserving today's wording via noteFor) + the TYPED durable
|
|
802
|
+
// ControlFailure, then settle('failed') so the planner replans at this seq.
|
|
803
|
+
const cutControlFailure = async (reason) => {
|
|
781
804
|
bundle.budgets.stepsUsed++;
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
793
|
-
retries++;
|
|
794
|
-
if (retries <= cfg.maxRetries) {
|
|
795
|
-
messages.push({
|
|
796
|
-
role: 'user',
|
|
797
|
-
content: `The previous attempt failed: ${res.error}. Retry the step.`,
|
|
798
|
-
});
|
|
799
|
-
await syncTranscript();
|
|
800
|
-
continue;
|
|
805
|
+
await writeControlFailure(noteFor(reason));
|
|
806
|
+
bundle.plannerPrivate += `\n[seq ${inFlight?.seq ?? bundle.nextSeq ?? 0} ${step.name} control-failed] ${noteFor(reason)}`;
|
|
807
|
+
if (inFlight) {
|
|
808
|
+
inFlight.phase = 'awaiting-replan';
|
|
809
|
+
const typedReason = reason === 'maxToolCalls' || reason === 'step-timeout'
|
|
810
|
+
? reason
|
|
811
|
+
: 'control-failure';
|
|
812
|
+
inFlight.controlFailure = {
|
|
813
|
+
reason: typedReason,
|
|
814
|
+
seq: inFlight.seq,
|
|
815
|
+
};
|
|
801
816
|
}
|
|
802
|
-
// Retries exhausted — feed the error back as the step result so the
|
|
803
|
-
// planner can replan on the next iteration.
|
|
804
|
-
bundle.budgets.stepsUsed++;
|
|
805
|
-
await writeControlFailure(`executor error: ${res.error}`);
|
|
806
|
-
bundle.plannerPrivate += `\n[step ${step.name} failed] ${res.error}`;
|
|
807
817
|
return settle('failed');
|
|
808
|
-
}
|
|
809
|
-
//
|
|
810
|
-
|
|
811
|
-
|
|
812
|
-
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
+
};
|
|
819
|
+
// Inner loop handles tool routing / error retries until the executor
|
|
820
|
+
// produces content for this step (or the step suspends on an external tool).
|
|
821
|
+
// The whole loop is wrapped so the step budget's timer is disposed on EVERY
|
|
822
|
+
// exit (settle / cut / suspend / abort / throw).
|
|
823
|
+
while (true) {
|
|
824
|
+
roundNo++;
|
|
825
|
+
// TIME/abort gate BEFORE each executor round (count is gated per tool call).
|
|
826
|
+
const rc = budget.shouldContinueRound(state());
|
|
827
|
+
if (!rc.continue)
|
|
828
|
+
return cutControlFailure(rc.reason);
|
|
829
|
+
// Form the per-round executor context: the immutable prefix + the strategy's
|
|
830
|
+
// rounds, then the bounded control tail (retries). NEVER a growing raw array.
|
|
831
|
+
const messages = (await strategy.form({ prefix: staticPrefix, queryText: step.instructions }, ctx.options)).concat(controlTail);
|
|
832
|
+
// Merged signal into the executor call. A reject WHILE the step budget is
|
|
833
|
+
// aborted is a step-timeout cut (NOT the executor-error retry); a normal
|
|
834
|
+
// return after the budget fired is ALSO a step-timeout cut.
|
|
835
|
+
let res;
|
|
836
|
+
try {
|
|
837
|
+
res = await deps.executor.send(messages, offeredTools, {
|
|
838
|
+
...ctx.options,
|
|
839
|
+
signal: callSignal,
|
|
818
840
|
});
|
|
819
|
-
await syncTranscript();
|
|
820
|
-
continue;
|
|
821
841
|
}
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
842
|
+
catch (e) {
|
|
843
|
+
if (budget.signal.aborted)
|
|
844
|
+
return cutControlFailure('step-timeout');
|
|
845
|
+
throw e;
|
|
846
|
+
}
|
|
847
|
+
if (budget.signal.aborted)
|
|
848
|
+
return cutControlFailure('step-timeout');
|
|
849
|
+
logUsage?.('executor', res.usage);
|
|
850
|
+
if (res.kind === 'content') {
|
|
851
|
+
// Hold the executor's result; the reviewer (NOT the executor) decides the
|
|
852
|
+
// outcome. Default reviewer (no deps.reviewer) approves as 'ok' (legacy).
|
|
853
|
+
let review = deps.reviewer
|
|
854
|
+
? await deps.reviewer.review(step, evidence, res.content, {
|
|
855
|
+
hint: deps.config.subagents.reviewer?.hint,
|
|
856
|
+
logUsage,
|
|
857
|
+
maxDigestChars: cfg.maxDigestChars ?? 500,
|
|
858
|
+
})
|
|
859
|
+
: {
|
|
860
|
+
kind: 'outcome',
|
|
861
|
+
outcome: {
|
|
862
|
+
status: 'ok',
|
|
863
|
+
approved: res.content,
|
|
864
|
+
remainder: '',
|
|
865
|
+
note: '',
|
|
866
|
+
digest: res.content.slice(0, cfg.maxDigestChars ?? 500),
|
|
867
|
+
},
|
|
868
|
+
};
|
|
869
|
+
// Judge failure (provider error / malformed / contradictory ok-with-empty)
|
|
870
|
+
// is NOT a step failure: re-ask within maxReviewRetries, then ABORT (the
|
|
871
|
+
// outcome is unverifiable). Never mapped to settle('failed')/replan.
|
|
872
|
+
let reviewRetries = 0;
|
|
873
|
+
while (review.kind === 'judge-failure') {
|
|
874
|
+
reviewRetries++;
|
|
875
|
+
if (reviewRetries > (cfg.maxReviewRetries ?? 2)) {
|
|
876
|
+
// The reviewer could not produce a usable verdict within the retry
|
|
877
|
+
// budget (provider error / unparsable). DEGRADE to a failed step so the
|
|
878
|
+
// planner replans, rather than aborting the whole run — the terminal
|
|
879
|
+
// backstop is maxStepAttempts/maxSteps, not a single unverifiable verdict.
|
|
880
|
+
bundle.budgets.stepsUsed++;
|
|
881
|
+
await writeControlFailure(`reviewer unverifiable after ${cfg.maxReviewRetries ?? 2} retries: ${review.reason}`);
|
|
882
|
+
bundle.plannerPrivate += `\n[seq ${bundle.inFlightStep?.seq ?? bundle.nextSeq ?? 0} ${step.name} failed] reviewer unverifiable after ${cfg.maxReviewRetries ?? 2} retries: ${review.reason}`;
|
|
883
|
+
return settle('failed');
|
|
884
|
+
}
|
|
885
|
+
review = await deps.reviewer.review(step, evidence, res.content, {
|
|
886
|
+
hint: deps.config.subagents.reviewer?.hint,
|
|
887
|
+
logUsage,
|
|
888
|
+
maxDigestChars: cfg.maxDigestChars ?? 500,
|
|
889
|
+
});
|
|
890
|
+
}
|
|
891
|
+
const outcome = review.outcome;
|
|
892
|
+
const seq = bundle.inFlightStep?.seq ?? bundle.nextSeq ?? 0;
|
|
893
|
+
const attempt = bundle.inFlightStep?.attempt ?? 0;
|
|
894
|
+
// ONE post-review write carrying the COMPLETE Outcome + identity.
|
|
895
|
+
bundle.writeOrdinal = (bundle.writeOrdinal ?? 0) + 1;
|
|
896
|
+
await writeArtifact(rag, {
|
|
897
|
+
...meta,
|
|
898
|
+
artifactType: 'step-result',
|
|
899
|
+
task: step.name,
|
|
900
|
+
runId: bundle.runId,
|
|
901
|
+
seq,
|
|
902
|
+
attempt,
|
|
903
|
+
status: outcome.status,
|
|
904
|
+
note: outcome.note,
|
|
905
|
+
remainder: outcome.remainder,
|
|
906
|
+
stepId: step.stepId,
|
|
907
|
+
digest: outcome.digest,
|
|
908
|
+
writeOrdinal: bundle.writeOrdinal,
|
|
909
|
+
content: outcome.approved,
|
|
910
|
+
}, ctx.options);
|
|
835
911
|
bundle.budgets.stepsUsed++;
|
|
836
|
-
|
|
837
|
-
bundle
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
912
|
+
const mapped = mapOutcome(outcome.status);
|
|
913
|
+
recordStepControl(bundle, {
|
|
914
|
+
seq: bundle.inFlightStep?.seq ?? seq,
|
|
915
|
+
name: step.name,
|
|
916
|
+
status: outcome.status,
|
|
917
|
+
note: outcome.note,
|
|
918
|
+
remainder: outcome.remainder,
|
|
919
|
+
});
|
|
920
|
+
return settle(mapped);
|
|
921
|
+
}
|
|
922
|
+
if (res.kind === 'error') {
|
|
923
|
+
retries++;
|
|
924
|
+
if (retries <= cfg.maxRetries) {
|
|
925
|
+
controlTail.push({
|
|
926
|
+
role: 'user',
|
|
927
|
+
content: `The previous attempt failed: ${res.error}. Retry the step.`,
|
|
928
|
+
});
|
|
929
|
+
await persistExchange();
|
|
930
|
+
continue;
|
|
931
|
+
}
|
|
932
|
+
// Retries exhausted — feed the error back as the step result so the
|
|
933
|
+
// planner can replan on the next iteration.
|
|
934
|
+
bundle.budgets.stepsUsed++;
|
|
935
|
+
await writeControlFailure(`executor error: ${res.error}`);
|
|
936
|
+
bundle.plannerPrivate += `\n[step ${step.name} failed] ${res.error}`;
|
|
937
|
+
return settle('failed');
|
|
938
|
+
}
|
|
939
|
+
// res.kind === 'tool_call' → route the FIRST tool call.
|
|
940
|
+
const firstCall = res.toolCalls[0];
|
|
941
|
+
if (firstCall === undefined) {
|
|
942
|
+
// Empty tool-call array → treat as an executor error (retry/replan).
|
|
943
|
+
retries++;
|
|
944
|
+
if (retries <= cfg.maxRetries) {
|
|
945
|
+
controlTail.push({
|
|
946
|
+
role: 'user',
|
|
947
|
+
content: 'The previous attempt produced an empty tool call. Retry the step.',
|
|
948
|
+
});
|
|
949
|
+
await persistExchange();
|
|
950
|
+
continue;
|
|
951
|
+
}
|
|
952
|
+
bundle.budgets.stepsUsed++;
|
|
953
|
+
await writeControlFailure('empty tool call');
|
|
954
|
+
bundle.plannerPrivate += `\n[step ${step.name} failed] empty tool call`;
|
|
955
|
+
return settle('failed');
|
|
956
|
+
}
|
|
957
|
+
// Normalize the StreamToolCall (full or delta) into an LlmToolCall inline.
|
|
958
|
+
const call = 'arguments' in firstCall &&
|
|
959
|
+
typeof firstCall.arguments === 'object' &&
|
|
960
|
+
firstCall.arguments !== null
|
|
961
|
+
? {
|
|
962
|
+
id: ('id' in firstCall && firstCall.id) || 'call',
|
|
963
|
+
name: ('name' in firstCall && firstCall.name) || '',
|
|
964
|
+
arguments: firstCall.arguments,
|
|
965
|
+
}
|
|
966
|
+
: (() => {
|
|
967
|
+
let iArgs = {};
|
|
968
|
+
const raw = 'arguments' in firstCall ? firstCall.arguments : undefined;
|
|
969
|
+
if (typeof raw === 'string' && raw.length > 0) {
|
|
970
|
+
try {
|
|
971
|
+
iArgs = JSON.parse(raw);
|
|
972
|
+
}
|
|
973
|
+
catch {
|
|
974
|
+
iArgs = {};
|
|
975
|
+
}
|
|
976
|
+
}
|
|
977
|
+
return {
|
|
978
|
+
id: ('id' in firstCall && firstCall.id) || 'call',
|
|
979
|
+
name: ('name' in firstCall && firstCall.name) || '',
|
|
980
|
+
arguments: iArgs,
|
|
981
|
+
};
|
|
982
|
+
})();
|
|
983
|
+
const name = call.name;
|
|
984
|
+
const args = call.arguments;
|
|
985
|
+
if (isExternalTool(name)) {
|
|
986
|
+
// External round-trips share the SAME budget as internal calls; the
|
|
987
|
+
// prospective count gate is consulted BEFORE the increment so an external
|
|
988
|
+
// tool cannot exceed the cap. Cut → control-failed replan at the same seq.
|
|
989
|
+
const g = budget.canExecuteTool(state());
|
|
990
|
+
if (!g.continue)
|
|
991
|
+
return cutControlFailure(g.reason);
|
|
992
|
+
// Snapshot the strategy state SO FAR before we suspend (the resume injection
|
|
993
|
+
// appends the external assistant/tool pair, recorded on the next invocation).
|
|
994
|
+
if (inFlight)
|
|
995
|
+
inFlight.contextStrategyState = strategy.snapshot();
|
|
996
|
+
const extId = externalToolCallId(name, args);
|
|
997
|
+
if (inFlight)
|
|
998
|
+
inFlight.toolCallCount += 1;
|
|
999
|
+
// The new marker REPLACES any prior pending (a fresh extId).
|
|
1000
|
+
bundle.pending = {
|
|
1001
|
+
kind: 'external-tool',
|
|
1002
|
+
extId,
|
|
1003
|
+
toolName: name,
|
|
1004
|
+
args,
|
|
1005
|
+
position: step.name,
|
|
842
1006
|
};
|
|
1007
|
+
bundle.runState = 'suspended';
|
|
1008
|
+
await persistBundle(deps.backend, sessionId, bundle);
|
|
1009
|
+
this.surfaceToolCall(ctx, { id: extId, name, arguments: args }, usageNow?.());
|
|
1010
|
+
return 'suspended';
|
|
1011
|
+
}
|
|
1012
|
+
// The executor may only call a tool that was OFFERED to it this step. A
|
|
1013
|
+
// name that is neither external nor in the internal top-K (hallucinated /
|
|
1014
|
+
// stale / out-of-scope) is rejected — NOT executed — and fed back as a
|
|
1015
|
+
// tool-not-available error so the executor retries with an offered tool.
|
|
1016
|
+
if (!offeredInternalNames.has(name)) {
|
|
1017
|
+
retries++;
|
|
1018
|
+
if (retries <= cfg.maxRetries) {
|
|
1019
|
+
controlTail.push({
|
|
1020
|
+
role: 'user',
|
|
1021
|
+
content: `Tool "${name}" is not available for this step. Use only the tools provided to you.`,
|
|
1022
|
+
});
|
|
1023
|
+
await persistExchange();
|
|
1024
|
+
continue;
|
|
1025
|
+
}
|
|
1026
|
+
bundle.budgets.stepsUsed++;
|
|
1027
|
+
await writeControlFailure(`requested unavailable tool ${name}`);
|
|
1028
|
+
bundle.plannerPrivate += `\n[step ${step.name} failed] requested unavailable tool ${name}`;
|
|
843
1029
|
return settle('failed');
|
|
844
1030
|
}
|
|
845
|
-
//
|
|
846
|
-
//
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
1031
|
+
// Prospective count gate BEFORE the increment (the before-increment model:
|
|
1032
|
+
// the increment happens only after canExecuteTool allows the call).
|
|
1033
|
+
const g = budget.canExecuteTool(state());
|
|
1034
|
+
if (!g.continue)
|
|
1035
|
+
return cutControlFailure(g.reason);
|
|
1036
|
+
// Durable round-trip count: ++ and persist BEFORE surfacing so it survives a
|
|
1037
|
+
// resume (never a per-resume local).
|
|
1038
|
+
if (inFlight) {
|
|
850
1039
|
inFlight.toolCallCount += 1;
|
|
851
|
-
|
|
852
|
-
bundle.pending = {
|
|
853
|
-
kind: 'external-tool',
|
|
854
|
-
extId,
|
|
855
|
-
toolName: name,
|
|
856
|
-
args,
|
|
857
|
-
position: step.name,
|
|
858
|
-
};
|
|
859
|
-
bundle.runState = 'suspended';
|
|
860
|
-
await persistBundle(deps.backend, sessionId, bundle);
|
|
861
|
-
this.surfaceToolCall(ctx, { id: extId, name, arguments: args }, usageNow?.());
|
|
862
|
-
return 'suspended';
|
|
863
|
-
}
|
|
864
|
-
// The executor may only call a tool that was OFFERED to it this step. A
|
|
865
|
-
// name that is neither external nor in the internal top-K (hallucinated /
|
|
866
|
-
// stale / out-of-scope) is rejected — NOT executed — and fed back as a
|
|
867
|
-
// tool-not-available error so the executor retries with an offered tool.
|
|
868
|
-
if (!offeredInternalNames.has(name)) {
|
|
869
|
-
retries++;
|
|
870
|
-
if (retries <= cfg.maxRetries) {
|
|
871
|
-
messages.push({
|
|
872
|
-
role: 'user',
|
|
873
|
-
content: `Tool "${name}" is not available for this step. Use only the tools provided to you.`,
|
|
874
|
-
});
|
|
875
|
-
await syncTranscript();
|
|
876
|
-
continue;
|
|
1040
|
+
await persistBundle(deps.backend, sessionId, bundle);
|
|
877
1041
|
}
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
// Controller-level failure (NOT a reviewer status): record durably and replan.
|
|
891
|
-
bundle.budgets.stepsUsed++;
|
|
892
|
-
await writeControlFailure('tool-call budget exhausted (maxToolCalls)');
|
|
893
|
-
bundle.plannerPrivate += `\n[seq ${inFlight?.seq ?? bundle.nextSeq ?? 0} ${step.name} control-failed] tool-call budget exhausted (maxToolCalls)`;
|
|
894
|
-
if (inFlight) {
|
|
895
|
-
inFlight.phase = 'awaiting-replan';
|
|
896
|
-
inFlight.controlFailure = {
|
|
897
|
-
reason: 'maxToolCalls',
|
|
898
|
-
seq: inFlight.seq,
|
|
899
|
-
};
|
|
1042
|
+
// Execute locally, memorize, re-send to the executor.
|
|
1043
|
+
// FAIL LOUD: surface an MCP-unavailable failure as a terminal abort (not a
|
|
1044
|
+
// silent empty response). The bridge (buildMcpBridge) throws an McpError
|
|
1045
|
+
// IFF the injected classifier deemed it 'unavailable'; a tool-level error
|
|
1046
|
+
// is returned as TEXT, never thrown. So ANY McpError reaching this catch is
|
|
1047
|
+
// already a classifier-unavailable verdict — trust that throw-contract
|
|
1048
|
+
// rather than re-checking the code (which would drop a CUSTOM classifier's
|
|
1049
|
+
// decision → rethrow → outer catch swallow → (no response)). A non-McpError
|
|
1050
|
+
// is a genuine unexpected error and is re-thrown for the outer handler.
|
|
1051
|
+
let result;
|
|
1052
|
+
try {
|
|
1053
|
+
result = await deps.callMcp(name, args, callSignal);
|
|
900
1054
|
}
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
1055
|
+
catch (mcpErr) {
|
|
1056
|
+
// A step-timeout cancellation aborts the merged signal → the bridge rejects.
|
|
1057
|
+
// Map that to a step-timeout control-failure BEFORE the McpError escalate so
|
|
1058
|
+
// it is NOT mis-classified as MCP-unavailable (20.4.0 escalate order is
|
|
1059
|
+
// otherwise unchanged).
|
|
1060
|
+
if (budget.signal.aborted)
|
|
1061
|
+
return cutControlFailure('step-timeout');
|
|
1062
|
+
if (mcpErr instanceof McpError) {
|
|
1063
|
+
const now = deps.now ?? (() => new Date().toISOString());
|
|
1064
|
+
const terminalTtlMs = deps.terminalTtlMs ?? 24 * 60 * 60 * 1000;
|
|
1065
|
+
await this.abortTerminal(ctx, sessionId, bundle, `MCP server unavailable: ${mcpErr.message}`, now, terminalTtlMs, usageNow?.());
|
|
1066
|
+
return 'aborted';
|
|
1067
|
+
}
|
|
1068
|
+
throw mcpErr;
|
|
1069
|
+
}
|
|
1070
|
+
// Record this exchange as a coherent assistant→tool ROUND (OpenAI protocol)
|
|
1071
|
+
// via the context strategy so the executor LLM continues from its own tool
|
|
1072
|
+
// call. The strategy owns the per-round context (Window keeps a bounded
|
|
1073
|
+
// buffer; RagRecall (Task 13) persists the mcp-result + recalls it) — the
|
|
1074
|
+
// handler no longer writes the mcp-result artifact or grows a raw transcript.
|
|
1075
|
+
// Durable monotonic write ordinal for this mcp-result write. RagRecall's
|
|
1076
|
+
// run-scoped dedup (isBetterMcp) tie-breaks on writeOrdinal FIRST (then
|
|
1077
|
+
// createdAt); since all mcp-result writes in a step share createdAt, a later
|
|
1078
|
+
// same-identityKey fetch only wins with a strictly-higher ordinal. Increment
|
|
1079
|
+
// BEFORE building the round so the value rides on it into strategy.record.
|
|
1080
|
+
bundle.writeOrdinal = (bundle.writeOrdinal ?? 0) + 1;
|
|
1081
|
+
const round = {
|
|
1082
|
+
assistant: {
|
|
1083
|
+
role: 'assistant',
|
|
1084
|
+
content: null,
|
|
1085
|
+
tool_calls: [
|
|
1086
|
+
{
|
|
1087
|
+
id: call.id,
|
|
1088
|
+
type: 'function',
|
|
1089
|
+
function: { name, arguments: JSON.stringify(args) },
|
|
1090
|
+
},
|
|
1091
|
+
],
|
|
931
1092
|
},
|
|
932
|
-
|
|
933
|
-
|
|
934
|
-
|
|
935
|
-
|
|
936
|
-
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
|
|
1093
|
+
results: [
|
|
1094
|
+
{
|
|
1095
|
+
role: 'tool',
|
|
1096
|
+
tool_call_id: call.id,
|
|
1097
|
+
content: result,
|
|
1098
|
+
},
|
|
1099
|
+
],
|
|
1100
|
+
// Stable fetch identity (tool+args) for run-scoped recall dedup. The
|
|
1101
|
+
// controller has no tool-level error classifier here — an unavailable MCP
|
|
1102
|
+
// server aborts BEFORE record; a returned string is a delivered result.
|
|
1103
|
+
meta: [
|
|
1104
|
+
{ identityKey: externalToolCallId(name, args), isError: false },
|
|
1105
|
+
],
|
|
1106
|
+
ordinal: bundle.writeOrdinal,
|
|
1107
|
+
roundId: undefined,
|
|
1108
|
+
};
|
|
1109
|
+
await strategy.record(round, ctx.options);
|
|
1110
|
+
// A recorded round supersedes any pending control retry → prune the tail.
|
|
1111
|
+
controlTail.length = 0;
|
|
1112
|
+
// The executor saw this round → make the strategy state durable before next.
|
|
1113
|
+
await persistExchange();
|
|
1114
|
+
}
|
|
1115
|
+
}
|
|
1116
|
+
finally {
|
|
1117
|
+
// Dispose the step budget (clears its wall-clock timer) on EVERY exit.
|
|
1118
|
+
budget.dispose();
|
|
941
1119
|
}
|
|
942
1120
|
}
|
|
943
1121
|
// -- Escalation & surfacing (mirror StepperCoordinatorHandler) ----------
|
|
@@ -1004,12 +1182,18 @@ export class ControllerCoordinatorHandler {
|
|
|
1004
1182
|
await persistBundle(deps.backend, sessionId, bundle);
|
|
1005
1183
|
let answer;
|
|
1006
1184
|
if (deps.finalizer && bundle.runId) {
|
|
1185
|
+
// Recall the skills block ONCE (not per finalize retry — re-embedding on
|
|
1186
|
+
// every attempt is wasteful; the recall is invariant across retries).
|
|
1187
|
+
const skillsBlock = deps.skillsRecall
|
|
1188
|
+
? await deps.skillsRecall(bundle.goal, ctx.options)
|
|
1189
|
+
: undefined;
|
|
1007
1190
|
while (answer === undefined) {
|
|
1008
1191
|
try {
|
|
1009
1192
|
const composed = await deps.finalizer.finalize(bundle.goal, request, approved, {
|
|
1010
1193
|
hint: deps.config.subagents.finalizer?.hint,
|
|
1011
1194
|
logUsage,
|
|
1012
1195
|
log: (m) => dlog(m),
|
|
1196
|
+
skillsBlock,
|
|
1013
1197
|
});
|
|
1014
1198
|
// Empty-but-ok finalizer output is a JUDGE failure (spec), not a valid
|
|
1015
1199
|
// answer → throw so it retries within maxFinalizeRetries.
|
|
@@ -1080,7 +1264,7 @@ export class ControllerCoordinatorHandler {
|
|
|
1080
1264
|
}
|
|
1081
1265
|
}
|
|
1082
1266
|
// ---------------------------------------------------------------------------
|
|
1083
|
-
//
|
|
1267
|
+
// Goal-clarification helpers
|
|
1084
1268
|
// ---------------------------------------------------------------------------
|
|
1085
1269
|
/** Bare confirmations that, on a goal clarify, commit the evaluator's proposed
|
|
1086
1270
|
* target rather than the literal answer. Anything else = a refinement. */
|
|
@@ -1111,7 +1295,6 @@ function isAffirmation(answer) {
|
|
|
1111
1295
|
.replace(/[.!]+$/, '');
|
|
1112
1296
|
return AFFIRMATIONS.has(t);
|
|
1113
1297
|
}
|
|
1114
|
-
/** Top-K tools surfaced from toolsRag per planner/step query. */
|
|
1115
1298
|
/** Agnostic executor system prompt. Domain specifics (e.g. SAP/ABAP fact kinds)
|
|
1116
1299
|
* are layered on via `subagents.executor.hint` (see {@link appendHint}). */
|
|
1117
1300
|
const EXECUTOR_SYSTEM = 'You are the executor. You have tools that read the live target system. ' +
|
|
@@ -1124,44 +1307,11 @@ const EXECUTOR_SYSTEM = 'You are the executor. You have tools that read the live
|
|
|
1124
1307
|
'if the step asks for a LIST or overview, return the list; do NOT then go and ' +
|
|
1125
1308
|
'fetch the full details of every listed item unless the step explicitly asks ' +
|
|
1126
1309
|
'for per-item details.';
|
|
1310
|
+
/** Top-K tools surfaced from toolsRag per planner/step query. */
|
|
1127
1311
|
const TOOL_SELECT_K = 20;
|
|
1128
|
-
/** Top-k recalled artifacts injected into the executor context per step. */
|
|
1129
|
-
/** Artifact types eligible for recall (excludes the 'controller-bundle' record
|
|
1130
|
-
* that shares the same backend). */
|
|
1131
|
-
const RECALL_ARTIFACT_TYPES = ['step-result', 'mcp-result'];
|
|
1132
|
-
/** Per-kind recall counts (distinct artifacts kept after dedup + cap). */
|
|
1133
|
-
const RECALL_K_STEP = 4;
|
|
1134
|
-
const RECALL_K_MCP = 4;
|
|
1135
|
-
/** SEPARATE char budgets per kind, so a huge step-result cannot starve MCP context. */
|
|
1136
|
-
const RECALL_MAX_CHARS_STEP = 2000;
|
|
1137
|
-
const RECALL_MAX_CHARS_MCP = 2000;
|
|
1138
|
-
/** Char budget for a single per-`requires` evidence extract handed to the reviewer. */
|
|
1139
|
-
const RECALL_EVIDENCE_CHARS = 800;
|
|
1140
1312
|
// ---------------------------------------------------------------------------
|
|
1141
1313
|
// Pure helpers
|
|
1142
1314
|
// ---------------------------------------------------------------------------
|
|
1143
|
-
/** Build a bounded "Relevant prior context" block from recalled artifacts under
|
|
1144
|
-
* the given char budget, or undefined when there is nothing to inject. */
|
|
1145
|
-
function buildRecallBlock(hits, maxChars) {
|
|
1146
|
-
if (hits.length === 0)
|
|
1147
|
-
return undefined;
|
|
1148
|
-
const parts = [];
|
|
1149
|
-
let used = 0;
|
|
1150
|
-
for (const h of hits) {
|
|
1151
|
-
const c = h.content ?? '';
|
|
1152
|
-
if (c.length === 0)
|
|
1153
|
-
continue;
|
|
1154
|
-
if (used + c.length > maxChars) {
|
|
1155
|
-
parts.push(c.slice(0, maxChars - used));
|
|
1156
|
-
break;
|
|
1157
|
-
}
|
|
1158
|
-
parts.push(c);
|
|
1159
|
-
used += c.length;
|
|
1160
|
-
}
|
|
1161
|
-
if (parts.length === 0)
|
|
1162
|
-
return undefined;
|
|
1163
|
-
return `Relevant prior context:\n${parts.join('\n')}`;
|
|
1164
|
-
}
|
|
1165
1315
|
/** Extract the user prompt from the request's textOrMessages. */
|
|
1166
1316
|
function extractPrompt(textOrMessages) {
|
|
1167
1317
|
if (typeof textOrMessages === 'string')
|
|
@@ -1172,131 +1322,6 @@ function extractPrompt(textOrMessages) {
|
|
|
1172
1322
|
}
|
|
1173
1323
|
return '';
|
|
1174
1324
|
}
|
|
1175
|
-
/** Parse a planner content string into a NextStep, defensively. */
|
|
1176
|
-
/** Parse the planner's reply into a NextStep, tolerating ```json fences and
|
|
1177
|
-
* surrounding prose. Returns null when no valid decision can be extracted — the
|
|
1178
|
-
* caller treats that as a FORMAT error (re-ask the planner), NOT a rewind, so a
|
|
1179
|
-
* badly-formatted reply never silently burns the rewind budget. */
|
|
1180
|
-
export function parseNextStep(content) {
|
|
1181
|
-
const json = extractJsonObject(content);
|
|
1182
|
-
if (json === null)
|
|
1183
|
-
return null;
|
|
1184
|
-
try {
|
|
1185
|
-
const obj = JSON.parse(json);
|
|
1186
|
-
if (obj.kind === 'done' && typeof obj.result === 'string')
|
|
1187
|
-
return { kind: 'done', result: obj.result };
|
|
1188
|
-
if (obj.kind === 'rewind' && typeof obj.reason === 'string')
|
|
1189
|
-
return { kind: 'rewind', reason: obj.reason };
|
|
1190
|
-
if (obj.kind === 'next' &&
|
|
1191
|
-
obj.step &&
|
|
1192
|
-
typeof obj.step.name === 'string' &&
|
|
1193
|
-
typeof obj.step.instructions === 'string') {
|
|
1194
|
-
// Validate requires[] so a non-string / empty / oversized reference never
|
|
1195
|
-
// reaches the semantic query / embedder; a malformed value is a parse
|
|
1196
|
-
// failure that drives the existing parse-retry.
|
|
1197
|
-
const req = validateRequires(obj.step.requires);
|
|
1198
|
-
if (req === false)
|
|
1199
|
-
return null;
|
|
1200
|
-
return {
|
|
1201
|
-
kind: 'next',
|
|
1202
|
-
step: {
|
|
1203
|
-
name: obj.step.name,
|
|
1204
|
-
instructions: obj.step.instructions,
|
|
1205
|
-
...(obj.step.type ? { type: obj.step.type } : {}),
|
|
1206
|
-
...(req ? { requires: req } : {}),
|
|
1207
|
-
},
|
|
1208
|
-
};
|
|
1209
|
-
}
|
|
1210
|
-
}
|
|
1211
|
-
catch {
|
|
1212
|
-
// fall through
|
|
1213
|
-
}
|
|
1214
|
-
return null;
|
|
1215
|
-
}
|
|
1216
|
-
/** Extract the first balanced JSON object from a planner reply, ignoring ```json
|
|
1217
|
-
* fences and prose around it. String-aware (braces inside strings don't count).
|
|
1218
|
-
* Returns null if no balanced object is present. */
|
|
1219
|
-
export function extractJsonObject(raw) {
|
|
1220
|
-
const fence = raw.match(/```(?:json)?\s*([\s\S]*?)```/i);
|
|
1221
|
-
const body = fence ? fence[1] : raw;
|
|
1222
|
-
const start = body.indexOf('{');
|
|
1223
|
-
if (start < 0)
|
|
1224
|
-
return null;
|
|
1225
|
-
let depth = 0;
|
|
1226
|
-
let inStr = false;
|
|
1227
|
-
let esc = false;
|
|
1228
|
-
for (let i = start; i < body.length; i++) {
|
|
1229
|
-
const ch = body[i];
|
|
1230
|
-
if (inStr) {
|
|
1231
|
-
if (esc)
|
|
1232
|
-
esc = false;
|
|
1233
|
-
else if (ch === '\\')
|
|
1234
|
-
esc = true;
|
|
1235
|
-
else if (ch === '"')
|
|
1236
|
-
inStr = false;
|
|
1237
|
-
continue;
|
|
1238
|
-
}
|
|
1239
|
-
if (ch === '"')
|
|
1240
|
-
inStr = true;
|
|
1241
|
-
else if (ch === '{')
|
|
1242
|
-
depth++;
|
|
1243
|
-
else if (ch === '}') {
|
|
1244
|
-
depth--;
|
|
1245
|
-
if (depth === 0)
|
|
1246
|
-
return body.slice(start, i + 1);
|
|
1247
|
-
}
|
|
1248
|
-
}
|
|
1249
|
-
return null;
|
|
1250
|
-
}
|
|
1251
|
-
/** Normalize a StreamToolCall (full or delta) into an LlmToolCall. */
|
|
1252
|
-
function toLlmToolCall(c) {
|
|
1253
|
-
if ('arguments' in c &&
|
|
1254
|
-
typeof c.arguments === 'object' &&
|
|
1255
|
-
c.arguments !== null) {
|
|
1256
|
-
// Full LlmToolCall: arguments is already a parsed object.
|
|
1257
|
-
return {
|
|
1258
|
-
id: ('id' in c && c.id) || 'call',
|
|
1259
|
-
name: ('name' in c && c.name) || '',
|
|
1260
|
-
arguments: c.arguments,
|
|
1261
|
-
};
|
|
1262
|
-
}
|
|
1263
|
-
// Delta: arguments is a (possibly partial) JSON string.
|
|
1264
|
-
let args = {};
|
|
1265
|
-
const raw = 'arguments' in c ? c.arguments : undefined;
|
|
1266
|
-
if (typeof raw === 'string' && raw.length > 0) {
|
|
1267
|
-
try {
|
|
1268
|
-
args = JSON.parse(raw);
|
|
1269
|
-
}
|
|
1270
|
-
catch {
|
|
1271
|
-
args = {};
|
|
1272
|
-
}
|
|
1273
|
-
}
|
|
1274
|
-
return {
|
|
1275
|
-
id: ('id' in c && c.id) || 'call',
|
|
1276
|
-
name: ('name' in c && c.name) || '',
|
|
1277
|
-
arguments: args,
|
|
1278
|
-
};
|
|
1279
|
-
}
|
|
1280
|
-
/** Reconstruct and render the live step-state board from artifacts.
|
|
1281
|
-
* Returns '' when there is no runId (the board has nothing to show yet). */
|
|
1282
|
-
async function renderLiveBoard(rag, bundle, budget) {
|
|
1283
|
-
const runId = bundle.runId;
|
|
1284
|
-
if (!runId)
|
|
1285
|
-
return '';
|
|
1286
|
-
const [structure, claims] = await Promise.all([
|
|
1287
|
-
readPlanDecisions(rag, runId),
|
|
1288
|
-
readClaims(rag, runId),
|
|
1289
|
-
]);
|
|
1290
|
-
const stepResults = await rag.list({ runId, artifactType: 'step-result' });
|
|
1291
|
-
const board = reconstructBoard({
|
|
1292
|
-
structure,
|
|
1293
|
-
stepResults,
|
|
1294
|
-
claims,
|
|
1295
|
-
inFlight: bundle.inFlightStep,
|
|
1296
|
-
pending: bundle.pending,
|
|
1297
|
-
});
|
|
1298
|
-
return renderBoard(board, budget);
|
|
1299
|
-
}
|
|
1300
1325
|
/** Synthesize the strict KnowledgeEntryMetadata for controller artifacts. */
|
|
1301
1326
|
function synthMeta(ctx, sessionId) {
|
|
1302
1327
|
const traceId = ctx.options?.trace?.traceId ?? sessionId;
|
|
@@ -1328,163 +1353,4 @@ function recordStepControl(bundle, rec) {
|
|
|
1328
1353
|
(rec.note ? ` ${rec.note}` : '') +
|
|
1329
1354
|
(rec.remainder ? ` remainder: ${rec.remainder}` : '');
|
|
1330
1355
|
}
|
|
1331
|
-
/** Gather the run's approved results, one per seq, resolved by outcome precedence
|
|
1332
|
-
* (ok/exists > partial > failed), ordered by seq. Reconstructs the complete
|
|
1333
|
-
* Outcome from artifact metadata (status/note/remainder) + content. */
|
|
1334
|
-
async function collectApproved(rag, runId) {
|
|
1335
|
-
const all = await rag.list({ runId, artifactType: 'step-result' });
|
|
1336
|
-
const bySeq = new Map();
|
|
1337
|
-
for (const e of all) {
|
|
1338
|
-
const seq = e.metadata.seq ?? 0;
|
|
1339
|
-
const o = {
|
|
1340
|
-
status: (e.metadata.status ?? 'failed'),
|
|
1341
|
-
approved: e.content,
|
|
1342
|
-
remainder: e.metadata.remainder ?? '',
|
|
1343
|
-
note: e.metadata.note ?? '',
|
|
1344
|
-
};
|
|
1345
|
-
const arr = bySeq.get(seq);
|
|
1346
|
-
if (arr)
|
|
1347
|
-
arr.push(o);
|
|
1348
|
-
else
|
|
1349
|
-
bySeq.set(seq, [o]);
|
|
1350
|
-
}
|
|
1351
|
-
const out = [];
|
|
1352
|
-
for (const [seq, outcomes] of [...bySeq.entries()].sort((a, b) => a[0] - b[0])) {
|
|
1353
|
-
const resolved = resolveByPrecedence(outcomes);
|
|
1354
|
-
if (resolved && resolved.status !== 'failed')
|
|
1355
|
-
out.push({ seq, content: resolved.approved });
|
|
1356
|
-
}
|
|
1357
|
-
return out;
|
|
1358
|
-
}
|
|
1359
|
-
/** The ONE run-scoped results-RAG recall primitive — used by BOTH the whole-step
|
|
1360
|
-
* recall AND the per-`requires` evidence. EMBEDDING-based similarity via the
|
|
1361
|
-
* backend's semantic query (NO homemade lexical scoring): the backend embeds the
|
|
1362
|
-
* query + ranks by vector similarity, with the `runId` filter applied PRE-cap.
|
|
1363
|
-
* Over-fetch `kPrime` (caller-supplied so the duplication bound is justified PER
|
|
1364
|
-
* KIND), then dedup and cap to `k`. Dedup: step-results (have `seq`) →
|
|
1365
|
-
* precedence-winner per seq; mcp-results → by `identityKey`. Embedding rank order
|
|
1366
|
-
* is preserved through the dedup.
|
|
1367
|
-
* `options` is forwarded into the embedder so recall-time embeds are metered. */
|
|
1368
|
-
export async function runScopedRecall(rag, text, k, runId, kPrime, artifactType, options) {
|
|
1369
|
-
const hits = await rag.query(text, {
|
|
1370
|
-
k: kPrime,
|
|
1371
|
-
filter: { runId, artifactType },
|
|
1372
|
-
options,
|
|
1373
|
-
});
|
|
1374
|
-
const bestStep = new Map();
|
|
1375
|
-
const bestMcp = new Map();
|
|
1376
|
-
for (const e of hits) {
|
|
1377
|
-
if (e.metadata.seq !== undefined && e.metadata.status !== undefined) {
|
|
1378
|
-
const prev = bestStep.get(e.metadata.seq);
|
|
1379
|
-
if (!prev || isBetterStep(e, prev))
|
|
1380
|
-
bestStep.set(e.metadata.seq, e);
|
|
1381
|
-
}
|
|
1382
|
-
else if (e.metadata.identityKey) {
|
|
1383
|
-
const prev = bestMcp.get(e.metadata.identityKey);
|
|
1384
|
-
if (!prev || isBetterMcp(e, prev))
|
|
1385
|
-
bestMcp.set(e.metadata.identityKey, e);
|
|
1386
|
-
}
|
|
1387
|
-
}
|
|
1388
|
-
// Walk hits in embedding-rank order; emit each (runId,seq) / identityKey once.
|
|
1389
|
-
const out = [];
|
|
1390
|
-
const seenSeq = new Set();
|
|
1391
|
-
const seenMcp = new Set();
|
|
1392
|
-
for (const e of hits) {
|
|
1393
|
-
if (e.metadata.seq !== undefined && e.metadata.status !== undefined) {
|
|
1394
|
-
if (seenSeq.has(e.metadata.seq))
|
|
1395
|
-
continue;
|
|
1396
|
-
seenSeq.add(e.metadata.seq);
|
|
1397
|
-
// biome-ignore lint/style/noNonNullAssertion: bestStep has this seq (set above).
|
|
1398
|
-
out.push(bestStep.get(e.metadata.seq));
|
|
1399
|
-
}
|
|
1400
|
-
else if (e.metadata.identityKey) {
|
|
1401
|
-
if (seenMcp.has(e.metadata.identityKey))
|
|
1402
|
-
continue;
|
|
1403
|
-
seenMcp.add(e.metadata.identityKey);
|
|
1404
|
-
// biome-ignore lint/style/noNonNullAssertion: bestMcp has this key (set above).
|
|
1405
|
-
out.push(bestMcp.get(e.metadata.identityKey));
|
|
1406
|
-
}
|
|
1407
|
-
else {
|
|
1408
|
-
out.push(e);
|
|
1409
|
-
}
|
|
1410
|
-
if (out.length >= k)
|
|
1411
|
-
break;
|
|
1412
|
-
}
|
|
1413
|
-
return out.slice(0, k);
|
|
1414
|
-
}
|
|
1415
|
-
/** Outcome-precedence rank for step-result dedup (ok/exists > partial > failed). */
|
|
1416
|
-
function rankStatus(s) {
|
|
1417
|
-
return s === 'ok' || s === 'exists'
|
|
1418
|
-
? 3
|
|
1419
|
-
: s === 'partial'
|
|
1420
|
-
? 2
|
|
1421
|
-
: s === 'failed'
|
|
1422
|
-
? 1
|
|
1423
|
-
: 0;
|
|
1424
|
-
}
|
|
1425
|
-
/** True when candidate `a` is a better winner than current `b` for step-result
|
|
1426
|
-
* dedup. Latest-wins by EXECUTION IDENTITY, not by semantic-rank position:
|
|
1427
|
-
* 1. Higher status rank wins; on tie →
|
|
1428
|
-
* 2. Higher attempt wins; on further tie →
|
|
1429
|
-
* 3. Higher writeOrdinal wins (tie-breaks same-timestamp artifacts from one run); on tie →
|
|
1430
|
-
* 4. Later createdAt wins (missing = older: compare with '' as sentinel). */
|
|
1431
|
-
function isBetterStep(a, b) {
|
|
1432
|
-
const ra = rankStatus(a.metadata.status);
|
|
1433
|
-
const rb = rankStatus(b.metadata.status);
|
|
1434
|
-
if (ra !== rb)
|
|
1435
|
-
return ra > rb;
|
|
1436
|
-
const aa = a.metadata.attempt ?? 0;
|
|
1437
|
-
const ba = b.metadata.attempt ?? 0;
|
|
1438
|
-
if (aa !== ba)
|
|
1439
|
-
return aa > ba;
|
|
1440
|
-
const ao = a.metadata.writeOrdinal ?? -1;
|
|
1441
|
-
const bo = b.metadata.writeOrdinal ?? -1;
|
|
1442
|
-
if (ao !== bo)
|
|
1443
|
-
return ao > bo;
|
|
1444
|
-
return (a.metadata.createdAt ?? '') > (b.metadata.createdAt ?? '');
|
|
1445
|
-
}
|
|
1446
|
-
/** True when candidate `a` is a better winner than current `b` for mcp-result
|
|
1447
|
-
* dedup. Latest-fetch wins by writeOrdinal first (handles same-timestamp), then
|
|
1448
|
-
* falls back to createdAt (missing = older). */
|
|
1449
|
-
function isBetterMcp(a, b) {
|
|
1450
|
-
const ao = a.metadata.writeOrdinal ?? -1;
|
|
1451
|
-
const bo = b.metadata.writeOrdinal ?? -1;
|
|
1452
|
-
if (ao !== bo)
|
|
1453
|
-
return ao > bo;
|
|
1454
|
-
return (a.metadata.createdAt ?? '') > (b.metadata.createdAt ?? '');
|
|
1455
|
-
}
|
|
1456
|
-
const MAX_EXTRACT_WINDOWS = 64;
|
|
1457
|
-
/** Return the ≤`maxChars` fragment of `content` most similar to `ref` by EMBEDDING
|
|
1458
|
-
* (NOT ASCII lexical overlap). DIRECT single-pass ranking: every candidate is
|
|
1459
|
-
* scored on its own. The SCORED window IS the RETURNED body: candidates are
|
|
1460
|
-
* `body = maxChars - 2` chars (head+tail '…' reserved up front), so the
|
|
1461
|
-
* highest-scoring fragment is never truncated by the markers. Stride is 50%
|
|
1462
|
-
* overlap, widened to span the whole content within MAX_EXTRACT_WINDOWS windows
|
|
1463
|
-
* (point coverage for content ≤ MAX_EXTRACT_WINDOWS×maxChars; larger thins to
|
|
1464
|
-
* non-overlapping, best-effort). Embeds are SEQUENTIAL and BOUNDED to ≤
|
|
1465
|
-
* MAX_EXTRACT_WINDOWS + 1 — touches NO public embedder API (batch is a deferred
|
|
1466
|
-
* optimization). Result STRICTLY ≤ maxChars; tiny maxChars (< 3) → bare slice.
|
|
1467
|
-
* The `requires` ref is English (planner invariant) → a normal embedder suffices. */
|
|
1468
|
-
export async function relevantExtract(content, ref, maxChars, embedder, options) {
|
|
1469
|
-
if (content.length <= maxChars)
|
|
1470
|
-
return content;
|
|
1471
|
-
if (maxChars < 3)
|
|
1472
|
-
return content.slice(0, Math.max(0, maxChars));
|
|
1473
|
-
const body = maxChars - 2;
|
|
1474
|
-
const stride = Math.max(Math.floor(body / 2), Math.ceil(content.length / MAX_EXTRACT_WINDOWS));
|
|
1475
|
-
const { vector: q } = await embedder.embed(ref, options);
|
|
1476
|
-
let bestStart = 0;
|
|
1477
|
-
let bestScore = Number.NEGATIVE_INFINITY;
|
|
1478
|
-
for (let s = 0; s < content.length; s += stride) {
|
|
1479
|
-
const { vector } = await embedder.embed(content.slice(s, s + body), options);
|
|
1480
|
-
const score = cosine(q, vector);
|
|
1481
|
-
if (score > bestScore) {
|
|
1482
|
-
bestScore = score;
|
|
1483
|
-
bestStart = s;
|
|
1484
|
-
}
|
|
1485
|
-
}
|
|
1486
|
-
const head = bestStart > 0 ? '…' : '';
|
|
1487
|
-
const tail = bestStart + body < content.length ? '…' : '';
|
|
1488
|
-
return head + content.slice(bestStart, bestStart + body) + tail;
|
|
1489
|
-
}
|
|
1490
1356
|
//# sourceMappingURL=controller-coordinator-handler.js.map
|