@mcp-abap-adt/llm-agent-server-libs 20.0.0 → 20.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. package/dist/builders/controller-skill-pipeline-builder.d.ts +57 -0
  2. package/dist/builders/controller-skill-pipeline-builder.d.ts.map +1 -0
  3. package/dist/builders/controller-skill-pipeline-builder.js +175 -0
  4. package/dist/builders/controller-skill-pipeline-builder.js.map +1 -0
  5. package/dist/factories/controller-factory.d.ts +18 -1
  6. package/dist/factories/controller-factory.d.ts.map +1 -1
  7. package/dist/factories/controller-factory.js +12 -1
  8. package/dist/factories/controller-factory.js.map +1 -1
  9. package/dist/generated/version.d.ts +1 -1
  10. package/dist/generated/version.js +1 -1
  11. package/dist/index.d.ts +1 -0
  12. package/dist/index.d.ts.map +1 -1
  13. package/dist/index.js +1 -0
  14. package/dist/index.js.map +1 -1
  15. package/dist/mcp/compose-auxiliary.d.ts +35 -0
  16. package/dist/mcp/compose-auxiliary.d.ts.map +1 -0
  17. package/dist/mcp/compose-auxiliary.js +64 -0
  18. package/dist/mcp/compose-auxiliary.js.map +1 -0
  19. package/dist/pipelines/controller.d.ts.map +1 -1
  20. package/dist/pipelines/controller.js +87 -5
  21. package/dist/pipelines/controller.js.map +1 -1
  22. package/dist/pipelines/coordinator-resolvers.d.ts +68 -0
  23. package/dist/pipelines/coordinator-resolvers.d.ts.map +1 -0
  24. package/dist/pipelines/coordinator-resolvers.js +97 -0
  25. package/dist/pipelines/coordinator-resolvers.js.map +1 -0
  26. package/dist/pipelines/parsers.d.ts +1 -1
  27. package/dist/pipelines/parsers.d.ts.map +1 -1
  28. package/dist/pipelines/parsers.js +5 -4
  29. package/dist/pipelines/parsers.js.map +1 -1
  30. package/dist/pipelines/register-skill-sources.d.ts.map +1 -1
  31. package/dist/pipelines/register-skill-sources.js +5 -0
  32. package/dist/pipelines/register-skill-sources.js.map +1 -1
  33. package/dist/smart-agent/build-stepper-root.d.ts +3 -2
  34. package/dist/smart-agent/build-stepper-root.d.ts.map +1 -1
  35. package/dist/smart-agent/build-stepper-root.js +2 -1
  36. package/dist/smart-agent/build-stepper-root.js.map +1 -1
  37. package/dist/smart-agent/config-reload-watcher.d.ts +30 -0
  38. package/dist/smart-agent/config-reload-watcher.d.ts.map +1 -0
  39. package/dist/smart-agent/config-reload-watcher.js +101 -0
  40. package/dist/smart-agent/config-reload-watcher.js.map +1 -0
  41. package/dist/smart-agent/config-validator.d.ts +21 -0
  42. package/dist/smart-agent/config-validator.d.ts.map +1 -0
  43. package/dist/smart-agent/config-validator.js +196 -0
  44. package/dist/smart-agent/config-validator.js.map +1 -0
  45. package/dist/smart-agent/config.d.ts +16 -254
  46. package/dist/smart-agent/config.d.ts.map +1 -1
  47. package/dist/smart-agent/config.js +17 -904
  48. package/dist/smart-agent/config.js.map +1 -1
  49. package/dist/smart-agent/controller/board.d.ts +6 -3
  50. package/dist/smart-agent/controller/board.d.ts.map +1 -1
  51. package/dist/smart-agent/controller/board.js +21 -0
  52. package/dist/smart-agent/controller/board.js.map +1 -1
  53. package/dist/smart-agent/controller/controller-coordinator-handler.d.ts +24 -50
  54. package/dist/smart-agent/controller/controller-coordinator-handler.d.ts.map +1 -1
  55. package/dist/smart-agent/controller/controller-coordinator-handler.js +542 -676
  56. package/dist/smart-agent/controller/controller-coordinator-handler.js.map +1 -1
  57. package/dist/smart-agent/controller/default-step-execution-control.d.ts +7 -0
  58. package/dist/smart-agent/controller/default-step-execution-control.d.ts.map +1 -0
  59. package/dist/smart-agent/controller/default-step-execution-control.js +36 -0
  60. package/dist/smart-agent/controller/default-step-execution-control.js.map +1 -0
  61. package/dist/smart-agent/controller/finalizer.d.ts +5 -0
  62. package/dist/smart-agent/controller/finalizer.d.ts.map +1 -1
  63. package/dist/smart-agent/controller/finalizer.js +12 -5
  64. package/dist/smart-agent/controller/finalizer.js.map +1 -1
  65. package/dist/smart-agent/controller/noop-run-execution-control.d.ts +6 -0
  66. package/dist/smart-agent/controller/noop-run-execution-control.d.ts.map +1 -0
  67. package/dist/smart-agent/controller/noop-run-execution-control.js +14 -0
  68. package/dist/smart-agent/controller/noop-run-execution-control.js.map +1 -0
  69. package/dist/smart-agent/controller/parser.d.ts +11 -0
  70. package/dist/smart-agent/controller/parser.d.ts.map +1 -0
  71. package/dist/smart-agent/controller/parser.js +77 -0
  72. package/dist/smart-agent/controller/parser.js.map +1 -0
  73. package/dist/smart-agent/controller/planner.js +1 -1
  74. package/dist/smart-agent/controller/planner.js.map +1 -1
  75. package/dist/smart-agent/controller/recall.d.ts +48 -0
  76. package/dist/smart-agent/controller/recall.d.ts.map +1 -0
  77. package/dist/smart-agent/controller/recall.js +196 -0
  78. package/dist/smart-agent/controller/recall.js.map +1 -0
  79. package/dist/smart-agent/controller/reviewer.d.ts.map +1 -1
  80. package/dist/smart-agent/controller/reviewer.js +1 -1
  81. package/dist/smart-agent/controller/reviewer.js.map +1 -1
  82. package/dist/smart-agent/controller/subagent-client.d.ts +2 -2
  83. package/dist/smart-agent/controller/subagent-client.d.ts.map +1 -1
  84. package/dist/smart-agent/controller/subagent-client.js +2 -2
  85. package/dist/smart-agent/controller/subagent-client.js.map +1 -1
  86. package/dist/smart-agent/controller/types.d.ts +16 -4
  87. package/dist/smart-agent/controller/types.d.ts.map +1 -1
  88. package/dist/smart-agent/controller/types.js.map +1 -1
  89. package/dist/smart-agent/controller/usage-logging.d.ts +16 -0
  90. package/dist/smart-agent/controller/usage-logging.d.ts.map +1 -0
  91. package/dist/smart-agent/controller/usage-logging.js +39 -0
  92. package/dist/smart-agent/controller/usage-logging.js.map +1 -0
  93. package/dist/smart-agent/http/adapter-route-handler.d.ts +13 -0
  94. package/dist/smart-agent/http/adapter-route-handler.d.ts.map +1 -0
  95. package/dist/smart-agent/http/adapter-route-handler.js +76 -0
  96. package/dist/smart-agent/http/adapter-route-handler.js.map +1 -0
  97. package/dist/smart-agent/http/chat-route-handler.d.ts +15 -0
  98. package/dist/smart-agent/http/chat-route-handler.d.ts.map +1 -0
  99. package/dist/smart-agent/http/chat-route-handler.js +327 -0
  100. package/dist/smart-agent/http/chat-route-handler.js.map +1 -0
  101. package/dist/smart-agent/http/config-route-handler.d.ts +21 -0
  102. package/dist/smart-agent/http/config-route-handler.d.ts.map +1 -0
  103. package/dist/smart-agent/http/config-route-handler.js +149 -0
  104. package/dist/smart-agent/http/config-route-handler.js.map +1 -0
  105. package/dist/smart-agent/http/health-route-handler.d.ts +12 -0
  106. package/dist/smart-agent/http/health-route-handler.d.ts.map +1 -0
  107. package/dist/smart-agent/http/health-route-handler.js +19 -0
  108. package/dist/smart-agent/http/health-route-handler.js.map +1 -0
  109. package/dist/smart-agent/http/models-route-handler.d.ts +17 -0
  110. package/dist/smart-agent/http/models-route-handler.d.ts.map +1 -0
  111. package/dist/smart-agent/http/models-route-handler.js +65 -0
  112. package/dist/smart-agent/http/models-route-handler.js.map +1 -0
  113. package/dist/smart-agent/http/response-helpers.d.ts +31 -0
  114. package/dist/smart-agent/http/response-helpers.d.ts.map +1 -0
  115. package/dist/smart-agent/http/response-helpers.js +59 -0
  116. package/dist/smart-agent/http/response-helpers.js.map +1 -0
  117. package/dist/smart-agent/http/route-table.d.ts +44 -0
  118. package/dist/smart-agent/http/route-table.d.ts.map +1 -0
  119. package/dist/smart-agent/http/route-table.js +35 -0
  120. package/dist/smart-agent/http/route-table.js.map +1 -0
  121. package/dist/smart-agent/http/session-cookie.d.ts +10 -0
  122. package/dist/smart-agent/http/session-cookie.d.ts.map +1 -0
  123. package/dist/smart-agent/http/session-cookie.js +16 -0
  124. package/dist/smart-agent/http/session-cookie.js.map +1 -0
  125. package/dist/smart-agent/http/sessions-route-handler.d.ts +8 -0
  126. package/dist/smart-agent/http/sessions-route-handler.d.ts.map +1 -0
  127. package/dist/smart-agent/http/sessions-route-handler.js +54 -0
  128. package/dist/smart-agent/http/sessions-route-handler.js.map +1 -0
  129. package/dist/smart-agent/http/usage-route-handler.d.ts +12 -0
  130. package/dist/smart-agent/http/usage-route-handler.d.ts.map +1 -0
  131. package/dist/smart-agent/http/usage-route-handler.js +28 -0
  132. package/dist/smart-agent/http/usage-route-handler.js.map +1 -0
  133. package/dist/smart-agent/knowledge/make-knowledge-backend.d.ts +13 -0
  134. package/dist/smart-agent/knowledge/make-knowledge-backend.d.ts.map +1 -0
  135. package/dist/smart-agent/knowledge/make-knowledge-backend.js +18 -0
  136. package/dist/smart-agent/knowledge/make-knowledge-backend.js.map +1 -0
  137. package/dist/smart-agent/llm/role-llm-resolver.d.ts +30 -0
  138. package/dist/smart-agent/llm/role-llm-resolver.d.ts.map +1 -0
  139. package/dist/smart-agent/llm/role-llm-resolver.js +44 -0
  140. package/dist/smart-agent/llm/role-llm-resolver.js.map +1 -0
  141. package/dist/smart-agent/llm-config-map.d.ts +38 -0
  142. package/dist/smart-agent/llm-config-map.d.ts.map +1 -0
  143. package/dist/smart-agent/llm-config-map.js +75 -0
  144. package/dist/smart-agent/llm-config-map.js.map +1 -0
  145. package/dist/smart-agent/mcp-readiness-monitor.d.ts +38 -0
  146. package/dist/smart-agent/mcp-readiness-monitor.d.ts.map +1 -0
  147. package/dist/smart-agent/mcp-readiness-monitor.js +81 -0
  148. package/dist/smart-agent/mcp-readiness-monitor.js.map +1 -0
  149. package/dist/smart-agent/mcp-readiness-registry.d.ts +37 -0
  150. package/dist/smart-agent/mcp-readiness-registry.d.ts.map +1 -0
  151. package/dist/smart-agent/mcp-readiness-registry.js +55 -0
  152. package/dist/smart-agent/mcp-readiness-registry.js.map +1 -0
  153. package/dist/smart-agent/resolve-config-sections.d.ts +15 -0
  154. package/dist/smart-agent/resolve-config-sections.d.ts.map +1 -0
  155. package/dist/smart-agent/resolve-config-sections.js +212 -0
  156. package/dist/smart-agent/resolve-config-sections.js.map +1 -0
  157. package/dist/smart-agent/session-lifecycle/index.d.ts +117 -0
  158. package/dist/smart-agent/session-lifecycle/index.d.ts.map +1 -0
  159. package/dist/smart-agent/session-lifecycle/index.js +152 -0
  160. package/dist/smart-agent/session-lifecycle/index.js.map +1 -0
  161. package/dist/smart-agent/skill-plugins-config.d.ts +9 -1
  162. package/dist/smart-agent/skill-plugins-config.d.ts.map +1 -1
  163. package/dist/smart-agent/skill-plugins-config.js +27 -1
  164. package/dist/smart-agent/skill-plugins-config.js.map +1 -1
  165. package/dist/smart-agent/skill-plugins-host-factory.d.ts +14 -2
  166. package/dist/smart-agent/skill-plugins-host-factory.d.ts.map +1 -1
  167. package/dist/smart-agent/skill-plugins-host-factory.js +16 -4
  168. package/dist/smart-agent/skill-plugins-host-factory.js.map +1 -1
  169. package/dist/smart-agent/smart-server.d.ts +169 -222
  170. package/dist/smart-agent/smart-server.d.ts.map +1 -1
  171. package/dist/smart-agent/smart-server.js +569 -1387
  172. package/dist/smart-agent/smart-server.js.map +1 -1
  173. package/dist/smart-agent/stepper-config.d.ts +138 -0
  174. package/dist/smart-agent/stepper-config.d.ts.map +1 -0
  175. package/dist/smart-agent/stepper-config.js +216 -0
  176. package/dist/smart-agent/stepper-config.js.map +1 -0
  177. package/dist/smart-agent/tools-rag-handle.d.ts +10 -0
  178. package/dist/smart-agent/tools-rag-handle.d.ts.map +1 -0
  179. package/dist/smart-agent/tools-rag-handle.js +76 -0
  180. package/dist/smart-agent/tools-rag-handle.js.map +1 -0
  181. package/dist/smart-agent/workers/worker-registry.d.ts +159 -0
  182. package/dist/smart-agent/workers/worker-registry.d.ts.map +1 -0
  183. package/dist/smart-agent/workers/worker-registry.js +203 -0
  184. package/dist/smart-agent/workers/worker-registry.js.map +1 -0
  185. package/dist/smart-agent/yaml-loader.d.ts +7 -0
  186. package/dist/smart-agent/yaml-loader.d.ts.map +1 -0
  187. package/dist/smart-agent/yaml-loader.js +146 -0
  188. package/dist/smart-agent/yaml-loader.js.map +1 -0
  189. package/package.json +7 -7
@@ -1,55 +1,34 @@
1
- import { externalToolCallId, } from '@mcp-abap-adt/llm-agent';
2
- import { summaryToUsage, } from '@mcp-abap-adt/llm-agent-libs';
3
- import { cosine } from '../embedder-knowledge-index.js';
4
- import { readClaims, readPlanDecisions, writePlanDecision, } from './artifacts.js';
5
- import { BoardOverBudgetError, reconstructBoard, renderBoard, } from './board.js';
1
+ import { externalToolCallId, McpError, } from '@mcp-abap-adt/llm-agent';
2
+ import { LegacyAccumulateContextStrategy, LegacyTranscriptContextStrategy, summaryToUsage, } from '@mcp-abap-adt/llm-agent-libs';
3
+ import { writePlanDecision } from './artifacts.js';
4
+ import { BoardOverBudgetError, renderLiveBoard, } from './board.js';
5
+ import { DefaultStepExecutionControl } from './default-step-execution-control.js';
6
6
  import { writeArtifact } from './memorizer.js';
7
7
  import { resolveByPrecedence } from './outcome.js';
8
8
  import { makeControllerPlanner } from './planner.js';
9
9
  import { appendHint } from './prompts.js';
10
+ import { buildRecallBlock, collectApproved, RECALL_ARTIFACT_TYPES, RECALL_EVIDENCE_CHARS, RECALL_K_STEP, RECALL_MAX_CHARS_STEP, relevantExtract, runScopedRecall, } from './recall.js';
10
11
  import { classifyRequest, readTerminal, writeTerminal } from './run-scope.js';
11
12
  import { hydrateBundle, persistBundle, resetRun } from './session-bundle.js';
12
13
  import { establishTargetState } from './target-state.js';
13
- import { validateRequires, } from './types.js';
14
+ import { makeLogUsage } from './usage-logging.js';
14
15
  // ---------------------------------------------------------------------------
15
16
  // Debug logging — gated behind DEBUG_CONTROLLER (e.g. DEBUG_CONTROLLER=1).
16
17
  // Surfaces the steps the planner delegates and per-role/total token usage to
17
18
  // stderr, for tuning step granularity and watching token spend. Off by default.
19
+ // (Also in usage-logging.ts — intentional small duplication; no 3rd copy exists.)
18
20
  // ---------------------------------------------------------------------------
19
21
  function dlog(msg) {
20
22
  if (process.env.DEBUG_CONTROLLER)
21
23
  console.error(`[controller] ${msg}`);
22
24
  }
23
- /**
24
- * Build a request-time `logUsage(role, usage)` that writes each subagent call
25
- * into the per-request `IRequestLogger` (the single aggregator), attributing the
26
- * role's configured model. The role is explicit at the call site, so the shared
27
- * planner/finalizer client is attributed correctly. `durationMs: 0` — the seam
28
- * carries no timing (matches the rag-query precedent).
29
- */
30
- export function makeLogUsage(requestLogger, requestId, models) {
31
- return (role, u) => {
32
- if (!u)
33
- return;
34
- const model = role === 'finalizer'
35
- ? (models.finalizer ?? models.planner)
36
- : role === 'reviewer'
37
- ? (models.reviewer ?? models.planner)
38
- : role === 'embedding'
39
- ? 'embedder'
40
- : (models[role] ?? 'unknown');
41
- requestLogger.logLlmCall({
42
- component: role,
43
- model,
44
- promptTokens: u.promptTokens ?? 0,
45
- completionTokens: u.completionTokens ?? 0,
46
- totalTokens: u.totalTokens ?? 0,
47
- durationMs: 0,
48
- requestId,
49
- });
50
- dlog(`tokens ${role}: prompt=${u.promptTokens} completion=${u.completionTokens} total=${u.totalTokens}`);
51
- };
52
- }
25
+ // ---------------------------------------------------------------------------
26
+ // Re-exported for import-path stability (helpers moved to sibling modules).
27
+ // ---------------------------------------------------------------------------
28
+ export { renderLiveBoard } from './board.js';
29
+ export { parseNextStep } from './parser.js';
30
+ export { relevantExtract, runScopedRecall } from './recall.js';
31
+ export { makeLogUsage } from './usage-logging.js';
53
32
  // ---------------------------------------------------------------------------
54
33
  // Handler
55
34
  // ---------------------------------------------------------------------------
@@ -585,183 +564,222 @@ export class ControllerCoordinatorHandler {
585
564
  const cfg = deps.config.budgets;
586
565
  const maxToolCalls = cfg.maxToolCalls ?? 10;
587
566
  const inFlight = bundle.inFlightStep; // set by the caller (block A or B)
588
- // Persist the step outcome ATOMICALLY: record lastOutcome (durable, so a
589
- // resume after a failed step replans instead of repeating it) AND advance the
590
- // planner cursor (onCommit) in the SAME persistBundle that records the step
591
- // result — never in a separate write, so a crash cannot replay a completed step.
592
- const settle = async (outcome) => {
593
- bundle.lastOutcome = outcome;
594
- onCommit?.(outcome);
595
- if (outcome === 'advanced' || outcome === 'partial') {
596
- bundle.nextSeq = (bundle.nextSeq ?? 0) + 1;
597
- bundle.inFlightStep = undefined;
598
- bundle.runPhase = 'planning';
567
+ // Per-step execution control: a wall-clock time budget (perStepTimeoutMs) +
568
+ // the prospective maxToolCalls gate, consumer-swappable via deps. Its signal
569
+ // is merged with the caller's cancel signal (below) and fed into both the
570
+ // executor.send and callMcp so a non-converging / hung step is CUT rather than
571
+ // livelocking. Absent perStepTimeoutMs → time never fires → count-only bound.
572
+ const stepControl = deps.stepExecutionControl ?? new DefaultStepExecutionControl();
573
+ const budget = stepControl.beginStep({
574
+ stepName: step.name,
575
+ seq: inFlight?.seq ?? 0,
576
+ attempt: inFlight?.attempt ?? 0,
577
+ budgets: { maxToolCalls, perStepTimeoutMs: cfg.perStepTimeoutMs },
578
+ });
579
+ // The budget owns a wall-clock timer that is NOT unref'd — it MUST be disposed
580
+ // on every step exit. Open the try IMMEDIATELY after beginStep so the
581
+ // budget-dependent pre-loop (recall / evidence embed / selectTools /
582
+ // strategy.record) is inside the SAME try…finally; a throw from any of those
583
+ // documented-fallible awaits (e.g. an embedder 429) would otherwise leak the
584
+ // timer and later fire controller.abort() on an orphaned signal.
585
+ try {
586
+ const stepStartedAt = Date.now();
587
+ // Merge the caller's request/cancel signal with the step budget: an inner call
588
+ // is cancelled by EITHER. The step-timeout DISCRIMINATOR remains
589
+ // budget.signal.aborted SPECIFICALLY, so a pure caller-cancel is NOT mis-mapped
590
+ // to a step-timeout control-failure.
591
+ const callSignal = ctx.options?.signal
592
+ ? AbortSignal.any([ctx.options.signal, budget.signal])
593
+ : budget.signal;
594
+ // Typed reason code (StepControlDecision.reason) → human note. Preserves
595
+ // today's exact wording so existing suites stay byte-identical.
596
+ const noteFor = (r) => r === 'maxToolCalls'
597
+ ? 'tool-call budget exhausted (maxToolCalls)'
598
+ : r === 'step-timeout'
599
+ ? 'step time budget exhausted (step-timeout)'
600
+ : r;
601
+ let roundNo = 0;
602
+ const state = () => ({
603
+ round: roundNo,
604
+ toolCallCount: inFlight?.toolCallCount ?? 0,
605
+ elapsedMs: Date.now() - stepStartedAt,
606
+ });
607
+ // Persist the step outcome ATOMICALLY: record lastOutcome (durable, so a
608
+ // resume after a failed step replans instead of repeating it) AND advance the
609
+ // planner cursor (onCommit) in the SAME persistBundle that records the step
610
+ // result — never in a separate write, so a crash cannot replay a completed step.
611
+ const settle = async (outcome) => {
612
+ bundle.lastOutcome = outcome;
613
+ onCommit?.(outcome);
614
+ if (outcome === 'advanced' || outcome === 'partial') {
615
+ bundle.nextSeq = (bundle.nextSeq ?? 0) + 1;
616
+ bundle.inFlightStep = undefined;
617
+ bundle.runPhase = 'planning';
618
+ }
619
+ else {
620
+ // 'failed' — keep the same seq, mark awaiting-replan in the SAME persist so
621
+ // recovery routes by durable phase.
622
+ if (bundle.inFlightStep)
623
+ bundle.inFlightStep.phase = 'awaiting-replan';
624
+ bundle.runPhase = 'executing';
625
+ }
626
+ await persistBundle(deps.backend, sessionId, bundle);
627
+ return outcome;
628
+ };
629
+ // The IMMUTABLE per-round prefix: system + step user message + the step-result
630
+ // recall block. Re-emitted verbatim every round via strategy.form(); the dynamic
631
+ // tool rounds are owned by the injected context strategy, NOT accumulated here.
632
+ const staticPrefix = [
633
+ {
634
+ role: 'system',
635
+ content: appendHint(EXECUTOR_SYSTEM, deps.config.subagents.executor?.hint),
636
+ },
637
+ {
638
+ role: 'user',
639
+ content: `Goal: ${bundle.goal}\nStep: ${step.name}\nInstructions: ${step.instructions}`,
640
+ },
641
+ ];
642
+ // Episodic recall: pull prior STEP-RESULT artifacts relevant to this step from
643
+ // session-memory and inject them as static context. The session-memory rag shares
644
+ // the bundle backend, so restrict to 'step-result' (excludes the
645
+ // 'controller-bundle' infrastructure record). Bounded by k and length. The
646
+ // per-round MCP context is now the context strategy's job (its form() supplies
647
+ // the mcp-result rounds — the Window keeps its own buffer, RagRecall recalls),
648
+ // so it is NOT part of the handler-built static prefix.
649
+ const recallText = step.instructions || step.name;
650
+ const maxAttempts = cfg.maxStepAttempts ?? 5;
651
+ const recalledSteps = await runScopedRecall(rag, recallText, RECALL_K_STEP, bundle.runId, RECALL_K_STEP * (maxAttempts + 1), ['step-result'], ctx.options);
652
+ const stepBlock = buildRecallBlock(recalledSteps, RECALL_MAX_CHARS_STEP);
653
+ if (stepBlock) {
654
+ staticPrefix.push({ role: 'user', content: stepBlock });
655
+ }
656
+ // Per-step tool-loop context strategy (record/form). Absent factory →
657
+ // LegacyAccumulateContextStrategy (byte-identical to the historical growing
658
+ // transcript).
659
+ const makeStrategy = () => (deps.toolLoopContextStrategyFactory ??
660
+ (() => new LegacyAccumulateContextStrategy()))({
661
+ run: { rag, runId: bundle.runId, meta, stepName: step.name },
662
+ });
663
+ // Resume / migration selection (Task 12). A step that suspended under the new
664
+ // design carries a serialized `contextStrategyState` → RESTORE it so the
665
+ // pre-suspend rounds (including any INTERNAL tool rounds before an external
666
+ // suspend) come back exactly as the executor last saw them. A PRE-RELEASE
667
+ // in-flight step carries only a raw `transcript` (no snapshot) → migrate it
668
+ // verbatim via the migration-only LegacyTranscriptContextStrategy (one release).
669
+ // Otherwise a fresh step.
670
+ let strategy;
671
+ if (inFlight?.contextStrategyState !== undefined) {
672
+ const state = inFlight.contextStrategyState;
673
+ // A migrated step persisted a LegacyTranscript snapshot ({rawMessages,
674
+ // newRounds}). Restore it through the SAME strategy type so its raw history
675
+ // + post-migration rounds survive a SECOND resume; a normal snapshot
676
+ // restores via the injected/default strategy. Discriminate on shape.
677
+ strategy =
678
+ state.rawMessages !== undefined
679
+ ? new LegacyTranscriptContextStrategy({ rawMessages: [] })
680
+ : makeStrategy();
681
+ strategy.restore(state);
682
+ }
683
+ else if (inFlight?.transcript?.length) {
684
+ strategy = new LegacyTranscriptContextStrategy({
685
+ rawMessages: inFlight.transcript,
686
+ });
687
+ // Adopted verbatim (rawMessages is copied) → clear the durable transcript so
688
+ // it only ever carries UN-recorded external-continuation pairs from here on.
689
+ // The next suspend snapshots this strategy → subsequent resumes take the
690
+ // restore branch above, so the raw history is never re-injected (no double).
691
+ inFlight.transcript.length = 0;
599
692
  }
600
693
  else {
601
- // 'failed' — keep the same seq, mark awaiting-replan in the SAME persist so
602
- // recovery routes by durable phase.
603
- if (bundle.inFlightStep)
604
- bundle.inFlightStep.phase = 'awaiting-replan';
605
- bundle.runPhase = 'executing';
694
+ strategy = makeStrategy(); // fresh step
606
695
  }
607
- await persistBundle(deps.backend, sessionId, bundle);
608
- return outcome;
609
- };
610
- const messages = [
611
- {
612
- role: 'system',
613
- content: appendHint(EXECUTOR_SYSTEM, deps.config.subagents.executor?.hint),
614
- },
615
- {
616
- role: 'user',
617
- content: `Goal: ${bundle.goal}\nStep: ${step.name}\nInstructions: ${step.instructions}`,
618
- },
619
- ];
620
- // Episodic recall: pull prior artifacts relevant to this step from
621
- // session-memory and inject them as context. The session-memory rag shares
622
- // the bundle backend, so restrict to artifact types (excludes the
623
- // 'controller-bundle' infrastructure record). Bounded by k and length.
624
- const recallText = step.instructions || step.name;
625
- const maxAttempts = cfg.maxStepAttempts ?? 5;
626
- const maxTool = cfg.maxToolCalls ?? 10;
627
- // Per-kind run-scoped recall with GUARANTEED over-fetch bounds: step-result
628
- // retries are bounded by maxStepAttempts (k×(maxStepAttempts+1)); mcp-results
629
- // are NOT deduped on re-fetch here and toolCallCount RESETS per attempt, so one
630
- // run can emit up to maxSteps × maxStepAttempts × maxToolCalls of them — over-
631
- // fetch that full run bound so every distinct identityKey is seen before the cap.
632
- const mcpBound = cfg.maxSteps * maxAttempts * maxTool;
633
- const recalledSteps = await runScopedRecall(rag, recallText, RECALL_K_STEP, bundle.runId, RECALL_K_STEP * (maxAttempts + 1), ['step-result'], ctx.options);
634
- const recalledMcp = await runScopedRecall(rag, recallText, RECALL_K_MCP, bundle.runId, mcpBound, ['mcp-result'], ctx.options);
635
- // SEPARATE character budgets per kind: a single huge step-result cannot consume
636
- // the whole budget and starve the MCP context (and vice-versa).
637
- const stepBlock = buildRecallBlock(recalledSteps, RECALL_MAX_CHARS_STEP);
638
- const mcpBlock = buildRecallBlock(recalledMcp, RECALL_MAX_CHARS_MCP);
639
- const recallBlock = [stepBlock, mcpBlock].filter(Boolean).join('\n\n');
640
- if (recallBlock) {
641
- messages.push({ role: 'user', content: recallBlock });
642
- }
643
- // Durable transcript = static prefix (system/user/recall) + the dynamic
644
- // executor/tool turns. On a resume/continuation the dynamic tail is rebuilt
645
- // from inFlightStep.transcript so the executor sees the FULL exchange it had
646
- // (prior tool rounds + the injected external result), not just a fragment.
647
- const staticLen = messages.length;
648
- if (inFlight && inFlight.transcript.length > 0) {
649
- messages.push(...inFlight.transcript);
650
- }
651
- // Persist the dynamic tail after every executor/tool exchange so a suspend or
652
- // crash never rebuilds with a shorter conversation than the executor saw.
653
- const syncTranscript = async () => {
696
+ // Durable, bounded control-message tail (retries only). Aliased IN PLACE so
697
+ // push/prune mutate the persisted field; a legacy call with no inFlightStep gets
698
+ // an ephemeral local (no durable tail to persist).
699
+ let controlTail;
654
700
  if (inFlight) {
655
- inFlight.transcript = messages.slice(staticLen);
656
- await persistBundle(deps.backend, sessionId, bundle);
701
+ inFlight.controlTail = inFlight.controlTail ?? [];
702
+ controlTail = inFlight.controlTail;
657
703
  }
658
- };
659
- // Per-reference evidence: one recall per requires[] reference. A non-empty
660
- // top-K does NOT prove the dependency is present — semantic recall returns the
661
- // NEAREST artifact even at low relevance — so we hand the reviewer the TOP
662
- // artifact's relevant fragment (Evidence.topArtifact) and let IT (the judging
663
- // role) decide whether the ref is actually satisfied. `hit` is a coarse
664
- // any-candidate flag. Gathered SEQUENTIALLY (NOT Promise.all): each
665
- // relevantExtract is itself bounded-sequential, so the outer sequential loop
666
- // keeps at most ONE embed request in flight at a time (rate-limit-safe).
667
- const refs = step.requires && step.requires.length > 0 ? step.requires : [recallText];
668
- const evBound = RECALL_K_STEP * (maxAttempts + 1) +
669
- cfg.maxSteps * maxAttempts * (cfg.maxToolCalls ?? 10);
670
- const evidence = [];
671
- for (const ref of refs) {
672
- const hits = await runScopedRecall(rag, ref, 1, bundle.runId, evBound, RECALL_ARTIFACT_TYPES, ctx.options);
673
- const topArtifact = hits[0]
674
- ? await relevantExtract(hits[0].content, ref, RECALL_EVIDENCE_CHARS,
675
- // biome-ignore lint/style/noNonNullAssertion: distance strategies require an embedder; the factory enforces it (Task 17).
676
- deps.embedder, ctx.options)
677
- : undefined;
678
- evidence.push({ ref, hit: hits.length > 0, topArtifact });
679
- }
680
- // Tools offered to the executor = the INTERNAL (MCP) tools semantically
681
- // relevant to THIS step (top-K from toolsRag) PLUS the per-request external
682
- // (consumer-supplied) tools. The executor decides which to call; internal
683
- // calls route through `callMcp`, external calls round-trip via `isExternalTool`.
684
- const relevant = await deps.selectTools(step.instructions || step.name, TOOL_SELECT_K, ctx.options);
685
- const offeredTools = [...relevant, ...(ctx.externalTools ?? [])];
686
- // The executor may ONLY call a tool that was offered to it: an internal tool
687
- // selected for this step, or a per-request external tool. Any other name
688
- // (hallucinated / stale / not in the top-K) is rejected — never executed —
689
- // so the semantic exposure boundary actually bounds what runs.
690
- const offeredInternalNames = new Set(relevant.map((t) => t.name));
691
- let retries = 0;
692
- // (D) Persist a 'failed' step-result artifact for controller-level failures
693
- // (reviewer unverifiable, executor error exhausted, maxToolCalls, unavailable
694
- // tool) so the board can project the step's terminal state from artifacts alone.
695
- const writeControlFailure = async (reason) => {
696
- const seq = bundle.inFlightStep?.seq ?? bundle.nextSeq ?? 0;
697
- const attempt = bundle.inFlightStep?.attempt ?? 0;
698
- bundle.writeOrdinal = (bundle.writeOrdinal ?? 0) + 1;
699
- await writeArtifact(rag, {
700
- ...meta,
701
- artifactType: 'step-result',
702
- task: step.name,
703
- runId: bundle.runId,
704
- seq,
705
- attempt,
706
- status: 'failed',
707
- note: reason,
708
- remainder: '',
709
- stepId: step.stepId,
710
- digest: reason.slice(0, cfg.maxDigestChars ?? 500),
711
- writeOrdinal: bundle.writeOrdinal,
712
- content: '',
713
- }, ctx.options);
714
- };
715
- // Inner loop handles tool routing / error retries until the executor
716
- // produces content for this step (or the step suspends on an external tool).
717
- while (true) {
718
- const res = await deps.executor.send(messages, offeredTools);
719
- logUsage?.('executor', res.usage);
720
- if (res.kind === 'content') {
721
- // Hold the executor's result; the reviewer (NOT the executor) decides the
722
- // outcome. Default reviewer (no deps.reviewer) approves as 'ok' (legacy).
723
- let review = deps.reviewer
724
- ? await deps.reviewer.review(step, evidence, res.content, {
725
- hint: deps.config.subagents.reviewer?.hint,
726
- logUsage,
727
- maxDigestChars: cfg.maxDigestChars ?? 500,
728
- })
729
- : {
730
- kind: 'outcome',
731
- outcome: {
732
- status: 'ok',
733
- approved: res.content,
734
- remainder: '',
735
- note: '',
736
- digest: res.content.slice(0, cfg.maxDigestChars ?? 500),
737
- },
738
- };
739
- // Judge failure (provider error / malformed / contradictory ok-with-empty)
740
- // is NOT a step failure: re-ask within maxReviewRetries, then ABORT (the
741
- // outcome is unverifiable). Never mapped to settle('failed')/replan.
742
- let reviewRetries = 0;
743
- while (review.kind === 'judge-failure') {
744
- reviewRetries++;
745
- if (reviewRetries > (cfg.maxReviewRetries ?? 2)) {
746
- // The reviewer could not produce a usable verdict within the retry
747
- // budget (provider error / unparsable). DEGRADE to a failed step so the
748
- // planner replans, rather than aborting the whole run — the terminal
749
- // backstop is maxStepAttempts/maxSteps, not a single unverifiable verdict.
750
- bundle.budgets.stepsUsed++;
751
- await writeControlFailure(`reviewer unverifiable after ${cfg.maxReviewRetries ?? 2} retries: ${review.reason}`);
752
- bundle.plannerPrivate += `\n[seq ${bundle.inFlightStep?.seq ?? bundle.nextSeq ?? 0} ${step.name} failed] reviewer unverifiable after ${cfg.maxReviewRetries ?? 2} retries: ${review.reason}`;
753
- return settle('failed');
704
+ else {
705
+ controlTail = [];
706
+ }
707
+ // External CONTINUATION bridge: the resume preamble APPENDS the freshly
708
+ // resolved external assistant/tool pair(s) to inFlight.transcript. Record them
709
+ // as rounds ON TOP of the restored strategy so the executor continues from its
710
+ // own tool call, then CLEAR the transcript — they now live in the strategy, so
711
+ // leaving them would double-record on the next external round-trip. (The
712
+ // LegacyTranscript migration above already adopted AND cleared its transcript,
713
+ // so here the transcript holds ONLY the just-injected external pair — never the
714
+ // migrated raw history; the two never double-inject the same rounds.)
715
+ if (inFlight && inFlight.transcript.length > 0) {
716
+ const t = inFlight.transcript;
717
+ let i = 0;
718
+ while (i < t.length) {
719
+ const assistant = t[i];
720
+ i++;
721
+ const results = [];
722
+ while (i < t.length && t[i]?.role === 'tool') {
723
+ results.push(t[i]);
724
+ i++;
754
725
  }
755
- review = await deps.reviewer.review(step, evidence, res.content, {
756
- hint: deps.config.subagents.reviewer?.hint,
757
- logUsage,
758
- maxDigestChars: cfg.maxDigestChars ?? 500,
759
- });
726
+ if (assistant)
727
+ await strategy.record({ assistant, results }, ctx.options);
760
728
  }
761
- const outcome = review.outcome;
729
+ inFlight.transcript.length = 0;
730
+ }
731
+ // Snapshot the strategy state + persist the bundle after every executor/tool
732
+ // exchange so a suspend or crash resumes with the same context the executor saw.
733
+ const persistExchange = async () => {
734
+ if (inFlight) {
735
+ inFlight.contextStrategyState = strategy.snapshot();
736
+ await persistBundle(deps.backend, sessionId, bundle);
737
+ }
738
+ };
739
+ // Per-reference evidence: one recall per requires[] reference. A non-empty
740
+ // top-K does NOT prove the dependency is present — semantic recall returns the
741
+ // NEAREST artifact even at low relevance — so we hand the reviewer the TOP
742
+ // artifact's relevant fragment (Evidence.topArtifact) and let IT (the judging
743
+ // role) decide whether the ref is actually satisfied. `hit` is a coarse
744
+ // any-candidate flag. Gathered SEQUENTIALLY (NOT Promise.all): each
745
+ // relevantExtract is itself bounded-sequential, so the outer sequential loop
746
+ // keeps at most ONE embed request in flight at a time (rate-limit-safe).
747
+ const refs = step.requires && step.requires.length > 0
748
+ ? step.requires
749
+ : [recallText];
750
+ const evBound = RECALL_K_STEP * (maxAttempts + 1) +
751
+ cfg.maxSteps * maxAttempts * (cfg.maxToolCalls ?? 10);
752
+ const evidence = [];
753
+ for (const ref of refs) {
754
+ const hits = await runScopedRecall(rag, ref, 1, bundle.runId, evBound, RECALL_ARTIFACT_TYPES, ctx.options);
755
+ const topArtifact = hits[0]
756
+ ? await relevantExtract(hits[0].content, ref, RECALL_EVIDENCE_CHARS,
757
+ // biome-ignore lint/style/noNonNullAssertion: distance strategies require an embedder; the factory enforces it (Task 17).
758
+ deps.embedder, ctx.options)
759
+ : undefined;
760
+ evidence.push({ ref, hit: hits.length > 0, topArtifact });
761
+ }
762
+ // Tools offered to the executor = the INTERNAL (MCP) tools semantically
763
+ // relevant to THIS step (top-K from toolsRag) PLUS the per-request external
764
+ // (consumer-supplied) tools. The executor decides which to call; internal
765
+ // calls route through `callMcp`, external calls round-trip via `isExternalTool`.
766
+ const relevant = await deps.selectTools(step.instructions || step.name, TOOL_SELECT_K, ctx.options);
767
+ const offeredTools = [
768
+ ...relevant,
769
+ ...(ctx.externalTools ?? []),
770
+ ];
771
+ // The executor may ONLY call a tool that was offered to it: an internal tool
772
+ // selected for this step, or a per-request external tool. Any other name
773
+ // (hallucinated / stale / not in the top-K) is rejected — never executed —
774
+ // so the semantic exposure boundary actually bounds what runs.
775
+ const offeredInternalNames = new Set(relevant.map((t) => t.name));
776
+ let retries = 0;
777
+ // (D) Persist a 'failed' step-result artifact for controller-level failures
778
+ // (reviewer unverifiable, executor error exhausted, maxToolCalls, unavailable
779
+ // tool) so the board can project the step's terminal state from artifacts alone.
780
+ const writeControlFailure = async (reason) => {
762
781
  const seq = bundle.inFlightStep?.seq ?? bundle.nextSeq ?? 0;
763
782
  const attempt = bundle.inFlightStep?.attempt ?? 0;
764
- // ONE post-review write carrying the COMPLETE Outcome + identity.
765
783
  bundle.writeOrdinal = (bundle.writeOrdinal ?? 0) + 1;
766
784
  await writeArtifact(rag, {
767
785
  ...meta,
@@ -770,174 +788,334 @@ export class ControllerCoordinatorHandler {
770
788
  runId: bundle.runId,
771
789
  seq,
772
790
  attempt,
773
- status: outcome.status,
774
- note: outcome.note,
775
- remainder: outcome.remainder,
791
+ status: 'failed',
792
+ note: reason,
793
+ remainder: '',
776
794
  stepId: step.stepId,
777
- digest: outcome.digest,
795
+ digest: reason.slice(0, cfg.maxDigestChars ?? 500),
778
796
  writeOrdinal: bundle.writeOrdinal,
779
- content: outcome.approved,
797
+ content: '',
780
798
  }, ctx.options);
799
+ };
800
+ // A controller-level (non-reviewer) cut: persist a 'failed' step-result +
801
+ // planner note (preserving today's wording via noteFor) + the TYPED durable
802
+ // ControlFailure, then settle('failed') so the planner replans at this seq.
803
+ const cutControlFailure = async (reason) => {
781
804
  bundle.budgets.stepsUsed++;
782
- const mapped = mapOutcome(outcome.status);
783
- recordStepControl(bundle, {
784
- seq: bundle.inFlightStep?.seq ?? seq,
785
- name: step.name,
786
- status: outcome.status,
787
- note: outcome.note,
788
- remainder: outcome.remainder,
789
- });
790
- return settle(mapped);
791
- }
792
- if (res.kind === 'error') {
793
- retries++;
794
- if (retries <= cfg.maxRetries) {
795
- messages.push({
796
- role: 'user',
797
- content: `The previous attempt failed: ${res.error}. Retry the step.`,
798
- });
799
- await syncTranscript();
800
- continue;
805
+ await writeControlFailure(noteFor(reason));
806
+ bundle.plannerPrivate += `\n[seq ${inFlight?.seq ?? bundle.nextSeq ?? 0} ${step.name} control-failed] ${noteFor(reason)}`;
807
+ if (inFlight) {
808
+ inFlight.phase = 'awaiting-replan';
809
+ const typedReason = reason === 'maxToolCalls' || reason === 'step-timeout'
810
+ ? reason
811
+ : 'control-failure';
812
+ inFlight.controlFailure = {
813
+ reason: typedReason,
814
+ seq: inFlight.seq,
815
+ };
801
816
  }
802
- // Retries exhausted — feed the error back as the step result so the
803
- // planner can replan on the next iteration.
804
- bundle.budgets.stepsUsed++;
805
- await writeControlFailure(`executor error: ${res.error}`);
806
- bundle.plannerPrivate += `\n[step ${step.name} failed] ${res.error}`;
807
817
  return settle('failed');
808
- }
809
- // res.kind === 'tool_call' → route the FIRST tool call.
810
- const firstCall = res.toolCalls[0];
811
- if (firstCall === undefined) {
812
- // Empty tool-call array → treat as an executor error (retry/replan).
813
- retries++;
814
- if (retries <= cfg.maxRetries) {
815
- messages.push({
816
- role: 'user',
817
- content: 'The previous attempt produced an empty tool call. Retry the step.',
818
+ };
819
+ // Inner loop handles tool routing / error retries until the executor
820
+ // produces content for this step (or the step suspends on an external tool).
821
+ // The whole loop is wrapped so the step budget's timer is disposed on EVERY
822
+ // exit (settle / cut / suspend / abort / throw).
823
+ while (true) {
824
+ roundNo++;
825
+ // TIME/abort gate BEFORE each executor round (count is gated per tool call).
826
+ const rc = budget.shouldContinueRound(state());
827
+ if (!rc.continue)
828
+ return cutControlFailure(rc.reason);
829
+ // Form the per-round executor context: the immutable prefix + the strategy's
830
+ // rounds, then the bounded control tail (retries). NEVER a growing raw array.
831
+ const messages = (await strategy.form({ prefix: staticPrefix, queryText: step.instructions }, ctx.options)).concat(controlTail);
832
+ // Merged signal into the executor call. A reject WHILE the step budget is
833
+ // aborted is a step-timeout cut (NOT the executor-error retry); a normal
834
+ // return after the budget fired is ALSO a step-timeout cut.
835
+ let res;
836
+ try {
837
+ res = await deps.executor.send(messages, offeredTools, {
838
+ ...ctx.options,
839
+ signal: callSignal,
818
840
  });
819
- await syncTranscript();
820
- continue;
821
841
  }
822
- bundle.budgets.stepsUsed++;
823
- await writeControlFailure('empty tool call');
824
- bundle.plannerPrivate += `\n[step ${step.name} failed] empty tool call`;
825
- return settle('failed');
826
- }
827
- const call = toLlmToolCall(firstCall);
828
- const name = call.name;
829
- const args = call.arguments;
830
- if (isExternalTool(name)) {
831
- // External round-trips share the SAME durable toolCallCount/maxToolCalls
832
- // bound as internal calls; check BEFORE surfacing so an external tool
833
- // cannot exceed the cap. Exhausted → control-failed replan at the same seq.
834
- if (inFlight && inFlight.toolCallCount + 1 > maxToolCalls) {
842
+ catch (e) {
843
+ if (budget.signal.aborted)
844
+ return cutControlFailure('step-timeout');
845
+ throw e;
846
+ }
847
+ if (budget.signal.aborted)
848
+ return cutControlFailure('step-timeout');
849
+ logUsage?.('executor', res.usage);
850
+ if (res.kind === 'content') {
851
+ // Hold the executor's result; the reviewer (NOT the executor) decides the
852
+ // outcome. Default reviewer (no deps.reviewer) approves as 'ok' (legacy).
853
+ let review = deps.reviewer
854
+ ? await deps.reviewer.review(step, evidence, res.content, {
855
+ hint: deps.config.subagents.reviewer?.hint,
856
+ logUsage,
857
+ maxDigestChars: cfg.maxDigestChars ?? 500,
858
+ })
859
+ : {
860
+ kind: 'outcome',
861
+ outcome: {
862
+ status: 'ok',
863
+ approved: res.content,
864
+ remainder: '',
865
+ note: '',
866
+ digest: res.content.slice(0, cfg.maxDigestChars ?? 500),
867
+ },
868
+ };
869
+ // Judge failure (provider error / malformed / contradictory ok-with-empty)
870
+ // is NOT a step failure: re-ask within maxReviewRetries, then ABORT (the
871
+ // outcome is unverifiable). Never mapped to settle('failed')/replan.
872
+ let reviewRetries = 0;
873
+ while (review.kind === 'judge-failure') {
874
+ reviewRetries++;
875
+ if (reviewRetries > (cfg.maxReviewRetries ?? 2)) {
876
+ // The reviewer could not produce a usable verdict within the retry
877
+ // budget (provider error / unparsable). DEGRADE to a failed step so the
878
+ // planner replans, rather than aborting the whole run — the terminal
879
+ // backstop is maxStepAttempts/maxSteps, not a single unverifiable verdict.
880
+ bundle.budgets.stepsUsed++;
881
+ await writeControlFailure(`reviewer unverifiable after ${cfg.maxReviewRetries ?? 2} retries: ${review.reason}`);
882
+ bundle.plannerPrivate += `\n[seq ${bundle.inFlightStep?.seq ?? bundle.nextSeq ?? 0} ${step.name} failed] reviewer unverifiable after ${cfg.maxReviewRetries ?? 2} retries: ${review.reason}`;
883
+ return settle('failed');
884
+ }
885
+ review = await deps.reviewer.review(step, evidence, res.content, {
886
+ hint: deps.config.subagents.reviewer?.hint,
887
+ logUsage,
888
+ maxDigestChars: cfg.maxDigestChars ?? 500,
889
+ });
890
+ }
891
+ const outcome = review.outcome;
892
+ const seq = bundle.inFlightStep?.seq ?? bundle.nextSeq ?? 0;
893
+ const attempt = bundle.inFlightStep?.attempt ?? 0;
894
+ // ONE post-review write carrying the COMPLETE Outcome + identity.
895
+ bundle.writeOrdinal = (bundle.writeOrdinal ?? 0) + 1;
896
+ await writeArtifact(rag, {
897
+ ...meta,
898
+ artifactType: 'step-result',
899
+ task: step.name,
900
+ runId: bundle.runId,
901
+ seq,
902
+ attempt,
903
+ status: outcome.status,
904
+ note: outcome.note,
905
+ remainder: outcome.remainder,
906
+ stepId: step.stepId,
907
+ digest: outcome.digest,
908
+ writeOrdinal: bundle.writeOrdinal,
909
+ content: outcome.approved,
910
+ }, ctx.options);
835
911
  bundle.budgets.stepsUsed++;
836
- await writeControlFailure('tool-call budget exhausted (maxToolCalls)');
837
- bundle.plannerPrivate += `\n[seq ${inFlight.seq} ${step.name} control-failed] tool-call budget exhausted (maxToolCalls)`;
838
- inFlight.phase = 'awaiting-replan';
839
- inFlight.controlFailure = {
840
- reason: 'maxToolCalls',
841
- seq: inFlight.seq,
912
+ const mapped = mapOutcome(outcome.status);
913
+ recordStepControl(bundle, {
914
+ seq: bundle.inFlightStep?.seq ?? seq,
915
+ name: step.name,
916
+ status: outcome.status,
917
+ note: outcome.note,
918
+ remainder: outcome.remainder,
919
+ });
920
+ return settle(mapped);
921
+ }
922
+ if (res.kind === 'error') {
923
+ retries++;
924
+ if (retries <= cfg.maxRetries) {
925
+ controlTail.push({
926
+ role: 'user',
927
+ content: `The previous attempt failed: ${res.error}. Retry the step.`,
928
+ });
929
+ await persistExchange();
930
+ continue;
931
+ }
932
+ // Retries exhausted — feed the error back as the step result so the
933
+ // planner can replan on the next iteration.
934
+ bundle.budgets.stepsUsed++;
935
+ await writeControlFailure(`executor error: ${res.error}`);
936
+ bundle.plannerPrivate += `\n[step ${step.name} failed] ${res.error}`;
937
+ return settle('failed');
938
+ }
939
+ // res.kind === 'tool_call' → route the FIRST tool call.
940
+ const firstCall = res.toolCalls[0];
941
+ if (firstCall === undefined) {
942
+ // Empty tool-call array → treat as an executor error (retry/replan).
943
+ retries++;
944
+ if (retries <= cfg.maxRetries) {
945
+ controlTail.push({
946
+ role: 'user',
947
+ content: 'The previous attempt produced an empty tool call. Retry the step.',
948
+ });
949
+ await persistExchange();
950
+ continue;
951
+ }
952
+ bundle.budgets.stepsUsed++;
953
+ await writeControlFailure('empty tool call');
954
+ bundle.plannerPrivate += `\n[step ${step.name} failed] empty tool call`;
955
+ return settle('failed');
956
+ }
957
+ // Normalize the StreamToolCall (full or delta) into an LlmToolCall inline.
958
+ const call = 'arguments' in firstCall &&
959
+ typeof firstCall.arguments === 'object' &&
960
+ firstCall.arguments !== null
961
+ ? {
962
+ id: ('id' in firstCall && firstCall.id) || 'call',
963
+ name: ('name' in firstCall && firstCall.name) || '',
964
+ arguments: firstCall.arguments,
965
+ }
966
+ : (() => {
967
+ let iArgs = {};
968
+ const raw = 'arguments' in firstCall ? firstCall.arguments : undefined;
969
+ if (typeof raw === 'string' && raw.length > 0) {
970
+ try {
971
+ iArgs = JSON.parse(raw);
972
+ }
973
+ catch {
974
+ iArgs = {};
975
+ }
976
+ }
977
+ return {
978
+ id: ('id' in firstCall && firstCall.id) || 'call',
979
+ name: ('name' in firstCall && firstCall.name) || '',
980
+ arguments: iArgs,
981
+ };
982
+ })();
983
+ const name = call.name;
984
+ const args = call.arguments;
985
+ if (isExternalTool(name)) {
986
+ // External round-trips share the SAME budget as internal calls; the
987
+ // prospective count gate is consulted BEFORE the increment so an external
988
+ // tool cannot exceed the cap. Cut → control-failed replan at the same seq.
989
+ const g = budget.canExecuteTool(state());
990
+ if (!g.continue)
991
+ return cutControlFailure(g.reason);
992
+ // Snapshot the strategy state SO FAR before we suspend (the resume injection
993
+ // appends the external assistant/tool pair, recorded on the next invocation).
994
+ if (inFlight)
995
+ inFlight.contextStrategyState = strategy.snapshot();
996
+ const extId = externalToolCallId(name, args);
997
+ if (inFlight)
998
+ inFlight.toolCallCount += 1;
999
+ // The new marker REPLACES any prior pending (a fresh extId).
1000
+ bundle.pending = {
1001
+ kind: 'external-tool',
1002
+ extId,
1003
+ toolName: name,
1004
+ args,
1005
+ position: step.name,
842
1006
  };
1007
+ bundle.runState = 'suspended';
1008
+ await persistBundle(deps.backend, sessionId, bundle);
1009
+ this.surfaceToolCall(ctx, { id: extId, name, arguments: args }, usageNow?.());
1010
+ return 'suspended';
1011
+ }
1012
+ // The executor may only call a tool that was OFFERED to it this step. A
1013
+ // name that is neither external nor in the internal top-K (hallucinated /
1014
+ // stale / out-of-scope) is rejected — NOT executed — and fed back as a
1015
+ // tool-not-available error so the executor retries with an offered tool.
1016
+ if (!offeredInternalNames.has(name)) {
1017
+ retries++;
1018
+ if (retries <= cfg.maxRetries) {
1019
+ controlTail.push({
1020
+ role: 'user',
1021
+ content: `Tool "${name}" is not available for this step. Use only the tools provided to you.`,
1022
+ });
1023
+ await persistExchange();
1024
+ continue;
1025
+ }
1026
+ bundle.budgets.stepsUsed++;
1027
+ await writeControlFailure(`requested unavailable tool ${name}`);
1028
+ bundle.plannerPrivate += `\n[step ${step.name} failed] requested unavailable tool ${name}`;
843
1029
  return settle('failed');
844
1030
  }
845
- // Sync the executor turns SO FAR into the durable transcript before we
846
- // suspend (the resume injection appends the external assistant/tool pair).
847
- await syncTranscript();
848
- const extId = externalToolCallId(name, args);
849
- if (inFlight)
1031
+ // Prospective count gate BEFORE the increment (the before-increment model:
1032
+ // the increment happens only after canExecuteTool allows the call).
1033
+ const g = budget.canExecuteTool(state());
1034
+ if (!g.continue)
1035
+ return cutControlFailure(g.reason);
1036
+ // Durable round-trip count: ++ and persist BEFORE surfacing so it survives a
1037
+ // resume (never a per-resume local).
1038
+ if (inFlight) {
850
1039
  inFlight.toolCallCount += 1;
851
- // The new marker REPLACES any prior pending (a fresh extId).
852
- bundle.pending = {
853
- kind: 'external-tool',
854
- extId,
855
- toolName: name,
856
- args,
857
- position: step.name,
858
- };
859
- bundle.runState = 'suspended';
860
- await persistBundle(deps.backend, sessionId, bundle);
861
- this.surfaceToolCall(ctx, { id: extId, name, arguments: args }, usageNow?.());
862
- return 'suspended';
863
- }
864
- // The executor may only call a tool that was OFFERED to it this step. A
865
- // name that is neither external nor in the internal top-K (hallucinated /
866
- // stale / out-of-scope) is rejected — NOT executed — and fed back as a
867
- // tool-not-available error so the executor retries with an offered tool.
868
- if (!offeredInternalNames.has(name)) {
869
- retries++;
870
- if (retries <= cfg.maxRetries) {
871
- messages.push({
872
- role: 'user',
873
- content: `Tool "${name}" is not available for this step. Use only the tools provided to you.`,
874
- });
875
- await syncTranscript();
876
- continue;
1040
+ await persistBundle(deps.backend, sessionId, bundle);
877
1041
  }
878
- bundle.budgets.stepsUsed++;
879
- await writeControlFailure(`requested unavailable tool ${name}`);
880
- bundle.plannerPrivate += `\n[step ${step.name} failed] requested unavailable tool ${name}`;
881
- return settle('failed');
882
- }
883
- // Durable round-trip count: ++ and persist BEFORE surfacing so it survives a
884
- // resume (never a per-resume local).
885
- if (inFlight) {
886
- inFlight.toolCallCount += 1;
887
- await persistBundle(deps.backend, sessionId, bundle);
888
- }
889
- if ((inFlight?.toolCallCount ?? 0) > maxToolCalls) {
890
- // Controller-level failure (NOT a reviewer status): record durably and replan.
891
- bundle.budgets.stepsUsed++;
892
- await writeControlFailure('tool-call budget exhausted (maxToolCalls)');
893
- bundle.plannerPrivate += `\n[seq ${inFlight?.seq ?? bundle.nextSeq ?? 0} ${step.name} control-failed] tool-call budget exhausted (maxToolCalls)`;
894
- if (inFlight) {
895
- inFlight.phase = 'awaiting-replan';
896
- inFlight.controlFailure = {
897
- reason: 'maxToolCalls',
898
- seq: inFlight.seq,
899
- };
1042
+ // Execute locally, memorize, re-send to the executor.
1043
+ // FAIL LOUD: surface an MCP-unavailable failure as a terminal abort (not a
1044
+ // silent empty response). The bridge (buildMcpBridge) throws an McpError
1045
+ // IFF the injected classifier deemed it 'unavailable'; a tool-level error
1046
+ // is returned as TEXT, never thrown. So ANY McpError reaching this catch is
1047
+ // already a classifier-unavailable verdict — trust that throw-contract
1048
+ // rather than re-checking the code (which would drop a CUSTOM classifier's
1049
+ // decision → rethrow → outer catch swallow → (no response)). A non-McpError
1050
+ // is a genuine unexpected error and is re-thrown for the outer handler.
1051
+ let result;
1052
+ try {
1053
+ result = await deps.callMcp(name, args, callSignal);
900
1054
  }
901
- return settle('failed');
902
- }
903
- // Execute locally, memorize, re-send to the executor.
904
- const result = await deps.callMcp(name, args);
905
- bundle.writeOrdinal = (bundle.writeOrdinal ?? 0) + 1;
906
- await writeArtifact(rag, {
907
- ...meta,
908
- artifactType: 'mcp-result',
909
- toolName: name,
910
- task: step.name,
911
- runId: bundle.runId,
912
- seq: inFlight?.seq,
913
- attempt: inFlight?.attempt,
914
- // Stable fetch identity (tool+args) for run-scoped recall dedup.
915
- identityKey: externalToolCallId(name, args),
916
- writeOrdinal: bundle.writeOrdinal,
917
- content: result,
918
- }, ctx.options);
919
- // Feed the result back as a coherent assistant→tool turn (OpenAI protocol)
920
- // so the executor LLM continues from its own tool call rather than seeing a
921
- // bare user message. The assistant message carries the tool_call it made;
922
- // the tool message carries the result keyed by the same id.
923
- messages.push({
924
- role: 'assistant',
925
- content: null,
926
- tool_calls: [
927
- {
928
- id: call.id,
929
- type: 'function',
930
- function: { name, arguments: JSON.stringify(args) },
1055
+ catch (mcpErr) {
1056
+ // A step-timeout cancellation aborts the merged signal → the bridge rejects.
1057
+ // Map that to a step-timeout control-failure BEFORE the McpError escalate so
1058
+ // it is NOT mis-classified as MCP-unavailable (20.4.0 escalate order is
1059
+ // otherwise unchanged).
1060
+ if (budget.signal.aborted)
1061
+ return cutControlFailure('step-timeout');
1062
+ if (mcpErr instanceof McpError) {
1063
+ const now = deps.now ?? (() => new Date().toISOString());
1064
+ const terminalTtlMs = deps.terminalTtlMs ?? 24 * 60 * 60 * 1000;
1065
+ await this.abortTerminal(ctx, sessionId, bundle, `MCP server unavailable: ${mcpErr.message}`, now, terminalTtlMs, usageNow?.());
1066
+ return 'aborted';
1067
+ }
1068
+ throw mcpErr;
1069
+ }
1070
+ // Record this exchange as a coherent assistant→tool ROUND (OpenAI protocol)
1071
+ // via the context strategy so the executor LLM continues from its own tool
1072
+ // call. The strategy owns the per-round context (Window keeps a bounded
1073
+ // buffer; RagRecall (Task 13) persists the mcp-result + recalls it) — the
1074
+ // handler no longer writes the mcp-result artifact or grows a raw transcript.
1075
+ // Durable monotonic write ordinal for this mcp-result write. RagRecall's
1076
+ // run-scoped dedup (isBetterMcp) tie-breaks on writeOrdinal FIRST (then
1077
+ // createdAt); since all mcp-result writes in a step share createdAt, a later
1078
+ // same-identityKey fetch only wins with a strictly-higher ordinal. Increment
1079
+ // BEFORE building the round so the value rides on it into strategy.record.
1080
+ bundle.writeOrdinal = (bundle.writeOrdinal ?? 0) + 1;
1081
+ const round = {
1082
+ assistant: {
1083
+ role: 'assistant',
1084
+ content: null,
1085
+ tool_calls: [
1086
+ {
1087
+ id: call.id,
1088
+ type: 'function',
1089
+ function: { name, arguments: JSON.stringify(args) },
1090
+ },
1091
+ ],
931
1092
  },
932
- ],
933
- });
934
- messages.push({
935
- role: 'tool',
936
- tool_call_id: call.id,
937
- content: result,
938
- });
939
- // The executor saw these turns → make them durable before the next round.
940
- await syncTranscript();
1093
+ results: [
1094
+ {
1095
+ role: 'tool',
1096
+ tool_call_id: call.id,
1097
+ content: result,
1098
+ },
1099
+ ],
1100
+ // Stable fetch identity (tool+args) for run-scoped recall dedup. The
1101
+ // controller has no tool-level error classifier here — an unavailable MCP
1102
+ // server aborts BEFORE record; a returned string is a delivered result.
1103
+ meta: [
1104
+ { identityKey: externalToolCallId(name, args), isError: false },
1105
+ ],
1106
+ ordinal: bundle.writeOrdinal,
1107
+ roundId: undefined,
1108
+ };
1109
+ await strategy.record(round, ctx.options);
1110
+ // A recorded round supersedes any pending control retry → prune the tail.
1111
+ controlTail.length = 0;
1112
+ // The executor saw this round → make the strategy state durable before next.
1113
+ await persistExchange();
1114
+ }
1115
+ }
1116
+ finally {
1117
+ // Dispose the step budget (clears its wall-clock timer) on EVERY exit.
1118
+ budget.dispose();
941
1119
  }
942
1120
  }
943
1121
  // -- Escalation & surfacing (mirror StepperCoordinatorHandler) ----------
@@ -1004,12 +1182,18 @@ export class ControllerCoordinatorHandler {
1004
1182
  await persistBundle(deps.backend, sessionId, bundle);
1005
1183
  let answer;
1006
1184
  if (deps.finalizer && bundle.runId) {
1185
+ // Recall the skills block ONCE (not per finalize retry — re-embedding on
1186
+ // every attempt is wasteful; the recall is invariant across retries).
1187
+ const skillsBlock = deps.skillsRecall
1188
+ ? await deps.skillsRecall(bundle.goal, ctx.options)
1189
+ : undefined;
1007
1190
  while (answer === undefined) {
1008
1191
  try {
1009
1192
  const composed = await deps.finalizer.finalize(bundle.goal, request, approved, {
1010
1193
  hint: deps.config.subagents.finalizer?.hint,
1011
1194
  logUsage,
1012
1195
  log: (m) => dlog(m),
1196
+ skillsBlock,
1013
1197
  });
1014
1198
  // Empty-but-ok finalizer output is a JUDGE failure (spec), not a valid
1015
1199
  // answer → throw so it retries within maxFinalizeRetries.
@@ -1080,7 +1264,7 @@ export class ControllerCoordinatorHandler {
1080
1264
  }
1081
1265
  }
1082
1266
  // ---------------------------------------------------------------------------
1083
- // Episodic recall tuning
1267
+ // Goal-clarification helpers
1084
1268
  // ---------------------------------------------------------------------------
1085
1269
  /** Bare confirmations that, on a goal clarify, commit the evaluator's proposed
1086
1270
  * target rather than the literal answer. Anything else = a refinement. */
@@ -1111,7 +1295,6 @@ function isAffirmation(answer) {
1111
1295
  .replace(/[.!]+$/, '');
1112
1296
  return AFFIRMATIONS.has(t);
1113
1297
  }
1114
- /** Top-K tools surfaced from toolsRag per planner/step query. */
1115
1298
  /** Agnostic executor system prompt. Domain specifics (e.g. SAP/ABAP fact kinds)
1116
1299
  * are layered on via `subagents.executor.hint` (see {@link appendHint}). */
1117
1300
  const EXECUTOR_SYSTEM = 'You are the executor. You have tools that read the live target system. ' +
@@ -1124,44 +1307,11 @@ const EXECUTOR_SYSTEM = 'You are the executor. You have tools that read the live
1124
1307
  'if the step asks for a LIST or overview, return the list; do NOT then go and ' +
1125
1308
  'fetch the full details of every listed item unless the step explicitly asks ' +
1126
1309
  'for per-item details.';
1310
+ /** Top-K tools surfaced from toolsRag per planner/step query. */
1127
1311
  const TOOL_SELECT_K = 20;
1128
- /** Top-k recalled artifacts injected into the executor context per step. */
1129
- /** Artifact types eligible for recall (excludes the 'controller-bundle' record
1130
- * that shares the same backend). */
1131
- const RECALL_ARTIFACT_TYPES = ['step-result', 'mcp-result'];
1132
- /** Per-kind recall counts (distinct artifacts kept after dedup + cap). */
1133
- const RECALL_K_STEP = 4;
1134
- const RECALL_K_MCP = 4;
1135
- /** SEPARATE char budgets per kind, so a huge step-result cannot starve MCP context. */
1136
- const RECALL_MAX_CHARS_STEP = 2000;
1137
- const RECALL_MAX_CHARS_MCP = 2000;
1138
- /** Char budget for a single per-`requires` evidence extract handed to the reviewer. */
1139
- const RECALL_EVIDENCE_CHARS = 800;
1140
1312
  // ---------------------------------------------------------------------------
1141
1313
  // Pure helpers
1142
1314
  // ---------------------------------------------------------------------------
1143
- /** Build a bounded "Relevant prior context" block from recalled artifacts under
1144
- * the given char budget, or undefined when there is nothing to inject. */
1145
- function buildRecallBlock(hits, maxChars) {
1146
- if (hits.length === 0)
1147
- return undefined;
1148
- const parts = [];
1149
- let used = 0;
1150
- for (const h of hits) {
1151
- const c = h.content ?? '';
1152
- if (c.length === 0)
1153
- continue;
1154
- if (used + c.length > maxChars) {
1155
- parts.push(c.slice(0, maxChars - used));
1156
- break;
1157
- }
1158
- parts.push(c);
1159
- used += c.length;
1160
- }
1161
- if (parts.length === 0)
1162
- return undefined;
1163
- return `Relevant prior context:\n${parts.join('\n')}`;
1164
- }
1165
1315
  /** Extract the user prompt from the request's textOrMessages. */
1166
1316
  function extractPrompt(textOrMessages) {
1167
1317
  if (typeof textOrMessages === 'string')
@@ -1172,131 +1322,6 @@ function extractPrompt(textOrMessages) {
1172
1322
  }
1173
1323
  return '';
1174
1324
  }
1175
- /** Parse a planner content string into a NextStep, defensively. */
1176
- /** Parse the planner's reply into a NextStep, tolerating ```json fences and
1177
- * surrounding prose. Returns null when no valid decision can be extracted — the
1178
- * caller treats that as a FORMAT error (re-ask the planner), NOT a rewind, so a
1179
- * badly-formatted reply never silently burns the rewind budget. */
1180
- export function parseNextStep(content) {
1181
- const json = extractJsonObject(content);
1182
- if (json === null)
1183
- return null;
1184
- try {
1185
- const obj = JSON.parse(json);
1186
- if (obj.kind === 'done' && typeof obj.result === 'string')
1187
- return { kind: 'done', result: obj.result };
1188
- if (obj.kind === 'rewind' && typeof obj.reason === 'string')
1189
- return { kind: 'rewind', reason: obj.reason };
1190
- if (obj.kind === 'next' &&
1191
- obj.step &&
1192
- typeof obj.step.name === 'string' &&
1193
- typeof obj.step.instructions === 'string') {
1194
- // Validate requires[] so a non-string / empty / oversized reference never
1195
- // reaches the semantic query / embedder; a malformed value is a parse
1196
- // failure that drives the existing parse-retry.
1197
- const req = validateRequires(obj.step.requires);
1198
- if (req === false)
1199
- return null;
1200
- return {
1201
- kind: 'next',
1202
- step: {
1203
- name: obj.step.name,
1204
- instructions: obj.step.instructions,
1205
- ...(obj.step.type ? { type: obj.step.type } : {}),
1206
- ...(req ? { requires: req } : {}),
1207
- },
1208
- };
1209
- }
1210
- }
1211
- catch {
1212
- // fall through
1213
- }
1214
- return null;
1215
- }
1216
- /** Extract the first balanced JSON object from a planner reply, ignoring ```json
1217
- * fences and prose around it. String-aware (braces inside strings don't count).
1218
- * Returns null if no balanced object is present. */
1219
- export function extractJsonObject(raw) {
1220
- const fence = raw.match(/```(?:json)?\s*([\s\S]*?)```/i);
1221
- const body = fence ? fence[1] : raw;
1222
- const start = body.indexOf('{');
1223
- if (start < 0)
1224
- return null;
1225
- let depth = 0;
1226
- let inStr = false;
1227
- let esc = false;
1228
- for (let i = start; i < body.length; i++) {
1229
- const ch = body[i];
1230
- if (inStr) {
1231
- if (esc)
1232
- esc = false;
1233
- else if (ch === '\\')
1234
- esc = true;
1235
- else if (ch === '"')
1236
- inStr = false;
1237
- continue;
1238
- }
1239
- if (ch === '"')
1240
- inStr = true;
1241
- else if (ch === '{')
1242
- depth++;
1243
- else if (ch === '}') {
1244
- depth--;
1245
- if (depth === 0)
1246
- return body.slice(start, i + 1);
1247
- }
1248
- }
1249
- return null;
1250
- }
1251
- /** Normalize a StreamToolCall (full or delta) into an LlmToolCall. */
1252
- function toLlmToolCall(c) {
1253
- if ('arguments' in c &&
1254
- typeof c.arguments === 'object' &&
1255
- c.arguments !== null) {
1256
- // Full LlmToolCall: arguments is already a parsed object.
1257
- return {
1258
- id: ('id' in c && c.id) || 'call',
1259
- name: ('name' in c && c.name) || '',
1260
- arguments: c.arguments,
1261
- };
1262
- }
1263
- // Delta: arguments is a (possibly partial) JSON string.
1264
- let args = {};
1265
- const raw = 'arguments' in c ? c.arguments : undefined;
1266
- if (typeof raw === 'string' && raw.length > 0) {
1267
- try {
1268
- args = JSON.parse(raw);
1269
- }
1270
- catch {
1271
- args = {};
1272
- }
1273
- }
1274
- return {
1275
- id: ('id' in c && c.id) || 'call',
1276
- name: ('name' in c && c.name) || '',
1277
- arguments: args,
1278
- };
1279
- }
1280
- /** Reconstruct and render the live step-state board from artifacts.
1281
- * Returns '' when there is no runId (the board has nothing to show yet). */
1282
- async function renderLiveBoard(rag, bundle, budget) {
1283
- const runId = bundle.runId;
1284
- if (!runId)
1285
- return '';
1286
- const [structure, claims] = await Promise.all([
1287
- readPlanDecisions(rag, runId),
1288
- readClaims(rag, runId),
1289
- ]);
1290
- const stepResults = await rag.list({ runId, artifactType: 'step-result' });
1291
- const board = reconstructBoard({
1292
- structure,
1293
- stepResults,
1294
- claims,
1295
- inFlight: bundle.inFlightStep,
1296
- pending: bundle.pending,
1297
- });
1298
- return renderBoard(board, budget);
1299
- }
1300
1325
  /** Synthesize the strict KnowledgeEntryMetadata for controller artifacts. */
1301
1326
  function synthMeta(ctx, sessionId) {
1302
1327
  const traceId = ctx.options?.trace?.traceId ?? sessionId;
@@ -1328,163 +1353,4 @@ function recordStepControl(bundle, rec) {
1328
1353
  (rec.note ? ` ${rec.note}` : '') +
1329
1354
  (rec.remainder ? ` remainder: ${rec.remainder}` : '');
1330
1355
  }
1331
- /** Gather the run's approved results, one per seq, resolved by outcome precedence
1332
- * (ok/exists > partial > failed), ordered by seq. Reconstructs the complete
1333
- * Outcome from artifact metadata (status/note/remainder) + content. */
1334
- async function collectApproved(rag, runId) {
1335
- const all = await rag.list({ runId, artifactType: 'step-result' });
1336
- const bySeq = new Map();
1337
- for (const e of all) {
1338
- const seq = e.metadata.seq ?? 0;
1339
- const o = {
1340
- status: (e.metadata.status ?? 'failed'),
1341
- approved: e.content,
1342
- remainder: e.metadata.remainder ?? '',
1343
- note: e.metadata.note ?? '',
1344
- };
1345
- const arr = bySeq.get(seq);
1346
- if (arr)
1347
- arr.push(o);
1348
- else
1349
- bySeq.set(seq, [o]);
1350
- }
1351
- const out = [];
1352
- for (const [seq, outcomes] of [...bySeq.entries()].sort((a, b) => a[0] - b[0])) {
1353
- const resolved = resolveByPrecedence(outcomes);
1354
- if (resolved && resolved.status !== 'failed')
1355
- out.push({ seq, content: resolved.approved });
1356
- }
1357
- return out;
1358
- }
1359
- /** The ONE run-scoped results-RAG recall primitive — used by BOTH the whole-step
1360
- * recall AND the per-`requires` evidence. EMBEDDING-based similarity via the
1361
- * backend's semantic query (NO homemade lexical scoring): the backend embeds the
1362
- * query + ranks by vector similarity, with the `runId` filter applied PRE-cap.
1363
- * Over-fetch `kPrime` (caller-supplied so the duplication bound is justified PER
1364
- * KIND), then dedup and cap to `k`. Dedup: step-results (have `seq`) →
1365
- * precedence-winner per seq; mcp-results → by `identityKey`. Embedding rank order
1366
- * is preserved through the dedup.
1367
- * `options` is forwarded into the embedder so recall-time embeds are metered. */
1368
- export async function runScopedRecall(rag, text, k, runId, kPrime, artifactType, options) {
1369
- const hits = await rag.query(text, {
1370
- k: kPrime,
1371
- filter: { runId, artifactType },
1372
- options,
1373
- });
1374
- const bestStep = new Map();
1375
- const bestMcp = new Map();
1376
- for (const e of hits) {
1377
- if (e.metadata.seq !== undefined && e.metadata.status !== undefined) {
1378
- const prev = bestStep.get(e.metadata.seq);
1379
- if (!prev || isBetterStep(e, prev))
1380
- bestStep.set(e.metadata.seq, e);
1381
- }
1382
- else if (e.metadata.identityKey) {
1383
- const prev = bestMcp.get(e.metadata.identityKey);
1384
- if (!prev || isBetterMcp(e, prev))
1385
- bestMcp.set(e.metadata.identityKey, e);
1386
- }
1387
- }
1388
- // Walk hits in embedding-rank order; emit each (runId,seq) / identityKey once.
1389
- const out = [];
1390
- const seenSeq = new Set();
1391
- const seenMcp = new Set();
1392
- for (const e of hits) {
1393
- if (e.metadata.seq !== undefined && e.metadata.status !== undefined) {
1394
- if (seenSeq.has(e.metadata.seq))
1395
- continue;
1396
- seenSeq.add(e.metadata.seq);
1397
- // biome-ignore lint/style/noNonNullAssertion: bestStep has this seq (set above).
1398
- out.push(bestStep.get(e.metadata.seq));
1399
- }
1400
- else if (e.metadata.identityKey) {
1401
- if (seenMcp.has(e.metadata.identityKey))
1402
- continue;
1403
- seenMcp.add(e.metadata.identityKey);
1404
- // biome-ignore lint/style/noNonNullAssertion: bestMcp has this key (set above).
1405
- out.push(bestMcp.get(e.metadata.identityKey));
1406
- }
1407
- else {
1408
- out.push(e);
1409
- }
1410
- if (out.length >= k)
1411
- break;
1412
- }
1413
- return out.slice(0, k);
1414
- }
1415
- /** Outcome-precedence rank for step-result dedup (ok/exists > partial > failed). */
1416
- function rankStatus(s) {
1417
- return s === 'ok' || s === 'exists'
1418
- ? 3
1419
- : s === 'partial'
1420
- ? 2
1421
- : s === 'failed'
1422
- ? 1
1423
- : 0;
1424
- }
1425
- /** True when candidate `a` is a better winner than current `b` for step-result
1426
- * dedup. Latest-wins by EXECUTION IDENTITY, not by semantic-rank position:
1427
- * 1. Higher status rank wins; on tie →
1428
- * 2. Higher attempt wins; on further tie →
1429
- * 3. Higher writeOrdinal wins (tie-breaks same-timestamp artifacts from one run); on tie →
1430
- * 4. Later createdAt wins (missing = older: compare with '' as sentinel). */
1431
- function isBetterStep(a, b) {
1432
- const ra = rankStatus(a.metadata.status);
1433
- const rb = rankStatus(b.metadata.status);
1434
- if (ra !== rb)
1435
- return ra > rb;
1436
- const aa = a.metadata.attempt ?? 0;
1437
- const ba = b.metadata.attempt ?? 0;
1438
- if (aa !== ba)
1439
- return aa > ba;
1440
- const ao = a.metadata.writeOrdinal ?? -1;
1441
- const bo = b.metadata.writeOrdinal ?? -1;
1442
- if (ao !== bo)
1443
- return ao > bo;
1444
- return (a.metadata.createdAt ?? '') > (b.metadata.createdAt ?? '');
1445
- }
1446
- /** True when candidate `a` is a better winner than current `b` for mcp-result
1447
- * dedup. Latest-fetch wins by writeOrdinal first (handles same-timestamp), then
1448
- * falls back to createdAt (missing = older). */
1449
- function isBetterMcp(a, b) {
1450
- const ao = a.metadata.writeOrdinal ?? -1;
1451
- const bo = b.metadata.writeOrdinal ?? -1;
1452
- if (ao !== bo)
1453
- return ao > bo;
1454
- return (a.metadata.createdAt ?? '') > (b.metadata.createdAt ?? '');
1455
- }
1456
- const MAX_EXTRACT_WINDOWS = 64;
1457
- /** Return the ≤`maxChars` fragment of `content` most similar to `ref` by EMBEDDING
1458
- * (NOT ASCII lexical overlap). DIRECT single-pass ranking: every candidate is
1459
- * scored on its own. The SCORED window IS the RETURNED body: candidates are
1460
- * `body = maxChars - 2` chars (head+tail '…' reserved up front), so the
1461
- * highest-scoring fragment is never truncated by the markers. Stride is 50%
1462
- * overlap, widened to span the whole content within MAX_EXTRACT_WINDOWS windows
1463
- * (point coverage for content ≤ MAX_EXTRACT_WINDOWS×maxChars; larger thins to
1464
- * non-overlapping, best-effort). Embeds are SEQUENTIAL and BOUNDED to ≤
1465
- * MAX_EXTRACT_WINDOWS + 1 — touches NO public embedder API (batch is a deferred
1466
- * optimization). Result STRICTLY ≤ maxChars; tiny maxChars (< 3) → bare slice.
1467
- * The `requires` ref is English (planner invariant) → a normal embedder suffices. */
1468
- export async function relevantExtract(content, ref, maxChars, embedder, options) {
1469
- if (content.length <= maxChars)
1470
- return content;
1471
- if (maxChars < 3)
1472
- return content.slice(0, Math.max(0, maxChars));
1473
- const body = maxChars - 2;
1474
- const stride = Math.max(Math.floor(body / 2), Math.ceil(content.length / MAX_EXTRACT_WINDOWS));
1475
- const { vector: q } = await embedder.embed(ref, options);
1476
- let bestStart = 0;
1477
- let bestScore = Number.NEGATIVE_INFINITY;
1478
- for (let s = 0; s < content.length; s += stride) {
1479
- const { vector } = await embedder.embed(content.slice(s, s + body), options);
1480
- const score = cosine(q, vector);
1481
- if (score > bestScore) {
1482
- bestScore = score;
1483
- bestStart = s;
1484
- }
1485
- }
1486
- const head = bestStart > 0 ? '…' : '';
1487
- const tail = bestStart + body < content.length ? '…' : '';
1488
- return head + content.slice(bestStart, bestStart + body) + tail;
1489
- }
1490
1356
  //# sourceMappingURL=controller-coordinator-handler.js.map