@dotdrelle/wiki-manager 0.11.10 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/README.md +32 -7
  2. package/agents.docker-compose.yml +3 -0
  3. package/docker-compose.yml +1 -0
  4. package/package.json +3 -2
  5. package/src/activity/activityAggregator.js +109 -0
  6. package/src/activity/activityAggregator.test.js +60 -0
  7. package/src/activity/activityDeduplicator.js +50 -0
  8. package/src/activity/progressCalculator.js +61 -0
  9. package/src/activity/runSynthesis.js +15 -0
  10. package/src/agent/graph.js +101 -38
  11. package/src/agent/graph.test.js +75 -5
  12. package/src/cli/wiki-manager.js +22 -0
  13. package/src/contracts/schemas.js +249 -13
  14. package/src/contracts/schemas.test.js +147 -2
  15. package/src/core/activity.js +8 -5
  16. package/src/core/agentEvents.js +260 -23
  17. package/src/core/agentEvents.test.js +53 -0
  18. package/src/core/compose.js +6 -1
  19. package/src/core/dockerCompose.test.js +12 -0
  20. package/src/core/documentIntake.js +33 -38
  21. package/src/core/documentIntake.test.js +25 -0
  22. package/src/core/env.js +11 -1
  23. package/src/core/jobQueue.js +26 -2
  24. package/src/core/mcp.js +30 -3
  25. package/src/core/mcp.test.js +69 -1
  26. package/src/core/plan.js +46 -4
  27. package/src/core/plan.test.js +16 -1
  28. package/src/core/planPatch.js +64 -3
  29. package/src/core/planPatch.test.js +110 -1
  30. package/src/core/queueStore.test.js +21 -0
  31. package/src/core/runtimeLog.js +119 -0
  32. package/src/core/runtimeLog.test.js +84 -0
  33. package/src/core/wikirc.js +22 -0
  34. package/src/core/wikirc.test.js +49 -1
  35. package/src/core/workflow.js +14 -3
  36. package/src/graph/graphAggregator.js +5 -0
  37. package/src/graph/graphPatch.js +8 -0
  38. package/src/graph/graphSnapshot.js +14 -0
  39. package/src/graph/graphVisibilityPolicy.js +40 -0
  40. package/src/graph/runGraphProjector.js +87 -0
  41. package/src/graph/runGraphProjector.test.js +111 -0
  42. package/src/orchestrator/agentRegistry.js +172 -0
  43. package/src/orchestrator/agentRegistry.test.js +114 -0
  44. package/src/orchestrator/approvalPolicy.js +126 -0
  45. package/src/orchestrator/approvalPolicy.test.js +78 -0
  46. package/src/orchestrator/assignmentManager.js +80 -0
  47. package/src/orchestrator/attemptManager.js +168 -0
  48. package/src/orchestrator/attemptManager.test.js +160 -0
  49. package/src/orchestrator/budgetManager.js +127 -0
  50. package/src/orchestrator/capabilityRegistry.js +64 -0
  51. package/src/orchestrator/capabilityRegistry.test.js +59 -0
  52. package/src/orchestrator/capabilityResolver.js +151 -0
  53. package/src/orchestrator/capabilityResolver.test.js +127 -0
  54. package/src/orchestrator/dependencyResolver.js +112 -0
  55. package/src/orchestrator/dispatcher.js +302 -0
  56. package/src/orchestrator/lockManager.js +51 -0
  57. package/src/orchestrator/planIntegrator.js +268 -0
  58. package/src/orchestrator/planIntegrator.test.js +213 -0
  59. package/src/orchestrator/planValidator.js +534 -0
  60. package/src/orchestrator/planValidator.test.js +262 -0
  61. package/src/orchestrator/resultAggregator.js +248 -0
  62. package/src/orchestrator/resultAggregator.test.js +211 -0
  63. package/src/orchestrator/scheduler.js +103 -0
  64. package/src/orchestrator/scheduler.test.js +138 -0
  65. package/src/runtime/approvals.js +130 -1
  66. package/src/runtime/controlMessages.js +43 -0
  67. package/src/runtime/controlMessages.test.js +21 -0
  68. package/src/runtime/donna-contract.test.js +296 -33
  69. package/src/runtime/recoveryManager.js +176 -0
  70. package/src/runtime/recoveryManager.test.js +162 -0
  71. package/src/runtime/runner.e2e.test.js +98 -12
  72. package/src/runtime/runner.js +261 -227
  73. package/src/runtime/runner.test.js +249 -452
  74. package/src/runtime/server.js +142 -24
  75. package/src/runtime/server.test.js +192 -62
  76. package/src/runtime/store.js +837 -2
  77. package/src/runtime/store.test.js +269 -28
  78. package/src/runtime/supervisor.js +33 -1
  79. package/src/runtime/supervisor.test.js +47 -1
  80. package/src/shell/RightPane.tsx +8 -5
  81. package/src/shell/repl.js +7 -11
  82. package/src/shell/repl.test.js +26 -5
  83. package/src/shell/tui.tsx +1 -0
  84. package/src/shell/useAgent.ts +5 -1
  85. package/src/shell/useSession.ts +75 -19
@@ -5,11 +5,11 @@ import {
5
5
  callMcpTool,
6
6
  formatMcpToolResult,
7
7
  formatMcpToolsForAgent,
8
- parseToolCallName,
8
+ resolveToolCallName,
9
9
  } from '../core/mcp.js';
10
10
  import { formatSkillsForAgent, readOptionalText } from '../core/skills.js';
11
11
  import { handleSlashCommand } from '../commands/slash.js';
12
- import { extractActivity, formatActivitySummary, parseJsonText } from '../core/activity.js';
12
+ import { extractActivity, formatActivitySummary, parseJsonText, sessionActivities } from '../core/activity.js';
13
13
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
14
14
  import { enqueueProductionJob, ensureJobQueue, formatQueue, productionLockBusy } from '../core/jobQueue.js';
15
15
  import { updateWorkspaceProfilePreference } from '../core/profile.js';
@@ -18,6 +18,14 @@ const MAX_TOOL_ITERATIONS = 80;
18
18
  const MAX_SPINNER_ARG_LENGTH = 96;
19
19
  const MAX_PROFILE_CHARS = 4000;
20
20
 
21
+ // Pseudo-servers handled directly by the tool executor (not present in
22
+ // session.mcp). Listed so unqualified names like "plan_set" resolve the same
23
+ // way as MCP tools in resolveToolCallName.
24
+ const INTERNAL_TOOL_SERVERS = {
25
+ wiki: ['plan_set', 'plan_done'],
26
+ shell: ['run_command', 'read_command', 'profile_update'],
27
+ };
28
+
21
29
  const AGENT_SLASH_COMMANDS = new Set([
22
30
  'help',
23
31
  'version',
@@ -124,17 +132,16 @@ const WIKI_PLAN_SET_TOOL = {
124
132
  properties: {
125
133
  id: { type: 'string' },
126
134
  description: { type: 'string' },
135
+ requiredCapability: { type: ['string', 'null'] },
127
136
  status: { type: 'string', enum: ['pending', 'queued', 'running', 'waiting', 'pending_approval', 'done', 'failed', 'cancelled', 'stalled', 'added_during_run'] },
128
137
  dependsOn: { type: 'array', items: { type: 'string' } },
129
- executor: { type: ['string', 'null'] },
130
- executorQuery: { type: ['object', 'null'], additionalProperties: true },
131
138
  outputRefs: { type: 'array', items: { type: 'string' } },
132
139
  },
133
140
  required: ['description'],
134
141
  },
135
142
  ],
136
143
  },
137
- description: 'Ordered steps. Backward-compatible strings are accepted; structured steps may include id, dependsOn, executor, executorQuery, outputRefs.',
144
+ description: 'Ordered steps. Backward-compatible strings are accepted; structured steps may include id, requiredCapability, dependsOn, outputRefs.',
138
145
  },
139
146
  },
140
147
  required: ['steps'],
@@ -180,6 +187,7 @@ const AgentState = Annotation.Root({
180
187
  }),
181
188
  toolIterations: Annotation({ default: () => 0 }),
182
189
  pendingToolCalls: Annotation(),
190
+ inputClassification: Annotation(),
183
191
  readyToStream: Annotation(),
184
192
  streamContext: Annotation(),
185
193
  streamedInline: Annotation(),
@@ -460,7 +468,7 @@ function handleWikiTool(session, tool, args) {
460
468
  return `Unknown wiki tool: ${tool}`;
461
469
  }
462
470
 
463
- function normalizeDeclaredPlanStep(raw, index, session) {
471
+ function normalizeDeclaredPlanStep(raw, index) {
464
472
  const item = raw && typeof raw === 'object' && !Array.isArray(raw)
465
473
  ? raw
466
474
  : { description: String(raw) };
@@ -471,8 +479,9 @@ function normalizeDeclaredPlanStep(raw, index, session) {
471
479
  description,
472
480
  status: item.status ?? 'pending',
473
481
  dependsOn: Array.isArray(item.dependsOn) ? item.dependsOn.map(String) : [],
474
- executor: item.executor ?? selectExecutorForStep(description, session),
475
- executorQuery: item.executorQuery ?? null,
482
+ requiredCapability: item.requiredCapability != null ? String(item.requiredCapability) : null,
483
+ executor: null,
484
+ executorQuery: null,
476
485
  outputRefs: Array.isArray(item.outputRefs) ? item.outputRefs.map(String) : [],
477
486
  };
478
487
  }
@@ -488,23 +497,6 @@ function slugStepId(description, index) {
488
497
  return slug || `task-${index + 1}`;
489
498
  }
490
499
 
491
- function selectExecutorForStep(description, session) {
492
- const text = String(description ?? '').toLowerCase();
493
- let fallback = null;
494
- for (const [serverName, value] of Object.entries(session.mcp ?? {})) {
495
- if (value.status !== 'connected') continue;
496
- for (const tool of value.tools ?? []) {
497
- const executor = `${serverName}.${tool.name}`;
498
- fallback ??= executor;
499
- const haystack = `${serverName} ${tool.name} ${tool.description ?? ''}`.toLowerCase();
500
- if (text.split(/[^a-z0-9]+/).filter((token) => token.length >= 4).some((token) => haystack.includes(token))) {
501
- return executor;
502
- }
503
- }
504
- }
505
- return fallback;
506
- }
507
-
508
500
  // The manager runs on the same host filesystem as the workspace directory
509
501
  // (this is the same local file wiki__profile_update writes to via its
510
502
  // volume-mounted container), so read it fresh on every turn instead of
@@ -535,6 +527,7 @@ export function buildAgentSystemPrompt(state) {
535
527
  `Current workspace: ${workspace}.`,
536
528
  `Current wikirc profile: ${wikirc}.`,
537
529
  `Available primitives: ${commandList(state.session)}.`,
530
+ 'Only announce or call slash commands that appear exactly in Available primitives. Do not invent command names, subcommands, or arguments.',
538
531
  'Connected MCP tools (use the server__tool naming convention for tool calls):',
539
532
  mcpTools,
540
533
  'Current local MCP job queue:',
@@ -556,8 +549,8 @@ export function buildAgentSystemPrompt(state) {
556
549
  '',
557
550
  'Task startup:',
558
551
  ' 1. If the next MCP tool returns _activity.plan.steps, call that tool directly; the shell will create the visible plan from the returned activity.',
559
- ' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, dependsOn, executor, executorQuery, outputRefs}; a legacy list of strings is still accepted.',
560
- ' Multi-tool example: wiki__plan_set(steps=[{id:"cme-export",description:"CME export",dependsOn:[],executor:"cme.cme_export_run",outputRefs:["raw/untracked"]},{id:"production",description:"Production pipeline",dependsOn:["cme-export"],executor:"production.production_start_job",outputRefs:["deliverables"]}])',
552
+ ' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, dependsOn, outputRefs}; a legacy list of strings is still accepted.',
553
+ ' Multi-tool example: wiki__plan_set(steps=[{id:"cme-export",description:"CME export",requiredCapability:"external-source.export",dependsOn:[],outputRefs:["raw/untracked"]},{id:"production",description:"Production pipeline",requiredCapability:"knowledge.pipeline",dependsOn:["cme-export"],outputRefs:["deliverables"]}])',
561
554
  ' 3. Immediately execute the first step using the appropriate MCP tool. Do not start step 2 in the same turn unless one async pipeline tool owns and declares the whole sequence.',
562
555
  ' For synchronous steps (result is immediate, no _activity polling), call wiki__plan_done(step=1) after confirming success.',
563
556
  ' For async MCP jobs (returns _activity with poll), the orchestrator tracks completion automatically.',
@@ -575,13 +568,13 @@ export function buildAgentSystemPrompt(state) {
575
568
  '',
576
569
  'On failure: if a completed activity is failed/error/cancelled, call wiki__plan_done(step=N, status="failed") then stop with a clear error report.',
577
570
  ].filter(Boolean).join('\n'),
578
- 'For service actions, recommend /services, /start, /stop or /logs with the exact service name.',
571
+ 'For service actions, recommend only available service primitives from Available primitives, with the exact service name when the primitive supports one.',
579
572
  'Disambiguate export requests carefully.',
580
- 'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`cme_export_run`, then `cme_export_status`). Never use production `type=export` for Confluence source export.',
581
- 'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`production_start_job` with `type:"export"` or pipeline steps). Require the deliverable path when exporting deliverables.',
582
- 'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts. To chain multiple sequential steps (e.g. build then polish), always use a single production_start_job call with type="pipeline" and steps=["build","polish"] — never start them as separate jobs: the first job is asynchronous and the second would run before it completes. For existing deliverables where content stability matters, pass stabilize:true so the build step preserves unchanged sections; keep polish in the pipeline when publication output is requested. Do not ask the user to confirm between steps; start the pipeline call directly.',
573
+ 'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`cme__cme_export_run`, then `cme__cme_export_status`). Never use production `type=export` for Confluence source export.',
574
+ 'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`production__production_start_job` with `type:"export"` or pipeline steps). Require the deliverable path when exporting deliverables.',
575
+ 'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts. To chain multiple sequential steps (e.g. build then polish), always use a single production__production_start_job call with type="pipeline" and steps=["build","polish"] — never start them as separate jobs: the first job is asynchronous and the second would run before it completes. For existing deliverables where content stability matters, pass stabilize:true so the build step preserves unchanged sections; keep polish in the pipeline when publication output is requested. Do not ask the user to confirm between steps; start the pipeline call directly.',
583
576
  'Long-running MCP jobs: do not call the same status tool more than once consecutively. When chaining jobs sequentially: (1) start the job, report job/activity id and status; (2) check status once — if done, proceed to the next step immediately; (3) if still running, report status, list the remaining steps, and return control; (4) when re-invoked, check status first, then proceed. Do not spin-poll (status → status → status with no new action between). The shell activity panel monitors non-terminal jobs automatically.',
584
- 'If production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
577
+ 'If production__production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
585
578
  'For diagnostics, use /wiki run doctor when the user asks for doctor. Use /workspace init <name> [path] for low-level non-interactive workspace creation. In the interactive TUI, /new <name> opens the setup wizard. Use /wiki for index, or /wiki run index through the explicit backup hatch. Use /wiki run init only for explicit current-workspace llm-wiki init.',
586
579
  'If an action requires tools or skills not available yet, explain the limitation and name the expected primitive.',
587
580
  workspaceProfile
@@ -617,6 +610,36 @@ export function formatLlmUnavailableMessage(reason) {
617
610
  return `⚠ LLM injoignable : ${clean || 'raison inconnue'}`;
618
611
  }
619
612
 
613
+ function classifyAgentInput(input, session) {
614
+ const lower = String(input ?? '').toLowerCase();
615
+ const hasActiveRun = session?.agentProjection?.status === 'running'
616
+ || sessionActivities(session).some((activity) => !activity.terminal);
617
+ if (/\b(valide tout|approve all|approve|approuve|valid[eé]|ok pour tout|go pour tout)\b/i.test(lower)) {
618
+ return { kind: 'approve', confidence: 0.86, reason: 'approval_request', activeRun: hasActiveRun };
619
+ }
620
+ if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort)\b/i.test(lower)) {
621
+ return { kind: 'cancel', confidence: 0.86, reason: 'cancel_request', activeRun: hasActiveRun };
622
+ }
623
+ if (/\b(plus tard|later|ensuite|apr[eè]s ce run|enqueue|mets en file|met en file|futur|next run|future run)\b/i.test(lower)) {
624
+ return { kind: 'enqueue_run', confidence: 0.8, reason: 'future_run_request', activeRun: hasActiveRun };
625
+ }
626
+ if (/\b(o[uù] en es[t-]|status|statut|progress|progression|run|job|queue|logs?|explique|explain|inspect|show|montre|quoi de neuf)\b/i.test(lower)) {
627
+ return { kind: 'observe', confidence: 0.86, reason: 'status_or_explanation_request', activeRun: hasActiveRun };
628
+ }
629
+ if (hasActiveRun && /\b(ajoute|add|change|modifie|modify|remplace|replace|retire|remove|skip|ignore|plan|step|t[aâ]che)\b/i.test(lower)) {
630
+ return { kind: 'modify_run', confidence: 0.78, reason: 'active_run_change_request', activeRun: hasActiveRun };
631
+ }
632
+ if (hasActiveRun && /\b(lance|run|g[eé]n[eè]re|build|export|cr[eé]e|create|send|envoie|ingest|convert|importe|import)\b/i.test(lower)) {
633
+ return { kind: 'ambiguous', confidence: 0.45, reason: 'active_run_action_is_ambiguous', activeRun: hasActiveRun };
634
+ }
635
+ return { kind: 'converse', confidence: 0.62, reason: 'plain_conversation', activeRun: hasActiveRun };
636
+ }
637
+
638
+ function toolsForClassification(classification, writeTools) {
639
+ if (classification.activeRun && ['converse', 'observe'].includes(classification.kind)) return [SHELL_READ_COMMAND_TOOL];
640
+ return [SHELL_READ_COMMAND_TOOL, ...writeTools];
641
+ }
642
+
620
643
  export function createAgentGraph(options = {}) {
621
644
  async function orchestratorNode(state) {
622
645
  const llm = state.session.llm ?? options.llm ?? null;
@@ -640,15 +663,33 @@ export function createAgentGraph(options = {}) {
640
663
  state.session._onStep?.('Agent: planning next action…');
641
664
  }
642
665
 
643
- const allTools = [
666
+ const classification = iterations === 0
667
+ ? classifyAgentInput(state.input, state.session)
668
+ : (state.inputClassification ?? { kind: 'modify_run', confidence: 1, reason: 'tool_iteration' });
669
+ if (iterations === 0) {
670
+ state.session._onStep?.(`Agent: classified input as ${classification.kind}`);
671
+ emitAgentEvent(state.session, 'control_message_received', 'agent_classifier', {
672
+ input: state.input,
673
+ classification,
674
+ });
675
+ }
676
+ if (iterations === 0 && classification.kind === 'ambiguous') {
677
+ return {
678
+ response: 'Je peux répondre sur le run en cours, modifier ce run, mettre une nouvelle demande en file, approuver, ou annuler. Peux-tu préciser ce que tu veux faire ?',
679
+ pendingToolCalls: null,
680
+ readyToStream: false,
681
+ inputClassification: classification,
682
+ };
683
+ }
684
+
685
+ const writeTools = [
644
686
  SHELL_RUN_COMMAND_TOOL,
645
- SHELL_READ_COMMAND_TOOL,
646
687
  SHELL_PROFILE_UPDATE_TOOL,
647
688
  WIKI_PLAN_SET_TOOL,
648
689
  WIKI_PLAN_DONE_TOOL,
649
690
  ...buildLlmTools(state.session.mcp),
650
691
  ];
651
- const tools = allTools;
692
+ const tools = toolsForClassification(classification, writeTools);
652
693
  const system = buildAgentSystemPrompt(state);
653
694
 
654
695
  // On iteration 0: prior history is in state.messages, user input must be appended.
@@ -690,6 +731,7 @@ export function createAgentGraph(options = {}) {
690
731
  messages: newMessages,
691
732
  toolIterations: iterations + 1,
692
733
  readyToStream: false,
734
+ inputClassification: classification,
693
735
  };
694
736
  }
695
737
 
@@ -705,6 +747,7 @@ export function createAgentGraph(options = {}) {
705
747
  readyToStream: false,
706
748
  streamedInline: true,
707
749
  messages: newMessages,
750
+ inputClassification: classification,
708
751
  };
709
752
  }
710
753
 
@@ -736,11 +779,19 @@ export function createAgentGraph(options = {}) {
736
779
  const toolResultMessages = [];
737
780
 
738
781
  for (const call of toolCalls) {
739
- const { server, tool } = parseToolCallName(call.function.name);
782
+ const resolved = resolveToolCallName(state.session.mcp, call.function.name, INTERNAL_TOOL_SERVERS);
783
+ const { server, tool } = resolved;
740
784
  const argsSummary = summarizeToolArguments(call.function.arguments);
741
785
  const isInternalWikiTool = server === 'wiki' && (tool === 'plan_set' || tool === 'plan_done');
742
786
  const serverLabel = server === 'shell' ? 'Shell' : isInternalWikiTool ? 'Plan' : 'MCP';
743
- const toolName = `${server}.${tool}`;
787
+ const toolName = server ? `${server}.${tool}` : call.function.name;
788
+ if (resolved.normalized) {
789
+ // Keep normalizations visible: the defensive routing must not hide
790
+ // prompt/skill regressions that reintroduce unqualified names.
791
+ state.session._onStep?.(
792
+ `tool name normalized: ${call.function.name} -> ${server}__${tool}`,
793
+ );
794
+ }
744
795
  state.session._onStep?.(
745
796
  `[${state.toolIterations}/${MAX_TOOL_ITERATIONS}] ${serverLabel} ${toolName}${argsSummary ? ` (${argsSummary})` : ''}`,
746
797
  );
@@ -761,6 +812,18 @@ export function createAgentGraph(options = {}) {
761
812
  let resultText;
762
813
  let ok = true;
763
814
  try {
815
+ if (!server) {
816
+ if (resolved.candidates.length > 1) {
817
+ throw new Error(
818
+ `Ambiguous unqualified tool name "${call.function.name}": several connected servers expose it. `
819
+ + `Use the <server>__<tool> form: ${resolved.candidates.map((s) => `${s}__${tool}`).join(', ')}.`,
820
+ );
821
+ }
822
+ throw new Error(
823
+ `Unqualified tool call name "${call.function.name}". Tool calls must use the <server>__<tool> `
824
+ + `naming convention (e.g. production__production_start_job); no connected server exposes a tool named "${tool}".`,
825
+ );
826
+ }
764
827
  let args = JSON.parse(call.function.arguments ?? '{}');
765
828
  if (server === 'production' && tool === 'production_start_job' && state.session.workspace && !args.callerLabel) {
766
829
  args = { ...args, callerLabel: `${state.session.workspace}/wiki-manager` };
@@ -834,7 +897,7 @@ export function createAgentGraph(options = {}) {
834
897
  err.name === 'ApprovalError'
835
898
  ) throw err;
836
899
  ok = false;
837
- resultText = `Error [${server}.${tool}]: ${err instanceof Error ? err.message : String(err)}`;
900
+ resultText = `Error [${toolName}]: ${err instanceof Error ? err.message : String(err)}`;
838
901
  if (minimalPlanActive && state.session.headlessPlan?.[0]?._activityKey === null) {
839
902
  emitAgentEvent(state.session, 'plan_step_updated', 'tool', { step: 1, status: 'failed' });
840
903
  }
@@ -300,7 +300,7 @@ test('agent graph waits for tool-level approval configured on endpoint', async (
300
300
  }
301
301
  });
302
302
 
303
- test('agent graph accepts structured wiki plan steps and selects MCP executors', async () => {
303
+ test('agent graph accepts structured wiki plan steps without selecting MCP executors implicitly', async () => {
304
304
  let calls = 0;
305
305
  const session = sessionBase({
306
306
  mcp: {
@@ -337,8 +337,20 @@ test('agent graph accepts structured wiki plan steps and selects MCP executors',
337
337
  name: 'wiki__plan_set',
338
338
  arguments: JSON.stringify({
339
339
  steps: [
340
- { id: 'cme-export', description: 'Export CME pages', outputRefs: ['raw/untracked'] },
341
- { id: 'build', description: 'Run production build', dependsOn: ['cme-export'] },
340
+ {
341
+ id: 'cme-export',
342
+ description: 'Export CME pages',
343
+ requiredCapability: 'external-source.export',
344
+ executor: 'cme.cme_export_run',
345
+ executorQuery: { capability: 'legacy export' },
346
+ outputRefs: ['raw/untracked'],
347
+ },
348
+ {
349
+ id: 'build',
350
+ description: 'Run production build',
351
+ requiredCapability: 'knowledge.pipeline',
352
+ dependsOn: ['cme-export'],
353
+ },
342
354
  ],
343
355
  }),
344
356
  },
@@ -359,8 +371,66 @@ test('agent graph accepts structured wiki plan steps and selects MCP executors',
359
371
 
360
372
  assert.equal(result.response, 'Plan ready.');
361
373
  assert.deepEqual(session.headlessPlan.map((step) => step.id), ['cme-export', 'build']);
362
- assert.equal(session.headlessPlan[0].executor, 'cme.cme_export_run');
363
- assert.equal(session.headlessPlan[1].executor, 'production.production_start_job');
374
+ assert.deepEqual(session.headlessPlan.map((step) => step.requiredCapability), ['external-source.export', 'knowledge.pipeline']);
375
+ assert.equal(session.headlessPlan[0].executor, null);
376
+ assert.equal(session.headlessPlan[0].executorQuery, null);
377
+ assert.equal(session.headlessPlan[1].executor, null);
364
378
  assert.deepEqual(session.headlessPlan[1].dependsOn, ['cme-export']);
365
379
  assert.deepEqual(session.headlessPlan[0].outputRefs, ['raw/untracked']);
366
380
  });
381
+
382
+ test('buildAgentSystemPrompt forbids inventing slash commands or arguments', () => {
383
+ const prompt = buildAgentSystemPrompt({ session: sessionBase({ commands: ['status', 'services'] }) });
384
+ assert.match(prompt, /Available primitives: \/status, \/services\./);
385
+ assert.match(prompt, /Do not invent command names, subcommands, or arguments/);
386
+ assert.doesNotMatch(prompt, /\/restart serve/);
387
+ assert.doesNotMatch(prompt, /executorQuery/);
388
+ assert.doesNotMatch(prompt, /executor:"/);
389
+ });
390
+
391
+ // Guard: the system prompt must never show a connected tool's bare name
392
+ // outside its qualified server__tool form. Bare mentions are what teach the
393
+ // model to emit unqualified tool calls (the cme_status incident). The bare
394
+ // name list comes from the session's declared servers, never from a manual
395
+ // list (amendment A6). New prompt text or injected skill descriptions that
396
+ // reintroduce a bare name must fail here.
397
+ test('buildAgentSystemPrompt contains no unqualified tool names for connected servers', () => {
398
+ const session = sessionBase({
399
+ mcp: {
400
+ production: {
401
+ status: 'connected',
402
+ tools: [
403
+ { name: 'production_start_job' }, { name: 'production_job_status' },
404
+ { name: 'production_job_logs' }, { name: 'production_cancel_job' },
405
+ { name: 'production_list_jobs' }, { name: 'production_list_templates' },
406
+ { name: 'production_status' }, { name: 'agent_describe' },
407
+ { name: 'agent_plan' }, { name: 'agent_execute' },
408
+ { name: 'agent_status' }, { name: 'agent_cancel' },
409
+ ],
410
+ },
411
+ cme: {
412
+ status: 'connected',
413
+ tools: [
414
+ { name: 'cme_status' }, { name: 'cme_setup' },
415
+ { name: 'cme_sources_list' }, { name: 'cme_source_add' },
416
+ { name: 'cme_source_remove' }, { name: 'cme_export_run' },
417
+ { name: 'cme_export_status' }, { name: 'cme_export_cancel' },
418
+ { name: 'agent_describe' }, { name: 'agent_execute' },
419
+ { name: 'agent_status' }, { name: 'agent_cancel' },
420
+ ],
421
+ },
422
+ },
423
+ });
424
+ const prompt = buildAgentSystemPrompt({ session });
425
+ const offenders = [];
426
+ for (const [serverName, value] of Object.entries(session.mcp)) {
427
+ for (const tool of value.tools) {
428
+ // A bare occurrence is the tool name not embedded in a wider
429
+ // identifier: `production__production_start_job` does not match
430
+ // because the inner occurrence is preceded by `_`.
431
+ const bare = new RegExp(`(?<![\\w])${tool.name}(?![\\w])`);
432
+ if (bare.test(prompt)) offenders.push(`${serverName}:${tool.name}`);
433
+ }
434
+ }
435
+ assert.deepEqual(offenders, [], `Unqualified tool names found in system prompt: ${offenders.join(', ')}`);
436
+ });
@@ -294,6 +294,7 @@ async function runRuntime(argv, agent) {
294
294
  }
295
295
  const { defaultRuntimeStateDir, openRuntimeStore, RECOVERABLE_QUEUE_STATUSES } = await import('../runtime/store.js');
296
296
  const { startRuntimeServer } = await import('../runtime/server.js');
297
+ const { recoverActiveRuns } = await import('../runtime/recoveryManager.js');
297
298
  const { emitRuntimeLog, startActivitySupervisor } = await import('../runtime/supervisor.js');
298
299
  const { resolveRuntimeAuthToken } = await import('../runtime/auth.js');
299
300
  const { createSqliteQueueStore } = await import('../runtime/queueStore.js');
@@ -484,6 +485,27 @@ async function runRuntime(argv, agent) {
484
485
  reason: `MCP unavailable: ${gaps.join(', ')}`,
485
486
  };
486
487
  }
488
+ const taskRecovery = await recoverActiveRuns({
489
+ store,
490
+ session: context.session,
491
+ workspace: context.workspace,
492
+ callTool: callMcpTool,
493
+ });
494
+ if (!taskRecovery.ok) {
495
+ const interrupted = store.interruptRuns({ workspace: context.workspace });
496
+ return {
497
+ workspace: context.workspace ?? workspace ?? null,
498
+ resumed: false,
499
+ interrupted,
500
+ reason: `Task recovery failed: ${taskRecovery.errors.map((item) => `${item.taskId}: ${item.error}`).join('; ')}`,
501
+ };
502
+ }
503
+ if (taskRecovery.recovered.length > 0 || taskRecovery.rescheduled.length > 0) {
504
+ emitRuntimeLog(
505
+ context.session,
506
+ `runtime: recovery attached ${taskRecovery.recovered.length} job(s), requeued ${taskRecovery.rescheduled.length} task(s)`,
507
+ );
508
+ }
487
509
  const recoverableRuns = store.listRecoverableRuns({ workspace: context.workspace });
488
510
  const runningRun = recoverableRuns.find((run) => run.status === 'running');
489
511
  const activeActivities = activeNonTerminalActivities(context.session);