@dotdrelle/wiki-manager 0.12.0 → 0.12.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/package.json +2 -2
  2. package/src/activity/activityAggregator.js +34 -3
  3. package/src/activity/activityAggregator.test.js +32 -0
  4. package/src/agent/graph.js +307 -29
  5. package/src/agent/graph.test.js +404 -1
  6. package/src/cli/wiki-manager.js +24 -1
  7. package/src/commands/slash.js +74 -1
  8. package/src/commands/slash.test.js +36 -0
  9. package/src/contracts/schemas.test.js +2 -2
  10. package/src/core/activity.js +4 -0
  11. package/src/core/activity.test.js +9 -1
  12. package/src/core/agentEvents.js +30 -0
  13. package/src/core/agentEvents.test.js +42 -1
  14. package/src/core/buildInfo.js +58 -0
  15. package/src/core/buildInfo.json +4 -0
  16. package/src/core/buildInfo.test.js +14 -0
  17. package/src/core/mcp.js +50 -3
  18. package/src/core/mcp.test.js +94 -1
  19. package/src/core/runtimeLog.js +6 -1
  20. package/src/core/runtimeLog.test.js +3 -1
  21. package/src/runtime/client.js +22 -0
  22. package/src/runtime/controlMessages.js +43 -0
  23. package/src/runtime/controlMessages.test.js +21 -0
  24. package/src/runtime/donna-contract.test.js +3 -3
  25. package/src/runtime/recoveryManager.js +54 -0
  26. package/src/runtime/recoveryManager.test.js +73 -0
  27. package/src/runtime/runner.js +49 -13
  28. package/src/runtime/runner.test.js +196 -0
  29. package/src/runtime/server.js +73 -13
  30. package/src/runtime/server.test.js +284 -67
  31. package/src/runtime/store.js +48 -3
  32. package/src/runtime/store.test.js +56 -24
  33. package/src/runtime/supervisor.js +77 -1
  34. package/src/shell/RightPane.tsx +124 -31
  35. package/src/shell/SetupWizard.tsx +13 -1
  36. package/src/shell/repl.js +81 -11
  37. package/src/shell/repl.test.js +115 -2
  38. package/src/shell/tui.tsx +4 -1
  39. package/src/shell/useAgent.ts +37 -4
  40. package/src/shell/useSession.ts +51 -10
@@ -5,7 +5,8 @@ import {
5
5
  callMcpTool,
6
6
  formatMcpToolResult,
7
7
  formatMcpToolsForAgent,
8
- parseToolCallName,
8
+ resolveToolCallName,
9
+ truncateToolResult,
9
10
  } from '../core/mcp.js';
10
11
  import { formatSkillsForAgent, readOptionalText } from '../core/skills.js';
11
12
  import { handleSlashCommand } from '../commands/slash.js';
@@ -13,11 +14,22 @@ import { extractActivity, formatActivitySummary, parseJsonText, sessionActivitie
13
14
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
14
15
  import { enqueueProductionJob, ensureJobQueue, formatQueue, productionLockBusy } from '../core/jobQueue.js';
15
16
  import { updateWorkspaceProfilePreference } from '../core/profile.js';
17
+ import { createCapabilityRegistry } from '../orchestrator/capabilityRegistry.js';
18
+ import { fetchRuntimeState, postRuntimeCancel, postRuntimeControl, postRuntimeKill } from '../runtime/client.js';
16
19
 
17
20
  const MAX_TOOL_ITERATIONS = 80;
18
21
  const MAX_SPINNER_ARG_LENGTH = 96;
19
22
  const MAX_PROFILE_CHARS = 4000;
20
23
 
24
+ // Pseudo-servers handled directly by the tool executor (not present in
25
+ // session.mcp). Listed so unqualified names like "plan_set" resolve the same
26
+ // way as MCP tools in resolveToolCallName.
27
+ const INTERNAL_TOOL_SERVERS = {
28
+ wiki: ['plan_set', 'plan_done'],
29
+ shell: ['run_command', 'read_command', 'profile_update'],
30
+ runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue'],
31
+ };
32
+
21
33
  const AGENT_SLASH_COMMANDS = new Set([
22
34
  'help',
23
35
  'version',
@@ -100,6 +112,59 @@ const SHELL_PROFILE_UPDATE_TOOL = {
100
112
  },
101
113
  };
102
114
 
115
+ // Runtime control tools: Donna interprets the user's intent ("supprime le
116
+ // job et la queue", "arrête tout", "où en est le run") and ACTS through
117
+ // these, instead of a hardcoded regex classifier answering with canned text.
118
+ const RUNTIME_KILL_TOOL = {
119
+ type: 'function',
120
+ function: {
121
+ name: 'runtime__kill',
122
+ description: 'Hard-stop the workspace runtime: abort the active run, cancel its agent jobs, mark persisted runs interrupted and purge the control queue. Use when the user asks to remove/kill/clean the current run, its jobs or the queue.',
123
+ parameters: { type: 'object', additionalProperties: false, properties: { runId: { type: 'string', description: 'Optional specific run id; omit to kill everything active in the workspace.' } } },
124
+ },
125
+ };
126
+
127
+ const RUNTIME_CANCEL_TOOL = {
128
+ type: 'function',
129
+ function: {
130
+ name: 'runtime__cancel',
131
+ description: 'Soft-cancel the active runtime run (graceful abort, no queue purge). Use for "annule le run" when the user does not ask to clean everything.',
132
+ parameters: { type: 'object', additionalProperties: false, properties: {} },
133
+ },
134
+ };
135
+
136
+ const RUNTIME_STATUS_TOOL = {
137
+ type: 'function',
138
+ function: {
139
+ name: 'runtime__status',
140
+ description: 'Read the runtime state: active run, plan steps, queue items, approvals. Use to answer questions about what is currently running or queued.',
141
+ parameters: { type: 'object', additionalProperties: false, properties: {} },
142
+ },
143
+ };
144
+
145
+ const RUNTIME_APPROVE_TOOL = {
146
+ type: 'function',
147
+ function: {
148
+ name: 'runtime__approve',
149
+ description: 'Grant the pending approval of the active runtime run (mutating tasks wait on it). Use when the user consents in ANY phrasing: "vas-y", "ok pour l\'export", "approuve", "valide". Confirm what was approved.',
150
+ parameters: { type: 'object', additionalProperties: false, properties: {} },
151
+ },
152
+ };
153
+
154
+ const RUNTIME_ENQUEUE_TOOL = {
155
+ type: 'function',
156
+ function: {
157
+ name: 'runtime__enqueue',
158
+ description: 'Queue a request to run AFTER the currently active runtime run finishes. Use when the user asks for a new action while a run is active and wants it done afterwards.',
159
+ parameters: {
160
+ type: 'object',
161
+ additionalProperties: false,
162
+ properties: { input: { type: 'string', description: 'The request to execute after the current run, phrased as a complete instruction.' } },
163
+ required: ['input'],
164
+ },
165
+ },
166
+ };
167
+
103
168
  const WIKI_PLAN_SET_TOOL = {
104
169
  type: 'function',
105
170
  function: {
@@ -128,6 +193,8 @@ const WIKI_PLAN_SET_TOOL = {
128
193
  status: { type: 'string', enum: ['pending', 'queued', 'running', 'waiting', 'pending_approval', 'done', 'failed', 'cancelled', 'stalled', 'added_during_run'] },
129
194
  dependsOn: { type: 'array', items: { type: 'string' } },
130
195
  outputRefs: { type: 'array', items: { type: 'string' } },
196
+ operation: { type: ['string', 'null'], description: 'Operation for the capability provider (e.g. ingest_plan, build).' },
197
+ arguments: { type: 'object', description: 'Arguments passed to the provider agent_execute for this step.' },
131
198
  },
132
199
  required: ['description'],
133
200
  },
@@ -440,9 +507,87 @@ function emitAgentEvent(session, type, origin, payload = {}) {
440
507
  dispatchAgentEvent(session, createAgentEvent(type, { origin, payload }));
441
508
  }
442
509
 
510
+ // Capability ids currently provided by discovered, orchestrable, healthy
511
+ // agents. This is the live registry the dispatcher will resolve against —
512
+ // a plan declaring anything outside this set can only stall forever.
513
+ export function knownCapabilityIds(session) {
514
+ const registry = session?.capabilityRegistry ?? createCapabilityRegistry({
515
+ agents: session?.agentRegistrySnapshot ?? session?.agents ?? [],
516
+ });
517
+ const snapshot = typeof registry.snapshot === 'function' ? registry.snapshot() : registry;
518
+ return [...new Set(Object.keys(snapshot ?? {}).map((key) => {
519
+ const index = key.lastIndexOf('@');
520
+ return index > 0 ? key.slice(0, index) : key;
521
+ }))].sort();
522
+ }
523
+
524
+ async function handleRuntimeControlTool(session, tool, args = {}) {
525
+ const url = session.runtime?.url ?? null;
526
+ if (!url) return 'Runtime not connected: no runtime URL available in this session.';
527
+ const workspace = session.workspace ?? null;
528
+ try {
529
+ if (tool === 'kill') {
530
+ const result = await postRuntimeKill({ url, workspace, runId: args.runId ?? null });
531
+ return `Runtime killed: ${result.runs ?? 0} run(s) interrupted, ${result.tasks ?? 0} task(s) cancelled, ${result.queued ?? 0} queued control request(s) purged.`;
532
+ }
533
+ if (tool === 'cancel') {
534
+ const result = await postRuntimeCancel({ url, workspace });
535
+ return result.cancelled ? 'Runtime run cancellation requested.' : `No active run to cancel${result.reason ? ` (${result.reason})` : ''}.`;
536
+ }
537
+ if (tool === 'approve') {
538
+ const result = await postRuntimeControl('message', { url, workspace, input: 'approve', intent: 'approve' });
539
+ return String(result?.explanation ?? (result?.accepted ? 'Approval granted.' : 'No pending approval found.'));
540
+ }
541
+ if (tool === 'enqueue') {
542
+ const result = await postRuntimeControl('message', { url, workspace, input: String(args.input ?? ''), intent: 'enqueue' });
543
+ return String(result?.explanation ?? 'Request queued for after the current run.');
544
+ }
545
+ if (tool === 'status') {
546
+ const state = await fetchRuntimeState({ url, workspace });
547
+ const plan = Array.isArray(state?.plan) ? state.plan : [];
548
+ const queue = Array.isArray(state?.queue) ? state.queue : [];
549
+ const controlQueue = Array.isArray(state?.controlQueue) ? state.controlQueue : [];
550
+ return JSON.stringify({
551
+ status: state?.status ?? 'unknown',
552
+ running: Boolean(state?.running),
553
+ runId: state?.runId ?? null,
554
+ plan: plan.map((step) => ({ id: step.id ?? step.step, description: step.description, status: step.status })),
555
+ queue: queue.map((item) => ({ id: item.id, status: item.status, tool: item.tool ?? item.type ?? null })),
556
+ controlQueue: controlQueue.filter((item) => item.status === 'queued').map((item) => ({ id: item.id, input: item.input })),
557
+ pendingApprovals: (Array.isArray(state?.approvals) ? state.approvals : [])
558
+ .filter((approval) => approval.status === 'pending_approval')
559
+ .map((approval) => ({ id: approval.id, reason: approval.reason ?? null })),
560
+ }, null, 2);
561
+ }
562
+ return `Unknown runtime tool: ${tool}`;
563
+ } catch (err) {
564
+ return `Runtime control error (${tool}): ${err instanceof Error ? err.message : String(err)}`;
565
+ }
566
+ }
567
+
443
568
  function handleWikiTool(session, tool, args) {
444
569
  if (tool === 'plan_set') {
445
570
  const steps = Array.isArray(args.steps) ? args.steps : [];
571
+ // Reject fantasy capabilities BEFORE the plan exists: once registered,
572
+ // unresolvable steps become tasks that wait forever and flood the queue.
573
+ // A tool-level error (not an exception) lets the LLM correct itself in
574
+ // the same turn. An empty registry (discovery not done yet) skips the
575
+ // check rather than blocking legitimate early plans.
576
+ const known = knownCapabilityIds(session);
577
+ if (known.length > 0) {
578
+ const unknown = [...new Set(steps
579
+ .map((step) => (step && typeof step === 'object' ? step.requiredCapability : null))
580
+ .filter(Boolean)
581
+ .map(String)
582
+ .filter((capability) => !known.includes(capability.includes('@') ? capability.slice(0, capability.lastIndexOf('@')) : capability)))];
583
+ if (unknown.length > 0) {
584
+ return `Plan rejected: unknown capabilities [${unknown.join(', ')}]. `
585
+ + `Available capabilities: ${known.join(', ')}. `
586
+ + 'Redeclare the plan using only available capabilities, or use requiredCapability: null for a step you execute yourself.';
587
+ }
588
+ } else if (steps.some((step) => step && typeof step === 'object' && step.requiredCapability)) {
589
+ session._onStep?.('plan_set: capability registry empty, validation skipped');
590
+ }
446
591
  emitAgentEvent(session, 'plan_set', 'tool', {
447
592
  steps: steps.map((raw, i) => normalizeDeclaredPlanStep(raw, i, session)),
448
593
  });
@@ -475,6 +620,21 @@ function normalizeDeclaredPlanStep(raw, index) {
475
620
  executor: null,
476
621
  executorQuery: null,
477
622
  outputRefs: Array.isArray(item.outputRefs) ? item.outputRefs.map(String) : [],
623
+ // Execution fields the deterministic dispatcher consumes (agent_execute):
624
+ // without them a capability step cannot actually run.
625
+ ...(item.operation != null ? { operation: String(item.operation) } : {}),
626
+ ...(item.arguments && typeof item.arguments === 'object' ? { arguments: item.arguments } : {}),
627
+ ...(item.groupId != null ? { groupId: String(item.groupId) } : {}),
628
+ ...(item.dependsOnGroup != null ? { dependsOnGroup: String(item.dependsOnGroup) } : {}),
629
+ ...(item.parallelizable != null ? { parallelizable: Boolean(item.parallelizable) } : {}),
630
+ ...(item.barrier ? { barrier: true } : {}),
631
+ ...(item.locks ? { locks: item.locks } : {}),
632
+ ...(item.requiresApproval != null ? { requiresApproval: Boolean(item.requiresApproval) } : {}),
633
+ ...(item.approvalClass ? { approvalClass: String(item.approvalClass) } : {}),
634
+ ...(item.approvalSummary ? { approvalSummary: String(item.approvalSummary) } : {}),
635
+ ...(item.idempotencyKey ? { idempotencyKey: String(item.idempotencyKey) } : {}),
636
+ ...(item.progressWeight != null ? { progressWeight: Number(item.progressWeight) } : {}),
637
+ ...(item.recommendedConcurrency != null ? { recommendedConcurrency: Number(item.recommendedConcurrency) } : {}),
478
638
  };
479
639
  }
480
640
 
@@ -539,9 +699,16 @@ export function buildAgentSystemPrompt(state) {
539
699
  'Prefer MCP tools that declare their own plan via _activity.plan.steps — when such a tool returns _activity, the shell creates and tracks the plan automatically without requiring wiki__plan_set.',
540
700
  'Use wiki__plan_set when the MCP tool cannot declare its own plan or when the task spans multiple independent tools (e.g. CME export then email report). For a single self-describing async job, wiki__plan_set is optional.',
541
701
  '',
702
+ (() => {
703
+ const capabilityIds = knownCapabilityIds(state.session);
704
+ return capabilityIds.length > 0
705
+ ? `Known orchestration capabilities — the ONLY values allowed in requiredCapability: ${capabilityIds.join(', ')}. Never invent capability names; a plan declaring an unknown capability will be rejected. A step you execute yourself directly takes requiredCapability: null.`
706
+ : 'No orchestration capabilities discovered yet: declare plan steps with requiredCapability: null and execute them yourself with the connected MCP tools.';
707
+ })(),
708
+ '',
542
709
  'Task startup:',
543
710
  ' 1. If the next MCP tool returns _activity.plan.steps, call that tool directly; the shell will create the visible plan from the returned activity.',
544
- ' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, dependsOn, outputRefs}; a legacy list of strings is still accepted.',
711
+ ' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, operation, arguments, dependsOn, outputRefs}; capability steps need operation+arguments for the dispatcher to execute them; a legacy list of strings is still accepted.',
545
712
  ' Multi-tool example: wiki__plan_set(steps=[{id:"cme-export",description:"CME export",requiredCapability:"external-source.export",dependsOn:[],outputRefs:["raw/untracked"]},{id:"production",description:"Production pipeline",requiredCapability:"knowledge.pipeline",dependsOn:["cme-export"],outputRefs:["deliverables"]}])',
546
713
  ' 3. Immediately execute the first step using the appropriate MCP tool. Do not start step 2 in the same turn unless one async pipeline tool owns and declares the whole sequence.',
547
714
  ' For synchronous steps (result is immediate, no _activity polling), call wiki__plan_done(step=1) after confirming success.',
@@ -561,17 +728,21 @@ export function buildAgentSystemPrompt(state) {
561
728
  'On failure: if a completed activity is failed/error/cancelled, call wiki__plan_done(step=N, status="failed") then stop with a clear error report.',
562
729
  ].filter(Boolean).join('\n'),
563
730
  'For service actions, recommend only available service primitives from Available primitives, with the exact service name when the primitive supports one.',
731
+ 'Scope discipline: execute ONLY the action(s) the user explicitly requested. Never chain additional mutating operations (ingest, build, export, polish, delete, send…) that the user did not ask for — even when diagnostics or recommendations suggest them. Finish the requested work, then list the suggested follow-ups in your final answer and stop. Example: "applique les recommandations de config" means apply the config; it does NOT authorize launching the ingest those recommendations mention.',
564
732
  'Disambiguate export requests carefully.',
565
- 'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`cme_export_run`, then `cme_export_status`). Never use production `type=export` for Confluence source export.',
566
- 'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`production_start_job` with `type:"export"` or pipeline steps). Require the deliverable path when exporting deliverables.',
567
- 'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts. To chain multiple sequential steps (e.g. build then polish), always use a single production_start_job call with type="pipeline" and steps=["build","polish"] — never start them as separate jobs: the first job is asynchronous and the second would run before it completes. For existing deliverables where content stability matters, pass stabilize:true so the build step preserves unchanged sections; keep polish in the pipeline when publication output is requested. Do not ask the user to confirm between steps; start the pipeline call directly.',
733
+ 'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`cme__cme_export_run`, then `cme__cme_export_status`). Never use production `type=export` for Confluence source export.',
734
+ 'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`production__production_start_job` with `type:"export"` or pipeline steps). Require the deliverable path when exporting deliverables.',
735
+ 'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts.',
736
+ 'MULTI-DOCUMENT ingest (more than 2 files, or "ingest everything pending"): call production__agent_plan first, e.g. {capability:"knowledge.update", operation:"ingest", constraints:{maxConcurrency:3, requireApprovalForMutations:true}}. The shell integrates the returned task graph as the plan automatically and the orchestrator dispatches the per-document tasks IN PARALLEL with an approval gate. Do not call production__production_start_job for multi-document ingest — that creates one monolithic sequential job.',
737
+ 'Single-document ingest or one-off jobs (doctor, one build, one export): production__production_start_job is fine. To chain sequential steps (e.g. build then polish) use ONE call with type="pipeline" and steps=["build","polish"] — never separate jobs (the first is asynchronous). For existing deliverables where content stability matters, pass stabilize:true. Do not ask the user to confirm between steps.',
568
738
  'Long-running MCP jobs: do not call the same status tool more than once consecutively. When chaining jobs sequentially: (1) start the job, report job/activity id and status; (2) check status once — if done, proceed to the next step immediately; (3) if still running, report status, list the remaining steps, and return control; (4) when re-invoked, check status first, then proceed. Do not spin-poll (status → status → status with no new action between). The shell activity panel monitors non-terminal jobs automatically.',
569
- 'If production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
570
- 'For diagnostics, use /wiki run doctor when the user asks for doctor. Use /workspace init <name> [path] for low-level non-interactive workspace creation. In the interactive TUI, /new <name> opens the setup wizard. Use /wiki for index, or /wiki run index through the explicit backup hatch. Use /wiki run init only for explicit current-workspace llm-wiki init.',
739
+ 'If production__production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
740
+ 'For diagnostics (doctor), use production__production_start_job with type="doctor" like any other production job; /wiki run doctor is only the fallback when the production MCP is not connected. Use /workspace init <name> [path] for low-level non-interactive workspace creation. In the interactive TUI, /new <name> opens the setup wizard. Use /wiki for index, or /wiki run index through the explicit backup hatch. Use /wiki run init only for explicit current-workspace llm-wiki init.',
571
741
  'If an action requires tools or skills not available yet, explain the limitation and name the expected primitive.',
572
742
  workspaceProfile
573
743
  ? `Workspace profile (.wiki/profile.md) — durable user preferences, apply these to every reply (tone, tutoiement/vouvoiement, formatting, etc.):\n${workspaceProfile}`
574
744
  : null,
745
+ 'Runtime control: you have runtime__status, runtime__cancel, runtime__kill, runtime__approve and runtime__enqueue. When the user asks to stop, remove, clean or kill the current run, its jobs or the queue ("supprime le job et la queue", "arr\u00eate tout"), call runtime__kill (or runtime__cancel for a soft stop of just the run) and confirm what was stopped. For questions about what is running or queued, call runtime__status and answer from its data. When the user consents to a pending approval in any phrasing ("vas-y", "ok pour l\'export"), call runtime__approve. When the user asks for a NEW action while a run is active, do not execute it: propose runtime__enqueue (run it after) or, if they insist it replaces the current work, runtime__kill then the new action.',
575
746
  'When the user explicitly asks you to remember, persist, or update durable preference/profile information, call wiki__profile_update when it is available; otherwise call shell__profile_update. Do not just acknowledge in text without calling a profile update tool.',
576
747
  ].filter(Boolean).join('\n');
577
748
 
@@ -602,20 +773,35 @@ export function formatLlmUnavailableMessage(reason) {
602
773
  return `⚠ LLM injoignable : ${clean || 'raison inconnue'}`;
603
774
  }
604
775
 
605
- function classifyAgentInput(input, session) {
776
+ // Verbs that clearly request work (a runtime run), in French and English.
777
+ // "configure/configurer" is an action; the nouns "config/configuration" are
778
+ // NOT matched here — asking for a config is an observe request.
779
+ const ACTION_REQUEST_PATTERN = /\b(lance|relance|d[eé]marre|start|ex[eé]cute|execute|g[eé]n[eè]re|generate|build|construis|exporte?|ingest\w*|ing[eè]re|importe?|convert(?:is|it|s)?|cr[eé]e|create|polish|publie|publish|d[eé]ploie|deploy|envoie|send|configure[rsz]?|setup|installe|update|mets? [aà] jour|supprime|delete|efface|nettoie|clean|r[eé]pare|fix|corrige)\b/i;
780
+
781
+ // Explicit explanation/question markers dominate action verbs: "explique le
782
+ // build" is a question about the build, not a request to build.
783
+ const EXPLANATION_REQUEST_PATTERN = /\b(explique|explain|pourquoi|why|comment|how|c'est quoi|qu'est[- ]ce)\b/i;
784
+
785
+ export function classifyAgentInput(input, session) {
606
786
  const lower = String(input ?? '').toLowerCase();
607
787
  const hasActiveRun = session?.agentProjection?.status === 'running'
608
788
  || sessionActivities(session).some((activity) => !activity.terminal);
609
789
  if (/\b(valide tout|approve all|approve|approuve|valid[eé]|ok pour tout|go pour tout)\b/i.test(lower)) {
610
790
  return { kind: 'approve', confidence: 0.86, reason: 'approval_request', activeRun: hasActiveRun };
611
791
  }
612
- if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort)\b/i.test(lower)) {
792
+ if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort|supprime|kill|tue|purge|vide la (file|queue)|nettoie la (file|queue))\b/i.test(lower)) {
613
793
  return { kind: 'cancel', confidence: 0.86, reason: 'cancel_request', activeRun: hasActiveRun };
614
794
  }
615
795
  if (/\b(plus tard|later|ensuite|apr[eè]s ce run|enqueue|mets en file|met en file|futur|next run|future run)\b/i.test(lower)) {
616
796
  return { kind: 'enqueue_run', confidence: 0.8, reason: 'future_run_request', activeRun: hasActiveRun };
617
797
  }
618
- if (/\b(o[uù] en es[t-]|status|statut|progress|progression|run|job|queue|logs?|explique|explain|inspect|show|montre|quoi de neuf)\b/i.test(lower)) {
798
+ if (EXPLANATION_REQUEST_PATTERN.test(lower)) {
799
+ return { kind: 'observe', confidence: 0.86, reason: 'explanation_request', activeRun: hasActiveRun };
800
+ }
801
+ // Observe markers only win when no action verb is present: "où en est le
802
+ // run" is observe, "lance le run" is an action request.
803
+ if (/\b(o[uù] en es[t-]|status|statut|progress|progression|run|job|queue|logs?|inspect|show|montre|affiche|donne|liste|list|quel(?:le)?s?|combien|config(?:uration)?|quoi de neuf)\b/i.test(lower)
804
+ && !ACTION_REQUEST_PATTERN.test(lower)) {
619
805
  return { kind: 'observe', confidence: 0.86, reason: 'status_or_explanation_request', activeRun: hasActiveRun };
620
806
  }
621
807
  if (hasActiveRun && /\b(ajoute|add|change|modifie|modify|remplace|replace|retire|remove|skip|ignore|plan|step|t[aâ]che)\b/i.test(lower)) {
@@ -624,12 +810,24 @@ function classifyAgentInput(input, session) {
624
810
  if (hasActiveRun && /\b(lance|run|g[eé]n[eè]re|build|export|cr[eé]e|create|send|envoie|ingest|convert|importe|import)\b/i.test(lower)) {
625
811
  return { kind: 'ambiguous', confidence: 0.45, reason: 'active_run_action_is_ambiguous', activeRun: hasActiveRun };
626
812
  }
813
+ if (ACTION_REQUEST_PATTERN.test(lower)) {
814
+ return { kind: 'start_run', confidence: 0.8, reason: 'action_request', activeRun: hasActiveRun };
815
+ }
627
816
  return { kind: 'converse', confidence: 0.62, reason: 'plain_conversation', activeRun: hasActiveRun };
628
817
  }
629
818
 
630
- function toolsForClassification(classification, writeTools) {
631
- if (classification.activeRun && ['converse', 'observe'].includes(classification.kind)) return [SHELL_READ_COMMAND_TOOL];
632
- return [SHELL_READ_COMMAND_TOOL, ...writeTools];
819
+ function toolsForClassification(classification, writeTools, session = null) {
820
+ const controlTools = session?.runtime?.url
821
+ ? [RUNTIME_STATUS_TOOL, RUNTIME_CANCEL_TOOL, RUNTIME_KILL_TOOL, RUNTIME_APPROVE_TOOL, RUNTIME_ENQUEUE_TOOL]
822
+ : [];
823
+ if (classification.activeRun && ['converse', 'observe', 'ambiguous', 'approve', 'cancel', 'enqueue_run'].includes(classification.kind)) {
824
+ // During an active run Donna gets read + profile + the runtime control
825
+ // suite: she can answer, approve, enqueue for later, soft-cancel or
826
+ // kill — but she must not fire new MCP jobs alongside the run (that is
827
+ // what runtime__enqueue is for). No canned regex answers anywhere.
828
+ return [SHELL_READ_COMMAND_TOOL, SHELL_PROFILE_UPDATE_TOOL, ...controlTools];
829
+ }
830
+ return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...writeTools];
633
831
  }
634
832
 
635
833
  export function createAgentGraph(options = {}) {
@@ -655,8 +853,17 @@ export function createAgentGraph(options = {}) {
655
853
  state.session._onStep?.('Agent: planning next action…');
656
854
  }
657
855
 
856
+ // Inside a runtime run the input IS the task to execute (the runtime
857
+ // already accepted it as a run): the interactive control-message
858
+ // classifier must not apply. Without this, agentProjection.status is
859
+ // 'running' during every run, so any action verb ("lance l'ingestion")
860
+ // matched the active-run 'ambiguous' branch and returned a canned
861
+ // clarification instead of executing — the run ended silently.
862
+ const runtimeExecution = Boolean(state.session._currentRunIdentity);
658
863
  const classification = iterations === 0
659
- ? classifyAgentInput(state.input, state.session)
864
+ ? (runtimeExecution
865
+ ? { kind: 'execute_run', confidence: 1, reason: 'runtime_run_execution', activeRun: true }
866
+ : classifyAgentInput(state.input, state.session))
660
867
  : (state.inputClassification ?? { kind: 'modify_run', confidence: 1, reason: 'tool_iteration' });
661
868
  if (iterations === 0) {
662
869
  state.session._onStep?.(`Agent: classified input as ${classification.kind}`);
@@ -665,14 +872,6 @@ export function createAgentGraph(options = {}) {
665
872
  classification,
666
873
  });
667
874
  }
668
- if (iterations === 0 && classification.kind === 'ambiguous') {
669
- return {
670
- response: 'Je peux répondre sur le run en cours, modifier ce run, mettre une nouvelle demande en file, approuver, ou annuler. Peux-tu préciser ce que tu veux faire ?',
671
- pendingToolCalls: null,
672
- readyToStream: false,
673
- inputClassification: classification,
674
- };
675
- }
676
875
 
677
876
  const writeTools = [
678
877
  SHELL_RUN_COMMAND_TOOL,
@@ -681,7 +880,7 @@ export function createAgentGraph(options = {}) {
681
880
  WIKI_PLAN_DONE_TOOL,
682
881
  ...buildLlmTools(state.session.mcp),
683
882
  ];
684
- const tools = toolsForClassification(classification, writeTools);
883
+ const tools = toolsForClassification(classification, writeTools, state.session);
685
884
  const system = buildAgentSystemPrompt(state);
686
885
 
687
886
  // On iteration 0: prior history is in state.messages, user input must be appended.
@@ -713,6 +912,13 @@ export function createAgentGraph(options = {}) {
713
912
 
714
913
  if (result.tool_calls?.length > 0) {
715
914
  state.session._onStreamReset?.();
915
+ // Close the streaming conversation entry now: the text streamed so
916
+ // far is this iteration's narration. Without this, the next
917
+ // iteration's deltas append to the SAME entry with no separator and
918
+ // the chat becomes one glued wall of text ("…de la config.Voyons…").
919
+ // An empty finalize keeps the accumulated content and just drops the
920
+ // streaming flag; it is a no-op when nothing was streamed.
921
+ emitAgentEvent(state.session, 'assistant_message', 'llm', { content: '' });
716
922
  state.session._onStep?.(`[${iterations + 1}/${MAX_TOOL_ITERATIONS}] ${result.tool_calls.length} MCP action${result.tool_calls.length > 1 ? 's' : ''} queued…`);
717
923
  // On iteration 0 persist the user message too so it survives the loop.
718
924
  const newMessages = iterations === 0
@@ -771,11 +977,19 @@ export function createAgentGraph(options = {}) {
771
977
  const toolResultMessages = [];
772
978
 
773
979
  for (const call of toolCalls) {
774
- const { server, tool } = parseToolCallName(call.function.name);
980
+ const resolved = resolveToolCallName(state.session.mcp, call.function.name, INTERNAL_TOOL_SERVERS);
981
+ const { server, tool } = resolved;
775
982
  const argsSummary = summarizeToolArguments(call.function.arguments);
776
983
  const isInternalWikiTool = server === 'wiki' && (tool === 'plan_set' || tool === 'plan_done');
777
984
  const serverLabel = server === 'shell' ? 'Shell' : isInternalWikiTool ? 'Plan' : 'MCP';
778
- const toolName = `${server}.${tool}`;
985
+ const toolName = server ? `${server}.${tool}` : call.function.name;
986
+ if (resolved.normalized) {
987
+ // Keep normalizations visible: the defensive routing must not hide
988
+ // prompt/skill regressions that reintroduce unqualified names.
989
+ state.session._onStep?.(
990
+ `tool name normalized: ${call.function.name} -> ${server}__${tool}`,
991
+ );
992
+ }
779
993
  state.session._onStep?.(
780
994
  `[${state.toolIterations}/${MAX_TOOL_ITERATIONS}] ${serverLabel} ${toolName}${argsSummary ? ` (${argsSummary})` : ''}`,
781
995
  );
@@ -796,6 +1010,18 @@ export function createAgentGraph(options = {}) {
796
1010
  let resultText;
797
1011
  let ok = true;
798
1012
  try {
1013
+ if (!server) {
1014
+ if (resolved.candidates.length > 1) {
1015
+ throw new Error(
1016
+ `Ambiguous unqualified tool name "${call.function.name}": several connected servers expose it. `
1017
+ + `Use the <server>__<tool> form: ${resolved.candidates.map((s) => `${s}__${tool}`).join(', ')}.`,
1018
+ );
1019
+ }
1020
+ throw new Error(
1021
+ `Unqualified tool call name "${call.function.name}". Tool calls must use the <server>__<tool> `
1022
+ + `naming convention (e.g. production__production_start_job); no connected server exposes a tool named "${tool}".`,
1023
+ );
1024
+ }
799
1025
  let args = JSON.parse(call.function.arguments ?? '{}');
800
1026
  if (server === 'production' && tool === 'production_start_job' && state.session.workspace && !args.callerLabel) {
801
1027
  args = { ...args, callerLabel: `${state.session.workspace}/wiki-manager` };
@@ -811,6 +1037,8 @@ export function createAgentGraph(options = {}) {
811
1037
  } else if (server === 'shell' && tool === 'profile_update') {
812
1038
  const result = await updateWorkspaceProfilePreference(state.session, args.preference);
813
1039
  resultText = JSON.stringify(result, null, 2);
1040
+ } else if (server === 'runtime') {
1041
+ resultText = await handleRuntimeControlTool(state.session, tool, args);
814
1042
  } else if (server !== 'shell') {
815
1043
  await awaitRunApproval(state.session, { runId, tool: toolName });
816
1044
  await awaitToolApproval(state.session, {
@@ -833,6 +1061,41 @@ export function createAgentGraph(options = {}) {
833
1061
  resultText = formatMcpToolResult(result);
834
1062
  }
835
1063
  }
1064
+ {
1065
+ const payload = parseJsonText(resultText);
1066
+ // agent_plan (any provider) returned a task-graph fragment: declare it as
1067
+ // the plan DETERMINISTICALLY. Asking the LLM to copy N tasks into
1068
+ // wiki__plan_set would lose fields (a small local model dropped
1069
+ // arguments/operations in testing) — the shell does the mapping.
1070
+ if (/(^|__)agent_plan$/.test(tool) && Array.isArray(payload?.tasks) && payload.tasks.length > 0) {
1071
+ const steps = payload.tasks.map((task, index) => normalizeDeclaredPlanStep({
1072
+ id: task.id,
1073
+ description: task.label ?? task.id ?? `Task ${index + 1}`,
1074
+ requiredCapability: task.requiredCapability ?? payload.capability ?? null,
1075
+ operation: task.operation ?? null,
1076
+ arguments: task.arguments ?? {},
1077
+ dependsOn: task.dependsOn ?? [],
1078
+ outputRefs: (task.expectedOutputRefs ?? []).map((ref) => (ref && typeof ref === 'object' ? String(ref.ref ?? '') : String(ref))).filter(Boolean),
1079
+ groupId: task.groupId ?? null,
1080
+ dependsOnGroup: task.dependsOnGroup ?? null,
1081
+ parallelizable: task.parallelizable,
1082
+ barrier: task.barrier,
1083
+ locks: task.locks,
1084
+ requiresApproval: task.requiresApproval,
1085
+ approvalClass: task.approvalClass,
1086
+ approvalSummary: task.approvalSummary,
1087
+ idempotencyKey: task.idempotencyKey,
1088
+ progressWeight: task.progressWeight,
1089
+ recommendedConcurrency: task.recommendedConcurrency,
1090
+ }, index, state.session));
1091
+ emitAgentEvent(state.session, 'plan_set', 'tool', { steps });
1092
+ state.session._onStep?.(`Plan: ${steps.length} task(s) declared from ${server} fragment`);
1093
+ resultText = `Task-graph fragment integrated as the current plan (${steps.length} task(s), groups: ${[...new Set(payload.tasks.map((task) => task.groupId).filter(Boolean))].join(', ') || 'none'}). The orchestrator will dispatch these tasks — do NOT call production tools for them yourself. Reply with a short summary and wait.`;
1094
+ }
1095
+ if (/(^|__)agent_plan$/.test(tool) && Array.isArray(payload?.tasks) && payload.tasks.length > 0) {
1096
+ minimalPlanActive = false;
1097
+ }
1098
+ }
836
1099
  if (server === 'production') {
837
1100
  let payload = parseJsonText(resultText);
838
1101
  if (tool === 'production_start_job' && payload?.ok === false && payload?.error === 'workspace_busy') {
@@ -869,22 +1132,26 @@ export function createAgentGraph(options = {}) {
869
1132
  err.name === 'ApprovalError'
870
1133
  ) throw err;
871
1134
  ok = false;
872
- resultText = `Error [${server}.${tool}]: ${err instanceof Error ? err.message : String(err)}`;
1135
+ resultText = `Error [${toolName}]: ${err instanceof Error ? err.message : String(err)}`;
873
1136
  if (minimalPlanActive && state.session.headlessPlan?.[0]?._activityKey === null) {
874
1137
  emitAgentEvent(state.session, 'plan_step_updated', 'tool', { step: 1, status: 'failed' });
875
1138
  }
876
1139
  }
1140
+ // Bound the result at its two exit points only (LLM context + display).
1141
+ // The full resultText above was already used for payload parsing and
1142
+ // _activity extraction, which must never see a truncated document.
1143
+ const boundedResult = truncateToolResult(resultText);
877
1144
  emitAgentEvent(state.session, 'tool_call_result', 'tool', {
878
1145
  callId: call.id,
879
1146
  name: toolName,
880
1147
  ok,
881
- result: resultText,
1148
+ result: boundedResult,
882
1149
  summary: ok ? 'done' : 'failed',
883
1150
  });
884
1151
  toolResultMessages.push({
885
1152
  role: 'tool',
886
1153
  tool_call_id: call.id,
887
- content: resultText,
1154
+ content: boundedResult,
888
1155
  });
889
1156
  }
890
1157
 
@@ -899,11 +1166,22 @@ export function createAgentGraph(options = {}) {
899
1166
  return END;
900
1167
  }
901
1168
 
902
- return new StateGraph(AgentState)
1169
+ const compiled = new StateGraph(AgentState)
903
1170
  .addNode('orchestrator', orchestratorNode)
904
1171
  .addNode('tool_executor', toolExecutorNode)
905
1172
  .addEdge(START, 'orchestrator')
906
1173
  .addConditionalEdges('orchestrator', routeOrchestrator)
907
1174
  .addEdge('tool_executor', 'orchestrator')
908
1175
  .compile();
1176
+
1177
+ // LangGraph's default recursionLimit is 25 super-steps. Each tool round
1178
+ // costs two of them (orchestrator + tool_executor), so runs died with
1179
+ // GRAPH_RECURSION_LIMIT around iteration 12 — far below the intended
1180
+ // MAX_TOOL_ITERATIONS budget — before finishing their work (observed as
1181
+ // `knowledge.update — error 0%` with no production job ever created).
1182
+ // Bake a limit matching the iteration budget into every invocation.
1183
+ const recursionLimit = MAX_TOOL_ITERATIONS * 2 + 10;
1184
+ return {
1185
+ invoke: (state, config = {}) => compiled.invoke(state, { recursionLimit, ...config }),
1186
+ };
909
1187
  }