@dotdrelle/wiki-manager 0.12.1 → 0.12.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/README.md +49 -8
  2. package/bin/wiki-manager +23 -12
  3. package/docker-compose.yml +0 -1
  4. package/mcp.endpoints.example.json +0 -6
  5. package/package.json +3 -2
  6. package/src/activity/activityAggregator.js +34 -3
  7. package/src/activity/activityAggregator.test.js +32 -0
  8. package/src/agent/graph.js +272 -22
  9. package/src/agent/graph.test.js +357 -1
  10. package/src/cli/wiki-manager.js +33 -2
  11. package/src/commands/slash.js +74 -1
  12. package/src/commands/slash.test.js +36 -0
  13. package/src/core/activity.js +4 -0
  14. package/src/core/activity.test.js +9 -1
  15. package/src/core/agentEvents.js +30 -0
  16. package/src/core/agentEvents.test.js +42 -1
  17. package/src/core/buildInfo.js +58 -0
  18. package/src/core/buildInfo.json +4 -0
  19. package/src/core/buildInfo.test.js +14 -0
  20. package/src/core/env.js +30 -1
  21. package/src/core/mcp.js +21 -1
  22. package/src/core/mcp.test.js +25 -0
  23. package/src/core/runtimeLog.js +6 -1
  24. package/src/core/runtimeLog.test.js +3 -1
  25. package/src/runtime/client.js +22 -0
  26. package/src/runtime/controlMessages.js +2 -2
  27. package/src/runtime/recoveryManager.js +54 -0
  28. package/src/runtime/recoveryManager.test.js +73 -0
  29. package/src/runtime/runner.js +49 -13
  30. package/src/runtime/runner.test.js +196 -0
  31. package/src/runtime/server.js +67 -8
  32. package/src/runtime/server.test.js +223 -6
  33. package/src/runtime/store.js +48 -3
  34. package/src/runtime/store.test.js +32 -0
  35. package/src/runtime/supervisor.js +77 -1
  36. package/src/shell/RightPane.tsx +124 -31
  37. package/src/shell/SetupWizard.tsx +13 -1
  38. package/src/shell/repl.js +78 -9
  39. package/src/shell/repl.test.js +114 -1
  40. package/src/shell/tui.tsx +4 -1
  41. package/src/shell/useAgent.ts +32 -3
  42. package/src/shell/useSession.ts +14 -1
  43. package/wiki-workspace +33 -0
@@ -6,6 +6,7 @@ import {
6
6
  formatMcpToolResult,
7
7
  formatMcpToolsForAgent,
8
8
  resolveToolCallName,
9
+ truncateToolResult,
9
10
  } from '../core/mcp.js';
10
11
  import { formatSkillsForAgent, readOptionalText } from '../core/skills.js';
11
12
  import { handleSlashCommand } from '../commands/slash.js';
@@ -13,6 +14,8 @@ import { extractActivity, formatActivitySummary, parseJsonText, sessionActivitie
13
14
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
14
15
  import { enqueueProductionJob, ensureJobQueue, formatQueue, productionLockBusy } from '../core/jobQueue.js';
15
16
  import { updateWorkspaceProfilePreference } from '../core/profile.js';
17
+ import { createCapabilityRegistry } from '../orchestrator/capabilityRegistry.js';
18
+ import { fetchRuntimeState, postRuntimeCancel, postRuntimeControl, postRuntimeKill } from '../runtime/client.js';
16
19
 
17
20
  const MAX_TOOL_ITERATIONS = 80;
18
21
  const MAX_SPINNER_ARG_LENGTH = 96;
@@ -24,6 +27,7 @@ const MAX_PROFILE_CHARS = 4000;
24
27
  const INTERNAL_TOOL_SERVERS = {
25
28
  wiki: ['plan_set', 'plan_done'],
26
29
  shell: ['run_command', 'read_command', 'profile_update'],
30
+ runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue'],
27
31
  };
28
32
 
29
33
  const AGENT_SLASH_COMMANDS = new Set([
@@ -108,6 +112,59 @@ const SHELL_PROFILE_UPDATE_TOOL = {
108
112
  },
109
113
  };
110
114
 
115
+ // Runtime control tools: Donna interprets the user's intent ("supprime le
116
+ // job et la queue", "arrête tout", "où en est le run") and ACTS through
117
+ // these, instead of a hardcoded regex classifier answering with canned text.
118
+ const RUNTIME_KILL_TOOL = {
119
+ type: 'function',
120
+ function: {
121
+ name: 'runtime__kill',
122
+ description: 'Hard-stop the workspace runtime: abort the active run, cancel its agent jobs, mark persisted runs interrupted and purge the control queue. Use when the user asks to remove/kill/clean the current run, its jobs or the queue.',
123
+ parameters: { type: 'object', additionalProperties: false, properties: { runId: { type: 'string', description: 'Optional specific run id; omit to kill everything active in the workspace.' } } },
124
+ },
125
+ };
126
+
127
+ const RUNTIME_CANCEL_TOOL = {
128
+ type: 'function',
129
+ function: {
130
+ name: 'runtime__cancel',
131
+ description: 'Soft-cancel the active runtime run (graceful abort, no queue purge). Use for "annule le run" when the user does not ask to clean everything.',
132
+ parameters: { type: 'object', additionalProperties: false, properties: {} },
133
+ },
134
+ };
135
+
136
+ const RUNTIME_STATUS_TOOL = {
137
+ type: 'function',
138
+ function: {
139
+ name: 'runtime__status',
140
+ description: 'Read the runtime state: active run, plan steps, queue items, approvals. Use to answer questions about what is currently running or queued.',
141
+ parameters: { type: 'object', additionalProperties: false, properties: {} },
142
+ },
143
+ };
144
+
145
+ const RUNTIME_APPROVE_TOOL = {
146
+ type: 'function',
147
+ function: {
148
+ name: 'runtime__approve',
149
+ description: 'Grant the pending approval of the active runtime run (mutating tasks wait on it). Use when the user consents in ANY phrasing: "vas-y", "ok pour l\'export", "approuve", "valide". Confirm what was approved.',
150
+ parameters: { type: 'object', additionalProperties: false, properties: {} },
151
+ },
152
+ };
153
+
154
+ const RUNTIME_ENQUEUE_TOOL = {
155
+ type: 'function',
156
+ function: {
157
+ name: 'runtime__enqueue',
158
+ description: 'Queue a request to run AFTER the currently active runtime run finishes. Use when the user asks for a new action while a run is active and wants it done afterwards.',
159
+ parameters: {
160
+ type: 'object',
161
+ additionalProperties: false,
162
+ properties: { input: { type: 'string', description: 'The request to execute after the current run, phrased as a complete instruction.' } },
163
+ required: ['input'],
164
+ },
165
+ },
166
+ };
167
+
111
168
  const WIKI_PLAN_SET_TOOL = {
112
169
  type: 'function',
113
170
  function: {
@@ -136,6 +193,8 @@ const WIKI_PLAN_SET_TOOL = {
136
193
  status: { type: 'string', enum: ['pending', 'queued', 'running', 'waiting', 'pending_approval', 'done', 'failed', 'cancelled', 'stalled', 'added_during_run'] },
137
194
  dependsOn: { type: 'array', items: { type: 'string' } },
138
195
  outputRefs: { type: 'array', items: { type: 'string' } },
196
+ operation: { type: ['string', 'null'], description: 'Operation for the capability provider (e.g. ingest_plan, build).' },
197
+ arguments: { type: 'object', description: 'Arguments passed to the provider agent_execute for this step.' },
139
198
  },
140
199
  required: ['description'],
141
200
  },
@@ -448,9 +507,87 @@ function emitAgentEvent(session, type, origin, payload = {}) {
448
507
  dispatchAgentEvent(session, createAgentEvent(type, { origin, payload }));
449
508
  }
450
509
 
510
+ // Capability ids currently provided by discovered, orchestrable, healthy
511
+ // agents. This is the live registry the dispatcher will resolve against —
512
+ // a plan declaring anything outside this set can only stall forever.
513
+ export function knownCapabilityIds(session) {
514
+ const registry = session?.capabilityRegistry ?? createCapabilityRegistry({
515
+ agents: session?.agentRegistrySnapshot ?? session?.agents ?? [],
516
+ });
517
+ const snapshot = typeof registry.snapshot === 'function' ? registry.snapshot() : registry;
518
+ return [...new Set(Object.keys(snapshot ?? {}).map((key) => {
519
+ const index = key.lastIndexOf('@');
520
+ return index > 0 ? key.slice(0, index) : key;
521
+ }))].sort();
522
+ }
523
+
524
+ async function handleRuntimeControlTool(session, tool, args = {}) {
525
+ const url = session.runtime?.url ?? null;
526
+ if (!url) return 'Runtime not connected: no runtime URL available in this session.';
527
+ const workspace = session.workspace ?? null;
528
+ try {
529
+ if (tool === 'kill') {
530
+ const result = await postRuntimeKill({ url, workspace, runId: args.runId ?? null });
531
+ return `Runtime killed: ${result.runs ?? 0} run(s) interrupted, ${result.tasks ?? 0} task(s) cancelled, ${result.queued ?? 0} queued control request(s) purged.`;
532
+ }
533
+ if (tool === 'cancel') {
534
+ const result = await postRuntimeCancel({ url, workspace });
535
+ return result.cancelled ? 'Runtime run cancellation requested.' : `No active run to cancel${result.reason ? ` (${result.reason})` : ''}.`;
536
+ }
537
+ if (tool === 'approve') {
538
+ const result = await postRuntimeControl('message', { url, workspace, input: 'approve', intent: 'approve' });
539
+ return String(result?.explanation ?? (result?.accepted ? 'Approval granted.' : 'No pending approval found.'));
540
+ }
541
+ if (tool === 'enqueue') {
542
+ const result = await postRuntimeControl('message', { url, workspace, input: String(args.input ?? ''), intent: 'enqueue' });
543
+ return String(result?.explanation ?? 'Request queued for after the current run.');
544
+ }
545
+ if (tool === 'status') {
546
+ const state = await fetchRuntimeState({ url, workspace });
547
+ const plan = Array.isArray(state?.plan) ? state.plan : [];
548
+ const queue = Array.isArray(state?.queue) ? state.queue : [];
549
+ const controlQueue = Array.isArray(state?.controlQueue) ? state.controlQueue : [];
550
+ return JSON.stringify({
551
+ status: state?.status ?? 'unknown',
552
+ running: Boolean(state?.running),
553
+ runId: state?.runId ?? null,
554
+ plan: plan.map((step) => ({ id: step.id ?? step.step, description: step.description, status: step.status })),
555
+ queue: queue.map((item) => ({ id: item.id, status: item.status, tool: item.tool ?? item.type ?? null })),
556
+ controlQueue: controlQueue.filter((item) => item.status === 'queued').map((item) => ({ id: item.id, input: item.input })),
557
+ pendingApprovals: (Array.isArray(state?.approvals) ? state.approvals : [])
558
+ .filter((approval) => approval.status === 'pending_approval')
559
+ .map((approval) => ({ id: approval.id, reason: approval.reason ?? null })),
560
+ }, null, 2);
561
+ }
562
+ return `Unknown runtime tool: ${tool}`;
563
+ } catch (err) {
564
+ return `Runtime control error (${tool}): ${err instanceof Error ? err.message : String(err)}`;
565
+ }
566
+ }
567
+
451
568
  function handleWikiTool(session, tool, args) {
452
569
  if (tool === 'plan_set') {
453
570
  const steps = Array.isArray(args.steps) ? args.steps : [];
571
+ // Reject fantasy capabilities BEFORE the plan exists: once registered,
572
+ // unresolvable steps become tasks that wait forever and flood the queue.
573
+ // A tool-level error (not an exception) lets the LLM correct itself in
574
+ // the same turn. An empty registry (discovery not done yet) skips the
575
+ // check rather than blocking legitimate early plans.
576
+ const known = knownCapabilityIds(session);
577
+ if (known.length > 0) {
578
+ const unknown = [...new Set(steps
579
+ .map((step) => (step && typeof step === 'object' ? step.requiredCapability : null))
580
+ .filter(Boolean)
581
+ .map(String)
582
+ .filter((capability) => !known.includes(capability.includes('@') ? capability.slice(0, capability.lastIndexOf('@')) : capability)))];
583
+ if (unknown.length > 0) {
584
+ return `Plan rejected: unknown capabilities [${unknown.join(', ')}]. `
585
+ + `Available capabilities: ${known.join(', ')}. `
586
+ + 'Redeclare the plan using only available capabilities, or use requiredCapability: null for a step you execute yourself.';
587
+ }
588
+ } else if (steps.some((step) => step && typeof step === 'object' && step.requiredCapability)) {
589
+ session._onStep?.('plan_set: capability registry empty, validation skipped');
590
+ }
454
591
  emitAgentEvent(session, 'plan_set', 'tool', {
455
592
  steps: steps.map((raw, i) => normalizeDeclaredPlanStep(raw, i, session)),
456
593
  });
@@ -483,6 +620,21 @@ function normalizeDeclaredPlanStep(raw, index) {
483
620
  executor: null,
484
621
  executorQuery: null,
485
622
  outputRefs: Array.isArray(item.outputRefs) ? item.outputRefs.map(String) : [],
623
+ // Execution fields the deterministic dispatcher consumes (agent_execute):
624
+ // without them a capability step cannot actually run.
625
+ ...(item.operation != null ? { operation: String(item.operation) } : {}),
626
+ ...(item.arguments && typeof item.arguments === 'object' ? { arguments: item.arguments } : {}),
627
+ ...(item.groupId != null ? { groupId: String(item.groupId) } : {}),
628
+ ...(item.dependsOnGroup != null ? { dependsOnGroup: String(item.dependsOnGroup) } : {}),
629
+ ...(item.parallelizable != null ? { parallelizable: Boolean(item.parallelizable) } : {}),
630
+ ...(item.barrier ? { barrier: true } : {}),
631
+ ...(item.locks ? { locks: item.locks } : {}),
632
+ ...(item.requiresApproval != null ? { requiresApproval: Boolean(item.requiresApproval) } : {}),
633
+ ...(item.approvalClass ? { approvalClass: String(item.approvalClass) } : {}),
634
+ ...(item.approvalSummary ? { approvalSummary: String(item.approvalSummary) } : {}),
635
+ ...(item.idempotencyKey ? { idempotencyKey: String(item.idempotencyKey) } : {}),
636
+ ...(item.progressWeight != null ? { progressWeight: Number(item.progressWeight) } : {}),
637
+ ...(item.recommendedConcurrency != null ? { recommendedConcurrency: Number(item.recommendedConcurrency) } : {}),
486
638
  };
487
639
  }
488
640
 
@@ -547,9 +699,16 @@ export function buildAgentSystemPrompt(state) {
547
699
  'Prefer MCP tools that declare their own plan via _activity.plan.steps — when such a tool returns _activity, the shell creates and tracks the plan automatically without requiring wiki__plan_set.',
548
700
  'Use wiki__plan_set when the MCP tool cannot declare its own plan or when the task spans multiple independent tools (e.g. CME export then email report). For a single self-describing async job, wiki__plan_set is optional.',
549
701
  '',
702
+ (() => {
703
+ const capabilityIds = knownCapabilityIds(state.session);
704
+ return capabilityIds.length > 0
705
+ ? `Known orchestration capabilities — the ONLY values allowed in requiredCapability: ${capabilityIds.join(', ')}. Never invent capability names; a plan declaring an unknown capability will be rejected. A step you execute yourself directly takes requiredCapability: null.`
706
+ : 'No orchestration capabilities discovered yet: declare plan steps with requiredCapability: null and execute them yourself with the connected MCP tools.';
707
+ })(),
708
+ '',
550
709
  'Task startup:',
551
710
  ' 1. If the next MCP tool returns _activity.plan.steps, call that tool directly; the shell will create the visible plan from the returned activity.',
552
- ' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, dependsOn, outputRefs}; a legacy list of strings is still accepted.',
711
+ ' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, operation, arguments, dependsOn, outputRefs}; capability steps need operation+arguments for the dispatcher to execute them; a legacy list of strings is still accepted.',
553
712
  ' Multi-tool example: wiki__plan_set(steps=[{id:"cme-export",description:"CME export",requiredCapability:"external-source.export",dependsOn:[],outputRefs:["raw/untracked"]},{id:"production",description:"Production pipeline",requiredCapability:"knowledge.pipeline",dependsOn:["cme-export"],outputRefs:["deliverables"]}])',
554
713
  ' 3. Immediately execute the first step using the appropriate MCP tool. Do not start step 2 in the same turn unless one async pipeline tool owns and declares the whole sequence.',
555
714
  ' For synchronous steps (result is immediate, no _activity polling), call wiki__plan_done(step=1) after confirming success.',
@@ -569,17 +728,21 @@ export function buildAgentSystemPrompt(state) {
569
728
  'On failure: if a completed activity is failed/error/cancelled, call wiki__plan_done(step=N, status="failed") then stop with a clear error report.',
570
729
  ].filter(Boolean).join('\n'),
571
730
  'For service actions, recommend only available service primitives from Available primitives, with the exact service name when the primitive supports one.',
731
+ 'Scope discipline: execute ONLY the action(s) the user explicitly requested. Never chain additional mutating operations (ingest, build, export, polish, delete, send…) that the user did not ask for — even when diagnostics or recommendations suggest them. Finish the requested work, then list the suggested follow-ups in your final answer and stop. Example: "applique les recommandations de config" means apply the config; it does NOT authorize launching the ingest those recommendations mention.',
572
732
  'Disambiguate export requests carefully.',
573
733
  'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`cme__cme_export_run`, then `cme__cme_export_status`). Never use production `type=export` for Confluence source export.',
574
734
  'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`production__production_start_job` with `type:"export"` or pipeline steps). Require the deliverable path when exporting deliverables.',
575
- 'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts. To chain multiple sequential steps (e.g. build then polish), always use a single production__production_start_job call with type="pipeline" and steps=["build","polish"] — never start them as separate jobs: the first job is asynchronous and the second would run before it completes. For existing deliverables where content stability matters, pass stabilize:true so the build step preserves unchanged sections; keep polish in the pipeline when publication output is requested. Do not ask the user to confirm between steps; start the pipeline call directly.',
735
+ 'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts.',
736
+ 'MULTI-DOCUMENT ingest (more than 2 files, or "ingest everything pending"): call production__agent_plan first, e.g. {capability:"knowledge.update", operation:"ingest", constraints:{maxConcurrency:3, requireApprovalForMutations:true}}. The shell integrates the returned task graph as the plan automatically and the orchestrator dispatches the per-document tasks IN PARALLEL with an approval gate. Do not call production__production_start_job for multi-document ingest — that creates one monolithic sequential job.',
737
+ 'Single-document ingest or one-off jobs (doctor, one build, one export): production__production_start_job is fine. To chain sequential steps (e.g. build then polish) use ONE call with type="pipeline" and steps=["build","polish"] — never separate jobs (the first is asynchronous). For existing deliverables where content stability matters, pass stabilize:true. Do not ask the user to confirm between steps.',
576
738
  'Long-running MCP jobs: do not call the same status tool more than once consecutively. When chaining jobs sequentially: (1) start the job, report job/activity id and status; (2) check status once — if done, proceed to the next step immediately; (3) if still running, report status, list the remaining steps, and return control; (4) when re-invoked, check status first, then proceed. Do not spin-poll (status → status → status with no new action between). The shell activity panel monitors non-terminal jobs automatically.',
577
739
  'If production__production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
578
- 'For diagnostics, use /wiki run doctor when the user asks for doctor. Use /workspace init <name> [path] for low-level non-interactive workspace creation. In the interactive TUI, /new <name> opens the setup wizard. Use /wiki for index, or /wiki run index through the explicit backup hatch. Use /wiki run init only for explicit current-workspace llm-wiki init.',
740
+ 'For diagnostics (doctor), use production__production_start_job with type="doctor" like any other production job; /wiki run doctor is only the fallback when the production MCP is not connected. Use /workspace init <name> [path] for low-level non-interactive workspace creation. In the interactive TUI, /new <name> opens the setup wizard. Use /wiki for index, or /wiki run index through the explicit backup hatch. Use /wiki run init only for explicit current-workspace llm-wiki init.',
579
741
  'If an action requires tools or skills not available yet, explain the limitation and name the expected primitive.',
580
742
  workspaceProfile
581
743
  ? `Workspace profile (.wiki/profile.md) — durable user preferences, apply these to every reply (tone, tutoiement/vouvoiement, formatting, etc.):\n${workspaceProfile}`
582
744
  : null,
745
+ 'Runtime control: you have runtime__status, runtime__cancel, runtime__kill, runtime__approve and runtime__enqueue. When the user asks to stop, remove, clean or kill the current run, its jobs or the queue ("supprime le job et la queue", "arr\u00eate tout"), call runtime__kill (or runtime__cancel for a soft stop of just the run) and confirm what was stopped. For questions about what is running or queued, call runtime__status and answer from its data. When the user consents to a pending approval in any phrasing ("vas-y", "ok pour l\'export"), call runtime__approve. When the user asks for a NEW action while a run is active, do not execute it: propose runtime__enqueue (run it after) or, if they insist it replaces the current work, runtime__kill then the new action.',
583
746
  'When the user explicitly asks you to remember, persist, or update durable preference/profile information, call wiki__profile_update when it is available; otherwise call shell__profile_update. Do not just acknowledge in text without calling a profile update tool.',
584
747
  ].filter(Boolean).join('\n');
585
748
 
@@ -610,20 +773,35 @@ export function formatLlmUnavailableMessage(reason) {
610
773
  return `⚠ LLM injoignable : ${clean || 'raison inconnue'}`;
611
774
  }
612
775
 
613
- function classifyAgentInput(input, session) {
776
+ // Verbs that clearly request work (a runtime run), in French and English.
777
+ // "configure/configurer" is an action; the nouns "config/configuration" are
778
+ // NOT matched here — asking for a config is an observe request.
779
+ const ACTION_REQUEST_PATTERN = /\b(lance|relance|d[eé]marre|start|ex[eé]cute|execute|g[eé]n[eè]re|generate|build|construis|exporte?|ingest\w*|ing[eè]re|importe?|convert(?:is|it|s)?|cr[eé]e|create|polish|publie|publish|d[eé]ploie|deploy|envoie|send|configure[rsz]?|setup|installe|update|mets? [aà] jour|supprime|delete|efface|nettoie|clean|r[eé]pare|fix|corrige)\b/i;
780
+
781
+ // Explicit explanation/question markers dominate action verbs: "explique le
782
+ // build" is a question about the build, not a request to build.
783
+ const EXPLANATION_REQUEST_PATTERN = /\b(explique|explain|pourquoi|why|comment|how|c'est quoi|qu'est[- ]ce)\b/i;
784
+
785
+ export function classifyAgentInput(input, session) {
614
786
  const lower = String(input ?? '').toLowerCase();
615
787
  const hasActiveRun = session?.agentProjection?.status === 'running'
616
788
  || sessionActivities(session).some((activity) => !activity.terminal);
617
789
  if (/\b(valide tout|approve all|approve|approuve|valid[eé]|ok pour tout|go pour tout)\b/i.test(lower)) {
618
790
  return { kind: 'approve', confidence: 0.86, reason: 'approval_request', activeRun: hasActiveRun };
619
791
  }
620
- if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort)\b/i.test(lower)) {
792
+ if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort|supprime|kill|tue|purge|vide la (file|queue)|nettoie la (file|queue))\b/i.test(lower)) {
621
793
  return { kind: 'cancel', confidence: 0.86, reason: 'cancel_request', activeRun: hasActiveRun };
622
794
  }
623
795
  if (/\b(plus tard|later|ensuite|apr[eè]s ce run|enqueue|mets en file|met en file|futur|next run|future run)\b/i.test(lower)) {
624
796
  return { kind: 'enqueue_run', confidence: 0.8, reason: 'future_run_request', activeRun: hasActiveRun };
625
797
  }
626
- if (/\b(o[uù] en es[t-]|status|statut|progress|progression|run|job|queue|logs?|explique|explain|inspect|show|montre|quoi de neuf)\b/i.test(lower)) {
798
+ if (EXPLANATION_REQUEST_PATTERN.test(lower)) {
799
+ return { kind: 'observe', confidence: 0.86, reason: 'explanation_request', activeRun: hasActiveRun };
800
+ }
801
+ // Observe markers only win when no action verb is present: "où en est le
802
+ // run" is observe, "lance le run" is an action request.
803
+ if (/\b(o[uù] en es[t-]|status|statut|progress|progression|run|job|queue|logs?|inspect|show|montre|affiche|donne|liste|list|quel(?:le)?s?|combien|config(?:uration)?|quoi de neuf)\b/i.test(lower)
804
+ && !ACTION_REQUEST_PATTERN.test(lower)) {
627
805
  return { kind: 'observe', confidence: 0.86, reason: 'status_or_explanation_request', activeRun: hasActiveRun };
628
806
  }
629
807
  if (hasActiveRun && /\b(ajoute|add|change|modifie|modify|remplace|replace|retire|remove|skip|ignore|plan|step|t[aâ]che)\b/i.test(lower)) {
@@ -632,12 +810,24 @@ function classifyAgentInput(input, session) {
632
810
  if (hasActiveRun && /\b(lance|run|g[eé]n[eè]re|build|export|cr[eé]e|create|send|envoie|ingest|convert|importe|import)\b/i.test(lower)) {
633
811
  return { kind: 'ambiguous', confidence: 0.45, reason: 'active_run_action_is_ambiguous', activeRun: hasActiveRun };
634
812
  }
813
+ if (ACTION_REQUEST_PATTERN.test(lower)) {
814
+ return { kind: 'start_run', confidence: 0.8, reason: 'action_request', activeRun: hasActiveRun };
815
+ }
635
816
  return { kind: 'converse', confidence: 0.62, reason: 'plain_conversation', activeRun: hasActiveRun };
636
817
  }
637
818
 
638
- function toolsForClassification(classification, writeTools) {
639
- if (classification.activeRun && ['converse', 'observe'].includes(classification.kind)) return [SHELL_READ_COMMAND_TOOL];
640
- return [SHELL_READ_COMMAND_TOOL, ...writeTools];
819
+ function toolsForClassification(classification, writeTools, session = null) {
820
+ const controlTools = session?.runtime?.url
821
+ ? [RUNTIME_STATUS_TOOL, RUNTIME_CANCEL_TOOL, RUNTIME_KILL_TOOL, RUNTIME_APPROVE_TOOL, RUNTIME_ENQUEUE_TOOL]
822
+ : [];
823
+ if (classification.activeRun && ['converse', 'observe', 'ambiguous', 'approve', 'cancel', 'enqueue_run'].includes(classification.kind)) {
824
+ // During an active run Donna gets read + profile + the runtime control
825
+ // suite: she can answer, approve, enqueue for later, soft-cancel or
826
+ // kill — but she must not fire new MCP jobs alongside the run (that is
827
+ // what runtime__enqueue is for). No canned regex answers anywhere.
828
+ return [SHELL_READ_COMMAND_TOOL, SHELL_PROFILE_UPDATE_TOOL, ...controlTools];
829
+ }
830
+ return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...writeTools];
641
831
  }
642
832
 
643
833
  export function createAgentGraph(options = {}) {
@@ -663,8 +853,17 @@ export function createAgentGraph(options = {}) {
663
853
  state.session._onStep?.('Agent: planning next action…');
664
854
  }
665
855
 
856
+ // Inside a runtime run the input IS the task to execute (the runtime
857
+ // already accepted it as a run): the interactive control-message
858
+ // classifier must not apply. Without this, agentProjection.status is
859
+ // 'running' during every run, so any action verb ("lance l'ingestion")
860
+ // matched the active-run 'ambiguous' branch and returned a canned
861
+ // clarification instead of executing — the run ended silently.
862
+ const runtimeExecution = Boolean(state.session._currentRunIdentity);
666
863
  const classification = iterations === 0
667
- ? classifyAgentInput(state.input, state.session)
864
+ ? (runtimeExecution
865
+ ? { kind: 'execute_run', confidence: 1, reason: 'runtime_run_execution', activeRun: true }
866
+ : classifyAgentInput(state.input, state.session))
668
867
  : (state.inputClassification ?? { kind: 'modify_run', confidence: 1, reason: 'tool_iteration' });
669
868
  if (iterations === 0) {
670
869
  state.session._onStep?.(`Agent: classified input as ${classification.kind}`);
@@ -673,14 +872,6 @@ export function createAgentGraph(options = {}) {
673
872
  classification,
674
873
  });
675
874
  }
676
- if (iterations === 0 && classification.kind === 'ambiguous') {
677
- return {
678
- response: 'Je peux répondre sur le run en cours, modifier ce run, mettre une nouvelle demande en file, approuver, ou annuler. Peux-tu préciser ce que tu veux faire ?',
679
- pendingToolCalls: null,
680
- readyToStream: false,
681
- inputClassification: classification,
682
- };
683
- }
684
875
 
685
876
  const writeTools = [
686
877
  SHELL_RUN_COMMAND_TOOL,
@@ -689,7 +880,7 @@ export function createAgentGraph(options = {}) {
689
880
  WIKI_PLAN_DONE_TOOL,
690
881
  ...buildLlmTools(state.session.mcp),
691
882
  ];
692
- const tools = toolsForClassification(classification, writeTools);
883
+ const tools = toolsForClassification(classification, writeTools, state.session);
693
884
  const system = buildAgentSystemPrompt(state);
694
885
 
695
886
  // On iteration 0: prior history is in state.messages, user input must be appended.
@@ -721,6 +912,13 @@ export function createAgentGraph(options = {}) {
721
912
 
722
913
  if (result.tool_calls?.length > 0) {
723
914
  state.session._onStreamReset?.();
915
+ // Close the streaming conversation entry now: the text streamed so
916
+ // far is this iteration's narration. Without this, the next
917
+ // iteration's deltas append to the SAME entry with no separator and
918
+ // the chat becomes one glued wall of text ("…de la config.Voyons…").
919
+ // An empty finalize keeps the accumulated content and just drops the
920
+ // streaming flag; it is a no-op when nothing was streamed.
921
+ emitAgentEvent(state.session, 'assistant_message', 'llm', { content: '' });
724
922
  state.session._onStep?.(`[${iterations + 1}/${MAX_TOOL_ITERATIONS}] ${result.tool_calls.length} MCP action${result.tool_calls.length > 1 ? 's' : ''} queued…`);
725
923
  // On iteration 0 persist the user message too so it survives the loop.
726
924
  const newMessages = iterations === 0
@@ -839,6 +1037,8 @@ export function createAgentGraph(options = {}) {
839
1037
  } else if (server === 'shell' && tool === 'profile_update') {
840
1038
  const result = await updateWorkspaceProfilePreference(state.session, args.preference);
841
1039
  resultText = JSON.stringify(result, null, 2);
1040
+ } else if (server === 'runtime') {
1041
+ resultText = await handleRuntimeControlTool(state.session, tool, args);
842
1042
  } else if (server !== 'shell') {
843
1043
  await awaitRunApproval(state.session, { runId, tool: toolName });
844
1044
  await awaitToolApproval(state.session, {
@@ -861,6 +1061,41 @@ export function createAgentGraph(options = {}) {
861
1061
  resultText = formatMcpToolResult(result);
862
1062
  }
863
1063
  }
1064
+ {
1065
+ const payload = parseJsonText(resultText);
1066
+ // agent_plan (any provider) returned a task-graph fragment: declare it as
1067
+ // the plan DETERMINISTICALLY. Asking the LLM to copy N tasks into
1068
+ // wiki__plan_set would lose fields (a small local model dropped
1069
+ // arguments/operations in testing) — the shell does the mapping.
1070
+ if (/(^|__)agent_plan$/.test(tool) && Array.isArray(payload?.tasks) && payload.tasks.length > 0) {
1071
+ const steps = payload.tasks.map((task, index) => normalizeDeclaredPlanStep({
1072
+ id: task.id,
1073
+ description: task.label ?? task.id ?? `Task ${index + 1}`,
1074
+ requiredCapability: task.requiredCapability ?? payload.capability ?? null,
1075
+ operation: task.operation ?? null,
1076
+ arguments: task.arguments ?? {},
1077
+ dependsOn: task.dependsOn ?? [],
1078
+ outputRefs: (task.expectedOutputRefs ?? []).map((ref) => (ref && typeof ref === 'object' ? String(ref.ref ?? '') : String(ref))).filter(Boolean),
1079
+ groupId: task.groupId ?? null,
1080
+ dependsOnGroup: task.dependsOnGroup ?? null,
1081
+ parallelizable: task.parallelizable,
1082
+ barrier: task.barrier,
1083
+ locks: task.locks,
1084
+ requiresApproval: task.requiresApproval,
1085
+ approvalClass: task.approvalClass,
1086
+ approvalSummary: task.approvalSummary,
1087
+ idempotencyKey: task.idempotencyKey,
1088
+ progressWeight: task.progressWeight,
1089
+ recommendedConcurrency: task.recommendedConcurrency,
1090
+ }, index, state.session));
1091
+ emitAgentEvent(state.session, 'plan_set', 'tool', { steps });
1092
+ state.session._onStep?.(`Plan: ${steps.length} task(s) declared from ${server} fragment`);
1093
+ resultText = `Task-graph fragment integrated as the current plan (${steps.length} task(s), groups: ${[...new Set(payload.tasks.map((task) => task.groupId).filter(Boolean))].join(', ') || 'none'}). The orchestrator will dispatch these tasks — do NOT call production tools for them yourself. Reply with a short summary and wait.`;
1094
+ }
1095
+ if (/(^|__)agent_plan$/.test(tool) && Array.isArray(payload?.tasks) && payload.tasks.length > 0) {
1096
+ minimalPlanActive = false;
1097
+ }
1098
+ }
864
1099
  if (server === 'production') {
865
1100
  let payload = parseJsonText(resultText);
866
1101
  if (tool === 'production_start_job' && payload?.ok === false && payload?.error === 'workspace_busy') {
@@ -902,17 +1137,21 @@ export function createAgentGraph(options = {}) {
902
1137
  emitAgentEvent(state.session, 'plan_step_updated', 'tool', { step: 1, status: 'failed' });
903
1138
  }
904
1139
  }
1140
+ // Bound the result at its two exit points only (LLM context + display).
1141
+ // The full resultText above was already used for payload parsing and
1142
+ // _activity extraction, which must never see a truncated document.
1143
+ const boundedResult = truncateToolResult(resultText);
905
1144
  emitAgentEvent(state.session, 'tool_call_result', 'tool', {
906
1145
  callId: call.id,
907
1146
  name: toolName,
908
1147
  ok,
909
- result: resultText,
1148
+ result: boundedResult,
910
1149
  summary: ok ? 'done' : 'failed',
911
1150
  });
912
1151
  toolResultMessages.push({
913
1152
  role: 'tool',
914
1153
  tool_call_id: call.id,
915
- content: resultText,
1154
+ content: boundedResult,
916
1155
  });
917
1156
  }
918
1157
 
@@ -927,11 +1166,22 @@ export function createAgentGraph(options = {}) {
927
1166
  return END;
928
1167
  }
929
1168
 
930
- return new StateGraph(AgentState)
1169
+ const compiled = new StateGraph(AgentState)
931
1170
  .addNode('orchestrator', orchestratorNode)
932
1171
  .addNode('tool_executor', toolExecutorNode)
933
1172
  .addEdge(START, 'orchestrator')
934
1173
  .addConditionalEdges('orchestrator', routeOrchestrator)
935
1174
  .addEdge('tool_executor', 'orchestrator')
936
1175
  .compile();
1176
+
1177
+ // LangGraph's default recursionLimit is 25 super-steps. Each tool round
1178
+ // costs two of them (orchestrator + tool_executor), so runs died with
1179
+ // GRAPH_RECURSION_LIMIT around iteration 12 — far below the intended
1180
+ // MAX_TOOL_ITERATIONS budget — before finishing their work (observed as
1181
+ // `knowledge.update — error 0%` with no production job ever created).
1182
+ // Bake a limit matching the iteration budget into every invocation.
1183
+ const recursionLimit = MAX_TOOL_ITERATIONS * 2 + 10;
1184
+ return {
1185
+ invoke: (state, config = {}) => compiled.invoke(state, { recursionLimit, ...config }),
1186
+ };
937
1187
  }