@dotdrelle/wiki-manager 0.12.12 → 0.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/docker-compose.yml +1 -1
  2. package/mcp.endpoints.example.json +7 -0
  3. package/package.json +2 -2
  4. package/src/agent/graph.js +354 -143
  5. package/src/agent/graph.test.js +516 -54
  6. package/src/agent/llm.js +5 -5
  7. package/src/cli/wiki-manager.js +234 -6
  8. package/src/cli/wiki-manager.test.js +28 -0
  9. package/src/commands/slash.js +32 -11
  10. package/src/commands/slash.test.js +9 -1
  11. package/src/core/agentEvents.js +7 -1
  12. package/src/core/agentEvents.test.js +13 -1
  13. package/src/core/buildInfo.json +2 -2
  14. package/src/core/mcp.js +46 -4
  15. package/src/core/skills.js +0 -28
  16. package/src/core/toolLoop.js +56 -0
  17. package/src/core/toolLoop.test.js +88 -0
  18. package/src/orchestrator/capabilityRegistry.js +14 -0
  19. package/src/orchestrator/capabilityRegistry.test.js +12 -1
  20. package/src/orchestrator/dependencyResolver.js +10 -1
  21. package/src/orchestrator/dispatcher.js +34 -3
  22. package/src/orchestrator/dispatcher.test.js +34 -0
  23. package/src/orchestrator/objectiveResolver.js +79 -0
  24. package/src/orchestrator/objectiveResolver.test.js +50 -0
  25. package/src/orchestrator/scheduler.test.js +25 -0
  26. package/src/runtime/client.js +32 -1
  27. package/src/runtime/lifecycle.js +32 -2
  28. package/src/runtime/recoveryManager.js +14 -7
  29. package/src/runtime/runner.js +112 -13
  30. package/src/runtime/runner.test.js +64 -1
  31. package/src/runtime/server.js +47 -3
  32. package/src/runtime/supervisor.js +4 -1
  33. package/src/runtime/supervisor.test.js +49 -0
  34. package/src/shell/repl.js +134 -55
  35. package/src/shell/repl.test.js +151 -12
  36. package/src/shell/useSession.ts +15 -3
@@ -14,8 +14,8 @@ import { extractActivity, formatActivitySummary, parseJsonText, sessionActivitie
14
14
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
15
15
  import { enqueueProductionJob, ensureJobQueue, formatQueue, productionLockBusy } from '../core/jobQueue.js';
16
16
  import { updateWorkspaceProfilePreference } from '../core/profile.js';
17
- import { createCapabilityRegistry } from '../orchestrator/capabilityRegistry.js';
18
- import { fetchRuntimeState, postRuntimeCancel, postRuntimeControl, postRuntimeKill } from '../runtime/client.js';
17
+ import { capabilityRegistryForSession } from '../orchestrator/capabilityRegistry.js';
18
+ import { fetchRuntimeState, postRuntimeApprove, postRuntimeCancel, postRuntimeControl, postRuntimeDelegate, postRuntimeKill } from '../runtime/client.js';
19
19
 
20
20
  const MAX_TOOL_ITERATIONS = 80;
21
21
  const MAX_SPINNER_ARG_LENGTH = 96;
@@ -27,7 +27,7 @@ const MAX_PROFILE_CHARS = 4000;
27
27
  const INTERNAL_TOOL_SERVERS = {
28
28
  wiki: ['plan_set', 'plan_done'],
29
29
  shell: ['run_command', 'read_command', 'profile_update'],
30
- runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue', 'start_capability_run'],
30
+ runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue', 'delegate'],
31
31
  };
32
32
 
33
33
  const AGENT_SLASH_COMMANDS = new Set([
@@ -40,8 +40,8 @@ const AGENT_SLASH_COMMANDS = new Set([
40
40
  'services',
41
41
  'skills',
42
42
  'upload',
43
- 'uploads',
44
43
  'queue',
44
+ 'openui',
45
45
  ]);
46
46
 
47
47
  const SHELL_RUN_COMMAND_TOOL = {
@@ -50,7 +50,7 @@ const SHELL_RUN_COMMAND_TOOL = {
50
50
  name: 'shell__run_command',
51
51
  description: [
52
52
  'Run a deterministic wiki-manager slash command inside the current shell session.',
53
- 'Allowed commands: /workspace list, /workspace init <name> [path], /use <workspace>, /config, /status, /services, /skills, /skills show <name>, /skills run <name>, /upload <path>, /upload convert <id|pending>, /uploads.',
53
+ 'Allowed commands: /workspace list, /workspace init <name> [path], /use <workspace>, /config, /status, /services, /skills, /skills show <name>, /skills run <name>, /upload <path>, /upload convert <id|pending>.',
54
54
  'Do not use for arbitrary system shell commands, /workspace delete, /mcp call, /wiki run, /start, /stop, /logs, or /exit.',
55
55
  ].join(' '),
56
56
  parameters: {
@@ -73,7 +73,7 @@ const SHELL_READ_COMMAND_TOOL = {
73
73
  name: 'shell__read_command',
74
74
  description: [
75
75
  'Run a read-only deterministic wiki-manager slash command inside the current shell session.',
76
- 'Allowed commands: /help, /version, /config, /config list, /config status, /status, /services, /skills, /skills list, /skills show <name>, /uploads, /uploads list, /queue.',
76
+ 'Allowed commands: /help, /version, /config, /config list, /config status, /status, /services, /skills, /skills list, /skills show <name>, /queue.',
77
77
  'Do not use for workspace creation/deletion, uploads conversion, service start/stop, MCP calls, wiki runs, or any mutation.',
78
78
  ].join(' '),
79
79
  parameters: {
@@ -165,20 +165,18 @@ const RUNTIME_ENQUEUE_TOOL = {
165
165
  },
166
166
  };
167
167
 
168
- const RUNTIME_CAPABILITY_RUN_TOOL = {
168
+ const RUNTIME_DELEGATE_TOOL = {
169
169
  type: 'function',
170
170
  function: {
171
- name: 'runtime__start_capability_run',
172
- description: 'Start a DETERMINISTIC orchestrated run for a discovered capability: the runtime asks the capable agent for its task graph (agent_plan), validates and integrates it, creates the approval requests, and dispatches the tasks in parallel. Use this whenever the user asks for multi-document or multi-step work covered by a known capability (e.g. ingest everything pending → capability "knowledge.update", operation "ingest"). Do NOT call production tools directly for such requests.',
171
+ name: 'runtime__delegate',
172
+ description: 'Delegate the user objective to the runtime. Pass the objective in natural language without choosing a capability, operation, agent, plan, file list, or implementation. The runtime resolves the agent, obtains and validates the real plan before accepting the run.',
173
173
  parameters: {
174
174
  type: 'object',
175
175
  additionalProperties: false,
176
176
  properties: {
177
- capability: { type: 'string', description: 'Capability id from the known list (e.g. knowledge.update).' },
178
- operation: { type: 'string', description: 'Operation supported by the capability (e.g. ingest, build).' },
179
- inputs: { type: 'array', items: { type: 'string' }, description: 'Optional file subset; omit to cover everything pending.' },
177
+ objective: { type: 'string', description: 'The complete user objective, preserving scope and constraints but containing no invented technical identifiers.' },
180
178
  },
181
- required: ['capability'],
179
+ required: ['objective'],
182
180
  },
183
181
  },
184
182
  };
@@ -264,15 +262,124 @@ const AgentState = Annotation.Root({
264
262
  }),
265
263
  toolIterations: Annotation({ default: () => 0 }),
266
264
  pendingToolCalls: Annotation(),
265
+ allowedToolNames: Annotation(),
267
266
  inputClassification: Annotation(),
268
267
  readyToStream: Annotation(),
269
268
  streamContext: Annotation(),
270
269
  streamedInline: Annotation(),
271
270
  retryWithoutTool: Annotation({ default: () => false }),
271
+ invalidResponseRetries: Annotation({ default: () => 0 }),
272
+ invalidToolCallRetries: Annotation({ default: () => 0 }),
273
+ forceDelegation: Annotation({ default: () => false }),
272
274
  });
273
275
 
276
+ function invalidToolCalls(toolCalls) {
277
+ if (!Array.isArray(toolCalls)) return [];
278
+ return toolCalls.filter((call) => {
279
+ if (!call?.id || !call?.function?.name) return true;
280
+ try {
281
+ const args = JSON.parse(call.function.arguments || '{}');
282
+ return !args || typeof args !== 'object' || Array.isArray(args);
283
+ } catch {
284
+ return true;
285
+ }
286
+ });
287
+ }
288
+
289
+ export function normalizeToolArgumentsFromSchema(args, parameters) {
290
+ if (!args || typeof args !== 'object' || Array.isArray(args)) return args;
291
+ const schema = parameters && typeof parameters === 'object' ? parameters : {};
292
+ const properties = schema.properties && typeof schema.properties === 'object' ? schema.properties : {};
293
+ const required = Array.isArray(schema.required) ? schema.required.filter((key) => typeof key === 'string') : [];
294
+ const missing = required.filter((key) => args[key] === undefined);
295
+ const unknown = Object.keys(args).filter((key) => properties[key] === undefined);
296
+ if (missing.length !== 1 || unknown.length !== 1) return args;
297
+ const target = missing[0];
298
+ const source = unknown[0];
299
+ if (!schemaValueMatches(args[source], properties[target])) return args;
300
+ const normalized = { ...args, [target]: args[source] };
301
+ delete normalized[source];
302
+ return normalized;
303
+ }
304
+
305
+ function schemaValueMatches(value, propertySchema) {
306
+ const types = Array.isArray(propertySchema?.type) ? propertySchema.type : [propertySchema?.type];
307
+ if (types.includes(undefined) || types.includes(null)) return true;
308
+ return types.some((type) => {
309
+ if (type === 'array') return Array.isArray(value);
310
+ if (type === 'object') return value !== null && typeof value === 'object' && !Array.isArray(value);
311
+ if (type === 'integer') return Number.isInteger(value);
312
+ if (type === 'number') return typeof value === 'number' && Number.isFinite(value);
313
+ if (type === 'null') return value === null;
314
+ return typeof value === type;
315
+ });
316
+ }
317
+
318
+ function toolDefinitionForCall(session, callName) {
319
+ const internal = [
320
+ SHELL_RUN_COMMAND_TOOL,
321
+ SHELL_READ_COMMAND_TOOL,
322
+ SHELL_PROFILE_UPDATE_TOOL,
323
+ RUNTIME_STATUS_TOOL,
324
+ RUNTIME_CANCEL_TOOL,
325
+ RUNTIME_KILL_TOOL,
326
+ RUNTIME_APPROVE_TOOL,
327
+ RUNTIME_ENQUEUE_TOOL,
328
+ RUNTIME_DELEGATE_TOOL,
329
+ WIKI_PLAN_SET_TOOL,
330
+ WIKI_PLAN_DONE_TOOL,
331
+ ];
332
+ return [...internal, ...buildLlmTools(session?.mcp)]
333
+ .find((item) => item?.function?.name === callName) ?? null;
334
+ }
335
+
274
336
  function commandList(session) {
275
- return session.commands.map((command) => `/${command}`).join(', ');
337
+ return session.commands
338
+ .filter((command) => AGENT_SLASH_COMMANDS.has(command))
339
+ .map((command) => `/${command}`)
340
+ .join(', ');
341
+ }
342
+
343
+ export function invalidSuggestedSlashCommands(content, session) {
344
+ const allowed = new Set((session?.commands ?? []).filter((command) => AGENT_SLASH_COMMANDS.has(command)));
345
+ const candidates = new Set();
346
+ for (const line of String(content ?? '').split(/\r?\n/)) {
347
+ const trimmed = line.trim();
348
+ const standalone = trimmed.match(/^\/([a-z][\w-]*)\b/i);
349
+ if (standalone) candidates.add(standalone[1].toLowerCase());
350
+ for (const match of line.matchAll(/`\/([a-z][\w-]*)\b/gi)) candidates.add(match[1].toLowerCase());
351
+ }
352
+ return [...candidates].filter((command) => !allowed.has(command)).sort();
353
+ }
354
+
355
+ export function invalidUserFacingToolNames(content, session) {
356
+ const text = String(content ?? '');
357
+ const connected = buildLlmTools(session?.mcp)
358
+ .map((item) => item?.function?.name)
359
+ .filter(Boolean)
360
+ .filter((name) => text.includes(name));
361
+ const syntactic = [...text.matchAll(/\b[a-z][a-z0-9_-]*__[a-z][a-z0-9_-]*\b/gi)].map((match) => match[0]);
362
+ return [...new Set([...connected, ...syntactic])].sort();
363
+ }
364
+
365
+ async function classifyRequestedAction(llm, input, signal) {
366
+ try {
367
+ const result = await llm.completeWithTools({
368
+ system: [
369
+ 'Classify whether the user explicitly requests a real state-changing action now.',
370
+ 'Actions include starting, stopping, importing, ingesting, building, exporting, configuring, writing, deleting, or sending.',
371
+ 'Questions, explanations, status questions, greetings, and hypothetical discussions are not actions.',
372
+ 'Return JSON only: {"action":true} or {"action":false}.',
373
+ ].join('\n'),
374
+ tools: [],
375
+ messages: [{ role: 'user', content: String(input ?? '') }],
376
+ signal,
377
+ });
378
+ const text = String(result?.content ?? '').trim().replace(/^```(?:json)?\s*/i, '').replace(/\s*```$/, '');
379
+ return JSON.parse(text)?.action === true;
380
+ } catch {
381
+ return false;
382
+ }
276
383
  }
277
384
 
278
385
  function summarizeToolArguments(rawArguments) {
@@ -384,8 +491,7 @@ function assertAgentReadSlashCommandAllowed(commandLine) {
384
491
  command === 'services' ||
385
492
  command === 'queue' ||
386
493
  (command === 'config' && ['', 'list', 'status'].includes(subcommand)) ||
387
- (command === 'skills' && ['', 'list', 'show'].includes(subcommand)) ||
388
- (command === 'uploads' && ['', 'list'].includes(subcommand));
494
+ (command === 'skills' && ['', 'list', 'show'].includes(subcommand));
389
495
  if (!allowed) {
390
496
  throw new Error(`Read-only command is not available to the agent: /${parts.join(' ')}`);
391
497
  }
@@ -530,9 +636,7 @@ function emitAgentEvent(session, type, origin, payload = {}) {
530
636
  // agents. This is the live registry the dispatcher will resolve against —
531
637
  // a plan declaring anything outside this set can only stall forever.
532
638
  export function knownCapabilityIds(session) {
533
- const registry = session?.capabilityRegistry ?? createCapabilityRegistry({
534
- agents: session?.agentRegistrySnapshot ?? session?.agents ?? [],
535
- });
639
+ const registry = capabilityRegistryForSession(session);
536
640
  const snapshot = typeof registry.snapshot === 'function' ? registry.snapshot() : registry;
537
641
  return [...new Set(Object.keys(snapshot ?? {}).map((key) => {
538
642
  const index = key.lastIndexOf('@');
@@ -581,25 +685,34 @@ async function handleRuntimeControlTool(session, tool, args = {}) {
581
685
  return result.cancelled ? 'Runtime run cancellation requested.' : `No active run to cancel${result.reason ? ` (${result.reason})` : ''}.`;
582
686
  }
583
687
  if (tool === 'approve') {
584
- const result = await postRuntimeControl('message', { url, workspace, input: 'approve', intent: 'approve' });
585
- return String(result?.explanation ?? (result?.accepted ? 'Approval granted.' : 'No pending approval found.'));
586
- }
587
- if (tool === 'start_capability_run') {
588
- const { postRuntimeRun } = await import('../runtime/client.js');
589
- const result = await postRuntimeRun(`Run de capability ${args.capability}${args.operation ? ` (${args.operation})` : ''} demandé par Donna.`, {
688
+ const state = await fetchRuntimeState({ url, workspace });
689
+ const pending = (Array.isArray(state?.approvals) ? state.approvals : [])
690
+ .filter((approval) => approval.status === 'pending_approval');
691
+ const runId = state?.runId
692
+ ?? state?.runs?.find((run) => ['running', 'pending_approval'].includes(run.status))?.id
693
+ ?? null;
694
+ if (!runId || pending.length === 0) return 'No pending approval found.';
695
+ const approvalClasses = [...new Set(pending.flatMap((approval) => {
696
+ const value = approval.approvalClasses ?? approval.approvalClass ?? [];
697
+ return Array.isArray(value) ? value : [value];
698
+ }).map(String).filter(Boolean))];
699
+ const result = await postRuntimeApprove({
590
700
  url,
591
701
  workspace,
592
- capabilityPlan: {
593
- capability: String(args.capability ?? ''),
594
- ...(args.operation ? { operation: String(args.operation) } : {}),
595
- ...(Array.isArray(args.inputs) && args.inputs.length > 0 ? { inputs: args.inputs.map(String) } : {}),
596
- // No concurrency dictated here: the runtime reads the provider's
597
- // own declared capacity (agent_describe limits).
598
- },
702
+ runId,
703
+ scope: 'run',
704
+ planRevision: state?.planRevision ?? null,
705
+ approvalClasses: approvalClasses.length > 0 ? approvalClasses : ['default'],
599
706
  });
707
+ return result?.approved ? 'Current validated plan approved.' : 'No pending approval found.';
708
+ }
709
+ if (tool === 'delegate') {
710
+ const objective = String(args.objective ?? '').trim();
711
+ if (!objective) return 'Delegation rejected: missing objective.';
712
+ const result = await postRuntimeDelegate(objective, { url, workspace });
600
713
  return result?.runId
601
- ? `Run de capability accepté (${String(result.runId).slice(0, 8)}) : le plan sera intégré et dispatché en parallèle ; une approbation sera demandée avant les mutations (l'utilisateur peut dire « valide tout » ou taper /approve). Suis la progression dans Activity.`
602
- : `Run non démarré : ${result?.explanation ?? result?.error ?? JSON.stringify(result)}`;
714
+ ? `Action lancée (${String(result.runId).slice(0, 8)}) après validation du plan réel : ${result.delegation?.tasks ?? 0} tâche(s), ${result.delegation?.agent ?? 'agent résolu'}. Exécution en cours.`
715
+ : `Délégation refusée : ${result?.error ?? JSON.stringify(result)}`;
603
716
  }
604
717
  if (tool === 'enqueue') {
605
718
  const result = await postRuntimeControl('message', { url, workspace, input: String(args.input ?? ''), intent: 'enqueue' });
@@ -728,7 +841,12 @@ export function buildAgentSystemPrompt(state) {
728
841
  const workspace = state.session.workspace ?? 'no workspace selected';
729
842
  const wikirc = state.session.wikirc?.profile ?? 'no profile loaded';
730
843
  const language = state.session.language ?? 'en-US';
731
- const mcpTools = formatMcpToolsForAgent(state.session.mcp);
844
+ // Advertise only the read-only tools Donna may call directly. Listing
845
+ // mutating provider tools (e.g. production__production_start_job) here teaches
846
+ // a capable model to invoke them directly and bypass runtime__delegate.
847
+ const mcpTools = formatMcpToolsForAgent(state.session.mcp, {
848
+ include: (qualifiedName) => !isOrchestrationBypassTool(qualifiedName),
849
+ });
732
850
  const skills = formatSkillsForAgent(state.session);
733
851
  const customPrompt = state.session.systemPrompt ?? null;
734
852
  const workspaceProfile = loadWorkspaceProfile(state.session.workspacePath);
@@ -743,74 +861,40 @@ export function buildAgentSystemPrompt(state) {
743
861
  `Current wikirc profile: ${wikirc}.`,
744
862
  `Available primitives: ${commandList(state.session)}.`,
745
863
  'Only announce or call slash commands that appear exactly in Available primitives. Do not invent command names, subcommands, or arguments.',
746
- 'Connected MCP tools (use the server__tool naming convention for tool calls):',
864
+ 'Connected MCP tools you may call directly (server__tool naming convention) — reads AND single-step actions like configuring or adding a connector source, converting a document, sending, or searching. Only the heavy multi-step operations (ingest, build, export, polish, pipeline) go through runtime__delegate to get their parallel plan. Everything listed below is directly callable:',
747
865
  mcpTools,
748
866
  'Current local MCP job queue:',
749
867
  formatQueue(state.session),
750
868
  'Available skills:',
751
869
  skills,
752
- 'You can call MCP tools directly using the provided tool functions.',
870
+ 'In interactive agent mode you may call only the read-only tools and runtime control/delegation tools actually provided to you.',
753
871
  'When the user asks for an action that can be performed with connected MCP tools or safe primitives, do not answer with future intent such as "I will call...", "I am going to run...", or "launching..." unless you also call the tool in the same turn. Either call the tool now, ask for the exact missing required arguments, or explain the concrete blocker.',
754
872
  'Execution truthfulness: never invent a job id, status, percentage, duration, generated file, file content, URL, command, or tool result. An action is executed only when you call an available tool and receive its result. Examples and placeholders are forbidden in execution reports.',
755
873
  'After any completed action, give a short factual summary based only on the tool result: outcome and concrete outputs or references actually returned. Mention a viewing primitive only when it exists in Available primitives and is relevant. Do not interpret generated content, propose verification checklists, invent next steps, or suggest commands unless the user explicitly asks.',
874
+ 'Never add a "Next steps", "Prochaines étapes", "À suivre", options, or suggestions section unless the user explicitly asks what to do next. End after the requested result or the concrete error.',
756
875
  'When calling a tool, emit no preliminary narration. Call it directly; the PLAN and Activity panels show progress. After completion, keep the final response concise and proportional to the result.',
757
- 'For connector configuration/setup/update requests, if a matching setup/configuration tool is connected and the required arguments are known, call it immediately. If the connector or tool is not connected, say which concrete capability is missing and recommend the exact service/status primitive to inspect it. Do not invent a pending connector action in plain text.',
758
- 'For workspace-scoped external MCP tools, the orchestrator enforces workspace injection. Use the active workspace for configuration, source, import, export, conversion, and generation tools unless a tool is explicitly job-scoped and only requires a job id.',
759
- 'Dynamic capability inputs are owned by the agent that provides the capability. To answer which inputs are pending or available, call the qualified `<provider>__agent_status` tool with {capability, operation} and report only its pendingInputs. Never infer this state from /uploads or a hardcoded workspace directory. If no capable agent/status tool is connected, say that the pending inputs cannot be determined.',
760
- 'You can call shell__run_command for safe manager slash commands such as /workspace list, /workspace init <name> [path], /use <workspace>, /config, /status, /services, /skills, /skills show <name>, and /skills run <name>.',
761
- 'Skills are workflow instructions, not executable code. When a user asks to run a skill, inspect it, propose the concrete primitive/tool plan, and ask for confirmation before costly or mutating actions.',
762
- [
763
- state.session.headless ? 'HEADLESS MODE ACTIVE. Execute the requested skill or task autonomously using available safe primitives and MCP tools. Do not ask for interactive confirmation unless the request is genuinely ambiguous or outside the loaded workspace.' : null,
764
- '',
765
- 'You have two internal planning tools: wiki__plan_set and wiki__plan_done.',
766
- 'Prefer MCP tools that declare their own plan via _activity.plan.steps — when such a tool returns _activity, the shell creates and tracks the plan automatically without requiring wiki__plan_set.',
767
- 'Use wiki__plan_set when the MCP tool cannot declare its own plan or when the task spans multiple independent tools (e.g. CME export then email report). For a single self-describing async job, wiki__plan_set is optional.',
768
- '',
769
- (() => {
770
- const capabilityIds = knownCapabilityIds(state.session);
771
- return capabilityIds.length > 0
772
- ? `Known orchestration capabilities — the ONLY values allowed in requiredCapability: ${capabilityIds.join(', ')}. Never invent capability names; a plan declaring an unknown capability will be rejected. A step you execute yourself directly takes requiredCapability: null.`
773
- : 'No orchestration capabilities discovered yet: declare plan steps with requiredCapability: null and execute them yourself with the connected MCP tools.';
774
- })(),
775
- '',
776
- 'Task startup:',
777
- ' 1. If the next MCP tool returns _activity.plan.steps, call that tool directly; the shell will create the visible plan from the returned activity.',
778
- ' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, operation, arguments, dependsOn, outputRefs}; capability steps need operation+arguments for the dispatcher to execute them; a legacy list of strings is still accepted.',
779
- ' Multi-tool example: wiki__plan_set(steps=[{id:"cme-export",description:"CME export",requiredCapability:"external-source.export",dependsOn:[],outputRefs:["raw/untracked"]},{id:"production",description:"Production pipeline",requiredCapability:"knowledge.pipeline",dependsOn:["cme-export"],outputRefs:["deliverables"]}])',
780
- ' 3. Immediately execute the first step using the appropriate MCP tool. Do not start step 2 in the same turn unless one async pipeline tool owns and declares the whole sequence.',
781
- ' For synchronous steps (result is immediate, no _activity polling), call wiki__plan_done(step=1) after confirming success.',
782
- ' For async MCP jobs (returns _activity with poll), the orchestrator tracks completion automatically.',
783
- '',
784
- state.session.headless ? [
785
- 'Headless follow-up turns — the orchestrator re-invokes you with:',
786
- ' (a) the original task,',
787
- ' (b) the current plan status — [✓] done / [✗] failed / [ ] pending,',
788
- ' (c) the just-completed activities.',
789
- ' Read the plan status. Find the first [ ] pending step. Execute it only.',
790
- ' Never re-execute a [✓] or [✗] step. Never skip a [ ] step.',
791
- '',
792
- 'Final turn — when all steps are [✓] or [✗]: respond with a concise summary. Do not start new actions.',
793
- ].join('\n') : null,
794
- '',
795
- 'On failure: if a completed activity is failed/error/cancelled, call wiki__plan_done(step=N, status="failed") then stop with a clear error report.',
796
- ].filter(Boolean).join('\n'),
876
+ 'Keep every response synthetic and information-dense. Use only the lines needed, and never exceed roughly 15 to 20 short lines even for a detailed answer. Never expose internal reasoning, repeated checks, tool-selection commentary, or a chronological diary. Prioritize the result, essential facts, concrete errors, and actual outputs.',
877
+ 'Only the heavy multi-step operations — ingest, build, export, polish, pipeline — are delegated via runtime__delegate (for their DAG and parallelism). Single-step actions — configuring or adding a connector source, converting a document, sending, searching — are called directly on the connected tool. Never call an agent orchestration-contract or plan tool directly.',
878
+ 'For any question about the current workspace inventory or what is waiting there, call wiki__wiki_workspace_status first and answer only from its result. This is the canonical read-only workspace state; do not reconstruct it from upload, connector, or production tools.',
879
+ 'Tool identifiers are private implementation details. Never print MCP tool names such as server__tool in a user-facing answer. Describe the human result instead.',
880
+ 'Never suggest a manual filesystem command or implementation workaround unless the user explicitly asks for manual instructions. For an action request, delegate the objective and let the specialized agent determine paths and operations from its live contract.',
881
+ 'Skills are documentation only in this stabilized version. Never execute a skill from conversation; delegate the user objective.',
797
882
  'For service actions, recommend only available service primitives from Available primitives, with the exact service name when the primitive supports one.',
798
- 'Scope discipline: execute ONLY the action(s) the user explicitly requested. Never chain additional mutating operations (ingest, build, export, polish, delete, send…) that the user did not ask for — even when diagnostics or recommendations suggest them. Finish the requested work, then list the suggested follow-ups in your final answer and stop. Example: "applique les recommandations de config" means apply the config; it does NOT authorize launching the ingest those recommendations mention.',
799
- 'Disambiguate export requests carefully.',
800
- 'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`cme__cme_export_run`, then `cme__cme_export_status`). Never use production `type=export` for Confluence source export.',
801
- 'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`production__production_start_job` with `type:"export"` or pipeline steps). Require the deliverable path when exporting deliverables.',
802
- 'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts.',
803
- 'MULTI-DOCUMENT ingest (more than 2 files, or "ingest everything pending") and any multi-step capability work: when runtime__start_capability_run is available, call it (e.g. {capability:"knowledge.update", operation:"ingest"}) — the runtime integrates the agent task graph deterministically and dispatches IN PARALLEL with an approval gate. Inside a runtime run (no runtime tools), call production__agent_plan instead; the shell integrates the fragment automatically. Never call production__production_start_job for multi-document ingest — that creates one monolithic sequential job.',
804
- 'Single-document ingest or one-off jobs (doctor, one build, one export): production__production_start_job is fine. To chain sequential steps (e.g. build then polish) use ONE call with type="pipeline" and steps=["build","polish"] — never separate jobs (the first is asynchronous). For existing deliverables where content stability matters, pass stabilize:true. Do not ask the user to confirm between steps.',
805
- 'Long-running MCP jobs: do not call the same status tool more than once consecutively. When chaining jobs sequentially: (1) start the job, report job/activity id and status; (2) check status once — if done, proceed to the next step immediately; (3) if still running, report status, list the remaining steps, and return control; (4) when re-invoked, check status first, then proceed. Do not spin-poll (status → status → status with no new action between). The shell activity panel monitors non-terminal jobs automatically.',
806
- 'If production__production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
807
- 'For diagnostics (doctor), use production__production_start_job with type="doctor" like any other production job; /wiki run doctor is only the fallback when the production MCP is not connected. Use /workspace init <name> [path] for low-level non-interactive workspace creation. In the interactive TUI, /new <name> opens the setup wizard. Use /wiki for index, or /wiki run index through the explicit backup hatch. Use /wiki run init only for explicit current-workspace llm-wiki init.',
883
+ 'Scope discipline: execute ONLY the action(s) the user explicitly requested. Never chain additional mutating operations (ingest, build, export, polish, delete, send…) that the user did not ask for — even when diagnostics or recommendations suggest them. Finish with the requested result and stop. Example: "applique les recommandations de config" means apply the config; it does NOT authorize launching the ingest those recommendations mention.',
884
+ state.session.runtime?.url
885
+ ? 'The runtime is connected and runtime__delegate is bound and available to you right now — it is a tool you call directly, not a slash command or a missing primitive. It is the ONLY way to execute an action (ingest, build, export, configure, send…). Never tell the user that delegation or the runtime is unavailable while it is connected; call runtime__delegate instead.'
886
+ : 'No runtime is connected, so you cannot execute actions. State that plainly and name the runtime connection as the missing capability — do not invent a workaround.',
887
+ 'If the connector or service needed for a requested read or action is absent from the Connected MCP tools above (its service is not running — e.g. CME, documents, or production), say plainly that this service is not connected and name it as the missing capability. Never redirect a simple read (e.g. "give me the CME config") to an "agent action", never invent its result, and never propose a workaround. Only requests you can actually serve with a listed tool are answered with data.',
888
+ 'For any requested action, call runtime__delegate with the user objective only. Never choose a capability, operation, agent, plan, or implementation yourself. The runtime resolves the registry and validates the provider plan before accepting. Never call <provider>__agent_plan, <provider>__agent_execute, legacy production__production_start_job, wiki__plan_set, or wiki__plan_done from interactive chat.',
889
+ 'Do not ask the user which sources, files, connectors, or templates to use for an ingest, build, or export: the specialized agent discovers them from the workspace. When the objective is clear (e.g. "lance une ingestion"), delegate it as stated, without a clarifying question.',
890
+ 'If runtime__delegate returns a blocker or no specialized provider is available, report only that concrete blocker concisely. Never replace the missing execution path with a suggested slash command, skill, MCP tool name, manual file move, administrator escalation, or alternative workflow unless the user explicitly asks for alternatives.',
891
+ 'For workspace inventory and page listings, use the connected wiki MCP read tools. Never invent or call a /wiki shell command through shell__run_command. Use /workspace init <name> [path] for low-level non-interactive workspace creation; in the interactive TUI, /new <name> opens the setup wizard.',
808
892
  'If an action requires tools or skills not available yet, explain the limitation and name the expected primitive.',
809
893
  workspaceProfile
810
894
  ? `Workspace profile (.wiki/profile.md) — durable user preferences, apply these to every reply (tone, tutoiement/vouvoiement, formatting, etc.):\n${workspaceProfile}`
811
895
  : null,
812
896
  'Runtime control: you have runtime__status, runtime__cancel, runtime__kill, runtime__approve and runtime__enqueue. When the user asks to stop, remove, clean or kill the current run, its jobs or the queue ("supprime le job et la queue", "arr\u00eate tout"), call runtime__kill (or runtime__cancel for a soft stop of just the run) and confirm what was stopped. For questions about what is running or queued, call runtime__status and answer from its data. When the user consents to a pending approval in any phrasing ("vas-y", "ok pour l\'export"), call runtime__approve. When the user asks for a NEW action while a run is active, do not execute it: propose runtime__enqueue (run it after) or, if they insist it replaces the current work, runtime__kill then the new action.',
813
- 'When the user explicitly asks you to remember, persist, or update durable preference/profile information, call wiki__profile_update when it is available; otherwise call shell__profile_update. Do not just acknowledge in text without calling a profile update tool.',
897
+ 'Durable profile updates are actions in this stabilized version: delegate them instead of writing directly.',
814
898
  ].filter(Boolean).join('\n');
815
899
 
816
900
  return customPrompt ? `${customPrompt}\n\n${agentContext}` : agentContext;
@@ -840,66 +924,75 @@ export function formatLlmUnavailableMessage(reason) {
840
924
  return `⚠ LLM injoignable : ${clean || 'raison inconnue'}`;
841
925
  }
842
926
 
843
- // Verbs that clearly request work (a runtime run), in French and English.
844
- // "configure/configurer" is an action; the nouns "config/configuration" are
845
- // NOT matched here — asking for a config is an observe request.
846
- const ACTION_REQUEST_PATTERN = /\b(lance|relance|d[eé]marre|start|ex[eé]cute|execute|g[eé]n[eè]re|generate|build|construis|exporte?|ingest\w*|ing[eè]re|importe?|convert(?:is|it|s)?|cr[eé]e|create|polish|publie|publish|d[eé]ploie|deploy|envoie|send|configure[rsz]?|setup|installe|update|mets? [aà] jour|supprime|delete|efface|nettoie|clean|r[eé]pare|fix|corrige)\b/i;
847
-
848
- // Explicit explanation/question markers dominate action verbs: "explique le
849
- // build" is a question about the build, not a request to build.
850
- const EXPLANATION_REQUEST_PATTERN = /\b(explique|explain|pourquoi|why|comment|how|c'est quoi|qu'est[- ]ce)\b/i;
851
-
852
- export function classifyAgentInput(input, session) {
853
- const lower = String(input ?? '').toLowerCase();
854
- const hasActiveRun = session?.agentProjection?.status === 'running'
855
- || sessionActivities(session).some((activity) => !activity.terminal);
856
- if (/\b(valide tout|approve all|approve|approuve|valid[eé]|ok pour tout|go pour tout)\b/i.test(lower)) {
857
- return { kind: 'approve', confidence: 0.86, reason: 'approval_request', activeRun: hasActiveRun };
858
- }
859
- if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort|supprime|kill|tue|purge|vide la (file|queue)|nettoie la (file|queue))\b/i.test(lower)) {
860
- return { kind: 'cancel', confidence: 0.86, reason: 'cancel_request', activeRun: hasActiveRun };
861
- }
862
- if (/\b(plus tard|later|ensuite|apr[eè]s ce run|enqueue|mets en file|met en file|futur|next run|future run)\b/i.test(lower)) {
863
- return { kind: 'enqueue_run', confidence: 0.8, reason: 'future_run_request', activeRun: hasActiveRun };
864
- }
865
- if (EXPLANATION_REQUEST_PATTERN.test(lower)) {
866
- return { kind: 'observe', confidence: 0.86, reason: 'explanation_request', activeRun: hasActiveRun };
867
- }
868
- // Observe markers only win when no action verb is present: "où en est le
869
- // run" is observe, "lance le run" is an action request.
870
- if (/\b(o[uù] en es[t-]|status|statut|progress|progression|run|job|queue|logs?|inspect|show|montre|affiche|donne|liste|list|quel(?:le)?s?|combien|config(?:uration)?|quoi de neuf)\b/i.test(lower)
871
- && !ACTION_REQUEST_PATTERN.test(lower)) {
872
- return { kind: 'observe', confidence: 0.86, reason: 'status_or_explanation_request', activeRun: hasActiveRun };
873
- }
874
- if (hasActiveRun && /\b(ajoute|add|change|modifie|modify|remplace|replace|retire|remove|skip|ignore|plan|step|t[aâ]che)\b/i.test(lower)) {
875
- return { kind: 'modify_run', confidence: 0.78, reason: 'active_run_change_request', activeRun: hasActiveRun };
876
- }
877
- if (hasActiveRun && /\b(lance|run|g[eé]n[eè]re|build|export|cr[eé]e|create|send|envoie|ingest|convert|importe|import)\b/i.test(lower)) {
878
- return { kind: 'ambiguous', confidence: 0.45, reason: 'active_run_action_is_ambiguous', activeRun: hasActiveRun };
879
- }
880
- if (ACTION_REQUEST_PATTERN.test(lower)) {
881
- return { kind: 'start_run', confidence: 0.8, reason: 'action_request', activeRun: hasActiveRun };
882
- }
883
- return { kind: 'converse', confidence: 0.62, reason: 'plain_conversation', activeRun: hasActiveRun };
884
- }
885
-
886
927
  function toolsForClassification(classification, writeTools, session = null) {
887
928
  const controlTools = session?.runtime?.url
888
929
  ? [RUNTIME_STATUS_TOOL, RUNTIME_CANCEL_TOOL, RUNTIME_KILL_TOOL, RUNTIME_APPROVE_TOOL, RUNTIME_ENQUEUE_TOOL]
889
930
  : [];
931
+ // Provider discovery and validation belong to the runtime. Hiding
932
+ // delegation while the shell snapshot is temporarily empty forced Donna
933
+ // to invent commands instead of submitting the objective.
890
934
  const capabilityRunTools = session?.runtime?.url && !classification.activeRun
891
- ? [RUNTIME_CAPABILITY_RUN_TOOL]
935
+ ? [RUNTIME_DELEGATE_TOOL]
892
936
  : [];
893
- if (classification.activeRun && ['converse', 'observe', 'ambiguous', 'approve', 'cancel', 'enqueue_run'].includes(classification.kind)) {
937
+ if (classification.activeRun) {
894
938
  // During an active run Donna gets read + profile + the runtime control
895
939
  // suite: she can answer, approve, enqueue for later, soft-cancel or
896
940
  // kill — but she must not fire new MCP jobs alongside the run (that is
897
941
  // what runtime__enqueue is for). No canned regex answers anywhere.
898
- return [SHELL_READ_COMMAND_TOOL, SHELL_PROFILE_UPDATE_TOOL, ...controlTools];
942
+ return [SHELL_READ_COMMAND_TOOL, ...controlTools];
943
+ }
944
+ if (session?.runtime?.url) {
945
+ // Offer every connected tool directly EXCEPT orchestration-bypass tools
946
+ // and raw shell write/profile mutation. Reads, configuration, connector
947
+ // setup — and any newly added MCP's tools — stay directly callable.
948
+ const directTools = writeTools.filter((item) => {
949
+ const name = item?.function?.name;
950
+ if (!name || name === 'shell__run_command' || name === 'shell__profile_update') return false;
951
+ return !isOrchestrationBypassTool(name);
952
+ });
953
+ return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...directTools];
899
954
  }
900
955
  return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...writeTools];
901
956
  }
902
957
 
958
+ export function isDonnaReadTool(item) {
959
+ const name = String(item?.function?.name ?? '');
960
+ if (!name || name.startsWith('shell__') || name === 'wiki__plan_set' || name === 'wiki__plan_done') return false;
961
+ if (item?.readOnly === true) return true;
962
+ const tool = name.includes('__') ? name.slice(name.indexOf('__') + 2) : name;
963
+ return tool === 'wiki_workspace_status'
964
+ || tool === 'agent_describe'
965
+ || tool === 'agent_status'
966
+ || /(?:^|_)(?:status|list|search|read|get)$/.test(tool);
967
+ }
968
+
969
+ // Two-tier tool policy. Donna may call any connected MCP tool directly
970
+ // (reads AND plain writes: cme_setup, connector setup, document conversion,
971
+ // send, search, and anything a newly added MCP exposes) EXCEPT the small set
972
+ // that must go through the runtime's orchestration: the universal five-tool
973
+ // contract executors (agent_plan/agent_execute), the legacy job starter, and
974
+ // direct plan mutation. Heavy multi-step work (ingest/build/export via the
975
+ // production agent) is delegated for its DAG/parallelism; plain single-step
976
+ // tools are called directly. This is a blocklist, not a whitelist, so adding a
977
+ // new MCP never silently disables its tools.
978
+ function isOrchestrationBypassTool(name) {
979
+ const full = String(name ?? '');
980
+ if (!full) return true;
981
+ if (full === 'wiki__plan_set' || full === 'wiki__plan_done') return true;
982
+ const sep = full.indexOf('__');
983
+ const tool = sep === -1 ? full : full.slice(sep + 2);
984
+ return tool === 'agent_plan' || tool === 'agent_execute' || tool === 'production_start_job';
985
+ }
986
+
987
+ function isReadOnlyMcpCall(session, server, tool) {
988
+ const descriptor = (session?.mcp?.[server]?.tools ?? [])
989
+ .find((item) => String(item?.name ?? '') === tool || String(item?.name ?? '').endsWith(`__${tool}`));
990
+ return isDonnaReadTool({
991
+ function: { name: `${server}__${tool}` },
992
+ readOnly: descriptor?.readOnly === true,
993
+ });
994
+ }
995
+
903
996
  export function createAgentGraph(options = {}) {
904
997
  async function orchestratorNode(state) {
905
998
  const llm = state.session.llm ?? options.llm ?? null;
@@ -933,7 +1026,13 @@ export function createAgentGraph(options = {}) {
933
1026
  const classification = iterations === 0
934
1027
  ? (runtimeExecution
935
1028
  ? { kind: 'execute_run', confidence: 1, reason: 'runtime_run_execution', activeRun: true }
936
- : classifyAgentInput(state.input, state.session))
1029
+ : {
1030
+ kind: 'agent_turn',
1031
+ confidence: 1,
1032
+ reason: 'agent_mode_llm_decision',
1033
+ activeRun: state.session?.agentProjection?.status === 'running'
1034
+ || sessionActivities(state.session).some((activity) => !activity.terminal),
1035
+ })
937
1036
  : (state.inputClassification ?? { kind: 'modify_run', confidence: 1, reason: 'tool_iteration' });
938
1037
  if (iterations === 0) {
939
1038
  state.session._onStep?.(`Agent: classified input as ${classification.kind}`);
@@ -962,28 +1061,54 @@ export function createAgentGraph(options = {}) {
962
1061
 
963
1062
  try {
964
1063
  const useStreamWithTools = typeof llm.streamWithTools === 'function';
965
- const suppressExecutionNarration = runtimeExecution && (iterations === 0 || state.retryWithoutTool);
1064
+ const toolChoice = state.forceDelegation
1065
+ ? { type: 'function', function: { name: 'runtime__delegate' } }
1066
+ : 'auto';
966
1067
  const result = useStreamWithTools
967
1068
  ? await llm.streamWithTools({
968
1069
  system,
969
1070
  tools,
970
1071
  messages: conversationMessages,
971
- onTextDelta: (delta) => {
972
- if (suppressExecutionNarration) return;
973
- emitAgentEvent(state.session, 'assistant_delta', 'llm', { delta });
974
- state.session._onStream?.(delta);
975
- },
1072
+ toolChoice,
1073
+ // Buffer until validation. Invalid commands and malformed tool
1074
+ // calls must never flash hundreds of lines before disappearing.
1075
+ onTextDelta: () => {},
976
1076
  signal: state.session._abortSignal,
977
1077
  })
978
1078
  : await llm.completeWithTools({
979
1079
  system,
980
1080
  tools,
981
1081
  messages: conversationMessages,
1082
+ toolChoice,
982
1083
  signal: state.session._abortSignal,
983
1084
  });
984
1085
 
985
1086
  if (result.tool_calls?.length > 0) {
986
1087
  state.session._onStreamReset?.();
1088
+ const malformed = invalidToolCalls(result.tool_calls);
1089
+ if (malformed.length > 0) {
1090
+ const retries = Number(state.invalidToolCallRetries ?? 0);
1091
+ if (retries < 2) {
1092
+ state.session._onStep?.('Agent: malformed tool call rejected; retrying…');
1093
+ return {
1094
+ pendingToolCalls: null,
1095
+ messages: [
1096
+ ...(iterations === 0 ? [{ role: 'user', content: state.input }] : []),
1097
+ {
1098
+ role: 'user',
1099
+ content: 'Your previous tool call was incomplete or contained invalid JSON arguments. Call the appropriate available tool again with one complete valid JSON object. Do not narrate or reproduce the broken call.',
1100
+ },
1101
+ ],
1102
+ toolIterations: iterations + 1,
1103
+ readyToStream: false,
1104
+ inputClassification: classification,
1105
+ invalidToolCallRetries: retries + 1,
1106
+ };
1107
+ }
1108
+ const failure = 'Action non exécutée : l’appel d’outil généré par le modèle était incomplet.';
1109
+ emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: failure });
1110
+ return { response: failure, pendingToolCalls: null, readyToStream: false };
1111
+ }
987
1112
  // Close the streaming conversation entry now: the text streamed so
988
1113
  // far is this iteration's narration. Without this, the next
989
1114
  // iteration's deltas append to the SAME entry with no separator and
@@ -998,6 +1123,7 @@ export function createAgentGraph(options = {}) {
998
1123
  : [result.message];
999
1124
  return {
1000
1125
  pendingToolCalls: result.tool_calls,
1126
+ allowedToolNames: tools.map((item) => item?.function?.name).filter(Boolean),
1001
1127
  messages: newMessages,
1002
1128
  toolIterations: iterations + 1,
1003
1129
  readyToStream: false,
@@ -1026,6 +1152,26 @@ export function createAgentGraph(options = {}) {
1026
1152
  };
1027
1153
  }
1028
1154
 
1155
+ const canDelegate = tools.some((item) => item?.function?.name === 'runtime__delegate');
1156
+ if (!runtimeExecution && iterations === 0 && canDelegate && !state.retryWithoutTool
1157
+ && await classifyRequestedAction(llm, state.input, state.session._abortSignal)) {
1158
+ state.session._onStreamReset?.();
1159
+ state.session._onStep?.('Agent: action response rejected — delegation required; retrying…');
1160
+ return {
1161
+ pendingToolCalls: null,
1162
+ messages: [
1163
+ { role: 'user', content: state.input },
1164
+ result.message ?? { role: 'assistant', content: result.content ?? '' },
1165
+ { role: 'user', content: 'This is an action request. Call runtime__delegate now with the original objective only. Do not provide instructions or narration.' },
1166
+ ],
1167
+ toolIterations: 1,
1168
+ readyToStream: false,
1169
+ inputClassification: classification,
1170
+ retryWithoutTool: true,
1171
+ forceDelegation: true,
1172
+ };
1173
+ }
1174
+
1029
1175
  if (runtimeExecution && state.retryWithoutTool) {
1030
1176
  state.session._onStreamReset?.();
1031
1177
  const failure = 'Action non exécutée : Donna n’a appelé aucun outil disponible. Aucun job ni résultat n’a été créé.';
@@ -1038,7 +1184,43 @@ export function createAgentGraph(options = {}) {
1038
1184
  };
1039
1185
  }
1040
1186
 
1187
+ const invalidCommands = invalidSuggestedSlashCommands(result.content, state.session);
1188
+ const leakedTools = invalidUserFacingToolNames(result.content, state.session);
1189
+ if (invalidCommands.length > 0 || leakedTools.length > 0) {
1190
+ state.session._onStreamReset?.();
1191
+ const retries = Number(state.invalidResponseRetries ?? 0);
1192
+ if (retries < 2) {
1193
+ const canDelegate = tools.some((item) => item?.function?.name === 'runtime__delegate');
1194
+ state.session._onStep?.('Agent: invalid user-facing implementation detail rejected; retrying…');
1195
+ return {
1196
+ pendingToolCalls: null,
1197
+ messages: [
1198
+ ...(iterations === 0 ? [{ role: 'user', content: state.input }] : []),
1199
+ result.message ?? { role: 'assistant', content: result.content ?? '' },
1200
+ {
1201
+ role: 'user',
1202
+ content: [
1203
+ 'Rewrite the answer for the end user without internal MCP tool identifiers or unsolicited shell commands.',
1204
+ invalidCommands.length > 0 ? `Unavailable slash commands: /${invalidCommands.join(', /')}.` : null,
1205
+ leakedTools.length > 0 ? 'Do not print tool names; use them internally if needed.' : null,
1206
+ 'If the user requested an action and runtime delegation is available, call runtime__delegate instead of giving manual instructions.',
1207
+ ].filter(Boolean).join(' '),
1208
+ },
1209
+ ],
1210
+ toolIterations: iterations + 1,
1211
+ readyToStream: false,
1212
+ inputClassification: classification,
1213
+ invalidResponseRetries: retries + 1,
1214
+ forceDelegation: canDelegate,
1215
+ };
1216
+ }
1217
+ const failure = 'Réponse rejetée : Donna a exposé une instruction interne ou une procédure manuelle incorrecte.';
1218
+ emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: failure });
1219
+ return { response: failure, pendingToolCalls: null, readyToStream: false };
1220
+ }
1221
+
1041
1222
  if (useStreamWithTools) {
1223
+ if (result.content) state.session._onStream?.(result.content);
1042
1224
  emitAgentEvent(state.session, 'assistant_message', 'llm', { content: result.content ?? '' });
1043
1225
  // Text was streamed inline via session._onStream — no second LLM call needed.
1044
1226
  const newMessages = iterations === 0
@@ -1088,6 +1270,26 @@ export function createAgentGraph(options = {}) {
1088
1270
  const isInternalWikiTool = server === 'wiki' && (tool === 'plan_set' || tool === 'plan_done');
1089
1271
  const serverLabel = server === 'shell' ? 'Shell' : isInternalWikiTool ? 'Plan' : 'MCP';
1090
1272
  const toolName = server ? `${server}.${tool}` : call.function.name;
1273
+ // Hard guardrail: only execute tools that were actually offered this turn
1274
+ // (read-only tools + runtime controls + delegate). A capable model that
1275
+ // spots a mutating provider tool in the prompt and calls it directly must
1276
+ // be refused and steered back to runtime__delegate — this is what keeps
1277
+ // the orchestration capability-driven regardless of model strength.
1278
+ // Only interactive turns are constrained. Inside a runtime run
1279
+ // (_currentRunIdentity set) the graph legitimately executes the
1280
+ // already-validated, already-approved delegated task via provider tools.
1281
+ const runtimeExecutionTurn = Boolean(state.session._currentRunIdentity);
1282
+ const allowedNames = !runtimeExecutionTurn && Array.isArray(state.allowedToolNames) ? state.allowedToolNames : null;
1283
+ const isInternalCall = server === 'shell' || server === 'runtime' || isInternalWikiTool;
1284
+ if (allowedNames && server && !isInternalCall && !allowedNames.includes(`${server}__${tool}`)) {
1285
+ const refusal = `${server}__${tool} is not available in interactive mode. Do not call provider tools directly. For any action or mutation, call runtime__delegate with the user objective; only read-only tools and runtime controls may be called directly.`;
1286
+ state.session._onStep?.(`tool call refused (not offered): ${server}__${tool}`);
1287
+ emitAgentEvent(state.session, 'tool_call_result', 'tool', {
1288
+ callId: call.id, name: toolName, ok: false, result: refusal, summary: 'refused',
1289
+ });
1290
+ toolResultMessages.push({ role: 'tool', tool_call_id: call.id, content: refusal });
1291
+ continue;
1292
+ }
1091
1293
  if (resolved.normalized) {
1092
1294
  // Keep normalizations visible: the defensive routing must not hide
1093
1295
  // prompt/skill regressions that reintroduce unqualified names.
@@ -1104,9 +1306,11 @@ export function createAgentGraph(options = {}) {
1104
1306
  args: call.function.arguments ?? '{}',
1105
1307
  summary: argsSummary || 'calling...',
1106
1308
  });
1107
- // Immediate visible plan for any MCP call that doesn't yet have an _activity plan.
1309
+ // A plan represents work, never observation. Read-only inventory/status
1310
+ // calls stay out of Plan even when Donna uses them to answer a question.
1108
1311
  let minimalPlanActive = false;
1109
- if (!isInternalWikiTool && server !== 'shell' && !state.session.headlessPlan) {
1312
+ if (!isInternalWikiTool && server !== 'shell' && server !== 'runtime'
1313
+ && !isReadOnlyMcpCall(state.session, server, tool) && !state.session.headlessPlan) {
1110
1314
  minimalPlanActive = true;
1111
1315
  emitAgentEvent(state.session, 'plan_set', 'tool', {
1112
1316
  steps: [{ step: 1, id: null, description: toolName, status: 'running', _activityKey: null }],
@@ -1128,6 +1332,8 @@ export function createAgentGraph(options = {}) {
1128
1332
  );
1129
1333
  }
1130
1334
  let args = JSON.parse(call.function.arguments ?? '{}');
1335
+ const definition = toolDefinitionForCall(state.session, call.function.name);
1336
+ args = normalizeToolArgumentsFromSchema(args, definition?.function?.parameters);
1131
1337
  if (server === 'production' && tool === 'production_start_job' && state.session.workspace && !args.callerLabel) {
1132
1338
  args = { ...args, callerLabel: `${state.session.workspace}/wiki-manager` };
1133
1339
  }
@@ -1244,12 +1450,17 @@ export function createAgentGraph(options = {}) {
1244
1450
  return {
1245
1451
  messages: toolResultMessages,
1246
1452
  pendingToolCalls: null,
1453
+ forceDelegation: false,
1454
+ invalidToolCallRetries: 0,
1455
+ invalidResponseRetries: 0,
1247
1456
  };
1248
1457
  }
1249
1458
 
1250
1459
  function routeOrchestrator(state) {
1251
1460
  if (state.pendingToolCalls?.length > 0) return 'tool_executor';
1252
1461
  if (state.retryWithoutTool) return 'orchestrator';
1462
+ if (state.invalidToolCallRetries > 0 && state.response == null && !state.streamedInline) return 'orchestrator';
1463
+ if (state.invalidResponseRetries > 0 && state.response == null && !state.streamedInline) return 'orchestrator';
1253
1464
  return END;
1254
1465
  }
1255
1466