@dotdrelle/wiki-manager 0.12.11 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/.env.example +6 -0
  2. package/docker-compose.yml +1 -1
  3. package/package.json +1 -1
  4. package/src/agent/graph.js +377 -142
  5. package/src/agent/graph.test.js +576 -34
  6. package/src/agent/llm.js +5 -5
  7. package/src/cli/wiki-manager.js +294 -9
  8. package/src/cli/wiki-manager.test.js +28 -0
  9. package/src/commands/slash.js +80 -13
  10. package/src/commands/slash.test.js +9 -1
  11. package/src/contracts/schemas.js +33 -0
  12. package/src/contracts/schemas.test.js +14 -0
  13. package/src/core/agentEvents.js +6 -0
  14. package/src/core/agentEvents.test.js +26 -0
  15. package/src/core/agentLoop.js +15 -16
  16. package/src/core/agentLoop.test.js +9 -7
  17. package/src/core/buildInfo.json +2 -2
  18. package/src/core/mcp.js +13 -6
  19. package/src/core/mcp.test.js +0 -12
  20. package/src/core/skills.js +0 -28
  21. package/src/orchestrator/capabilityRegistry.js +14 -0
  22. package/src/orchestrator/capabilityRegistry.test.js +12 -1
  23. package/src/orchestrator/dependencyResolver.js +10 -1
  24. package/src/orchestrator/dispatcher.js +34 -3
  25. package/src/orchestrator/dispatcher.test.js +34 -0
  26. package/src/orchestrator/objectiveResolver.js +79 -0
  27. package/src/orchestrator/objectiveResolver.test.js +50 -0
  28. package/src/orchestrator/scheduler.js +24 -0
  29. package/src/orchestrator/scheduler.test.js +65 -1
  30. package/src/runtime/client.js +34 -2
  31. package/src/runtime/lifecycle.js +1 -1
  32. package/src/runtime/recoveryManager.js +14 -7
  33. package/src/runtime/runner.js +214 -14
  34. package/src/runtime/runner.test.js +100 -2
  35. package/src/runtime/server.js +43 -3
  36. package/src/runtime/supervisor.js +65 -1
  37. package/src/runtime/supervisor.test.js +80 -0
  38. package/src/shell/LeftPane.tsx +9 -2
  39. package/src/shell/repl.js +57 -42
  40. package/src/shell/repl.test.js +81 -12
  41. package/src/shell/tui.tsx +26 -3
  42. package/src/shell/useSession.ts +15 -3
@@ -14,8 +14,8 @@ import { extractActivity, formatActivitySummary, parseJsonText, sessionActivitie
14
14
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
15
15
  import { enqueueProductionJob, ensureJobQueue, formatQueue, productionLockBusy } from '../core/jobQueue.js';
16
16
  import { updateWorkspaceProfilePreference } from '../core/profile.js';
17
- import { createCapabilityRegistry } from '../orchestrator/capabilityRegistry.js';
18
- import { fetchRuntimeState, postRuntimeCancel, postRuntimeControl, postRuntimeKill } from '../runtime/client.js';
17
+ import { capabilityRegistryForSession } from '../orchestrator/capabilityRegistry.js';
18
+ import { fetchRuntimeState, postRuntimeApprove, postRuntimeCancel, postRuntimeControl, postRuntimeDelegate, postRuntimeKill } from '../runtime/client.js';
19
19
 
20
20
  const MAX_TOOL_ITERATIONS = 80;
21
21
  const MAX_SPINNER_ARG_LENGTH = 96;
@@ -27,7 +27,7 @@ const MAX_PROFILE_CHARS = 4000;
27
27
  const INTERNAL_TOOL_SERVERS = {
28
28
  wiki: ['plan_set', 'plan_done'],
29
29
  shell: ['run_command', 'read_command', 'profile_update'],
30
- runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue'],
30
+ runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue', 'delegate'],
31
31
  };
32
32
 
33
33
  const AGENT_SLASH_COMMANDS = new Set([
@@ -40,8 +40,8 @@ const AGENT_SLASH_COMMANDS = new Set([
40
40
  'services',
41
41
  'skills',
42
42
  'upload',
43
- 'uploads',
44
43
  'queue',
44
+ 'openui',
45
45
  ]);
46
46
 
47
47
  const SHELL_RUN_COMMAND_TOOL = {
@@ -50,7 +50,7 @@ const SHELL_RUN_COMMAND_TOOL = {
50
50
  name: 'shell__run_command',
51
51
  description: [
52
52
  'Run a deterministic wiki-manager slash command inside the current shell session.',
53
- 'Allowed commands: /workspace list, /workspace init <name> [path], /use <workspace>, /config, /status, /services, /skills, /skills show <name>, /skills run <name>, /upload <path>, /upload convert <id|pending>, /uploads.',
53
+ 'Allowed commands: /workspace list, /workspace init <name> [path], /use <workspace>, /config, /status, /services, /skills, /skills show <name>, /skills run <name>, /upload <path>, /upload convert <id|pending>.',
54
54
  'Do not use for arbitrary system shell commands, /workspace delete, /mcp call, /wiki run, /start, /stop, /logs, or /exit.',
55
55
  ].join(' '),
56
56
  parameters: {
@@ -73,7 +73,7 @@ const SHELL_READ_COMMAND_TOOL = {
73
73
  name: 'shell__read_command',
74
74
  description: [
75
75
  'Run a read-only deterministic wiki-manager slash command inside the current shell session.',
76
- 'Allowed commands: /help, /version, /config, /config list, /config status, /status, /services, /skills, /skills list, /skills show <name>, /uploads, /uploads list, /queue.',
76
+ 'Allowed commands: /help, /version, /config, /config list, /config status, /status, /services, /skills, /skills list, /skills show <name>, /queue.',
77
77
  'Do not use for workspace creation/deletion, uploads conversion, service start/stop, MCP calls, wiki runs, or any mutation.',
78
78
  ].join(' '),
79
79
  parameters: {
@@ -165,6 +165,22 @@ const RUNTIME_ENQUEUE_TOOL = {
165
165
  },
166
166
  };
167
167
 
168
+ const RUNTIME_DELEGATE_TOOL = {
169
+ type: 'function',
170
+ function: {
171
+ name: 'runtime__delegate',
172
+ description: 'Delegate the user objective to the runtime. Pass the objective in natural language without choosing a capability, operation, agent, plan, file list, or implementation. The runtime resolves the agent, obtains and validates the real plan before accepting the run.',
173
+ parameters: {
174
+ type: 'object',
175
+ additionalProperties: false,
176
+ properties: {
177
+ objective: { type: 'string', description: 'The complete user objective, preserving scope and constraints but containing no invented technical identifiers.' },
178
+ },
179
+ required: ['objective'],
180
+ },
181
+ },
182
+ };
183
+
168
184
  const WIKI_PLAN_SET_TOOL = {
169
185
  type: 'function',
170
186
  function: {
@@ -246,15 +262,124 @@ const AgentState = Annotation.Root({
246
262
  }),
247
263
  toolIterations: Annotation({ default: () => 0 }),
248
264
  pendingToolCalls: Annotation(),
265
+ allowedToolNames: Annotation(),
249
266
  inputClassification: Annotation(),
250
267
  readyToStream: Annotation(),
251
268
  streamContext: Annotation(),
252
269
  streamedInline: Annotation(),
253
270
  retryWithoutTool: Annotation({ default: () => false }),
271
+ invalidResponseRetries: Annotation({ default: () => 0 }),
272
+ invalidToolCallRetries: Annotation({ default: () => 0 }),
273
+ forceDelegation: Annotation({ default: () => false }),
254
274
  });
255
275
 
276
+ function invalidToolCalls(toolCalls) {
277
+ if (!Array.isArray(toolCalls)) return [];
278
+ return toolCalls.filter((call) => {
279
+ if (!call?.id || !call?.function?.name) return true;
280
+ try {
281
+ const args = JSON.parse(call.function.arguments || '{}');
282
+ return !args || typeof args !== 'object' || Array.isArray(args);
283
+ } catch {
284
+ return true;
285
+ }
286
+ });
287
+ }
288
+
289
+ export function normalizeToolArgumentsFromSchema(args, parameters) {
290
+ if (!args || typeof args !== 'object' || Array.isArray(args)) return args;
291
+ const schema = parameters && typeof parameters === 'object' ? parameters : {};
292
+ const properties = schema.properties && typeof schema.properties === 'object' ? schema.properties : {};
293
+ const required = Array.isArray(schema.required) ? schema.required.filter((key) => typeof key === 'string') : [];
294
+ const missing = required.filter((key) => args[key] === undefined);
295
+ const unknown = Object.keys(args).filter((key) => properties[key] === undefined);
296
+ if (missing.length !== 1 || unknown.length !== 1) return args;
297
+ const target = missing[0];
298
+ const source = unknown[0];
299
+ if (!schemaValueMatches(args[source], properties[target])) return args;
300
+ const normalized = { ...args, [target]: args[source] };
301
+ delete normalized[source];
302
+ return normalized;
303
+ }
304
+
305
+ function schemaValueMatches(value, propertySchema) {
306
+ const types = Array.isArray(propertySchema?.type) ? propertySchema.type : [propertySchema?.type];
307
+ if (types.includes(undefined) || types.includes(null)) return true;
308
+ return types.some((type) => {
309
+ if (type === 'array') return Array.isArray(value);
310
+ if (type === 'object') return value !== null && typeof value === 'object' && !Array.isArray(value);
311
+ if (type === 'integer') return Number.isInteger(value);
312
+ if (type === 'number') return typeof value === 'number' && Number.isFinite(value);
313
+ if (type === 'null') return value === null;
314
+ return typeof value === type;
315
+ });
316
+ }
317
+
318
+ function toolDefinitionForCall(session, callName) {
319
+ const internal = [
320
+ SHELL_RUN_COMMAND_TOOL,
321
+ SHELL_READ_COMMAND_TOOL,
322
+ SHELL_PROFILE_UPDATE_TOOL,
323
+ RUNTIME_STATUS_TOOL,
324
+ RUNTIME_CANCEL_TOOL,
325
+ RUNTIME_KILL_TOOL,
326
+ RUNTIME_APPROVE_TOOL,
327
+ RUNTIME_ENQUEUE_TOOL,
328
+ RUNTIME_DELEGATE_TOOL,
329
+ WIKI_PLAN_SET_TOOL,
330
+ WIKI_PLAN_DONE_TOOL,
331
+ ];
332
+ return [...internal, ...buildLlmTools(session?.mcp)]
333
+ .find((item) => item?.function?.name === callName) ?? null;
334
+ }
335
+
256
336
  function commandList(session) {
257
- return session.commands.map((command) => `/${command}`).join(', ');
337
+ return session.commands
338
+ .filter((command) => AGENT_SLASH_COMMANDS.has(command))
339
+ .map((command) => `/${command}`)
340
+ .join(', ');
341
+ }
342
+
343
+ export function invalidSuggestedSlashCommands(content, session) {
344
+ const allowed = new Set((session?.commands ?? []).filter((command) => AGENT_SLASH_COMMANDS.has(command)));
345
+ const candidates = new Set();
346
+ for (const line of String(content ?? '').split(/\r?\n/)) {
347
+ const trimmed = line.trim();
348
+ const standalone = trimmed.match(/^\/([a-z][\w-]*)\b/i);
349
+ if (standalone) candidates.add(standalone[1].toLowerCase());
350
+ for (const match of line.matchAll(/`\/([a-z][\w-]*)\b/gi)) candidates.add(match[1].toLowerCase());
351
+ }
352
+ return [...candidates].filter((command) => !allowed.has(command)).sort();
353
+ }
354
+
355
+ export function invalidUserFacingToolNames(content, session) {
356
+ const text = String(content ?? '');
357
+ const connected = buildLlmTools(session?.mcp)
358
+ .map((item) => item?.function?.name)
359
+ .filter(Boolean)
360
+ .filter((name) => text.includes(name));
361
+ const syntactic = [...text.matchAll(/\b[a-z][a-z0-9_-]*__[a-z][a-z0-9_-]*\b/gi)].map((match) => match[0]);
362
+ return [...new Set([...connected, ...syntactic])].sort();
363
+ }
364
+
365
+ async function classifyRequestedAction(llm, input, signal) {
366
+ try {
367
+ const result = await llm.completeWithTools({
368
+ system: [
369
+ 'Classify whether the user explicitly requests a real state-changing action now.',
370
+ 'Actions include starting, stopping, importing, ingesting, building, exporting, configuring, writing, deleting, or sending.',
371
+ 'Questions, explanations, status questions, greetings, and hypothetical discussions are not actions.',
372
+ 'Return JSON only: {"action":true} or {"action":false}.',
373
+ ].join('\n'),
374
+ tools: [],
375
+ messages: [{ role: 'user', content: String(input ?? '') }],
376
+ signal,
377
+ });
378
+ const text = String(result?.content ?? '').trim().replace(/^```(?:json)?\s*/i, '').replace(/\s*```$/, '');
379
+ return JSON.parse(text)?.action === true;
380
+ } catch {
381
+ return false;
382
+ }
258
383
  }
259
384
 
260
385
  function summarizeToolArguments(rawArguments) {
@@ -366,8 +491,7 @@ function assertAgentReadSlashCommandAllowed(commandLine) {
366
491
  command === 'services' ||
367
492
  command === 'queue' ||
368
493
  (command === 'config' && ['', 'list', 'status'].includes(subcommand)) ||
369
- (command === 'skills' && ['', 'list', 'show'].includes(subcommand)) ||
370
- (command === 'uploads' && ['', 'list'].includes(subcommand));
494
+ (command === 'skills' && ['', 'list', 'show'].includes(subcommand));
371
495
  if (!allowed) {
372
496
  throw new Error(`Read-only command is not available to the agent: /${parts.join(' ')}`);
373
497
  }
@@ -512,9 +636,7 @@ function emitAgentEvent(session, type, origin, payload = {}) {
512
636
  // agents. This is the live registry the dispatcher will resolve against —
513
637
  // a plan declaring anything outside this set can only stall forever.
514
638
  export function knownCapabilityIds(session) {
515
- const registry = session?.capabilityRegistry ?? createCapabilityRegistry({
516
- agents: session?.agentRegistrySnapshot ?? session?.agents ?? [],
517
- });
639
+ const registry = capabilityRegistryForSession(session);
518
640
  const snapshot = typeof registry.snapshot === 'function' ? registry.snapshot() : registry;
519
641
  return [...new Set(Object.keys(snapshot ?? {}).map((key) => {
520
642
  const index = key.lastIndexOf('@');
@@ -522,6 +644,33 @@ export function knownCapabilityIds(session) {
522
644
  }))].sort();
523
645
  }
524
646
 
647
+ // Deterministic fragment→plan mapping, shared by the agent_plan tool bridge
648
+ // and the /ingest-style direct capability runs: the parallel path must not
649
+ // depend on an LLM copying fields correctly.
650
+ export function planStepsFromFragment(payload) {
651
+ const tasks = Array.isArray(payload?.tasks) ? payload.tasks : [];
652
+ return tasks.map((task, index) => normalizeDeclaredPlanStep({
653
+ id: task.id,
654
+ description: task.label ?? task.id ?? `Task ${index + 1}`,
655
+ requiredCapability: task.requiredCapability ?? payload.capability ?? null,
656
+ operation: task.operation ?? null,
657
+ arguments: task.arguments ?? {},
658
+ dependsOn: task.dependsOn ?? [],
659
+ outputRefs: (task.expectedOutputRefs ?? []).map((ref) => (ref && typeof ref === 'object' ? String(ref.ref ?? '') : String(ref))).filter(Boolean),
660
+ groupId: task.groupId ?? null,
661
+ dependsOnGroup: task.dependsOnGroup ?? null,
662
+ parallelizable: task.parallelizable,
663
+ barrier: task.barrier,
664
+ locks: task.locks,
665
+ requiresApproval: task.requiresApproval,
666
+ approvalClass: task.approvalClass,
667
+ approvalSummary: task.approvalSummary,
668
+ idempotencyKey: task.idempotencyKey,
669
+ progressWeight: task.progressWeight,
670
+ recommendedConcurrency: task.recommendedConcurrency,
671
+ }, index));
672
+ }
673
+
525
674
  async function handleRuntimeControlTool(session, tool, args = {}) {
526
675
  const url = session.runtime?.url ?? null;
527
676
  if (!url) return 'Runtime not connected: no runtime URL available in this session.';
@@ -536,8 +685,34 @@ async function handleRuntimeControlTool(session, tool, args = {}) {
536
685
  return result.cancelled ? 'Runtime run cancellation requested.' : `No active run to cancel${result.reason ? ` (${result.reason})` : ''}.`;
537
686
  }
538
687
  if (tool === 'approve') {
539
- const result = await postRuntimeControl('message', { url, workspace, input: 'approve', intent: 'approve' });
540
- return String(result?.explanation ?? (result?.accepted ? 'Approval granted.' : 'No pending approval found.'));
688
+ const state = await fetchRuntimeState({ url, workspace });
689
+ const pending = (Array.isArray(state?.approvals) ? state.approvals : [])
690
+ .filter((approval) => approval.status === 'pending_approval');
691
+ const runId = state?.runId
692
+ ?? state?.runs?.find((run) => ['running', 'pending_approval'].includes(run.status))?.id
693
+ ?? null;
694
+ if (!runId || pending.length === 0) return 'No pending approval found.';
695
+ const approvalClasses = [...new Set(pending.flatMap((approval) => {
696
+ const value = approval.approvalClasses ?? approval.approvalClass ?? [];
697
+ return Array.isArray(value) ? value : [value];
698
+ }).map(String).filter(Boolean))];
699
+ const result = await postRuntimeApprove({
700
+ url,
701
+ workspace,
702
+ runId,
703
+ scope: 'run',
704
+ planRevision: state?.planRevision ?? null,
705
+ approvalClasses: approvalClasses.length > 0 ? approvalClasses : ['default'],
706
+ });
707
+ return result?.approved ? 'Current validated plan approved.' : 'No pending approval found.';
708
+ }
709
+ if (tool === 'delegate') {
710
+ const objective = String(args.objective ?? '').trim();
711
+ if (!objective) return 'Delegation rejected: missing objective.';
712
+ const result = await postRuntimeDelegate(objective, { url, workspace });
713
+ return result?.runId
714
+ ? `Délégation acceptée (${String(result.runId).slice(0, 8)}) après validation du plan réel : ${result.delegation?.tasks ?? 0} tâche(s), ${result.delegation?.agent ?? 'agent résolu'}. L'approbation porte sur ce plan.`
715
+ : `Délégation refusée : ${result?.error ?? JSON.stringify(result)}`;
541
716
  }
542
717
  if (tool === 'enqueue') {
543
718
  const result = await postRuntimeControl('message', { url, workspace, input: String(args.input ?? ''), intent: 'enqueue' });
@@ -666,7 +841,15 @@ export function buildAgentSystemPrompt(state) {
666
841
  const workspace = state.session.workspace ?? 'no workspace selected';
667
842
  const wikirc = state.session.wikirc?.profile ?? 'no profile loaded';
668
843
  const language = state.session.language ?? 'en-US';
669
- const mcpTools = formatMcpToolsForAgent(state.session.mcp);
844
+ // Advertise only the read-only tools Donna may call directly. Listing
845
+ // mutating provider tools (e.g. production__production_start_job) here teaches
846
+ // a capable model to invoke them directly and bypass runtime__delegate.
847
+ const mcpTools = formatMcpToolsForAgent(state.session.mcp, {
848
+ include: (qualifiedName, tool) => isDonnaReadTool({
849
+ function: { name: qualifiedName },
850
+ readOnly: tool?.annotations?.readOnlyHint === true,
851
+ }),
852
+ });
670
853
  const skills = formatSkillsForAgent(state.session);
671
854
  const customPrompt = state.session.systemPrompt ?? null;
672
855
  const workspaceProfile = loadWorkspaceProfile(state.session.workspacePath);
@@ -681,73 +864,39 @@ export function buildAgentSystemPrompt(state) {
681
864
  `Current wikirc profile: ${wikirc}.`,
682
865
  `Available primitives: ${commandList(state.session)}.`,
683
866
  'Only announce or call slash commands that appear exactly in Available primitives. Do not invent command names, subcommands, or arguments.',
684
- 'Connected MCP tools (use the server__tool naming convention for tool calls):',
867
+ 'Connected read-only MCP tools you may call directly to answer questions (server__tool naming convention). Every mutation or action — ingest, build, export, configure, send, write — goes through runtime__delegate, never a direct provider tool call:',
685
868
  mcpTools,
686
869
  'Current local MCP job queue:',
687
870
  formatQueue(state.session),
688
871
  'Available skills:',
689
872
  skills,
690
- 'You can call MCP tools directly using the provided tool functions.',
873
+ 'In interactive agent mode you may call only the read-only tools and runtime control/delegation tools actually provided to you.',
691
874
  'When the user asks for an action that can be performed with connected MCP tools or safe primitives, do not answer with future intent such as "I will call...", "I am going to run...", or "launching..." unless you also call the tool in the same turn. Either call the tool now, ask for the exact missing required arguments, or explain the concrete blocker.',
692
875
  'Execution truthfulness: never invent a job id, status, percentage, duration, generated file, file content, URL, command, or tool result. An action is executed only when you call an available tool and receive its result. Examples and placeholders are forbidden in execution reports.',
693
876
  'After any completed action, give a short factual summary based only on the tool result: outcome and concrete outputs or references actually returned. Mention a viewing primitive only when it exists in Available primitives and is relevant. Do not interpret generated content, propose verification checklists, invent next steps, or suggest commands unless the user explicitly asks.',
877
+ 'Never add a "Next steps", "Prochaines étapes", "À suivre", options, or suggestions section unless the user explicitly asks what to do next. End after the requested result or the concrete error.',
694
878
  'When calling a tool, emit no preliminary narration. Call it directly; the PLAN and Activity panels show progress. After completion, keep the final response concise and proportional to the result.',
695
- 'For connector configuration/setup/update requests, if a matching setup/configuration tool is connected and the required arguments are known, call it immediately. If the connector or tool is not connected, say which concrete capability is missing and recommend the exact service/status primitive to inspect it. Do not invent a pending connector action in plain text.',
696
- 'For workspace-scoped external MCP tools, the orchestrator enforces workspace injection. Use the active workspace for configuration, source, import, export, conversion, and generation tools unless a tool is explicitly job-scoped and only requires a job id.',
697
- 'You can call shell__run_command for safe manager slash commands such as /workspace list, /workspace init <name> [path], /use <workspace>, /config, /status, /services, /skills, /skills show <name>, and /skills run <name>.',
698
- 'Skills are workflow instructions, not executable code. When a user asks to run a skill, inspect it, propose the concrete primitive/tool plan, and ask for confirmation before costly or mutating actions.',
699
- [
700
- state.session.headless ? 'HEADLESS MODE ACTIVE. Execute the requested skill or task autonomously using available safe primitives and MCP tools. Do not ask for interactive confirmation unless the request is genuinely ambiguous or outside the loaded workspace.' : null,
701
- '',
702
- 'You have two internal planning tools: wiki__plan_set and wiki__plan_done.',
703
- 'Prefer MCP tools that declare their own plan via _activity.plan.steps — when such a tool returns _activity, the shell creates and tracks the plan automatically without requiring wiki__plan_set.',
704
- 'Use wiki__plan_set when the MCP tool cannot declare its own plan or when the task spans multiple independent tools (e.g. CME export then email report). For a single self-describing async job, wiki__plan_set is optional.',
705
- '',
706
- (() => {
707
- const capabilityIds = knownCapabilityIds(state.session);
708
- return capabilityIds.length > 0
709
- ? `Known orchestration capabilities — the ONLY values allowed in requiredCapability: ${capabilityIds.join(', ')}. Never invent capability names; a plan declaring an unknown capability will be rejected. A step you execute yourself directly takes requiredCapability: null.`
710
- : 'No orchestration capabilities discovered yet: declare plan steps with requiredCapability: null and execute them yourself with the connected MCP tools.';
711
- })(),
712
- '',
713
- 'Task startup:',
714
- ' 1. If the next MCP tool returns _activity.plan.steps, call that tool directly; the shell will create the visible plan from the returned activity.',
715
- ' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, operation, arguments, dependsOn, outputRefs}; capability steps need operation+arguments for the dispatcher to execute them; a legacy list of strings is still accepted.',
716
- ' Multi-tool example: wiki__plan_set(steps=[{id:"cme-export",description:"CME export",requiredCapability:"external-source.export",dependsOn:[],outputRefs:["raw/untracked"]},{id:"production",description:"Production pipeline",requiredCapability:"knowledge.pipeline",dependsOn:["cme-export"],outputRefs:["deliverables"]}])',
717
- ' 3. Immediately execute the first step using the appropriate MCP tool. Do not start step 2 in the same turn unless one async pipeline tool owns and declares the whole sequence.',
718
- ' For synchronous steps (result is immediate, no _activity polling), call wiki__plan_done(step=1) after confirming success.',
719
- ' For async MCP jobs (returns _activity with poll), the orchestrator tracks completion automatically.',
720
- '',
721
- state.session.headless ? [
722
- 'Headless follow-up turns — the orchestrator re-invokes you with:',
723
- ' (a) the original task,',
724
- ' (b) the current plan status — [✓] done / [✗] failed / [ ] pending,',
725
- ' (c) the just-completed activities.',
726
- ' Read the plan status. Find the first [ ] pending step. Execute it only.',
727
- ' Never re-execute a [✓] or [✗] step. Never skip a [ ] step.',
728
- '',
729
- 'Final turn — when all steps are [✓] or [✗]: respond with a concise summary. Do not start new actions.',
730
- ].join('\n') : null,
731
- '',
732
- 'On failure: if a completed activity is failed/error/cancelled, call wiki__plan_done(step=N, status="failed") then stop with a clear error report.',
733
- ].filter(Boolean).join('\n'),
879
+ 'Keep every response synthetic and information-dense. Use only the lines needed, and never exceed roughly 15 to 20 short lines even for a detailed answer. Never expose internal reasoning, repeated checks, tool-selection commentary, or a chronological diary. Prioritize the result, essential facts, concrete errors, and actual outputs.',
880
+ 'Configuration, connector, import, export, conversion, generation, and every other mutation are actions: delegate the objective to the runtime instead of calling an external tool directly.',
881
+ 'For any question about the current workspace inventory or what is waiting there, call wiki__wiki_workspace_status first and answer only from its result. This is the canonical read-only workspace state; do not reconstruct it from upload, connector, or production tools.',
882
+ 'Tool identifiers are private implementation details. Never print MCP tool names such as server__tool in a user-facing answer. Describe the human result instead.',
883
+ 'Never suggest a manual filesystem command or implementation workaround unless the user explicitly asks for manual instructions. For an action request, delegate the objective and let the specialized agent determine paths and operations from its live contract.',
884
+ 'Skills are documentation only in this stabilized version. Never execute a skill from conversation; delegate the user objective.',
734
885
  'For service actions, recommend only available service primitives from Available primitives, with the exact service name when the primitive supports one.',
735
- 'Scope discipline: execute ONLY the action(s) the user explicitly requested. Never chain additional mutating operations (ingest, build, export, polish, delete, send…) that the user did not ask for — even when diagnostics or recommendations suggest them. Finish the requested work, then list the suggested follow-ups in your final answer and stop. Example: "applique les recommandations de config" means apply the config; it does NOT authorize launching the ingest those recommendations mention.',
736
- 'Disambiguate export requests carefully.',
737
- 'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`cme__cme_export_run`, then `cme__cme_export_status`). Never use production `type=export` for Confluence source export.',
738
- 'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`production__production_start_job` with `type:"export"` or pipeline steps). Require the deliverable path when exporting deliverables.',
739
- 'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts.',
740
- 'MULTI-DOCUMENT ingest (more than 2 files, or "ingest everything pending"): call production__agent_plan first, e.g. {capability:"knowledge.update", operation:"ingest", constraints:{maxConcurrency:3, requireApprovalForMutations:true}}. The shell integrates the returned task graph as the plan automatically and the orchestrator dispatches the per-document tasks IN PARALLEL with an approval gate. Do not call production__production_start_job for multi-document ingest — that creates one monolithic sequential job.',
741
- 'Single-document ingest or one-off jobs (doctor, one build, one export): production__production_start_job is fine. To chain sequential steps (e.g. build then polish) use ONE call with type="pipeline" and steps=["build","polish"] — never separate jobs (the first is asynchronous). For existing deliverables where content stability matters, pass stabilize:true. Do not ask the user to confirm between steps.',
742
- 'Long-running MCP jobs: do not call the same status tool more than once consecutively. When chaining jobs sequentially: (1) start the job, report job/activity id and status; (2) check status once — if done, proceed to the next step immediately; (3) if still running, report status, list the remaining steps, and return control; (4) when re-invoked, check status first, then proceed. Do not spin-poll (status → status → status with no new action between). The shell activity panel monitors non-terminal jobs automatically.',
743
- 'If production__production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
744
- 'For diagnostics (doctor), use production__production_start_job with type="doctor" like any other production job; /wiki run doctor is only the fallback when the production MCP is not connected. Use /workspace init <name> [path] for low-level non-interactive workspace creation. In the interactive TUI, /new <name> opens the setup wizard. Use /wiki for index, or /wiki run index through the explicit backup hatch. Use /wiki run init only for explicit current-workspace llm-wiki init.',
886
+ 'Scope discipline: execute ONLY the action(s) the user explicitly requested. Never chain additional mutating operations (ingest, build, export, polish, delete, send…) that the user did not ask for — even when diagnostics or recommendations suggest them. Finish with the requested result and stop. Example: "applique les recommandations de config" means apply the config; it does NOT authorize launching the ingest those recommendations mention.',
887
+ state.session.runtime?.url
888
+ ? 'The runtime is connected and runtime__delegate is bound and available to you right now — it is a tool you call directly, not a slash command or a missing primitive. It is the ONLY way to execute an action (ingest, build, export, configure, send…). Never tell the user that delegation or the runtime is unavailable while it is connected; call runtime__delegate instead.'
889
+ : 'No runtime is connected, so you cannot execute actions. State that plainly and name the runtime connection as the missing capability — do not invent a workaround.',
890
+ 'For any requested action, call runtime__delegate with the user objective only. Never choose a capability, operation, agent, plan, or implementation yourself. The runtime resolves the registry and validates the provider plan before accepting. Never call <provider>__agent_plan, <provider>__agent_execute, legacy production__production_start_job, wiki__plan_set, or wiki__plan_done from interactive chat.',
891
+ 'Do not ask the user which sources, files, connectors, or templates to use for an ingest, build, or export: the specialized agent discovers them from the workspace. When the objective is clear (e.g. "lance une ingestion"), delegate it as stated, without a clarifying question.',
892
+ 'If runtime__delegate returns a blocker or no specialized provider is available, report only that concrete blocker concisely. Never replace the missing execution path with a suggested slash command, skill, MCP tool name, manual file move, administrator escalation, or alternative workflow unless the user explicitly asks for alternatives.',
893
+ 'For workspace inventory and page listings, use the connected wiki MCP read tools. Never invent or call a /wiki shell command through shell__run_command. Use /workspace init <name> [path] for low-level non-interactive workspace creation; in the interactive TUI, /new <name> opens the setup wizard.',
745
894
  'If an action requires tools or skills not available yet, explain the limitation and name the expected primitive.',
746
895
  workspaceProfile
747
896
  ? `Workspace profile (.wiki/profile.md) — durable user preferences, apply these to every reply (tone, tutoiement/vouvoiement, formatting, etc.):\n${workspaceProfile}`
748
897
  : null,
749
898
  'Runtime control: you have runtime__status, runtime__cancel, runtime__kill, runtime__approve and runtime__enqueue. When the user asks to stop, remove, clean or kill the current run, its jobs or the queue ("supprime le job et la queue", "arr\u00eate tout"), call runtime__kill (or runtime__cancel for a soft stop of just the run) and confirm what was stopped. For questions about what is running or queued, call runtime__status and answer from its data. When the user consents to a pending approval in any phrasing ("vas-y", "ok pour l\'export"), call runtime__approve. When the user asks for a NEW action while a run is active, do not execute it: propose runtime__enqueue (run it after) or, if they insist it replaces the current work, runtime__kill then the new action.',
750
- 'When the user explicitly asks you to remember, persist, or update durable preference/profile information, call wiki__profile_update when it is available; otherwise call shell__profile_update. Do not just acknowledge in text without calling a profile update tool.',
899
+ 'Durable profile updates are actions in this stabilized version: delegate them instead of writing directly.',
751
900
  ].filter(Boolean).join('\n');
752
901
 
753
902
  return customPrompt ? `${customPrompt}\n\n${agentContext}` : agentContext;
@@ -777,61 +926,48 @@ export function formatLlmUnavailableMessage(reason) {
777
926
  return `⚠ LLM injoignable : ${clean || 'raison inconnue'}`;
778
927
  }
779
928
 
780
- // Verbs that clearly request work (a runtime run), in French and English.
781
- // "configure/configurer" is an action; the nouns "config/configuration" are
782
- // NOT matched here — asking for a config is an observe request.
783
- const ACTION_REQUEST_PATTERN = /\b(lance|relance|d[eé]marre|start|ex[eé]cute|execute|g[eé]n[eè]re|generate|build|construis|exporte?|ingest\w*|ing[eè]re|importe?|convert(?:is|it|s)?|cr[eé]e|create|polish|publie|publish|d[eé]ploie|deploy|envoie|send|configure[rsz]?|setup|installe|update|mets? [aà] jour|supprime|delete|efface|nettoie|clean|r[eé]pare|fix|corrige)\b/i;
784
-
785
- // Explicit explanation/question markers dominate action verbs: "explique le
786
- // build" is a question about the build, not a request to build.
787
- const EXPLANATION_REQUEST_PATTERN = /\b(explique|explain|pourquoi|why|comment|how|c'est quoi|qu'est[- ]ce)\b/i;
788
-
789
- export function classifyAgentInput(input, session) {
790
- const lower = String(input ?? '').toLowerCase();
791
- const hasActiveRun = session?.agentProjection?.status === 'running'
792
- || sessionActivities(session).some((activity) => !activity.terminal);
793
- if (/\b(valide tout|approve all|approve|approuve|valid[eé]|ok pour tout|go pour tout)\b/i.test(lower)) {
794
- return { kind: 'approve', confidence: 0.86, reason: 'approval_request', activeRun: hasActiveRun };
795
- }
796
- if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort|supprime|kill|tue|purge|vide la (file|queue)|nettoie la (file|queue))\b/i.test(lower)) {
797
- return { kind: 'cancel', confidence: 0.86, reason: 'cancel_request', activeRun: hasActiveRun };
798
- }
799
- if (/\b(plus tard|later|ensuite|apr[eè]s ce run|enqueue|mets en file|met en file|futur|next run|future run)\b/i.test(lower)) {
800
- return { kind: 'enqueue_run', confidence: 0.8, reason: 'future_run_request', activeRun: hasActiveRun };
801
- }
802
- if (EXPLANATION_REQUEST_PATTERN.test(lower)) {
803
- return { kind: 'observe', confidence: 0.86, reason: 'explanation_request', activeRun: hasActiveRun };
804
- }
805
- // Observe markers only win when no action verb is present: "où en est le
806
- // run" is observe, "lance le run" is an action request.
807
- if (/\b(o[uù] en es[t-]|status|statut|progress|progression|run|job|queue|logs?|inspect|show|montre|affiche|donne|liste|list|quel(?:le)?s?|combien|config(?:uration)?|quoi de neuf)\b/i.test(lower)
808
- && !ACTION_REQUEST_PATTERN.test(lower)) {
809
- return { kind: 'observe', confidence: 0.86, reason: 'status_or_explanation_request', activeRun: hasActiveRun };
810
- }
811
- if (hasActiveRun && /\b(ajoute|add|change|modifie|modify|remplace|replace|retire|remove|skip|ignore|plan|step|t[aâ]che)\b/i.test(lower)) {
812
- return { kind: 'modify_run', confidence: 0.78, reason: 'active_run_change_request', activeRun: hasActiveRun };
813
- }
814
- if (hasActiveRun && /\b(lance|run|g[eé]n[eè]re|build|export|cr[eé]e|create|send|envoie|ingest|convert|importe|import)\b/i.test(lower)) {
815
- return { kind: 'ambiguous', confidence: 0.45, reason: 'active_run_action_is_ambiguous', activeRun: hasActiveRun };
816
- }
817
- if (ACTION_REQUEST_PATTERN.test(lower)) {
818
- return { kind: 'start_run', confidence: 0.8, reason: 'action_request', activeRun: hasActiveRun };
819
- }
820
- return { kind: 'converse', confidence: 0.62, reason: 'plain_conversation', activeRun: hasActiveRun };
821
- }
822
-
823
929
  function toolsForClassification(classification, writeTools, session = null) {
824
930
  const controlTools = session?.runtime?.url
825
931
  ? [RUNTIME_STATUS_TOOL, RUNTIME_CANCEL_TOOL, RUNTIME_KILL_TOOL, RUNTIME_APPROVE_TOOL, RUNTIME_ENQUEUE_TOOL]
826
932
  : [];
827
- if (classification.activeRun && ['converse', 'observe', 'ambiguous', 'approve', 'cancel', 'enqueue_run'].includes(classification.kind)) {
933
+ // Provider discovery and validation belong to the runtime. Hiding
934
+ // delegation while the shell snapshot is temporarily empty forced Donna
935
+ // to invent commands instead of submitting the objective.
936
+ const capabilityRunTools = session?.runtime?.url && !classification.activeRun
937
+ ? [RUNTIME_DELEGATE_TOOL]
938
+ : [];
939
+ if (classification.activeRun) {
828
940
  // During an active run Donna gets read + profile + the runtime control
829
941
  // suite: she can answer, approve, enqueue for later, soft-cancel or
830
942
  // kill — but she must not fire new MCP jobs alongside the run (that is
831
943
  // what runtime__enqueue is for). No canned regex answers anywhere.
832
- return [SHELL_READ_COMMAND_TOOL, SHELL_PROFILE_UPDATE_TOOL, ...controlTools];
944
+ return [SHELL_READ_COMMAND_TOOL, ...controlTools];
945
+ }
946
+ if (session?.runtime?.url) {
947
+ const readTools = writeTools.filter(isDonnaReadTool);
948
+ return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...readTools];
833
949
  }
834
- return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...writeTools];
950
+ return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...writeTools];
951
+ }
952
+
953
+ function isDonnaReadTool(item) {
954
+ const name = String(item?.function?.name ?? '');
955
+ if (!name || name.startsWith('shell__') || name === 'wiki__plan_set' || name === 'wiki__plan_done') return false;
956
+ if (item?.readOnly === true) return true;
957
+ const tool = name.includes('__') ? name.slice(name.indexOf('__') + 2) : name;
958
+ return tool === 'wiki_workspace_status'
959
+ || tool === 'agent_describe'
960
+ || tool === 'agent_status'
961
+ || /(?:^|_)(?:status|list|search|read|get)$/.test(tool);
962
+ }
963
+
964
+ function isReadOnlyMcpCall(session, server, tool) {
965
+ const descriptor = (session?.mcp?.[server]?.tools ?? [])
966
+ .find((item) => String(item?.name ?? '') === tool || String(item?.name ?? '').endsWith(`__${tool}`));
967
+ return isDonnaReadTool({
968
+ function: { name: `${server}__${tool}` },
969
+ readOnly: descriptor?.readOnly === true,
970
+ });
835
971
  }
836
972
 
837
973
  export function createAgentGraph(options = {}) {
@@ -867,7 +1003,13 @@ export function createAgentGraph(options = {}) {
867
1003
  const classification = iterations === 0
868
1004
  ? (runtimeExecution
869
1005
  ? { kind: 'execute_run', confidence: 1, reason: 'runtime_run_execution', activeRun: true }
870
- : classifyAgentInput(state.input, state.session))
1006
+ : {
1007
+ kind: 'agent_turn',
1008
+ confidence: 1,
1009
+ reason: 'agent_mode_llm_decision',
1010
+ activeRun: state.session?.agentProjection?.status === 'running'
1011
+ || sessionActivities(state.session).some((activity) => !activity.terminal),
1012
+ })
871
1013
  : (state.inputClassification ?? { kind: 'modify_run', confidence: 1, reason: 'tool_iteration' });
872
1014
  if (iterations === 0) {
873
1015
  state.session._onStep?.(`Agent: classified input as ${classification.kind}`);
@@ -896,28 +1038,54 @@ export function createAgentGraph(options = {}) {
896
1038
 
897
1039
  try {
898
1040
  const useStreamWithTools = typeof llm.streamWithTools === 'function';
899
- const suppressExecutionNarration = runtimeExecution && (iterations === 0 || state.retryWithoutTool);
1041
+ const toolChoice = state.forceDelegation
1042
+ ? { type: 'function', function: { name: 'runtime__delegate' } }
1043
+ : 'auto';
900
1044
  const result = useStreamWithTools
901
1045
  ? await llm.streamWithTools({
902
1046
  system,
903
1047
  tools,
904
1048
  messages: conversationMessages,
905
- onTextDelta: (delta) => {
906
- if (suppressExecutionNarration) return;
907
- emitAgentEvent(state.session, 'assistant_delta', 'llm', { delta });
908
- state.session._onStream?.(delta);
909
- },
1049
+ toolChoice,
1050
+ // Buffer until validation. Invalid commands and malformed tool
1051
+ // calls must never flash hundreds of lines before disappearing.
1052
+ onTextDelta: () => {},
910
1053
  signal: state.session._abortSignal,
911
1054
  })
912
1055
  : await llm.completeWithTools({
913
1056
  system,
914
1057
  tools,
915
1058
  messages: conversationMessages,
1059
+ toolChoice,
916
1060
  signal: state.session._abortSignal,
917
1061
  });
918
1062
 
919
1063
  if (result.tool_calls?.length > 0) {
920
1064
  state.session._onStreamReset?.();
1065
+ const malformed = invalidToolCalls(result.tool_calls);
1066
+ if (malformed.length > 0) {
1067
+ const retries = Number(state.invalidToolCallRetries ?? 0);
1068
+ if (retries < 2) {
1069
+ state.session._onStep?.('Agent: malformed tool call rejected; retrying…');
1070
+ return {
1071
+ pendingToolCalls: null,
1072
+ messages: [
1073
+ ...(iterations === 0 ? [{ role: 'user', content: state.input }] : []),
1074
+ {
1075
+ role: 'user',
1076
+ content: 'Your previous tool call was incomplete or contained invalid JSON arguments. Call the appropriate available tool again with one complete valid JSON object. Do not narrate or reproduce the broken call.',
1077
+ },
1078
+ ],
1079
+ toolIterations: iterations + 1,
1080
+ readyToStream: false,
1081
+ inputClassification: classification,
1082
+ invalidToolCallRetries: retries + 1,
1083
+ };
1084
+ }
1085
+ const failure = 'Action non exécutée : l’appel d’outil généré par le modèle était incomplet.';
1086
+ emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: failure });
1087
+ return { response: failure, pendingToolCalls: null, readyToStream: false };
1088
+ }
921
1089
  // Close the streaming conversation entry now: the text streamed so
922
1090
  // far is this iteration's narration. Without this, the next
923
1091
  // iteration's deltas append to the SAME entry with no separator and
@@ -932,6 +1100,7 @@ export function createAgentGraph(options = {}) {
932
1100
  : [result.message];
933
1101
  return {
934
1102
  pendingToolCalls: result.tool_calls,
1103
+ allowedToolNames: tools.map((item) => item?.function?.name).filter(Boolean),
935
1104
  messages: newMessages,
936
1105
  toolIterations: iterations + 1,
937
1106
  readyToStream: false,
@@ -960,6 +1129,26 @@ export function createAgentGraph(options = {}) {
960
1129
  };
961
1130
  }
962
1131
 
1132
+ const canDelegate = tools.some((item) => item?.function?.name === 'runtime__delegate');
1133
+ if (!runtimeExecution && iterations === 0 && canDelegate && !state.retryWithoutTool
1134
+ && await classifyRequestedAction(llm, state.input, state.session._abortSignal)) {
1135
+ state.session._onStreamReset?.();
1136
+ state.session._onStep?.('Agent: action response rejected — delegation required; retrying…');
1137
+ return {
1138
+ pendingToolCalls: null,
1139
+ messages: [
1140
+ { role: 'user', content: state.input },
1141
+ result.message ?? { role: 'assistant', content: result.content ?? '' },
1142
+ { role: 'user', content: 'This is an action request. Call runtime__delegate now with the original objective only. Do not provide instructions or narration.' },
1143
+ ],
1144
+ toolIterations: 1,
1145
+ readyToStream: false,
1146
+ inputClassification: classification,
1147
+ retryWithoutTool: true,
1148
+ forceDelegation: true,
1149
+ };
1150
+ }
1151
+
963
1152
  if (runtimeExecution && state.retryWithoutTool) {
964
1153
  state.session._onStreamReset?.();
965
1154
  const failure = 'Action non exécutée : Donna n’a appelé aucun outil disponible. Aucun job ni résultat n’a été créé.';
@@ -972,7 +1161,43 @@ export function createAgentGraph(options = {}) {
972
1161
  };
973
1162
  }
974
1163
 
1164
+ const invalidCommands = invalidSuggestedSlashCommands(result.content, state.session);
1165
+ const leakedTools = invalidUserFacingToolNames(result.content, state.session);
1166
+ if (invalidCommands.length > 0 || leakedTools.length > 0) {
1167
+ state.session._onStreamReset?.();
1168
+ const retries = Number(state.invalidResponseRetries ?? 0);
1169
+ if (retries < 2) {
1170
+ const canDelegate = tools.some((item) => item?.function?.name === 'runtime__delegate');
1171
+ state.session._onStep?.('Agent: invalid user-facing implementation detail rejected; retrying…');
1172
+ return {
1173
+ pendingToolCalls: null,
1174
+ messages: [
1175
+ ...(iterations === 0 ? [{ role: 'user', content: state.input }] : []),
1176
+ result.message ?? { role: 'assistant', content: result.content ?? '' },
1177
+ {
1178
+ role: 'user',
1179
+ content: [
1180
+ 'Rewrite the answer for the end user without internal MCP tool identifiers or unsolicited shell commands.',
1181
+ invalidCommands.length > 0 ? `Unavailable slash commands: /${invalidCommands.join(', /')}.` : null,
1182
+ leakedTools.length > 0 ? 'Do not print tool names; use them internally if needed.' : null,
1183
+ 'If the user requested an action and runtime delegation is available, call runtime__delegate instead of giving manual instructions.',
1184
+ ].filter(Boolean).join(' '),
1185
+ },
1186
+ ],
1187
+ toolIterations: iterations + 1,
1188
+ readyToStream: false,
1189
+ inputClassification: classification,
1190
+ invalidResponseRetries: retries + 1,
1191
+ forceDelegation: canDelegate,
1192
+ };
1193
+ }
1194
+ const failure = 'Réponse rejetée : Donna a exposé une instruction interne ou une procédure manuelle incorrecte.';
1195
+ emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: failure });
1196
+ return { response: failure, pendingToolCalls: null, readyToStream: false };
1197
+ }
1198
+
975
1199
  if (useStreamWithTools) {
1200
+ if (result.content) state.session._onStream?.(result.content);
976
1201
  emitAgentEvent(state.session, 'assistant_message', 'llm', { content: result.content ?? '' });
977
1202
  // Text was streamed inline via session._onStream — no second LLM call needed.
978
1203
  const newMessages = iterations === 0
@@ -1022,6 +1247,26 @@ export function createAgentGraph(options = {}) {
1022
1247
  const isInternalWikiTool = server === 'wiki' && (tool === 'plan_set' || tool === 'plan_done');
1023
1248
  const serverLabel = server === 'shell' ? 'Shell' : isInternalWikiTool ? 'Plan' : 'MCP';
1024
1249
  const toolName = server ? `${server}.${tool}` : call.function.name;
1250
+ // Hard guardrail: only execute tools that were actually offered this turn
1251
+ // (read-only tools + runtime controls + delegate). A capable model that
1252
+ // spots a mutating provider tool in the prompt and calls it directly must
1253
+ // be refused and steered back to runtime__delegate — this is what keeps
1254
+ // the orchestration capability-driven regardless of model strength.
1255
+ // Only interactive turns are constrained. Inside a runtime run
1256
+ // (_currentRunIdentity set) the graph legitimately executes the
1257
+ // already-validated, already-approved delegated task via provider tools.
1258
+ const runtimeExecutionTurn = Boolean(state.session._currentRunIdentity);
1259
+ const allowedNames = !runtimeExecutionTurn && Array.isArray(state.allowedToolNames) ? state.allowedToolNames : null;
1260
+ const isInternalCall = server === 'shell' || server === 'runtime' || isInternalWikiTool;
1261
+ if (allowedNames && server && !isInternalCall && !allowedNames.includes(`${server}__${tool}`)) {
1262
+ const refusal = `${server}__${tool} is not available in interactive mode. Do not call provider tools directly. For any action or mutation, call runtime__delegate with the user objective; only read-only tools and runtime controls may be called directly.`;
1263
+ state.session._onStep?.(`tool call refused (not offered): ${server}__${tool}`);
1264
+ emitAgentEvent(state.session, 'tool_call_result', 'tool', {
1265
+ callId: call.id, name: toolName, ok: false, result: refusal, summary: 'refused',
1266
+ });
1267
+ toolResultMessages.push({ role: 'tool', tool_call_id: call.id, content: refusal });
1268
+ continue;
1269
+ }
1025
1270
  if (resolved.normalized) {
1026
1271
  // Keep normalizations visible: the defensive routing must not hide
1027
1272
  // prompt/skill regressions that reintroduce unqualified names.
@@ -1038,9 +1283,11 @@ export function createAgentGraph(options = {}) {
1038
1283
  args: call.function.arguments ?? '{}',
1039
1284
  summary: argsSummary || 'calling...',
1040
1285
  });
1041
- // Immediate visible plan for any MCP call that doesn't yet have an _activity plan.
1286
+ // A plan represents work, never observation. Read-only inventory/status
1287
+ // calls stay out of Plan even when Donna uses them to answer a question.
1042
1288
  let minimalPlanActive = false;
1043
- if (!isInternalWikiTool && server !== 'shell' && !state.session.headlessPlan) {
1289
+ if (!isInternalWikiTool && server !== 'shell' && server !== 'runtime'
1290
+ && !isReadOnlyMcpCall(state.session, server, tool) && !state.session.headlessPlan) {
1044
1291
  minimalPlanActive = true;
1045
1292
  emitAgentEvent(state.session, 'plan_set', 'tool', {
1046
1293
  steps: [{ step: 1, id: null, description: toolName, status: 'running', _activityKey: null }],
@@ -1062,6 +1309,8 @@ export function createAgentGraph(options = {}) {
1062
1309
  );
1063
1310
  }
1064
1311
  let args = JSON.parse(call.function.arguments ?? '{}');
1312
+ const definition = toolDefinitionForCall(state.session, call.function.name);
1313
+ args = normalizeToolArgumentsFromSchema(args, definition?.function?.parameters);
1065
1314
  if (server === 'production' && tool === 'production_start_job' && state.session.workspace && !args.callerLabel) {
1066
1315
  args = { ...args, callerLabel: `${state.session.workspace}/wiki-manager` };
1067
1316
  }
@@ -1107,26 +1356,7 @@ export function createAgentGraph(options = {}) {
1107
1356
  // wiki__plan_set would lose fields (a small local model dropped
1108
1357
  // arguments/operations in testing) — the shell does the mapping.
1109
1358
  if (/(^|__)agent_plan$/.test(tool) && Array.isArray(payload?.tasks) && payload.tasks.length > 0) {
1110
- const steps = payload.tasks.map((task, index) => normalizeDeclaredPlanStep({
1111
- id: task.id,
1112
- description: task.label ?? task.id ?? `Task ${index + 1}`,
1113
- requiredCapability: task.requiredCapability ?? payload.capability ?? null,
1114
- operation: task.operation ?? null,
1115
- arguments: task.arguments ?? {},
1116
- dependsOn: task.dependsOn ?? [],
1117
- outputRefs: (task.expectedOutputRefs ?? []).map((ref) => (ref && typeof ref === 'object' ? String(ref.ref ?? '') : String(ref))).filter(Boolean),
1118
- groupId: task.groupId ?? null,
1119
- dependsOnGroup: task.dependsOnGroup ?? null,
1120
- parallelizable: task.parallelizable,
1121
- barrier: task.barrier,
1122
- locks: task.locks,
1123
- requiresApproval: task.requiresApproval,
1124
- approvalClass: task.approvalClass,
1125
- approvalSummary: task.approvalSummary,
1126
- idempotencyKey: task.idempotencyKey,
1127
- progressWeight: task.progressWeight,
1128
- recommendedConcurrency: task.recommendedConcurrency,
1129
- }, index, state.session));
1359
+ const steps = planStepsFromFragment(payload);
1130
1360
  emitAgentEvent(state.session, 'plan_set', 'tool', { steps });
1131
1361
  state.session._onStep?.(`Plan: ${steps.length} task(s) declared from ${server} fragment`);
1132
1362
  resultText = `Task-graph fragment integrated as the current plan (${steps.length} task(s), groups: ${[...new Set(payload.tasks.map((task) => task.groupId).filter(Boolean))].join(', ') || 'none'}). The orchestrator will dispatch these tasks — do NOT call production tools for them yourself. Reply with a short summary and wait.`;
@@ -1197,12 +1427,17 @@ export function createAgentGraph(options = {}) {
1197
1427
  return {
1198
1428
  messages: toolResultMessages,
1199
1429
  pendingToolCalls: null,
1430
+ forceDelegation: false,
1431
+ invalidToolCallRetries: 0,
1432
+ invalidResponseRetries: 0,
1200
1433
  };
1201
1434
  }
1202
1435
 
1203
1436
  function routeOrchestrator(state) {
1204
1437
  if (state.pendingToolCalls?.length > 0) return 'tool_executor';
1205
1438
  if (state.retryWithoutTool) return 'orchestrator';
1439
+ if (state.invalidToolCallRetries > 0 && state.response == null && !state.streamedInline) return 'orchestrator';
1440
+ if (state.invalidResponseRetries > 0 && state.response == null && !state.streamedInline) return 'orchestrator';
1206
1441
  return END;
1207
1442
  }
1208
1443