@dotdrelle/wiki-manager 0.12.12 → 0.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/docker-compose.yml +1 -1
  2. package/mcp.endpoints.example.json +7 -0
  3. package/package.json +2 -2
  4. package/src/agent/graph.js +354 -143
  5. package/src/agent/graph.test.js +516 -54
  6. package/src/agent/llm.js +5 -5
  7. package/src/cli/wiki-manager.js +234 -6
  8. package/src/cli/wiki-manager.test.js +28 -0
  9. package/src/commands/slash.js +32 -11
  10. package/src/commands/slash.test.js +9 -1
  11. package/src/core/agentEvents.js +7 -1
  12. package/src/core/agentEvents.test.js +13 -1
  13. package/src/core/buildInfo.json +2 -2
  14. package/src/core/mcp.js +46 -4
  15. package/src/core/skills.js +0 -28
  16. package/src/core/toolLoop.js +56 -0
  17. package/src/core/toolLoop.test.js +88 -0
  18. package/src/orchestrator/capabilityRegistry.js +14 -0
  19. package/src/orchestrator/capabilityRegistry.test.js +12 -1
  20. package/src/orchestrator/dependencyResolver.js +10 -1
  21. package/src/orchestrator/dispatcher.js +34 -3
  22. package/src/orchestrator/dispatcher.test.js +34 -0
  23. package/src/orchestrator/objectiveResolver.js +79 -0
  24. package/src/orchestrator/objectiveResolver.test.js +50 -0
  25. package/src/orchestrator/scheduler.test.js +25 -0
  26. package/src/runtime/client.js +32 -1
  27. package/src/runtime/lifecycle.js +32 -2
  28. package/src/runtime/recoveryManager.js +14 -7
  29. package/src/runtime/runner.js +112 -13
  30. package/src/runtime/runner.test.js +64 -1
  31. package/src/runtime/server.js +47 -3
  32. package/src/runtime/supervisor.js +4 -1
  33. package/src/runtime/supervisor.test.js +49 -0
  34. package/src/shell/repl.js +134 -55
  35. package/src/shell/repl.test.js +151 -12
  36. package/src/shell/useSession.ts +15 -3
package/src/agent/llm.js CHANGED
@@ -55,7 +55,7 @@ export function createLlmClientFromWikiConfig(config) {
55
55
  }
56
56
  return content;
57
57
  },
58
- async completeWithTools({ system, tools = [], messages = [], signal }) {
58
+ async completeWithTools({ system, tools = [], messages = [], toolChoice = 'auto', signal }) {
59
59
  const allMessages = [
60
60
  { role: 'system', content: system },
61
61
  ...messages,
@@ -67,7 +67,7 @@ export function createLlmClientFromWikiConfig(config) {
67
67
  };
68
68
  if (tools.length > 0) {
69
69
  body.tools = tools;
70
- body.tool_choice = 'auto';
70
+ body.tool_choice = toolChoice;
71
71
  }
72
72
  const response = await fetch(`${baseUrl}/chat/completions`, {
73
73
  method: 'POST',
@@ -90,7 +90,7 @@ export function createLlmClientFromWikiConfig(config) {
90
90
  message: { role: 'assistant', content: msg?.content ?? null, tool_calls: msg?.tool_calls },
91
91
  };
92
92
  },
93
- async streamWithTools({ system, tools = [], messages = [], onTextDelta, signal }) {
93
+ async streamWithTools({ system, tools = [], messages = [], toolChoice = 'auto', onTextDelta, signal }) {
94
94
  const allMessages = [
95
95
  { role: 'system', content: system },
96
96
  ...messages,
@@ -103,7 +103,7 @@ export function createLlmClientFromWikiConfig(config) {
103
103
  };
104
104
  if (tools.length > 0) {
105
105
  body.tools = tools;
106
- body.tool_choice = 'auto';
106
+ body.tool_choice = toolChoice;
107
107
  }
108
108
  const response = await fetch(`${baseUrl}/chat/completions`, {
109
109
  method: 'POST',
@@ -119,7 +119,7 @@ export function createLlmClientFromWikiConfig(config) {
119
119
  throw new Error(`HTTP ${response.status} ${text.slice(0, 240)}`);
120
120
  }
121
121
  if (!response.body) {
122
- const result = await this.completeWithTools({ system, tools, messages, signal });
122
+ const result = await this.completeWithTools({ system, tools, messages, toolChoice, signal });
123
123
  if (result.content) onTextDelta?.(result.content);
124
124
  return result;
125
125
  }
@@ -16,6 +16,7 @@ import { syncActivitiesToPlan, formatPlanStatus } from '../core/plan.js';
16
16
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
17
17
  import { runAgentTurn, runAgenticLoop } from '../core/agentLoop.js';
18
18
  import { resolveCapabilityConcurrency } from '../orchestrator/scheduler.js';
19
+ import { capabilityRegistryForSession } from '../orchestrator/capabilityRegistry.js';
19
20
  // Runtime modules use node:sqlite (Node.js built-in unavailable in Bun).
20
21
  // They are imported dynamically so the shell / TUI path never loads them.
21
22
 
@@ -56,6 +57,11 @@ function createSession() {
56
57
  };
57
58
  }
58
59
 
60
+ export async function forwardRuntimeApproval(getWorkspaceContext, request = {}) {
61
+ const context = await getWorkspaceContext(request.workspace ?? null);
62
+ return context.approvalManager?.approve(request) ?? { approved: false };
63
+ }
64
+
59
65
  function timestampForFile() {
60
66
  return new Date().toISOString().replace(/[:.]/g, '-');
61
67
  }
@@ -198,6 +204,109 @@ async function runHeadlessAgenticLoop(agent, session, initialInput, log, { timeo
198
204
  return { exitCode: result.ok ? 0 : (result.waitResult?.exitCode ?? 1) };
199
205
  }
200
206
 
207
+ // Observe a runtime-delegated run from headless: the run executes server-side,
208
+ // so poll /state and mirror status transitions, new logs and the final plan
209
+ // into the headless log until the run reaches a terminal state.
210
+ async function waitForRuntimeRun(session, log, { timeoutMs, pollMs = 1500, autoApprove = false, priorRunIds = [] } = {}) {
211
+ const { fetchRuntimeState, postRuntimeApprove } = await import('../runtime/client.js');
212
+ const url = session.runtime?.url;
213
+ const workspace = session.workspace ?? null;
214
+ if (!url) return { exitCode: 0 };
215
+ // Scope strictly to the run this turn created: any run already present before
216
+ // the turn (including a stuck/zombie run) must be ignored, or the wait would
217
+ // observe/approve the wrong run and never finish.
218
+ const priorSet = new Set((priorRunIds ?? []).map(String));
219
+ const terminal = new Set(['succeeded', 'success', 'done', 'complete', 'completed', 'failed', 'error', 'cancelled', 'canceled']);
220
+ const deadline = Date.now() + timeoutMs;
221
+ const graceDeadline = Date.now() + 8000;
222
+ let lastStatus = null;
223
+ let lastLogCount = 0;
224
+ let sawRun = false;
225
+ const approvedRevisions = new Set();
226
+ while (Date.now() < deadline) {
227
+ let state;
228
+ try {
229
+ state = await fetchRuntimeState({ url, workspace });
230
+ } catch (err) {
231
+ const line = `runtime-wait: state fetch failed (${err instanceof Error ? err.message : String(err)})`;
232
+ log.push(line); console.error(line);
233
+ return { exitCode: 1 };
234
+ }
235
+ const logs = Array.isArray(state?.logs) ? state.logs : [];
236
+ for (const entry of logs.slice(lastLogCount)) {
237
+ const text = typeof entry === 'string' ? entry : String(entry?.message ?? JSON.stringify(entry));
238
+ log.push(`runtime: ${text}`); console.log(`[runtime] ${text}`);
239
+ }
240
+ lastLogCount = logs.length;
241
+ const runs = Array.isArray(state?.runs) ? state.runs : [];
242
+ const currentRun = runs.find((run) => run?.id && !priorSet.has(String(run.id)));
243
+ if (!currentRun) {
244
+ if (sawRun) return { exitCode: 0 };
245
+ if (Date.now() >= graceDeadline) {
246
+ const line = 'runtime-wait: no run was delegated this turn (Donna answered without starting a run).';
247
+ log.push(line); console.log(line);
248
+ return { exitCode: 0 };
249
+ }
250
+ await new Promise((resolve) => setTimeout(resolve, pollMs));
251
+ continue;
252
+ }
253
+ sawRun = true;
254
+ const status = String(currentRun.status ?? 'running').toLowerCase();
255
+ if (status !== lastStatus) {
256
+ log.push(`runtime-status: ${status} (run ${currentRun.id})`); console.log(`[runtime] status=${status}`);
257
+ lastStatus = status;
258
+ }
259
+ // Approval is granted per task, so the run status stays "running" while a
260
+ // task waits — detect the block via state.approvals, scoped to this run.
261
+ const pendingApprovals = (Array.isArray(state?.approvals) ? state.approvals : [])
262
+ .filter((approval) => approval.status === 'pending_approval'
263
+ && (approval.runId == null || String(approval.runId) === String(currentRun.id)));
264
+ if (pendingApprovals.length > 0) {
265
+ if (!autoApprove) {
266
+ const line = `runtime-wait: run ${currentRun.id} waiting for approval (${pendingApprovals.length} task(s)); re-run with --auto-approve to drive it through.`;
267
+ log.push(line); console.log(line);
268
+ return { exitCode: 0 };
269
+ }
270
+ const planRevision = state?.planRevision ?? currentRun.planRevision ?? 0;
271
+ if (!approvedRevisions.has(planRevision)) {
272
+ approvedRevisions.add(planRevision);
273
+ const approvalClasses = [...new Set(pendingApprovals.flatMap((approval) => {
274
+ const value = approval.approvalClasses ?? approval.approvalClass ?? [];
275
+ return Array.isArray(value) ? value : [value];
276
+ }).map(String).filter(Boolean))];
277
+ try {
278
+ const result = await postRuntimeApprove({
279
+ url,
280
+ workspace,
281
+ runId: currentRun.id,
282
+ scope: 'run',
283
+ planRevision,
284
+ approvalClasses: approvalClasses.length > 0 ? approvalClasses : ['default'],
285
+ });
286
+ const line = `runtime-wait: auto-approved run ${currentRun.id} (revision ${planRevision})${result?.approved ? '' : ' [no pending approval matched]'}`;
287
+ log.push(line); console.log(line);
288
+ } catch (err) {
289
+ const line = `runtime-wait: auto-approve failed (${err instanceof Error ? err.message : String(err)})`;
290
+ log.push(line); console.error(line);
291
+ return { exitCode: 1 };
292
+ }
293
+ }
294
+ }
295
+ if (terminal.has(status)) {
296
+ const plan = Array.isArray(currentRun.plan) ? currentRun.plan
297
+ : (Array.isArray(state?.plan) ? state.plan : []);
298
+ if (plan.length > 0) {
299
+ log.push(`runtime-plan:\n${plan.map((planStep) => ` - ${planStep.description ?? planStep.id ?? planStep.step ?? ''}: ${planStep.status ?? ''}`).join('\n')}`);
300
+ }
301
+ return { exitCode: status === 'failed' || status === 'error' ? 1 : 0 };
302
+ }
303
+ await new Promise((resolve) => setTimeout(resolve, pollMs));
304
+ }
305
+ const line = 'runtime-wait: timeout waiting for the delegated run to finish.';
306
+ log.push(line); console.error(line);
307
+ return { exitCode: 1 };
308
+ }
309
+
201
310
  async function runHeadless(argv, agent) {
202
311
  const workspaceName = valueAfter(argv, '--workspace');
203
312
  const skillName = valueAfter(argv, '--skill');
@@ -229,6 +338,37 @@ async function runHeadless(argv, agent) {
229
338
  if (!session.workspacePath) throw new Error(useResult.output || `Workspace not loaded: ${workspaceName}`);
230
339
  if (!session.llm) throw new Error(`Workspace ${workspaceName} has no usable LLM config.`);
231
340
 
341
+ // Agent-mode parity: the interactive TUI runs every turn against the
342
+ // runtime (delegation + run control) with MCP connected. Without a
343
+ // runtime, the graph exposes no runtime__delegate tool, so any action
344
+ // request degrades to a chat-like text answer instead of a real delegated
345
+ // run — which is exactly why headless "looked like chat mode". Connect the
346
+ // same way the TUI does. Use --no-runtime for the legacy direct-MCP path.
347
+ if (!argv.includes('--no-runtime')) {
348
+ try {
349
+ const { ensureRuntime } = await import('../runtime/lifecycle.js');
350
+ const runtime = await ensureRuntime();
351
+ session.runtime = runtime?.url ? { url: runtime.url, started: Boolean(runtime.started) } : null;
352
+ step(`runtime: ${session.runtime ? `connected ${session.runtime.url}` : 'unavailable'}`);
353
+ } catch (err) {
354
+ session.runtime = null;
355
+ step(`runtime: unavailable (${err instanceof Error ? err.message : String(err)})`);
356
+ }
357
+ }
358
+ await refreshMcpRuntimeStatus(session);
359
+ step(`mcp: ${Object.values(session.mcp ?? {}).filter((value) => value.status === 'connected').length} connected`);
360
+ // Surface which tools Donna is actually offered this turn: if a factual
361
+ // question is answered without the matching read tool appearing here, the
362
+ // problem is discovery/connection, not the model.
363
+ for (const [name, value] of Object.entries(session.mcp ?? {})) {
364
+ if (value?.status !== 'connected') continue;
365
+ const toolNames = (value.tools ?? []).map((tool) => tool.name).join(', ');
366
+ step(`mcp-tools ${name}: ${toolNames || '(none discovered)'}`);
367
+ }
368
+ // Wire the graph's step trace (classification, tool calls, retries) into
369
+ // the headless log so the agent-mode decision is observable.
370
+ session._onStep = step;
371
+
232
372
  let input = prompt;
233
373
  if (skillName) {
234
374
  const skillResult = await handleSlashCommand(`/skills run ${skillName}`, { packageJson, session, onStep: step });
@@ -260,11 +400,26 @@ async function runHeadless(argv, agent) {
260
400
  if (useAgenticLoop) {
261
401
  ({ exitCode } = await runHeadlessAgenticLoop(agent, session, input, log, { timeoutMs, maxTurns }));
262
402
  } else {
403
+ // Snapshot existing runs so the wait scopes strictly to the run this turn
404
+ // creates and never observes a pre-existing / zombie run.
405
+ let priorRunIds = [];
406
+ if (session.runtime?.url && wait) {
407
+ try {
408
+ const { fetchRuntimeState } = await import('../runtime/client.js');
409
+ const before = await fetchRuntimeState({ url: session.runtime.url, workspace: session.workspace ?? null });
410
+ priorRunIds = (Array.isArray(before?.runs) ? before.runs : []).map((run) => run?.id).filter(Boolean);
411
+ } catch { priorRunIds = []; }
412
+ }
263
413
  const response = await runAgentTurn(agent, session, input);
264
414
  log.push('response:');
265
415
  log.push(response);
266
416
  console.log(response);
267
- ({ exitCode } = await runHeadlessActivityLoop(session, log, { wait, timeoutMs }));
417
+ // When connected to the runtime, an action turn delegates a run that
418
+ // executes server-side — its progress lives in runtime state, not in the
419
+ // local session. Poll it so the headless log shows the real outcome.
420
+ ({ exitCode } = session.runtime?.url && wait
421
+ ? await waitForRuntimeRun(session, log, { timeoutMs, autoApprove: argv.includes('--auto-approve'), priorRunIds })
422
+ : await runHeadlessActivityLoop(session, log, { wait, timeoutMs }));
268
423
  }
269
424
  const saved = await writeHeadlessLog(session, log, logFile);
270
425
  console.log(`Headless log: ${saved}`);
@@ -591,6 +746,55 @@ async function runRuntime(argv, agent) {
591
746
  };
592
747
  }
593
748
 
749
+ async function prepareDelegation(context, { objective }) {
750
+ const { resolveObjective } = await import('../orchestrator/objectiveResolver.js');
751
+ const { validateFragment } = await import('../orchestrator/planValidator.js');
752
+ const session = context.session;
753
+ const selection = await resolveObjective(objective, session);
754
+ const provider = selection.provider;
755
+ const fragment = parseJsonText(formatMcpToolResult(await callMcpTool(
756
+ session.mcp,
757
+ provider.serverName,
758
+ 'agent_plan',
759
+ {
760
+ capability: selection.capability,
761
+ operation: selection.operation,
762
+ objective,
763
+ workspace: { revision: String(Date.now()) },
764
+ constraints: {
765
+ maxConcurrency: resolveCapabilityConcurrency(
766
+ provider,
767
+ undefined,
768
+ process.env.WIKI_MANAGER_CAPABILITY_CONCURRENCY,
769
+ ),
770
+ requireApprovalForMutations: true,
771
+ },
772
+ },
773
+ )));
774
+ if (!Array.isArray(fragment?.tasks) || fragment.tasks.length === 0) {
775
+ throw new Error(fragment?.summary?.initialSynthesis?.[0] ?? `No task was planned for ${selection.capability}/${selection.operation}.`);
776
+ }
777
+ const validation = validateFragment(fragment, {
778
+ registry: capabilityRegistryForSession(session),
779
+ run: { plannerAgentInstanceId: provider.agentInstanceId ?? provider.serverName },
780
+ });
781
+ if (!validation.ok) {
782
+ throw new Error(`Delegated plan rejected: ${validation.errors.map((error) => error.message ?? error.code ?? String(error)).join('; ')}`);
783
+ }
784
+ return {
785
+ capability: selection.capability,
786
+ operation: selection.operation,
787
+ provider: { serverName: provider.serverName, agentInstanceId: provider.agentInstanceId ?? provider.serverName },
788
+ fragment: validation.normalizedFragment,
789
+ summary: {
790
+ capability: selection.capability,
791
+ operation: selection.operation,
792
+ agent: provider.agentInstanceId ?? provider.serverName,
793
+ tasks: validation.normalizedFragment.tasks.length,
794
+ },
795
+ };
796
+ }
797
+
594
798
  async function executeRun(context, body, { signal } = {}) {
595
799
  const session = context.session;
596
800
  const supervisor = context.supervisor;
@@ -629,6 +833,31 @@ async function runRuntime(argv, agent) {
629
833
  : undefined;
630
834
  supervisor?.setRunSignal(signal);
631
835
  session._onStep = (message) => emitRuntimeLog(session, message);
836
+ if (body.preparedDelegation?.fragment) {
837
+ const { integrate } = await import('../orchestrator/planIntegrator.js');
838
+ const prepared = body.preparedDelegation;
839
+ const integrated = integrate(runId, prepared.fragment, {
840
+ registry: capabilityRegistryForSession(session),
841
+ session,
842
+ store,
843
+ workspace: session.workspace ?? null,
844
+ enforceApprovalCoverage: true,
845
+ });
846
+ if (!integrated.ok) {
847
+ throw new Error(`Delegated plan integration failed: ${(integrated.errors ?? []).map((error) => error.message ?? error.code ?? String(error)).join('; ')}`);
848
+ }
849
+ emitRuntimeLog(session, `delegation: ${prepared.fragment.tasks.length} validated task(s) integrated from ${prepared.provider.serverName}.agent_plan (${prepared.capability}/${prepared.operation})`);
850
+ // Demandé = consenti: a directly-delegated run carries the user's
851
+ // explicit consent, so auto-approve its initial plan. Persisting a
852
+ // run-scope grant (via the approval manager) makes the scheduler's
853
+ // readyTasks approval check pass, so the tasks run without re-prompting.
854
+ // Replanned tasks are integrated later without a fresh grant.
855
+ if (context.approvalManager?.approve) {
856
+ context.approvalManager.approve({ scope: 'run', runId });
857
+ emitRuntimeLog(session, `approval: run ${runId} auto-approved (user-requested action)`);
858
+ }
859
+ body._planReady?.resolve?.({ runId, planRevision: session.agentProjection?.planRevision ?? 0 });
860
+ }
632
861
  // Deterministic capability run (/ingest): ask the capable agent for its
633
862
  // task-graph fragment and integrate it as the plan BEFORE any LLM turn.
634
863
  // The parallel path must not depend on a small model deciding to call
@@ -636,7 +865,7 @@ async function runRuntime(argv, agent) {
636
865
  if (body.capabilityPlan?.capability) {
637
866
  const { validateFragment } = await import('../orchestrator/planValidator.js');
638
867
  const { integrate } = await import('../orchestrator/planIntegrator.js');
639
- const registry = session.capabilityRegistry ?? null;
868
+ const registry = capabilityRegistryForSession(session);
640
869
  const agents = session.agentRegistry?.snapshot?.() ?? session.agentRegistrySnapshot ?? [];
641
870
  const provider = agents.find((item) => (item.description?.capabilities ?? [])
642
871
  .some((capability) => capability.id === body.capabilityPlan.capability));
@@ -704,6 +933,7 @@ async function runRuntime(argv, agent) {
704
933
  ...(maxReplans === undefined ? {} : { maxReplans }),
705
934
  });
706
935
  } catch (err) {
936
+ body._planReady?.reject?.(err);
707
937
  if (err?.name === 'AbortError') {
708
938
  // Cancel the asynchronous agent jobs the run started: aborting only
709
939
  // the manager loop left ingest subprocesses running for minutes with
@@ -747,12 +977,10 @@ async function runRuntime(argv, agent) {
747
977
  .filter((context) => context?.running)
748
978
  .map((context) => ({ workspace: context.workspace ?? null, runId: context.currentRunId ?? null })),
749
979
  run: executeRun,
980
+ delegate: prepareDelegation,
750
981
  cancel: (context) => emitRuntimeLog(context.session, 'runtime: cancel requested'),
751
982
  resume: ({ workspace }) => recoverRuntime({ workspace, manual: true }),
752
- approve: async ({ workspace, runId, itemId, approvalId }) => {
753
- const context = await getWorkspaceContext(workspace);
754
- return context.approvalManager?.approve({ runId, itemId, approvalId }) ?? { approved: false };
755
- },
983
+ approve: (request) => forwardRuntimeApproval(getWorkspaceContext, request),
756
984
  configProfiles: async (context) => {
757
985
  const profiles = listWikircProfiles(context.session.workspacePath);
758
986
  return {
@@ -0,0 +1,28 @@
1
+ import assert from 'node:assert/strict';
2
+ import test from 'node:test';
3
+ import { forwardRuntimeApproval } from './wiki-manager.js';
4
+
5
+ test('runtime approval bridge preserves the complete run-scoped grant', async () => {
6
+ let forwarded = null;
7
+ const request = {
8
+ workspace: 'test4',
9
+ workspaceId: 'test4',
10
+ runId: 'run-1',
11
+ scope: 'run',
12
+ planRevision: 3,
13
+ approvalClasses: ['mutation'],
14
+ };
15
+
16
+ const result = await forwardRuntimeApproval(async (workspace) => ({
17
+ approvalManager: {
18
+ approve(value) {
19
+ assert.equal(workspace, 'test4');
20
+ forwarded = value;
21
+ return { approved: true };
22
+ },
23
+ },
24
+ }), request);
25
+
26
+ assert.deepEqual(forwarded, request);
27
+ assert.deepEqual(result, { approved: true });
28
+ });
@@ -676,6 +676,20 @@ function rawCommandResult(command, output) {
676
676
  };
677
677
  }
678
678
 
679
+ export function localizedOperationResult({ operation, target, status = 'succeeded' }) {
680
+ const facts = JSON.stringify({ operation, target, status });
681
+ return {
682
+ output: facts,
683
+ rawOutput: true,
684
+ agentTrigger: [
685
+ 'Formule le résultat structuré suivant dans la langue et le ton demandés par le profil du workspace.',
686
+ 'Réponds par une seule phrase humaine et naturelle.',
687
+ 'Ne mentionne aucune commande, syntaxe shell, étape suivante ou détail technique.',
688
+ `Résultat: ${facts}`,
689
+ ].join('\n'),
690
+ };
691
+ }
692
+
679
693
  function formatRuntimeRunStatus(state) {
680
694
  const status = state?.status ?? 'unknown';
681
695
  const runId = state?.runId ? ` run=${state.runId}` : '';
@@ -711,13 +725,14 @@ export async function handleSlashCommand(line, context) {
711
725
  const args = line.slice(1).trim().split(/\s+/).filter(Boolean);
712
726
  const [command] = args;
713
727
  const step = context.onStep ?? (() => {});
714
- const runAgentCommand = async (fn, verb, commandLabel) => {
728
+ const runAgentCommand = async (fn, verb) => {
715
729
  try {
716
730
  step(`Agents: ${verb}ing external agents…`);
717
- const output = await fn();
718
- return output
719
- ? rawCommandResult(commandLabel, output)
720
- : { output: `Agents ${verb}ed.` };
731
+ await fn();
732
+ return localizedOperationResult({
733
+ operation: verb,
734
+ target: 'agents',
735
+ });
721
736
  } catch (err) {
722
737
  step(formatActivityError('agents', verb, err));
723
738
  return { output: err instanceof Error ? err.message : String(err) };
@@ -876,13 +891,16 @@ export async function handleSlashCommand(line, context) {
876
891
  // do not remap it to undefined, that bypasses any custom "all" target list and always
877
892
  // falls back to the hardcoded COMPOSE_SERVICES constant instead.
878
893
  const service = args[1];
879
- if (service === 'agents') return runAgentCommand(startAgents, 'start', '/start agents');
894
+ if (service === 'agents') return runAgentCommand(startAgents, 'start');
880
895
  try {
881
896
  step(`Services: starting ${service ?? 'workspace services'}…`);
882
- const output = await startService(context.session, service);
897
+ await startService(context.session, service);
883
898
  step('Services: refreshing MCP runtime…');
884
899
  await refreshMcpRuntimeStatus(context.session);
885
- return rawCommandResult(`/start${service ? ` ${service}` : ''}`, output);
900
+ return localizedOperationResult({
901
+ operation: 'start',
902
+ target: service || 'workspace-services',
903
+ });
886
904
  } catch (err) {
887
905
  const message = err instanceof Error ? err.message : String(err);
888
906
  step(formatActivityError('services', 'stop', err));
@@ -891,13 +909,16 @@ export async function handleSlashCommand(line, context) {
891
909
  }
892
910
  case 'stop': {
893
911
  const service = args[1];
894
- if (service === 'agents') return runAgentCommand(stopAgents, 'stop', '/stop agents');
912
+ if (service === 'agents') return runAgentCommand(stopAgents, 'stop');
895
913
  try {
896
914
  step(`Services: stopping ${service ?? 'workspace services'}…`);
897
- const output = await stopService(context.session, service);
915
+ await stopService(context.session, service);
898
916
  step('Services: refreshing MCP runtime…');
899
917
  await refreshMcpRuntimeStatus(context.session);
900
- return rawCommandResult(`/stop${service ? ` ${service}` : ''}`, output);
918
+ return localizedOperationResult({
919
+ operation: 'stop',
920
+ target: service || 'workspace-services',
921
+ });
901
922
  } catch (err) {
902
923
  const message = err instanceof Error ? err.message : String(err);
903
924
  step(formatActivityError('services', 'logs', err));
@@ -4,9 +4,17 @@ import { mkdtemp } from 'node:fs/promises';
4
4
  import { tmpdir } from 'node:os';
5
5
  import { join } from 'node:path';
6
6
  import test from 'node:test';
7
- import { handleSlashCommand } from './slash.js';
7
+ import { handleSlashCommand, localizedOperationResult } from './slash.js';
8
8
  import { completionContext } from '../shell/repl.js';
9
9
 
10
+ test('deterministic operation results ask Donna to localize compact facts without leaking commands', () => {
11
+ const result = localizedOperationResult({ operation: 'start', target: 'agents' });
12
+ assert.equal(result.rawOutput, true);
13
+ assert.deepEqual(JSON.parse(result.output), { operation: 'start', target: 'agents', status: 'succeeded' });
14
+ assert.match(result.agentTrigger, /une seule phrase humaine et naturelle/);
15
+ assert.doesNotMatch(result.agentTrigger, /\/start|Docker|compose/);
16
+ });
17
+
10
18
  test('/workspace delete removes files and clears current session context after confirmation', async () => {
11
19
  const root = await mkdtemp(join(tmpdir(), 'wiki-manager-delete-workspace-'));
12
20
  const registryRoot = join(root, 'registry');
@@ -196,7 +196,13 @@ export function applyAgentProjectionToSession(session, projection) {
196
196
  function applyEvent(state, event) {
197
197
  switch (event.type) {
198
198
  case 'run_started':
199
- state.status = 'running';
199
+ // Only a real runtime run (origin 'runtime') marks the projection
200
+ // 'running'. An interactive turn (origin 'user') is NOT a run: forcing
201
+ // 'running' here made the graph classify activeRun=true and hide Donna's
202
+ // MCP read tools (cme_status, wiki_workspace_status…), so questions about
203
+ // MCP state failed. Preserve the existing status (idle, or a genuinely
204
+ // active runtime run synced from the runtime) for interactive turns.
205
+ if (event.origin === 'runtime') state.status = 'running';
200
206
  state.plan = null;
201
207
  state.chain = [];
202
208
  state.activities = {};
@@ -20,13 +20,25 @@ test('reduceAgentEvents: run_started clears stale plan', () => {
20
20
  origin: 'tool',
21
21
  payload: { steps: ['Old action'] },
22
22
  }),
23
- createAgentEvent('run_started', { origin: 'user' }),
23
+ createAgentEvent('run_started', { origin: 'runtime' }),
24
24
  ]);
25
25
  assert.equal(projection.plan, null);
26
26
  assert.equal(projection.activities.length, 0);
27
27
  assert.equal(projection.status, 'running');
28
28
  });
29
29
 
30
+ test('reduceAgentEvents: interactive (user) run_started clears state but is not a running run', () => {
31
+ const projection = reduceAgentEvents([
32
+ createAgentEvent('plan_set', { origin: 'tool', payload: { steps: ['Old action'] } }),
33
+ createAgentEvent('run_started', { origin: 'user' }),
34
+ ]);
35
+ // An interactive turn clears stale plan/activities but must NOT mark the
36
+ // projection 'running' — otherwise the graph classifies activeRun=true and
37
+ // hides Donna's MCP read tools.
38
+ assert.equal(projection.plan, null);
39
+ assert.notEqual(projection.status, 'running');
40
+ });
41
+
30
42
  test('reduceAgentEvents: tracks manual plan and step updates', () => {
31
43
  const projection = reduceAgentEvents([
32
44
  createAgentEvent('plan_set', {
@@ -1,4 +1,4 @@
1
1
  {
2
- "version": "0.12.12",
3
- "commit": "e1e43ae"
2
+ "version": "0.14.1",
3
+ "commit": "f6e95e4"
4
4
  }
package/src/core/mcp.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import { existsSync, readFileSync } from 'node:fs';
2
2
  import { managerEnvFile, managerMcpEndpointsFile, readEnvFile } from './env.js';
3
3
 
4
- const WIKI_MANAGER_VERSION = '0.12.12';
4
+ const WIKI_MANAGER_VERSION = '0.14.1';
5
5
 
6
6
  function envValue(key) {
7
7
  const filePath = managerEnvFile();
@@ -43,6 +43,29 @@ function normalizeExternalUrlForRuntime(url) {
43
43
  return url;
44
44
  }
45
45
 
46
+ // Config-driven policy for the /chat read-only toolset — NOT /agent, which has
47
+ // the full toolset and ignores this. The endpoints file's "chatAccess" block
48
+ // declares, per server, which tools /chat may call ("*" or a list), plus a
49
+ // maxToolIterations budget. Operator-owned, agnostic allow-list. Returns null
50
+ // when not configured — then /chat stays a plain, tool-less conversation.
51
+ export function readChatAccessConfig() {
52
+ const filePath = managerMcpEndpointsFile();
53
+ if (!existsSync(filePath)) return null;
54
+ let raw;
55
+ try { raw = JSON.parse(readFileSync(filePath, 'utf8')); } catch { return null; }
56
+ const chatAccess = raw?.chatAccess;
57
+ if (!chatAccess || typeof chatAccess !== 'object' || Array.isArray(chatAccess)) return null;
58
+ const servers = {};
59
+ for (const [name, entry] of Object.entries(chatAccess.servers ?? {})) {
60
+ if (entry?.allow === '*') servers[name] = { allow: '*' };
61
+ else if (Array.isArray(entry?.allow)) servers[name] = { allow: entry.allow.map(String).filter(Boolean) };
62
+ }
63
+ const maxToolIterations = Number.isFinite(Number(chatAccess.maxToolIterations)) && Number(chatAccess.maxToolIterations) > 0
64
+ ? Math.floor(Number(chatAccess.maxToolIterations))
65
+ : null;
66
+ return { maxToolIterations, servers };
67
+ }
68
+
46
69
  function readExternalMcpEndpoints() {
47
70
  const filePath = managerMcpEndpointsFile();
48
71
  if (!existsSync(filePath)) return {};
@@ -59,6 +82,14 @@ function readExternalMcpEndpoints() {
59
82
  url: normalizeExternalUrlForRuntime(interpolateEnv(String(endpoint.url))),
60
83
  configuredUrl: interpolateEnv(String(endpoint.url)),
61
84
  headers: normalizeHeaders(endpoint.headers),
85
+ // Tools the endpoint marks approval-gated: Donna may still call them
86
+ // directly (they are single-step tools), but toolRequiresApproval
87
+ // makes the call wait for the user's confirmation first (e.g. a
88
+ // destructive cme_export_run). Agent/operator owned — no hard-coded
89
+ // business name in the manager.
90
+ requireApproval: Array.isArray(endpoint.requireApproval)
91
+ ? endpoint.requireApproval.map(String).filter(Boolean)
92
+ : undefined,
62
93
  external: true,
63
94
  },
64
95
  ]),
@@ -95,6 +126,9 @@ const DEFAULT_MCP_RETRY_POLICY = {
95
126
  };
96
127
 
97
128
  export function buildMcpStatus(session) {
129
+ // Attach the /chat read-tool policy to the session alongside MCP status.
130
+ // Only /chat (repl.js) reads session.chatAccess; /agent ignores it.
131
+ if (session) session.chatAccess = readChatAccessConfig();
98
132
  const workspaceEnv = session.workspaceEnv ?? {};
99
133
  const wikiMcpToken = session.wikircConfig?.mcp?.accessKey;
100
134
  const wikiMcpDetail = workspaceEnv.WIKI_MCP_PORT
@@ -479,15 +513,22 @@ export function formatMcpToolSummary(mcpStatus) {
479
513
  return lines.length > 0 ? lines.join('\n') : 'No connected MCP tools discovered.';
480
514
  }
481
515
 
482
- export function formatMcpToolsForAgent(mcpStatus) {
516
+ export function formatMcpToolsForAgent(mcpStatus, { include } = {}) {
483
517
  const sections = [];
484
518
  for (const [name, value] of Object.entries(mcpStatus ?? {})) {
485
519
  if (value.status !== 'connected') continue;
486
- const tools = value.tools ?? [];
487
- if (tools.length === 0) {
520
+ const allTools = value.tools ?? [];
521
+ if (allTools.length === 0) {
488
522
  sections.push(`${name}: connected, tools not discovered yet`);
489
523
  continue;
490
524
  }
525
+ // Optional filter: callers (e.g. the interactive prompt) advertise only
526
+ // the tools Donna is actually allowed to call, so a capable model is not
527
+ // tempted to invoke a mutating provider tool directly instead of delegating.
528
+ const tools = typeof include === 'function'
529
+ ? allTools.filter((tool) => include(`${name}__${tool.name}`, tool, name))
530
+ : allTools;
531
+ if (tools.length === 0) continue;
491
532
  // Always advertise the qualified call name (server__tool): showing bare
492
533
  // tool names here is what teaches the model to emit unqualified calls.
493
534
  sections.push(`${name}: ${tools.map((tool) => `${name}__${tool.name}`).join(', ')}`);
@@ -502,6 +543,7 @@ export function buildLlmTools(mcpStatus) {
502
543
  for (const tool of value.tools ?? []) {
503
544
  tools.push({
504
545
  type: 'function',
546
+ readOnly: tool.annotations?.readOnlyHint === true,
505
547
  function: {
506
548
  name: `${serverName}__${tool.name}`,
507
549
  description: clarifyToolDescription(serverName, tool.name, tool.description),