@dotdrelle/wiki-manager 0.14.0 → 0.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -25,5 +25,12 @@
25
25
  "Authorization": "Bearer ${DOCUMENTS_MCP_AUTH_TOKEN}"
26
26
  }
27
27
  }
28
+ },
29
+ "chatAccess": {
30
+ "maxToolIterations": 6,
31
+ "servers": {
32
+ "production": { "allow": ["production_job_status", "production_jobs_list"] },
33
+ "cme": { "allow": ["cme_status", "cme_sources_list", "cme_export_status"] }
34
+ }
28
35
  }
29
36
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dotdrelle/wiki-manager",
3
- "version": "0.14.0",
3
+ "version": "0.14.1",
4
4
  "description": "Agentic shell and orchestration cockpit for llm-wiki workspaces.",
5
5
  "license": "PolyForm-Noncommercial-1.0.0",
6
6
  "author": "dotrelle",
@@ -11,7 +11,7 @@
11
11
  },
12
12
  "scripts": {
13
13
  "start": "bun ./bin/wiki-manager.js",
14
- "test": "node --test src/agent/graph.test.js src/contracts/schemas.test.js src/core/activity.test.js src/core/buildInfo.test.js src/core/agentEvents.test.js src/core/runtimeLog.test.js src/activity/activityAggregator.test.js src/graph/runGraphProjector.test.js src/core/workflow.test.js src/core/planPatch.test.js src/core/agentLoop.test.js src/core/plan.test.js src/core/mcp.test.js src/core/documentIntake.test.js src/core/dockerCompose.test.js src/core/wikiWorkspace.test.js src/core/wikirc.test.js src/core/modelFetch.test.js src/core/startupCheck.test.js src/core/queueStore.test.js src/orchestrator/agentRegistry.test.js src/orchestrator/capabilityRegistry.test.js src/orchestrator/capabilityResolver.test.js src/orchestrator/planValidator.test.js src/orchestrator/planIntegrator.test.js src/orchestrator/scheduler.test.js src/orchestrator/attemptManager.test.js src/orchestrator/resultAggregator.test.js src/orchestrator/approvalPolicy.test.js src/commands/slash.test.js src/shell/repl.test.js src/runtime/store.test.js src/runtime/controlMessages.test.js src/runtime/recoveryManager.test.js src/runtime/server.test.js src/runtime/supervisor.test.js src/runtime/runner.test.js src/runtime/runner.e2e.test.js src/runtime/donna-contract.test.js src/runtime/auth.test.js",
14
+ "test": "node --test src/agent/graph.test.js src/contracts/schemas.test.js src/core/activity.test.js src/core/buildInfo.test.js src/core/agentEvents.test.js src/core/runtimeLog.test.js src/activity/activityAggregator.test.js src/graph/runGraphProjector.test.js src/core/workflow.test.js src/core/planPatch.test.js src/core/agentLoop.test.js src/core/plan.test.js src/core/mcp.test.js src/core/toolLoop.test.js src/core/documentIntake.test.js src/core/dockerCompose.test.js src/core/wikiWorkspace.test.js src/core/wikirc.test.js src/core/modelFetch.test.js src/core/startupCheck.test.js src/core/queueStore.test.js src/orchestrator/agentRegistry.test.js src/orchestrator/capabilityRegistry.test.js src/orchestrator/capabilityResolver.test.js src/orchestrator/planValidator.test.js src/orchestrator/planIntegrator.test.js src/orchestrator/scheduler.test.js src/orchestrator/attemptManager.test.js src/orchestrator/resultAggregator.test.js src/orchestrator/approvalPolicy.test.js src/commands/slash.test.js src/shell/repl.test.js src/runtime/store.test.js src/runtime/controlMessages.test.js src/runtime/recoveryManager.test.js src/runtime/server.test.js src/runtime/supervisor.test.js src/runtime/runner.test.js src/runtime/runner.e2e.test.js src/runtime/donna-contract.test.js src/runtime/auth.test.js",
15
15
  "check-versions": "node scripts/check-versions.js",
16
16
  "prepack": "node scripts/check-versions.js",
17
17
  "prepublishOnly": "node scripts/check-versions.js",
@@ -711,7 +711,7 @@ async function handleRuntimeControlTool(session, tool, args = {}) {
711
711
  if (!objective) return 'Delegation rejected: missing objective.';
712
712
  const result = await postRuntimeDelegate(objective, { url, workspace });
713
713
  return result?.runId
714
- ? `Délégation acceptée (${String(result.runId).slice(0, 8)}) après validation du plan réel : ${result.delegation?.tasks ?? 0} tâche(s), ${result.delegation?.agent ?? 'agent résolu'}. L'approbation porte sur ce plan.`
714
+ ? `Action lancée (${String(result.runId).slice(0, 8)}) après validation du plan réel : ${result.delegation?.tasks ?? 0} tâche(s), ${result.delegation?.agent ?? 'agent résolu'}. Exécution en cours.`
715
715
  : `Délégation refusée : ${result?.error ?? JSON.stringify(result)}`;
716
716
  }
717
717
  if (tool === 'enqueue') {
@@ -845,10 +845,7 @@ export function buildAgentSystemPrompt(state) {
845
845
  // mutating provider tools (e.g. production__production_start_job) here teaches
846
846
  // a capable model to invoke them directly and bypass runtime__delegate.
847
847
  const mcpTools = formatMcpToolsForAgent(state.session.mcp, {
848
- include: (qualifiedName, tool) => isDonnaReadTool({
849
- function: { name: qualifiedName },
850
- readOnly: tool?.annotations?.readOnlyHint === true,
851
- }),
848
+ include: (qualifiedName) => !isOrchestrationBypassTool(qualifiedName),
852
849
  });
853
850
  const skills = formatSkillsForAgent(state.session);
854
851
  const customPrompt = state.session.systemPrompt ?? null;
@@ -864,7 +861,7 @@ export function buildAgentSystemPrompt(state) {
864
861
  `Current wikirc profile: ${wikirc}.`,
865
862
  `Available primitives: ${commandList(state.session)}.`,
866
863
  'Only announce or call slash commands that appear exactly in Available primitives. Do not invent command names, subcommands, or arguments.',
867
- 'Connected read-only MCP tools you may call directly to answer questions (server__tool naming convention). Every mutation or action — ingest, build, export, configure, send, write — goes through runtime__delegate, never a direct provider tool call:',
864
+ 'Connected MCP tools you may call directly (server__tool naming convention) — reads AND single-step actions like configuring or adding a connector source, converting a document, sending, or searching. Only the heavy multi-step operations (ingest, build, export, polish, pipeline) go through runtime__delegate to get their parallel plan. Everything listed below is directly callable:',
868
865
  mcpTools,
869
866
  'Current local MCP job queue:',
870
867
  formatQueue(state.session),
@@ -877,7 +874,7 @@ export function buildAgentSystemPrompt(state) {
877
874
  'Never add a "Next steps", "Prochaines étapes", "À suivre", options, or suggestions section unless the user explicitly asks what to do next. End after the requested result or the concrete error.',
878
875
  'When calling a tool, emit no preliminary narration. Call it directly; the PLAN and Activity panels show progress. After completion, keep the final response concise and proportional to the result.',
879
876
  'Keep every response synthetic and information-dense. Use only the lines needed, and never exceed roughly 15 to 20 short lines even for a detailed answer. Never expose internal reasoning, repeated checks, tool-selection commentary, or a chronological diary. Prioritize the result, essential facts, concrete errors, and actual outputs.',
880
- 'Configuration, connector, import, export, conversion, generation, and every other mutation are actions: delegate the objective to the runtime instead of calling an external tool directly.',
877
+ 'Only the heavy multi-step operations — ingest, build, export, polish, pipeline — are delegated via runtime__delegate (for their DAG and parallelism). Single-step actions — configuring or adding a connector source, converting a document, sending, searching — are called directly on the connected tool. Never call an agent orchestration-contract or plan tool directly.',
881
878
  'For any question about the current workspace inventory or what is waiting there, call wiki__wiki_workspace_status first and answer only from its result. This is the canonical read-only workspace state; do not reconstruct it from upload, connector, or production tools.',
882
879
  'Tool identifiers are private implementation details. Never print MCP tool names such as server__tool in a user-facing answer. Describe the human result instead.',
883
880
  'Never suggest a manual filesystem command or implementation workaround unless the user explicitly asks for manual instructions. For an action request, delegate the objective and let the specialized agent determine paths and operations from its live contract.',
@@ -887,6 +884,7 @@ export function buildAgentSystemPrompt(state) {
887
884
  state.session.runtime?.url
888
885
  ? 'The runtime is connected and runtime__delegate is bound and available to you right now — it is a tool you call directly, not a slash command or a missing primitive. It is the ONLY way to execute an action (ingest, build, export, configure, send…). Never tell the user that delegation or the runtime is unavailable while it is connected; call runtime__delegate instead.'
889
886
  : 'No runtime is connected, so you cannot execute actions. State that plainly and name the runtime connection as the missing capability — do not invent a workaround.',
887
+ 'If the connector or service needed for a requested read or action is absent from the Connected MCP tools above (its service is not running — e.g. CME, documents, or production), say plainly that this service is not connected and name it as the missing capability. Never redirect a simple read (e.g. "give me the CME config") to an "agent action", never invent its result, and never propose a workaround. Only requests you can actually serve with a listed tool are answered with data.',
890
888
  'For any requested action, call runtime__delegate with the user objective only. Never choose a capability, operation, agent, plan, or implementation yourself. The runtime resolves the registry and validates the provider plan before accepting. Never call <provider>__agent_plan, <provider>__agent_execute, legacy production__production_start_job, wiki__plan_set, or wiki__plan_done from interactive chat.',
891
889
  'Do not ask the user which sources, files, connectors, or templates to use for an ingest, build, or export: the specialized agent discovers them from the workspace. When the objective is clear (e.g. "lance une ingestion"), delegate it as stated, without a clarifying question.',
892
890
  'If runtime__delegate returns a blocker or no specialized provider is available, report only that concrete blocker concisely. Never replace the missing execution path with a suggested slash command, skill, MCP tool name, manual file move, administrator escalation, or alternative workflow unless the user explicitly asks for alternatives.',
@@ -944,13 +942,20 @@ function toolsForClassification(classification, writeTools, session = null) {
944
942
  return [SHELL_READ_COMMAND_TOOL, ...controlTools];
945
943
  }
946
944
  if (session?.runtime?.url) {
947
- const readTools = writeTools.filter(isDonnaReadTool);
948
- return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...readTools];
945
+ // Offer every connected tool directly EXCEPT orchestration-bypass tools
946
+ // and raw shell write/profile mutation. Reads, configuration, connector
947
+ // setup — and any newly added MCP's tools — stay directly callable.
948
+ const directTools = writeTools.filter((item) => {
949
+ const name = item?.function?.name;
950
+ if (!name || name === 'shell__run_command' || name === 'shell__profile_update') return false;
951
+ return !isOrchestrationBypassTool(name);
952
+ });
953
+ return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...directTools];
949
954
  }
950
955
  return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...writeTools];
951
956
  }
952
957
 
953
- function isDonnaReadTool(item) {
958
+ export function isDonnaReadTool(item) {
954
959
  const name = String(item?.function?.name ?? '');
955
960
  if (!name || name.startsWith('shell__') || name === 'wiki__plan_set' || name === 'wiki__plan_done') return false;
956
961
  if (item?.readOnly === true) return true;
@@ -961,6 +966,24 @@ function isDonnaReadTool(item) {
961
966
  || /(?:^|_)(?:status|list|search|read|get)$/.test(tool);
962
967
  }
963
968
 
969
+ // Two-tier tool policy. Donna may call any connected MCP tool directly
970
+ // (reads AND plain writes: cme_setup, connector setup, document conversion,
971
+ // send, search, and anything a newly added MCP exposes) EXCEPT the small set
972
+ // that must go through the runtime's orchestration: the universal five-tool
973
+ // contract executors (agent_plan/agent_execute), the legacy job starter, and
974
+ // direct plan mutation. Heavy multi-step work (ingest/build/export via the
975
+ // production agent) is delegated for its DAG/parallelism; plain single-step
976
+ // tools are called directly. This is a blocklist, not a whitelist, so adding a
977
+ // new MCP never silently disables its tools.
978
+ function isOrchestrationBypassTool(name) {
979
+ const full = String(name ?? '');
980
+ if (!full) return true;
981
+ if (full === 'wiki__plan_set' || full === 'wiki__plan_done') return true;
982
+ const sep = full.indexOf('__');
983
+ const tool = sep === -1 ? full : full.slice(sep + 2);
984
+ return tool === 'agent_plan' || tool === 'agent_execute' || tool === 'production_start_job';
985
+ }
986
+
964
987
  function isReadOnlyMcpCall(session, server, tool) {
965
988
  const descriptor = (session?.mcp?.[server]?.tools ?? [])
966
989
  .find((item) => String(item?.name ?? '') === tool || String(item?.name ?? '').endsWith(`__${tool}`));
@@ -847,6 +847,15 @@ async function runRuntime(argv, agent) {
847
847
  throw new Error(`Delegated plan integration failed: ${(integrated.errors ?? []).map((error) => error.message ?? error.code ?? String(error)).join('; ')}`);
848
848
  }
849
849
  emitRuntimeLog(session, `delegation: ${prepared.fragment.tasks.length} validated task(s) integrated from ${prepared.provider.serverName}.agent_plan (${prepared.capability}/${prepared.operation})`);
850
+ // Demandé = consenti: a directly-delegated run carries the user's
851
+ // explicit consent, so auto-approve its initial plan. Persisting a
852
+ // run-scope grant (via the approval manager) makes the scheduler's
853
+ // readyTasks approval check pass, so the tasks run without re-prompting.
854
+ // Replanned tasks are integrated later without a fresh grant.
855
+ if (context.approvalManager?.approve) {
856
+ context.approvalManager.approve({ scope: 'run', runId });
857
+ emitRuntimeLog(session, `approval: run ${runId} auto-approved (user-requested action)`);
858
+ }
850
859
  body._planReady?.resolve?.({ runId, planRevision: session.agentProjection?.planRevision ?? 0 });
851
860
  }
852
861
  // Deterministic capability run (/ingest): ask the capable agent for its
@@ -196,7 +196,13 @@ export function applyAgentProjectionToSession(session, projection) {
196
196
  function applyEvent(state, event) {
197
197
  switch (event.type) {
198
198
  case 'run_started':
199
- state.status = 'running';
199
+ // Only a real runtime run (origin 'runtime') marks the projection
200
+ // 'running'. An interactive turn (origin 'user') is NOT a run: forcing
201
+ // 'running' here made the graph classify activeRun=true and hide Donna's
202
+ // MCP read tools (cme_status, wiki_workspace_status…), so questions about
203
+ // MCP state failed. Preserve the existing status (idle, or a genuinely
204
+ // active runtime run synced from the runtime) for interactive turns.
205
+ if (event.origin === 'runtime') state.status = 'running';
200
206
  state.plan = null;
201
207
  state.chain = [];
202
208
  state.activities = {};
@@ -20,13 +20,25 @@ test('reduceAgentEvents: run_started clears stale plan', () => {
20
20
  origin: 'tool',
21
21
  payload: { steps: ['Old action'] },
22
22
  }),
23
- createAgentEvent('run_started', { origin: 'user' }),
23
+ createAgentEvent('run_started', { origin: 'runtime' }),
24
24
  ]);
25
25
  assert.equal(projection.plan, null);
26
26
  assert.equal(projection.activities.length, 0);
27
27
  assert.equal(projection.status, 'running');
28
28
  });
29
29
 
30
+ test('reduceAgentEvents: interactive (user) run_started clears state but is not a running run', () => {
31
+ const projection = reduceAgentEvents([
32
+ createAgentEvent('plan_set', { origin: 'tool', payload: { steps: ['Old action'] } }),
33
+ createAgentEvent('run_started', { origin: 'user' }),
34
+ ]);
35
+ // An interactive turn clears stale plan/activities but must NOT mark the
36
+ // projection 'running' — otherwise the graph classifies activeRun=true and
37
+ // hides Donna's MCP read tools.
38
+ assert.equal(projection.plan, null);
39
+ assert.notEqual(projection.status, 'running');
40
+ });
41
+
30
42
  test('reduceAgentEvents: tracks manual plan and step updates', () => {
31
43
  const projection = reduceAgentEvents([
32
44
  createAgentEvent('plan_set', {
@@ -1,4 +1,4 @@
1
1
  {
2
- "version": "0.14.0",
3
- "commit": "fb0b915"
2
+ "version": "0.14.1",
3
+ "commit": "f6e95e4"
4
4
  }
package/src/core/mcp.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import { existsSync, readFileSync } from 'node:fs';
2
2
  import { managerEnvFile, managerMcpEndpointsFile, readEnvFile } from './env.js';
3
3
 
4
- const WIKI_MANAGER_VERSION = '0.14.0';
4
+ const WIKI_MANAGER_VERSION = '0.14.1';
5
5
 
6
6
  function envValue(key) {
7
7
  const filePath = managerEnvFile();
@@ -43,6 +43,29 @@ function normalizeExternalUrlForRuntime(url) {
43
43
  return url;
44
44
  }
45
45
 
46
+ // Config-driven policy for the /chat read-only toolset — NOT /agent, which has
47
+ // the full toolset and ignores this. The endpoints file's "chatAccess" block
48
+ // declares, per server, which tools /chat may call ("*" or a list), plus a
49
+ // maxToolIterations budget. Operator-owned, agnostic allow-list. Returns null
50
+ // when not configured — then /chat stays a plain, tool-less conversation.
51
+ export function readChatAccessConfig() {
52
+ const filePath = managerMcpEndpointsFile();
53
+ if (!existsSync(filePath)) return null;
54
+ let raw;
55
+ try { raw = JSON.parse(readFileSync(filePath, 'utf8')); } catch { return null; }
56
+ const chatAccess = raw?.chatAccess;
57
+ if (!chatAccess || typeof chatAccess !== 'object' || Array.isArray(chatAccess)) return null;
58
+ const servers = {};
59
+ for (const [name, entry] of Object.entries(chatAccess.servers ?? {})) {
60
+ if (entry?.allow === '*') servers[name] = { allow: '*' };
61
+ else if (Array.isArray(entry?.allow)) servers[name] = { allow: entry.allow.map(String).filter(Boolean) };
62
+ }
63
+ const maxToolIterations = Number.isFinite(Number(chatAccess.maxToolIterations)) && Number(chatAccess.maxToolIterations) > 0
64
+ ? Math.floor(Number(chatAccess.maxToolIterations))
65
+ : null;
66
+ return { maxToolIterations, servers };
67
+ }
68
+
46
69
  function readExternalMcpEndpoints() {
47
70
  const filePath = managerMcpEndpointsFile();
48
71
  if (!existsSync(filePath)) return {};
@@ -59,6 +82,14 @@ function readExternalMcpEndpoints() {
59
82
  url: normalizeExternalUrlForRuntime(interpolateEnv(String(endpoint.url))),
60
83
  configuredUrl: interpolateEnv(String(endpoint.url)),
61
84
  headers: normalizeHeaders(endpoint.headers),
85
+ // Tools the endpoint marks approval-gated: Donna may still call them
86
+ // directly (they are single-step tools), but toolRequiresApproval
87
+ // makes the call wait for the user's confirmation first (e.g. a
88
+ // destructive cme_export_run). Agent/operator owned — no hard-coded
89
+ // business name in the manager.
90
+ requireApproval: Array.isArray(endpoint.requireApproval)
91
+ ? endpoint.requireApproval.map(String).filter(Boolean)
92
+ : undefined,
62
93
  external: true,
63
94
  },
64
95
  ]),
@@ -95,6 +126,9 @@ const DEFAULT_MCP_RETRY_POLICY = {
95
126
  };
96
127
 
97
128
  export function buildMcpStatus(session) {
129
+ // Attach the /chat read-tool policy to the session alongside MCP status.
130
+ // Only /chat (repl.js) reads session.chatAccess; /agent ignores it.
131
+ if (session) session.chatAccess = readChatAccessConfig();
98
132
  const workspaceEnv = session.workspaceEnv ?? {};
99
133
  const wikiMcpToken = session.wikircConfig?.mcp?.accessKey;
100
134
  const wikiMcpDetail = workspaceEnv.WIKI_MCP_PORT
@@ -0,0 +1,56 @@
1
+ // Minimal, side-effect-free bounded tool-use loop.
2
+ //
3
+ // This is the shared mechanic of "ask the LLM with a tool set, run the tool
4
+ // calls it emits, feed results back, repeat up to a cap". The caller injects
5
+ // the ONLY policy that varies: `executeCall(call) -> string` decides whether a
6
+ // requested tool is allowed and produces its textual result (allow-list check,
7
+ // MCP dispatch, error formatting). The loop itself owns no plan, no delegation,
8
+ // no run identity and no agent events — deliberately unlike the /agent
9
+ // orchestration loop in createAgentGraph, which is a stateful LangGraph node
10
+ // graph and stays separate. Use this for stateless tool-answer turns (e.g.
11
+ // /chat read-only questions).
12
+ //
13
+ // `executeCall` may throw to abort the whole loop (e.g. an AbortError on
14
+ // cancel); anything it returns is treated as the tool result for that call.
15
+ export async function runBoundedToolLoop({
16
+ llm,
17
+ system,
18
+ messages,
19
+ tools,
20
+ executeCall,
21
+ maxIterations = 4,
22
+ signal,
23
+ onStep,
24
+ } = {}) {
25
+ const cap = Math.max(1, Math.floor(maxIterations) || 1);
26
+ const convo = [...(messages ?? [])];
27
+ for (let i = 0; i < cap; i += 1) {
28
+ onStep?.(i + 1, cap);
29
+ const result = await llm.completeWithTools({
30
+ system,
31
+ tools,
32
+ messages: convo,
33
+ toolChoice: 'auto',
34
+ signal,
35
+ });
36
+ const calls = result?.tool_calls ?? [];
37
+ if (calls.length === 0) {
38
+ return {
39
+ content: result?.content ?? result?.message?.content ?? '',
40
+ iterations: i + 1,
41
+ capped: false,
42
+ };
43
+ }
44
+ convo.push(result.message ?? { role: 'assistant', content: result.content ?? '', tool_calls: calls });
45
+ // Tool calls within one turn are independent: dispatch concurrently, then
46
+ // replay results in the model's call order so the transcript stays stable.
47
+ const outcomes = await Promise.all(calls.map(async (call) => ({
48
+ tool_call_id: call.id,
49
+ content: await executeCall(call),
50
+ })));
51
+ for (const outcome of outcomes) {
52
+ convo.push({ role: 'tool', tool_call_id: outcome.tool_call_id, content: outcome.content });
53
+ }
54
+ }
55
+ return { content: '', iterations: cap, capped: true };
56
+ }
@@ -0,0 +1,88 @@
1
+ import assert from 'node:assert/strict';
2
+ import test from 'node:test';
3
+ import { runBoundedToolLoop } from './toolLoop.js';
4
+
5
+ function toolCall(id, name, args = '{}') {
6
+ return { id, function: { name, arguments: args } };
7
+ }
8
+
9
+ test('returns the model answer directly when no tool is called', async () => {
10
+ const llm = {
11
+ async completeWithTools() {
12
+ return { content: 'plain answer', tool_calls: [] };
13
+ },
14
+ };
15
+ const out = await runBoundedToolLoop({ llm, tools: [], executeCall: async () => 'unused' });
16
+ assert.deepEqual(out, { content: 'plain answer', iterations: 1, capped: false });
17
+ });
18
+
19
+ test('dispatches a tool call, feeds the result back, then returns the final answer', async () => {
20
+ let round = 0;
21
+ const seen = [];
22
+ const llm = {
23
+ async completeWithTools({ messages }) {
24
+ round += 1;
25
+ if (round === 1) return { message: { role: 'assistant', content: '', tool_calls: [toolCall('c1', 'cme__cme_status')] }, tool_calls: [toolCall('c1', 'cme__cme_status')] };
26
+ seen.push(messages.find((m) => m.role === 'tool')?.content);
27
+ return { content: 'configured', tool_calls: [] };
28
+ },
29
+ };
30
+ const out = await runBoundedToolLoop({
31
+ llm,
32
+ tools: [{ function: { name: 'cme__cme_status' } }],
33
+ executeCall: async (call) => `RESULT(${call.function.name})`,
34
+ });
35
+ assert.equal(out.content, 'configured');
36
+ assert.equal(out.iterations, 2);
37
+ assert.equal(out.capped, false);
38
+ assert.deepEqual(seen, ['RESULT(cme__cme_status)']);
39
+ });
40
+
41
+ test('runs concurrent tool calls and replays results in call order', async () => {
42
+ let round = 0;
43
+ const order = [];
44
+ const llm = {
45
+ async completeWithTools({ messages }) {
46
+ round += 1;
47
+ if (round === 1) {
48
+ const calls = [toolCall('a', 's__list'), toolCall('b', 's__status')];
49
+ return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
50
+ }
51
+ order.push(...messages.filter((m) => m.role === 'tool').map((m) => m.tool_call_id));
52
+ return { content: 'done', tool_calls: [] };
53
+ },
54
+ };
55
+ const out = await runBoundedToolLoop({
56
+ llm,
57
+ tools: [],
58
+ executeCall: async (call) => call.id,
59
+ });
60
+ assert.equal(out.content, 'done');
61
+ assert.deepEqual(order, ['a', 'b']); // preserved model call order
62
+ });
63
+
64
+ test('reports capped when the model keeps calling tools past the cap', async () => {
65
+ const llm = {
66
+ async completeWithTools() {
67
+ const calls = [toolCall('x', 's__status')];
68
+ return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
69
+ },
70
+ };
71
+ const out = await runBoundedToolLoop({ llm, tools: [], executeCall: async () => 'r', maxIterations: 3 });
72
+ assert.equal(out.capped, true);
73
+ assert.equal(out.iterations, 3);
74
+ });
75
+
76
+ test('propagates an abort thrown by executeCall', async () => {
77
+ const llm = {
78
+ async completeWithTools() {
79
+ const calls = [toolCall('x', 's__status')];
80
+ return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
81
+ },
82
+ };
83
+ const abort = Object.assign(new Error('aborted'), { name: 'AbortError' });
84
+ await assert.rejects(
85
+ runBoundedToolLoop({ llm, tools: [], executeCall: async () => { throw abort; } }),
86
+ /aborted/,
87
+ );
88
+ });
@@ -1,5 +1,6 @@
1
1
  import { execFile, spawn } from 'node:child_process';
2
- import { dirname, resolve } from 'node:path';
2
+ import { readdirSync, statSync } from 'node:fs';
3
+ import { dirname, join, resolve } from 'node:path';
3
4
  import { fileURLToPath } from 'node:url';
4
5
  import { activeCacertPath } from '../core/cacert.js';
5
6
  import { checkRuntimeHealth, postRuntimeShutdown, runtimeUrlFromEnv } from './client.js';
@@ -10,6 +11,26 @@ const __dirname = dirname(fileURLToPath(import.meta.url));
10
11
  const managerRoot = resolve(__dirname, '../..');
11
12
  const binPath = resolve(managerRoot, 'bin/wiki-manager.js');
12
13
 
14
+ // Newest mtime (ms) of the manager's own source tree. Used to detect that the
15
+ // code was edited after a reused runtime started, so ensureRuntime can restart
16
+ // it instead of serving stale code. Returns 0 if the source tree is unreadable
17
+ // (e.g. running from a packed install) — in that case staleness is not checked.
18
+ function newestManagerSourceMtimeMs() {
19
+ const srcDir = join(managerRoot, 'src');
20
+ let newest = 0;
21
+ try {
22
+ for (const entry of readdirSync(srcDir, { recursive: true })) {
23
+ const name = String(entry);
24
+ if (!(name.endsWith('.js') || name.endsWith('.ts') || name.endsWith('.tsx'))) continue;
25
+ try {
26
+ const mtime = statSync(join(srcDir, name)).mtimeMs;
27
+ if (mtime > newest) newest = mtime;
28
+ } catch { /* file vanished mid-scan */ }
29
+ }
30
+ } catch { return 0; }
31
+ return newest;
32
+ }
33
+
13
34
  export function runtimeNodeExecutable() {
14
35
  return process.versions.bun
15
36
  ? (process.env.WIKI_MANAGER_NODE_BIN ?? 'node')
@@ -47,13 +68,22 @@ export async function ensureRuntime({
47
68
  if (existing) {
48
69
  const expectedCacertPath = activeCacertPath();
49
70
  const actualCacertPath = existing.cacertPath ? resolve(existing.cacertPath) : null;
71
+ // Dev staleness: if the manager source was edited after this runtime
72
+ // started, the reused process would keep serving old code (the recurring
73
+ // "my change is not taking effect" trap). Treat it as stale and restart.
74
+ // Packed installs report mtime 0 (unreadable src) → never flagged stale.
75
+ // Opt out with WIKI_MANAGER_RUNTIME_NO_STALE_CHECK=1.
76
+ const startedAtMs = Number(existing.startedAtMs) || 0;
77
+ const sourceMtimeMs = process.env.WIKI_MANAGER_RUNTIME_NO_STALE_CHECK === '1' ? 0 : newestManagerSourceMtimeMs();
78
+ const stale = startedAtMs > 0 && sourceMtimeMs > startedAtMs;
50
79
  // forceRestart: the caller knows the manager configuration just changed
51
80
  // (e.g. mcp.endpoints.json scaffolded on first run) — a runtime started
52
81
  // BEFORE that only knows the old endpoints and would keep answering
53
82
  // without the agents until manually restarted.
54
- if (!forceRestart && actualCacertPath === expectedCacertPath) {
83
+ if (!forceRestart && !stale && actualCacertPath === expectedCacertPath) {
55
84
  return { url, started: false, health: existing, token: auth.token, tokenPath: auth.tokenPath };
56
85
  }
86
+ if (stale) console.error('runtime: source changed since start — restarting for fresh code.');
57
87
  await postRuntimeShutdown({ url, token: auth.token });
58
88
  await waitForRuntimeShutdown(url, auth.token, 2500);
59
89
  }
@@ -25,6 +25,9 @@ export function startRuntimeServer({
25
25
  exitOnShutdown = process.env.WIKI_MANAGER_RUNTIME_CHILD === '1',
26
26
  } = {}) {
27
27
  const clients = new Set();
28
+ // When this runtime process started — used by ensureRuntime to detect that
29
+ // the manager source has been edited since (dev staleness) and auto-restart.
30
+ const runtimeStartedAtMs = Date.now();
28
31
  const defaultContext = { workspace: null, session, running: false, currentAbortController: null, currentRunId: null };
29
32
  const resolvedGetContext = getContext ?? (() => defaultContext);
30
33
 
@@ -62,6 +65,7 @@ export function startRuntimeServer({
62
65
  status: context?.running ? 'running' : 'idle',
63
66
  workspace: context?.workspace ?? workspace ?? null,
64
67
  activeRuns,
68
+ startedAtMs: runtimeStartedAtMs,
65
69
  dbPath: store.dbPath,
66
70
  cacertPath: activeCacertPath(),
67
71
  nodeExtraCaCerts: process.env.NODE_EXTRA_CA_CERTS ?? null,
package/src/shell/repl.js CHANGED
@@ -5,12 +5,13 @@ import { execFileSync } from 'node:child_process';
5
5
  import { stdin as input, stdout as output } from 'node:process';
6
6
  import { marked } from 'marked';
7
7
  import { markedTerminal } from 'marked-terminal';
8
- import { buildAgentSystemPrompt, formatLlmUnavailableMessage } from '../agent/graph.js';
8
+ import { buildAgentSystemPrompt, formatLlmUnavailableMessage, isDonnaReadTool } from '../agent/graph.js';
9
9
  import { handleSlashCommand } from '../commands/slash.js';
10
10
  import { serviceDescription, serviceNames as composeServiceNames } from '../core/compose.js';
11
11
  import { extractActivity, parseJsonText, sessionActivities } from '../core/activity.js';
12
12
  import { syncActivitiesToPlan } from '../core/plan.js';
13
- import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
13
+ import { buildLlmTools, callMcpTool, formatMcpToolResult, parseToolCallName, resolveToolCallName } from '../core/mcp.js';
14
+ import { runBoundedToolLoop } from '../core/toolLoop.js';
14
15
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
15
16
  import { listSkills } from '../core/skills.js';
16
17
  import { listWikircProfiles } from '../core/wikirc.js';
@@ -285,15 +286,37 @@ function isDonnaRole(role) {
285
286
  return role === 'donna' || role === LEGACY_DONNA_ROLE;
286
287
  }
287
288
 
289
+ // Read-only MCP tools exposed to /chat. The operator declares which tools per
290
+ // server in the `chatAccess` config (mcp.endpoints.json). /chat is extended to
291
+ // those tools ONLY, filtered through the same read-only test Donna's /agent
292
+ // mode uses (isDonnaReadTool), and only when their server is connected — so
293
+ // /chat can answer live state questions ("le CME est-il configuré ?") but can
294
+ // never mutate or delegate. Actions still belong to /agent, which has all tools.
295
+ export function chatReadTools(session) {
296
+ const servers = session?.chatAccess?.servers;
297
+ if (!servers) return [];
298
+ const scopedMcp = Object.fromEntries(
299
+ Object.entries(session.mcp ?? {}).filter(([name]) => Object.hasOwn(servers, name)),
300
+ );
301
+ return buildLlmTools(scopedMcp).filter((item) => {
302
+ const { server, tool } = parseToolCallName(item.function.name);
303
+ const entry = servers[server];
304
+ const declared = entry.allow === '*' ? true : (Array.isArray(entry.allow) && entry.allow.includes(tool));
305
+ return declared && isDonnaReadTool(item);
306
+ });
307
+ }
308
+
288
309
  function buildDirectChatSystemPrompt(session) {
289
310
  const workspace = session.workspace ?? 'no workspace selected';
290
311
  const wikirc = session.wikirc?.profile ?? 'no profile loaded';
291
312
  const language = session.language ?? 'en-US';
292
313
  return [
293
314
  'You are Donna, the llm-wiki-manager chat assistant.',
294
- 'Answer directly and concisely. Do not claim to have called tools or changed files.',
315
+ 'You have a small READ-ONLY toolset — the tools provided to you for this turn, which may be none. Use them to answer questions about live state (e.g. "le CME est-il configuré", "quelles pages sont en attente"), and answer only from their results.',
316
+ 'If no provided tool covers the request — or the request is an action or mutation (ingest, build, export, configure, send, delete…), or needs a service that is not connected — say plainly you cannot do it in chat mode and to switch to agent mode (/agent). Do not pretend to execute it and never guess.',
317
+ 'Answer directly and concisely. Do not claim to have called tools or changed files beyond the tools actually provided.',
295
318
  'Never add a "Next steps", "Prochaines étapes", "À suivre", options, or suggestions section unless the user explicitly asks what to do next. End after answering the question.',
296
- 'If the user asks for an action that needs workspace commands, MCP tools, services, files, or mutations, say to ask as an agent action instead of pretending to execute it.',
319
+ 'Never invent a tool name, command, job id, status, or result (e.g. do not fabricate names like "check_cme_configuration"). If you cannot know something with the tools you were given, say so.',
297
320
  `Reply language: ${language}.`,
298
321
  `Current workspace: ${workspace}.`,
299
322
  `Current wikirc profile: ${wikirc}.`,
@@ -1164,6 +1187,49 @@ async function runAgentTurn(input, { agent, session, onUpdate, onStep, displayIn
1164
1187
  return {};
1165
1188
  }
1166
1189
 
1190
+ // Bounded read-only tool loop for /chat. Only the tools declared in chatAccess
1191
+ // (and read-only) are offered; every call goes through callMcpTool, and any
1192
+ // tool the model names outside the offered set is refused — /chat can never
1193
+ // mutate or delegate. maxToolIterations caps the loop.
1194
+ async function runChatReadToolLoop({ input, session, history, donnaMessage, onUpdate, onStep, readTools }) {
1195
+ const allowed = new Set(readTools.map((item) => item.function.name));
1196
+ // /chat only supplies the policy; the loop mechanic lives in runBoundedToolLoop.
1197
+ // executeCall enforces the allow-list and turns each call into a text result;
1198
+ // it never mutates and refuses anything outside the offered read set.
1199
+ const executeCall = async (call) => {
1200
+ const rawName = call.function?.name ?? '';
1201
+ const { server, tool } = resolveToolCallName(session.mcp, rawName);
1202
+ const qualified = server ? `${server}__${tool}` : null;
1203
+ if (!qualified || !allowed.has(qualified)) {
1204
+ return `Refused: "${rawName}" is not an available read-only tool in chat mode. Actions and other tools live in agent mode (/agent).`;
1205
+ }
1206
+ let args = {};
1207
+ try { args = call.function?.arguments ? JSON.parse(call.function.arguments) : {}; } catch { args = {}; }
1208
+ try {
1209
+ onStep?.(`Chat: read ${server} ${tool}…`);
1210
+ const res = await callMcpTool(session.mcp, server, tool, args, session._abortSignal);
1211
+ return formatMcpToolResult(res);
1212
+ } catch (err) {
1213
+ if (err.name === 'AbortError' && session._abortSignal?.aborted) throw err;
1214
+ return `Error [${qualified}]: ${err instanceof Error ? err.message : String(err)}`;
1215
+ }
1216
+ };
1217
+ const { content, capped } = await runBoundedToolLoop({
1218
+ llm: session.llm,
1219
+ system: buildDirectChatSystemPrompt(session),
1220
+ messages: [...history, { role: 'user', content: input }],
1221
+ tools: readTools,
1222
+ executeCall,
1223
+ maxIterations: Math.min(8, Number(session?.chatAccess?.maxToolIterations) || 4),
1224
+ signal: session._abortSignal,
1225
+ onStep: (i, cap) => onStep?.(`Chat: consulting… [${i}/${cap}]`),
1226
+ });
1227
+ donnaMessage.content = capped
1228
+ ? 'Je n’ai pas pu conclure dans la limite d’itérations du mode chat. Passe en /agent si besoin.'
1229
+ : (stripDsmlArtifacts(content).trimEnd() || formatLlmUnavailableMessage('reponse vide'));
1230
+ onUpdate?.();
1231
+ }
1232
+
1167
1233
  async function runDirectChatTurn(input, { session, onUpdate, onStep }) {
1168
1234
  if (!session.llm?.stream) {
1169
1235
  conversationMessages(session).push({ role: 'command', content: directChatUnavailableText(session) });
@@ -1176,24 +1242,30 @@ async function runDirectChatTurn(input, { session, onUpdate, onStep }) {
1176
1242
  const donnaMessage = { role: 'donna', content: '' };
1177
1243
  messages.push(donnaMessage);
1178
1244
  onUpdate?.();
1245
+ const readTools = chatReadTools(session);
1246
+ const canUseReadTools = readTools.length > 0 && typeof session.llm.completeWithTools === 'function';
1179
1247
  try {
1180
- onStep?.('Chat: streaming direct answer…');
1181
- for await (const delta of session.llm.stream({
1182
- system: buildDirectChatSystemPrompt(session),
1183
- messages: [...history, { role: 'user', content: input }],
1184
- signal: session._abortSignal,
1185
- })) {
1186
- const cleanDelta = stripDsmlArtifacts(delta);
1187
- if (cleanDelta) {
1188
- donnaMessage.content += cleanDelta;
1248
+ if (canUseReadTools) {
1249
+ await runChatReadToolLoop({ input, session, history, donnaMessage, onUpdate, onStep, readTools });
1250
+ } else {
1251
+ onStep?.('Chat: streaming direct answer…');
1252
+ for await (const delta of session.llm.stream({
1253
+ system: buildDirectChatSystemPrompt(session),
1254
+ messages: [...history, { role: 'user', content: input }],
1255
+ signal: session._abortSignal,
1256
+ })) {
1257
+ const cleanDelta = stripDsmlArtifacts(delta);
1258
+ if (cleanDelta) {
1259
+ donnaMessage.content += cleanDelta;
1260
+ onUpdate?.();
1261
+ }
1262
+ }
1263
+ donnaMessage.content = stripDsmlArtifacts(donnaMessage.content).trimEnd();
1264
+ if (!donnaMessage.content.trim()) {
1265
+ donnaMessage.content = formatLlmUnavailableMessage('flux vide');
1189
1266
  onUpdate?.();
1190
1267
  }
1191
1268
  }
1192
- donnaMessage.content = stripDsmlArtifacts(donnaMessage.content).trimEnd();
1193
- if (!donnaMessage.content.trim()) {
1194
- donnaMessage.content = formatLlmUnavailableMessage('flux vide');
1195
- onUpdate?.();
1196
- }
1197
1269
  } catch (err) {
1198
1270
  if (err.name === 'AbortError') {
1199
1271
  messages.pop();
@@ -5,6 +5,7 @@ import { tmpdir } from 'node:os';
5
5
  import { join } from 'node:path';
6
6
  import {
7
7
  applyRuntimeStateToShellSession,
8
+ chatReadTools,
8
9
  createSession,
9
10
  conversationMessages,
10
11
  recordRuntimeUnavailableAgentInput,
@@ -373,3 +374,72 @@ test('submitRuntimeRun sends a control message instead of /run while a run is ac
373
374
  restore();
374
375
  }
375
376
  });
377
+
378
+ test('chatReadTools exposes only declared, read-only MCP tools to /chat', () => {
379
+ const session = {
380
+ chatAccess: {
381
+ servers: {
382
+ cme: { allow: ['cme_status', 'cme_sources_list', 'cme_export_run'] },
383
+ },
384
+ },
385
+ mcp: {
386
+ cme: {
387
+ status: 'connected',
388
+ tools: [
389
+ { name: 'cme_status', inputSchema: { type: 'object', properties: {} } },
390
+ { name: 'cme_sources_list', inputSchema: { type: 'object', properties: {} } },
391
+ { name: 'cme_setup', inputSchema: { type: 'object', properties: {} } },
392
+ { name: 'cme_export_run', inputSchema: { type: 'object', properties: {} } },
393
+ ],
394
+ },
395
+ documents: {
396
+ status: 'connected',
397
+ tools: [{ name: 'documents_status', inputSchema: { type: 'object', properties: {} } }],
398
+ },
399
+ },
400
+ };
401
+ const names = chatReadTools(session).map((item) => item.function.name).sort();
402
+ // cme_setup: not declared. cme_export_run: declared but a write (excluded by
403
+ // the read-only guard). documents_status: server absent from chatAccess.
404
+ assert.deepEqual(names, ['cme__cme_sources_list', 'cme__cme_status']);
405
+ });
406
+
407
+ test('chatReadTools is empty when no chatAccess is configured', () => {
408
+ const session = { mcp: { cme: { status: 'connected', tools: [{ name: 'cme_status', inputSchema: {} }] } } };
409
+ assert.deepEqual(chatReadTools(session), []);
410
+ });
411
+
412
+ test('/chat uses the tool-capable path when read tools are declared', async () => {
413
+ const session = createSession();
414
+ session.chatMode = true;
415
+ session.chatAccess = { maxToolIterations: 4, servers: { cme: { allow: ['cme_status'] } } };
416
+ session.mcp = { cme: { status: 'connected', tools: [{ name: 'cme_status', inputSchema: { type: 'object', properties: {} } }] } };
417
+ let usedComplete = false;
418
+ session.llm = {
419
+ async *stream() { yield 'STREAM_FALLBACK'; },
420
+ async completeWithTools() {
421
+ usedComplete = true;
422
+ return { tool_calls: [], content: 'Réponse via outils.', message: { role: 'assistant', content: 'Réponse via outils.' } };
423
+ },
424
+ };
425
+ await runLine('le cme est-il configuré', { session, chatMode: true });
426
+ const last = conversationMessages(session).at(-1);
427
+ assert.ok(usedComplete, 'completeWithTools path was taken');
428
+ assert.match(last.content, /Réponse via outils/);
429
+ assert.doesNotMatch(last.content, /STREAM_FALLBACK/);
430
+ });
431
+
432
+ test('/chat falls back to the plain stream when no read tools are declared', async () => {
433
+ const session = createSession();
434
+ session.chatMode = true;
435
+ session.chatAccess = null;
436
+ session.mcp = {};
437
+ session.llm = {
438
+ async *stream() { yield 'PLAIN_STREAM'; },
439
+ async completeWithTools() { return { tool_calls: [], content: 'SHOULD_NOT_APPEAR' }; },
440
+ };
441
+ await runLine('bonjour', { session, chatMode: true });
442
+ const last = conversationMessages(session).at(-1);
443
+ assert.match(last.content, /PLAIN_STREAM/);
444
+ assert.doesNotMatch(last.content, /SHOULD_NOT_APPEAR/);
445
+ });