@dotdrelle/wiki-manager 0.12.12 → 0.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/docker-compose.yml +1 -1
  2. package/mcp.endpoints.example.json +7 -0
  3. package/package.json +2 -2
  4. package/src/agent/graph.js +354 -143
  5. package/src/agent/graph.test.js +516 -54
  6. package/src/agent/llm.js +5 -5
  7. package/src/cli/wiki-manager.js +234 -6
  8. package/src/cli/wiki-manager.test.js +28 -0
  9. package/src/commands/slash.js +32 -11
  10. package/src/commands/slash.test.js +9 -1
  11. package/src/core/agentEvents.js +7 -1
  12. package/src/core/agentEvents.test.js +13 -1
  13. package/src/core/buildInfo.json +2 -2
  14. package/src/core/mcp.js +46 -4
  15. package/src/core/skills.js +0 -28
  16. package/src/core/toolLoop.js +56 -0
  17. package/src/core/toolLoop.test.js +88 -0
  18. package/src/orchestrator/capabilityRegistry.js +14 -0
  19. package/src/orchestrator/capabilityRegistry.test.js +12 -1
  20. package/src/orchestrator/dependencyResolver.js +10 -1
  21. package/src/orchestrator/dispatcher.js +34 -3
  22. package/src/orchestrator/dispatcher.test.js +34 -0
  23. package/src/orchestrator/objectiveResolver.js +79 -0
  24. package/src/orchestrator/objectiveResolver.test.js +50 -0
  25. package/src/orchestrator/scheduler.test.js +25 -0
  26. package/src/runtime/client.js +32 -1
  27. package/src/runtime/lifecycle.js +32 -2
  28. package/src/runtime/recoveryManager.js +14 -7
  29. package/src/runtime/runner.js +112 -13
  30. package/src/runtime/runner.test.js +64 -1
  31. package/src/runtime/server.js +47 -3
  32. package/src/runtime/supervisor.js +4 -1
  33. package/src/runtime/supervisor.test.js +49 -0
  34. package/src/shell/repl.js +134 -55
  35. package/src/shell/repl.test.js +151 -12
  36. package/src/shell/useSession.ts +15 -3
package/src/shell/repl.js CHANGED
@@ -5,12 +5,13 @@ import { execFileSync } from 'node:child_process';
5
5
  import { stdin as input, stdout as output } from 'node:process';
6
6
  import { marked } from 'marked';
7
7
  import { markedTerminal } from 'marked-terminal';
8
- import { buildAgentSystemPrompt, classifyAgentInput, formatLlmUnavailableMessage } from '../agent/graph.js';
8
+ import { buildAgentSystemPrompt, formatLlmUnavailableMessage, isDonnaReadTool } from '../agent/graph.js';
9
9
  import { handleSlashCommand } from '../commands/slash.js';
10
10
  import { serviceDescription, serviceNames as composeServiceNames } from '../core/compose.js';
11
11
  import { extractActivity, parseJsonText, sessionActivities } from '../core/activity.js';
12
12
  import { syncActivitiesToPlan } from '../core/plan.js';
13
- import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
13
+ import { buildLlmTools, callMcpTool, formatMcpToolResult, parseToolCallName, resolveToolCallName } from '../core/mcp.js';
14
+ import { runBoundedToolLoop } from '../core/toolLoop.js';
14
15
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
15
16
  import { listSkills } from '../core/skills.js';
16
17
  import { listWikircProfiles } from '../core/wikirc.js';
@@ -285,14 +286,37 @@ function isDonnaRole(role) {
285
286
  return role === 'donna' || role === LEGACY_DONNA_ROLE;
286
287
  }
287
288
 
289
+ // Read-only MCP tools exposed to /chat. The operator declares which tools per
290
+ // server in the `chatAccess` config (mcp.endpoints.json). /chat is extended to
291
+ // those tools ONLY, filtered through the same read-only test Donna's /agent
292
+ // mode uses (isDonnaReadTool), and only when their server is connected — so
293
+ // /chat can answer live state questions ("le CME est-il configuré ?") but can
294
+ // never mutate or delegate. Actions still belong to /agent, which has all tools.
295
+ export function chatReadTools(session) {
296
+ const servers = session?.chatAccess?.servers;
297
+ if (!servers) return [];
298
+ const scopedMcp = Object.fromEntries(
299
+ Object.entries(session.mcp ?? {}).filter(([name]) => Object.hasOwn(servers, name)),
300
+ );
301
+ return buildLlmTools(scopedMcp).filter((item) => {
302
+ const { server, tool } = parseToolCallName(item.function.name);
303
+ const entry = servers[server];
304
+ const declared = entry.allow === '*' ? true : (Array.isArray(entry.allow) && entry.allow.includes(tool));
305
+ return declared && isDonnaReadTool(item);
306
+ });
307
+ }
308
+
288
309
  function buildDirectChatSystemPrompt(session) {
289
310
  const workspace = session.workspace ?? 'no workspace selected';
290
311
  const wikirc = session.wikirc?.profile ?? 'no profile loaded';
291
312
  const language = session.language ?? 'en-US';
292
313
  return [
293
314
  'You are Donna, the llm-wiki-manager chat assistant.',
294
- 'Answer directly and concisely. Do not claim to have called tools or changed files.',
295
- 'If the user asks for an action that needs workspace commands, MCP tools, services, files, or mutations, say to ask as an agent action instead of pretending to execute it.',
315
+ 'You have a small READ-ONLY toolset — the tools provided to you for this turn, which may be none. Use them to answer questions about live state (e.g. "le CME est-il configuré", "quelles pages sont en attente"), and answer only from their results.',
316
+ 'If no provided tool covers the request — or the request is an action or mutation (ingest, build, export, configure, send, delete…), or needs a service that is not connected — say plainly you cannot do it in chat mode and to switch to agent mode (/agent). Do not pretend to execute it and never guess.',
317
+ 'Answer directly and concisely. Do not claim to have called tools or changed files beyond the tools actually provided.',
318
+ 'Never add a "Next steps", "Prochaines étapes", "À suivre", options, or suggestions section unless the user explicitly asks what to do next. End after answering the question.',
319
+ 'Never invent a tool name, command, job id, status, or result (e.g. do not fabricate names like "check_cme_configuration"). If you cannot know something with the tools you were given, say so.',
296
320
  `Reply language: ${language}.`,
297
321
  `Current workspace: ${workspace}.`,
298
322
  `Current wikirc profile: ${wikirc}.`,
@@ -774,18 +798,21 @@ function rememberProductionActivity(session, payload) {
774
798
 
775
799
  export function applyRuntimeStateToShellSession(session, state) {
776
800
  if (!state || typeof state !== 'object') return false;
801
+ const displayState = sanitizeRuntimeStateForDisplay(state);
777
802
  session.agentProjection = {
778
- conversation: Array.isArray(state.conversation) ? state.conversation.map((message) => ({ ...message })) : [],
779
- chain: Array.isArray(state.chain) ? state.chain.map((step) => ({ ...step })) : [],
780
- plan: Array.isArray(state.plan) ? state.plan.map((step) => ({ ...step })) : null,
781
- activities: Array.isArray(state.activities) ? state.activities.map((activity) => ({ ...activity })) : [],
782
- logs: Array.isArray(state.logs) ? [...state.logs] : [],
783
- summary: state.summary ?? null,
784
- status: state.status ?? 'idle',
785
- planRevision: state.planRevision ?? 0,
786
- planPatches: Array.isArray(state.planPatches) ? state.planPatches.map((patch) => ({ ...patch })) : [],
803
+ conversation: Array.isArray(displayState.conversation) ? displayState.conversation.map((message) => ({ ...message })) : [],
804
+ chain: Array.isArray(displayState.chain) ? displayState.chain.map((step) => ({ ...step })) : [],
805
+ plan: Array.isArray(displayState.plan) && displayState.plan.length > 0
806
+ ? displayState.plan.map((step) => ({ ...step }))
807
+ : null,
808
+ activities: Array.isArray(displayState.activities) ? displayState.activities.map((activity) => ({ ...activity })) : [],
809
+ logs: Array.isArray(displayState.logs) ? [...displayState.logs] : [],
810
+ summary: displayState.summary ?? null,
811
+ status: displayState.status ?? 'idle',
812
+ planRevision: displayState.planRevision ?? 0,
813
+ planPatches: Array.isArray(displayState.planPatches) ? displayState.planPatches.map((patch) => ({ ...patch })) : [],
787
814
  };
788
- session.workflow = state.workflow && typeof state.workflow === 'object'
815
+ session.workflow = displayState.workflow && typeof displayState.workflow === 'object'
789
816
  ? {
790
817
  ...state.workflow,
791
818
  nodes: Array.isArray(state.workflow.nodes) ? state.workflow.nodes.map((node) => ({ ...node })) : [],
@@ -803,7 +830,7 @@ export function applyRuntimeStateToShellSession(session, state) {
803
830
  // Runtime queue items replace the local jobQueue wholesale on every sync.
804
831
  // Tag their origin so /queue cancel can refuse to fake-cancel them locally
805
832
  // (a local status flip would be silently reverted by the next SSE sync).
806
- if (Array.isArray(state.queue)) session.jobQueue = state.queue.map((item) => ({ ...item, origin: 'runtime' }));
833
+ if (Array.isArray(displayState.queue)) session.jobQueue = displayState.queue.map((item) => ({ ...item, origin: 'runtime' }));
807
834
  const production = session.agentProjection.activities.filter((activity) => activity.source === 'production').at(-1);
808
835
  if (production) {
809
836
  session.productionActivity = {
@@ -817,21 +844,29 @@ export function applyRuntimeStateToShellSession(session, state) {
817
844
  return true;
818
845
  }
819
846
 
820
- // Submits a prompt to the shared runtime. If the workspace is already busy
821
- // (HTTP 409 from POST /run), route the input through the runtime control lane
822
- // so status questions and plan-change proposals do not become future runs.
823
- // A plain question or small talk must never start a runtime run (nor be
824
- // enqueued as a future one): it only needs an answer. Route converse/observe
825
- // to the local agent EVEN during an active run — the chat is supposed to stay
826
- // available, and the graph already restricts tools to read-only in that case.
827
- // Actions/cancels/approvals still go to the runtime.
828
- export function shouldHandleFreeTextLocally(line, session, { llmAvailable = Boolean(session?.llm) } = {}) {
829
- const classification = classifyAgentInput(line, session);
830
- // cancel intents included: Donna interprets "supprime le job et la queue"
831
- // and calls runtime__kill / runtime__cancel herself — no hardcoded regex
832
- // deciding between soft and hard stop. The control lane remains the
833
- // deterministic fallback when the local LLM is down.
834
- if (!['converse', 'observe', 'cancel', 'approve', 'enqueue_run', 'ambiguous'].includes(classification.kind)) return { local: false, classification };
847
+ export function sanitizeRuntimeStateForDisplay(state) {
848
+ if (!state || typeof state !== 'object') return state;
849
+ const status = String(state.status ?? 'idle').toLowerCase();
850
+ const visible = ['running', 'queued', 'pending', 'waiting', 'pending_approval', 'error', 'failed']
851
+ .includes(status);
852
+ if (visible) return state;
853
+ return {
854
+ ...state,
855
+ conversation: [],
856
+ chain: [],
857
+ plan: [],
858
+ activities: [],
859
+ workflow: null,
860
+ logs: [],
861
+ summary: null,
862
+ planPatches: [],
863
+ };
864
+ }
865
+
866
+ // In agent mode, Donna receives every free-text turn and decides whether to
867
+ // answer or call an exposed tool. Slash commands remain the deterministic UI.
868
+ export function shouldHandleFreeTextLocally(_line, session, { llmAvailable = Boolean(session?.llm) } = {}) {
869
+ const classification = { kind: 'agent_turn', confidence: 1, reason: 'agent_mode_llm_decision' };
835
870
  if (!llmAvailable) return { local: false, classification, fallbackReason: 'local LLM unavailable' };
836
871
  return { local: true, classification };
837
872
  }
@@ -1049,17 +1084,12 @@ async function runAgentTurn(input, { agent, session, onUpdate, onStep, displayIn
1049
1084
  };
1050
1085
  session._onStreamReset = () => {
1051
1086
  if (!donnaMessage) return;
1052
- if (donnaMessage.content.trim()) {
1053
- // Intermediate streamed text before tool calls: keep it, add separator.
1054
- donnaMessage.content += '\n\n';
1055
- onUpdate?.();
1056
- } else {
1057
- // Still empty ("Thinking…"): remove it cleanly.
1058
- const index = messages.indexOf(donnaMessage);
1059
- if (index !== -1) messages.splice(index, 1);
1060
- donnaMessage = null;
1061
- onUpdate?.();
1062
- }
1087
+ // Text emitted before a tool call is provisional narration, not an answer.
1088
+ // Remove it; the post-tool result gets a fresh Donna bubble.
1089
+ const index = messages.indexOf(donnaMessage);
1090
+ if (index !== -1) messages.splice(index, 1);
1091
+ donnaMessage = null;
1092
+ onUpdate?.();
1063
1093
  };
1064
1094
 
1065
1095
  let agentResult;
@@ -1157,6 +1187,49 @@ async function runAgentTurn(input, { agent, session, onUpdate, onStep, displayIn
1157
1187
  return {};
1158
1188
  }
1159
1189
 
1190
+ // Bounded read-only tool loop for /chat. Only the tools declared in chatAccess
1191
+ // (and read-only) are offered; every call goes through callMcpTool, and any
1192
+ // tool the model names outside the offered set is refused — /chat can never
1193
+ // mutate or delegate. maxToolIterations caps the loop.
1194
+ async function runChatReadToolLoop({ input, session, history, donnaMessage, onUpdate, onStep, readTools }) {
1195
+ const allowed = new Set(readTools.map((item) => item.function.name));
1196
+ // /chat only supplies the policy; the loop mechanic lives in runBoundedToolLoop.
1197
+ // executeCall enforces the allow-list and turns each call into a text result;
1198
+ // it never mutates and refuses anything outside the offered read set.
1199
+ const executeCall = async (call) => {
1200
+ const rawName = call.function?.name ?? '';
1201
+ const { server, tool } = resolveToolCallName(session.mcp, rawName);
1202
+ const qualified = server ? `${server}__${tool}` : null;
1203
+ if (!qualified || !allowed.has(qualified)) {
1204
+ return `Refused: "${rawName}" is not an available read-only tool in chat mode. Actions and other tools live in agent mode (/agent).`;
1205
+ }
1206
+ let args = {};
1207
+ try { args = call.function?.arguments ? JSON.parse(call.function.arguments) : {}; } catch { args = {}; }
1208
+ try {
1209
+ onStep?.(`Chat: read ${server} ${tool}…`);
1210
+ const res = await callMcpTool(session.mcp, server, tool, args, session._abortSignal);
1211
+ return formatMcpToolResult(res);
1212
+ } catch (err) {
1213
+ if (err.name === 'AbortError' && session._abortSignal?.aborted) throw err;
1214
+ return `Error [${qualified}]: ${err instanceof Error ? err.message : String(err)}`;
1215
+ }
1216
+ };
1217
+ const { content, capped } = await runBoundedToolLoop({
1218
+ llm: session.llm,
1219
+ system: buildDirectChatSystemPrompt(session),
1220
+ messages: [...history, { role: 'user', content: input }],
1221
+ tools: readTools,
1222
+ executeCall,
1223
+ maxIterations: Math.min(8, Number(session?.chatAccess?.maxToolIterations) || 4),
1224
+ signal: session._abortSignal,
1225
+ onStep: (i, cap) => onStep?.(`Chat: consulting… [${i}/${cap}]`),
1226
+ });
1227
+ donnaMessage.content = capped
1228
+ ? 'Je n’ai pas pu conclure dans la limite d’itérations du mode chat. Passe en /agent si besoin.'
1229
+ : (stripDsmlArtifacts(content).trimEnd() || formatLlmUnavailableMessage('reponse vide'));
1230
+ onUpdate?.();
1231
+ }
1232
+
1160
1233
  async function runDirectChatTurn(input, { session, onUpdate, onStep }) {
1161
1234
  if (!session.llm?.stream) {
1162
1235
  conversationMessages(session).push({ role: 'command', content: directChatUnavailableText(session) });
@@ -1169,24 +1242,30 @@ async function runDirectChatTurn(input, { session, onUpdate, onStep }) {
1169
1242
  const donnaMessage = { role: 'donna', content: '' };
1170
1243
  messages.push(donnaMessage);
1171
1244
  onUpdate?.();
1245
+ const readTools = chatReadTools(session);
1246
+ const canUseReadTools = readTools.length > 0 && typeof session.llm.completeWithTools === 'function';
1172
1247
  try {
1173
- onStep?.('Chat: streaming direct answer…');
1174
- for await (const delta of session.llm.stream({
1175
- system: buildDirectChatSystemPrompt(session),
1176
- messages: [...history, { role: 'user', content: input }],
1177
- signal: session._abortSignal,
1178
- })) {
1179
- const cleanDelta = stripDsmlArtifacts(delta);
1180
- if (cleanDelta) {
1181
- donnaMessage.content += cleanDelta;
1248
+ if (canUseReadTools) {
1249
+ await runChatReadToolLoop({ input, session, history, donnaMessage, onUpdate, onStep, readTools });
1250
+ } else {
1251
+ onStep?.('Chat: streaming direct answer…');
1252
+ for await (const delta of session.llm.stream({
1253
+ system: buildDirectChatSystemPrompt(session),
1254
+ messages: [...history, { role: 'user', content: input }],
1255
+ signal: session._abortSignal,
1256
+ })) {
1257
+ const cleanDelta = stripDsmlArtifacts(delta);
1258
+ if (cleanDelta) {
1259
+ donnaMessage.content += cleanDelta;
1260
+ onUpdate?.();
1261
+ }
1262
+ }
1263
+ donnaMessage.content = stripDsmlArtifacts(donnaMessage.content).trimEnd();
1264
+ if (!donnaMessage.content.trim()) {
1265
+ donnaMessage.content = formatLlmUnavailableMessage('flux vide');
1182
1266
  onUpdate?.();
1183
1267
  }
1184
1268
  }
1185
- donnaMessage.content = stripDsmlArtifacts(donnaMessage.content).trimEnd();
1186
- if (!donnaMessage.content.trim()) {
1187
- donnaMessage.content = formatLlmUnavailableMessage('flux vide');
1188
- onUpdate?.();
1189
- }
1190
1269
  } catch (err) {
1191
1270
  if (err.name === 'AbortError') {
1192
1271
  messages.pop();
@@ -5,16 +5,45 @@ import { tmpdir } from 'node:os';
5
5
  import { join } from 'node:path';
6
6
  import {
7
7
  applyRuntimeStateToShellSession,
8
+ chatReadTools,
8
9
  createSession,
9
10
  conversationMessages,
10
11
  recordRuntimeUnavailableAgentInput,
11
12
  runLine,
13
+ sanitizeRuntimeStateForDisplay,
12
14
  runtimeStatusLine,
13
15
  runtimeUnavailableAgentMessage,
14
16
  shouldHandleFreeTextLocally,
15
17
  submitRuntimeRun,
16
18
  } from './repl.js';
17
19
 
20
+ test('runtime display preserves a failed plan and its diagnostic evidence', () => {
21
+ const state = {
22
+ status: 'error',
23
+ plan: [{ id: 'apply', status: 'failed' }],
24
+ activities: [{ id: 'job-1', status: 'failed', error: 'exitCode=1' }],
25
+ logs: ['run_error: ingest_apply exitCode=1'],
26
+ conversation: [{ role: 'assistant', content: 'Échec de l’ingestion.' }],
27
+ };
28
+
29
+ assert.equal(sanitizeRuntimeStateForDisplay(state), state);
30
+ });
31
+
32
+ test('runtime display still clears completed historical execution state', () => {
33
+ const display = sanitizeRuntimeStateForDisplay({
34
+ status: 'done',
35
+ plan: [{ id: 'old', status: 'done' }],
36
+ activities: [{ id: 'old-job', status: 'done' }],
37
+ logs: ['old log'],
38
+ conversation: [{ role: 'assistant', content: 'Old run.' }],
39
+ });
40
+
41
+ assert.deepEqual(display.plan, []);
42
+ assert.deepEqual(display.activities, []);
43
+ assert.deepEqual(display.logs, []);
44
+ assert.deepEqual(display.conversation, []);
45
+ });
46
+
18
47
  function stubFetch(handler) {
19
48
  const original = globalThis.fetch;
20
49
  globalThis.fetch = handler;
@@ -74,6 +103,50 @@ test('applyRuntimeStateToShellSession projects runtime state into shell session'
74
103
  assert.deepEqual(conversationMessages(session), []);
75
104
  });
76
105
 
106
+ test('applyRuntimeStateToShellSession clears terminal plan and activities when runtime is idle', () => {
107
+ const session = createSession();
108
+ session.headlessPlan = [{ step: 1, description: 'Old read', status: 'failed' }];
109
+ session.activities = { old: { key: 'old', status: 'failed', terminal: true } };
110
+
111
+ applyRuntimeStateToShellSession(session, {
112
+ status: 'idle',
113
+ conversation: [{ role: 'assistant', content: 'Old failed answer' }],
114
+ chain: [{ id: 'old-step' }],
115
+ plan: [{ step: 1, description: 'Old read', status: 'failed' }],
116
+ activities: [{ key: 'old', status: 'failed', terminal: true }],
117
+ workflow: { nodes: [{ id: 'task:old' }], relations: [] },
118
+ logs: ['Runtime evaluator rejected the old run'],
119
+ summary: 'Old run failed',
120
+ planPatches: [{ id: 'old-patch' }],
121
+ });
122
+
123
+ assert.equal(session.headlessPlan, null);
124
+ assert.deepEqual(session.activities, {});
125
+ assert.equal(session.workflow, null);
126
+ assert.deepEqual(session.agentProjection.logs, []);
127
+ assert.equal(session.agentProjection.summary, null);
128
+ assert.deepEqual(session.agentProjection.conversation, []);
129
+ assert.deepEqual(session.agentProjection.chain, []);
130
+ assert.deepEqual(session.agentProjection.planPatches, []);
131
+ });
132
+
133
+ test('direct chat system prompt forbids unsolicited next steps', async () => {
134
+ const session = createSession();
135
+ let systemPrompt = '';
136
+ session.llm = {
137
+ async *stream({ system }) {
138
+ systemPrompt = system;
139
+ yield 'Réponse concise.';
140
+ },
141
+ };
142
+
143
+ await runLine('bonjour', { agent: null, packageJson: { version: 'test' }, session, chatMode: true });
144
+
145
+ assert.match(systemPrompt, /Never add a "Next steps", "Prochaines étapes", "À suivre"/);
146
+ assert.match(systemPrompt, /unless the user explicitly asks what to do next/);
147
+ assert.equal(conversationMessages(session).at(-1).content, 'Réponse concise.');
148
+ });
149
+
77
150
  test('submitRuntimeRun reports acceptance without throwing', async () => {
78
151
  const restore = stubFetch(async (url) => {
79
152
  assert.equal(pathOf(url), '/run');
@@ -241,37 +314,34 @@ test('/queue cancel on a runtime workflow id points to run cancellation commands
241
314
  assert.match(conversationMessages(session).at(-1).content, /\/run kill/);
242
315
  });
243
316
 
244
- test('free text routing keeps questions local and sends actions to the runtime', () => {
317
+ test('agent mode sends every free-text turn to Donna', () => {
245
318
  const session = createSession();
246
319
  session.llm = { completeWithTools: () => {} };
247
320
 
248
- // The original incident: a config question must never start a run.
249
321
  const question = shouldHandleFreeTextLocally('donne moi la config du cme', session);
250
322
  assert.equal(question.local, true);
251
- assert.equal(question.classification.kind, 'observe');
323
+ assert.equal(question.classification.kind, 'agent_turn');
252
324
 
253
325
  const smallTalk = shouldHandleFreeTextLocally('bonjour', session);
254
326
  assert.equal(smallTalk.local, true);
255
327
 
256
328
  const action = shouldHandleFreeTextLocally('lance le pipeline complet', session);
257
- assert.equal(action.local, false);
258
- assert.equal(action.classification.kind, 'start_run');
329
+ assert.equal(action.local, true);
330
+ assert.equal(action.classification.kind, 'agent_turn');
331
+
332
+ const pending = shouldHandleFreeTextLocally('as ton des fichier en attente d ingestion', session);
333
+ assert.equal(pending.local, true);
334
+ assert.equal(pending.classification.kind, 'agent_turn');
259
335
  });
260
336
 
261
- test('free text routing keeps questions local even during an active run', () => {
262
- // The chat must stay available during a run: a status question or small
263
- // talk answered locally (read-only tools) — never enqueued as a future run.
337
+ test('Donna keeps receiving free text during an active run', () => {
264
338
  const session = createSession();
265
339
  session.llm = { completeWithTools: () => {} };
266
340
  session.agentProjection = { status: 'running', activities: [], conversation: [] };
267
341
  assert.equal(shouldHandleFreeTextLocally('où en est le run', session).local, true);
268
342
  assert.equal(shouldHandleFreeTextLocally('salut', session).local, true);
269
- // Cancel intents are handled by Donna locally (runtime__kill/cancel tools);
270
- // approvals stay on the deterministic control lane.
271
343
  assert.equal(shouldHandleFreeTextLocally('stop le job', session).local, true);
272
344
  assert.equal(shouldHandleFreeTextLocally('supprime le job et la queue', session).local, true);
273
- // Approvals and "later" requests too: Donna owns runtime__approve and
274
- // runtime__enqueue. Only plan modifications and new runs bypass her.
275
345
  assert.equal(shouldHandleFreeTextLocally('approuve le run', session).local, true);
276
346
  assert.equal(shouldHandleFreeTextLocally('fais le build plus tard', session).local, true);
277
347
 
@@ -304,3 +374,72 @@ test('submitRuntimeRun sends a control message instead of /run while a run is ac
304
374
  restore();
305
375
  }
306
376
  });
377
+
378
+ test('chatReadTools exposes only declared, read-only MCP tools to /chat', () => {
379
+ const session = {
380
+ chatAccess: {
381
+ servers: {
382
+ cme: { allow: ['cme_status', 'cme_sources_list', 'cme_export_run'] },
383
+ },
384
+ },
385
+ mcp: {
386
+ cme: {
387
+ status: 'connected',
388
+ tools: [
389
+ { name: 'cme_status', inputSchema: { type: 'object', properties: {} } },
390
+ { name: 'cme_sources_list', inputSchema: { type: 'object', properties: {} } },
391
+ { name: 'cme_setup', inputSchema: { type: 'object', properties: {} } },
392
+ { name: 'cme_export_run', inputSchema: { type: 'object', properties: {} } },
393
+ ],
394
+ },
395
+ documents: {
396
+ status: 'connected',
397
+ tools: [{ name: 'documents_status', inputSchema: { type: 'object', properties: {} } }],
398
+ },
399
+ },
400
+ };
401
+ const names = chatReadTools(session).map((item) => item.function.name).sort();
402
+ // cme_setup: not declared. cme_export_run: declared but a write (excluded by
403
+ // the read-only guard). documents_status: server absent from chatAccess.
404
+ assert.deepEqual(names, ['cme__cme_sources_list', 'cme__cme_status']);
405
+ });
406
+
407
+ test('chatReadTools is empty when no chatAccess is configured', () => {
408
+ const session = { mcp: { cme: { status: 'connected', tools: [{ name: 'cme_status', inputSchema: {} }] } } };
409
+ assert.deepEqual(chatReadTools(session), []);
410
+ });
411
+
412
+ test('/chat uses the tool-capable path when read tools are declared', async () => {
413
+ const session = createSession();
414
+ session.chatMode = true;
415
+ session.chatAccess = { maxToolIterations: 4, servers: { cme: { allow: ['cme_status'] } } };
416
+ session.mcp = { cme: { status: 'connected', tools: [{ name: 'cme_status', inputSchema: { type: 'object', properties: {} } }] } };
417
+ let usedComplete = false;
418
+ session.llm = {
419
+ async *stream() { yield 'STREAM_FALLBACK'; },
420
+ async completeWithTools() {
421
+ usedComplete = true;
422
+ return { tool_calls: [], content: 'Réponse via outils.', message: { role: 'assistant', content: 'Réponse via outils.' } };
423
+ },
424
+ };
425
+ await runLine('le cme est-il configuré', { session, chatMode: true });
426
+ const last = conversationMessages(session).at(-1);
427
+ assert.ok(usedComplete, 'completeWithTools path was taken');
428
+ assert.match(last.content, /Réponse via outils/);
429
+ assert.doesNotMatch(last.content, /STREAM_FALLBACK/);
430
+ });
431
+
432
+ test('/chat falls back to the plain stream when no read tools are declared', async () => {
433
+ const session = createSession();
434
+ session.chatMode = true;
435
+ session.chatAccess = null;
436
+ session.mcp = {};
437
+ session.llm = {
438
+ async *stream() { yield 'PLAIN_STREAM'; },
439
+ async completeWithTools() { return { tool_calls: [], content: 'SHOULD_NOT_APPEAR' }; },
440
+ };
441
+ await runLine('bonjour', { session, chatMode: true });
442
+ const last = conversationMessages(session).at(-1);
443
+ assert.match(last.content, /PLAIN_STREAM/);
444
+ assert.doesNotMatch(last.content, /SHOULD_NOT_APPEAR/);
445
+ });
@@ -17,6 +17,7 @@ import {
17
17
  conversationMessages,
18
18
  createSession,
19
19
  runtimeUnavailableAgentMessage,
20
+ sanitizeRuntimeStateForDisplay,
20
21
  } from './repl.js';
21
22
  import { useAgent } from './useAgent';
22
23
 
@@ -340,10 +341,20 @@ export function useSession(props: { agent: unknown; packageJson: Record<string,
340
341
  function mergeRuntimeConversation(state: any) {
341
342
  const workspace = (session as any).workspace || '__global__';
342
343
  const runtimeConversation = Array.isArray(state?.conversation) ? state.conversation : [];
343
- if (runtimeConversation.length === 0) return;
344
344
  const target = conversationMessages(session);
345
345
  const backed = runtimeConversationRefsByWorkspace.get(workspace) ?? [];
346
346
  runtimeConversationRefsByWorkspace.set(workspace, backed);
347
+ if (runtimeConversation.length === 0) {
348
+ // Idle runtime display state deliberately has no historical run
349
+ // conversation. Remove only entries previously merged from that
350
+ // runtime; preserve local slash-command output and the current input.
351
+ for (const entry of backed) {
352
+ const index = target.indexOf(entry);
353
+ if (index !== -1) target.splice(index, 1);
354
+ }
355
+ backed.length = 0;
356
+ return;
357
+ }
347
358
  // The merge is index-aligned with the runtime conversation array. If that
348
359
  // array got SHORTER (runtime restart, projection reset), keeping stale
349
360
  // refs would make every new runtime entry silently overwrite an old
@@ -388,8 +399,9 @@ export function useSession(props: { agent: unknown; packageJson: Record<string,
388
399
  function syncRuntimeState() {
389
400
  void fetchRuntimeState({ url: props.runtime.url, workspace: (session as any).workspace ?? null })
390
401
  .then((state) => {
391
- setRuntimeState(state);
392
- mergeRuntimeConversation(state);
402
+ const displayState = sanitizeRuntimeStateForDisplay(state);
403
+ setRuntimeState(displayState);
404
+ mergeRuntimeConversation(displayState);
393
405
  setRuntimeStatus('connected');
394
406
  refresh();
395
407
  })