@dotdrelle/wiki-manager 0.12.11 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/.env.example +6 -0
  2. package/docker-compose.yml +1 -1
  3. package/package.json +1 -1
  4. package/src/agent/graph.js +377 -142
  5. package/src/agent/graph.test.js +576 -34
  6. package/src/agent/llm.js +5 -5
  7. package/src/cli/wiki-manager.js +294 -9
  8. package/src/cli/wiki-manager.test.js +28 -0
  9. package/src/commands/slash.js +80 -13
  10. package/src/commands/slash.test.js +9 -1
  11. package/src/contracts/schemas.js +33 -0
  12. package/src/contracts/schemas.test.js +14 -0
  13. package/src/core/agentEvents.js +6 -0
  14. package/src/core/agentEvents.test.js +26 -0
  15. package/src/core/agentLoop.js +15 -16
  16. package/src/core/agentLoop.test.js +9 -7
  17. package/src/core/buildInfo.json +2 -2
  18. package/src/core/mcp.js +13 -6
  19. package/src/core/mcp.test.js +0 -12
  20. package/src/core/skills.js +0 -28
  21. package/src/orchestrator/capabilityRegistry.js +14 -0
  22. package/src/orchestrator/capabilityRegistry.test.js +12 -1
  23. package/src/orchestrator/dependencyResolver.js +10 -1
  24. package/src/orchestrator/dispatcher.js +34 -3
  25. package/src/orchestrator/dispatcher.test.js +34 -0
  26. package/src/orchestrator/objectiveResolver.js +79 -0
  27. package/src/orchestrator/objectiveResolver.test.js +50 -0
  28. package/src/orchestrator/scheduler.js +24 -0
  29. package/src/orchestrator/scheduler.test.js +65 -1
  30. package/src/runtime/client.js +34 -2
  31. package/src/runtime/lifecycle.js +1 -1
  32. package/src/runtime/recoveryManager.js +14 -7
  33. package/src/runtime/runner.js +214 -14
  34. package/src/runtime/runner.test.js +100 -2
  35. package/src/runtime/server.js +43 -3
  36. package/src/runtime/supervisor.js +65 -1
  37. package/src/runtime/supervisor.test.js +80 -0
  38. package/src/shell/LeftPane.tsx +9 -2
  39. package/src/shell/repl.js +57 -42
  40. package/src/shell/repl.test.js +81 -12
  41. package/src/shell/tui.tsx +26 -3
  42. package/src/shell/useSession.ts +15 -3
@@ -15,6 +15,7 @@ export function startRuntimeServer({
15
15
  session = null,
16
16
  getContext,
17
17
  run,
18
+ delegate,
18
19
  cancel,
19
20
  resume,
20
21
  approve,
@@ -271,6 +272,36 @@ export function startRuntimeServer({
271
272
  }
272
273
  return;
273
274
  }
275
+ if (request.method === 'POST' && url.pathname === '/delegate') {
276
+ const { body, context } = await resolveBodyContext(request, url);
277
+ const objective = String(body.objective ?? '').trim();
278
+ if (!objective) {
279
+ sendJson(response, 400, { error: 'Missing objective.' });
280
+ return;
281
+ }
282
+ if (context.running) {
283
+ sendJson(response, 409, { error: 'A runtime run is already active.' });
284
+ return;
285
+ }
286
+ if (typeof delegate !== 'function') {
287
+ sendJson(response, 501, { error: 'Runtime delegation is unavailable.' });
288
+ return;
289
+ }
290
+ try {
291
+ const prepared = await delegate(context, { objective, workspace: body.workspace ?? context.workspace ?? null });
292
+ const started = startRuntimeRun(context, {
293
+ input: objective,
294
+ workspace: body.workspace ?? context.workspace ?? null,
295
+ preparedDelegation: prepared,
296
+ evaluate: false,
297
+ }, { waitForPlan: true });
298
+ await started.ready;
299
+ sendJson(response, 202, { accepted: true, runId: started.runId, workspace: started.workspace, delegation: prepared.summary ?? null });
300
+ } catch (err) {
301
+ sendJson(response, 422, { error: err instanceof Error ? err.message : String(err) });
302
+ }
303
+ return;
304
+ }
274
305
  if (request.method === 'POST' && url.pathname === '/cancel') {
275
306
  const workspace = workspaceFromUrl(url);
276
307
  const context = await resolveContext({ workspace });
@@ -391,14 +422,22 @@ export function startRuntimeServer({
391
422
  return { killed: true, workspace: targetWorkspace, runId: targetRunId, runs, tasks, queued };
392
423
  }
393
424
 
394
- function startRuntimeRun(context, body, { controlItemId = null } = {}) {
425
+ function startRuntimeRun(context, body, { controlItemId = null, waitForPlan = false } = {}) {
395
426
  const runId = randomUUID();
396
427
  const runWorkspace = context.workspace ?? body.workspace ?? null;
397
428
  context.running = true;
398
429
  context.currentAbortController = new AbortController();
399
430
  context.currentRunId = runId;
400
431
  context.currentRunWorkspace = runWorkspace;
401
- const runBody = { ...body, workspace: runWorkspace, runId };
432
+ let resolveReady;
433
+ let rejectReady;
434
+ const ready = waitForPlan ? new Promise((resolve, reject) => { resolveReady = resolve; rejectReady = reject; }) : null;
435
+ const runBody = {
436
+ ...body,
437
+ workspace: runWorkspace,
438
+ runId,
439
+ ...(waitForPlan ? { _planReady: { resolve: resolveReady, reject: rejectReady } } : {}),
440
+ };
402
441
  if (controlItemId) {
403
442
  dispatchAgentEvent(context.session, createAgentEvent('control_started', {
404
443
  origin: 'runtime',
@@ -410,6 +449,7 @@ export function startRuntimeServer({
410
449
  const runPromise = run(context, runBody, { signal: context.currentAbortController.signal, runId });
411
450
  runPromise
412
451
  .catch((err) => {
452
+ rejectReady?.(err);
413
453
  context.session?._onRuntimeError?.(err);
414
454
  })
415
455
  .finally(() => {
@@ -420,7 +460,7 @@ export function startRuntimeServer({
420
460
  publishState(runWorkspace, context);
421
461
  void startNextControlRequest(context);
422
462
  });
423
- return { accepted: true, runId, workspace: runWorkspace };
463
+ return { accepted: true, runId, workspace: runWorkspace, ...(ready ? { ready } : {}) };
424
464
  }
425
465
 
426
466
  function startNextControlRequest(context) {
@@ -1,3 +1,5 @@
1
+ import { openSync, readSync, closeSync, fstatSync } from 'node:fs';
2
+ import { isAbsolute, join, normalize, resolve } from 'node:path';
1
3
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
2
4
  import { extractActivity, parseJsonText, sessionActivities } from '../core/activity.js';
3
5
  import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
@@ -117,7 +119,10 @@ export async function pollActivitiesOnce(session, {
117
119
  const retry = progress.retryAt
118
120
  ? `retry ${progress.retryAt}`
119
121
  : (progress.waitMs ? `wait ${progress.waitMs}ms` : null);
120
- const progressKey = `${polledActivity.status}:${Number.isFinite(percent) ? Math.round(percent) : ''}:${detail}:${batch ?? ''}:${lastEvent ?? ''}:${retry ?? ''}`;
122
+ // retryAt/waitMs are scheduling metadata and may be recomputed on every
123
+ // status poll. They must remain visible in the first log line, but must
124
+ // not turn an unchanged quota/backoff state into a new trace event.
125
+ const progressKey = `${polledActivity.status}:${Number.isFinite(percent) ? Math.round(percent) : ''}:${detail}:${batch ?? ''}:${lastEvent ?? ''}`;
121
126
  session._activityLogKeys ??= {};
122
127
  if (session._activityLogKeys[key] !== progressKey) {
123
128
  session._activityLogKeys[key] = progressKey;
@@ -144,9 +149,19 @@ export async function pollActivitiesOnce(session, {
144
149
  }
145
150
  session._activityLogCursors[key] = logTail.at(-1);
146
151
  }
152
+ // The job log is terse — the actual narrative (per-document plans,
153
+ // LLM calls with token counts, apply operations) lives in the
154
+ // engine's trace file, whose path the status exposes. Tail it from
155
+ // the host and surface the significant events.
156
+ if (progress.traceFile) {
157
+ for (const line of readNewTraceLines(session, key, progress.traceFile)) {
158
+ emitRuntimeLog(session, `trace: ${line}`);
159
+ }
160
+ }
147
161
  if (polledActivity.terminal) {
148
162
  delete session._activityLogKeys[key];
149
163
  delete session._activityLogCursors?.[key];
164
+ delete session._activityTraceCursors?.[key];
150
165
  await startNextQueuedJob(session, {
151
166
  addLog: (message) => emitRuntimeLog(session, message),
152
167
  });
@@ -210,3 +225,52 @@ export async function cancelActiveActivityJobs(session, { callTool = callMcpTool
210
225
  }
211
226
  return cancelled;
212
227
  }
228
+
229
+ // Trace events worth surfacing in the chat-side log: plan/apply milestones,
230
+ // LLM calls (with token counts), warnings and errors. The raw trace is far
231
+ // chattier — streaming everything would recreate the noise the dedupe killed.
232
+ const TRACE_EVENT_PATTERN = /\b(llm:start|llm:end|llm:json|llm:error|ingest:plan|ingest:operations|ingest:apply|ingest:source|build:template|retrieval:|embedding:|WARN|ERROR)\b/;
233
+ const TRACE_MAX_BYTES_PER_POLL = 64 * 1024;
234
+ const TRACE_MAX_LINES_PER_POLL = 12;
235
+
236
+ export function readNewTraceLines(session, key, traceFile) {
237
+ const workspacePath = session?.workspacePath;
238
+ if (!workspacePath) return [];
239
+ // The trace path comes from an agent payload: never let it escape the
240
+ // workspace directory.
241
+ const resolved = resolve(workspacePath, normalize(String(traceFile)));
242
+ if (isAbsolute(String(traceFile)) || !resolved.startsWith(resolve(workspacePath))) return [];
243
+ let fd;
244
+ try {
245
+ fd = openSync(resolved, 'r');
246
+ const size = fstatSync(fd).size;
247
+ session._activityTraceCursors ??= {};
248
+ let offset = session._activityTraceCursors[key];
249
+ // First sighting (or file rotation): start near the end, not at byte 0 —
250
+ // replaying a long history would flood the panel.
251
+ if (offset == null || offset > size) offset = Math.max(0, size - 4096);
252
+ if (size <= offset) return [];
253
+ const length = Math.min(size - offset, TRACE_MAX_BYTES_PER_POLL);
254
+ const buffer = Buffer.alloc(length);
255
+ readSync(fd, buffer, 0, length, offset);
256
+ const chunk = buffer.toString('utf8');
257
+ // Only advance past COMPLETE lines so a partially-written line is
258
+ // re-read whole on the next poll.
259
+ const lastNewline = chunk.lastIndexOf('\n');
260
+ if (lastNewline === -1) return [];
261
+ session._activityTraceCursors[key] = offset + Buffer.byteLength(chunk.slice(0, lastNewline + 1), 'utf8');
262
+ return chunk
263
+ .slice(0, lastNewline)
264
+ .split('\n')
265
+ .map((line) => line.trim())
266
+ .filter((line) => line && TRACE_EVENT_PATTERN.test(line))
267
+ .slice(0, TRACE_MAX_LINES_PER_POLL)
268
+ // Strip the ISO timestamp + elapsed prefix: the runtime log adds its
269
+ // own clock, and the double timestamp ate half the panel width.
270
+ .map((line) => line.replace(/^\S+\s+\+[\d.]+(?:ms|s|m|h)\s+(INFO|WARN|ERROR)\s+/, (_m, level) => (level === 'INFO' ? '' : `${level} `)));
271
+ } catch {
272
+ return [];
273
+ } finally {
274
+ if (fd !== undefined) closeSync(fd);
275
+ }
276
+ }
@@ -41,6 +41,55 @@ test('pollActivitiesOnce updates activity through the event reducer', async () =
41
41
  assert.ok(session.agentProjection.logs.some((line) => line.includes('activity:')));
42
42
  });
43
43
 
44
+ test('pollActivitiesOnce does not repeat an unchanged retry state when retryAt moves', async () => {
45
+ const session = {
46
+ mcp: { production: { status: 'connected' } },
47
+ activities: {},
48
+ headlessPlan: null,
49
+ jobQueue: [],
50
+ };
51
+ dispatchAgentEvent(session, createAgentEvent('activity_upserted', {
52
+ payload: {
53
+ activity: {
54
+ id: 'job-quota',
55
+ source: 'production',
56
+ label: 'Production · ingest',
57
+ status: 'running',
58
+ poll: { server: 'production', tool: 'production_job_status', args: { jobId: 'job-quota' }, intervalMs: 0 },
59
+ },
60
+ },
61
+ }));
62
+
63
+ let poll = 0;
64
+ const callTool = async () => {
65
+ poll += 1;
66
+ return {
67
+ content: [{ type: 'text', text: JSON.stringify({
68
+ _activity: {
69
+ id: 'job-quota',
70
+ source: 'production',
71
+ label: 'Production · ingest',
72
+ status: 'running',
73
+ terminal: false,
74
+ progress: {
75
+ percent: 15,
76
+ detail: 'LLM quota wait',
77
+ lastEvent: 'llm:rate-limit-wait',
78
+ retryAt: `2026-07-10T20:31:5${poll}.000Z`,
79
+ },
80
+ },
81
+ }) }],
82
+ };
83
+ };
84
+
85
+ await pollActivitiesOnce(session, { callTool });
86
+ await pollActivitiesOnce(session, { callTool });
87
+
88
+ const lines = session.agentProjection.logs.filter((line) => line.includes('activity: Production · ingest'));
89
+ assert.equal(lines.length, 1);
90
+ assert.match(lines[0], /retry 2026-07-10T20:31:51\.000Z/);
91
+ });
92
+
44
93
  test('pollActivitiesOnce retries transient MCP poll failures', async () => {
45
94
  const originalFetch = globalThis.fetch;
46
95
  let attempts = 0;
@@ -284,3 +333,34 @@ async function waitFor(predicate, timeoutMs = 250) {
284
333
  }
285
334
  assert.fail('Timed out waiting for condition.');
286
335
  }
336
+
337
+ test('readNewTraceLines streams significant trace events with a per-activity cursor', async () => {
338
+ const { readNewTraceLines } = await import('./supervisor.js');
339
+ const { mkdtempSync, writeFileSync, appendFileSync, rmSync } = await import('node:fs');
340
+ const { tmpdir } = await import('node:os');
341
+ const { join } = await import('node:path');
342
+ const workspacePath = mkdtempSync(join(tmpdir(), 'trace-tail-'));
343
+ try {
344
+ const rel = '.wiki/logs/ingest-test.log';
345
+ const abs = join(workspacePath, rel);
346
+ (await import('node:fs')).mkdirSync(join(workspacePath, '.wiki/logs'), { recursive: true });
347
+ writeFileSync(abs, '2026-07-10T13:00:00.000Z +10ms INFO llm:start label=ingest_plan promptChars=42000\n');
348
+ const session = { workspacePath };
349
+
350
+ const first = readNewTraceLines(session, 'k1', rel);
351
+ assert.deepEqual(first, ['llm:start label=ingest_plan promptChars=42000']);
352
+
353
+ // No new content → nothing re-emitted.
354
+ assert.deepEqual(readNewTraceLines(session, 'k1', rel), []);
355
+
356
+ appendFileSync(abs, '2026-07-10T13:01:00.000Z +70s INFO noise: irrelevant heartbeat\n2026-07-10T13:02:00.000Z +130s WARN embedding:neutralized-input status=413\n');
357
+ const second = readNewTraceLines(session, 'k1', rel);
358
+ assert.deepEqual(second, ['WARN embedding:neutralized-input status=413']);
359
+
360
+ // Path traversal attempts are refused.
361
+ assert.deepEqual(readNewTraceLines(session, 'k1', '../../etc/passwd'), []);
362
+ assert.deepEqual(readNewTraceLines(session, 'k1', '/etc/passwd'), []);
363
+ } finally {
364
+ rmSync(workspacePath, { recursive: true, force: true });
365
+ }
366
+ });
@@ -272,9 +272,16 @@ function renderMarkdownLines(lines: Array<{ text: string; isCode: boolean }>, ro
272
272
  for (let index = 0; index < lines.length; index += 1) {
273
273
  const { text, isCode } = lines[index];
274
274
  if (isCode) {
275
- output.push(...wrapLine(text || ' ', columns).map((piece) => ({
276
- segments: [{ text: piece || ' ', color: '#D6DEE8', bg: '#1A2235' }],
275
+ const blockStarts = index === 0 || !lines[index - 1].isCode;
276
+ const blockEnds = index === lines.length - 1 || !lines[index + 1].isCode;
277
+ // Blank line before/after the block + 2-space inner padding: fenced
278
+ // blocks used to render as a dense background slab glued to the text.
279
+ if (blockStarts) output.push({ segments: [{ text: ' ', color: '#D6DEE8' }] });
280
+ const innerWidth = Math.max(8, columns - 4);
281
+ output.push(...wrapLine(text || ' ', innerWidth).map((piece) => ({
282
+ segments: [{ text: ` ${(piece || ' ').padEnd(innerWidth)} `, color: '#D6DEE8', bg: '#1A2235' }],
277
283
  })));
284
+ if (blockEnds) output.push({ segments: [{ text: ' ', color: '#D6DEE8' }] });
278
285
  continue;
279
286
  }
280
287
 
package/src/shell/repl.js CHANGED
@@ -5,7 +5,7 @@ import { execFileSync } from 'node:child_process';
5
5
  import { stdin as input, stdout as output } from 'node:process';
6
6
  import { marked } from 'marked';
7
7
  import { markedTerminal } from 'marked-terminal';
8
- import { buildAgentSystemPrompt, classifyAgentInput, formatLlmUnavailableMessage } from '../agent/graph.js';
8
+ import { buildAgentSystemPrompt, formatLlmUnavailableMessage } from '../agent/graph.js';
9
9
  import { handleSlashCommand } from '../commands/slash.js';
10
10
  import { serviceDescription, serviceNames as composeServiceNames } from '../core/compose.js';
11
11
  import { extractActivity, parseJsonText, sessionActivities } from '../core/activity.js';
@@ -18,7 +18,15 @@ import { listWorkspaces } from '../core/workspaces.js';
18
18
  import { fetchRuntimeState, postRuntimeApprove, postRuntimeCancel, postRuntimeControl, postRuntimeRun, postRuntimeShutdown, streamRuntimeEvents } from '../runtime/client.js';
19
19
  import { versionWithBuild } from '../core/buildInfo.js';
20
20
 
21
- marked.use(markedTerminal());
21
+ // Code blocks: marked-terminal's default paints a dense background block
22
+ // glued to the surrounding text. Indent the content, keep a plain style and
23
+ // guarantee a blank line before/after so fenced blocks breathe in the chat.
24
+ const CODE_INDENT = ' ';
25
+ const CODE_TINT = '\u001b[38;5;152m'; // soft blue-grey, readable on dark bg
26
+ const CODE_RESET = '\u001b[0m';
27
+ marked.use(markedTerminal({
28
+ code: (code) => `\n${String(code).split('\n').map((line) => `${CODE_INDENT}${CODE_TINT}${line}${CODE_RESET}`).join('\n')}\n`,
29
+ }));
22
30
  // marked-terminal's text renderer extracts token.text (raw string) instead of
23
31
  // calling parseInline(token.tokens), so inline Markdown inside list items is
24
32
  // silently dropped. Patch it to call parseInline when tokens are available.
@@ -75,7 +83,7 @@ const COMMAND_COMPLETION_DESCRIPTIONS = {
75
83
  '/chat': 'Switch free text to direct LLM chat without tools.',
76
84
  '/agent': 'Switch free text to the LangGraph agent with tools.',
77
85
  '/openui': 'Open the workspace web UI in the browser.',
78
- '/run': 'Inspect, cancel, or kill runtime runs.',
86
+ '/run': 'Inspect, cancel, kill runtime runs, or start a capability run.',
79
87
  '/approve': 'Approve a pending runtime run or tool.',
80
88
  };
81
89
 
@@ -141,7 +149,7 @@ export function createSession() {
141
149
  wikircConfig: null,
142
150
  language: null,
143
151
  mcp: null,
144
- commands: ['help', 'version', 'exit', 'workspace', 'new', 'use', 'config', 'status', 'services', 'start', 'stop', 'logs', 'mcp', 'wiki', 'skills', 'upload', 'uploads', 'clear', 'chat', 'agent', 'openui', 'run', 'queue', 'approve'],
152
+ commands: ['help', 'version', 'exit', 'workspace', 'new', 'use', 'config', 'status', 'services', 'start', 'stop', 'logs', 'mcp', 'wiki', 'skills', 'upload', 'uploads', 'clear', 'chat', 'agent', 'openui', 'run', 'cancel', 'queue', 'approve'],
145
153
  chatMode: true,
146
154
  llm: null,
147
155
  activities: {},
@@ -246,7 +254,7 @@ function completionValuesFor(parts, inputBuffer, session) {
246
254
  if (command === '/upload' && parts[1] === 'convert' && tokenIndex === 2) return ['pending'];
247
255
  if (command === '/uploads' && tokenIndex === 1) return ['clean', 'list'];
248
256
  if (command === '/uploads' && previousToken === 'clean') return ['--older-than'];
249
- if (command === '/run' && tokenIndex === 1) return ['status', 'cancel', 'kill'];
257
+ if (command === '/run' && tokenIndex === 1) return ['status', 'cancel', 'kill', 'capability'];
250
258
  if (command === '/queue' && tokenIndex === 1) return ['cancel', 'clear'];
251
259
  if (command === '/queue' && previousToken === 'cancel') {
252
260
  return (session.jobQueue ?? [])
@@ -284,6 +292,7 @@ function buildDirectChatSystemPrompt(session) {
284
292
  return [
285
293
  'You are Donna, the llm-wiki-manager chat assistant.',
286
294
  'Answer directly and concisely. Do not claim to have called tools or changed files.',
295
+ 'Never add a "Next steps", "Prochaines étapes", "À suivre", options, or suggestions section unless the user explicitly asks what to do next. End after answering the question.',
287
296
  'If the user asks for an action that needs workspace commands, MCP tools, services, files, or mutations, say to ask as an agent action instead of pretending to execute it.',
288
297
  `Reply language: ${language}.`,
289
298
  `Current workspace: ${workspace}.`,
@@ -766,18 +775,21 @@ function rememberProductionActivity(session, payload) {
766
775
 
767
776
  export function applyRuntimeStateToShellSession(session, state) {
768
777
  if (!state || typeof state !== 'object') return false;
778
+ const displayState = sanitizeRuntimeStateForDisplay(state);
769
779
  session.agentProjection = {
770
- conversation: Array.isArray(state.conversation) ? state.conversation.map((message) => ({ ...message })) : [],
771
- chain: Array.isArray(state.chain) ? state.chain.map((step) => ({ ...step })) : [],
772
- plan: Array.isArray(state.plan) ? state.plan.map((step) => ({ ...step })) : null,
773
- activities: Array.isArray(state.activities) ? state.activities.map((activity) => ({ ...activity })) : [],
774
- logs: Array.isArray(state.logs) ? [...state.logs] : [],
775
- summary: state.summary ?? null,
776
- status: state.status ?? 'idle',
777
- planRevision: state.planRevision ?? 0,
778
- planPatches: Array.isArray(state.planPatches) ? state.planPatches.map((patch) => ({ ...patch })) : [],
780
+ conversation: Array.isArray(displayState.conversation) ? displayState.conversation.map((message) => ({ ...message })) : [],
781
+ chain: Array.isArray(displayState.chain) ? displayState.chain.map((step) => ({ ...step })) : [],
782
+ plan: Array.isArray(displayState.plan) && displayState.plan.length > 0
783
+ ? displayState.plan.map((step) => ({ ...step }))
784
+ : null,
785
+ activities: Array.isArray(displayState.activities) ? displayState.activities.map((activity) => ({ ...activity })) : [],
786
+ logs: Array.isArray(displayState.logs) ? [...displayState.logs] : [],
787
+ summary: displayState.summary ?? null,
788
+ status: displayState.status ?? 'idle',
789
+ planRevision: displayState.planRevision ?? 0,
790
+ planPatches: Array.isArray(displayState.planPatches) ? displayState.planPatches.map((patch) => ({ ...patch })) : [],
779
791
  };
780
- session.workflow = state.workflow && typeof state.workflow === 'object'
792
+ session.workflow = displayState.workflow && typeof displayState.workflow === 'object'
781
793
  ? {
782
794
  ...state.workflow,
783
795
  nodes: Array.isArray(state.workflow.nodes) ? state.workflow.nodes.map((node) => ({ ...node })) : [],
@@ -795,7 +807,7 @@ export function applyRuntimeStateToShellSession(session, state) {
795
807
  // Runtime queue items replace the local jobQueue wholesale on every sync.
796
808
  // Tag their origin so /queue cancel can refuse to fake-cancel them locally
797
809
  // (a local status flip would be silently reverted by the next SSE sync).
798
- if (Array.isArray(state.queue)) session.jobQueue = state.queue.map((item) => ({ ...item, origin: 'runtime' }));
810
+ if (Array.isArray(displayState.queue)) session.jobQueue = displayState.queue.map((item) => ({ ...item, origin: 'runtime' }));
799
811
  const production = session.agentProjection.activities.filter((activity) => activity.source === 'production').at(-1);
800
812
  if (production) {
801
813
  session.productionActivity = {
@@ -809,21 +821,29 @@ export function applyRuntimeStateToShellSession(session, state) {
809
821
  return true;
810
822
  }
811
823
 
812
- // Submits a prompt to the shared runtime. If the workspace is already busy
813
- // (HTTP 409 from POST /run), route the input through the runtime control lane
814
- // so status questions and plan-change proposals do not become future runs.
815
- // A plain question or small talk must never start a runtime run (nor be
816
- // enqueued as a future one): it only needs an answer. Route converse/observe
817
- // to the local agent EVEN during an active run — the chat is supposed to stay
818
- // available, and the graph already restricts tools to read-only in that case.
819
- // Actions/cancels/approvals still go to the runtime.
820
- export function shouldHandleFreeTextLocally(line, session, { llmAvailable = Boolean(session?.llm) } = {}) {
821
- const classification = classifyAgentInput(line, session);
822
- // cancel intents included: Donna interprets "supprime le job et la queue"
823
- // and calls runtime__kill / runtime__cancel herself — no hardcoded regex
824
- // deciding between soft and hard stop. The control lane remains the
825
- // deterministic fallback when the local LLM is down.
826
- if (!['converse', 'observe', 'cancel', 'approve', 'enqueue_run', 'ambiguous'].includes(classification.kind)) return { local: false, classification };
824
+ export function sanitizeRuntimeStateForDisplay(state) {
825
+ if (!state || typeof state !== 'object') return state;
826
+ const status = String(state.status ?? 'idle').toLowerCase();
827
+ const visible = ['running', 'queued', 'pending', 'waiting', 'pending_approval', 'error', 'failed']
828
+ .includes(status);
829
+ if (visible) return state;
830
+ return {
831
+ ...state,
832
+ conversation: [],
833
+ chain: [],
834
+ plan: [],
835
+ activities: [],
836
+ workflow: null,
837
+ logs: [],
838
+ summary: null,
839
+ planPatches: [],
840
+ };
841
+ }
842
+
843
+ // In agent mode, Donna receives every free-text turn and decides whether to
844
+ // answer or call an exposed tool. Slash commands remain the deterministic UI.
845
+ export function shouldHandleFreeTextLocally(_line, session, { llmAvailable = Boolean(session?.llm) } = {}) {
846
+ const classification = { kind: 'agent_turn', confidence: 1, reason: 'agent_mode_llm_decision' };
827
847
  if (!llmAvailable) return { local: false, classification, fallbackReason: 'local LLM unavailable' };
828
848
  return { local: true, classification };
829
849
  }
@@ -1041,17 +1061,12 @@ async function runAgentTurn(input, { agent, session, onUpdate, onStep, displayIn
1041
1061
  };
1042
1062
  session._onStreamReset = () => {
1043
1063
  if (!donnaMessage) return;
1044
- if (donnaMessage.content.trim()) {
1045
- // Intermediate streamed text before tool calls: keep it, add separator.
1046
- donnaMessage.content += '\n\n';
1047
- onUpdate?.();
1048
- } else {
1049
- // Still empty ("Thinking…"): remove it cleanly.
1050
- const index = messages.indexOf(donnaMessage);
1051
- if (index !== -1) messages.splice(index, 1);
1052
- donnaMessage = null;
1053
- onUpdate?.();
1054
- }
1064
+ // Text emitted before a tool call is provisional narration, not an answer.
1065
+ // Remove it; the post-tool result gets a fresh Donna bubble.
1066
+ const index = messages.indexOf(donnaMessage);
1067
+ if (index !== -1) messages.splice(index, 1);
1068
+ donnaMessage = null;
1069
+ onUpdate?.();
1055
1070
  };
1056
1071
 
1057
1072
  let agentResult;
@@ -9,12 +9,40 @@ import {
9
9
  conversationMessages,
10
10
  recordRuntimeUnavailableAgentInput,
11
11
  runLine,
12
+ sanitizeRuntimeStateForDisplay,
12
13
  runtimeStatusLine,
13
14
  runtimeUnavailableAgentMessage,
14
15
  shouldHandleFreeTextLocally,
15
16
  submitRuntimeRun,
16
17
  } from './repl.js';
17
18
 
19
+ test('runtime display preserves a failed plan and its diagnostic evidence', () => {
20
+ const state = {
21
+ status: 'error',
22
+ plan: [{ id: 'apply', status: 'failed' }],
23
+ activities: [{ id: 'job-1', status: 'failed', error: 'exitCode=1' }],
24
+ logs: ['run_error: ingest_apply exitCode=1'],
25
+ conversation: [{ role: 'assistant', content: 'Échec de l’ingestion.' }],
26
+ };
27
+
28
+ assert.equal(sanitizeRuntimeStateForDisplay(state), state);
29
+ });
30
+
31
+ test('runtime display still clears completed historical execution state', () => {
32
+ const display = sanitizeRuntimeStateForDisplay({
33
+ status: 'done',
34
+ plan: [{ id: 'old', status: 'done' }],
35
+ activities: [{ id: 'old-job', status: 'done' }],
36
+ logs: ['old log'],
37
+ conversation: [{ role: 'assistant', content: 'Old run.' }],
38
+ });
39
+
40
+ assert.deepEqual(display.plan, []);
41
+ assert.deepEqual(display.activities, []);
42
+ assert.deepEqual(display.logs, []);
43
+ assert.deepEqual(display.conversation, []);
44
+ });
45
+
18
46
  function stubFetch(handler) {
19
47
  const original = globalThis.fetch;
20
48
  globalThis.fetch = handler;
@@ -74,6 +102,50 @@ test('applyRuntimeStateToShellSession projects runtime state into shell session'
74
102
  assert.deepEqual(conversationMessages(session), []);
75
103
  });
76
104
 
105
+ test('applyRuntimeStateToShellSession clears terminal plan and activities when runtime is idle', () => {
106
+ const session = createSession();
107
+ session.headlessPlan = [{ step: 1, description: 'Old read', status: 'failed' }];
108
+ session.activities = { old: { key: 'old', status: 'failed', terminal: true } };
109
+
110
+ applyRuntimeStateToShellSession(session, {
111
+ status: 'idle',
112
+ conversation: [{ role: 'assistant', content: 'Old failed answer' }],
113
+ chain: [{ id: 'old-step' }],
114
+ plan: [{ step: 1, description: 'Old read', status: 'failed' }],
115
+ activities: [{ key: 'old', status: 'failed', terminal: true }],
116
+ workflow: { nodes: [{ id: 'task:old' }], relations: [] },
117
+ logs: ['Runtime evaluator rejected the old run'],
118
+ summary: 'Old run failed',
119
+ planPatches: [{ id: 'old-patch' }],
120
+ });
121
+
122
+ assert.equal(session.headlessPlan, null);
123
+ assert.deepEqual(session.activities, {});
124
+ assert.equal(session.workflow, null);
125
+ assert.deepEqual(session.agentProjection.logs, []);
126
+ assert.equal(session.agentProjection.summary, null);
127
+ assert.deepEqual(session.agentProjection.conversation, []);
128
+ assert.deepEqual(session.agentProjection.chain, []);
129
+ assert.deepEqual(session.agentProjection.planPatches, []);
130
+ });
131
+
132
+ test('direct chat system prompt forbids unsolicited next steps', async () => {
133
+ const session = createSession();
134
+ let systemPrompt = '';
135
+ session.llm = {
136
+ async *stream({ system }) {
137
+ systemPrompt = system;
138
+ yield 'Réponse concise.';
139
+ },
140
+ };
141
+
142
+ await runLine('bonjour', { agent: null, packageJson: { version: 'test' }, session, chatMode: true });
143
+
144
+ assert.match(systemPrompt, /Never add a "Next steps", "Prochaines étapes", "À suivre"/);
145
+ assert.match(systemPrompt, /unless the user explicitly asks what to do next/);
146
+ assert.equal(conversationMessages(session).at(-1).content, 'Réponse concise.');
147
+ });
148
+
77
149
  test('submitRuntimeRun reports acceptance without throwing', async () => {
78
150
  const restore = stubFetch(async (url) => {
79
151
  assert.equal(pathOf(url), '/run');
@@ -241,37 +313,34 @@ test('/queue cancel on a runtime workflow id points to run cancellation commands
241
313
  assert.match(conversationMessages(session).at(-1).content, /\/run kill/);
242
314
  });
243
315
 
244
- test('free text routing keeps questions local and sends actions to the runtime', () => {
316
+ test('agent mode sends every free-text turn to Donna', () => {
245
317
  const session = createSession();
246
318
  session.llm = { completeWithTools: () => {} };
247
319
 
248
- // The original incident: a config question must never start a run.
249
320
  const question = shouldHandleFreeTextLocally('donne moi la config du cme', session);
250
321
  assert.equal(question.local, true);
251
- assert.equal(question.classification.kind, 'observe');
322
+ assert.equal(question.classification.kind, 'agent_turn');
252
323
 
253
324
  const smallTalk = shouldHandleFreeTextLocally('bonjour', session);
254
325
  assert.equal(smallTalk.local, true);
255
326
 
256
327
  const action = shouldHandleFreeTextLocally('lance le pipeline complet', session);
257
- assert.equal(action.local, false);
258
- assert.equal(action.classification.kind, 'start_run');
328
+ assert.equal(action.local, true);
329
+ assert.equal(action.classification.kind, 'agent_turn');
330
+
331
+ const pending = shouldHandleFreeTextLocally('as ton des fichier en attente d ingestion', session);
332
+ assert.equal(pending.local, true);
333
+ assert.equal(pending.classification.kind, 'agent_turn');
259
334
  });
260
335
 
261
- test('free text routing keeps questions local even during an active run', () => {
262
- // The chat must stay available during a run: a status question or small
263
- // talk answered locally (read-only tools) — never enqueued as a future run.
336
+ test('Donna keeps receiving free text during an active run', () => {
264
337
  const session = createSession();
265
338
  session.llm = { completeWithTools: () => {} };
266
339
  session.agentProjection = { status: 'running', activities: [], conversation: [] };
267
340
  assert.equal(shouldHandleFreeTextLocally('où en est le run', session).local, true);
268
341
  assert.equal(shouldHandleFreeTextLocally('salut', session).local, true);
269
- // Cancel intents are handled by Donna locally (runtime__kill/cancel tools);
270
- // approvals stay on the deterministic control lane.
271
342
  assert.equal(shouldHandleFreeTextLocally('stop le job', session).local, true);
272
343
  assert.equal(shouldHandleFreeTextLocally('supprime le job et la queue', session).local, true);
273
- // Approvals and "later" requests too: Donna owns runtime__approve and
274
- // runtime__enqueue. Only plan modifications and new runs bypass her.
275
344
  assert.equal(shouldHandleFreeTextLocally('approuve le run', session).local, true);
276
345
  assert.equal(shouldHandleFreeTextLocally('fais le build plus tard', session).local, true);
277
346
 
package/src/shell/tui.tsx CHANGED
@@ -116,6 +116,29 @@ function App(props: {
116
116
  // sync at every call site.
117
117
  const [screen, setScreen] = createSignal<'startup' | 'setup' | 'main'>('startup');
118
118
  let ctrlCTimer: ReturnType<typeof setTimeout> | null = null;
119
+ let exiting = false;
120
+ // Single exit path: the owned-runtime shutdown MUST happen here, on the
121
+ // user's actual exit gesture. render() resolves at MOUNT, so code placed
122
+ // after `await runOpenTuiShell(...)` runs while the shell is still on
123
+ // screen — 0.12.9 shipped that and killed the runtime mid-session.
124
+ const exitShell = () => {
125
+ if (exiting) return;
126
+ exiting = true;
127
+ void (async () => {
128
+ const messages: string[] = [];
129
+ try {
130
+ if (props.runtime?.url) {
131
+ const { shutdownOwnedRuntime } = await import('../runtime/lifecycle.js');
132
+ await shutdownOwnedRuntime(props.runtime, { log: (message: string) => { messages.push(message); } });
133
+ }
134
+ } catch {
135
+ // Best effort: never block the exit on runtime cleanup.
136
+ }
137
+ renderer.destroy();
138
+ // Print AFTER destroy so the note survives on the restored terminal.
139
+ for (const message of messages) console.log(`[wiki-manager] ${message}`);
140
+ })();
141
+ };
119
142
  let copyHintTimer: ReturnType<typeof setTimeout> | null = null;
120
143
  let selectionCopyTimer: ReturnType<typeof setTimeout> | null = null;
121
144
  let startupKeyboardEventId = 0;
@@ -141,7 +164,7 @@ function App(props: {
141
164
  return;
142
165
  }
143
166
  void state.submitInput(value).then((result) => {
144
- if (result?.exit) renderer.destroy();
167
+ if (result?.exit) exitShell();
145
168
  });
146
169
  };
147
170
 
@@ -263,7 +286,7 @@ function App(props: {
263
286
  return;
264
287
  }
265
288
  if (exitHint()) {
266
- renderer.destroy();
289
+ exitShell();
267
290
  return;
268
291
  }
269
292
  setExitHint(true);
@@ -378,7 +401,7 @@ function App(props: {
378
401
  height={dimensions().height}
379
402
  keyboardEvent={startupKeyboardEvent()}
380
403
  onSelect={openAction}
381
- onQuit={() => renderer.destroy()}
404
+ onQuit={() => exitShell()}
382
405
  />
383
406
  </Show>
384
407
  );