@dotdrelle/wiki-manager 0.14.12 → 0.14.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/.env.example +11 -0
  2. package/docker-compose.yml +1 -1
  3. package/package.json +1 -1
  4. package/src/activity/activityAggregator.js +43 -13
  5. package/src/activity/activityAggregator.test.js +51 -2
  6. package/src/agent/graph.js +79 -11
  7. package/src/agent/graph.test.js +43 -3
  8. package/src/cli/wiki-manager.js +44 -3
  9. package/src/commands/slash.js +10 -3
  10. package/src/commands/slash.test.js +24 -0
  11. package/src/core/activity.js +4 -0
  12. package/src/core/activity.test.js +31 -0
  13. package/src/core/agentEvents.js +11 -2
  14. package/src/core/buildInfo.json +2 -2
  15. package/src/core/dockerCompose.test.js +4 -0
  16. package/src/core/env.js +16 -4
  17. package/src/core/env.test.js +3 -0
  18. package/src/core/mcp.js +1 -1
  19. package/src/core/wikiSetup.js +43 -1
  20. package/src/core/wikiWorkspace.test.js +20 -0
  21. package/src/core/workspaces.js +10 -2
  22. package/src/orchestrator/dependencyResolver.js +19 -1
  23. package/src/orchestrator/objectiveResolver.js +24 -0
  24. package/src/orchestrator/objectiveResolver.test.js +23 -1
  25. package/src/orchestrator/scheduler.test.js +22 -1
  26. package/src/runtime/auth.test.js +65 -1
  27. package/src/runtime/client.js +4 -0
  28. package/src/runtime/donna-contract.test.js +2 -0
  29. package/src/runtime/lifecycle.js +21 -12
  30. package/src/runtime/runner.js +130 -17
  31. package/src/runtime/runner.test.js +62 -1
  32. package/src/runtime/server.js +13 -2
  33. package/src/runtime/server.test.js +30 -1
  34. package/src/shell/FileEditorDialog.tsx +2 -2
  35. package/src/shell/LeftPane.tsx +60 -15
  36. package/src/shell/RightPane.tsx +147 -54
  37. package/src/shell/StartupScreen.tsx +3 -7
  38. package/src/shell/renderer.ts +1 -0
  39. package/src/shell/repl.js +7 -81
  40. package/src/shell/repl.test.js +128 -38
  41. package/src/shell/tui.tsx +59 -64
  42. package/src/shell/useSession.ts +45 -6
  43. package/wiki-workspace +36 -3
package/.env.example CHANGED
@@ -63,6 +63,17 @@ DOCUMENTS_MCP_AUTH_TOKEN=
63
63
  # Add your own entries when you declare additional endpoints.
64
64
 
65
65
  # ── Orchestration (optional) ───────────────────────────────────────────────────
66
+ # The runtime runs on the host while `llm-wiki serve` runs in Docker and
67
+ # reaches it through host.docker.internal. Listen on all host interfaces so
68
+ # the container can connect; exposed runtimes are protected by the generated
69
+ # WIKI_MANAGER_RUNTIME_TOKEN.
70
+ #
71
+ # WIKI_MANAGER_RUNTIME_PORT=7788
72
+ # Set to 0 to skip pulling and renewing already-running containers at startup.
73
+ # WIKI_MANAGER_AUTO_UPDATE=1
74
+ # WIKI_MANAGER_RUNTIME_HOST=0.0.0.0
75
+
76
+
66
77
  # Parallel tasks dispatched at once for capability runs. Defaults to what the
67
78
  # agent itself declares (agent_describe limits); set only to constrain it.
68
79
  # Example constraint (never raises an agent's declared capacity):
@@ -59,7 +59,7 @@ services:
59
59
  - DOCUMENT_MAX_UPLOAD_BYTES=${DOCUMENT_MAX_UPLOAD_BYTES:-52428800}
60
60
  - WIKI_MCP_PROXY_URL=http://host.docker.internal:${WIKI_MCP_PORT:-3101}/mcp
61
61
  - PRODUCTION_MCP_PROXY_URL=http://host.docker.internal:${PRODUCTION_MCP_PORT:-3102}/mcp/
62
- - WIKI_MANAGER_RUNTIME_URL=http://host.docker.internal:7788
62
+ - WIKI_MANAGER_RUNTIME_URL=http://host.docker.internal:${WIKI_MANAGER_RUNTIME_PORT:-7788}
63
63
  - WIKI_MANAGER_RUNTIME_TOKEN=${WIKI_MANAGER_RUNTIME_TOKEN:-}
64
64
  # HTTPS — set paths inside the container (e.g. /certs/server.crt) and uncomment the volume above
65
65
  #- WIKI_SERVE_TLS_CERT_PATH=/certs/server.crt
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dotdrelle/wiki-manager",
3
- "version": "0.14.12",
3
+ "version": "0.14.16",
4
4
  "description": "Agentic shell and orchestration cockpit for llm-wiki workspaces.",
5
5
  "license": "PolyForm-Noncommercial-1.0.0",
6
6
  "author": "dotrelle",
@@ -71,21 +71,37 @@ function groupLine(group, activities) {
71
71
  const running = group.tasks.filter((task) => ACTIVE.has(statusOf(task)));
72
72
  const failed = group.tasks.find((task) => statusOf(task) === 'failed');
73
73
  const waitingApproval = group.tasks.some((task) => ['pending_approval', 'waiting_approval'].includes(statusOf(task)));
74
- const activeAgents = new Set(running.map((task) => task.agentInstanceId).filter(Boolean)).size;
75
- const activeProgress = running
76
- .map((task) => progressForTask(task, activities))
77
- .find((value) => value != null);
74
+ // Activity polling can advance before the persisted task projection catches
75
+ // up. Treat a live, linked activity as authoritative instead of rendering
76
+ // the whole group as "validation 0%" while a worker visibly runs at 35%.
77
+ const activePair = group.tasks
78
+ .map((task) => ({ task, activity: activityForTask(task, activities) }))
79
+ .find(({ activity }) => activity && !activity.terminal && !DONE.has(statusOf(activity)));
80
+ const activeTask = activePair?.task ?? running[0] ?? null;
81
+ const activeActivity = activePair?.activity
82
+ ?? running.map((task) => activityForTask(task, activities)).find(Boolean);
83
+ const activeAgents = new Set([
84
+ ...running.map((task) => task.agentInstanceId),
85
+ activeTask?.agentInstanceId,
86
+ ].filter(Boolean)).size;
87
+ const activeProgress = Number.isFinite(Number(activeActivity?.progress?.percent))
88
+ ? Number(activeActivity.progress.percent)
89
+ : running.map((task) => Number(task?.progress?.percent)).find(Number.isFinite);
78
90
  let icon = '[ ]';
79
91
  let status = 'en attente';
80
- if (failed) {
92
+ // A group may contain an earlier failure while another independent task is
93
+ // still progressing. Show the live worker as running; surface the group
94
+ // failure once no work remains. Otherwise its business label was rendered
95
+ // red even though that exact task was healthy and advancing.
96
+ if (running.length > 0 || activeActivity) {
97
+ icon = '[...]';
98
+ status = activeProgress != null ? `${Math.round(activeProgress)} %` : `${done}/${total}`;
99
+ } else if (failed) {
81
100
  icon = '[!]';
82
101
  status = 'error';
83
102
  } else if (done === total) {
84
103
  icon = '[x]';
85
104
  status = 'done';
86
- } else if (running.length > 0) {
87
- icon = '[...]';
88
- status = activeProgress != null ? `${Math.round(activeProgress)} %` : `${done}/${total}`;
89
105
  } else if (waitingApproval) {
90
106
  icon = '[!]';
91
107
  status = 'validation';
@@ -100,12 +116,28 @@ function groupLine(group, activities) {
100
116
  // text. Fall back to the task-completion ratio when nothing is running.
101
117
  const percent = running.length > 0 && activeProgress != null
102
118
  ? Math.round(activeProgress)
119
+ : activeActivity && activeProgress != null
120
+ ? Math.round(activeProgress)
103
121
  : (total > 0 ? Math.round((done / total) * 100) : null);
122
+ const phaseTasks = activeTask
123
+ ? group.tasks.filter((task) => String(task.operation ?? '') === String(activeTask.operation ?? ''))
124
+ : [];
125
+ const taskIndex = activeTask ? phaseTasks.indexOf(activeTask) + 1 : null;
126
+ const taskTotal = phaseTasks.length || null;
104
127
  return {
105
128
  id: `group:${group.id}`,
106
129
  label: `${icon} ${group.label} - ${status}${agents}`,
107
130
  status,
108
- progress: { done, total, percent },
131
+ // Preserve the active worker's business progress (source/template/
132
+ // deliverable, phase and detail). ShellUI can then show the same useful
133
+ // information as the direct wiki CLI instead of only "knowledge.update".
134
+ progress: {
135
+ ...(activeActivity?.progress ?? {}),
136
+ done,
137
+ total,
138
+ percent,
139
+ ...(taskIndex ? { taskIndex, taskTotal, taskOperation: activeTask?.operation ?? null } : {}),
140
+ },
109
141
  activeAgents,
110
142
  };
111
143
  }
@@ -121,15 +153,13 @@ function activityLine(activity) {
121
153
  };
122
154
  }
123
155
 
124
- function progressForTask(task, activities) {
156
+ function activityForTask(task, activities) {
125
157
  const taskId = String(task.id ?? task.step ?? '');
126
158
  const activityKey = task.activityKey ?? task.ownerActivityKey ?? null;
127
- const match = activities.find((activity) =>
159
+ return activities.find((activity) =>
128
160
  (activityKey && (activity.key === activityKey || activity.id === activityKey))
129
161
  || String(activity?.progress?.stepId ?? '') === taskId,
130
162
  );
131
- const value = Number(match?.progress?.percent ?? task.progress?.percent);
132
- return Number.isFinite(value) ? value : null;
133
163
  }
134
164
 
135
165
  function statusOf(task) {
@@ -37,10 +37,10 @@ test('aggregateActivity exposes initial synthesis and grouped display lines', ()
37
37
  const activity = aggregateActivity({
38
38
  plan: [
39
39
  { id: 'collect', label: 'Collecte externe', groupId: 'collect', status: 'done', progressWeight: 1 },
40
- { id: 'enrich', label: 'Enrichissement commercial', groupId: 'enrich', requiredCapability: 'customer-data.enrich', status: 'running', progressWeight: 1, activityKey: 'activity-enrich' },
40
+ { id: 'enrich', label: 'Enrichissement commercial', groupId: 'enrich', requiredCapability: 'customer-data.enrich', operation: 'export', status: 'running', progressWeight: 1, activityKey: 'activity-enrich' },
41
41
  { id: 'publish', label: 'Publication', groupId: 'publish', status: 'pending', progressWeight: 1 },
42
42
  ],
43
- activities: [{ key: 'activity-enrich', label: 'Enrichissement commercial', status: 'running', progress: { percent: 63, stepId: 'enrich' } }],
43
+ activities: [{ key: 'activity-enrich', label: 'Enrichissement commercial', status: 'running', progress: { percent: 63, stepId: 'enrich', label: 'Export rapport.md', detail: 'Rendering PDF', currentStep: 'export' } }],
44
44
  }, [{
45
45
  type: 'plan.received',
46
46
  payload: {
@@ -56,6 +56,13 @@ test('aggregateActivity exposes initial synthesis and grouped display lines', ()
56
56
  assert.equal(activity.progress.percent, 54);
57
57
  assert.ok(activity.lines.some((line) => /\[x\] collect - done/.test(line.label)));
58
58
  assert.ok(activity.lines.some((line) => /\[\.\.\.\] customer-data\.enrich - 63 %/.test(line.label)));
59
+ const enrichLine = activity.lines.find((line) => /customer-data\.enrich/.test(line.label));
60
+ assert.equal(enrichLine.progress.label, 'Export rapport.md');
61
+ assert.equal(enrichLine.progress.detail, 'Rendering PDF');
62
+ assert.equal(enrichLine.progress.currentStep, 'export');
63
+ assert.equal(enrichLine.progress.taskIndex, 1);
64
+ assert.equal(enrichLine.progress.taskTotal, 1);
65
+ assert.equal(enrichLine.progress.taskOperation, 'export');
59
66
  assert.ok(activity.lines.some((line) => /\[ \] publish - en attente/.test(line.label)));
60
67
  });
61
68
 
@@ -90,3 +97,45 @@ test('aggregateActivity keeps activities not attached to any plan task visible',
90
97
  const ingestLine = aggregated.lines.find((line) => /Ingest/.test(line.label));
91
98
  assert.equal(ingestLine.status, 'running');
92
99
  });
100
+
101
+ test('aggregateActivity trusts live worker progress while the task projection lags', () => {
102
+ const aggregated = aggregateActivity({
103
+ plan: [
104
+ { id: 'plan-a', groupId: 'knowledge.update', operation: 'ingest_plan', status: 'pending_approval', activityKey: 'activity-plan-a' },
105
+ { id: 'plan-b', groupId: 'knowledge.update', operation: 'ingest_plan', status: 'pending' },
106
+ ],
107
+ activities: [{
108
+ key: 'activity-plan-a',
109
+ status: 'running',
110
+ terminal: false,
111
+ progress: { percent: 35, stepId: 'plan-a', label: 'Ingest source-a.md', stepIndex: 1, stepTotal: 1 },
112
+ }],
113
+ }, []);
114
+
115
+ const line = aggregated.lines.find((item) => /knowledge\.update/.test(item.label));
116
+ assert.match(line.label, /35 %/);
117
+ assert.equal(line.progress.percent, 35);
118
+ assert.equal(line.progress.label, 'Ingest source-a.md');
119
+ assert.equal(line.progress.taskIndex, 1);
120
+ assert.equal(line.progress.taskTotal, 2);
121
+ });
122
+
123
+ test('aggregateActivity keeps a healthy active task out of the error color when a sibling failed', () => {
124
+ const aggregated = aggregateActivity({
125
+ plan: [
126
+ { id: 'failed-a', groupId: 'ingest', operation: 'ingest_plan', status: 'failed' },
127
+ { id: 'running-b', groupId: 'ingest', operation: 'ingest_plan', status: 'running', activityKey: 'activity-b' },
128
+ ],
129
+ activities: [{
130
+ key: 'activity-b',
131
+ status: 'running',
132
+ terminal: false,
133
+ progress: { percent: 35, stepId: 'running-b', label: 'Ingest application-orea.md', detail: 'LLM running' },
134
+ }],
135
+ }, []);
136
+
137
+ const line = aggregated.lines[0];
138
+ assert.equal(line.status, '35 %');
139
+ assert.match(line.label, /^\[\.\.\.\]/);
140
+ assert.equal(line.progress.label, 'Ingest application-orea.md');
141
+ });
@@ -274,6 +274,7 @@ const AgentState = Annotation.Root({
274
274
  invalidResponseRetries: Annotation({ default: () => 0 }),
275
275
  invalidToolCallRetries: Annotation({ default: () => 0 }),
276
276
  forceDelegation: Annotation({ default: () => false }),
277
+ terminalToolFailure: Annotation({ default: () => false }),
277
278
  });
278
279
 
279
280
  function invalidToolCalls(toolCalls) {
@@ -365,21 +366,62 @@ export function invalidUserFacingToolNames(content, session) {
365
366
  return [...new Set([...connected, ...syntactic])].sort();
366
367
  }
367
368
 
369
+ function parseActionJson(text) {
370
+ const cleaned = String(text ?? '').trim().replace(/^```(?:json)?\s*/i, '').replace(/\s*```$/, '');
371
+ if (!cleaned) return null;
372
+ return JSON.parse(cleaned)?.action === true;
373
+ }
374
+
368
375
  async function classifyRequestedAction(llm, input, signal) {
376
+ const system = [
377
+ 'Classify whether the user explicitly requests a real state-changing action now.',
378
+ 'Actions include starting, stopping, importing, ingesting, building, exporting, configuring, writing, deleting, or sending.',
379
+ 'Questions, explanations, status questions, greetings, and hypothetical discussions are not actions.',
380
+ 'Return JSON only: {"action":true} or {"action":false}.',
381
+ ].join('\n');
382
+ const messages = [{ role: 'user', content: String(input ?? '') }];
383
+
384
+ // Preferred path: a forced structured tool call, reliable on providers that
385
+ // honour tool_choice. But an OpenAI-compatible gateway (e.g. Albert / gpt-oss)
386
+ // may reject a forced tool_choice or return neither tool_calls nor parsable
387
+ // content. Without a fallback that made EVERY request classify as a non-action
388
+ // (catch → false), so Donna silently stopped delegating in agent mode. Fall
389
+ // back to a plain JSON-text completion, and only give up if both paths fail.
369
390
  try {
391
+ const classifier = {
392
+ type: 'function',
393
+ function: {
394
+ name: 'classify_action_request',
395
+ description: 'Classify whether the user explicitly requests a real state-changing action now.',
396
+ parameters: {
397
+ type: 'object',
398
+ additionalProperties: false,
399
+ properties: { action: { type: 'boolean' } },
400
+ required: ['action'],
401
+ },
402
+ },
403
+ };
370
404
  const result = await llm.completeWithTools({
371
- system: [
372
- 'Classify whether the user explicitly requests a real state-changing action now.',
373
- 'Actions include starting, stopping, importing, ingesting, building, exporting, configuring, writing, deleting, or sending.',
374
- 'Questions, explanations, status questions, greetings, and hypothetical discussions are not actions.',
375
- 'Return JSON only: {"action":true} or {"action":false}.',
376
- ].join('\n'),
377
- tools: [],
378
- messages: [{ role: 'user', content: String(input ?? '') }],
405
+ system,
406
+ tools: [classifier],
407
+ toolChoice: { type: 'function', function: { name: 'classify_action_request' } },
408
+ messages,
379
409
  signal,
380
410
  });
381
- const text = String(result?.content ?? '').trim().replace(/^```(?:json)?\s*/i, '').replace(/\s*```$/, '');
382
- return JSON.parse(text)?.action === true;
411
+ const call = (result?.tool_calls ?? []).find((item) => item?.function?.name === 'classify_action_request');
412
+ if (call) {
413
+ const parsed = JSON.parse(call.function.arguments ?? '{}')?.action;
414
+ if (typeof parsed === 'boolean') return parsed;
415
+ }
416
+ const fromText = parseActionJson(result?.content);
417
+ if (fromText !== null) return fromText;
418
+ } catch {
419
+ // Fall through to the toolless path below.
420
+ }
421
+
422
+ try {
423
+ const result = await llm.completeWithTools({ system, tools: [], messages, signal });
424
+ return parseActionJson(result?.content) === true;
383
425
  } catch {
384
426
  return false;
385
427
  }
@@ -1327,6 +1369,7 @@ export function createAgentGraph(options = {}) {
1327
1369
  async function toolExecutorNode(state) {
1328
1370
  const toolCalls = state.pendingToolCalls ?? [];
1329
1371
  const toolResultMessages = [];
1372
+ let terminalFailure = null;
1330
1373
 
1331
1374
  for (const call of toolCalls) {
1332
1375
  const resolved = resolveToolCallName(state.session.mcp, call.function.name, INTERNAL_TOOL_SERVERS);
@@ -1415,6 +1458,12 @@ export function createAgentGraph(options = {}) {
1415
1458
  resultText = JSON.stringify(result, null, 2);
1416
1459
  } else if (server === 'runtime') {
1417
1460
  resultText = await handleRuntimeControlTool(state.session, tool, args);
1461
+ if (tool === 'delegate' && /^Runtime control error \(delegate\):/i.test(resultText)) {
1462
+ terminalFailure = resultText
1463
+ .replace(/^Runtime control error \(delegate\):\s*/i, '')
1464
+ .replace(/^Delegation failed during objective_resolution:\s*/i, '');
1465
+ ok = false;
1466
+ }
1418
1467
  } else if (server !== 'shell') {
1419
1468
  await awaitRunApproval(state.session, { runId, tool: toolName });
1420
1469
  await awaitToolApproval(state.session, {
@@ -1510,17 +1559,36 @@ export function createAgentGraph(options = {}) {
1510
1559
  tool_call_id: call.id,
1511
1560
  content: boundedResult,
1512
1561
  });
1562
+ if (terminalFailure) break;
1513
1563
  }
1514
1564
 
1565
+ if (terminalFailure) {
1566
+ const response = `Action non lancée : ${terminalFailure}`;
1567
+ emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: response });
1568
+ return {
1569
+ response,
1570
+ messages: toolResultMessages,
1571
+ pendingToolCalls: null,
1572
+ forceDelegation: false,
1573
+ terminalToolFailure: true,
1574
+ invalidToolCallRetries: 0,
1575
+ invalidResponseRetries: 0,
1576
+ };
1577
+ }
1515
1578
  return {
1516
1579
  messages: toolResultMessages,
1517
1580
  pendingToolCalls: null,
1518
1581
  forceDelegation: false,
1519
1582
  invalidToolCallRetries: 0,
1520
1583
  invalidResponseRetries: 0,
1584
+ terminalToolFailure: false,
1521
1585
  };
1522
1586
  }
1523
1587
 
1588
+ function routeToolExecutor(state) {
1589
+ return state.terminalToolFailure ? END : 'orchestrator';
1590
+ }
1591
+
1524
1592
  function routeOrchestrator(state) {
1525
1593
  if (state.pendingToolCalls?.length > 0) return 'tool_executor';
1526
1594
  if (state.retryWithoutTool) return 'orchestrator';
@@ -1534,7 +1602,7 @@ export function createAgentGraph(options = {}) {
1534
1602
  .addNode('tool_executor', toolExecutorNode)
1535
1603
  .addEdge(START, 'orchestrator')
1536
1604
  .addConditionalEdges('orchestrator', routeOrchestrator)
1537
- .addEdge('tool_executor', 'orchestrator')
1605
+ .addConditionalEdges('tool_executor', routeToolExecutor)
1538
1606
  .compile();
1539
1607
 
1540
1608
  // LangGraph's default recursionLimit is 25 super-steps. Each tool round
@@ -25,7 +25,9 @@ test('Donna cannot answer an explicit action with manual instructions instead of
25
25
  runtime: { url: 'http://runtime.test' },
26
26
  llm: {
27
27
  async completeWithTools({ tools }) {
28
- if (tools.length === 0) return { content: '{"action":true}', message: { role: 'assistant', content: '{"action":true}' }, tool_calls: null };
28
+ if (tools.some((tool) => tool.function?.name === 'classify_action_request')) {
29
+ return { content: null, message: { role: 'assistant', content: null }, tool_calls: [{ id: 'classify', type: 'function', function: { name: 'classify_action_request', arguments: '{"action":true}' } }] };
30
+ }
29
31
  mainCalls += 1;
30
32
  if (mainCalls === 1) {
31
33
  return {
@@ -910,8 +912,8 @@ test('forced delegation is cleared after one valid tool call and does not loop',
910
912
  commands: ['status'],
911
913
  llm: {
912
914
  async completeWithTools({ toolChoice, tools }) {
913
- if (tools.length === 0) {
914
- return { content: '{"action":true}', message: { role: 'assistant', content: '{"action":true}' }, tool_calls: null };
915
+ if (tools.some((tool) => tool.function?.name === 'classify_action_request')) {
916
+ return { content: null, message: { role: 'assistant', content: null }, tool_calls: [{ id: 'classify', type: 'function', function: { name: 'classify_action_request', arguments: '{"action":true}' } }] };
915
917
  }
916
918
  calls += 1;
917
919
  choices.push(toolChoice);
@@ -946,6 +948,44 @@ test('forced delegation is cleared after one valid tool call and does not loop',
946
948
  }
947
949
  });
948
950
 
951
+ test('a rejected runtime delegation is terminal and never loops', async () => {
952
+ const originalFetch = globalThis.fetch;
953
+ globalThis.fetch = async () => ({
954
+ ok: false,
955
+ status: 422,
956
+ json: async () => ({
957
+ error: 'Delegation failed during objective_resolution: No orchestrable capability is currently available.',
958
+ }),
959
+ });
960
+ let calls = 0;
961
+ const session = sessionBase({
962
+ runtime: { url: 'http://runtime.test' },
963
+ llm: {
964
+ async completeWithTools() {
965
+ calls += 1;
966
+ return {
967
+ content: null,
968
+ message: { role: 'assistant', content: null },
969
+ tool_calls: [{
970
+ id: 'delegate-failure',
971
+ type: 'function',
972
+ function: { name: 'runtime__delegate', arguments: '{"objective":"Lance ingestion"}' },
973
+ }],
974
+ };
975
+ },
976
+ },
977
+ });
978
+
979
+ try {
980
+ const result = await createAgentGraph().invoke({ input: 'lance ingestion', session });
981
+ assert.equal(calls, 1);
982
+ assert.equal(result.response, 'Action non lancée : No orchestrable capability is currently available.');
983
+ assert.equal(result.terminalToolFailure, true);
984
+ } finally {
985
+ globalThis.fetch = originalFetch;
986
+ }
987
+ });
988
+
949
989
  // Guard: the system prompt must never show a connected tool's bare name
950
990
  // outside its qualified server__tool form. Bare mentions are what teach the
951
991
  // model to emit unqualified tool calls (the cme_status incident). The bare
@@ -8,6 +8,7 @@ import { createAgentGraph } from '../agent/graph.js';
8
8
  import { handleSlashCommand, printHelp, printVersion, refreshMcpRuntimeStatus } from '../commands/slash.js';
9
9
  import { runShell, runHeadlessChatTurn } from '../shell/repl.js';
10
10
  import { runPreflightChecks, withRuntimePreflight } from '../core/startupCheck.js';
11
+ import { refreshRunningContainers } from '../core/wikiSetup.js';
11
12
  import { applySessionWikircProfile } from '../core/sessionConfig.js';
12
13
  import { listWikircProfiles } from '../core/wikirc.js';
13
14
  import { callMcpTool, formatMcpToolResult, readChatAccessConfig } from '../core/mcp.js';
@@ -17,6 +18,7 @@ import { createAgentEvent, dispatchAgentEvent, reduceAgentEvents } from '../core
17
18
  import { runAgentTurn, runAgenticLoop } from '../core/agentLoop.js';
18
19
  import { resolveCapabilityConcurrency } from '../orchestrator/scheduler.js';
19
20
  import { capabilityRegistryForSession } from '../orchestrator/capabilityRegistry.js';
21
+ import { listWorkspaces } from '../core/workspaces.js';
20
22
  // Runtime modules use node:sqlite (Node.js built-in unavailable in Bun).
21
23
  // They are imported dynamically so the shell / TUI path never loads them.
22
24
 
@@ -422,7 +424,7 @@ async function runHeadless(argv, agent) {
422
424
  let input = prompt;
423
425
  if (skillName) {
424
426
  const skillResult = await handleSlashCommand(`/skills run ${skillName}`, { packageJson, session, onStep: step });
425
- if (skillResult.output) log.push(skillResult.output);
427
+ if (skillResult.output && !skillResult.rawOutput) log.push(skillResult.output);
426
428
  if (String(skillResult.output ?? '').startsWith('Skill not found')) throw new Error(`Skill not found: ${skillName}`);
427
429
  input = skillResult.agentTrigger
428
430
  ? [
@@ -501,7 +503,7 @@ async function runRuntime(argv, agent) {
501
503
  const { defaultRuntimeStateDir, openRuntimeStore, RECOVERABLE_QUEUE_STATUSES } = await import('../runtime/store.js');
502
504
  const { startRuntimeServer } = await import('../runtime/server.js');
503
505
  const { recoverActiveRuns } = await import('../runtime/recoveryManager.js');
504
- const { emitRuntimeLog, startActivitySupervisor, cancelActiveActivityJobs } = await import('../runtime/supervisor.js');
506
+ const { emitRuntimeLog, startActivitySupervisor, cancelActiveActivityJobs, discoverAgentsOnce } = await import('../runtime/supervisor.js');
505
507
  const { resolveRuntimeAuthToken } = await import('../runtime/auth.js');
506
508
  const { createSqliteQueueStore } = await import('../runtime/queueStore.js');
507
509
  const { createApprovalManager } = await import('../runtime/approvals.js');
@@ -803,6 +805,12 @@ async function runRuntime(argv, agent) {
803
805
  const { resolveObjective } = await import('../orchestrator/objectiveResolver.js');
804
806
  const { validateFragment } = await import('../orchestrator/planValidator.js');
805
807
  const session = context.session;
808
+ // The supervisor starts discovery asynchronously. A delegation submitted
809
+ // immediately after opening ShellUI must not observe the transient empty
810
+ // registry and fail while the provider is already healthy. Refresh the
811
+ // live endpoints and await one discovery pass before resolving.
812
+ await refreshMcpRuntimeStatus(session);
813
+ await discoverAgentsOnce(session, { registry: session.agentRegistry });
806
814
  let selection;
807
815
  try {
808
816
  selection = await resolveObjective(objective, session);
@@ -1162,10 +1170,22 @@ async function runRuntime(argv, agent) {
1162
1170
  await new Promise(() => {});
1163
1171
  }
1164
1172
 
1173
+ // One place for the skipped-image-update warnings so the runtime and TUI
1174
+ // startup paths report refresh failures identically.
1175
+ function logImageRefreshErrors(imageRefresh) {
1176
+ for (const error of imageRefresh?.errors ?? []) {
1177
+ console.warn(`[wiki-manager] image update skipped: ${error}`);
1178
+ }
1179
+ }
1180
+
1165
1181
  export async function runCli(argv) {
1166
1182
  if (argv[0] === 'runtime') {
1167
1183
  const scaffolded = ensureManagerScaffold({ log: (message) => console.log(`[wiki-manager] ${message}`) });
1168
1184
  if (scaffolded.length > 0) loadManagerEnv();
1185
+ const imageRefresh = await refreshRunningContainers({
1186
+ onStep: (message) => console.log(`[wiki-manager] ${message}`),
1187
+ });
1188
+ logImageRefreshErrors(imageRefresh);
1169
1189
  const agent = createAgentGraph();
1170
1190
  await runRuntime(argv.slice(1), agent);
1171
1191
  return;
@@ -1218,6 +1238,10 @@ export async function runCli(argv) {
1218
1238
  if (!process.versions.bun) {
1219
1239
  throw new Error('Interactive TUI requires Bun. Run: bun ./bin/wiki-manager.js');
1220
1240
  }
1241
+ const initialWorkspaceName = valueAfter(argv, '--workspace');
1242
+ if (initialWorkspaceName && !listWorkspaces().some((workspace) => workspace.name === initialWorkspaceName)) {
1243
+ throw new Error(`Workspace not found: ${initialWorkspaceName}`);
1244
+ }
1221
1245
  const { runOpenTuiShell, runStartupWizard } = await import('../shell/tui.tsx');
1222
1246
  // Fresh directory → copy mcp.endpoints.json/.env from the packaged
1223
1247
  // examples so external agents (cme, mailer, documents) connect out of
@@ -1242,6 +1266,17 @@ export async function runCli(argv) {
1242
1266
  // configuration. Re-read everything before drawing the home screen.
1243
1267
  preflight = await runPreflightChecks();
1244
1268
  }
1269
+ const dockerReady = preflight.checks.some((check) => check.kind === 'docker' && check.ok);
1270
+ const internetReady = preflight.checks.some((check) => check.kind === 'internet' && check.ok);
1271
+ if (dockerReady && internetReady) {
1272
+ void refreshRunningContainers({
1273
+ onStep: (message) => console.log(`[wiki-manager] ${message}`),
1274
+ }).then((imageRefresh) => {
1275
+ logImageRefreshErrors(imageRefresh);
1276
+ }).catch((error) => {
1277
+ console.warn(`[wiki-manager] image update skipped: ${error instanceof Error ? error.message : String(error)}`);
1278
+ });
1279
+ }
1245
1280
  let runtime = null;
1246
1281
  try {
1247
1282
  const { ensureRuntime } = await import('../runtime/lifecycle.js');
@@ -1257,7 +1292,13 @@ export async function runCli(argv) {
1257
1292
  // (see tui.tsx onShellExit): render() resolves at MOUNT, so anything
1258
1293
  // after this await would run while the shell is still on screen —
1259
1294
  // 0.12.9 shipped exactly that bug and killed the runtime under the user.
1260
- await runOpenTuiShell({ agent, packageJson, runtime, preflight });
1295
+ await runOpenTuiShell({
1296
+ agent,
1297
+ packageJson,
1298
+ runtime,
1299
+ preflight,
1300
+ initialWorkspaceName,
1301
+ });
1261
1302
  return;
1262
1303
  }
1263
1304
 
@@ -391,7 +391,10 @@ function skillDetailText(skill) {
391
391
 
392
392
  function buildSkillRunPrompt(skill) {
393
393
  return [
394
- `Execute the "${skill.name}" skill for the current workspace.`,
394
+ `The user asked to run the "${skill.name}" skill for the current workspace.`,
395
+ 'First explain concisely, in the user language, what will be launched and its intended outcome.',
396
+ 'Do not quote, reproduce, or display the raw skill content.',
397
+ 'Then execute the workflow, using the available tools when required.',
395
398
  'Follow the workflow steps below. Call MCP tools and shell commands as needed for each step.',
396
399
  'Report progress as you go. Ask for confirmation before irreversible or costly actions not already defined in the skill.',
397
400
  '',
@@ -413,7 +416,11 @@ function skillActionCommand(session, action, name) {
413
416
  return { output: `Skill not found: ${name}.${hint}` };
414
417
  }
415
418
  if (action === 'run') {
416
- return { output: `Skill: ${skill.name} — launching…`, agentTrigger: buildSkillRunPrompt(skill) };
419
+ return {
420
+ output: JSON.stringify({ operation: 'run-skill', skill: skill.name }),
421
+ rawOutput: true,
422
+ agentTrigger: buildSkillRunPrompt(skill),
423
+ };
417
424
  }
418
425
  return { output: skillDetailText(skill) };
419
426
  }
@@ -603,7 +610,7 @@ Options:
603
610
  --cacert <path> Trust a local CA; Docker must be able to read this host path
604
611
  --once <prompt> Run one agent turn and exit
605
612
  --headless Run a workspace task non-interactively
606
- --workspace <name> Workspace for --headless
613
+ --workspace <name> Initial workspace (interactive or --headless)
607
614
  --skill <name> Skill to run in --headless (implies --wait)
608
615
  --prompt <text> Task or extra instruction for --headless
609
616
  --log-file <path> Optional headless log path
@@ -93,6 +93,30 @@ test('/new without a name shows usage', async () => {
93
93
  assert.match(result.output ?? '', /Usage/i);
94
94
  });
95
95
 
96
+ test('/skills run sends the private skill body to Donna without rendering it as command output', async () => {
97
+ const root = await mkdtemp(join(tmpdir(), 'wiki-manager-skill-run-'));
98
+ const skillDir = join(root, '.wiki', 'skills');
99
+ mkdirSync(skillDir, { recursive: true });
100
+ writeFileSync(join(skillDir, 'pipeline.md'), [
101
+ '---',
102
+ 'name: pipeline',
103
+ 'description: Build the deliverables',
104
+ '---',
105
+ 'SECRET WORKFLOW BODY',
106
+ '',
107
+ ].join('\n'), 'utf8');
108
+
109
+ const result = await handleSlashCommand('/skills run pipeline', {
110
+ packageJson: { version: 'test' },
111
+ session: { workspacePath: root },
112
+ });
113
+
114
+ assert.equal(result.rawOutput, true);
115
+ assert.doesNotMatch(result.output, /SECRET WORKFLOW BODY/);
116
+ assert.match(result.agentTrigger, /SECRET WORKFLOW BODY/);
117
+ assert.match(result.agentTrigger, /Do not quote, reproduce, or display the raw skill content/);
118
+ });
119
+
96
120
  test('/use loads only workspaces and /config use switches wikirc profiles', async () => {
97
121
  const root = await mkdtemp(join(tmpdir(), 'wiki-manager-use-profile-'));
98
122
  const registryRoot = join(root, 'registry');
@@ -31,6 +31,9 @@ function normalizePlanSteps(steps) {
31
31
  executor: s?.executor ?? null,
32
32
  executorQuery: s?.executorQuery ?? null,
33
33
  outputRefs: Array.isArray(s?.outputRefs) ? s.outputRefs.map(String) : [],
34
+ ...(s?.status != null ? { status: String(s.status) } : {}),
35
+ ...(s?.startedAt != null ? { startedAt: s.startedAt } : {}),
36
+ ...(s?.finishedAt != null ? { finishedAt: s.finishedAt } : {}),
34
37
  }));
35
38
  }
36
39
 
@@ -149,6 +152,7 @@ function productionActivityFromPayload(payload, context = {}) {
149
152
  step,
150
153
  ...(payload?.taskId ? { stepId: String(payload.taskId) } : {}),
151
154
  },
155
+ plan: Array.isArray(progress?.steps) ? { steps: progress.steps } : null,
152
156
  poll: jobId ? {
153
157
  server: 'production',
154
158
  tool: context.tool === 'agent_status' ? 'agent_status' : 'production_job_status',
@@ -109,6 +109,37 @@ test('extractActivity: documents conversion activity carries plan and percent',
109
109
  assert.deepEqual(activity.plan.steps.map((step) => step.id), ['resolve', 'convert', 'write']);
110
110
  });
111
111
 
112
+ test('extractActivity: agent_status carries structured processing progress', () => {
113
+ const activity = extractActivity({
114
+ jobId: 'job-build',
115
+ taskId: 'build-report',
116
+ operation: 'build',
117
+ status: 'running',
118
+ progress: {
119
+ percent: 47,
120
+ phase: 'build',
121
+ detail: 'Batch 2/4 · LLM throttled',
122
+ stepIndex: 1,
123
+ stepTotal: 2,
124
+ steps: [
125
+ { id: 'build', name: 'build', status: 'running' },
126
+ { id: 'export', name: 'export', status: 'pending' },
127
+ ],
128
+ batch: { index: 2, total: 4, status: 'running' },
129
+ throttling: { active: true, waitMs: 1200, retryAt: '2026-07-21T10:00:01Z' },
130
+ processing: { instructionCount: 8 },
131
+ },
132
+ }, { server: 'production', tool: 'agent_status' });
133
+
134
+ assert.deepEqual(activity.plan.steps.map((step) => [step.id, step.status]), [
135
+ ['build', 'running'],
136
+ ['export', 'pending'],
137
+ ]);
138
+ assert.equal(activity.progress.batch.index, 2);
139
+ assert.equal(activity.progress.throttling.active, true);
140
+ assert.equal(activity.progress.processing.instructionCount, 8);
141
+ });
142
+
112
143
  test('rememberActivity: returns normalized activity on success', () => {
113
144
  const session = {};
114
145
  const result = rememberActivity(session, { id: '1', status: 'running', source: 'x', kind: 'job' });