@dotdrelle/wiki-manager 0.12.10 → 0.12.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.env.example CHANGED
@@ -29,8 +29,10 @@ WORKSPACES_ROOT=/path/to/workspaces
29
29
  CME_MCP_AUTH_TOKEN=
30
30
  DOCUMENTS_MCP_AUTH_TOKEN=
31
31
 
32
- # ── Mailer (MailerSend) ────────────────────────────────────────────────────────
32
+ # ── Agent ports (optional, change only if defaults conflict) ───────────────────
33
33
 
34
+ # CME_MCP_PORT=3336
35
+ # DOCUMENTS_MCP_PORT=3337
34
36
 
35
37
  # ── Documents LLM OCR / Mermaid (optional) ─────────────────────────────────────
36
38
 
@@ -65,8 +67,3 @@ DOCUMENTS_MCP_AUTH_TOKEN=
65
67
  # Runtime approvals can pause runs or protected tools until /approve is called.
66
68
  # WIKI_MANAGER_APPROVAL_TIMEOUT_MS=600000
67
69
  # WIKI_MANAGER_REQUIRE_APPROVAL_TOOLS=production.production_start_job
68
-
69
- # ── Agent ports (optional, change only if defaults conflict) ───────────────────
70
-
71
- # CME_MCP_PORT=3336
72
- # DOCUMENTS_MCP_PORT=3337
package/README.md CHANGED
@@ -48,7 +48,7 @@ Use a local installation when the manager should be pinned in a project's
48
48
 
49
49
  ```bash
50
50
  npm install @dotdrelle/wiki-manager
51
- npm approve-scripts bun # only when npm reports that Bun's postinstall is pending
51
+ npm approve-scripts bun@1.3.14 # only when npm reports that Bun's postinstall is pending
52
52
  npx wiki-manager
53
53
  npx wiki-workspace --help
54
54
  ```
@@ -57,8 +57,8 @@ npx wiki-workspace --help
57
57
  global `PATH`. Run local executables with `npx` (or `npm exec wiki-manager` and
58
58
  `npm exec wiki-workspace`). Bun is installed automatically as a package runtime;
59
59
  you do not need to add `~/.bun/bin` to `PATH`. Recent npm versions may require
60
- the explicit `npm approve-scripts bun` security approval shown above before the
61
- first launch.
60
+ the explicit `npm approve-scripts bun@1.3.14` security approval shown above
61
+ before the first launch.
62
62
 
63
63
  ### Global installation
64
64
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dotdrelle/wiki-manager",
3
- "version": "0.12.10",
3
+ "version": "0.12.11",
4
4
  "description": "Agentic shell and orchestration cockpit for llm-wiki workspaces.",
5
5
  "license": "PolyForm-Noncommercial-1.0.0",
6
6
  "author": "dotrelle",
@@ -250,6 +250,7 @@ const AgentState = Annotation.Root({
250
250
  readyToStream: Annotation(),
251
251
  streamContext: Annotation(),
252
252
  streamedInline: Annotation(),
253
+ retryWithoutTool: Annotation({ default: () => false }),
253
254
  });
254
255
 
255
256
  function commandList(session) {
@@ -688,6 +689,9 @@ export function buildAgentSystemPrompt(state) {
688
689
  skills,
689
690
  'You can call MCP tools directly using the provided tool functions.',
690
691
  'When the user asks for an action that can be performed with connected MCP tools or safe primitives, do not answer with future intent such as "I will call...", "I am going to run...", or "launching..." unless you also call the tool in the same turn. Either call the tool now, ask for the exact missing required arguments, or explain the concrete blocker.',
692
+ 'Execution truthfulness: never invent a job id, status, percentage, duration, generated file, file content, URL, command, or tool result. An action is executed only when you call an available tool and receive its result. Examples and placeholders are forbidden in execution reports.',
693
+ 'After any completed action, give a short factual summary based only on the tool result: outcome and concrete outputs or references actually returned. Mention a viewing primitive only when it exists in Available primitives and is relevant. Do not interpret generated content, propose verification checklists, invent next steps, or suggest commands unless the user explicitly asks.',
694
+ 'When calling a tool, emit no preliminary narration. Call it directly; the PLAN and Activity panels show progress. After completion, keep the final response concise and proportional to the result.',
691
695
  'For connector configuration/setup/update requests, if a matching setup/configuration tool is connected and the required arguments are known, call it immediately. If the connector or tool is not connected, say which concrete capability is missing and recommend the exact service/status primitive to inspect it. Do not invent a pending connector action in plain text.',
692
696
  'For workspace-scoped external MCP tools, the orchestrator enforces workspace injection. Use the active workspace for configuration, source, import, export, conversion, and generation tools unless a tool is explicitly job-scoped and only requires a job id.',
693
697
  'You can call shell__run_command for safe manager slash commands such as /workspace list, /workspace init <name> [path], /use <workspace>, /config, /status, /services, /skills, /skills show <name>, and /skills run <name>.',
@@ -892,12 +896,14 @@ export function createAgentGraph(options = {}) {
892
896
 
893
897
  try {
894
898
  const useStreamWithTools = typeof llm.streamWithTools === 'function';
899
+ const suppressExecutionNarration = runtimeExecution && (iterations === 0 || state.retryWithoutTool);
895
900
  const result = useStreamWithTools
896
901
  ? await llm.streamWithTools({
897
902
  system,
898
903
  tools,
899
904
  messages: conversationMessages,
900
905
  onTextDelta: (delta) => {
906
+ if (suppressExecutionNarration) return;
901
907
  emitAgentEvent(state.session, 'assistant_delta', 'llm', { delta });
902
908
  state.session._onStream?.(delta);
903
909
  },
@@ -930,6 +936,39 @@ export function createAgentGraph(options = {}) {
930
936
  toolIterations: iterations + 1,
931
937
  readyToStream: false,
932
938
  inputClassification: classification,
939
+ retryWithoutTool: false,
940
+ };
941
+ }
942
+
943
+ if (runtimeExecution && iterations === 0 && !state.retryWithoutTool) {
944
+ state.session._onStreamReset?.();
945
+ state.session._onStep?.('Agent: action response rejected — no tool was called; retrying…');
946
+ return {
947
+ pendingToolCalls: null,
948
+ messages: [
949
+ { role: 'user', content: state.input },
950
+ result.message ?? { role: 'assistant', content: result.content ?? '' },
951
+ {
952
+ role: 'user',
953
+ content: 'Your previous response did not execute the requested action because it called no tool. Do not narrate or simulate execution. Call the appropriate available tool now. Never invent results.',
954
+ },
955
+ ],
956
+ toolIterations: 1,
957
+ readyToStream: false,
958
+ inputClassification: classification,
959
+ retryWithoutTool: true,
960
+ };
961
+ }
962
+
963
+ if (runtimeExecution && state.retryWithoutTool) {
964
+ state.session._onStreamReset?.();
965
+ const failure = 'Action non exécutée : Donna n’a appelé aucun outil disponible. Aucun job ni résultat n’a été créé.';
966
+ emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: failure });
967
+ return {
968
+ response: failure,
969
+ pendingToolCalls: null,
970
+ readyToStream: false,
971
+ retryWithoutTool: false,
933
972
  };
934
973
  }
935
974
 
@@ -1163,6 +1202,7 @@ export function createAgentGraph(options = {}) {
1163
1202
 
1164
1203
  function routeOrchestrator(state) {
1165
1204
  if (state.pendingToolCalls?.length > 0) return 'tool_executor';
1205
+ if (state.retryWithoutTool) return 'orchestrator';
1166
1206
  return END;
1167
1207
  }
1168
1208
 
@@ -686,6 +686,61 @@ test('agent graph survives more than 12 tool iterations (recursion limit)', asyn
686
686
  }
687
687
  });
688
688
 
689
+ test('runtime action retries a text-only hallucination and requires a real tool call', async () => {
690
+ const originalFetch = globalThis.fetch;
691
+ globalThis.fetch = async () => ({
692
+ ok: true,
693
+ status: 200,
694
+ headers: { get: () => null },
695
+ text: async () => JSON.stringify({ result: { content: [{ type: 'text', text: '{"ok":true,"outputs":["deliverables/result.md"]}' }] } }),
696
+ });
697
+ let calls = 0;
698
+ let retryMessages = [];
699
+ const session = sessionBase({
700
+ _currentRunIdentity: { runId: 'run-build', turnId: 'run-build:turn-1', workspace: 'docs' },
701
+ llm: {
702
+ async completeWithTools({ messages }) {
703
+ calls += 1;
704
+ if (calls === 1) {
705
+ return {
706
+ content: 'Build terminé, faux-job-123, rapport.pdf.',
707
+ message: { role: 'assistant', content: 'Build terminé, faux-job-123, rapport.pdf.' },
708
+ tool_calls: null,
709
+ };
710
+ }
711
+ if (calls === 2) {
712
+ retryMessages = messages;
713
+ return {
714
+ content: null,
715
+ message: { role: 'assistant', content: null },
716
+ tool_calls: [{
717
+ id: 'build-call',
718
+ type: 'function',
719
+ function: { name: 'production__production_start_job', arguments: '{"type":"build"}' },
720
+ }],
721
+ };
722
+ }
723
+ return {
724
+ content: 'Build terminé. Résultat : deliverables/result.md. Disponible dans /openui.',
725
+ message: { role: 'assistant', content: 'Build terminé. Résultat : deliverables/result.md. Disponible dans /openui.' },
726
+ tool_calls: null,
727
+ };
728
+ },
729
+ },
730
+ });
731
+
732
+ try {
733
+ const result = await createAgentGraph().invoke({ input: 'lance le build', session });
734
+ assert.equal(calls, 3);
735
+ assert.match(retryMessages.at(-1).content, /called no tool/);
736
+ assert.doesNotMatch(result.response, /faux-job-123|rapport\.pdf/);
737
+ assert.match(result.response, /deliverables\/result\.md/);
738
+ assert.equal(session.headlessPlan?.[0]?.status, 'done');
739
+ } finally {
740
+ globalThis.fetch = originalFetch;
741
+ }
742
+ });
743
+
689
744
  test('agent graph auto-declares the plan from an agent_plan task-graph fragment', async () => {
690
745
  // The bridge that makes parallel ingestion real: when the LLM calls
691
746
  // production__agent_plan, the shell integrates the fragment as the plan
@@ -513,9 +513,6 @@ export async function refreshMcpRuntimeStatus(session) {
513
513
 
514
514
  async function statusText(session) {
515
515
  const states = await refreshMcpRuntimeStatus(session);
516
- const services = session.workspacePath
517
- ? await composeServices(session).catch(() => [])
518
- : [];
519
516
  const workspaceStats = collectWorkspaceStats(session);
520
517
  const workspaceColumn = sectionBlock('Workspace', [
521
518
  `workspace: ${session.workspace ?? '-'}`,
@@ -530,9 +527,6 @@ async function statusText(session) {
530
527
  `model: ${session.wikircConfig?.llm?.model ?? '-'}`,
531
528
  `baseUrl: ${session.wikircConfig?.llm?.baseUrl ?? '-'}`,
532
529
  ]);
533
- const servicesColumn = sectionBlock('Services', services.length > 0
534
- ? services.map((service) => `- ${service}`)
535
- : ['No workspace loaded.']);
536
530
  const runtimeColumn = sectionBlock('Runtime', (states ? serviceStatesText(states) : 'Docker runtime not available or no workspace loaded.').split('\n'));
537
531
  const mcpColumn = sectionBlock('MCP', formatMcpStatus(session.mcp).split('\n'));
538
532
  const mcpToolsColumn = sectionBlock('MCP tool summary', formatMcpToolSummary(session.mcp).split('\n'));
@@ -542,7 +536,7 @@ async function statusText(session) {
542
536
  '',
543
537
  workspaceStatsText(workspaceStats),
544
538
  '',
545
- twoColumns(servicesColumn, runtimeColumn),
539
+ runtimeColumn,
546
540
  '',
547
541
  twoColumns(mcpColumn, mcpToolsColumn),
548
542
  ].join('\n');
@@ -66,6 +66,7 @@ export async function runAgenticLoop(agent, session, initialInput, {
66
66
  onPendingSteps = null,
67
67
  onActivitiesStarted = null,
68
68
  onActivitiesCompleted = null,
69
+ deterministicTerminalSummary = false,
69
70
  onMaxTurns = null,
70
71
  abortMessage = 'Agent run cancelled.',
71
72
  parallelHandoff = false,
@@ -149,6 +150,11 @@ export async function runAgenticLoop(agent, session, initialInput, {
149
150
  const completed = waitResult.completed ?? [];
150
151
  const summary = formatCompletedActivities(completed);
151
152
  onActivitiesCompleted?.({ completed, summary });
153
+ const unfinished = (session.headlessPlan ?? []).some((step) =>
154
+ ['pending', 'pending_approval', 'running', 'starting', 'queued'].includes(String(step.status ?? '').toLowerCase()));
155
+ if (deterministicTerminalSummary && !unfinished) {
156
+ return { ok: true, completed, summary, deterministicSummary: true };
157
+ }
152
158
  if (parallelHandoff && readyPlanTasks(session.headlessPlan).length > 1) {
153
159
  return { ok: true, handoff: true };
154
160
  }
@@ -77,8 +77,56 @@ test('runAgenticLoop waits for new activities and continues with a completion su
77
77
  assert.equal(result.ok, true);
78
78
  assert.equal(inputs.length, 2);
79
79
  assert.match(inputs[1], /Completed activities:/);
80
- assert.match(inputs[1], /production job-1: done/);
81
- assert.deepEqual(callbacks, ['started:1', '- production job-1: done']);
80
+ assert.match(inputs[1], /- job: done/);
81
+ assert.deepEqual(callbacks, ['started:1', '- job: done']);
82
+ });
83
+
84
+ test('runAgenticLoop can finish from terminal activity facts without another LLM turn', async () => {
85
+ const session = { activities: {}, headlessPlan: null };
86
+ let turns = 0;
87
+ const result = await runAgenticLoop({
88
+ async invoke({ session: turnSession }) {
89
+ turns += 1;
90
+ dispatchAgentEvent(turnSession, createAgentEvent('activity_upserted', {
91
+ payload: {
92
+ activity: {
93
+ id: 'job-build',
94
+ source: 'production',
95
+ kind: 'build',
96
+ label: 'Build workspace',
97
+ status: 'running',
98
+ terminal: false,
99
+ },
100
+ },
101
+ }));
102
+ return { response: 'Job started.' };
103
+ },
104
+ }, session, 'Build workspace', {
105
+ maxTurns: 3,
106
+ timeoutMs: 1000,
107
+ deterministicTerminalSummary: true,
108
+ waitForActivities: async (turnSession) => {
109
+ dispatchAgentEvent(turnSession, createAgentEvent('activity_upserted', {
110
+ payload: {
111
+ activity: {
112
+ id: 'job-build',
113
+ source: 'production',
114
+ kind: 'build',
115
+ label: 'Build workspace',
116
+ status: 'done',
117
+ terminal: true,
118
+ outputRefs: ['deliverables/result.md'],
119
+ },
120
+ },
121
+ }));
122
+ return { ok: true, completed: Object.values(turnSession.activities) };
123
+ },
124
+ });
125
+
126
+ assert.equal(turns, 1);
127
+ assert.equal(result.deterministicSummary, true);
128
+ assert.match(result.summary, /build: done/);
129
+ assert.match(result.summary, /output: deliverables\/result\.md/);
82
130
  });
83
131
 
84
132
  test('runAgenticLoop extracts a fallback numbered plan from first response', async () => {
@@ -1,4 +1,4 @@
1
1
  {
2
- "version": "0.12.10",
3
- "commit": "e256e16"
2
+ "version": "0.12.11",
3
+ "commit": "d8eae0b"
4
4
  }
package/src/core/mcp.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import { existsSync, readFileSync } from 'node:fs';
2
2
  import { managerEnvFile, managerMcpEndpointsFile, readEnvFile } from './env.js';
3
3
 
4
- const WIKI_MANAGER_VERSION = '0.12.10';
4
+ const WIKI_MANAGER_VERSION = '0.12.11';
5
5
 
6
6
  function envValue(key) {
7
7
  const filePath = managerEnvFile();
package/src/core/plan.js CHANGED
@@ -141,10 +141,17 @@ export function formatConfigValue(value) {
141
141
  }
142
142
 
143
143
  export function formatCompletedActivities(activities) {
144
- return activities
145
- .filter((a) => a.terminal)
146
- .map((a) => `- ${a.source} ${a.id ?? a.kind}: ${a.status}${a.error ? ` (${a.error})` : ''}`)
147
- .join('\n');
144
+ const terminal = activities.filter((activity) => activity.terminal);
145
+ const lines = terminal.map((activity) => {
146
+ const label = activity.kind ?? activity.label ?? `${activity.source} ${activity.id ?? 'activity'}`;
147
+ return `- ${label}: ${activity.status}${activity.error ? ` (${activity.error})` : ''}`;
148
+ });
149
+ const outputs = [...new Set(terminal.flatMap((activity) => activity.outputRefs ?? []).map((ref) => {
150
+ if (ref && typeof ref === 'object') return String(ref.ref ?? ref.path ?? ref.url ?? '').trim();
151
+ return String(ref ?? '').trim();
152
+ }).filter(Boolean))];
153
+ if (outputs.length > 0) lines.push(...outputs.map((output) => `- output: ${output}`));
154
+ return lines.join('\n');
148
155
  }
149
156
 
150
157
  function findMatchingPlanStepByStructure(plan, activity) {
@@ -50,6 +50,7 @@ export async function runRuntimeAgenticLoop(agent, session, initialInput, { sign
50
50
  maxTurns,
51
51
  runId,
52
52
  parallelHandoff,
53
+ deterministicTerminalSummary: true,
53
54
  abortMessage: 'Runtime run cancelled.',
54
55
  waitForActivities: (turnSession, startedActivities, waitOptions) =>
55
56
  waitForRuntimeActivities(turnSession, startedActivities, { ...waitOptions, pollBusy }),
@@ -78,6 +79,14 @@ export async function runRuntimeAgenticLoop(agent, session, initialInput, { sign
78
79
  onActivitiesStarted: ({ activities }) => {
79
80
  emitRuntimeLog(session, `agentic-loop: ${activities.length} new activity(s), waiting`);
80
81
  },
82
+ onActivitiesCompleted: ({ summary }) => {
83
+ emitRuntimeLog(session, `agentic-loop: completed activities:\n${summary}`);
84
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', {
85
+ origin: 'runtime',
86
+ runId,
87
+ payload: { content: summary || 'Action terminée.' },
88
+ }));
89
+ },
81
90
  onMaxTurns: ({ maxTurns: totalTurns }) => {
82
91
  emitRuntimeLog(session, `agentic-loop: max turns (${totalTurns}) reached`);
83
92
  },
@@ -375,7 +375,10 @@ function conversationLines(messages: Array<{ role: string; content: string }>, c
375
375
  let inFence = false;
376
376
  const lines: Array<{ text: string; isCode: boolean }> = [];
377
377
  for (const line of raw.split('\n')) {
378
- if (/^(`{2,3}|~{2,3})/.test(line)) { inFence = !inFence; continue; }
378
+ // LLM responses often indent fenced blocks as part of a list. Accept
379
+ // leading whitespace so ```bash / ~~~ fences are rendered as code
380
+ // instead of leaking their Markdown markers into the conversation.
381
+ if (/^\s*(`{3,}|~{3,})/.test(line)) { inFence = !inFence; continue; }
379
382
  lines.push({ text: line, isCode: inFence });
380
383
  }
381
384
  return [