@dotdrelle/wiki-manager 0.12.10 → 0.12.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.env.example CHANGED
@@ -29,8 +29,10 @@ WORKSPACES_ROOT=/path/to/workspaces
29
29
  CME_MCP_AUTH_TOKEN=
30
30
  DOCUMENTS_MCP_AUTH_TOKEN=
31
31
 
32
- # ── Mailer (MailerSend) ────────────────────────────────────────────────────────
32
+ # ── Agent ports (optional, change only if defaults conflict) ───────────────────
33
33
 
34
+ # CME_MCP_PORT=3336
35
+ # DOCUMENTS_MCP_PORT=3337
34
36
 
35
37
  # ── Documents LLM OCR / Mermaid (optional) ─────────────────────────────────────
36
38
 
@@ -46,6 +48,12 @@ DOCUMENTS_MCP_AUTH_TOKEN=
46
48
  # lives HERE and is referenced from mcp.endpoints.json as ${VAR_NAME}.
47
49
  # Add your own entries when you declare additional endpoints.
48
50
 
51
+ # ── Orchestration (optional) ───────────────────────────────────────────────────
52
+ # Parallel tasks dispatched at once for capability runs. Defaults to what the
53
+ # agent itself declares (agent_describe limits); set only to constrain it.
54
+ # Example constraint (never raises an agent's declared capacity):
55
+ # WIKI_MANAGER_CAPABILITY_CONCURRENCY=3
56
+
49
57
  # ── MCP retry policy (optional) ────────────────────────────────────────────────
50
58
 
51
59
  # Tool calls are retried on transient HTTP/MCP errors before the run fails.
@@ -65,8 +73,3 @@ DOCUMENTS_MCP_AUTH_TOKEN=
65
73
  # Runtime approvals can pause runs or protected tools until /approve is called.
66
74
  # WIKI_MANAGER_APPROVAL_TIMEOUT_MS=600000
67
75
  # WIKI_MANAGER_REQUIRE_APPROVAL_TOOLS=production.production_start_job
68
-
69
- # ── Agent ports (optional, change only if defaults conflict) ───────────────────
70
-
71
- # CME_MCP_PORT=3336
72
- # DOCUMENTS_MCP_PORT=3337
package/README.md CHANGED
@@ -48,7 +48,7 @@ Use a local installation when the manager should be pinned in a project's
48
48
 
49
49
  ```bash
50
50
  npm install @dotdrelle/wiki-manager
51
- npm approve-scripts bun # only when npm reports that Bun's postinstall is pending
51
+ npm approve-scripts bun@1.3.14 # only when npm reports that Bun's postinstall is pending
52
52
  npx wiki-manager
53
53
  npx wiki-workspace --help
54
54
  ```
@@ -57,8 +57,8 @@ npx wiki-workspace --help
57
57
  global `PATH`. Run local executables with `npx` (or `npm exec wiki-manager` and
58
58
  `npm exec wiki-workspace`). Bun is installed automatically as a package runtime;
59
59
  you do not need to add `~/.bun/bin` to `PATH`. Recent npm versions may require
60
- the explicit `npm approve-scripts bun` security approval shown above before the
61
- first launch.
60
+ the explicit `npm approve-scripts bun@1.3.14` security approval shown above
61
+ before the first launch.
62
62
 
63
63
  ### Global installation
64
64
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dotdrelle/wiki-manager",
3
- "version": "0.12.10",
3
+ "version": "0.12.12",
4
4
  "description": "Agentic shell and orchestration cockpit for llm-wiki workspaces.",
5
5
  "license": "PolyForm-Noncommercial-1.0.0",
6
6
  "author": "dotrelle",
@@ -27,7 +27,7 @@ const MAX_PROFILE_CHARS = 4000;
27
27
  const INTERNAL_TOOL_SERVERS = {
28
28
  wiki: ['plan_set', 'plan_done'],
29
29
  shell: ['run_command', 'read_command', 'profile_update'],
30
- runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue'],
30
+ runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue', 'start_capability_run'],
31
31
  };
32
32
 
33
33
  const AGENT_SLASH_COMMANDS = new Set([
@@ -165,6 +165,24 @@ const RUNTIME_ENQUEUE_TOOL = {
165
165
  },
166
166
  };
167
167
 
168
+ const RUNTIME_CAPABILITY_RUN_TOOL = {
169
+ type: 'function',
170
+ function: {
171
+ name: 'runtime__start_capability_run',
172
+ description: 'Start a DETERMINISTIC orchestrated run for a discovered capability: the runtime asks the capable agent for its task graph (agent_plan), validates and integrates it, creates the approval requests, and dispatches the tasks in parallel. Use this whenever the user asks for multi-document or multi-step work covered by a known capability (e.g. ingest everything pending → capability "knowledge.update", operation "ingest"). Do NOT call production tools directly for such requests.',
173
+ parameters: {
174
+ type: 'object',
175
+ additionalProperties: false,
176
+ properties: {
177
+ capability: { type: 'string', description: 'Capability id from the known list (e.g. knowledge.update).' },
178
+ operation: { type: 'string', description: 'Operation supported by the capability (e.g. ingest, build).' },
179
+ inputs: { type: 'array', items: { type: 'string' }, description: 'Optional file subset; omit to cover everything pending.' },
180
+ },
181
+ required: ['capability'],
182
+ },
183
+ },
184
+ };
185
+
168
186
  const WIKI_PLAN_SET_TOOL = {
169
187
  type: 'function',
170
188
  function: {
@@ -250,6 +268,7 @@ const AgentState = Annotation.Root({
250
268
  readyToStream: Annotation(),
251
269
  streamContext: Annotation(),
252
270
  streamedInline: Annotation(),
271
+ retryWithoutTool: Annotation({ default: () => false }),
253
272
  });
254
273
 
255
274
  function commandList(session) {
@@ -521,6 +540,33 @@ export function knownCapabilityIds(session) {
521
540
  }))].sort();
522
541
  }
523
542
 
543
+ // Deterministic fragment→plan mapping, shared by the agent_plan tool bridge
544
+ // and the /ingest-style direct capability runs: the parallel path must not
545
+ // depend on an LLM copying fields correctly.
546
+ export function planStepsFromFragment(payload) {
547
+ const tasks = Array.isArray(payload?.tasks) ? payload.tasks : [];
548
+ return tasks.map((task, index) => normalizeDeclaredPlanStep({
549
+ id: task.id,
550
+ description: task.label ?? task.id ?? `Task ${index + 1}`,
551
+ requiredCapability: task.requiredCapability ?? payload.capability ?? null,
552
+ operation: task.operation ?? null,
553
+ arguments: task.arguments ?? {},
554
+ dependsOn: task.dependsOn ?? [],
555
+ outputRefs: (task.expectedOutputRefs ?? []).map((ref) => (ref && typeof ref === 'object' ? String(ref.ref ?? '') : String(ref))).filter(Boolean),
556
+ groupId: task.groupId ?? null,
557
+ dependsOnGroup: task.dependsOnGroup ?? null,
558
+ parallelizable: task.parallelizable,
559
+ barrier: task.barrier,
560
+ locks: task.locks,
561
+ requiresApproval: task.requiresApproval,
562
+ approvalClass: task.approvalClass,
563
+ approvalSummary: task.approvalSummary,
564
+ idempotencyKey: task.idempotencyKey,
565
+ progressWeight: task.progressWeight,
566
+ recommendedConcurrency: task.recommendedConcurrency,
567
+ }, index));
568
+ }
569
+
524
570
  async function handleRuntimeControlTool(session, tool, args = {}) {
525
571
  const url = session.runtime?.url ?? null;
526
572
  if (!url) return 'Runtime not connected: no runtime URL available in this session.';
@@ -538,6 +584,23 @@ async function handleRuntimeControlTool(session, tool, args = {}) {
538
584
  const result = await postRuntimeControl('message', { url, workspace, input: 'approve', intent: 'approve' });
539
585
  return String(result?.explanation ?? (result?.accepted ? 'Approval granted.' : 'No pending approval found.'));
540
586
  }
587
+ if (tool === 'start_capability_run') {
588
+ const { postRuntimeRun } = await import('../runtime/client.js');
589
+ const result = await postRuntimeRun(`Run de capability ${args.capability}${args.operation ? ` (${args.operation})` : ''} demandé par Donna.`, {
590
+ url,
591
+ workspace,
592
+ capabilityPlan: {
593
+ capability: String(args.capability ?? ''),
594
+ ...(args.operation ? { operation: String(args.operation) } : {}),
595
+ ...(Array.isArray(args.inputs) && args.inputs.length > 0 ? { inputs: args.inputs.map(String) } : {}),
596
+ // No concurrency dictated here: the runtime reads the provider's
597
+ // own declared capacity (agent_describe limits).
598
+ },
599
+ });
600
+ return result?.runId
601
+ ? `Run de capability accepté (${String(result.runId).slice(0, 8)}) : le plan sera intégré et dispatché en parallèle ; une approbation sera demandée avant les mutations (l'utilisateur peut dire « valide tout » ou taper /approve). Suis la progression dans Activity.`
602
+ : `Run non démarré : ${result?.explanation ?? result?.error ?? JSON.stringify(result)}`;
603
+ }
541
604
  if (tool === 'enqueue') {
542
605
  const result = await postRuntimeControl('message', { url, workspace, input: String(args.input ?? ''), intent: 'enqueue' });
543
606
  return String(result?.explanation ?? 'Request queued for after the current run.');
@@ -688,8 +751,12 @@ export function buildAgentSystemPrompt(state) {
688
751
  skills,
689
752
  'You can call MCP tools directly using the provided tool functions.',
690
753
  'When the user asks for an action that can be performed with connected MCP tools or safe primitives, do not answer with future intent such as "I will call...", "I am going to run...", or "launching..." unless you also call the tool in the same turn. Either call the tool now, ask for the exact missing required arguments, or explain the concrete blocker.',
754
+ 'Execution truthfulness: never invent a job id, status, percentage, duration, generated file, file content, URL, command, or tool result. An action is executed only when you call an available tool and receive its result. Examples and placeholders are forbidden in execution reports.',
755
+ 'After any completed action, give a short factual summary based only on the tool result: outcome and concrete outputs or references actually returned. Mention a viewing primitive only when it exists in Available primitives and is relevant. Do not interpret generated content, propose verification checklists, invent next steps, or suggest commands unless the user explicitly asks.',
756
+ 'When calling a tool, emit no preliminary narration. Call it directly; the PLAN and Activity panels show progress. After completion, keep the final response concise and proportional to the result.',
691
757
  'For connector configuration/setup/update requests, if a matching setup/configuration tool is connected and the required arguments are known, call it immediately. If the connector or tool is not connected, say which concrete capability is missing and recommend the exact service/status primitive to inspect it. Do not invent a pending connector action in plain text.',
692
758
  'For workspace-scoped external MCP tools, the orchestrator enforces workspace injection. Use the active workspace for configuration, source, import, export, conversion, and generation tools unless a tool is explicitly job-scoped and only requires a job id.',
759
+ 'Dynamic capability inputs are owned by the agent that provides the capability. To answer which inputs are pending or available, call the qualified `<provider>__agent_status` tool with {capability, operation} and report only its pendingInputs. Never infer this state from /uploads or a hardcoded workspace directory. If no capable agent/status tool is connected, say that the pending inputs cannot be determined.',
693
760
  'You can call shell__run_command for safe manager slash commands such as /workspace list, /workspace init <name> [path], /use <workspace>, /config, /status, /services, /skills, /skills show <name>, and /skills run <name>.',
694
761
  'Skills are workflow instructions, not executable code. When a user asks to run a skill, inspect it, propose the concrete primitive/tool plan, and ask for confirmation before costly or mutating actions.',
695
762
  [
@@ -733,7 +800,7 @@ export function buildAgentSystemPrompt(state) {
733
800
  'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`cme__cme_export_run`, then `cme__cme_export_status`). Never use production `type=export` for Confluence source export.',
734
801
  'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`production__production_start_job` with `type:"export"` or pipeline steps). Require the deliverable path when exporting deliverables.',
735
802
  'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts.',
736
- 'MULTI-DOCUMENT ingest (more than 2 files, or "ingest everything pending"): call production__agent_plan first, e.g. {capability:"knowledge.update", operation:"ingest", constraints:{maxConcurrency:3, requireApprovalForMutations:true}}. The shell integrates the returned task graph as the plan automatically and the orchestrator dispatches the per-document tasks IN PARALLEL with an approval gate. Do not call production__production_start_job for multi-document ingest — that creates one monolithic sequential job.',
803
+ 'MULTI-DOCUMENT ingest (more than 2 files, or "ingest everything pending") and any multi-step capability work: when runtime__start_capability_run is available, call it (e.g. {capability:"knowledge.update", operation:"ingest"}) — the runtime integrates the agent task graph deterministically and dispatches IN PARALLEL with an approval gate. Inside a runtime run (no runtime tools), call production__agent_plan instead; the shell integrates the fragment automatically. Never call production__production_start_job for multi-document ingest — that creates one monolithic sequential job.',
737
804
  'Single-document ingest or one-off jobs (doctor, one build, one export): production__production_start_job is fine. To chain sequential steps (e.g. build then polish) use ONE call with type="pipeline" and steps=["build","polish"] — never separate jobs (the first is asynchronous). For existing deliverables where content stability matters, pass stabilize:true. Do not ask the user to confirm between steps.',
738
805
  'Long-running MCP jobs: do not call the same status tool more than once consecutively. When chaining jobs sequentially: (1) start the job, report job/activity id and status; (2) check status once — if done, proceed to the next step immediately; (3) if still running, report status, list the remaining steps, and return control; (4) when re-invoked, check status first, then proceed. Do not spin-poll (status → status → status with no new action between). The shell activity panel monitors non-terminal jobs automatically.',
739
806
  'If production__production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
@@ -820,6 +887,9 @@ function toolsForClassification(classification, writeTools, session = null) {
820
887
  const controlTools = session?.runtime?.url
821
888
  ? [RUNTIME_STATUS_TOOL, RUNTIME_CANCEL_TOOL, RUNTIME_KILL_TOOL, RUNTIME_APPROVE_TOOL, RUNTIME_ENQUEUE_TOOL]
822
889
  : [];
890
+ const capabilityRunTools = session?.runtime?.url && !classification.activeRun
891
+ ? [RUNTIME_CAPABILITY_RUN_TOOL]
892
+ : [];
823
893
  if (classification.activeRun && ['converse', 'observe', 'ambiguous', 'approve', 'cancel', 'enqueue_run'].includes(classification.kind)) {
824
894
  // During an active run Donna gets read + profile + the runtime control
825
895
  // suite: she can answer, approve, enqueue for later, soft-cancel or
@@ -827,7 +897,7 @@ function toolsForClassification(classification, writeTools, session = null) {
827
897
  // what runtime__enqueue is for). No canned regex answers anywhere.
828
898
  return [SHELL_READ_COMMAND_TOOL, SHELL_PROFILE_UPDATE_TOOL, ...controlTools];
829
899
  }
830
- return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...writeTools];
900
+ return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...writeTools];
831
901
  }
832
902
 
833
903
  export function createAgentGraph(options = {}) {
@@ -892,12 +962,14 @@ export function createAgentGraph(options = {}) {
892
962
 
893
963
  try {
894
964
  const useStreamWithTools = typeof llm.streamWithTools === 'function';
965
+ const suppressExecutionNarration = runtimeExecution && (iterations === 0 || state.retryWithoutTool);
895
966
  const result = useStreamWithTools
896
967
  ? await llm.streamWithTools({
897
968
  system,
898
969
  tools,
899
970
  messages: conversationMessages,
900
971
  onTextDelta: (delta) => {
972
+ if (suppressExecutionNarration) return;
901
973
  emitAgentEvent(state.session, 'assistant_delta', 'llm', { delta });
902
974
  state.session._onStream?.(delta);
903
975
  },
@@ -930,6 +1002,39 @@ export function createAgentGraph(options = {}) {
930
1002
  toolIterations: iterations + 1,
931
1003
  readyToStream: false,
932
1004
  inputClassification: classification,
1005
+ retryWithoutTool: false,
1006
+ };
1007
+ }
1008
+
1009
+ if (runtimeExecution && iterations === 0 && !state.retryWithoutTool) {
1010
+ state.session._onStreamReset?.();
1011
+ state.session._onStep?.('Agent: action response rejected — no tool was called; retrying…');
1012
+ return {
1013
+ pendingToolCalls: null,
1014
+ messages: [
1015
+ { role: 'user', content: state.input },
1016
+ result.message ?? { role: 'assistant', content: result.content ?? '' },
1017
+ {
1018
+ role: 'user',
1019
+ content: 'Your previous response did not execute the requested action because it called no tool. Do not narrate or simulate execution. Call the appropriate available tool now. Never invent results.',
1020
+ },
1021
+ ],
1022
+ toolIterations: 1,
1023
+ readyToStream: false,
1024
+ inputClassification: classification,
1025
+ retryWithoutTool: true,
1026
+ };
1027
+ }
1028
+
1029
+ if (runtimeExecution && state.retryWithoutTool) {
1030
+ state.session._onStreamReset?.();
1031
+ const failure = 'Action non exécutée : Donna n’a appelé aucun outil disponible. Aucun job ni résultat n’a été créé.';
1032
+ emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: failure });
1033
+ return {
1034
+ response: failure,
1035
+ pendingToolCalls: null,
1036
+ readyToStream: false,
1037
+ retryWithoutTool: false,
933
1038
  };
934
1039
  }
935
1040
 
@@ -1068,26 +1173,7 @@ export function createAgentGraph(options = {}) {
1068
1173
  // wiki__plan_set would lose fields (a small local model dropped
1069
1174
  // arguments/operations in testing) — the shell does the mapping.
1070
1175
  if (/(^|__)agent_plan$/.test(tool) && Array.isArray(payload?.tasks) && payload.tasks.length > 0) {
1071
- const steps = payload.tasks.map((task, index) => normalizeDeclaredPlanStep({
1072
- id: task.id,
1073
- description: task.label ?? task.id ?? `Task ${index + 1}`,
1074
- requiredCapability: task.requiredCapability ?? payload.capability ?? null,
1075
- operation: task.operation ?? null,
1076
- arguments: task.arguments ?? {},
1077
- dependsOn: task.dependsOn ?? [],
1078
- outputRefs: (task.expectedOutputRefs ?? []).map((ref) => (ref && typeof ref === 'object' ? String(ref.ref ?? '') : String(ref))).filter(Boolean),
1079
- groupId: task.groupId ?? null,
1080
- dependsOnGroup: task.dependsOnGroup ?? null,
1081
- parallelizable: task.parallelizable,
1082
- barrier: task.barrier,
1083
- locks: task.locks,
1084
- requiresApproval: task.requiresApproval,
1085
- approvalClass: task.approvalClass,
1086
- approvalSummary: task.approvalSummary,
1087
- idempotencyKey: task.idempotencyKey,
1088
- progressWeight: task.progressWeight,
1089
- recommendedConcurrency: task.recommendedConcurrency,
1090
- }, index, state.session));
1176
+ const steps = planStepsFromFragment(payload);
1091
1177
  emitAgentEvent(state.session, 'plan_set', 'tool', { steps });
1092
1178
  state.session._onStep?.(`Plan: ${steps.length} task(s) declared from ${server} fragment`);
1093
1179
  resultText = `Task-graph fragment integrated as the current plan (${steps.length} task(s), groups: ${[...new Set(payload.tasks.map((task) => task.groupId).filter(Boolean))].join(', ') || 'none'}). The orchestrator will dispatch these tasks — do NOT call production tools for them yourself. Reply with a short summary and wait.`;
@@ -1163,6 +1249,7 @@ export function createAgentGraph(options = {}) {
1163
1249
 
1164
1250
  function routeOrchestrator(state) {
1165
1251
  if (state.pendingToolCalls?.length > 0) return 'tool_executor';
1252
+ if (state.retryWithoutTool) return 'orchestrator';
1166
1253
  return END;
1167
1254
  }
1168
1255
 
@@ -686,6 +686,61 @@ test('agent graph survives more than 12 tool iterations (recursion limit)', asyn
686
686
  }
687
687
  });
688
688
 
689
+ test('runtime action retries a text-only hallucination and requires a real tool call', async () => {
690
+ const originalFetch = globalThis.fetch;
691
+ globalThis.fetch = async () => ({
692
+ ok: true,
693
+ status: 200,
694
+ headers: { get: () => null },
695
+ text: async () => JSON.stringify({ result: { content: [{ type: 'text', text: '{"ok":true,"outputs":["deliverables/result.md"]}' }] } }),
696
+ });
697
+ let calls = 0;
698
+ let retryMessages = [];
699
+ const session = sessionBase({
700
+ _currentRunIdentity: { runId: 'run-build', turnId: 'run-build:turn-1', workspace: 'docs' },
701
+ llm: {
702
+ async completeWithTools({ messages }) {
703
+ calls += 1;
704
+ if (calls === 1) {
705
+ return {
706
+ content: 'Build terminé, faux-job-123, rapport.pdf.',
707
+ message: { role: 'assistant', content: 'Build terminé, faux-job-123, rapport.pdf.' },
708
+ tool_calls: null,
709
+ };
710
+ }
711
+ if (calls === 2) {
712
+ retryMessages = messages;
713
+ return {
714
+ content: null,
715
+ message: { role: 'assistant', content: null },
716
+ tool_calls: [{
717
+ id: 'build-call',
718
+ type: 'function',
719
+ function: { name: 'production__production_start_job', arguments: '{"type":"build"}' },
720
+ }],
721
+ };
722
+ }
723
+ return {
724
+ content: 'Build terminé. Résultat : deliverables/result.md. Disponible dans /openui.',
725
+ message: { role: 'assistant', content: 'Build terminé. Résultat : deliverables/result.md. Disponible dans /openui.' },
726
+ tool_calls: null,
727
+ };
728
+ },
729
+ },
730
+ });
731
+
732
+ try {
733
+ const result = await createAgentGraph().invoke({ input: 'lance le build', session });
734
+ assert.equal(calls, 3);
735
+ assert.match(retryMessages.at(-1).content, /called no tool/);
736
+ assert.doesNotMatch(result.response, /faux-job-123|rapport\.pdf/);
737
+ assert.match(result.response, /deliverables\/result\.md/);
738
+ assert.equal(session.headlessPlan?.[0]?.status, 'done');
739
+ } finally {
740
+ globalThis.fetch = originalFetch;
741
+ }
742
+ });
743
+
689
744
  test('agent graph auto-declares the plan from an agent_plan task-graph fragment', async () => {
690
745
  // The bridge that makes parallel ingestion real: when the LLM calls
691
746
  // production__agent_plan, the shell integrates the fragment as the plan
@@ -790,3 +845,83 @@ test('Donna interprets a cleanup request and calls runtime__kill herself', async
790
845
  globalThis.fetch = originalFetch;
791
846
  }
792
847
  });
848
+
849
+ test('buildAgentSystemPrompt delegates pending inputs to the capability provider without filesystem fallback', () => {
850
+ const workspacePath = mkdtempSync(join(tmpdir(), 'facts-'));
851
+ try {
852
+ mkdirSync(join(workspacePath, 'raw', 'untracked'), { recursive: true });
853
+ writeFileSync(join(workspacePath, 'raw', 'untracked', 'note-a.md'), '# a\n');
854
+ const prompt = buildAgentSystemPrompt({ session: sessionBase({ workspacePath }) });
855
+ assert.match(prompt, /call the qualified `<provider>__agent_status` tool with \{capability, operation\}/);
856
+ assert.match(prompt, /report only its pendingInputs/);
857
+ assert.doesNotMatch(prompt, /note-a\.md/);
858
+ assert.doesNotMatch(prompt, /Workspace facts:/);
859
+ } finally {
860
+ rmSync(workspacePath, { recursive: true, force: true });
861
+ }
862
+ });
863
+
864
+ test('Donna reads pending inputs from the qualified capability provider status tool', async () => {
865
+ const originalFetch = globalThis.fetch;
866
+ let requestedArguments = null;
867
+ globalThis.fetch = async (_url, options) => {
868
+ const request = JSON.parse(options.body);
869
+ requestedArguments = request.params.arguments;
870
+ return {
871
+ ok: true,
872
+ status: 200,
873
+ headers: { get: () => null },
874
+ text: async () => JSON.stringify({ result: { content: [{
875
+ type: 'text',
876
+ text: JSON.stringify({
877
+ contractVersion: '1',
878
+ agentInstanceId: 'production-main',
879
+ capability: 'knowledge.update',
880
+ operation: 'ingest',
881
+ available: true,
882
+ pendingInputs: [{ type: 'file', ref: 'provider/source-a', label: 'source-a.md', mediaType: 'text/markdown' }],
883
+ }),
884
+ }] } }),
885
+ };
886
+ };
887
+ let calls = 0;
888
+ const session = sessionBase({
889
+ mcp: {
890
+ production: {
891
+ status: 'connected',
892
+ url: 'http://127.0.0.1:3000/mcp/',
893
+ tools: [{
894
+ name: 'agent_status',
895
+ description: 'Read a job or discover capability inputs.',
896
+ inputSchema: { type: 'object', properties: { capability: { type: 'string' }, operation: { type: 'string' } } },
897
+ }],
898
+ },
899
+ },
900
+ llm: {
901
+ async completeWithTools({ messages }) {
902
+ calls += 1;
903
+ if (calls === 1) {
904
+ return {
905
+ content: null,
906
+ message: { role: 'assistant', content: null },
907
+ tool_calls: [{
908
+ id: 'status-call',
909
+ type: 'function',
910
+ function: { name: 'production__agent_status', arguments: JSON.stringify({ capability: 'knowledge.update', operation: 'ingest' }) },
911
+ }],
912
+ };
913
+ }
914
+ assert.match(JSON.stringify(messages), /provider\/source-a/);
915
+ return { content: 'Un fichier est en attente : source-a.md.', message: { role: 'assistant', content: 'Un fichier est en attente : source-a.md.' }, tool_calls: null };
916
+ },
917
+ },
918
+ });
919
+
920
+ try {
921
+ const result = await createAgentGraph().invoke({ input: 'Quels fichiers sont en attente d’ingestion ?', session });
922
+ assert.equal(result.response, 'Un fichier est en attente : source-a.md.');
923
+ assert.deepEqual(requestedArguments, { capability: 'knowledge.update', operation: 'ingest' });
924
+ } finally {
925
+ globalThis.fetch = originalFetch;
926
+ }
927
+ });
@@ -15,6 +15,7 @@ import { extractActivity, parseJsonText, sessionActivities, terminalFailures } f
15
15
  import { syncActivitiesToPlan, formatPlanStatus } from '../core/plan.js';
16
16
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
17
17
  import { runAgentTurn, runAgenticLoop } from '../core/agentLoop.js';
18
+ import { resolveCapabilityConcurrency } from '../orchestrator/scheduler.js';
18
19
  // Runtime modules use node:sqlite (Node.js built-in unavailable in Bun).
19
20
  // They are imported dynamically so the shell / TUI path never loads them.
20
21
 
@@ -628,6 +629,71 @@ async function runRuntime(argv, agent) {
628
629
  : undefined;
629
630
  supervisor?.setRunSignal(signal);
630
631
  session._onStep = (message) => emitRuntimeLog(session, message);
632
+ // Deterministic capability run (/ingest): ask the capable agent for its
633
+ // task-graph fragment and integrate it as the plan BEFORE any LLM turn.
634
+ // The parallel path must not depend on a small model deciding to call
635
+ // agent_plan by itself.
636
+ if (body.capabilityPlan?.capability) {
637
+ const { validateFragment } = await import('../orchestrator/planValidator.js');
638
+ const { integrate } = await import('../orchestrator/planIntegrator.js');
639
+ const registry = session.capabilityRegistry ?? null;
640
+ const agents = session.agentRegistry?.snapshot?.() ?? session.agentRegistrySnapshot ?? [];
641
+ const provider = agents.find((item) => (item.description?.capabilities ?? [])
642
+ .some((capability) => capability.id === body.capabilityPlan.capability));
643
+ if (!provider?.serverName) {
644
+ throw new Error(`No agent provides capability ${body.capabilityPlan.capability}.`);
645
+ }
646
+ const fragment = parseJsonText(formatMcpToolResult(await callMcpTool(session.mcp, provider.serverName, 'agent_plan', {
647
+ capability: body.capabilityPlan.capability,
648
+ operation: body.capabilityPlan.operation ?? undefined,
649
+ workspace: { revision: String(Date.now()) },
650
+ constraints: {
651
+ // The agent declares its capacity. Request/env values are only
652
+ // constraints: they may lower that capacity, never raise it.
653
+ maxConcurrency: resolveCapabilityConcurrency(
654
+ provider,
655
+ body.capabilityPlan.maxConcurrency,
656
+ process.env.WIKI_MANAGER_CAPABILITY_CONCURRENCY,
657
+ ),
658
+ requireApprovalForMutations: body.capabilityPlan.requireApproval !== false,
659
+ },
660
+ ...(Array.isArray(body.capabilityPlan.inputs) && body.capabilityPlan.inputs.length > 0
661
+ ? { arguments: { inputs: body.capabilityPlan.inputs } }
662
+ : {}),
663
+ })));
664
+ if (!Array.isArray(fragment?.tasks) || fragment.tasks.length === 0) {
665
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', {
666
+ origin: 'runtime',
667
+ runId,
668
+ payload: { content: `Aucune tâche à planifier pour ${body.capabilityPlan.capability} (${fragment?.summary?.initialSynthesis?.[0] ?? 'fragment vide'}).` },
669
+ }));
670
+ dispatchAgentEvent(session, createAgentEvent('run_done', { origin: 'runtime', runId, payload: { runId } }));
671
+ return;
672
+ }
673
+ // Full official integration path — NOT a bare plan_set: integrate()
674
+ // validates the fragment, persists the tasks, and CREATES the
675
+ // approval requests the scheduler's approvalCovered() filter waits
676
+ // for. A bare plan_set left requiresApproval tasks unreachable
677
+ // forever (stalled as no_ready_plan_task with nothing to approve).
678
+ const validation = validateFragment(fragment, {
679
+ registry,
680
+ run: { plannerAgentInstanceId: provider.agentInstanceId ?? provider.serverName },
681
+ });
682
+ if (!validation.ok) {
683
+ throw new Error(`Capability plan rejected: ${validation.errors.map((error) => error.message ?? error.code ?? String(error)).join('; ')}`);
684
+ }
685
+ const integrated = integrate(runId, validation.normalizedFragment, {
686
+ registry,
687
+ session,
688
+ store,
689
+ workspace: session.workspace ?? null,
690
+ enforceApprovalCoverage: true,
691
+ });
692
+ if (!integrated.ok) {
693
+ throw new Error(`Capability plan integration failed: ${(integrated.errors ?? []).map((error) => error.message ?? error.code ?? String(error)).join('; ')}`);
694
+ }
695
+ emitRuntimeLog(session, `capability-plan: ${fragment.tasks.length} task(s) integrated from ${provider.serverName}.agent_plan (${body.capabilityPlan.capability}); approvals: ${(session.agentProjection?.approvals ?? []).filter((approval) => approval.status === 'pending_approval').length} pending`);
696
+ }
631
697
  await runRuntimeAgenticWorkflow(agent, session, input, {
632
698
  signal,
633
699
  timeoutMs,
@@ -811,11 +877,11 @@ export async function runCli(argv) {
811
877
  runtime = unavailableRuntime(err);
812
878
  console.error(`Runtime unavailable: ${runtime.error}`);
813
879
  }
880
+ // The owned-runtime shutdown happens inside the TUI's own exit paths
881
+ // (see tui.tsx onShellExit): render() resolves at MOUNT, so anything
882
+ // after this await would run while the shell is still on screen —
883
+ // 0.12.9 shipped exactly that bug and killed the runtime under the user.
814
884
  await runOpenTuiShell({ agent, packageJson, runtime });
815
- if (runtime?.url) {
816
- const { shutdownOwnedRuntime } = await import('../runtime/lifecycle.js');
817
- await shutdownOwnedRuntime(runtime, { log: (message) => console.log(`[wiki-manager] ${message}`) });
818
- }
819
885
  return;
820
886
  }
821
887
 
@@ -43,7 +43,7 @@ import {
43
43
  listDocumentUploads,
44
44
  storeAndMaybeConvertDocument,
45
45
  } from '../core/documentIntake.js';
46
- import { fetchRuntimeState, postRuntimeCancel, postRuntimeKill } from '../runtime/client.js';
46
+ import { fetchRuntimeState, postRuntimeCancel, postRuntimeControl, postRuntimeKill, postRuntimeRun } from '../runtime/client.js';
47
47
  import { versionWithBuild } from '../core/buildInfo.js';
48
48
 
49
49
  export function printVersion(packageJson) {
@@ -513,9 +513,6 @@ export async function refreshMcpRuntimeStatus(session) {
513
513
 
514
514
  async function statusText(session) {
515
515
  const states = await refreshMcpRuntimeStatus(session);
516
- const services = session.workspacePath
517
- ? await composeServices(session).catch(() => [])
518
- : [];
519
516
  const workspaceStats = collectWorkspaceStats(session);
520
517
  const workspaceColumn = sectionBlock('Workspace', [
521
518
  `workspace: ${session.workspace ?? '-'}`,
@@ -530,9 +527,6 @@ async function statusText(session) {
530
527
  `model: ${session.wikircConfig?.llm?.model ?? '-'}`,
531
528
  `baseUrl: ${session.wikircConfig?.llm?.baseUrl ?? '-'}`,
532
529
  ]);
533
- const servicesColumn = sectionBlock('Services', services.length > 0
534
- ? services.map((service) => `- ${service}`)
535
- : ['No workspace loaded.']);
536
530
  const runtimeColumn = sectionBlock('Runtime', (states ? serviceStatesText(states) : 'Docker runtime not available or no workspace loaded.').split('\n'));
537
531
  const mcpColumn = sectionBlock('MCP', formatMcpStatus(session.mcp).split('\n'));
538
532
  const mcpToolsColumn = sectionBlock('MCP tool summary', formatMcpToolSummary(session.mcp).split('\n'));
@@ -542,7 +536,7 @@ async function statusText(session) {
542
536
  '',
543
537
  workspaceStatsText(workspaceStats),
544
538
  '',
545
- twoColumns(servicesColumn, runtimeColumn),
539
+ runtimeColumn,
546
540
  '',
547
541
  twoColumns(mcpColumn, mcpToolsColumn),
548
542
  ].join('\n');
@@ -635,6 +629,8 @@ ${helpPair('/wiki', 'Run wiki index', '/wiki run <args>', 'Raw wiki CLI')}
635
629
  ${helpPair('/chat', 'Chat mode', '/agent', 'Agent mode')}
636
630
  ${helpPair('/openui', 'Open web UI in browser', '', '')}
637
631
  ${helpPair('/run status', 'Runtime status', '/run kill', 'Kill runtime run(s)')}
632
+ ${helpPair('/run capability <id>', 'Deterministic capability run', '/approve', 'Grant pending approval')}
633
+ ${helpPair('/cancel', 'Cancel active run', '', '')}
638
634
  ${helpPair('/run cancel', 'Cancel active run', '', '')}
639
635
  ${helpPair('/queue', 'MCP job queue', '/queue clear', 'Clear finished')}
640
636
  ${helpPair('/queue cancel <id>', 'Cancel queued/running', '', '')}
@@ -972,6 +968,27 @@ export async function handleSlashCommand(line, context) {
972
968
  }
973
969
  return { output: 'Usage: /mcp <status|endpoints|tools|call> [mcp]' };
974
970
  }
971
+ case 'cancel': {
972
+ // Alias of /run cancel — people type /cancel when they want out.
973
+ const runtime = context.runtime ?? {};
974
+ if (!runtime.url) return { output: 'Runtime unavailable. Start/connect the runtime before using /cancel.' };
975
+ const result = await postRuntimeCancel({ url: runtime.url, workspace: context.session.workspace ?? null });
976
+ return { output: result.cancelled ? 'Runtime cancel requested.' : `Nothing to cancel${result.reason ? ` (${result.reason})` : ''} — use /run kill to purge everything.` };
977
+ }
978
+ case 'approve': {
979
+ // /approve was only wired in the legacy REPL — in the opentui TUI it
980
+ // returned "Unknown command", which made every approval time out and
981
+ // every requiresApproval plan stall forever.
982
+ const runtime = context.runtime ?? {};
983
+ if (!runtime.url) return { output: 'Runtime unavailable. Start/connect the runtime before using /approve.' };
984
+ const result = await postRuntimeControl('message', {
985
+ url: runtime.url,
986
+ workspace: context.session.workspace ?? null,
987
+ input: args.slice(1).join(' ') || 'approve',
988
+ intent: 'approve',
989
+ });
990
+ return { output: String(result?.explanation ?? (result?.accepted ? 'Approval granted.' : 'No pending approval found.')) };
991
+ }
975
992
  case 'run': {
976
993
  const subcommand = args[1] ?? 'status';
977
994
  const runtime = context.runtime ?? {};
@@ -985,11 +1002,34 @@ export async function handleSlashCommand(line, context) {
985
1002
  const result = await postRuntimeCancel({ url, workspace: context.session.workspace ?? null });
986
1003
  return { output: result.cancelled ? 'Runtime cancel requested.' : `Runtime cancel skipped: ${result.reason ?? 'no active run'}` };
987
1004
  }
1005
+ if (subcommand === 'capability') {
1006
+ // Business-agnostic deterministic run: mirrors the capability
1007
+ // registry instead of hardcoding an application verb. The agent's
1008
+ // task graph is validated/integrated server-side before any LLM turn.
1009
+ const capability = args[2];
1010
+ if (!capability) return { output: 'Usage: /run capability <capability-id> [operation] [files…]' };
1011
+ if (!context.session.workspace) return { output: 'No workspace loaded. Use /use <workspace> first.' };
1012
+ const operation = args[3] && !args[3].includes('.') && !args[3].includes('/') ? args[3] : undefined;
1013
+ const inputs = args.slice(operation ? 4 : 3);
1014
+ const result = await postRuntimeRun(`Run de capability ${capability}${operation ? ` (${operation})` : ''} demandé via /run capability.`, {
1015
+ url,
1016
+ workspace: context.session.workspace,
1017
+ capabilityPlan: {
1018
+ capability,
1019
+ ...(operation ? { operation } : {}),
1020
+ ...(inputs.length > 0 ? { inputs } : {}),
1021
+ },
1022
+ });
1023
+ if (result?.runId) {
1024
+ return { output: `▶ Run de capability accepté (${String(result.runId).slice(0, 8)}) — le plan de l'agent sera intégré et dispatché en parallèle ; approbation demandée avant les mutations (« valide tout » ou /approve).` };
1025
+ }
1026
+ return { output: `Run non démarré: ${result?.explanation ?? result?.error ?? JSON.stringify(result)}` };
1027
+ }
988
1028
  if (subcommand === 'kill') {
989
1029
  const result = await postRuntimeKill({ url, workspace: context.session.workspace ?? null, runId: args[2] ?? null });
990
1030
  return { output: `Runtime kill requested: ${result.runs ?? 0} run${result.runs === 1 ? '' : 's'}, ${result.tasks ?? 0} task${result.tasks === 1 ? '' : 's'} cancelled.` };
991
1031
  }
992
- return { output: 'Usage: /run [status|cancel|kill [runId]]' };
1032
+ return { output: 'Usage: /run [status|cancel|kill [runId]|capability <id> [operation] [files…]]' };
993
1033
  }
994
1034
  case 'queue': {
995
1035
  const subcommand = args[1] ?? 'list';