@dotdrelle/wiki-manager 0.14.13 → 0.14.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/.env.example +29 -4
  2. package/README.md +19 -0
  3. package/docker-compose.yml +6 -1
  4. package/package.json +1 -1
  5. package/src/activity/activityAggregator.js +50 -16
  6. package/src/activity/activityAggregator.test.js +67 -4
  7. package/src/agent/graph.js +79 -11
  8. package/src/agent/graph.test.js +43 -3
  9. package/src/cli/wiki-manager.js +74 -11
  10. package/src/cli/wiki-manager.test.js +40 -1
  11. package/src/commands/slash.js +10 -3
  12. package/src/commands/slash.test.js +24 -0
  13. package/src/core/buildInfo.json +2 -2
  14. package/src/core/dockerCompose.test.js +4 -0
  15. package/src/core/env.test.js +3 -0
  16. package/src/core/mcp.js +1 -1
  17. package/src/core/wikiSetup.js +35 -0
  18. package/src/core/wikiWorkspace.test.js +20 -0
  19. package/src/core/workflow.js +72 -0
  20. package/src/core/workflow.test.js +57 -0
  21. package/src/core/workspaces.js +10 -2
  22. package/src/orchestrator/dependencyResolver.js +19 -1
  23. package/src/orchestrator/objectiveResolver.js +24 -0
  24. package/src/orchestrator/objectiveResolver.test.js +23 -1
  25. package/src/orchestrator/scheduler.js +27 -7
  26. package/src/orchestrator/scheduler.test.js +45 -1
  27. package/src/runtime/auth.test.js +65 -1
  28. package/src/runtime/client.js +4 -0
  29. package/src/runtime/donna-contract.test.js +2 -0
  30. package/src/runtime/lifecycle.js +21 -12
  31. package/src/runtime/runner.js +141 -18
  32. package/src/runtime/runner.test.js +30 -0
  33. package/src/runtime/server.js +13 -2
  34. package/src/runtime/server.test.js +30 -1
  35. package/src/runtime/store.js +54 -0
  36. package/src/runtime/store.test.js +42 -0
  37. package/src/shell/FileEditorDialog.tsx +2 -2
  38. package/src/shell/LeftPane.tsx +60 -15
  39. package/src/shell/RightPane.tsx +168 -56
  40. package/src/shell/StartupScreen.tsx +3 -7
  41. package/src/shell/renderer.ts +1 -0
  42. package/src/shell/repl.js +7 -81
  43. package/src/shell/repl.test.js +141 -38
  44. package/src/shell/tui.tsx +65 -65
  45. package/src/shell/useSession.ts +92 -6
  46. package/wiki-workspace +28 -0
package/.env.example CHANGED
@@ -63,10 +63,35 @@ DOCUMENTS_MCP_AUTH_TOKEN=
63
63
  # Add your own entries when you declare additional endpoints.
64
64
 
65
65
  # ── Orchestration (optional) ───────────────────────────────────────────────────
66
- # Parallel tasks dispatched at once for capability runs. Defaults to what the
67
- # agent itself declares (agent_describe limits); set only to constrain it.
68
- # Example constraint (never raises an agent's declared capacity):
69
- # WIKI_MANAGER_CAPABILITY_CONCURRENCY=3
66
+ # The runtime runs on the host while `llm-wiki serve` runs in Docker and
67
+ # reaches it through host.docker.internal. Listen on all host interfaces so
68
+ # the container can connect; exposed runtimes are protected by the generated
69
+ # WIKI_MANAGER_RUNTIME_TOKEN.
70
+ #
71
+ # WIKI_MANAGER_RUNTIME_PORT=7788
72
+ # Set to 0 to skip pulling and renewing already-running containers at startup.
73
+ # WIKI_MANAGER_AUTO_UPDATE=1
74
+ # WIKI_MANAGER_RUNTIME_HOST=0.0.0.0
75
+
76
+
77
+ # ── Parallelism & throughput ───────────────────────────────────────────────────
78
+ # Effective concurrency = MIN(agent recommendedConcurrency, agent maxConcurrency,
79
+ # this ceiling, per-task limits). So the PRIMARY levers live on the production
80
+ # agent (PRODUCTION_RECOMMENDED_CONCURRENCY / PRODUCTION_MAX_CONCURRENCY); this
81
+ # manager variable can only LOWER the result, never raise it. Full explanation
82
+ # and low/high profiles in docs/configuration.md § "Parallelism & throughput".
83
+ #
84
+ # Manager ceiling — leave unset to let the agent decide. Set to constrain:
85
+ # WIKI_MANAGER_CAPABILITY_CONCURRENCY=4
86
+ #
87
+ # Production agent capacity (passed through by docker-compose). Intermediate
88
+ # defaults are 4/8 (effective ≈ 4 parallel). Profiles:
89
+ # low → PRODUCTION_RECOMMENDED_CONCURRENCY=2 PRODUCTION_MAX_CONCURRENCY=4
90
+ # high → PRODUCTION_RECOMMENDED_CONCURRENCY=8 PRODUCTION_MAX_CONCURRENCY=16
91
+ # The wiki LLM backend must accept this many concurrent requests, and
92
+ # ingest_apply stays serialized regardless (global workspace-write lock).
93
+ # PRODUCTION_RECOMMENDED_CONCURRENCY=4
94
+ # PRODUCTION_MAX_CONCURRENCY=8
70
95
 
71
96
  # ── MCP retry policy (optional) ────────────────────────────────────────────────
72
97
 
package/README.md CHANGED
@@ -550,6 +550,25 @@ or the shell command `/approve item <id>`. The approval timeout defaults to 10
550
550
  minutes and can be changed with `WIKI_MANAGER_APPROVAL_TIMEOUT_MS` or
551
551
  `approvalTimeoutMs` in the `/run` body.
552
552
 
553
+ Directly-launched capability runs (ingest, pipeline) now **wait for approval by
554
+ default** before their mutating tasks: reply "valide tout", run `/approve`, or
555
+ click Approve in either UI (Shell right-pane banner, or the `serve` banner above
556
+ the composer). Auto-approval only happens when the run is started with
557
+ `autoApprove: true` (headless/CI).
558
+
559
+ ### Parallelism & throughput
560
+
561
+ The number of tasks that run at once is `MIN(agent recommendedConcurrency, agent
562
+ maxConcurrency, WIKI_MANAGER_CAPABILITY_CONCURRENCY, per-task limits)` — a
563
+ minimum, so the manager ceiling can only lower it. The production agent ships
564
+ intermediate defaults (`PRODUCTION_RECOMMENDED_CONCURRENCY=4` /
565
+ `PRODUCTION_MAX_CONCURRENCY=8`, ≈ 4 parallel); locks then cap real parallelism
566
+ per phase (`ingest_apply` stays serial). The resolved value is shown in both
567
+ UIs' run summary and on the run node of the execution graph, with an amber
568
+ "(ceiling)" marker when the manager ceiling binds. Low/high profiles, the lock
569
+ model and the LLM-backend caveat are in
570
+ [docs/configuration.md § "Parallelism & throughput"](docs/configuration.md).
571
+
553
572
  While a run is active, `GET`/`POST /control` still answers without waiting for
554
573
  it to finish: `{"action":"status"}` returns the current run/plan/queue state,
555
574
  `{"action":"explain"}` adds a one-line plain-language summary, and
@@ -59,7 +59,7 @@ services:
59
59
  - DOCUMENT_MAX_UPLOAD_BYTES=${DOCUMENT_MAX_UPLOAD_BYTES:-52428800}
60
60
  - WIKI_MCP_PROXY_URL=http://host.docker.internal:${WIKI_MCP_PORT:-3101}/mcp
61
61
  - PRODUCTION_MCP_PROXY_URL=http://host.docker.internal:${PRODUCTION_MCP_PORT:-3102}/mcp/
62
- - WIKI_MANAGER_RUNTIME_URL=http://host.docker.internal:7788
62
+ - WIKI_MANAGER_RUNTIME_URL=http://host.docker.internal:${WIKI_MANAGER_RUNTIME_PORT:-7788}
63
63
  - WIKI_MANAGER_RUNTIME_TOKEN=${WIKI_MANAGER_RUNTIME_TOKEN:-}
64
64
  # HTTPS — set paths inside the container (e.g. /certs/server.crt) and uncomment the volume above
65
65
  #- WIKI_SERVE_TLS_CERT_PATH=/certs/server.crt
@@ -110,6 +110,11 @@ services:
110
110
  - WIKI_CONFIG_PATH=${WIKI_CONFIG_PATH:-}
111
111
  - PRODUCTION_ALLOWED_STEPS=${PRODUCTION_ALLOWED_STEPS:-doctor,ingest,ingest_plan,ingest_apply,build,export,polish,pipeline}
112
112
  - PRODUCTION_REQUIRE_CONFIRMATION=${PRODUCTION_REQUIRE_CONFIRMATION:-false}
113
+ # Parallelism levers — effective concurrency ≈ recommendedConcurrency.
114
+ # Intermediate defaults (4/8). Low profile 2/4, high profile 8/16.
115
+ # See docs/configuration.md § "Parallelism & throughput".
116
+ - PRODUCTION_RECOMMENDED_CONCURRENCY=${PRODUCTION_RECOMMENDED_CONCURRENCY:-4}
117
+ - PRODUCTION_MAX_CONCURRENCY=${PRODUCTION_MAX_CONCURRENCY:-8}
113
118
  - PRODUCTION_JOBS_DIR=${PRODUCTION_JOBS_DIR:-/workspace/.wiki/production-jobs}
114
119
  - PRODUCTION_LOCKS_DIR=${PRODUCTION_LOCKS_DIR:-/workspace/.wiki/production-jobs/locks}
115
120
  ports:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@dotdrelle/wiki-manager",
3
- "version": "0.14.13",
3
+ "version": "0.14.20",
4
4
  "description": "Agentic shell and orchestration cockpit for llm-wiki workspaces.",
5
5
  "license": "PolyForm-Noncommercial-1.0.0",
6
6
  "author": "dotrelle",
@@ -71,23 +71,43 @@ function groupLine(group, activities) {
71
71
  const running = group.tasks.filter((task) => ACTIVE.has(statusOf(task)));
72
72
  const failed = group.tasks.find((task) => statusOf(task) === 'failed');
73
73
  const waitingApproval = group.tasks.some((task) => ['pending_approval', 'waiting_approval'].includes(statusOf(task)));
74
- const activeAgents = new Set(running.map((task) => task.agentInstanceId).filter(Boolean)).size;
75
- const activeProgress = running
76
- .map((task) => progressForTask(task, activities))
77
- .find((value) => value != null);
74
+ // Activity polling can advance before the persisted task projection catches
75
+ // up. Treat a live, linked activity as authoritative instead of rendering
76
+ // the whole group as "validation 0%" while a worker visibly runs at 35%.
77
+ const activePair = group.tasks
78
+ .map((task) => ({ task, activity: activityForTask(task, activities) }))
79
+ .find(({ activity }) => activity && !activity.terminal && !DONE.has(statusOf(activity)));
80
+ const activeTask = activePair?.task ?? running[0] ?? null;
81
+ const activeActivity = activePair?.activity
82
+ ?? running.map((task) => activityForTask(task, activities)).find(Boolean);
83
+ const activeAgents = new Set([
84
+ ...running.map((task) => task.agentInstanceId),
85
+ activeTask?.agentInstanceId,
86
+ ].filter(Boolean)).size;
87
+ const activeProgress = Number.isFinite(Number(activeActivity?.progress?.percent))
88
+ ? Number(activeActivity.progress.percent)
89
+ : running.map((task) => Number(task?.progress?.percent)).find(Number.isFinite);
78
90
  let icon = '[ ]';
79
91
  let status = 'en attente';
80
- if (failed) {
81
- icon = '[!]';
92
+ // A group may contain an earlier failure while another independent task is
93
+ // still progressing. Show the live worker as running; surface the group
94
+ // failure once no work remains. Otherwise its business label was rendered
95
+ // red even though that exact task was healthy and advancing.
96
+ // Glyphs mirror the Shell PlanPanel so the two never disagree: done is a
97
+ // check (not a cross — "[x]" read as a failure X), failure is a distinct
98
+ // cross, and an approval wait gets its own pause glyph instead of reusing the
99
+ // failure "[!]".
100
+ if (running.length > 0 || activeActivity) {
101
+ icon = '[...]';
102
+ status = activeProgress != null ? `${Math.round(activeProgress)} %` : `${done}/${total}`;
103
+ } else if (failed) {
104
+ icon = '[✗]';
82
105
  status = 'error';
83
106
  } else if (done === total) {
84
- icon = '[x]';
107
+ icon = '[✓]';
85
108
  status = 'done';
86
- } else if (running.length > 0) {
87
- icon = '[...]';
88
- status = activeProgress != null ? `${Math.round(activeProgress)} %` : `${done}/${total}`;
89
109
  } else if (waitingApproval) {
90
- icon = '[!]';
110
+ icon = '[⏸]';
91
111
  status = 'validation';
92
112
  } else if (done > 0) {
93
113
  icon = '[...]';
@@ -100,12 +120,28 @@ function groupLine(group, activities) {
100
120
  // text. Fall back to the task-completion ratio when nothing is running.
101
121
  const percent = running.length > 0 && activeProgress != null
102
122
  ? Math.round(activeProgress)
123
+ : activeActivity && activeProgress != null
124
+ ? Math.round(activeProgress)
103
125
  : (total > 0 ? Math.round((done / total) * 100) : null);
126
+ const phaseTasks = activeTask
127
+ ? group.tasks.filter((task) => String(task.operation ?? '') === String(activeTask.operation ?? ''))
128
+ : [];
129
+ const taskIndex = activeTask ? phaseTasks.indexOf(activeTask) + 1 : null;
130
+ const taskTotal = phaseTasks.length || null;
104
131
  return {
105
132
  id: `group:${group.id}`,
106
133
  label: `${icon} ${group.label} - ${status}${agents}`,
107
134
  status,
108
- progress: { done, total, percent },
135
+ // Preserve the active worker's business progress (source/template/
136
+ // deliverable, phase and detail). ShellUI can then show the same useful
137
+ // information as the direct wiki CLI instead of only "knowledge.update".
138
+ progress: {
139
+ ...(activeActivity?.progress ?? {}),
140
+ done,
141
+ total,
142
+ percent,
143
+ ...(taskIndex ? { taskIndex, taskTotal, taskOperation: activeTask?.operation ?? null } : {}),
144
+ },
109
145
  activeAgents,
110
146
  };
111
147
  }
@@ -121,15 +157,13 @@ function activityLine(activity) {
121
157
  };
122
158
  }
123
159
 
124
- function progressForTask(task, activities) {
160
+ function activityForTask(task, activities) {
125
161
  const taskId = String(task.id ?? task.step ?? '');
126
162
  const activityKey = task.activityKey ?? task.ownerActivityKey ?? null;
127
- const match = activities.find((activity) =>
163
+ return activities.find((activity) =>
128
164
  (activityKey && (activity.key === activityKey || activity.id === activityKey))
129
165
  || String(activity?.progress?.stepId ?? '') === taskId,
130
166
  );
131
- const value = Number(match?.progress?.percent ?? task.progress?.percent);
132
- return Number.isFinite(value) ? value : null;
133
167
  }
134
168
 
135
169
  function statusOf(task) {
@@ -5,6 +5,20 @@ import { aggregateActivity } from './activityAggregator.js';
5
5
  import { visibleActivityEvents } from './activityDeduplicator.js';
6
6
  import { calculateWeightedProgress } from './progressCalculator.js';
7
7
 
8
+ test('a group awaiting approval renders a distinct pause glyph, not a failure', () => {
9
+ const aggregated = aggregateActivity({
10
+ plan: [
11
+ { id: 'publish', label: 'Publication', groupId: 'publish', status: 'pending_approval', progressWeight: 1 },
12
+ ],
13
+ activities: [],
14
+ });
15
+ const line = aggregated.lines.find((item) => /publish|Publication/.test(item.label));
16
+ assert.ok(line, 'the awaiting-approval group is present');
17
+ assert.match(line.label, /\[⏸\]/, 'uses the pause glyph');
18
+ assert.doesNotMatch(line.label, /\[!\]|\[✗\]/, 'never reuses the failure glyph');
19
+ assert.equal(line.status, 'validation');
20
+ });
21
+
8
22
  test('activityDeduplicator keeps one visible entry for repeated 2 percent polls', () => {
9
23
  const events = Array.from({ length: 50 }, () => ({
10
24
  type: 'activity_upserted',
@@ -37,10 +51,10 @@ test('aggregateActivity exposes initial synthesis and grouped display lines', ()
37
51
  const activity = aggregateActivity({
38
52
  plan: [
39
53
  { id: 'collect', label: 'Collecte externe', groupId: 'collect', status: 'done', progressWeight: 1 },
40
- { id: 'enrich', label: 'Enrichissement commercial', groupId: 'enrich', requiredCapability: 'customer-data.enrich', status: 'running', progressWeight: 1, activityKey: 'activity-enrich' },
54
+ { id: 'enrich', label: 'Enrichissement commercial', groupId: 'enrich', requiredCapability: 'customer-data.enrich', operation: 'export', status: 'running', progressWeight: 1, activityKey: 'activity-enrich' },
41
55
  { id: 'publish', label: 'Publication', groupId: 'publish', status: 'pending', progressWeight: 1 },
42
56
  ],
43
- activities: [{ key: 'activity-enrich', label: 'Enrichissement commercial', status: 'running', progress: { percent: 63, stepId: 'enrich' } }],
57
+ activities: [{ key: 'activity-enrich', label: 'Enrichissement commercial', status: 'running', progress: { percent: 63, stepId: 'enrich', label: 'Export rapport.md', detail: 'Rendering PDF', currentStep: 'export' } }],
44
58
  }, [{
45
59
  type: 'plan.received',
46
60
  payload: {
@@ -54,8 +68,15 @@ test('aggregateActivity exposes initial synthesis and grouped display lines', ()
54
68
 
55
69
  assert.deepEqual(activity.initialSynthesis, ['120 sources detectees', '6 traitements simultanes recommandes']);
56
70
  assert.equal(activity.progress.percent, 54);
57
- assert.ok(activity.lines.some((line) => /\[x\] collect - done/.test(line.label)));
71
+ assert.ok(activity.lines.some((line) => /\[✓\] collect - done/.test(line.label)));
58
72
  assert.ok(activity.lines.some((line) => /\[\.\.\.\] customer-data\.enrich - 63 %/.test(line.label)));
73
+ const enrichLine = activity.lines.find((line) => /customer-data\.enrich/.test(line.label));
74
+ assert.equal(enrichLine.progress.label, 'Export rapport.md');
75
+ assert.equal(enrichLine.progress.detail, 'Rendering PDF');
76
+ assert.equal(enrichLine.progress.currentStep, 'export');
77
+ assert.equal(enrichLine.progress.taskIndex, 1);
78
+ assert.equal(enrichLine.progress.taskTotal, 1);
79
+ assert.equal(enrichLine.progress.taskOperation, 'export');
59
80
  assert.ok(activity.lines.some((line) => /\[ \] publish - en attente/.test(line.label)));
60
81
  });
61
82
 
@@ -85,8 +106,50 @@ test('aggregateActivity keeps activities not attached to any plan task visible',
85
106
 
86
107
  const aggregated = aggregateActivity(state, []);
87
108
  const labels = aggregated.lines.map((line) => line.label).join('\n');
88
- assert.match(labels, /\[x\] .* done/, 'the done plan group stays visible');
109
+ assert.match(labels, /\[✓\] .* done/, 'the done plan group stays visible');
89
110
  assert.match(labels, /Ingest b87acaf6/, 'the unattached running ingest must appear');
90
111
  const ingestLine = aggregated.lines.find((line) => /Ingest/.test(line.label));
91
112
  assert.equal(ingestLine.status, 'running');
92
113
  });
114
+
115
+ test('aggregateActivity trusts live worker progress while the task projection lags', () => {
116
+ const aggregated = aggregateActivity({
117
+ plan: [
118
+ { id: 'plan-a', groupId: 'knowledge.update', operation: 'ingest_plan', status: 'pending_approval', activityKey: 'activity-plan-a' },
119
+ { id: 'plan-b', groupId: 'knowledge.update', operation: 'ingest_plan', status: 'pending' },
120
+ ],
121
+ activities: [{
122
+ key: 'activity-plan-a',
123
+ status: 'running',
124
+ terminal: false,
125
+ progress: { percent: 35, stepId: 'plan-a', label: 'Ingest source-a.md', stepIndex: 1, stepTotal: 1 },
126
+ }],
127
+ }, []);
128
+
129
+ const line = aggregated.lines.find((item) => /knowledge\.update/.test(item.label));
130
+ assert.match(line.label, /35 %/);
131
+ assert.equal(line.progress.percent, 35);
132
+ assert.equal(line.progress.label, 'Ingest source-a.md');
133
+ assert.equal(line.progress.taskIndex, 1);
134
+ assert.equal(line.progress.taskTotal, 2);
135
+ });
136
+
137
+ test('aggregateActivity keeps a healthy active task out of the error color when a sibling failed', () => {
138
+ const aggregated = aggregateActivity({
139
+ plan: [
140
+ { id: 'failed-a', groupId: 'ingest', operation: 'ingest_plan', status: 'failed' },
141
+ { id: 'running-b', groupId: 'ingest', operation: 'ingest_plan', status: 'running', activityKey: 'activity-b' },
142
+ ],
143
+ activities: [{
144
+ key: 'activity-b',
145
+ status: 'running',
146
+ terminal: false,
147
+ progress: { percent: 35, stepId: 'running-b', label: 'Ingest application-orea.md', detail: 'LLM running' },
148
+ }],
149
+ }, []);
150
+
151
+ const line = aggregated.lines[0];
152
+ assert.equal(line.status, '35 %');
153
+ assert.match(line.label, /^\[\.\.\.\]/);
154
+ assert.equal(line.progress.label, 'Ingest application-orea.md');
155
+ });
@@ -274,6 +274,7 @@ const AgentState = Annotation.Root({
274
274
  invalidResponseRetries: Annotation({ default: () => 0 }),
275
275
  invalidToolCallRetries: Annotation({ default: () => 0 }),
276
276
  forceDelegation: Annotation({ default: () => false }),
277
+ terminalToolFailure: Annotation({ default: () => false }),
277
278
  });
278
279
 
279
280
  function invalidToolCalls(toolCalls) {
@@ -365,21 +366,62 @@ export function invalidUserFacingToolNames(content, session) {
365
366
  return [...new Set([...connected, ...syntactic])].sort();
366
367
  }
367
368
 
369
+ function parseActionJson(text) {
370
+ const cleaned = String(text ?? '').trim().replace(/^```(?:json)?\s*/i, '').replace(/\s*```$/, '');
371
+ if (!cleaned) return null;
372
+ return JSON.parse(cleaned)?.action === true;
373
+ }
374
+
368
375
  async function classifyRequestedAction(llm, input, signal) {
376
+ const system = [
377
+ 'Classify whether the user explicitly requests a real state-changing action now.',
378
+ 'Actions include starting, stopping, importing, ingesting, building, exporting, configuring, writing, deleting, or sending.',
379
+ 'Questions, explanations, status questions, greetings, and hypothetical discussions are not actions.',
380
+ 'Return JSON only: {"action":true} or {"action":false}.',
381
+ ].join('\n');
382
+ const messages = [{ role: 'user', content: String(input ?? '') }];
383
+
384
+ // Preferred path: a forced structured tool call, reliable on providers that
385
+ // honour tool_choice. But an OpenAI-compatible gateway (e.g. Albert / gpt-oss)
386
+ // may reject a forced tool_choice or return neither tool_calls nor parsable
387
+ // content. Without a fallback that made EVERY request classify as a non-action
388
+ // (catch → false), so Donna silently stopped delegating in agent mode. Fall
389
+ // back to a plain JSON-text completion, and only give up if both paths fail.
369
390
  try {
391
+ const classifier = {
392
+ type: 'function',
393
+ function: {
394
+ name: 'classify_action_request',
395
+ description: 'Classify whether the user explicitly requests a real state-changing action now.',
396
+ parameters: {
397
+ type: 'object',
398
+ additionalProperties: false,
399
+ properties: { action: { type: 'boolean' } },
400
+ required: ['action'],
401
+ },
402
+ },
403
+ };
370
404
  const result = await llm.completeWithTools({
371
- system: [
372
- 'Classify whether the user explicitly requests a real state-changing action now.',
373
- 'Actions include starting, stopping, importing, ingesting, building, exporting, configuring, writing, deleting, or sending.',
374
- 'Questions, explanations, status questions, greetings, and hypothetical discussions are not actions.',
375
- 'Return JSON only: {"action":true} or {"action":false}.',
376
- ].join('\n'),
377
- tools: [],
378
- messages: [{ role: 'user', content: String(input ?? '') }],
405
+ system,
406
+ tools: [classifier],
407
+ toolChoice: { type: 'function', function: { name: 'classify_action_request' } },
408
+ messages,
379
409
  signal,
380
410
  });
381
- const text = String(result?.content ?? '').trim().replace(/^```(?:json)?\s*/i, '').replace(/\s*```$/, '');
382
- return JSON.parse(text)?.action === true;
411
+ const call = (result?.tool_calls ?? []).find((item) => item?.function?.name === 'classify_action_request');
412
+ if (call) {
413
+ const parsed = JSON.parse(call.function.arguments ?? '{}')?.action;
414
+ if (typeof parsed === 'boolean') return parsed;
415
+ }
416
+ const fromText = parseActionJson(result?.content);
417
+ if (fromText !== null) return fromText;
418
+ } catch {
419
+ // Fall through to the toolless path below.
420
+ }
421
+
422
+ try {
423
+ const result = await llm.completeWithTools({ system, tools: [], messages, signal });
424
+ return parseActionJson(result?.content) === true;
383
425
  } catch {
384
426
  return false;
385
427
  }
@@ -1327,6 +1369,7 @@ export function createAgentGraph(options = {}) {
1327
1369
  async function toolExecutorNode(state) {
1328
1370
  const toolCalls = state.pendingToolCalls ?? [];
1329
1371
  const toolResultMessages = [];
1372
+ let terminalFailure = null;
1330
1373
 
1331
1374
  for (const call of toolCalls) {
1332
1375
  const resolved = resolveToolCallName(state.session.mcp, call.function.name, INTERNAL_TOOL_SERVERS);
@@ -1415,6 +1458,12 @@ export function createAgentGraph(options = {}) {
1415
1458
  resultText = JSON.stringify(result, null, 2);
1416
1459
  } else if (server === 'runtime') {
1417
1460
  resultText = await handleRuntimeControlTool(state.session, tool, args);
1461
+ if (tool === 'delegate' && /^Runtime control error \(delegate\):/i.test(resultText)) {
1462
+ terminalFailure = resultText
1463
+ .replace(/^Runtime control error \(delegate\):\s*/i, '')
1464
+ .replace(/^Delegation failed during objective_resolution:\s*/i, '');
1465
+ ok = false;
1466
+ }
1418
1467
  } else if (server !== 'shell') {
1419
1468
  await awaitRunApproval(state.session, { runId, tool: toolName });
1420
1469
  await awaitToolApproval(state.session, {
@@ -1510,17 +1559,36 @@ export function createAgentGraph(options = {}) {
1510
1559
  tool_call_id: call.id,
1511
1560
  content: boundedResult,
1512
1561
  });
1562
+ if (terminalFailure) break;
1513
1563
  }
1514
1564
 
1565
+ if (terminalFailure) {
1566
+ const response = `Action non lancée : ${terminalFailure}`;
1567
+ emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: response });
1568
+ return {
1569
+ response,
1570
+ messages: toolResultMessages,
1571
+ pendingToolCalls: null,
1572
+ forceDelegation: false,
1573
+ terminalToolFailure: true,
1574
+ invalidToolCallRetries: 0,
1575
+ invalidResponseRetries: 0,
1576
+ };
1577
+ }
1515
1578
  return {
1516
1579
  messages: toolResultMessages,
1517
1580
  pendingToolCalls: null,
1518
1581
  forceDelegation: false,
1519
1582
  invalidToolCallRetries: 0,
1520
1583
  invalidResponseRetries: 0,
1584
+ terminalToolFailure: false,
1521
1585
  };
1522
1586
  }
1523
1587
 
1588
+ function routeToolExecutor(state) {
1589
+ return state.terminalToolFailure ? END : 'orchestrator';
1590
+ }
1591
+
1524
1592
  function routeOrchestrator(state) {
1525
1593
  if (state.pendingToolCalls?.length > 0) return 'tool_executor';
1526
1594
  if (state.retryWithoutTool) return 'orchestrator';
@@ -1534,7 +1602,7 @@ export function createAgentGraph(options = {}) {
1534
1602
  .addNode('tool_executor', toolExecutorNode)
1535
1603
  .addEdge(START, 'orchestrator')
1536
1604
  .addConditionalEdges('orchestrator', routeOrchestrator)
1537
- .addEdge('tool_executor', 'orchestrator')
1605
+ .addConditionalEdges('tool_executor', routeToolExecutor)
1538
1606
  .compile();
1539
1607
 
1540
1608
  // LangGraph's default recursionLimit is 25 super-steps. Each tool round
@@ -25,7 +25,9 @@ test('Donna cannot answer an explicit action with manual instructions instead of
25
25
  runtime: { url: 'http://runtime.test' },
26
26
  llm: {
27
27
  async completeWithTools({ tools }) {
28
- if (tools.length === 0) return { content: '{"action":true}', message: { role: 'assistant', content: '{"action":true}' }, tool_calls: null };
28
+ if (tools.some((tool) => tool.function?.name === 'classify_action_request')) {
29
+ return { content: null, message: { role: 'assistant', content: null }, tool_calls: [{ id: 'classify', type: 'function', function: { name: 'classify_action_request', arguments: '{"action":true}' } }] };
30
+ }
29
31
  mainCalls += 1;
30
32
  if (mainCalls === 1) {
31
33
  return {
@@ -910,8 +912,8 @@ test('forced delegation is cleared after one valid tool call and does not loop',
910
912
  commands: ['status'],
911
913
  llm: {
912
914
  async completeWithTools({ toolChoice, tools }) {
913
- if (tools.length === 0) {
914
- return { content: '{"action":true}', message: { role: 'assistant', content: '{"action":true}' }, tool_calls: null };
915
+ if (tools.some((tool) => tool.function?.name === 'classify_action_request')) {
916
+ return { content: null, message: { role: 'assistant', content: null }, tool_calls: [{ id: 'classify', type: 'function', function: { name: 'classify_action_request', arguments: '{"action":true}' } }] };
915
917
  }
916
918
  calls += 1;
917
919
  choices.push(toolChoice);
@@ -946,6 +948,44 @@ test('forced delegation is cleared after one valid tool call and does not loop',
946
948
  }
947
949
  });
948
950
 
951
+ test('a rejected runtime delegation is terminal and never loops', async () => {
952
+ const originalFetch = globalThis.fetch;
953
+ globalThis.fetch = async () => ({
954
+ ok: false,
955
+ status: 422,
956
+ json: async () => ({
957
+ error: 'Delegation failed during objective_resolution: No orchestrable capability is currently available.',
958
+ }),
959
+ });
960
+ let calls = 0;
961
+ const session = sessionBase({
962
+ runtime: { url: 'http://runtime.test' },
963
+ llm: {
964
+ async completeWithTools() {
965
+ calls += 1;
966
+ return {
967
+ content: null,
968
+ message: { role: 'assistant', content: null },
969
+ tool_calls: [{
970
+ id: 'delegate-failure',
971
+ type: 'function',
972
+ function: { name: 'runtime__delegate', arguments: '{"objective":"Lance ingestion"}' },
973
+ }],
974
+ };
975
+ },
976
+ },
977
+ });
978
+
979
+ try {
980
+ const result = await createAgentGraph().invoke({ input: 'lance ingestion', session });
981
+ assert.equal(calls, 1);
982
+ assert.equal(result.response, 'Action non lancée : No orchestrable capability is currently available.');
983
+ assert.equal(result.terminalToolFailure, true);
984
+ } finally {
985
+ globalThis.fetch = originalFetch;
986
+ }
987
+ });
988
+
949
989
  // Guard: the system prompt must never show a connected tool's bare name
950
990
  // outside its qualified server__tool form. Bare mentions are what teach the
951
991
  // model to emit unqualified tool calls (the cme_status incident). The bare