@dotdrelle/wiki-manager 0.12.1 → 0.12.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/README.md +49 -8
  2. package/bin/wiki-manager +23 -12
  3. package/docker-compose.yml +0 -1
  4. package/mcp.endpoints.example.json +0 -6
  5. package/package.json +3 -2
  6. package/src/activity/activityAggregator.js +34 -3
  7. package/src/activity/activityAggregator.test.js +32 -0
  8. package/src/agent/graph.js +272 -22
  9. package/src/agent/graph.test.js +357 -1
  10. package/src/cli/wiki-manager.js +33 -2
  11. package/src/commands/slash.js +74 -1
  12. package/src/commands/slash.test.js +36 -0
  13. package/src/core/activity.js +4 -0
  14. package/src/core/activity.test.js +9 -1
  15. package/src/core/agentEvents.js +30 -0
  16. package/src/core/agentEvents.test.js +42 -1
  17. package/src/core/buildInfo.js +58 -0
  18. package/src/core/buildInfo.json +4 -0
  19. package/src/core/buildInfo.test.js +14 -0
  20. package/src/core/env.js +30 -1
  21. package/src/core/mcp.js +21 -1
  22. package/src/core/mcp.test.js +25 -0
  23. package/src/core/runtimeLog.js +6 -1
  24. package/src/core/runtimeLog.test.js +3 -1
  25. package/src/runtime/client.js +22 -0
  26. package/src/runtime/controlMessages.js +2 -2
  27. package/src/runtime/recoveryManager.js +54 -0
  28. package/src/runtime/recoveryManager.test.js +73 -0
  29. package/src/runtime/runner.js +49 -13
  30. package/src/runtime/runner.test.js +196 -0
  31. package/src/runtime/server.js +67 -8
  32. package/src/runtime/server.test.js +223 -6
  33. package/src/runtime/store.js +48 -3
  34. package/src/runtime/store.test.js +32 -0
  35. package/src/runtime/supervisor.js +77 -1
  36. package/src/shell/RightPane.tsx +124 -31
  37. package/src/shell/SetupWizard.tsx +13 -1
  38. package/src/shell/repl.js +78 -9
  39. package/src/shell/repl.test.js +114 -1
  40. package/src/shell/tui.tsx +4 -1
  41. package/src/shell/useAgent.ts +32 -3
  42. package/src/shell/useSession.ts +14 -1
  43. package/wiki-workspace +33 -0
@@ -39,6 +39,7 @@ const SESSION_PROJECTION_EVENTS = new Set([
39
39
  'approval.rejected',
40
40
  'control_enqueued',
41
41
  'control_started',
42
+ 'control_cancelled',
42
43
  'agent.registered',
43
44
  'agent.health_changed',
44
45
  'run_done',
@@ -479,6 +480,11 @@ function applyEvent(state, event) {
479
480
  case 'run_cancelled':
480
481
  state.status = 'cancelled';
481
482
  state.logs.push(String(event.payload?.message ?? 'Agent run cancelled.'));
483
+ // A cancelled run must not leave its plan steps "running/pending" and
484
+ // its activities spinning in the panels: mark every non-terminal one
485
+ // cancelled so the display reflects reality immediately.
486
+ cancelPendingPlanSteps(state.plan);
487
+ cancelActiveActivities(state.activities, event.ts);
482
488
  finishControlByRun(state.controlQueue, event.runId ?? event.payload?.runId ?? null, 'cancelled', event.ts);
483
489
  return;
484
490
  case 'run_error':
@@ -505,6 +511,14 @@ function applyEvent(state, event) {
505
511
  updatedAt: event.ts,
506
512
  });
507
513
  return;
514
+ case 'control_cancelled':
515
+ upsertControlItem(state.controlQueue, {
516
+ id: event.payload?.id ?? null,
517
+ status: 'cancelled',
518
+ finishedAt: event.payload?.finishedAt ?? event.ts,
519
+ updatedAt: event.ts,
520
+ });
521
+ return;
508
522
  case 'agent.registered':
509
523
  upsertAgent(state, event.payload?.agent, event.ts);
510
524
  return;
@@ -652,6 +666,22 @@ function markCoveredApprovalsApproved(approvals, grant, ts) {
652
666
  }
653
667
  }
654
668
 
669
+ function cancelPendingPlanSteps(plan) {
670
+ for (const step of plan ?? []) {
671
+ if (!['done', 'failed', 'cancelled'].includes(String(step.status ?? ''))) step.status = 'cancelled';
672
+ }
673
+ }
674
+
675
+ function cancelActiveActivities(activities, ts) {
676
+ for (const activity of Object.values(activities ?? {})) {
677
+ if (activity && activity.terminal !== true) {
678
+ activity.status = 'cancelled';
679
+ activity.terminal = true;
680
+ activity.updatedAt = ts ?? activity.updatedAt;
681
+ }
682
+ }
683
+ }
684
+
655
685
  function finishPendingPlanSteps(plan) {
656
686
  for (const step of plan ?? []) {
657
687
  if (step.status === 'running' || step.status === 'pending') {
@@ -279,6 +279,16 @@ test('reduceAgentEvents: control queue is event sourced and follows run status',
279
279
  createdAt: '2026-01-01T00:00:00.000Z',
280
280
  },
281
281
  }),
282
+ createAgentEvent('control_enqueued', {
283
+ origin: 'runtime',
284
+ workspace: 'docs',
285
+ payload: {
286
+ id: 'control-2',
287
+ workspace: 'docs',
288
+ input: 'Never run',
289
+ createdAt: '2026-01-01T00:00:01.000Z',
290
+ },
291
+ }),
282
292
  createAgentEvent('control_started', {
283
293
  origin: 'runtime',
284
294
  runId: 'run-control-1',
@@ -290,12 +300,19 @@ test('reduceAgentEvents: control queue is event sourced and follows run status',
290
300
  runId: 'run-control-1',
291
301
  workspace: 'docs',
292
302
  }),
303
+ createAgentEvent('control_cancelled', {
304
+ origin: 'runtime',
305
+ workspace: 'docs',
306
+ payload: { id: 'control-2' },
307
+ }),
293
308
  ]);
294
309
 
295
- assert.equal(projection.controlQueue.length, 1);
310
+ assert.equal(projection.controlQueue.length, 2);
296
311
  assert.equal(projection.controlQueue[0].id, 'control-1');
297
312
  assert.equal(projection.controlQueue[0].status, 'done');
298
313
  assert.equal(projection.controlQueue[0].runId, 'run-control-1');
314
+ assert.equal(projection.controlQueue[1].id, 'control-2');
315
+ assert.equal(projection.controlQueue[1].status, 'cancelled');
299
316
  });
300
317
 
301
318
  test('reduceAgentEvents: activity-owned plan is used when no orchestrator plan exists', () => {
@@ -399,3 +416,27 @@ test('reduceAgentEvents: plan patches are proposed, approved and applied with re
399
416
  assert.equal(projection.plan[1].status, 'pending');
400
417
  assert.equal(projection.planPatches[0].status, 'applied');
401
418
  });
419
+
420
+ test('streamed narration split across tool iterations yields separate conversation entries', () => {
421
+ // graph.js finalizes the streaming entry (assistant_message content:'')
422
+ // before each tool batch so per-iteration narrations do not glue together
423
+ // into one wall of text.
424
+ const session = {};
425
+ dispatchAgentEvent(session, createAgentEvent('assistant_delta', { origin: 'llm', payload: { delta: 'Analyse des jobs récents.' } }));
426
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', { origin: 'llm', payload: { content: '' } }));
427
+ dispatchAgentEvent(session, createAgentEvent('assistant_delta', { origin: 'llm', payload: { delta: 'Voyons les logs.' } }));
428
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', { origin: 'llm', payload: { content: 'Voyons les logs.' } }));
429
+
430
+ const conversation = session.agentProjection.conversation;
431
+ assert.equal(conversation.length, 2);
432
+ assert.equal(conversation[0].content, 'Analyse des jobs récents.');
433
+ assert.equal(conversation[0].streaming ?? false, false);
434
+ assert.equal(conversation[1].content, 'Voyons les logs.');
435
+ });
436
+
437
+ test('empty assistant_message finalize is a no-op without a streaming entry', () => {
438
+ const session = {};
439
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', { origin: 'llm', payload: { content: 'Réponse finale.' } }));
440
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', { origin: 'llm', payload: { content: '' } }));
441
+ assert.equal(session.agentProjection.conversation.length, 1);
442
+ });
@@ -0,0 +1,58 @@
1
+ import { execFileSync } from 'node:child_process';
2
+ import { readFileSync } from 'node:fs';
3
+ import { dirname, join, resolve } from 'node:path';
4
+ import { fileURLToPath } from 'node:url';
5
+
6
+ const here = dirname(fileURLToPath(import.meta.url));
7
+ const packageRoot = resolve(here, '..', '..');
8
+
9
+ let cachedCommit;
10
+
11
+ // Short git commit identifying the code actually running. Resolution order:
12
+ // 1. Live git HEAD when running from the development repository — accurate
13
+ // even between releases (dirty trees still show the base commit).
14
+ // 2. buildInfo.json generated at pack time (scripts/check-versions.js) —
15
+ // what a published/global install carries.
16
+ // 3. null — displayed as "+dev" so an untraceable build is visible at a
17
+ // glance instead of silently pretending to match the repo.
18
+ export function buildCommit() {
19
+ if (cachedCommit !== undefined) return cachedCommit;
20
+ cachedCommit = liveGitCommit() ?? packagedCommit();
21
+ return cachedCommit;
22
+ }
23
+
24
+ export function versionWithBuild(packageJson) {
25
+ const version = String(packageJson?.version ?? '').trim();
26
+ const commit = buildCommit();
27
+ return commit ? `${version}+${commit}` : `${version}+dev`;
28
+ }
29
+
30
+ function liveGitCommit() {
31
+ try {
32
+ // Guard against walking up into an unrelated parent repository when the
33
+ // package is installed under a directory that happens to be git-tracked.
34
+ const toplevel = git(['rev-parse', '--show-toplevel']);
35
+ if (!toplevel || resolve(toplevel) !== packageRoot) return null;
36
+ return git(['rev-parse', '--short', 'HEAD']);
37
+ } catch {
38
+ return null;
39
+ }
40
+ }
41
+
42
+ function git(args) {
43
+ const output = execFileSync('git', args, {
44
+ cwd: packageRoot,
45
+ stdio: ['ignore', 'pipe', 'ignore'],
46
+ timeout: 2000,
47
+ }).toString().trim();
48
+ return output || null;
49
+ }
50
+
51
+ function packagedCommit() {
52
+ try {
53
+ const info = JSON.parse(readFileSync(join(here, 'buildInfo.json'), 'utf8'));
54
+ return info?.commit ? String(info.commit) : null;
55
+ } catch {
56
+ return null;
57
+ }
58
+ }
@@ -0,0 +1,4 @@
1
+ {
2
+ "version": "0.12.8",
3
+ "commit": "eefd59f"
4
+ }
@@ -0,0 +1,14 @@
1
+ import assert from 'node:assert/strict';
2
+ import test from 'node:test';
3
+ import { buildCommit, versionWithBuild } from './buildInfo.js';
4
+
5
+ test('versionWithBuild always exposes a provenance suffix', () => {
6
+ const formatted = versionWithBuild({ version: '9.9.9' });
7
+ // From the development repo the live git short sha is used; from a packed
8
+ // install the buildInfo.json commit; '+dev' only when neither is available.
9
+ assert.match(formatted, /^9\.9\.9\+(?:[0-9a-f]{4,40}|dev)$/);
10
+ });
11
+
12
+ test('buildCommit is stable across calls (cached)', () => {
13
+ assert.equal(buildCommit(), buildCommit());
14
+ });
package/src/core/env.js CHANGED
@@ -1,5 +1,6 @@
1
- import { existsSync, readFileSync } from 'node:fs';
1
+ import { copyFileSync, existsSync, readFileSync } from 'node:fs';
2
2
  import { dirname, isAbsolute, join, resolve } from 'node:path';
3
+ import { fileURLToPath } from 'node:url';
3
4
 
4
5
  export function userManagerDir() {
5
6
  return process.cwd();
@@ -31,6 +32,34 @@ export function managerMcpEndpointsFile() {
31
32
  return join(managerStateDir(), 'mcp.endpoints.json');
32
33
  }
33
34
 
35
+ const packageRoot = resolve(dirname(fileURLToPath(import.meta.url)), '..', '..');
36
+
37
+ // First-run scaffolding: a fresh install directory has neither
38
+ // mcp.endpoints.json nor .env, so the external agents (cme, mailer,
39
+ // documents) silently never connect — /status shows no agents and Donna has
40
+ // no CME tools to configure anything with. Copy the packaged examples so a
41
+ // fresh directory works out of the box with the default agent ports; the
42
+ // user only has to fill in tokens/keys.
43
+ export function ensureManagerScaffold({ log = () => {} } = {}) {
44
+ const created = [];
45
+ const endpointsFile = managerMcpEndpointsFile();
46
+ const endpointsExample = join(packageRoot, 'mcp.endpoints.example.json');
47
+ if (!existsSync(endpointsFile) && existsSync(endpointsExample)) {
48
+ copyFileSync(endpointsExample, endpointsFile);
49
+ created.push('mcp.endpoints.json');
50
+ }
51
+ const envFile = managerEnvFile();
52
+ const envExample = join(packageRoot, '.env.example');
53
+ if (!existsSync(envFile) && existsSync(envExample)) {
54
+ copyFileSync(envExample, envFile);
55
+ created.push('.env');
56
+ }
57
+ if (created.length > 0) {
58
+ log(`scaffold: created ${created.join(' and ')} in ${managerStateDir()} from packaged example(s) — fill in tokens/keys (WIKI_WORKSPACES_DIR, *_MCP_AUTH_TOKEN…) before starting agents.`);
59
+ }
60
+ return created;
61
+ }
62
+
34
63
  // Single source of truth for where `.agents-data` lives, shared by the
35
64
  // manager's own document intake and by the host path it mounts into agent
36
65
  // containers — keeping them in sync avoids the two silently drifting apart.
package/src/core/mcp.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import { existsSync, readFileSync } from 'node:fs';
2
2
  import { managerEnvFile, managerMcpEndpointsFile, readEnvFile } from './env.js';
3
3
 
4
- const WIKI_MANAGER_VERSION = '0.12.1';
4
+ const WIKI_MANAGER_VERSION = '0.12.8';
5
5
 
6
6
  function envValue(key) {
7
7
  const filePath = managerEnvFile();
@@ -302,6 +302,26 @@ export function formatMcpToolResult(result) {
302
302
  .trim() || 'No result.';
303
303
  }
304
304
 
305
+ const DEFAULT_TOOL_RESULT_MAX_CHARS = 16000;
306
+
307
+ function toolResultMaxChars() {
308
+ const parsed = Number(process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS);
309
+ return Number.isFinite(parsed) && parsed > 0 ? Math.floor(parsed) : DEFAULT_TOOL_RESULT_MAX_CHARS;
310
+ }
311
+
312
+ // Bound what a tool result injects into the LLM context and the conversation
313
+ // display. Apply this ONLY at those two exit points — never before payload
314
+ // parsing (extractActivity/_activity detection needs the full text).
315
+ // Head + tail are kept because errors and job ids often live at either end.
316
+ export function truncateToolResult(text, maxChars = toolResultMaxChars()) {
317
+ const full = String(text ?? '');
318
+ if (full.length <= maxChars) return full;
319
+ const headLength = Math.floor(maxChars * 0.7);
320
+ const tailLength = Math.floor(maxChars * 0.2);
321
+ const omitted = full.length - headLength - tailLength;
322
+ return `${full.slice(0, headLength)}\n\n[… ${omitted} caractères tronqués — résultat complet dans les logs runtime …]\n\n${full.slice(-tailLength)}`;
323
+ }
324
+
305
325
  let _cachedEnvRetryPolicy = null;
306
326
  function getEnvRetryPolicy() {
307
327
  if (!_cachedEnvRetryPolicy) {
@@ -10,6 +10,7 @@ import {
10
10
  formatMcpToolsForAgent,
11
11
  resolveRetryPolicy,
12
12
  resolveToolCallName,
13
+ truncateToolResult,
13
14
  } from './mcp.js';
14
15
 
15
16
  const resolveFixtureStatus = {
@@ -516,3 +517,27 @@ test('callMcpTool parses SSE responses after keepalive comments', async () => {
516
517
  globalThis.fetch = originalFetch;
517
518
  }
518
519
  });
520
+
521
+ test('truncateToolResult keeps short results intact and bounds long ones head+tail', () => {
522
+ assert.equal(truncateToolResult('short result', 100), 'short result');
523
+
524
+ const long = `START-${'x'.repeat(50000)}-END`;
525
+ const bounded = truncateToolResult(long, 1000);
526
+ assert.ok(bounded.length < 1200, `bounded length ${bounded.length} should stay near the cap`);
527
+ assert.match(bounded, /^START-/);
528
+ assert.match(bounded, /-END$/);
529
+ assert.match(bounded, /caractères tronqués/);
530
+ });
531
+
532
+ test('truncateToolResult honours WIKI_MANAGER_TOOL_RESULT_MAX_CHARS', () => {
533
+ const previous = process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS;
534
+ process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS = '500';
535
+ try {
536
+ const bounded = truncateToolResult('y'.repeat(5000));
537
+ assert.ok(bounded.length < 700);
538
+ assert.match(bounded, /caractères tronqués/);
539
+ } finally {
540
+ if (previous === undefined) delete process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS;
541
+ else process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS = previous;
542
+ }
543
+ });
@@ -61,7 +61,12 @@ export function normalizeRuntimeLog(input, { session = null } = {}) {
61
61
  }
62
62
 
63
63
  export function formatRuntimeLogPayload(payload = {}, ts = null) {
64
- if (payload?.message != null && !payload.event) return String(payload.message);
64
+ // Plain messages get the same time prefix as structured events: untimed
65
+ // lines ended up visually glued at the bottom of Logs/Trace, out of
66
+ // chronology with the shell's own timestamped lines.
67
+ if (payload?.message != null && !payload.event) {
68
+ return [timeLabel(ts), String(payload.message)].filter(Boolean).join(' ');
69
+ }
65
70
  const time = timeLabel(ts);
66
71
  const event = eventLabel(payload.event);
67
72
  const fields = ORDERED_FIELDS
@@ -80,5 +80,7 @@ test('emitRuntimeLog accepts structured payloads and preserves legacy strings',
80
80
  assert.match(session.agentProjection.logs[0], /ASSIGNED/);
81
81
  assert.match(session.agentProjection.logs[0], /run=run-structured/);
82
82
  assert.match(session.agentProjection.logs[0], /workspace=docs/);
83
- assert.equal(session.agentProjection.logs[1], 'legacy line');
83
+ // Legacy plain messages now carry the same HH:MM:SS prefix as structured
84
+ // events so the Logs/Trace panel stays chronologically readable.
85
+ assert.match(session.agentProjection.logs[1], /^\d{2}:\d{2}:\d{2} legacy line$/);
84
86
  });
@@ -10,6 +10,14 @@ function runtimeEndpoint(url, path, workspace = null) {
10
10
  return endpoint.toString();
11
11
  }
12
12
 
13
+ function runtimeEndpointWithParams(url, path, params = {}) {
14
+ const endpoint = new URL(`${base(url)}${path}`);
15
+ for (const [key, value] of Object.entries(params)) {
16
+ if (value != null && value !== '') endpoint.searchParams.set(key, String(value));
17
+ }
18
+ return endpoint.toString();
19
+ }
20
+
13
21
  export function runtimeUrlFromEnv() {
14
22
  return process.env.WIKI_MANAGER_RUNTIME_URL ?? 'http://127.0.0.1:7788';
15
23
  }
@@ -97,6 +105,20 @@ export async function postRuntimeCancel({
97
105
  return response.json();
98
106
  }
99
107
 
108
+ export async function postRuntimeKill({
109
+ url = runtimeUrlFromEnv(),
110
+ token = runtimeToken(),
111
+ workspace = null,
112
+ runId = null,
113
+ } = {}) {
114
+ const response = await fetch(runtimeEndpointWithParams(url, '/kill', { workspace, runId }), {
115
+ method: 'POST',
116
+ headers: runtimeHeaders(token),
117
+ });
118
+ if (!response.ok && response.status !== 501) throw new Error(`Runtime kill failed: HTTP ${response.status}`);
119
+ return response.json();
120
+ }
121
+
100
122
  export async function postRuntimeShutdown({
101
123
  url = runtimeUrlFromEnv(),
102
124
  token = runtimeToken(),
@@ -18,8 +18,8 @@ const CONTROL_MESSAGES = {
18
18
  fr: 'Modification de plan proposée. Approuvez-la explicitement pour l’appliquer au plan actif.',
19
19
  },
20
20
  ambiguous_control: {
21
- en: 'The runtime cannot safely classify this message.',
22
- fr: 'Le runtime ne peut pas classer ce message de façon sûre.',
21
+ en: 'A run is already active, and this looks like a new action. Say "queue it" to run it after the current run, "modify the run" to change the active plan, "cancel" to stop the current run first — or wait for it to finish.',
22
+ fr: 'Un run est déjà actif et ta demande ressemble à une nouvelle action. Dis « mets en file » pour l\'exécuter après le run en cours, « modifie le run » pour changer le plan actif, « annule » pour arrêter le run actuel — ou attends la fin.',
23
23
  },
24
24
  converse_while_running: {
25
25
  en: 'Runtime run is still active. This message was treated as conversation and did not create a queued run.',
@@ -1,6 +1,7 @@
1
1
  import { parseJsonText } from '../core/activity.js';
2
2
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
3
3
  import { formatMcpToolResult, callMcpTool as defaultCallMcpTool } from '../core/mcp.js';
4
+ import { createCapabilityRegistry } from '../orchestrator/capabilityRegistry.js';
4
5
  import { accept as acceptResult } from '../orchestrator/resultAggregator.js';
5
6
 
6
7
  const ACTIVE_TASK_STATUSES = new Set(['running', 'queued', 'starting', 'assigned']);
@@ -24,9 +25,11 @@ export async function recoverActiveRuns({
24
25
  for (const run of runs) {
25
26
  const tasks = store.listTasks?.({ runId: run.id }) ?? [];
26
27
  const activeTasks = tasks.filter((task) => ACTIVE_TASK_STATUSES.has(String(task.status ?? '').toLowerCase()));
28
+ const runOutcomes = [];
27
29
  for (const task of activeTasks) {
28
30
  try {
29
31
  const outcome = await recoverTask({ store, session, run, task, callTool, resultAggregator });
32
+ runOutcomes.push(outcome ?? null);
30
33
  if (outcome?.status === 'recovered') recovered.push(outcome);
31
34
  else if (outcome?.status === 'rescheduled') rescheduled.push(outcome);
32
35
  else if (outcome?.status === 'interrupted') interrupted.push(outcome);
@@ -34,6 +37,21 @@ export async function recoverActiveRuns({
34
37
  errors.push({ runId: run.id, taskId: task.id, error: error instanceof Error ? error.message : String(error) });
35
38
  }
36
39
  }
40
+ // A run in which nothing was recovered or rescheduled can never progress:
41
+ // leaving it 'running' in the store would re-attach it as a zombie on
42
+ // every subsequent boot. Close it for good.
43
+ if (activeTasks.length > 0 && runOutcomes.length === activeTasks.length
44
+ && runOutcomes.every((outcome) => outcome?.status === 'interrupted')) {
45
+ const changed = store.interruptRuns?.({ workspace: run.workspace ?? null, runId: run.id, reason: 'Recovery found no recoverable task.' }) ?? 0;
46
+ if (changed > 0) {
47
+ dispatch(session, store, 'runtime_log', {
48
+ origin: 'recovery_manager',
49
+ runId: run.id,
50
+ workspace: run.workspace ?? workspaceFromSession(session),
51
+ payload: { message: `recovery: run ${run.id} interrupted (no recoverable task)` },
52
+ });
53
+ }
54
+ }
37
55
  }
38
56
 
39
57
  return {
@@ -46,6 +64,28 @@ export async function recoverActiveRuns({
46
64
  }
47
65
 
48
66
  async function recoverTask({ store, session, run, task, callTool, resultAggregator }) {
67
+ // A task whose capability no longer resolves can never be dispatched:
68
+ // re-attaching it would recreate the forever-waiting queue it came from.
69
+ // Fail it explicitly instead. No registry information at all (discovery
70
+ // not run yet) keeps the current behavior.
71
+ if (task.requiredCapability && !capabilityResolvable(session, task.requiredCapability)) {
72
+ dispatch(session, store, 'plan_step_updated', {
73
+ origin: 'recovery_manager',
74
+ runId: run.id,
75
+ taskId: task.id,
76
+ workspace: run.workspace ?? workspaceFromSession(session),
77
+ payload: {
78
+ taskId: task.id,
79
+ status: 'failed',
80
+ recovery: {
81
+ reason: 'unresolvable_capability',
82
+ capability: task.requiredCapability,
83
+ },
84
+ },
85
+ });
86
+ return interruptTask({ store, session, run, task, reason: `unresolvable capability: ${task.requiredCapability}` });
87
+ }
88
+
49
89
  const attempt = latestAttempt(store.listTaskAttempts?.({ taskId: task.id }) ?? []);
50
90
  const assignment = latestAssignment(store.listTaskAssignments?.({ taskId: task.id }) ?? [], attempt?.attemptId);
51
91
  if (!attempt?.jobId || !assignment?.agentInstanceId) {
@@ -126,6 +166,20 @@ function isTerminal(status) {
126
166
  return TERMINAL_STATUSES.has(String(status ?? '').toLowerCase());
127
167
  }
128
168
 
169
+ function capabilityResolvable(session, capability) {
170
+ const registry = session.capabilityRegistry
171
+ ?? ((session.agentRegistrySnapshot ?? []).length > 0
172
+ ? createCapabilityRegistry({ agents: session.agentRegistrySnapshot })
173
+ : null);
174
+ if (!registry || typeof registry.providersFor !== 'function') return true;
175
+ // Only trust a registry that actually knows about capabilities. An empty
176
+ // one (discovery not finished, or agents described without capability
177
+ // lists) cannot distinguish "nothing provides X" from "no information".
178
+ const snapshot = typeof registry.snapshot === 'function' ? registry.snapshot() : {};
179
+ if (Object.keys(snapshot ?? {}).length === 0) return true;
180
+ return registry.providersFor(capability).length > 0;
181
+ }
182
+
129
183
  function parseToolPayload(result) {
130
184
  if (result && typeof result === 'object' && !Array.isArray(result) && !Array.isArray(result.content)) return result;
131
185
  return parseJsonText(formatMcpToolResult(result)) ?? {};
@@ -160,3 +160,76 @@ function recoverySession() {
160
160
  }],
161
161
  };
162
162
  }
163
+
164
+ test('recoveryManager fails unresolvable-capability tasks and interrupts the run', async () => {
165
+ const { store, root, runId, taskId } = storeWithActiveTask();
166
+ const session = recoverySession();
167
+ // A registry that DOES know capabilities, but not the task's one: the plan
168
+ // came from a hallucinated capability (the 0.12.1 incident) — re-attaching
169
+ // it would recreate a forever-waiting queue on every boot.
170
+ session.agentRegistrySnapshot = [{
171
+ agentInstanceId: 'production-main',
172
+ serverName: 'production',
173
+ health: 'available',
174
+ description: {
175
+ agentType: 'production',
176
+ contractVersion: '1',
177
+ capabilities: [{ id: 'knowledge.pipeline', version: '1' }],
178
+ },
179
+ }];
180
+ store.hydrateSession(session, { workspace: 'docs' });
181
+ const statusCalls = [];
182
+
183
+ try {
184
+ const result = await recoverActiveRuns({
185
+ store,
186
+ session,
187
+ workspace: 'docs',
188
+ callTool: async (_mcp, serverName, toolName, args) => {
189
+ statusCalls.push({ serverName, toolName, args });
190
+ return { content: [{ type: 'text', text: '{}' }] };
191
+ },
192
+ });
193
+
194
+ assert.equal(result.ok, true);
195
+ assert.equal(result.recovered.length, 0);
196
+ assert.equal(result.rescheduled.length, 0);
197
+ assert.equal(result.interrupted.length, 1);
198
+ assert.match(result.interrupted[0].reason, /unresolvable capability: document\.build/);
199
+ assert.deepEqual(statusCalls, [], 'no agent_status poll for an unresolvable task');
200
+ assert.equal(store.listTasks({ runId })[0].status, 'failed');
201
+ // The run must not come back as a zombie on the next boot.
202
+ assert.deepEqual(store.listRecoverableRuns({ workspace: 'docs' }), []);
203
+ } finally {
204
+ store.close();
205
+ rmSync(root, { recursive: true, force: true });
206
+ }
207
+ });
208
+
209
+ test('recoveryManager keeps recovering when the registry has no capability information', async () => {
210
+ const { store, root, runId, taskId } = storeWithActiveTask();
211
+ const session = recoverySession(); // snapshot without capability lists
212
+ store.hydrateSession(session, { workspace: 'docs' });
213
+
214
+ try {
215
+ const result = await recoverActiveRuns({
216
+ store,
217
+ session,
218
+ workspace: 'docs',
219
+ callTool: async (_mcp, _serverName, _toolName, args) => ({
220
+ content: [{ type: 'text', text: JSON.stringify({
221
+ jobId: args.jobId,
222
+ taskId,
223
+ status: 'done',
224
+ result: { status: 'succeeded', outputRefs: [], metrics: {} },
225
+ }) }],
226
+ }),
227
+ });
228
+
229
+ assert.equal(result.recovered.length, 1, 'uninformative registry must not block recovery');
230
+ assert.equal(store.listTasks({ runId })[0].status, 'done');
231
+ } finally {
232
+ store.close();
233
+ rmSync(root, { recursive: true, force: true });
234
+ }
235
+ });