@dotdrelle/wiki-manager 0.12.1 → 0.12.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/src/activity/activityAggregator.js +34 -3
- package/src/activity/activityAggregator.test.js +32 -0
- package/src/agent/graph.js +272 -22
- package/src/agent/graph.test.js +357 -1
- package/src/cli/wiki-manager.js +24 -1
- package/src/commands/slash.js +74 -1
- package/src/commands/slash.test.js +36 -0
- package/src/core/activity.js +4 -0
- package/src/core/activity.test.js +9 -1
- package/src/core/agentEvents.js +30 -0
- package/src/core/agentEvents.test.js +42 -1
- package/src/core/buildInfo.js +58 -0
- package/src/core/buildInfo.json +4 -0
- package/src/core/buildInfo.test.js +14 -0
- package/src/core/mcp.js +21 -1
- package/src/core/mcp.test.js +25 -0
- package/src/core/runtimeLog.js +6 -1
- package/src/core/runtimeLog.test.js +3 -1
- package/src/runtime/client.js +22 -0
- package/src/runtime/controlMessages.js +2 -2
- package/src/runtime/recoveryManager.js +54 -0
- package/src/runtime/recoveryManager.test.js +73 -0
- package/src/runtime/runner.js +49 -13
- package/src/runtime/runner.test.js +196 -0
- package/src/runtime/server.js +67 -8
- package/src/runtime/server.test.js +223 -6
- package/src/runtime/store.js +48 -3
- package/src/runtime/store.test.js +32 -0
- package/src/runtime/supervisor.js +77 -1
- package/src/shell/RightPane.tsx +124 -31
- package/src/shell/SetupWizard.tsx +13 -1
- package/src/shell/repl.js +78 -9
- package/src/shell/repl.test.js +114 -1
- package/src/shell/tui.tsx +4 -1
- package/src/shell/useAgent.ts +32 -3
- package/src/shell/useSession.ts +14 -1
|
@@ -279,6 +279,16 @@ test('reduceAgentEvents: control queue is event sourced and follows run status',
|
|
|
279
279
|
createdAt: '2026-01-01T00:00:00.000Z',
|
|
280
280
|
},
|
|
281
281
|
}),
|
|
282
|
+
createAgentEvent('control_enqueued', {
|
|
283
|
+
origin: 'runtime',
|
|
284
|
+
workspace: 'docs',
|
|
285
|
+
payload: {
|
|
286
|
+
id: 'control-2',
|
|
287
|
+
workspace: 'docs',
|
|
288
|
+
input: 'Never run',
|
|
289
|
+
createdAt: '2026-01-01T00:00:01.000Z',
|
|
290
|
+
},
|
|
291
|
+
}),
|
|
282
292
|
createAgentEvent('control_started', {
|
|
283
293
|
origin: 'runtime',
|
|
284
294
|
runId: 'run-control-1',
|
|
@@ -290,12 +300,19 @@ test('reduceAgentEvents: control queue is event sourced and follows run status',
|
|
|
290
300
|
runId: 'run-control-1',
|
|
291
301
|
workspace: 'docs',
|
|
292
302
|
}),
|
|
303
|
+
createAgentEvent('control_cancelled', {
|
|
304
|
+
origin: 'runtime',
|
|
305
|
+
workspace: 'docs',
|
|
306
|
+
payload: { id: 'control-2' },
|
|
307
|
+
}),
|
|
293
308
|
]);
|
|
294
309
|
|
|
295
|
-
assert.equal(projection.controlQueue.length,
|
|
310
|
+
assert.equal(projection.controlQueue.length, 2);
|
|
296
311
|
assert.equal(projection.controlQueue[0].id, 'control-1');
|
|
297
312
|
assert.equal(projection.controlQueue[0].status, 'done');
|
|
298
313
|
assert.equal(projection.controlQueue[0].runId, 'run-control-1');
|
|
314
|
+
assert.equal(projection.controlQueue[1].id, 'control-2');
|
|
315
|
+
assert.equal(projection.controlQueue[1].status, 'cancelled');
|
|
299
316
|
});
|
|
300
317
|
|
|
301
318
|
test('reduceAgentEvents: activity-owned plan is used when no orchestrator plan exists', () => {
|
|
@@ -399,3 +416,27 @@ test('reduceAgentEvents: plan patches are proposed, approved and applied with re
|
|
|
399
416
|
assert.equal(projection.plan[1].status, 'pending');
|
|
400
417
|
assert.equal(projection.planPatches[0].status, 'applied');
|
|
401
418
|
});
|
|
419
|
+
|
|
420
|
+
test('streamed narration split across tool iterations yields separate conversation entries', () => {
|
|
421
|
+
// graph.js finalizes the streaming entry (assistant_message content:'')
|
|
422
|
+
// before each tool batch so per-iteration narrations do not glue together
|
|
423
|
+
// into one wall of text.
|
|
424
|
+
const session = {};
|
|
425
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_delta', { origin: 'llm', payload: { delta: 'Analyse des jobs récents.' } }));
|
|
426
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_message', { origin: 'llm', payload: { content: '' } }));
|
|
427
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_delta', { origin: 'llm', payload: { delta: 'Voyons les logs.' } }));
|
|
428
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_message', { origin: 'llm', payload: { content: 'Voyons les logs.' } }));
|
|
429
|
+
|
|
430
|
+
const conversation = session.agentProjection.conversation;
|
|
431
|
+
assert.equal(conversation.length, 2);
|
|
432
|
+
assert.equal(conversation[0].content, 'Analyse des jobs récents.');
|
|
433
|
+
assert.equal(conversation[0].streaming ?? false, false);
|
|
434
|
+
assert.equal(conversation[1].content, 'Voyons les logs.');
|
|
435
|
+
});
|
|
436
|
+
|
|
437
|
+
test('empty assistant_message finalize is a no-op without a streaming entry', () => {
|
|
438
|
+
const session = {};
|
|
439
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_message', { origin: 'llm', payload: { content: 'Réponse finale.' } }));
|
|
440
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_message', { origin: 'llm', payload: { content: '' } }));
|
|
441
|
+
assert.equal(session.agentProjection.conversation.length, 1);
|
|
442
|
+
});
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
import { execFileSync } from 'node:child_process';
|
|
2
|
+
import { readFileSync } from 'node:fs';
|
|
3
|
+
import { dirname, join, resolve } from 'node:path';
|
|
4
|
+
import { fileURLToPath } from 'node:url';
|
|
5
|
+
|
|
6
|
+
const here = dirname(fileURLToPath(import.meta.url));
|
|
7
|
+
const packageRoot = resolve(here, '..', '..');
|
|
8
|
+
|
|
9
|
+
let cachedCommit;
|
|
10
|
+
|
|
11
|
+
// Short git commit identifying the code actually running. Resolution order:
|
|
12
|
+
// 1. Live git HEAD when running from the development repository — accurate
|
|
13
|
+
// even between releases (dirty trees still show the base commit).
|
|
14
|
+
// 2. buildInfo.json generated at pack time (scripts/check-versions.js) —
|
|
15
|
+
// what a published/global install carries.
|
|
16
|
+
// 3. null — displayed as "+dev" so an untraceable build is visible at a
|
|
17
|
+
// glance instead of silently pretending to match the repo.
|
|
18
|
+
export function buildCommit() {
|
|
19
|
+
if (cachedCommit !== undefined) return cachedCommit;
|
|
20
|
+
cachedCommit = liveGitCommit() ?? packagedCommit();
|
|
21
|
+
return cachedCommit;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export function versionWithBuild(packageJson) {
|
|
25
|
+
const version = String(packageJson?.version ?? '').trim();
|
|
26
|
+
const commit = buildCommit();
|
|
27
|
+
return commit ? `${version}+${commit}` : `${version}+dev`;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
function liveGitCommit() {
|
|
31
|
+
try {
|
|
32
|
+
// Guard against walking up into an unrelated parent repository when the
|
|
33
|
+
// package is installed under a directory that happens to be git-tracked.
|
|
34
|
+
const toplevel = git(['rev-parse', '--show-toplevel']);
|
|
35
|
+
if (!toplevel || resolve(toplevel) !== packageRoot) return null;
|
|
36
|
+
return git(['rev-parse', '--short', 'HEAD']);
|
|
37
|
+
} catch {
|
|
38
|
+
return null;
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function git(args) {
|
|
43
|
+
const output = execFileSync('git', args, {
|
|
44
|
+
cwd: packageRoot,
|
|
45
|
+
stdio: ['ignore', 'pipe', 'ignore'],
|
|
46
|
+
timeout: 2000,
|
|
47
|
+
}).toString().trim();
|
|
48
|
+
return output || null;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function packagedCommit() {
|
|
52
|
+
try {
|
|
53
|
+
const info = JSON.parse(readFileSync(join(here, 'buildInfo.json'), 'utf8'));
|
|
54
|
+
return info?.commit ? String(info.commit) : null;
|
|
55
|
+
} catch {
|
|
56
|
+
return null;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import test from 'node:test';
|
|
3
|
+
import { buildCommit, versionWithBuild } from './buildInfo.js';
|
|
4
|
+
|
|
5
|
+
test('versionWithBuild always exposes a provenance suffix', () => {
|
|
6
|
+
const formatted = versionWithBuild({ version: '9.9.9' });
|
|
7
|
+
// From the development repo the live git short sha is used; from a packed
|
|
8
|
+
// install the buildInfo.json commit; '+dev' only when neither is available.
|
|
9
|
+
assert.match(formatted, /^9\.9\.9\+(?:[0-9a-f]{4,40}|dev)$/);
|
|
10
|
+
});
|
|
11
|
+
|
|
12
|
+
test('buildCommit is stable across calls (cached)', () => {
|
|
13
|
+
assert.equal(buildCommit(), buildCommit());
|
|
14
|
+
});
|
package/src/core/mcp.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { existsSync, readFileSync } from 'node:fs';
|
|
2
2
|
import { managerEnvFile, managerMcpEndpointsFile, readEnvFile } from './env.js';
|
|
3
3
|
|
|
4
|
-
const WIKI_MANAGER_VERSION = '0.12.
|
|
4
|
+
const WIKI_MANAGER_VERSION = '0.12.7';
|
|
5
5
|
|
|
6
6
|
function envValue(key) {
|
|
7
7
|
const filePath = managerEnvFile();
|
|
@@ -302,6 +302,26 @@ export function formatMcpToolResult(result) {
|
|
|
302
302
|
.trim() || 'No result.';
|
|
303
303
|
}
|
|
304
304
|
|
|
305
|
+
const DEFAULT_TOOL_RESULT_MAX_CHARS = 16000;
|
|
306
|
+
|
|
307
|
+
function toolResultMaxChars() {
|
|
308
|
+
const parsed = Number(process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS);
|
|
309
|
+
return Number.isFinite(parsed) && parsed > 0 ? Math.floor(parsed) : DEFAULT_TOOL_RESULT_MAX_CHARS;
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
// Bound what a tool result injects into the LLM context and the conversation
|
|
313
|
+
// display. Apply this ONLY at those two exit points — never before payload
|
|
314
|
+
// parsing (extractActivity/_activity detection needs the full text).
|
|
315
|
+
// Head + tail are kept because errors and job ids often live at either end.
|
|
316
|
+
export function truncateToolResult(text, maxChars = toolResultMaxChars()) {
|
|
317
|
+
const full = String(text ?? '');
|
|
318
|
+
if (full.length <= maxChars) return full;
|
|
319
|
+
const headLength = Math.floor(maxChars * 0.7);
|
|
320
|
+
const tailLength = Math.floor(maxChars * 0.2);
|
|
321
|
+
const omitted = full.length - headLength - tailLength;
|
|
322
|
+
return `${full.slice(0, headLength)}\n\n[… ${omitted} caractères tronqués — résultat complet dans les logs runtime …]\n\n${full.slice(-tailLength)}`;
|
|
323
|
+
}
|
|
324
|
+
|
|
305
325
|
let _cachedEnvRetryPolicy = null;
|
|
306
326
|
function getEnvRetryPolicy() {
|
|
307
327
|
if (!_cachedEnvRetryPolicy) {
|
package/src/core/mcp.test.js
CHANGED
|
@@ -10,6 +10,7 @@ import {
|
|
|
10
10
|
formatMcpToolsForAgent,
|
|
11
11
|
resolveRetryPolicy,
|
|
12
12
|
resolveToolCallName,
|
|
13
|
+
truncateToolResult,
|
|
13
14
|
} from './mcp.js';
|
|
14
15
|
|
|
15
16
|
const resolveFixtureStatus = {
|
|
@@ -516,3 +517,27 @@ test('callMcpTool parses SSE responses after keepalive comments', async () => {
|
|
|
516
517
|
globalThis.fetch = originalFetch;
|
|
517
518
|
}
|
|
518
519
|
});
|
|
520
|
+
|
|
521
|
+
test('truncateToolResult keeps short results intact and bounds long ones head+tail', () => {
|
|
522
|
+
assert.equal(truncateToolResult('short result', 100), 'short result');
|
|
523
|
+
|
|
524
|
+
const long = `START-${'x'.repeat(50000)}-END`;
|
|
525
|
+
const bounded = truncateToolResult(long, 1000);
|
|
526
|
+
assert.ok(bounded.length < 1200, `bounded length ${bounded.length} should stay near the cap`);
|
|
527
|
+
assert.match(bounded, /^START-/);
|
|
528
|
+
assert.match(bounded, /-END$/);
|
|
529
|
+
assert.match(bounded, /caractères tronqués/);
|
|
530
|
+
});
|
|
531
|
+
|
|
532
|
+
test('truncateToolResult honours WIKI_MANAGER_TOOL_RESULT_MAX_CHARS', () => {
|
|
533
|
+
const previous = process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS;
|
|
534
|
+
process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS = '500';
|
|
535
|
+
try {
|
|
536
|
+
const bounded = truncateToolResult('y'.repeat(5000));
|
|
537
|
+
assert.ok(bounded.length < 700);
|
|
538
|
+
assert.match(bounded, /caractères tronqués/);
|
|
539
|
+
} finally {
|
|
540
|
+
if (previous === undefined) delete process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS;
|
|
541
|
+
else process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS = previous;
|
|
542
|
+
}
|
|
543
|
+
});
|
package/src/core/runtimeLog.js
CHANGED
|
@@ -61,7 +61,12 @@ export function normalizeRuntimeLog(input, { session = null } = {}) {
|
|
|
61
61
|
}
|
|
62
62
|
|
|
63
63
|
export function formatRuntimeLogPayload(payload = {}, ts = null) {
|
|
64
|
-
|
|
64
|
+
// Plain messages get the same time prefix as structured events: untimed
|
|
65
|
+
// lines ended up visually glued at the bottom of Logs/Trace, out of
|
|
66
|
+
// chronology with the shell's own timestamped lines.
|
|
67
|
+
if (payload?.message != null && !payload.event) {
|
|
68
|
+
return [timeLabel(ts), String(payload.message)].filter(Boolean).join(' ');
|
|
69
|
+
}
|
|
65
70
|
const time = timeLabel(ts);
|
|
66
71
|
const event = eventLabel(payload.event);
|
|
67
72
|
const fields = ORDERED_FIELDS
|
|
@@ -80,5 +80,7 @@ test('emitRuntimeLog accepts structured payloads and preserves legacy strings',
|
|
|
80
80
|
assert.match(session.agentProjection.logs[0], /ASSIGNED/);
|
|
81
81
|
assert.match(session.agentProjection.logs[0], /run=run-structured/);
|
|
82
82
|
assert.match(session.agentProjection.logs[0], /workspace=docs/);
|
|
83
|
-
|
|
83
|
+
// Legacy plain messages now carry the same HH:MM:SS prefix as structured
|
|
84
|
+
// events so the Logs/Trace panel stays chronologically readable.
|
|
85
|
+
assert.match(session.agentProjection.logs[1], /^\d{2}:\d{2}:\d{2} legacy line$/);
|
|
84
86
|
});
|
package/src/runtime/client.js
CHANGED
|
@@ -10,6 +10,14 @@ function runtimeEndpoint(url, path, workspace = null) {
|
|
|
10
10
|
return endpoint.toString();
|
|
11
11
|
}
|
|
12
12
|
|
|
13
|
+
function runtimeEndpointWithParams(url, path, params = {}) {
|
|
14
|
+
const endpoint = new URL(`${base(url)}${path}`);
|
|
15
|
+
for (const [key, value] of Object.entries(params)) {
|
|
16
|
+
if (value != null && value !== '') endpoint.searchParams.set(key, String(value));
|
|
17
|
+
}
|
|
18
|
+
return endpoint.toString();
|
|
19
|
+
}
|
|
20
|
+
|
|
13
21
|
export function runtimeUrlFromEnv() {
|
|
14
22
|
return process.env.WIKI_MANAGER_RUNTIME_URL ?? 'http://127.0.0.1:7788';
|
|
15
23
|
}
|
|
@@ -97,6 +105,20 @@ export async function postRuntimeCancel({
|
|
|
97
105
|
return response.json();
|
|
98
106
|
}
|
|
99
107
|
|
|
108
|
+
export async function postRuntimeKill({
|
|
109
|
+
url = runtimeUrlFromEnv(),
|
|
110
|
+
token = runtimeToken(),
|
|
111
|
+
workspace = null,
|
|
112
|
+
runId = null,
|
|
113
|
+
} = {}) {
|
|
114
|
+
const response = await fetch(runtimeEndpointWithParams(url, '/kill', { workspace, runId }), {
|
|
115
|
+
method: 'POST',
|
|
116
|
+
headers: runtimeHeaders(token),
|
|
117
|
+
});
|
|
118
|
+
if (!response.ok && response.status !== 501) throw new Error(`Runtime kill failed: HTTP ${response.status}`);
|
|
119
|
+
return response.json();
|
|
120
|
+
}
|
|
121
|
+
|
|
100
122
|
export async function postRuntimeShutdown({
|
|
101
123
|
url = runtimeUrlFromEnv(),
|
|
102
124
|
token = runtimeToken(),
|
|
@@ -18,8 +18,8 @@ const CONTROL_MESSAGES = {
|
|
|
18
18
|
fr: 'Modification de plan proposée. Approuvez-la explicitement pour l’appliquer au plan actif.',
|
|
19
19
|
},
|
|
20
20
|
ambiguous_control: {
|
|
21
|
-
en: '
|
|
22
|
-
fr: '
|
|
21
|
+
en: 'A run is already active, and this looks like a new action. Say "queue it" to run it after the current run, "modify the run" to change the active plan, "cancel" to stop the current run first — or wait for it to finish.',
|
|
22
|
+
fr: 'Un run est déjà actif et ta demande ressemble à une nouvelle action. Dis « mets en file » pour l\'exécuter après le run en cours, « modifie le run » pour changer le plan actif, « annule » pour arrêter le run actuel — ou attends la fin.',
|
|
23
23
|
},
|
|
24
24
|
converse_while_running: {
|
|
25
25
|
en: 'Runtime run is still active. This message was treated as conversation and did not create a queued run.',
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { parseJsonText } from '../core/activity.js';
|
|
2
2
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
3
3
|
import { formatMcpToolResult, callMcpTool as defaultCallMcpTool } from '../core/mcp.js';
|
|
4
|
+
import { createCapabilityRegistry } from '../orchestrator/capabilityRegistry.js';
|
|
4
5
|
import { accept as acceptResult } from '../orchestrator/resultAggregator.js';
|
|
5
6
|
|
|
6
7
|
const ACTIVE_TASK_STATUSES = new Set(['running', 'queued', 'starting', 'assigned']);
|
|
@@ -24,9 +25,11 @@ export async function recoverActiveRuns({
|
|
|
24
25
|
for (const run of runs) {
|
|
25
26
|
const tasks = store.listTasks?.({ runId: run.id }) ?? [];
|
|
26
27
|
const activeTasks = tasks.filter((task) => ACTIVE_TASK_STATUSES.has(String(task.status ?? '').toLowerCase()));
|
|
28
|
+
const runOutcomes = [];
|
|
27
29
|
for (const task of activeTasks) {
|
|
28
30
|
try {
|
|
29
31
|
const outcome = await recoverTask({ store, session, run, task, callTool, resultAggregator });
|
|
32
|
+
runOutcomes.push(outcome ?? null);
|
|
30
33
|
if (outcome?.status === 'recovered') recovered.push(outcome);
|
|
31
34
|
else if (outcome?.status === 'rescheduled') rescheduled.push(outcome);
|
|
32
35
|
else if (outcome?.status === 'interrupted') interrupted.push(outcome);
|
|
@@ -34,6 +37,21 @@ export async function recoverActiveRuns({
|
|
|
34
37
|
errors.push({ runId: run.id, taskId: task.id, error: error instanceof Error ? error.message : String(error) });
|
|
35
38
|
}
|
|
36
39
|
}
|
|
40
|
+
// A run in which nothing was recovered or rescheduled can never progress:
|
|
41
|
+
// leaving it 'running' in the store would re-attach it as a zombie on
|
|
42
|
+
// every subsequent boot. Close it for good.
|
|
43
|
+
if (activeTasks.length > 0 && runOutcomes.length === activeTasks.length
|
|
44
|
+
&& runOutcomes.every((outcome) => outcome?.status === 'interrupted')) {
|
|
45
|
+
const changed = store.interruptRuns?.({ workspace: run.workspace ?? null, runId: run.id, reason: 'Recovery found no recoverable task.' }) ?? 0;
|
|
46
|
+
if (changed > 0) {
|
|
47
|
+
dispatch(session, store, 'runtime_log', {
|
|
48
|
+
origin: 'recovery_manager',
|
|
49
|
+
runId: run.id,
|
|
50
|
+
workspace: run.workspace ?? workspaceFromSession(session),
|
|
51
|
+
payload: { message: `recovery: run ${run.id} interrupted (no recoverable task)` },
|
|
52
|
+
});
|
|
53
|
+
}
|
|
54
|
+
}
|
|
37
55
|
}
|
|
38
56
|
|
|
39
57
|
return {
|
|
@@ -46,6 +64,28 @@ export async function recoverActiveRuns({
|
|
|
46
64
|
}
|
|
47
65
|
|
|
48
66
|
async function recoverTask({ store, session, run, task, callTool, resultAggregator }) {
|
|
67
|
+
// A task whose capability no longer resolves can never be dispatched:
|
|
68
|
+
// re-attaching it would recreate the forever-waiting queue it came from.
|
|
69
|
+
// Fail it explicitly instead. No registry information at all (discovery
|
|
70
|
+
// not run yet) keeps the current behavior.
|
|
71
|
+
if (task.requiredCapability && !capabilityResolvable(session, task.requiredCapability)) {
|
|
72
|
+
dispatch(session, store, 'plan_step_updated', {
|
|
73
|
+
origin: 'recovery_manager',
|
|
74
|
+
runId: run.id,
|
|
75
|
+
taskId: task.id,
|
|
76
|
+
workspace: run.workspace ?? workspaceFromSession(session),
|
|
77
|
+
payload: {
|
|
78
|
+
taskId: task.id,
|
|
79
|
+
status: 'failed',
|
|
80
|
+
recovery: {
|
|
81
|
+
reason: 'unresolvable_capability',
|
|
82
|
+
capability: task.requiredCapability,
|
|
83
|
+
},
|
|
84
|
+
},
|
|
85
|
+
});
|
|
86
|
+
return interruptTask({ store, session, run, task, reason: `unresolvable capability: ${task.requiredCapability}` });
|
|
87
|
+
}
|
|
88
|
+
|
|
49
89
|
const attempt = latestAttempt(store.listTaskAttempts?.({ taskId: task.id }) ?? []);
|
|
50
90
|
const assignment = latestAssignment(store.listTaskAssignments?.({ taskId: task.id }) ?? [], attempt?.attemptId);
|
|
51
91
|
if (!attempt?.jobId || !assignment?.agentInstanceId) {
|
|
@@ -126,6 +166,20 @@ function isTerminal(status) {
|
|
|
126
166
|
return TERMINAL_STATUSES.has(String(status ?? '').toLowerCase());
|
|
127
167
|
}
|
|
128
168
|
|
|
169
|
+
function capabilityResolvable(session, capability) {
|
|
170
|
+
const registry = session.capabilityRegistry
|
|
171
|
+
?? ((session.agentRegistrySnapshot ?? []).length > 0
|
|
172
|
+
? createCapabilityRegistry({ agents: session.agentRegistrySnapshot })
|
|
173
|
+
: null);
|
|
174
|
+
if (!registry || typeof registry.providersFor !== 'function') return true;
|
|
175
|
+
// Only trust a registry that actually knows about capabilities. An empty
|
|
176
|
+
// one (discovery not finished, or agents described without capability
|
|
177
|
+
// lists) cannot distinguish "nothing provides X" from "no information".
|
|
178
|
+
const snapshot = typeof registry.snapshot === 'function' ? registry.snapshot() : {};
|
|
179
|
+
if (Object.keys(snapshot ?? {}).length === 0) return true;
|
|
180
|
+
return registry.providersFor(capability).length > 0;
|
|
181
|
+
}
|
|
182
|
+
|
|
129
183
|
function parseToolPayload(result) {
|
|
130
184
|
if (result && typeof result === 'object' && !Array.isArray(result) && !Array.isArray(result.content)) return result;
|
|
131
185
|
return parseJsonText(formatMcpToolResult(result)) ?? {};
|
|
@@ -160,3 +160,76 @@ function recoverySession() {
|
|
|
160
160
|
}],
|
|
161
161
|
};
|
|
162
162
|
}
|
|
163
|
+
|
|
164
|
+
test('recoveryManager fails unresolvable-capability tasks and interrupts the run', async () => {
|
|
165
|
+
const { store, root, runId, taskId } = storeWithActiveTask();
|
|
166
|
+
const session = recoverySession();
|
|
167
|
+
// A registry that DOES know capabilities, but not the task's one: the plan
|
|
168
|
+
// came from a hallucinated capability (the 0.12.1 incident) — re-attaching
|
|
169
|
+
// it would recreate a forever-waiting queue on every boot.
|
|
170
|
+
session.agentRegistrySnapshot = [{
|
|
171
|
+
agentInstanceId: 'production-main',
|
|
172
|
+
serverName: 'production',
|
|
173
|
+
health: 'available',
|
|
174
|
+
description: {
|
|
175
|
+
agentType: 'production',
|
|
176
|
+
contractVersion: '1',
|
|
177
|
+
capabilities: [{ id: 'knowledge.pipeline', version: '1' }],
|
|
178
|
+
},
|
|
179
|
+
}];
|
|
180
|
+
store.hydrateSession(session, { workspace: 'docs' });
|
|
181
|
+
const statusCalls = [];
|
|
182
|
+
|
|
183
|
+
try {
|
|
184
|
+
const result = await recoverActiveRuns({
|
|
185
|
+
store,
|
|
186
|
+
session,
|
|
187
|
+
workspace: 'docs',
|
|
188
|
+
callTool: async (_mcp, serverName, toolName, args) => {
|
|
189
|
+
statusCalls.push({ serverName, toolName, args });
|
|
190
|
+
return { content: [{ type: 'text', text: '{}' }] };
|
|
191
|
+
},
|
|
192
|
+
});
|
|
193
|
+
|
|
194
|
+
assert.equal(result.ok, true);
|
|
195
|
+
assert.equal(result.recovered.length, 0);
|
|
196
|
+
assert.equal(result.rescheduled.length, 0);
|
|
197
|
+
assert.equal(result.interrupted.length, 1);
|
|
198
|
+
assert.match(result.interrupted[0].reason, /unresolvable capability: document\.build/);
|
|
199
|
+
assert.deepEqual(statusCalls, [], 'no agent_status poll for an unresolvable task');
|
|
200
|
+
assert.equal(store.listTasks({ runId })[0].status, 'failed');
|
|
201
|
+
// The run must not come back as a zombie on the next boot.
|
|
202
|
+
assert.deepEqual(store.listRecoverableRuns({ workspace: 'docs' }), []);
|
|
203
|
+
} finally {
|
|
204
|
+
store.close();
|
|
205
|
+
rmSync(root, { recursive: true, force: true });
|
|
206
|
+
}
|
|
207
|
+
});
|
|
208
|
+
|
|
209
|
+
test('recoveryManager keeps recovering when the registry has no capability information', async () => {
|
|
210
|
+
const { store, root, runId, taskId } = storeWithActiveTask();
|
|
211
|
+
const session = recoverySession(); // snapshot without capability lists
|
|
212
|
+
store.hydrateSession(session, { workspace: 'docs' });
|
|
213
|
+
|
|
214
|
+
try {
|
|
215
|
+
const result = await recoverActiveRuns({
|
|
216
|
+
store,
|
|
217
|
+
session,
|
|
218
|
+
workspace: 'docs',
|
|
219
|
+
callTool: async (_mcp, _serverName, _toolName, args) => ({
|
|
220
|
+
content: [{ type: 'text', text: JSON.stringify({
|
|
221
|
+
jobId: args.jobId,
|
|
222
|
+
taskId,
|
|
223
|
+
status: 'done',
|
|
224
|
+
result: { status: 'succeeded', outputRefs: [], metrics: {} },
|
|
225
|
+
}) }],
|
|
226
|
+
}),
|
|
227
|
+
});
|
|
228
|
+
|
|
229
|
+
assert.equal(result.recovered.length, 1, 'uninformative registry must not block recovery');
|
|
230
|
+
assert.equal(store.listTasks({ runId })[0].status, 'done');
|
|
231
|
+
} finally {
|
|
232
|
+
store.close();
|
|
233
|
+
rmSync(root, { recursive: true, force: true });
|
|
234
|
+
}
|
|
235
|
+
});
|
package/src/runtime/runner.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
2
|
-
import { sessionActivities, terminalFailures } from '../core/activity.js';
|
|
2
|
+
import { isCancelledStatus, sessionActivities, terminalFailures } from '../core/activity.js';
|
|
3
3
|
import { runAgenticLoop, throwIfAborted } from '../core/agentLoop.js';
|
|
4
|
-
import { formatPlanStatus } from '../core/plan.js';
|
|
4
|
+
import { formatPlanStatus, formatPlanStep } from '../core/plan.js';
|
|
5
5
|
import { readyPlanTasks, sanitizePlanForExecution } from '../core/planPatch.js';
|
|
6
6
|
import { createAssignmentManager } from '../orchestrator/assignmentManager.js';
|
|
7
7
|
import { createAttemptManager } from '../orchestrator/attemptManager.js';
|
|
@@ -134,6 +134,19 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
|
|
|
134
134
|
continue;
|
|
135
135
|
}
|
|
136
136
|
}
|
|
137
|
+
if (!trigger && isCancelledOnlyLoopResult(result)) {
|
|
138
|
+
dispatchAgentEvent(session, createAgentEvent('run_cancelled', {
|
|
139
|
+
origin: 'runtime',
|
|
140
|
+
runId,
|
|
141
|
+
payload: {
|
|
142
|
+
runId,
|
|
143
|
+
cancelled: true,
|
|
144
|
+
message: 'Runtime run cancelled by user.',
|
|
145
|
+
},
|
|
146
|
+
}));
|
|
147
|
+
emitRuntimeLog(session, 'runtime: run ended by user cancellation (no replan)');
|
|
148
|
+
return { ok: false, result, cancelled: true };
|
|
149
|
+
}
|
|
137
150
|
dispatchAgentEvent(session, createAgentEvent('run_error', {
|
|
138
151
|
origin: 'runtime',
|
|
139
152
|
runId,
|
|
@@ -800,6 +813,7 @@ export async function replanRuntimeRun(session, input, trigger, {
|
|
|
800
813
|
'Given the original objective, current plan, and failure reason, return only the remaining steps required.',
|
|
801
814
|
'Do not include steps that are already done.',
|
|
802
815
|
'Return only JSON with this exact shape: {"steps":["..."]}.',
|
|
816
|
+
'Each step MUST be a plain string, not an object.',
|
|
803
817
|
].join('\n'),
|
|
804
818
|
tools: [],
|
|
805
819
|
messages: [{ role: 'user', content: buildReplanPrompt(input, session, trigger) }],
|
|
@@ -825,10 +839,12 @@ export async function replanRuntimeRun(session, input, trigger, {
|
|
|
825
839
|
steps: mergedSteps,
|
|
826
840
|
},
|
|
827
841
|
}));
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
842
|
+
// Every replan requires approval: deciding "is this step mutating?" with
|
|
843
|
+
// a verb regex was a safety judgement made by pattern-matching — a
|
|
844
|
+
// missed verb silently skipped the approval gate. Replans are rare
|
|
845
|
+
// (technical failures only) and re-executing work deserves a human OK.
|
|
846
|
+
session._runApprovalRequired = true;
|
|
847
|
+
session._runApprovalResolved = false;
|
|
832
848
|
return { ok: true, steps };
|
|
833
849
|
} catch (err) {
|
|
834
850
|
return { ok: false, reason: err instanceof Error ? err.message : String(err) };
|
|
@@ -847,7 +863,12 @@ function replanTriggerFromLoopResult(result) {
|
|
|
847
863
|
};
|
|
848
864
|
}
|
|
849
865
|
const failures = terminalFailures(result.completed ?? []);
|
|
850
|
-
const
|
|
866
|
+
const technical = failures.filter((failure) => !isCancelledStatus(failure.status));
|
|
867
|
+
const cancelled = failures.filter((failure) => isCancelledStatus(failure.status));
|
|
868
|
+
if (cancelled.length > 0 && technical.length === 0 && (result.failures ?? []).every(isCancelledTaskFailure)) {
|
|
869
|
+
return null;
|
|
870
|
+
}
|
|
871
|
+
const failure = technical[0];
|
|
851
872
|
if (failure) {
|
|
852
873
|
return {
|
|
853
874
|
kind: 'activity_error',
|
|
@@ -859,7 +880,7 @@ function replanTriggerFromLoopResult(result) {
|
|
|
859
880
|
// The parallel scheduler can fail a task before it ever produces an
|
|
860
881
|
// _activity (e.g. a thrown error on the first turn) — that failure lives
|
|
861
882
|
// in result.failures, not in any activity, so it must be checked too.
|
|
862
|
-
const taskFailure = (result.failures ?? [])
|
|
883
|
+
const taskFailure = (result.failures ?? []).find((failure) => !isCancelledTaskFailure(failure));
|
|
863
884
|
if (!taskFailure) return null;
|
|
864
885
|
return {
|
|
865
886
|
kind: 'task_error',
|
|
@@ -869,6 +890,25 @@ function replanTriggerFromLoopResult(result) {
|
|
|
869
890
|
};
|
|
870
891
|
}
|
|
871
892
|
|
|
893
|
+
function isCancelledTaskFailure(failure) {
|
|
894
|
+
return failure?.cancelled === true
|
|
895
|
+
|| failure?.result?.cancelled === true
|
|
896
|
+
|| isCancelledStatus(failure?.status)
|
|
897
|
+
|| isCancelledStatus(failure?.result?.status);
|
|
898
|
+
}
|
|
899
|
+
|
|
900
|
+
function isCancelledOnlyLoopResult(result) {
|
|
901
|
+
const failures = terminalFailures(result.completed ?? []);
|
|
902
|
+
const technicalActivities = failures.filter((failure) => !isCancelledStatus(failure.status));
|
|
903
|
+
const cancelledActivities = failures.filter((failure) => isCancelledStatus(failure.status));
|
|
904
|
+
const taskFailures = result.failures ?? [];
|
|
905
|
+
const technicalTasks = taskFailures.filter((failure) => !isCancelledTaskFailure(failure));
|
|
906
|
+
const cancelledTasks = taskFailures.filter(isCancelledTaskFailure);
|
|
907
|
+
return technicalActivities.length === 0
|
|
908
|
+
&& technicalTasks.length === 0
|
|
909
|
+
&& (cancelledActivities.length > 0 || cancelledTasks.length > 0);
|
|
910
|
+
}
|
|
911
|
+
|
|
872
912
|
function runtimeLoopErrorMessage(result) {
|
|
873
913
|
if (result.timedOut) return 'Runtime agentic loop timed out.';
|
|
874
914
|
if (result.maxTurns) return 'Runtime agentic loop reached max turns.';
|
|
@@ -912,15 +952,11 @@ function buildReplannedRunPrompt(input, trigger, steps) {
|
|
|
912
952
|
function normalizeReplan(steps) {
|
|
913
953
|
if (!Array.isArray(steps)) return [];
|
|
914
954
|
return steps
|
|
915
|
-
.map((step) =>
|
|
955
|
+
.map((step) => formatPlanStep(step).trim())
|
|
916
956
|
.filter(Boolean)
|
|
917
957
|
.slice(0, 12);
|
|
918
958
|
}
|
|
919
959
|
|
|
920
|
-
function hasMutatingReplanStep(steps) {
|
|
921
|
-
return steps.some((step) => /\b(build|copy|ingest|import|export|polish|pipeline|write|create|delete|update|send|deploy|publish|generate|construire|copier|importer|exporter|publier|envoyer|supprimer|modifier|creer|générer|generer)\b/i.test(step));
|
|
922
|
-
}
|
|
923
|
-
|
|
924
960
|
function mergeReplanWithCompleted(plan, steps) {
|
|
925
961
|
const completed = (plan ?? [])
|
|
926
962
|
.filter((step) => step.status === 'done')
|