@dotdrelle/wiki-manager 0.12.11 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +6 -0
- package/docker-compose.yml +1 -1
- package/package.json +1 -1
- package/src/agent/graph.js +377 -142
- package/src/agent/graph.test.js +576 -34
- package/src/agent/llm.js +5 -5
- package/src/cli/wiki-manager.js +294 -9
- package/src/cli/wiki-manager.test.js +28 -0
- package/src/commands/slash.js +80 -13
- package/src/commands/slash.test.js +9 -1
- package/src/contracts/schemas.js +33 -0
- package/src/contracts/schemas.test.js +14 -0
- package/src/core/agentEvents.js +6 -0
- package/src/core/agentEvents.test.js +26 -0
- package/src/core/agentLoop.js +15 -16
- package/src/core/agentLoop.test.js +9 -7
- package/src/core/buildInfo.json +2 -2
- package/src/core/mcp.js +13 -6
- package/src/core/mcp.test.js +0 -12
- package/src/core/skills.js +0 -28
- package/src/orchestrator/capabilityRegistry.js +14 -0
- package/src/orchestrator/capabilityRegistry.test.js +12 -1
- package/src/orchestrator/dependencyResolver.js +10 -1
- package/src/orchestrator/dispatcher.js +34 -3
- package/src/orchestrator/dispatcher.test.js +34 -0
- package/src/orchestrator/objectiveResolver.js +79 -0
- package/src/orchestrator/objectiveResolver.test.js +50 -0
- package/src/orchestrator/scheduler.js +24 -0
- package/src/orchestrator/scheduler.test.js +65 -1
- package/src/runtime/client.js +34 -2
- package/src/runtime/lifecycle.js +1 -1
- package/src/runtime/recoveryManager.js +14 -7
- package/src/runtime/runner.js +214 -14
- package/src/runtime/runner.test.js +100 -2
- package/src/runtime/server.js +43 -3
- package/src/runtime/supervisor.js +65 -1
- package/src/runtime/supervisor.test.js +80 -0
- package/src/shell/LeftPane.tsx +9 -2
- package/src/shell/repl.js +57 -42
- package/src/shell/repl.test.js +81 -12
- package/src/shell/tui.tsx +26 -3
- package/src/shell/useSession.ts +15 -3
package/src/runtime/client.js
CHANGED
|
@@ -52,6 +52,7 @@ export async function postRuntimeRun(input, {
|
|
|
52
52
|
workspace = null,
|
|
53
53
|
evaluate = undefined,
|
|
54
54
|
replans = undefined,
|
|
55
|
+
capabilityPlan = undefined,
|
|
55
56
|
} = {}) {
|
|
56
57
|
const response = await fetch(runtimeEndpoint(url, '/run', workspace), {
|
|
57
58
|
method: 'POST',
|
|
@@ -59,7 +60,7 @@ export async function postRuntimeRun(input, {
|
|
|
59
60
|
...runtimeHeaders(token),
|
|
60
61
|
'Content-Type': 'application/json',
|
|
61
62
|
},
|
|
62
|
-
body: JSON.stringify(Object.assign({ input, workspace }, evaluate !== undefined && { evaluate }, replans !== undefined && { replans })),
|
|
63
|
+
body: JSON.stringify(Object.assign({ input, workspace }, evaluate !== undefined && { evaluate }, replans !== undefined && { replans }, capabilityPlan !== undefined && { capabilityPlan })),
|
|
63
64
|
});
|
|
64
65
|
if (!response.ok) {
|
|
65
66
|
const err = new Error(`Runtime run failed: HTTP ${response.status}`);
|
|
@@ -69,6 +70,25 @@ export async function postRuntimeRun(input, {
|
|
|
69
70
|
return response.json();
|
|
70
71
|
}
|
|
71
72
|
|
|
73
|
+
export async function postRuntimeDelegate(objective, {
|
|
74
|
+
url = runtimeUrlFromEnv(),
|
|
75
|
+
token = runtimeToken(),
|
|
76
|
+
workspace = null,
|
|
77
|
+
} = {}) {
|
|
78
|
+
const response = await fetch(runtimeEndpoint(url, '/delegate', workspace), {
|
|
79
|
+
method: 'POST',
|
|
80
|
+
headers: { ...runtimeHeaders(token), 'Content-Type': 'application/json' },
|
|
81
|
+
body: JSON.stringify({ objective, workspace }),
|
|
82
|
+
});
|
|
83
|
+
const payload = await response.json().catch(() => ({}));
|
|
84
|
+
if (!response.ok) {
|
|
85
|
+
const err = new Error(payload.error ?? `Runtime delegation failed: HTTP ${response.status}`);
|
|
86
|
+
err.status = response.status;
|
|
87
|
+
throw err;
|
|
88
|
+
}
|
|
89
|
+
return payload;
|
|
90
|
+
}
|
|
91
|
+
|
|
72
92
|
export async function postRuntimeControl(action, {
|
|
73
93
|
url = runtimeUrlFromEnv(),
|
|
74
94
|
token = runtimeToken(),
|
|
@@ -151,6 +171,9 @@ export async function postRuntimeApprove({
|
|
|
151
171
|
runId = null,
|
|
152
172
|
itemId = null,
|
|
153
173
|
approvalId = null,
|
|
174
|
+
scope = null,
|
|
175
|
+
planRevision = null,
|
|
176
|
+
approvalClasses = null,
|
|
154
177
|
} = {}) {
|
|
155
178
|
const endpoint = runtimeEndpoint(url, '/approve', workspace);
|
|
156
179
|
const parsed = new URL(endpoint);
|
|
@@ -159,7 +182,16 @@ export async function postRuntimeApprove({
|
|
|
159
182
|
if (approvalId) parsed.searchParams.set('approvalId', approvalId);
|
|
160
183
|
const response = await fetch(parsed.toString(), {
|
|
161
184
|
method: 'POST',
|
|
162
|
-
headers: runtimeHeaders(token),
|
|
185
|
+
headers: { ...runtimeHeaders(token), 'Content-Type': 'application/json' },
|
|
186
|
+
body: JSON.stringify({
|
|
187
|
+
workspace,
|
|
188
|
+
runId,
|
|
189
|
+
itemId,
|
|
190
|
+
approvalId,
|
|
191
|
+
scope,
|
|
192
|
+
planRevision,
|
|
193
|
+
approvalClasses,
|
|
194
|
+
}),
|
|
163
195
|
});
|
|
164
196
|
if (!response.ok) throw new Error(`Runtime approve failed: HTTP ${response.status}`);
|
|
165
197
|
return response.json();
|
package/src/runtime/lifecycle.js
CHANGED
|
@@ -109,7 +109,7 @@ export async function runtimeHealthOrNull(url = runtimeUrlFromEnv(), token = run
|
|
|
109
109
|
// alive after exit produced zombie runtimes running yesterday's code and
|
|
110
110
|
// yesterday's endpoints. Nuance preserved: if a run is active anywhere, the
|
|
111
111
|
// runtime is left alive so the run survives the shell (that promise stays).
|
|
112
|
-
export async function shutdownOwnedRuntime(runtime, { log = () => {} } = {}) {
|
|
112
|
+
export async function shutdownOwnedRuntime(runtime, { log = (_message) => {} } = {}) {
|
|
113
113
|
if (!runtime?.url || !runtime?.started) return { action: 'kept', reason: 'not_owned' };
|
|
114
114
|
try {
|
|
115
115
|
const health = await runtimeHealthOrNull(runtime.url, runtime.token);
|
|
@@ -37,18 +37,25 @@ export async function recoverActiveRuns({
|
|
|
37
37
|
errors.push({ runId: run.id, taskId: task.id, error: error instanceof Error ? error.message : String(error) });
|
|
38
38
|
}
|
|
39
39
|
}
|
|
40
|
-
// A run
|
|
41
|
-
//
|
|
42
|
-
// every
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
40
|
+
// A run that recovery cannot move forward must be closed for good, or it
|
|
41
|
+
// re-attaches as a blocking "a runtime run is already active" zombie on
|
|
42
|
+
// every boot. This covers BOTH cases that can never resume on a fresh boot:
|
|
43
|
+
// - runs whose active tasks were all interrupted, and
|
|
44
|
+
// - runs with no active task to recover at all (e.g. left waiting for
|
|
45
|
+
// approval, or with only un-started pending tasks). Nothing here will
|
|
46
|
+
// ever progress, so finalize it now.
|
|
47
|
+
const progressed = runOutcomes.some((outcome) => outcome?.status === 'recovered' || outcome?.status === 'rescheduled');
|
|
48
|
+
if (!progressed) {
|
|
49
|
+
const reason = activeTasks.length > 0
|
|
50
|
+
? 'Recovery found no recoverable task.'
|
|
51
|
+
: 'Recovery found no active task to resume.';
|
|
52
|
+
const changed = store.interruptRuns?.({ workspace: run.workspace ?? null, runId: run.id, reason }) ?? 0;
|
|
46
53
|
if (changed > 0) {
|
|
47
54
|
dispatch(session, store, 'runtime_log', {
|
|
48
55
|
origin: 'recovery_manager',
|
|
49
56
|
runId: run.id,
|
|
50
57
|
workspace: run.workspace ?? workspaceFromSession(session),
|
|
51
|
-
payload: { message: `recovery: run ${run.id} interrupted (
|
|
58
|
+
payload: { message: `recovery: run ${run.id} interrupted (${reason})` },
|
|
52
59
|
});
|
|
53
60
|
}
|
|
54
61
|
}
|
package/src/runtime/runner.js
CHANGED
|
@@ -9,10 +9,16 @@ import { createBudgetManager, BudgetExceededError } from '../orchestrator/budget
|
|
|
9
9
|
import { createDispatcher } from '../orchestrator/dispatcher.js';
|
|
10
10
|
import { assertValidatedFragment } from '../orchestrator/planValidator.js';
|
|
11
11
|
import { createResultAggregator } from '../orchestrator/resultAggregator.js';
|
|
12
|
-
import { drainActive,
|
|
12
|
+
import { drainActive, resolvePlanConcurrency, startReadyTasks } from '../orchestrator/scheduler.js';
|
|
13
13
|
import { emitRuntimeLog, pollActivitiesOnce } from './supervisor.js';
|
|
14
14
|
|
|
15
|
-
|
|
15
|
+
// 0 by default: automatic replans turn evaluator/replanner TEXT into
|
|
16
|
+
// executable pseudo-tasks (no capability, no operation) that stall at 0%
|
|
17
|
+
// and pile up as replan-1/2/3 ghost work — the same disease as the removed
|
|
18
|
+
// text-plan extraction. Failures now end with an honest report; the user
|
|
19
|
+
// (or a stronger model) decides what to do next. Re-enable explicitly with
|
|
20
|
+
// WIKI_MANAGER_REPLANNER_MAX_REPLANS if desired.
|
|
21
|
+
const DEFAULT_MAX_REPLANS = 0;
|
|
16
22
|
|
|
17
23
|
async function waitForRuntimeActivities(session, startedActivities, { timeoutMs, signal, pollBusy }) {
|
|
18
24
|
const deadline = Date.now() + timeoutMs;
|
|
@@ -43,13 +49,35 @@ async function waitForRuntimeActivities(session, startedActivities, { timeoutMs,
|
|
|
43
49
|
return { ok: false, timedOut: true, completed: tracked };
|
|
44
50
|
}
|
|
45
51
|
|
|
46
|
-
|
|
52
|
+
// Last chat exchanges (user/assistant) that preceded this run, so the run's
|
|
53
|
+
// LLM knows WHAT was agreed before acting. Long messages are clipped: the
|
|
54
|
+
// context is for grounding, not for re-reading novels.
|
|
55
|
+
// Env knobs (documented in .env.example): every tunable introduced by the
|
|
56
|
+
// grounding/orchestration work is overridable — nothing business-critical
|
|
57
|
+
// is frozen in code.
|
|
58
|
+
export function conversationSeed(session, currentInput, { limit = 12, maxChars = 2000 } = {}) {
|
|
59
|
+
const conversation = Array.isArray(session.agentProjection?.conversation)
|
|
60
|
+
? session.agentProjection.conversation
|
|
61
|
+
: [];
|
|
62
|
+
const seed = conversation
|
|
63
|
+
.filter((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim())
|
|
64
|
+
.slice(-limit)
|
|
65
|
+
.map((message) => ({ role: message.role, content: String(message.content).slice(0, maxChars) }));
|
|
66
|
+
// The run's own triggering user message is appended by the loop itself —
|
|
67
|
+
// drop it from the seed to avoid sending it twice.
|
|
68
|
+
const last = seed.at(-1);
|
|
69
|
+
if (last && last.role === 'user' && last.content === String(currentInput ?? '').slice(0, maxChars)) seed.pop();
|
|
70
|
+
return seed;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export async function runRuntimeAgenticLoop(agent, session, initialInput, { signal, timeoutMs, maxTurns, runId, pollBusy, parallelHandoff = false, initialMessages = [] }) {
|
|
47
74
|
return runAgenticLoop(agent, session, initialInput, {
|
|
48
75
|
signal,
|
|
49
76
|
timeoutMs,
|
|
50
77
|
maxTurns,
|
|
51
78
|
runId,
|
|
52
79
|
parallelHandoff,
|
|
80
|
+
initialMessages,
|
|
53
81
|
deterministicTerminalSummary: true,
|
|
54
82
|
abortMessage: 'Runtime run cancelled.',
|
|
55
83
|
waitForActivities: (turnSession, startedActivities, waitOptions) =>
|
|
@@ -103,10 +131,13 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
|
|
|
103
131
|
evaluate = true,
|
|
104
132
|
maxReplans = resolveMaxReplans(),
|
|
105
133
|
callTool = null,
|
|
106
|
-
dispatcherPollIntervalMs =
|
|
134
|
+
dispatcherPollIntervalMs = 2500,
|
|
107
135
|
} = {}) {
|
|
108
136
|
let currentInput = initialInput ?? input;
|
|
109
137
|
let replansLeft = Math.max(0, Math.floor(Number(maxReplans) || 0));
|
|
138
|
+
// Computed ONCE at run start: the pre-run chat. Re-computing inside the
|
|
139
|
+
// loop would re-ingest this run's own turns and duplicate them.
|
|
140
|
+
const runConversationSeed = conversationSeed(session, currentInput);
|
|
110
141
|
|
|
111
142
|
while (true) {
|
|
112
143
|
sanitizeSessionPlanForExecution(session, runId);
|
|
@@ -127,6 +158,7 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
|
|
|
127
158
|
runId,
|
|
128
159
|
pollBusy,
|
|
129
160
|
parallelHandoff: true,
|
|
161
|
+
initialMessages: runConversationSeed,
|
|
130
162
|
});
|
|
131
163
|
if (result.ok && result.handoff) continue;
|
|
132
164
|
if (!result.ok) {
|
|
@@ -212,6 +244,16 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
|
|
|
212
244
|
continue;
|
|
213
245
|
}
|
|
214
246
|
}
|
|
247
|
+
// Surface the verdict in the CHAT: the work that ran stays done, the
|
|
248
|
+
// user sees why the evaluator was unsatisfied and decides — no
|
|
249
|
+
// self-generated follow-up tasks.
|
|
250
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_message', {
|
|
251
|
+
origin: 'runtime',
|
|
252
|
+
runId,
|
|
253
|
+
payload: {
|
|
254
|
+
content: `Le run est terminé mais l'évaluation le juge incomplet : ${evaluation.reason}`,
|
|
255
|
+
},
|
|
256
|
+
}));
|
|
215
257
|
dispatchAgentEvent(session, createAgentEvent('run_error', {
|
|
216
258
|
origin: 'runtime',
|
|
217
259
|
runId,
|
|
@@ -240,7 +282,7 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
240
282
|
maxTurns,
|
|
241
283
|
runId = null,
|
|
242
284
|
pollBusy,
|
|
243
|
-
concurrency =
|
|
285
|
+
concurrency = null,
|
|
244
286
|
fragment = null,
|
|
245
287
|
assignmentManager = null,
|
|
246
288
|
attemptManager = null,
|
|
@@ -249,10 +291,18 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
249
291
|
budgetManager = null,
|
|
250
292
|
budgets = {},
|
|
251
293
|
callTool = null,
|
|
252
|
-
dispatcherPollIntervalMs =
|
|
294
|
+
dispatcherPollIntervalMs = 2500,
|
|
253
295
|
} = {}) {
|
|
254
296
|
if (fragment != null) assertValidatedFragment(fragment);
|
|
255
|
-
const
|
|
297
|
+
const agents = session.agentRegistry?.snapshot?.() ?? session.agentRegistrySnapshot ?? [];
|
|
298
|
+
const configuredConcurrency = Number(concurrency) > 0
|
|
299
|
+
? Number(concurrency)
|
|
300
|
+
: Number(process.env.WIKI_MANAGER_CAPABILITY_CONCURRENCY || process.env.WIKI_MANAGER_SCHEDULER_CONCURRENCY);
|
|
301
|
+
const limit = resolvePlanConcurrency({
|
|
302
|
+
plan: session.headlessPlan ?? [],
|
|
303
|
+
agents,
|
|
304
|
+
configured: configuredConcurrency,
|
|
305
|
+
});
|
|
256
306
|
const active = new Map();
|
|
257
307
|
const attempts = attemptManager ?? createAttemptManager();
|
|
258
308
|
const assigner = assignmentManager ?? createAssignmentManager({ session });
|
|
@@ -278,6 +328,15 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
278
328
|
sanitizeSessionPlanForExecution(session, runId);
|
|
279
329
|
ensurePlanProjection(session, runId);
|
|
280
330
|
emitRuntimeLog(session, `scheduler: parallel plan enabled (concurrency ${limit})`);
|
|
331
|
+
let approvalNoticeSent = false;
|
|
332
|
+
// Interactive approvals do NOT expire: the user has /approve, "valide
|
|
333
|
+
// tout", /cancel and /run kill — an arbitrary timer only created mystery
|
|
334
|
+
// failures. A deadline exists only when explicitly configured (headless
|
|
335
|
+
// runs, CI) via the session or the env escape hatch.
|
|
336
|
+
const configuredApprovalWait = Number(session._approvalTimeoutMs) > 0
|
|
337
|
+
? Number(session._approvalTimeoutMs)
|
|
338
|
+
: (Number(process.env.WIKI_MANAGER_APPROVAL_TIMEOUT_MS) > 0 ? Number(process.env.WIKI_MANAGER_APPROVAL_TIMEOUT_MS) : null);
|
|
339
|
+
const approvalDeadline = configuredApprovalWait ? Date.now() + configuredApprovalWait : Infinity;
|
|
281
340
|
|
|
282
341
|
try {
|
|
283
342
|
while (true) {
|
|
@@ -336,8 +395,9 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
336
395
|
emitRuntimeLog(session, taskLogPayload('task.starting', task, { runId, detail: `starting task ${taskId}` }));
|
|
337
396
|
},
|
|
338
397
|
startTask: (task, attempt) => {
|
|
398
|
+
const executableTask = materializeTaskInputs(task, session.headlessPlan ?? []);
|
|
339
399
|
const taskAbort = createTaskAbortSignal(signal);
|
|
340
|
-
const promise = runDispatchedTask(
|
|
400
|
+
const promise = runDispatchedTask(executableTask, {
|
|
341
401
|
session,
|
|
342
402
|
assignmentManager: assigner,
|
|
343
403
|
dispatcher: executor,
|
|
@@ -371,6 +431,44 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
371
431
|
emitRuntimeLog(session, `scheduler: budget exceeded (${exceeded.reason})`);
|
|
372
432
|
return { ok: false, budgetExceeded: true, reason: exceeded.reason, budget: exceeded, completed: sessionActivities(session), failures };
|
|
373
433
|
}
|
|
434
|
+
const needingApproval = pending.filter((step) => step.requiresApproval === true);
|
|
435
|
+
if (needingApproval.length > 0) {
|
|
436
|
+
// The plan is only blocked on a HUMAN decision — wait for it
|
|
437
|
+
// (bounded) instead of declaring the run stalled. Announce once in
|
|
438
|
+
// the chat: users cannot approve what they never saw asked.
|
|
439
|
+
if (!approvalNoticeSent) {
|
|
440
|
+
approvalNoticeSent = true;
|
|
441
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_message', {
|
|
442
|
+
origin: 'runtime',
|
|
443
|
+
runId,
|
|
444
|
+
payload: {
|
|
445
|
+
content: [
|
|
446
|
+
`⏸ Approbation requise avant exécution : ${needingApproval.length} tâche(s) mutante(s) en attente.`,
|
|
447
|
+
...needingApproval.slice(0, 5).map((step) => ` - ${step.description ?? step.id}`),
|
|
448
|
+
needingApproval.length > 5 ? ` … et ${needingApproval.length - 5} autre(s).` : null,
|
|
449
|
+
'Réponds « valide tout » (ou tape /approve) pour lancer, « annule » pour abandonner.',
|
|
450
|
+
].filter(Boolean).join('\n'),
|
|
451
|
+
},
|
|
452
|
+
}));
|
|
453
|
+
emitRuntimeLog(session, `scheduler: waiting for approval (${needingApproval.length} task(s))`);
|
|
454
|
+
}
|
|
455
|
+
if (Date.now() < approvalDeadline) {
|
|
456
|
+
await new Promise((resolveDelay) => setTimeout(resolveDelay, 500));
|
|
457
|
+
continue;
|
|
458
|
+
}
|
|
459
|
+
// Configured timeout (headless/CI) reached: say it PLAINLY in the
|
|
460
|
+
// chat and let run_error clean the plan/activities so nothing
|
|
461
|
+
// lingers in the panels.
|
|
462
|
+
emitRuntimeLog(session, 'scheduler: approval wait timed out');
|
|
463
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_message', {
|
|
464
|
+
origin: 'runtime',
|
|
465
|
+
runId,
|
|
466
|
+
payload: {
|
|
467
|
+
content: `⏱ Approbation non reçue dans le délai imparti — run arrêté, ${needingApproval.length} tâche(s) annulée(s). Relance la demande quand tu veux.`,
|
|
468
|
+
},
|
|
469
|
+
}));
|
|
470
|
+
return { ok: false, stalled: true, reason: 'awaiting_approval', completed: sessionActivities(session), failures };
|
|
471
|
+
}
|
|
374
472
|
const reason = pending.every((step) => approvalWaitingStatus(step.status)) ? 'awaiting_approval' : 'no_ready_plan_task';
|
|
375
473
|
emitRuntimeLog(session, `scheduler: stalled (${reason})`);
|
|
376
474
|
return { ok: false, stalled: true, reason, completed: sessionActivities(session), failures };
|
|
@@ -428,6 +526,41 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
428
526
|
}
|
|
429
527
|
}
|
|
430
528
|
|
|
529
|
+
export function materializeTaskInputs(task, plan = []) {
|
|
530
|
+
const dependencies = new Set(Array.isArray(task?.dependsOn) ? task.dependsOn.map(String) : []);
|
|
531
|
+
const replacements = new Map();
|
|
532
|
+
for (const dependency of plan ?? []) {
|
|
533
|
+
if (!dependencies.has(String(dependency?.id ?? dependency?.step))) continue;
|
|
534
|
+
const expected = Array.isArray(dependency?.expectedOutputRefs) ? dependency.expectedOutputRefs : [];
|
|
535
|
+
const actual = Array.isArray(dependency?.outputRefs) ? dependency.outputRefs : [];
|
|
536
|
+
for (let index = 0; index < Math.min(expected.length, actual.length); index += 1) {
|
|
537
|
+
const expectedRef = refValue(expected[index]);
|
|
538
|
+
const actualRef = refValue(actual[index]);
|
|
539
|
+
if (expectedRef && actualRef && expectedRef !== actualRef) replacements.set(expectedRef, actualRef);
|
|
540
|
+
}
|
|
541
|
+
}
|
|
542
|
+
if (replacements.size === 0) return task;
|
|
543
|
+
return {
|
|
544
|
+
...task,
|
|
545
|
+
arguments: replaceRefValues(task?.arguments, replacements),
|
|
546
|
+
inputRefs: replaceRefValues(task?.inputRefs, replacements),
|
|
547
|
+
};
|
|
548
|
+
}
|
|
549
|
+
|
|
550
|
+
function replaceRefValues(value, replacements) {
|
|
551
|
+
if (typeof value === 'string') return replacements.get(value) ?? value;
|
|
552
|
+
if (Array.isArray(value)) return value.map((item) => replaceRefValues(item, replacements));
|
|
553
|
+
if (value && typeof value === 'object') {
|
|
554
|
+
return Object.fromEntries(Object.entries(value).map(([key, item]) => [key, replaceRefValues(item, replacements)]));
|
|
555
|
+
}
|
|
556
|
+
return value;
|
|
557
|
+
}
|
|
558
|
+
|
|
559
|
+
function refValue(value) {
|
|
560
|
+
if (typeof value === 'string') return value;
|
|
561
|
+
return value && typeof value === 'object' ? String(value.ref ?? '') : '';
|
|
562
|
+
}
|
|
563
|
+
|
|
431
564
|
function sanitizeSessionPlanForExecution(session, runId = null) {
|
|
432
565
|
if (!session.headlessPlan) return;
|
|
433
566
|
const sanitized = sanitizePlanForExecution(session.headlessPlan);
|
|
@@ -603,7 +736,7 @@ async function runDispatchedTask(task, {
|
|
|
603
736
|
error: {
|
|
604
737
|
code: 'dispatcher_error',
|
|
605
738
|
message: err instanceof Error ? err.message : String(err),
|
|
606
|
-
retryable:
|
|
739
|
+
retryable: transientRuntimeError(err),
|
|
607
740
|
},
|
|
608
741
|
};
|
|
609
742
|
await resultAggregator.accept(result, { task, assignment });
|
|
@@ -623,8 +756,16 @@ async function runDispatchedTask(task, {
|
|
|
623
756
|
}
|
|
624
757
|
}
|
|
625
758
|
|
|
626
|
-
function shouldUseParallelScheduler(plan) {
|
|
627
|
-
|
|
759
|
+
export function shouldUseParallelScheduler(plan) {
|
|
760
|
+
// A validated provider plan enters the scheduler even while every task is
|
|
761
|
+
// waiting for approval. Looking only at readyPlanTasks() made an all-
|
|
762
|
+
// approval plan fall through to the conversational Donna loop; that loop
|
|
763
|
+
// then ignored the integrated TaskGraph and marked the run done.
|
|
764
|
+
return (plan ?? []).some((task) =>
|
|
765
|
+
task?.requiredCapability
|
|
766
|
+
&& task?.operation
|
|
767
|
+
&& pendingSchedulerStatus(task.status),
|
|
768
|
+
);
|
|
628
769
|
}
|
|
629
770
|
|
|
630
771
|
function taskLogPayload(event, task, {
|
|
@@ -718,6 +859,11 @@ export async function evaluateRuntimeRun(session, input, {
|
|
|
718
859
|
evaluate = true,
|
|
719
860
|
} = {}) {
|
|
720
861
|
if (!shouldEvaluate(evaluate)) return null;
|
|
862
|
+
const structured = structuredPlanEvaluation(session.headlessPlan);
|
|
863
|
+
if (structured) {
|
|
864
|
+
emitRuntimeLog(session, 'runtime: evaluating completed structured plan');
|
|
865
|
+
return structured;
|
|
866
|
+
}
|
|
721
867
|
const llm = session.llm;
|
|
722
868
|
if (!llm || typeof llm.completeWithTools !== 'function') {
|
|
723
869
|
return fallbackEvaluation('Evaluator unavailable: no LLM completeWithTools client.');
|
|
@@ -741,6 +887,40 @@ export async function evaluateRuntimeRun(session, input, {
|
|
|
741
887
|
}
|
|
742
888
|
}
|
|
743
889
|
|
|
890
|
+
function structuredPlanEvaluation(plan) {
|
|
891
|
+
if (!Array.isArray(plan) || plan.length === 0) return null;
|
|
892
|
+
// Only provider TaskGraph tasks are authoritative. Legacy conversational
|
|
893
|
+
// plans contain prose/tool labels and still use the compatibility evaluator.
|
|
894
|
+
if (!plan.every((step) => step?.requiredCapability && step?.operation)) return null;
|
|
895
|
+
const statuses = plan.map((step) => String(step?.status ?? '').toLowerCase());
|
|
896
|
+
const failed = plan.filter((step) => ['failed', 'error', 'cancelled', 'canceled', 'stalled'].includes(String(step?.status ?? '').toLowerCase()));
|
|
897
|
+
const incomplete = plan.filter((step) => !['done', 'complete', 'completed', 'success', 'succeeded'].includes(String(step?.status ?? '').toLowerCase()));
|
|
898
|
+
if (failed.length > 0) {
|
|
899
|
+
return {
|
|
900
|
+
ok: false,
|
|
901
|
+
reason: `${failed.length} tâche(s) du plan ont échoué : ${failed.map((step) => step.label ?? step.description ?? step.id ?? step.step).join(', ')}.`,
|
|
902
|
+
suggestedAction: null,
|
|
903
|
+
};
|
|
904
|
+
}
|
|
905
|
+
if (incomplete.length > 0) {
|
|
906
|
+
return {
|
|
907
|
+
ok: false,
|
|
908
|
+
reason: `${incomplete.length} tâche(s) du plan ne sont pas terminées.`,
|
|
909
|
+
suggestedAction: null,
|
|
910
|
+
};
|
|
911
|
+
}
|
|
912
|
+
return {
|
|
913
|
+
ok: statuses.length > 0,
|
|
914
|
+
reason: `${statuses.length} tâche(s) du plan terminées avec succès.`,
|
|
915
|
+
suggestedAction: null,
|
|
916
|
+
};
|
|
917
|
+
}
|
|
918
|
+
|
|
919
|
+
function transientRuntimeError(error) {
|
|
920
|
+
const value = error instanceof Error ? error.message : String(error ?? '');
|
|
921
|
+
return /(?:429|timeout|temporar|throttl|rate.?limit|quota|busy|unavailable)/i.test(value);
|
|
922
|
+
}
|
|
923
|
+
|
|
744
924
|
function shouldEvaluate(value) {
|
|
745
925
|
if (value === false) return false;
|
|
746
926
|
const env = String(process.env.WIKI_MANAGER_EVALUATOR ?? '').trim().toLowerCase();
|
|
@@ -860,10 +1040,30 @@ export async function replanRuntimeRun(session, input, trigger, {
|
|
|
860
1040
|
}
|
|
861
1041
|
}
|
|
862
1042
|
|
|
1043
|
+
function isBusyFailure(failure) {
|
|
1044
|
+
const fields = [
|
|
1045
|
+
failure?.error,
|
|
1046
|
+
failure?.status,
|
|
1047
|
+
failure?.result?.error?.code,
|
|
1048
|
+
failure?.result?.error?.message,
|
|
1049
|
+
failure?.result?.status,
|
|
1050
|
+
].map((value) => String(value ?? '').toLowerCase());
|
|
1051
|
+
return fields.some((value) => value.includes('busy') || value.includes('locked'));
|
|
1052
|
+
}
|
|
1053
|
+
|
|
863
1054
|
function replanTriggerFromLoopResult(result) {
|
|
864
1055
|
// 'awaiting_approval' is not a dead end — it means to wait for a human
|
|
865
1056
|
// decision, not to replan around it.
|
|
866
1057
|
if (result.stalled && result.reason !== 'awaiting_approval') {
|
|
1058
|
+
// A stall caused only by transient lock contention (target_busy /
|
|
1059
|
+
// workspace_busy) must NOT be replanned: the plan is correct, the
|
|
1060
|
+
// workspace was momentarily locked. Replanning around it re-runs the same
|
|
1061
|
+
// tasks, hits the lock again and spins the run into a zombie. Fail cleanly
|
|
1062
|
+
// so the run finalizes instead of lingering 'running'.
|
|
1063
|
+
const blocking = [...(result.failures ?? []), ...terminalFailures(result.completed ?? [])];
|
|
1064
|
+
if (blocking.length > 0 && blocking.every(isBusyFailure)) {
|
|
1065
|
+
return null;
|
|
1066
|
+
}
|
|
867
1067
|
return {
|
|
868
1068
|
kind: 'plan_stalled',
|
|
869
1069
|
reason: `Plan is stalled: ${result.reason ?? 'no ready task'} (pending steps exist but none have their dependencies satisfied).`,
|
|
@@ -872,7 +1072,7 @@ function replanTriggerFromLoopResult(result) {
|
|
|
872
1072
|
};
|
|
873
1073
|
}
|
|
874
1074
|
const failures = terminalFailures(result.completed ?? []);
|
|
875
|
-
const technical = failures.filter((failure) => !isCancelledStatus(failure.status));
|
|
1075
|
+
const technical = failures.filter((failure) => !isCancelledStatus(failure.status) && !isBusyFailure(failure));
|
|
876
1076
|
const cancelled = failures.filter((failure) => isCancelledStatus(failure.status));
|
|
877
1077
|
if (cancelled.length > 0 && technical.length === 0 && (result.failures ?? []).every(isCancelledTaskFailure)) {
|
|
878
1078
|
return null;
|
|
@@ -889,7 +1089,7 @@ function replanTriggerFromLoopResult(result) {
|
|
|
889
1089
|
// The parallel scheduler can fail a task before it ever produces an
|
|
890
1090
|
// _activity (e.g. a thrown error on the first turn) — that failure lives
|
|
891
1091
|
// in result.failures, not in any activity, so it must be checked too.
|
|
892
|
-
const taskFailure = (result.failures ?? []).find((failure) => !isCancelledTaskFailure(failure));
|
|
1092
|
+
const taskFailure = (result.failures ?? []).find((failure) => !isCancelledTaskFailure(failure) && !isBusyFailure(failure));
|
|
893
1093
|
if (!taskFailure) return null;
|
|
894
1094
|
return {
|
|
895
1095
|
kind: 'task_error',
|
|
@@ -1016,7 +1216,7 @@ function formatRecentConversation(session, n = 12) {
|
|
|
1016
1216
|
.join('\n');
|
|
1017
1217
|
}
|
|
1018
1218
|
|
|
1019
|
-
function resolveMaxReplans(value
|
|
1219
|
+
function resolveMaxReplans(value) {
|
|
1020
1220
|
const parsed = Number(value);
|
|
1021
1221
|
return Number.isFinite(parsed) ? Math.max(0, Math.floor(parsed)) : DEFAULT_MAX_REPLANS;
|
|
1022
1222
|
}
|
|
@@ -1,7 +1,70 @@
|
|
|
1
1
|
import assert from 'node:assert/strict';
|
|
2
2
|
import test from 'node:test';
|
|
3
3
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
4
|
-
import { finishRuntimeRun, replanRuntimeRun, runRuntimeAgenticWorkflow, runRuntimeParallelPlan } from './runner.js';
|
|
4
|
+
import { evaluateRuntimeRun, finishRuntimeRun, materializeTaskInputs, replanRuntimeRun, runRuntimeAgenticWorkflow, runRuntimeParallelPlan, shouldUseParallelScheduler } from './runner.js';
|
|
5
|
+
|
|
6
|
+
test('validated tasks waiting for approval stay in the scheduler instead of falling back to Donna', () => {
|
|
7
|
+
assert.equal(shouldUseParallelScheduler([
|
|
8
|
+
{
|
|
9
|
+
id: 'ingest-plan',
|
|
10
|
+
requiredCapability: 'knowledge.update',
|
|
11
|
+
operation: 'ingest_plan',
|
|
12
|
+
status: 'waiting_approval',
|
|
13
|
+
},
|
|
14
|
+
{
|
|
15
|
+
id: 'ingest-apply',
|
|
16
|
+
requiredCapability: 'knowledge.update',
|
|
17
|
+
operation: 'ingest_apply',
|
|
18
|
+
status: 'waiting_approval',
|
|
19
|
+
},
|
|
20
|
+
]), true);
|
|
21
|
+
assert.equal(shouldUseParallelScheduler([
|
|
22
|
+
{ id: 'finished', requiredCapability: 'knowledge.update', operation: 'ingest', status: 'done' },
|
|
23
|
+
]), false);
|
|
24
|
+
});
|
|
25
|
+
|
|
26
|
+
test('downstream tasks consume actual dependency outputs instead of planned placeholder refs', () => {
|
|
27
|
+
const plannedRef = '.wiki/ingest-plans/planned.json';
|
|
28
|
+
const actualRef = '.wiki/ingest-plans/ingest-2026-07-10.json';
|
|
29
|
+
const plan = [{
|
|
30
|
+
id: 'ingest-plan',
|
|
31
|
+
expectedOutputRefs: [{ type: 'file', ref: plannedRef }],
|
|
32
|
+
outputRefs: [{ type: 'file', ref: actualRef }],
|
|
33
|
+
}];
|
|
34
|
+
const apply = {
|
|
35
|
+
id: 'ingest-apply',
|
|
36
|
+
dependsOn: ['ingest-plan'],
|
|
37
|
+
arguments: { inputs: [plannedRef] },
|
|
38
|
+
inputRefs: [{ type: 'file', ref: plannedRef }],
|
|
39
|
+
};
|
|
40
|
+
|
|
41
|
+
const executable = materializeTaskInputs(apply, plan);
|
|
42
|
+
assert.deepEqual(executable.arguments.inputs, [actualRef]);
|
|
43
|
+
assert.equal(executable.inputRefs[0].ref, actualRef);
|
|
44
|
+
assert.deepEqual(apply.arguments.inputs, [plannedRef], 'the validated plan declaration remains immutable');
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
test('evaluateRuntimeRun trusts a completed provider TaskGraph without asking an LLM to reinterpret it', async () => {
|
|
48
|
+
const session = {
|
|
49
|
+
headlessPlan: [{
|
|
50
|
+
id: 'ingest-a',
|
|
51
|
+
label: 'Ingest A.md',
|
|
52
|
+
requiredCapability: 'knowledge.update',
|
|
53
|
+
operation: 'ingest_plan',
|
|
54
|
+
status: 'done',
|
|
55
|
+
}],
|
|
56
|
+
llm: {
|
|
57
|
+
async completeWithTools() {
|
|
58
|
+
assert.fail('a structured terminal TaskGraph must not be reinterpreted by an LLM');
|
|
59
|
+
},
|
|
60
|
+
},
|
|
61
|
+
};
|
|
62
|
+
|
|
63
|
+
const evaluation = await evaluateRuntimeRun(session, 'ingère A.md');
|
|
64
|
+
|
|
65
|
+
assert.equal(evaluation.ok, true);
|
|
66
|
+
assert.match(evaluation.reason, /1 tâche/);
|
|
67
|
+
});
|
|
5
68
|
|
|
6
69
|
test('runRuntimeAgenticWorkflow completes conversational turns without evaluation or replan', async () => {
|
|
7
70
|
const events = [];
|
|
@@ -713,7 +776,9 @@ test('runRuntimeParallelPlan fails cleanly when scheduler budget is exceeded', a
|
|
|
713
776
|
assert.equal(result.budgetExceeded, true);
|
|
714
777
|
assert.equal(result.reason, 'max_tasks_exceeded');
|
|
715
778
|
assert.ok(session.agentEvents.some((event) => event.type === 'run_error' && event.payload?.budget?.reason === 'max_tasks_exceeded'));
|
|
716
|
-
|
|
779
|
+
// run_error now cancels leftover pending steps: a dead run must not leave
|
|
780
|
+
// ghost work in the panels or reappear at the next relaunch.
|
|
781
|
+
assert.equal(session.headlessPlan[0].status, 'cancelled');
|
|
717
782
|
});
|
|
718
783
|
|
|
719
784
|
test('runRuntimeParallelPlan retries a retryable task on a fallback agent', async () => {
|
|
@@ -915,3 +980,36 @@ async function waitFor(predicate, timeoutMs = 500) {
|
|
|
915
980
|
}
|
|
916
981
|
assert.fail('condition was not met before timeout');
|
|
917
982
|
}
|
|
983
|
+
|
|
984
|
+
test('runtime runs are seeded with the chat that preceded them', async () => {
|
|
985
|
+
// "Donna n'a pas de contexte" was literally true: runs started with an
|
|
986
|
+
// empty history, so the model reinvented the missing context. The seed
|
|
987
|
+
// carries the prior exchanges, minus the run's own triggering message.
|
|
988
|
+
const { conversationSeed } = await import('./runner.js');
|
|
989
|
+
const session = {
|
|
990
|
+
agentProjection: {
|
|
991
|
+
conversation: [
|
|
992
|
+
{ role: 'user', content: 'peux-tu configurer le CME ?' },
|
|
993
|
+
{ role: 'assistant', content: 'CME configuré sur confluent.meteo.fr. Veux-tu ingérer les documents en attente ?' },
|
|
994
|
+
{ role: 'user', content: 'oui lance l\'ingestion' },
|
|
995
|
+
],
|
|
996
|
+
},
|
|
997
|
+
};
|
|
998
|
+
|
|
999
|
+
const seed = conversationSeed(session, "oui lance l'ingestion");
|
|
1000
|
+
assert.deepEqual(seed.map((message) => message.role), ['user', 'assistant']);
|
|
1001
|
+
assert.match(seed[1].content, /confluent\.meteo\.fr/);
|
|
1002
|
+
|
|
1003
|
+
// Long entries are clipped, empty/technical roles dropped.
|
|
1004
|
+
const noisy = {
|
|
1005
|
+
agentProjection: {
|
|
1006
|
+
conversation: [
|
|
1007
|
+
{ role: 'command', content: 'Runtime is idle.' },
|
|
1008
|
+
{ role: 'assistant', content: 'x'.repeat(5000) },
|
|
1009
|
+
],
|
|
1010
|
+
},
|
|
1011
|
+
};
|
|
1012
|
+
const clipped = conversationSeed(noisy, 'autre demande');
|
|
1013
|
+
assert.equal(clipped.length, 1);
|
|
1014
|
+
assert.equal(clipped[0].content.length, 2000);
|
|
1015
|
+
});
|