@dotdrelle/wiki-manager 0.12.11 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/.env.example +6 -0
  2. package/docker-compose.yml +1 -1
  3. package/package.json +1 -1
  4. package/src/agent/graph.js +377 -142
  5. package/src/agent/graph.test.js +576 -34
  6. package/src/agent/llm.js +5 -5
  7. package/src/cli/wiki-manager.js +294 -9
  8. package/src/cli/wiki-manager.test.js +28 -0
  9. package/src/commands/slash.js +80 -13
  10. package/src/commands/slash.test.js +9 -1
  11. package/src/contracts/schemas.js +33 -0
  12. package/src/contracts/schemas.test.js +14 -0
  13. package/src/core/agentEvents.js +6 -0
  14. package/src/core/agentEvents.test.js +26 -0
  15. package/src/core/agentLoop.js +15 -16
  16. package/src/core/agentLoop.test.js +9 -7
  17. package/src/core/buildInfo.json +2 -2
  18. package/src/core/mcp.js +13 -6
  19. package/src/core/mcp.test.js +0 -12
  20. package/src/core/skills.js +0 -28
  21. package/src/orchestrator/capabilityRegistry.js +14 -0
  22. package/src/orchestrator/capabilityRegistry.test.js +12 -1
  23. package/src/orchestrator/dependencyResolver.js +10 -1
  24. package/src/orchestrator/dispatcher.js +34 -3
  25. package/src/orchestrator/dispatcher.test.js +34 -0
  26. package/src/orchestrator/objectiveResolver.js +79 -0
  27. package/src/orchestrator/objectiveResolver.test.js +50 -0
  28. package/src/orchestrator/scheduler.js +24 -0
  29. package/src/orchestrator/scheduler.test.js +65 -1
  30. package/src/runtime/client.js +34 -2
  31. package/src/runtime/lifecycle.js +1 -1
  32. package/src/runtime/recoveryManager.js +14 -7
  33. package/src/runtime/runner.js +214 -14
  34. package/src/runtime/runner.test.js +100 -2
  35. package/src/runtime/server.js +43 -3
  36. package/src/runtime/supervisor.js +65 -1
  37. package/src/runtime/supervisor.test.js +80 -0
  38. package/src/shell/LeftPane.tsx +9 -2
  39. package/src/shell/repl.js +57 -42
  40. package/src/shell/repl.test.js +81 -12
  41. package/src/shell/tui.tsx +26 -3
  42. package/src/shell/useSession.ts +15 -3
@@ -52,6 +52,7 @@ export async function postRuntimeRun(input, {
52
52
  workspace = null,
53
53
  evaluate = undefined,
54
54
  replans = undefined,
55
+ capabilityPlan = undefined,
55
56
  } = {}) {
56
57
  const response = await fetch(runtimeEndpoint(url, '/run', workspace), {
57
58
  method: 'POST',
@@ -59,7 +60,7 @@ export async function postRuntimeRun(input, {
59
60
  ...runtimeHeaders(token),
60
61
  'Content-Type': 'application/json',
61
62
  },
62
- body: JSON.stringify(Object.assign({ input, workspace }, evaluate !== undefined && { evaluate }, replans !== undefined && { replans })),
63
+ body: JSON.stringify(Object.assign({ input, workspace }, evaluate !== undefined && { evaluate }, replans !== undefined && { replans }, capabilityPlan !== undefined && { capabilityPlan })),
63
64
  });
64
65
  if (!response.ok) {
65
66
  const err = new Error(`Runtime run failed: HTTP ${response.status}`);
@@ -69,6 +70,25 @@ export async function postRuntimeRun(input, {
69
70
  return response.json();
70
71
  }
71
72
 
73
+ export async function postRuntimeDelegate(objective, {
74
+ url = runtimeUrlFromEnv(),
75
+ token = runtimeToken(),
76
+ workspace = null,
77
+ } = {}) {
78
+ const response = await fetch(runtimeEndpoint(url, '/delegate', workspace), {
79
+ method: 'POST',
80
+ headers: { ...runtimeHeaders(token), 'Content-Type': 'application/json' },
81
+ body: JSON.stringify({ objective, workspace }),
82
+ });
83
+ const payload = await response.json().catch(() => ({}));
84
+ if (!response.ok) {
85
+ const err = new Error(payload.error ?? `Runtime delegation failed: HTTP ${response.status}`);
86
+ err.status = response.status;
87
+ throw err;
88
+ }
89
+ return payload;
90
+ }
91
+
72
92
  export async function postRuntimeControl(action, {
73
93
  url = runtimeUrlFromEnv(),
74
94
  token = runtimeToken(),
@@ -151,6 +171,9 @@ export async function postRuntimeApprove({
151
171
  runId = null,
152
172
  itemId = null,
153
173
  approvalId = null,
174
+ scope = null,
175
+ planRevision = null,
176
+ approvalClasses = null,
154
177
  } = {}) {
155
178
  const endpoint = runtimeEndpoint(url, '/approve', workspace);
156
179
  const parsed = new URL(endpoint);
@@ -159,7 +182,16 @@ export async function postRuntimeApprove({
159
182
  if (approvalId) parsed.searchParams.set('approvalId', approvalId);
160
183
  const response = await fetch(parsed.toString(), {
161
184
  method: 'POST',
162
- headers: runtimeHeaders(token),
185
+ headers: { ...runtimeHeaders(token), 'Content-Type': 'application/json' },
186
+ body: JSON.stringify({
187
+ workspace,
188
+ runId,
189
+ itemId,
190
+ approvalId,
191
+ scope,
192
+ planRevision,
193
+ approvalClasses,
194
+ }),
163
195
  });
164
196
  if (!response.ok) throw new Error(`Runtime approve failed: HTTP ${response.status}`);
165
197
  return response.json();
@@ -109,7 +109,7 @@ export async function runtimeHealthOrNull(url = runtimeUrlFromEnv(), token = run
109
109
  // alive after exit produced zombie runtimes running yesterday's code and
110
110
  // yesterday's endpoints. Nuance preserved: if a run is active anywhere, the
111
111
  // runtime is left alive so the run survives the shell (that promise stays).
112
- export async function shutdownOwnedRuntime(runtime, { log = () => {} } = {}) {
112
+ export async function shutdownOwnedRuntime(runtime, { log = (_message) => {} } = {}) {
113
113
  if (!runtime?.url || !runtime?.started) return { action: 'kept', reason: 'not_owned' };
114
114
  try {
115
115
  const health = await runtimeHealthOrNull(runtime.url, runtime.token);
@@ -37,18 +37,25 @@ export async function recoverActiveRuns({
37
37
  errors.push({ runId: run.id, taskId: task.id, error: error instanceof Error ? error.message : String(error) });
38
38
  }
39
39
  }
40
- // A run in which nothing was recovered or rescheduled can never progress:
41
- // leaving it 'running' in the store would re-attach it as a zombie on
42
- // every subsequent boot. Close it for good.
43
- if (activeTasks.length > 0 && runOutcomes.length === activeTasks.length
44
- && runOutcomes.every((outcome) => outcome?.status === 'interrupted')) {
45
- const changed = store.interruptRuns?.({ workspace: run.workspace ?? null, runId: run.id, reason: 'Recovery found no recoverable task.' }) ?? 0;
40
+ // A run that recovery cannot move forward must be closed for good, or it
41
+ // re-attaches as a blocking "a runtime run is already active" zombie on
42
+ // every boot. This covers BOTH cases that can never resume on a fresh boot:
43
+ // - runs whose active tasks were all interrupted, and
44
+ // - runs with no active task to recover at all (e.g. left waiting for
45
+ // approval, or with only un-started pending tasks). Nothing here will
46
+ // ever progress, so finalize it now.
47
+ const progressed = runOutcomes.some((outcome) => outcome?.status === 'recovered' || outcome?.status === 'rescheduled');
48
+ if (!progressed) {
49
+ const reason = activeTasks.length > 0
50
+ ? 'Recovery found no recoverable task.'
51
+ : 'Recovery found no active task to resume.';
52
+ const changed = store.interruptRuns?.({ workspace: run.workspace ?? null, runId: run.id, reason }) ?? 0;
46
53
  if (changed > 0) {
47
54
  dispatch(session, store, 'runtime_log', {
48
55
  origin: 'recovery_manager',
49
56
  runId: run.id,
50
57
  workspace: run.workspace ?? workspaceFromSession(session),
51
- payload: { message: `recovery: run ${run.id} interrupted (no recoverable task)` },
58
+ payload: { message: `recovery: run ${run.id} interrupted (${reason})` },
52
59
  });
53
60
  }
54
61
  }
@@ -9,10 +9,16 @@ import { createBudgetManager, BudgetExceededError } from '../orchestrator/budget
9
9
  import { createDispatcher } from '../orchestrator/dispatcher.js';
10
10
  import { assertValidatedFragment } from '../orchestrator/planValidator.js';
11
11
  import { createResultAggregator } from '../orchestrator/resultAggregator.js';
12
- import { drainActive, resolveSchedulerConcurrency, startReadyTasks } from '../orchestrator/scheduler.js';
12
+ import { drainActive, resolvePlanConcurrency, startReadyTasks } from '../orchestrator/scheduler.js';
13
13
  import { emitRuntimeLog, pollActivitiesOnce } from './supervisor.js';
14
14
 
15
- const DEFAULT_MAX_REPLANS = 2;
15
+ // 0 by default: automatic replans turn evaluator/replanner TEXT into
16
+ // executable pseudo-tasks (no capability, no operation) that stall at 0%
17
+ // and pile up as replan-1/2/3 ghost work — the same disease as the removed
18
+ // text-plan extraction. Failures now end with an honest report; the user
19
+ // (or a stronger model) decides what to do next. Re-enable explicitly with
20
+ // WIKI_MANAGER_REPLANNER_MAX_REPLANS if desired.
21
+ const DEFAULT_MAX_REPLANS = 0;
16
22
 
17
23
  async function waitForRuntimeActivities(session, startedActivities, { timeoutMs, signal, pollBusy }) {
18
24
  const deadline = Date.now() + timeoutMs;
@@ -43,13 +49,35 @@ async function waitForRuntimeActivities(session, startedActivities, { timeoutMs,
43
49
  return { ok: false, timedOut: true, completed: tracked };
44
50
  }
45
51
 
46
- export async function runRuntimeAgenticLoop(agent, session, initialInput, { signal, timeoutMs, maxTurns, runId, pollBusy, parallelHandoff = false }) {
52
+ // Last chat exchanges (user/assistant) that preceded this run, so the run's
53
+ // LLM knows WHAT was agreed before acting. Long messages are clipped: the
54
+ // context is for grounding, not for re-reading novels.
55
+ // Env knobs (documented in .env.example): every tunable introduced by the
56
+ // grounding/orchestration work is overridable — nothing business-critical
57
+ // is frozen in code.
58
+ export function conversationSeed(session, currentInput, { limit = 12, maxChars = 2000 } = {}) {
59
+ const conversation = Array.isArray(session.agentProjection?.conversation)
60
+ ? session.agentProjection.conversation
61
+ : [];
62
+ const seed = conversation
63
+ .filter((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim())
64
+ .slice(-limit)
65
+ .map((message) => ({ role: message.role, content: String(message.content).slice(0, maxChars) }));
66
+ // The run's own triggering user message is appended by the loop itself —
67
+ // drop it from the seed to avoid sending it twice.
68
+ const last = seed.at(-1);
69
+ if (last && last.role === 'user' && last.content === String(currentInput ?? '').slice(0, maxChars)) seed.pop();
70
+ return seed;
71
+ }
72
+
73
+ export async function runRuntimeAgenticLoop(agent, session, initialInput, { signal, timeoutMs, maxTurns, runId, pollBusy, parallelHandoff = false, initialMessages = [] }) {
47
74
  return runAgenticLoop(agent, session, initialInput, {
48
75
  signal,
49
76
  timeoutMs,
50
77
  maxTurns,
51
78
  runId,
52
79
  parallelHandoff,
80
+ initialMessages,
53
81
  deterministicTerminalSummary: true,
54
82
  abortMessage: 'Runtime run cancelled.',
55
83
  waitForActivities: (turnSession, startedActivities, waitOptions) =>
@@ -103,10 +131,13 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
103
131
  evaluate = true,
104
132
  maxReplans = resolveMaxReplans(),
105
133
  callTool = null,
106
- dispatcherPollIntervalMs = 250,
134
+ dispatcherPollIntervalMs = 2500,
107
135
  } = {}) {
108
136
  let currentInput = initialInput ?? input;
109
137
  let replansLeft = Math.max(0, Math.floor(Number(maxReplans) || 0));
138
+ // Computed ONCE at run start: the pre-run chat. Re-computing inside the
139
+ // loop would re-ingest this run's own turns and duplicate them.
140
+ const runConversationSeed = conversationSeed(session, currentInput);
110
141
 
111
142
  while (true) {
112
143
  sanitizeSessionPlanForExecution(session, runId);
@@ -127,6 +158,7 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
127
158
  runId,
128
159
  pollBusy,
129
160
  parallelHandoff: true,
161
+ initialMessages: runConversationSeed,
130
162
  });
131
163
  if (result.ok && result.handoff) continue;
132
164
  if (!result.ok) {
@@ -212,6 +244,16 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
212
244
  continue;
213
245
  }
214
246
  }
247
+ // Surface the verdict in the CHAT: the work that ran stays done, the
248
+ // user sees why the evaluator was unsatisfied and decides — no
249
+ // self-generated follow-up tasks.
250
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', {
251
+ origin: 'runtime',
252
+ runId,
253
+ payload: {
254
+ content: `Le run est terminé mais l'évaluation le juge incomplet : ${evaluation.reason}`,
255
+ },
256
+ }));
215
257
  dispatchAgentEvent(session, createAgentEvent('run_error', {
216
258
  origin: 'runtime',
217
259
  runId,
@@ -240,7 +282,7 @@ export async function runRuntimeParallelPlan(agent, session, input, {
240
282
  maxTurns,
241
283
  runId = null,
242
284
  pollBusy,
243
- concurrency = resolveSchedulerConcurrency(),
285
+ concurrency = null,
244
286
  fragment = null,
245
287
  assignmentManager = null,
246
288
  attemptManager = null,
@@ -249,10 +291,18 @@ export async function runRuntimeParallelPlan(agent, session, input, {
249
291
  budgetManager = null,
250
292
  budgets = {},
251
293
  callTool = null,
252
- dispatcherPollIntervalMs = 250,
294
+ dispatcherPollIntervalMs = 2500,
253
295
  } = {}) {
254
296
  if (fragment != null) assertValidatedFragment(fragment);
255
- const limit = Math.max(1, Math.floor(Number(concurrency) || 1));
297
+ const agents = session.agentRegistry?.snapshot?.() ?? session.agentRegistrySnapshot ?? [];
298
+ const configuredConcurrency = Number(concurrency) > 0
299
+ ? Number(concurrency)
300
+ : Number(process.env.WIKI_MANAGER_CAPABILITY_CONCURRENCY || process.env.WIKI_MANAGER_SCHEDULER_CONCURRENCY);
301
+ const limit = resolvePlanConcurrency({
302
+ plan: session.headlessPlan ?? [],
303
+ agents,
304
+ configured: configuredConcurrency,
305
+ });
256
306
  const active = new Map();
257
307
  const attempts = attemptManager ?? createAttemptManager();
258
308
  const assigner = assignmentManager ?? createAssignmentManager({ session });
@@ -278,6 +328,15 @@ export async function runRuntimeParallelPlan(agent, session, input, {
278
328
  sanitizeSessionPlanForExecution(session, runId);
279
329
  ensurePlanProjection(session, runId);
280
330
  emitRuntimeLog(session, `scheduler: parallel plan enabled (concurrency ${limit})`);
331
+ let approvalNoticeSent = false;
332
+ // Interactive approvals do NOT expire: the user has /approve, "valide
333
+ // tout", /cancel and /run kill — an arbitrary timer only created mystery
334
+ // failures. A deadline exists only when explicitly configured (headless
335
+ // runs, CI) via the session or the env escape hatch.
336
+ const configuredApprovalWait = Number(session._approvalTimeoutMs) > 0
337
+ ? Number(session._approvalTimeoutMs)
338
+ : (Number(process.env.WIKI_MANAGER_APPROVAL_TIMEOUT_MS) > 0 ? Number(process.env.WIKI_MANAGER_APPROVAL_TIMEOUT_MS) : null);
339
+ const approvalDeadline = configuredApprovalWait ? Date.now() + configuredApprovalWait : Infinity;
281
340
 
282
341
  try {
283
342
  while (true) {
@@ -336,8 +395,9 @@ export async function runRuntimeParallelPlan(agent, session, input, {
336
395
  emitRuntimeLog(session, taskLogPayload('task.starting', task, { runId, detail: `starting task ${taskId}` }));
337
396
  },
338
397
  startTask: (task, attempt) => {
398
+ const executableTask = materializeTaskInputs(task, session.headlessPlan ?? []);
339
399
  const taskAbort = createTaskAbortSignal(signal);
340
- const promise = runDispatchedTask(task, {
400
+ const promise = runDispatchedTask(executableTask, {
341
401
  session,
342
402
  assignmentManager: assigner,
343
403
  dispatcher: executor,
@@ -371,6 +431,44 @@ export async function runRuntimeParallelPlan(agent, session, input, {
371
431
  emitRuntimeLog(session, `scheduler: budget exceeded (${exceeded.reason})`);
372
432
  return { ok: false, budgetExceeded: true, reason: exceeded.reason, budget: exceeded, completed: sessionActivities(session), failures };
373
433
  }
434
+ const needingApproval = pending.filter((step) => step.requiresApproval === true);
435
+ if (needingApproval.length > 0) {
436
+ // The plan is only blocked on a HUMAN decision — wait for it
437
+ // (bounded) instead of declaring the run stalled. Announce once in
438
+ // the chat: users cannot approve what they never saw asked.
439
+ if (!approvalNoticeSent) {
440
+ approvalNoticeSent = true;
441
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', {
442
+ origin: 'runtime',
443
+ runId,
444
+ payload: {
445
+ content: [
446
+ `⏸ Approbation requise avant exécution : ${needingApproval.length} tâche(s) mutante(s) en attente.`,
447
+ ...needingApproval.slice(0, 5).map((step) => ` - ${step.description ?? step.id}`),
448
+ needingApproval.length > 5 ? ` … et ${needingApproval.length - 5} autre(s).` : null,
449
+ 'Réponds « valide tout » (ou tape /approve) pour lancer, « annule » pour abandonner.',
450
+ ].filter(Boolean).join('\n'),
451
+ },
452
+ }));
453
+ emitRuntimeLog(session, `scheduler: waiting for approval (${needingApproval.length} task(s))`);
454
+ }
455
+ if (Date.now() < approvalDeadline) {
456
+ await new Promise((resolveDelay) => setTimeout(resolveDelay, 500));
457
+ continue;
458
+ }
459
+ // Configured timeout (headless/CI) reached: say it PLAINLY in the
460
+ // chat and let run_error clean the plan/activities so nothing
461
+ // lingers in the panels.
462
+ emitRuntimeLog(session, 'scheduler: approval wait timed out');
463
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', {
464
+ origin: 'runtime',
465
+ runId,
466
+ payload: {
467
+ content: `⏱ Approbation non reçue dans le délai imparti — run arrêté, ${needingApproval.length} tâche(s) annulée(s). Relance la demande quand tu veux.`,
468
+ },
469
+ }));
470
+ return { ok: false, stalled: true, reason: 'awaiting_approval', completed: sessionActivities(session), failures };
471
+ }
374
472
  const reason = pending.every((step) => approvalWaitingStatus(step.status)) ? 'awaiting_approval' : 'no_ready_plan_task';
375
473
  emitRuntimeLog(session, `scheduler: stalled (${reason})`);
376
474
  return { ok: false, stalled: true, reason, completed: sessionActivities(session), failures };
@@ -428,6 +526,41 @@ export async function runRuntimeParallelPlan(agent, session, input, {
428
526
  }
429
527
  }
430
528
 
529
+ export function materializeTaskInputs(task, plan = []) {
530
+ const dependencies = new Set(Array.isArray(task?.dependsOn) ? task.dependsOn.map(String) : []);
531
+ const replacements = new Map();
532
+ for (const dependency of plan ?? []) {
533
+ if (!dependencies.has(String(dependency?.id ?? dependency?.step))) continue;
534
+ const expected = Array.isArray(dependency?.expectedOutputRefs) ? dependency.expectedOutputRefs : [];
535
+ const actual = Array.isArray(dependency?.outputRefs) ? dependency.outputRefs : [];
536
+ for (let index = 0; index < Math.min(expected.length, actual.length); index += 1) {
537
+ const expectedRef = refValue(expected[index]);
538
+ const actualRef = refValue(actual[index]);
539
+ if (expectedRef && actualRef && expectedRef !== actualRef) replacements.set(expectedRef, actualRef);
540
+ }
541
+ }
542
+ if (replacements.size === 0) return task;
543
+ return {
544
+ ...task,
545
+ arguments: replaceRefValues(task?.arguments, replacements),
546
+ inputRefs: replaceRefValues(task?.inputRefs, replacements),
547
+ };
548
+ }
549
+
550
+ function replaceRefValues(value, replacements) {
551
+ if (typeof value === 'string') return replacements.get(value) ?? value;
552
+ if (Array.isArray(value)) return value.map((item) => replaceRefValues(item, replacements));
553
+ if (value && typeof value === 'object') {
554
+ return Object.fromEntries(Object.entries(value).map(([key, item]) => [key, replaceRefValues(item, replacements)]));
555
+ }
556
+ return value;
557
+ }
558
+
559
+ function refValue(value) {
560
+ if (typeof value === 'string') return value;
561
+ return value && typeof value === 'object' ? String(value.ref ?? '') : '';
562
+ }
563
+
431
564
  function sanitizeSessionPlanForExecution(session, runId = null) {
432
565
  if (!session.headlessPlan) return;
433
566
  const sanitized = sanitizePlanForExecution(session.headlessPlan);
@@ -603,7 +736,7 @@ async function runDispatchedTask(task, {
603
736
  error: {
604
737
  code: 'dispatcher_error',
605
738
  message: err instanceof Error ? err.message : String(err),
606
- retryable: false,
739
+ retryable: transientRuntimeError(err),
607
740
  },
608
741
  };
609
742
  await resultAggregator.accept(result, { task, assignment });
@@ -623,8 +756,16 @@ async function runDispatchedTask(task, {
623
756
  }
624
757
  }
625
758
 
626
- function shouldUseParallelScheduler(plan) {
627
- return readyPlanTasks(plan).some((task) => task.requiredCapability && task.operation);
759
+ export function shouldUseParallelScheduler(plan) {
760
+ // A validated provider plan enters the scheduler even while every task is
761
+ // waiting for approval. Looking only at readyPlanTasks() made an all-
762
+ // approval plan fall through to the conversational Donna loop; that loop
763
+ // then ignored the integrated TaskGraph and marked the run done.
764
+ return (plan ?? []).some((task) =>
765
+ task?.requiredCapability
766
+ && task?.operation
767
+ && pendingSchedulerStatus(task.status),
768
+ );
628
769
  }
629
770
 
630
771
  function taskLogPayload(event, task, {
@@ -718,6 +859,11 @@ export async function evaluateRuntimeRun(session, input, {
718
859
  evaluate = true,
719
860
  } = {}) {
720
861
  if (!shouldEvaluate(evaluate)) return null;
862
+ const structured = structuredPlanEvaluation(session.headlessPlan);
863
+ if (structured) {
864
+ emitRuntimeLog(session, 'runtime: evaluating completed structured plan');
865
+ return structured;
866
+ }
721
867
  const llm = session.llm;
722
868
  if (!llm || typeof llm.completeWithTools !== 'function') {
723
869
  return fallbackEvaluation('Evaluator unavailable: no LLM completeWithTools client.');
@@ -741,6 +887,40 @@ export async function evaluateRuntimeRun(session, input, {
741
887
  }
742
888
  }
743
889
 
890
+ function structuredPlanEvaluation(plan) {
891
+ if (!Array.isArray(plan) || plan.length === 0) return null;
892
+ // Only provider TaskGraph tasks are authoritative. Legacy conversational
893
+ // plans contain prose/tool labels and still use the compatibility evaluator.
894
+ if (!plan.every((step) => step?.requiredCapability && step?.operation)) return null;
895
+ const statuses = plan.map((step) => String(step?.status ?? '').toLowerCase());
896
+ const failed = plan.filter((step) => ['failed', 'error', 'cancelled', 'canceled', 'stalled'].includes(String(step?.status ?? '').toLowerCase()));
897
+ const incomplete = plan.filter((step) => !['done', 'complete', 'completed', 'success', 'succeeded'].includes(String(step?.status ?? '').toLowerCase()));
898
+ if (failed.length > 0) {
899
+ return {
900
+ ok: false,
901
+ reason: `${failed.length} tâche(s) du plan ont échoué : ${failed.map((step) => step.label ?? step.description ?? step.id ?? step.step).join(', ')}.`,
902
+ suggestedAction: null,
903
+ };
904
+ }
905
+ if (incomplete.length > 0) {
906
+ return {
907
+ ok: false,
908
+ reason: `${incomplete.length} tâche(s) du plan ne sont pas terminées.`,
909
+ suggestedAction: null,
910
+ };
911
+ }
912
+ return {
913
+ ok: statuses.length > 0,
914
+ reason: `${statuses.length} tâche(s) du plan terminées avec succès.`,
915
+ suggestedAction: null,
916
+ };
917
+ }
918
+
919
+ function transientRuntimeError(error) {
920
+ const value = error instanceof Error ? error.message : String(error ?? '');
921
+ return /(?:429|timeout|temporar|throttl|rate.?limit|quota|busy|unavailable)/i.test(value);
922
+ }
923
+
744
924
  function shouldEvaluate(value) {
745
925
  if (value === false) return false;
746
926
  const env = String(process.env.WIKI_MANAGER_EVALUATOR ?? '').trim().toLowerCase();
@@ -860,10 +1040,30 @@ export async function replanRuntimeRun(session, input, trigger, {
860
1040
  }
861
1041
  }
862
1042
 
1043
+ function isBusyFailure(failure) {
1044
+ const fields = [
1045
+ failure?.error,
1046
+ failure?.status,
1047
+ failure?.result?.error?.code,
1048
+ failure?.result?.error?.message,
1049
+ failure?.result?.status,
1050
+ ].map((value) => String(value ?? '').toLowerCase());
1051
+ return fields.some((value) => value.includes('busy') || value.includes('locked'));
1052
+ }
1053
+
863
1054
  function replanTriggerFromLoopResult(result) {
864
1055
  // 'awaiting_approval' is not a dead end — it means to wait for a human
865
1056
  // decision, not to replan around it.
866
1057
  if (result.stalled && result.reason !== 'awaiting_approval') {
1058
+ // A stall caused only by transient lock contention (target_busy /
1059
+ // workspace_busy) must NOT be replanned: the plan is correct, the
1060
+ // workspace was momentarily locked. Replanning around it re-runs the same
1061
+ // tasks, hits the lock again and spins the run into a zombie. Fail cleanly
1062
+ // so the run finalizes instead of lingering 'running'.
1063
+ const blocking = [...(result.failures ?? []), ...terminalFailures(result.completed ?? [])];
1064
+ if (blocking.length > 0 && blocking.every(isBusyFailure)) {
1065
+ return null;
1066
+ }
867
1067
  return {
868
1068
  kind: 'plan_stalled',
869
1069
  reason: `Plan is stalled: ${result.reason ?? 'no ready task'} (pending steps exist but none have their dependencies satisfied).`,
@@ -872,7 +1072,7 @@ function replanTriggerFromLoopResult(result) {
872
1072
  };
873
1073
  }
874
1074
  const failures = terminalFailures(result.completed ?? []);
875
- const technical = failures.filter((failure) => !isCancelledStatus(failure.status));
1075
+ const technical = failures.filter((failure) => !isCancelledStatus(failure.status) && !isBusyFailure(failure));
876
1076
  const cancelled = failures.filter((failure) => isCancelledStatus(failure.status));
877
1077
  if (cancelled.length > 0 && technical.length === 0 && (result.failures ?? []).every(isCancelledTaskFailure)) {
878
1078
  return null;
@@ -889,7 +1089,7 @@ function replanTriggerFromLoopResult(result) {
889
1089
  // The parallel scheduler can fail a task before it ever produces an
890
1090
  // _activity (e.g. a thrown error on the first turn) — that failure lives
891
1091
  // in result.failures, not in any activity, so it must be checked too.
892
- const taskFailure = (result.failures ?? []).find((failure) => !isCancelledTaskFailure(failure));
1092
+ const taskFailure = (result.failures ?? []).find((failure) => !isCancelledTaskFailure(failure) && !isBusyFailure(failure));
893
1093
  if (!taskFailure) return null;
894
1094
  return {
895
1095
  kind: 'task_error',
@@ -1016,7 +1216,7 @@ function formatRecentConversation(session, n = 12) {
1016
1216
  .join('\n');
1017
1217
  }
1018
1218
 
1019
- function resolveMaxReplans(value = process.env.WIKI_MANAGER_REPLANNER_MAX_REPLANS) {
1219
+ function resolveMaxReplans(value) {
1020
1220
  const parsed = Number(value);
1021
1221
  return Number.isFinite(parsed) ? Math.max(0, Math.floor(parsed)) : DEFAULT_MAX_REPLANS;
1022
1222
  }
@@ -1,7 +1,70 @@
1
1
  import assert from 'node:assert/strict';
2
2
  import test from 'node:test';
3
3
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
4
- import { finishRuntimeRun, replanRuntimeRun, runRuntimeAgenticWorkflow, runRuntimeParallelPlan } from './runner.js';
4
+ import { evaluateRuntimeRun, finishRuntimeRun, materializeTaskInputs, replanRuntimeRun, runRuntimeAgenticWorkflow, runRuntimeParallelPlan, shouldUseParallelScheduler } from './runner.js';
5
+
6
+ test('validated tasks waiting for approval stay in the scheduler instead of falling back to Donna', () => {
7
+ assert.equal(shouldUseParallelScheduler([
8
+ {
9
+ id: 'ingest-plan',
10
+ requiredCapability: 'knowledge.update',
11
+ operation: 'ingest_plan',
12
+ status: 'waiting_approval',
13
+ },
14
+ {
15
+ id: 'ingest-apply',
16
+ requiredCapability: 'knowledge.update',
17
+ operation: 'ingest_apply',
18
+ status: 'waiting_approval',
19
+ },
20
+ ]), true);
21
+ assert.equal(shouldUseParallelScheduler([
22
+ { id: 'finished', requiredCapability: 'knowledge.update', operation: 'ingest', status: 'done' },
23
+ ]), false);
24
+ });
25
+
26
+ test('downstream tasks consume actual dependency outputs instead of planned placeholder refs', () => {
27
+ const plannedRef = '.wiki/ingest-plans/planned.json';
28
+ const actualRef = '.wiki/ingest-plans/ingest-2026-07-10.json';
29
+ const plan = [{
30
+ id: 'ingest-plan',
31
+ expectedOutputRefs: [{ type: 'file', ref: plannedRef }],
32
+ outputRefs: [{ type: 'file', ref: actualRef }],
33
+ }];
34
+ const apply = {
35
+ id: 'ingest-apply',
36
+ dependsOn: ['ingest-plan'],
37
+ arguments: { inputs: [plannedRef] },
38
+ inputRefs: [{ type: 'file', ref: plannedRef }],
39
+ };
40
+
41
+ const executable = materializeTaskInputs(apply, plan);
42
+ assert.deepEqual(executable.arguments.inputs, [actualRef]);
43
+ assert.equal(executable.inputRefs[0].ref, actualRef);
44
+ assert.deepEqual(apply.arguments.inputs, [plannedRef], 'the validated plan declaration remains immutable');
45
+ });
46
+
47
+ test('evaluateRuntimeRun trusts a completed provider TaskGraph without asking an LLM to reinterpret it', async () => {
48
+ const session = {
49
+ headlessPlan: [{
50
+ id: 'ingest-a',
51
+ label: 'Ingest A.md',
52
+ requiredCapability: 'knowledge.update',
53
+ operation: 'ingest_plan',
54
+ status: 'done',
55
+ }],
56
+ llm: {
57
+ async completeWithTools() {
58
+ assert.fail('a structured terminal TaskGraph must not be reinterpreted by an LLM');
59
+ },
60
+ },
61
+ };
62
+
63
+ const evaluation = await evaluateRuntimeRun(session, 'ingère A.md');
64
+
65
+ assert.equal(evaluation.ok, true);
66
+ assert.match(evaluation.reason, /1 tâche/);
67
+ });
5
68
 
6
69
  test('runRuntimeAgenticWorkflow completes conversational turns without evaluation or replan', async () => {
7
70
  const events = [];
@@ -713,7 +776,9 @@ test('runRuntimeParallelPlan fails cleanly when scheduler budget is exceeded', a
713
776
  assert.equal(result.budgetExceeded, true);
714
777
  assert.equal(result.reason, 'max_tasks_exceeded');
715
778
  assert.ok(session.agentEvents.some((event) => event.type === 'run_error' && event.payload?.budget?.reason === 'max_tasks_exceeded'));
716
- assert.equal(session.headlessPlan[0].status, 'pending');
779
+ // run_error now cancels leftover pending steps: a dead run must not leave
780
+ // ghost work in the panels or reappear at the next relaunch.
781
+ assert.equal(session.headlessPlan[0].status, 'cancelled');
717
782
  });
718
783
 
719
784
  test('runRuntimeParallelPlan retries a retryable task on a fallback agent', async () => {
@@ -915,3 +980,36 @@ async function waitFor(predicate, timeoutMs = 500) {
915
980
  }
916
981
  assert.fail('condition was not met before timeout');
917
982
  }
983
+
984
+ test('runtime runs are seeded with the chat that preceded them', async () => {
985
+ // "Donna n'a pas de contexte" was literally true: runs started with an
986
+ // empty history, so the model reinvented the missing context. The seed
987
+ // carries the prior exchanges, minus the run's own triggering message.
988
+ const { conversationSeed } = await import('./runner.js');
989
+ const session = {
990
+ agentProjection: {
991
+ conversation: [
992
+ { role: 'user', content: 'peux-tu configurer le CME ?' },
993
+ { role: 'assistant', content: 'CME configuré sur confluent.meteo.fr. Veux-tu ingérer les documents en attente ?' },
994
+ { role: 'user', content: 'oui lance l\'ingestion' },
995
+ ],
996
+ },
997
+ };
998
+
999
+ const seed = conversationSeed(session, "oui lance l'ingestion");
1000
+ assert.deepEqual(seed.map((message) => message.role), ['user', 'assistant']);
1001
+ assert.match(seed[1].content, /confluent\.meteo\.fr/);
1002
+
1003
+ // Long entries are clipped, empty/technical roles dropped.
1004
+ const noisy = {
1005
+ agentProjection: {
1006
+ conversation: [
1007
+ { role: 'command', content: 'Runtime is idle.' },
1008
+ { role: 'assistant', content: 'x'.repeat(5000) },
1009
+ ],
1010
+ },
1011
+ };
1012
+ const clipped = conversationSeed(noisy, 'autre demande');
1013
+ assert.equal(clipped.length, 1);
1014
+ assert.equal(clipped[0].content.length, 2000);
1015
+ });