@dotdrelle/wiki-manager 0.11.4 → 0.11.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,359 @@
1
+ import assert from 'node:assert/strict';
2
+ import test from 'node:test';
3
+ import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
4
+ import { runRuntimeAgenticWorkflow } from './runner.js';
5
+ import { startRuntimeServer } from './server.js';
6
+
7
+ // Executable version of plan-0.11.5-hotfix-final.md §3 ("La recette — le seul
8
+ // critère de sortie qui compte"). Each test below is one row of that table,
9
+ // referenced by its recipe number. Correctifs 1-3 and 8 (runner.js's
10
+ // conversational short-circuit, plan sanitization, and no-replan-on-vague
11
+ // paths) are exercised with a mocked agent standing in for the LLM's
12
+ // per-turn decision, exactly like the existing runner.test.js unit tests --
13
+ // this file's job is traceability to the recipe, not a new test seam.
14
+ // Recipes 4-6 drive the real scheduler (runAgenticLoop / runRuntimeParallelPlan,
15
+ // imported and executed for real, never reimplemented). Recipe 7 drives the
16
+ // real control-lane HTTP endpoint. Per plan §4.1: this file is the CI gate --
17
+ // no release ships while it is red.
18
+
19
+ function baseSession(overrides = {}) {
20
+ return { activities: {}, headlessPlan: null, ...overrides };
21
+ }
22
+
23
+ function eventTypes(session) {
24
+ return (session.agentEvents ?? []).map((event) => event.type);
25
+ }
26
+
27
+ async function waitFor(predicate, timeoutMs = 500) {
28
+ const deadline = Date.now() + timeoutMs;
29
+ while (Date.now() < deadline) {
30
+ if (predicate()) return;
31
+ await new Promise((resolve) => setTimeout(resolve, 1));
32
+ }
33
+ assert.fail('condition was not met before timeout');
34
+ }
35
+
36
+ test('Recipe #1 — "salut" gets a plain reply: no plan, no activity, no job, run done', async () => {
37
+ const session = baseSession({
38
+ llm: {
39
+ async completeWithTools() {
40
+ assert.fail('a plain greeting must never reach the evaluator or replanner');
41
+ },
42
+ },
43
+ });
44
+ const agent = {
45
+ async invoke() {
46
+ return { response: 'Bonjour ! Comment puis-je vous aider ?' };
47
+ },
48
+ };
49
+
50
+ const result = await runRuntimeAgenticWorkflow(agent, session, 'salut', {
51
+ runId: 'recipe-1',
52
+ timeoutMs: 1000,
53
+ maxTurns: 1,
54
+ maxReplans: 1,
55
+ });
56
+
57
+ assert.equal(result.ok, true);
58
+ assert.equal(session.headlessPlan, null);
59
+ assert.equal(Object.keys(session.activities).length, 0);
60
+ const types = eventTypes(session);
61
+ assert.ok(types.includes('run_done'));
62
+ assert.equal(types.includes('run_evaluated'), false);
63
+ assert.equal(types.includes('run_replanned'), false);
64
+ assert.equal(types.includes('tool_call_started'), false);
65
+ assert.equal(types.includes('plan_set'), false);
66
+ });
67
+
68
+ test('Recipe #2 — "quel est le profil actif ?" answers in conversation, never poses a plan', async () => {
69
+ const session = baseSession({
70
+ llm: {
71
+ async completeWithTools() {
72
+ assert.fail('a read-only config question must never reach the evaluator or replanner');
73
+ },
74
+ },
75
+ });
76
+ const agent = {
77
+ async invoke({ session: turnSession }) {
78
+ dispatchAgentEvent(turnSession, createAgentEvent('tool_call_started', {
79
+ origin: 'tool',
80
+ payload: { name: 'shell.status', args: '{}', summary: 'calling...' },
81
+ }));
82
+ dispatchAgentEvent(turnSession, createAgentEvent('tool_call_result', {
83
+ origin: 'tool',
84
+ payload: { name: 'shell.status', ok: true, result: 'profile: albert-openai (mistral-large)', summary: 'done' },
85
+ }));
86
+ return { response: 'Le profil actif est albert-openai (mistral-large).' };
87
+ },
88
+ };
89
+
90
+ const result = await runRuntimeAgenticWorkflow(agent, session, 'quel est le profil actif ?', {
91
+ runId: 'recipe-2',
92
+ timeoutMs: 1000,
93
+ maxTurns: 1,
94
+ maxReplans: 1,
95
+ });
96
+
97
+ assert.equal(result.ok, true);
98
+ assert.equal(session.headlessPlan, null);
99
+ const types = eventTypes(session);
100
+ assert.ok(types.includes('tool_call_started'));
101
+ assert.ok(types.includes('run_done'));
102
+ assert.equal(types.includes('plan_set'), false);
103
+ assert.equal(types.includes('run_evaluated'), false);
104
+ });
105
+
106
+ test('Recipe #3 — "où en est le dernier build ?" reads status, starts no new production run', async () => {
107
+ const session = baseSession({
108
+ llm: {
109
+ async completeWithTools() {
110
+ assert.fail('a status question must never reach the evaluator or replanner');
111
+ },
112
+ },
113
+ });
114
+ const agent = {
115
+ async invoke({ session: turnSession }) {
116
+ dispatchAgentEvent(turnSession, createAgentEvent('tool_call_started', {
117
+ origin: 'tool',
118
+ payload: { name: 'production.production_job_status', args: '{}', summary: 'calling...' },
119
+ }));
120
+ dispatchAgentEvent(turnSession, createAgentEvent('tool_call_result', {
121
+ origin: 'tool',
122
+ payload: { name: 'production.production_job_status', ok: true, result: 'last build: done', summary: 'done' },
123
+ }));
124
+ return { response: 'Le dernier build est terminé avec succès.' };
125
+ },
126
+ };
127
+
128
+ const result = await runRuntimeAgenticWorkflow(agent, session, 'où en est le dernier build ?', {
129
+ runId: 'recipe-3',
130
+ timeoutMs: 1000,
131
+ maxTurns: 1,
132
+ maxReplans: 1,
133
+ });
134
+
135
+ assert.equal(result.ok, true);
136
+ assert.equal(session.headlessPlan, null);
137
+ assert.equal(Object.keys(session.activities).length, 0);
138
+ const types = eventTypes(session);
139
+ assert.equal(types.includes('activity_upserted'), false);
140
+ assert.equal(types.includes('plan_set'), false);
141
+ });
142
+
143
+ function buildSingleTaskAgent({ taskId, description, finalResponse }) {
144
+ let turn = 0;
145
+ return {
146
+ async invoke({ session: turnSession }) {
147
+ turn += 1;
148
+ if (turn === 1) {
149
+ dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
150
+ origin: 'tool',
151
+ payload: { steps: [{ step: 1, id: taskId, description, status: 'pending', dependsOn: [] }] },
152
+ }));
153
+ return { response: `Je lance ${description}.` };
154
+ }
155
+ dispatchAgentEvent(turnSession, createAgentEvent('plan_step_updated', {
156
+ origin: 'tool',
157
+ payload: { step: 1, status: 'done' },
158
+ }));
159
+ return { response: finalResponse };
160
+ },
161
+ };
162
+ }
163
+
164
+ test('Recipe #4 — "lance le doctor": one-task plan, progress visible, summarized result', async () => {
165
+ const session = baseSession();
166
+ const agent = buildSingleTaskAgent({
167
+ taskId: 'doctor',
168
+ description: 'Diagnostic doctor',
169
+ finalResponse: 'Doctor terminé : aucun problème détecté.',
170
+ });
171
+
172
+ const result = await runRuntimeAgenticWorkflow(agent, session, 'lance le doctor', {
173
+ runId: 'recipe-4',
174
+ timeoutMs: 1000,
175
+ maxTurns: 3,
176
+ maxReplans: 1,
177
+ evaluate: false,
178
+ });
179
+
180
+ assert.equal(result.ok, true);
181
+ assert.equal(session.headlessPlan.length, 1);
182
+ assert.equal(session.headlessPlan[0].status, 'done');
183
+ const types = eventTypes(session);
184
+ assert.deepEqual(types.filter((type) => type === 'plan_set' || type === 'plan_step_updated'), ['plan_set', 'plan_step_updated']);
185
+ assert.ok(types.includes('run_done'));
186
+ assert.equal(session.agentProjection.conversation.at(-1).content, 'Doctor terminé : aucun problème détecté.');
187
+ });
188
+
189
+ test('Recipe #5 — "construis le livrable X": plan posed, job runs, progress, done, summarized', async () => {
190
+ const session = baseSession();
191
+ const agent = buildSingleTaskAgent({
192
+ taskId: 'build-x',
193
+ description: 'Construire le livrable X',
194
+ finalResponse: 'Le livrable X est construit.',
195
+ });
196
+
197
+ const result = await runRuntimeAgenticWorkflow(agent, session, 'construis le livrable X', {
198
+ runId: 'recipe-5',
199
+ timeoutMs: 1000,
200
+ maxTurns: 3,
201
+ maxReplans: 1,
202
+ evaluate: false,
203
+ });
204
+
205
+ assert.equal(result.ok, true);
206
+ assert.equal(session.headlessPlan[0].status, 'done');
207
+ assert.ok(eventTypes(session).includes('run_done'));
208
+ assert.equal(session.agentProjection.conversation.at(-1).content, 'Le livrable X est construit.');
209
+ });
210
+
211
+ function buildTwoParallelTasksAgent() {
212
+ const started = [];
213
+ const release = {};
214
+ let parentTurnDone = false;
215
+ return {
216
+ started,
217
+ release,
218
+ async invoke({ input, session: turnSession }) {
219
+ const taskMatch = input.match(/Task id: (\w+)/);
220
+ if (!taskMatch) {
221
+ if (!parentTurnDone) {
222
+ parentTurnDone = true;
223
+ dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
224
+ origin: 'tool',
225
+ payload: {
226
+ steps: [
227
+ { step: 1, id: 'x', description: 'Construire X', status: 'pending', dependsOn: [] },
228
+ { step: 2, id: 'y', description: 'Construire Y', status: 'pending', dependsOn: [] },
229
+ ],
230
+ },
231
+ }));
232
+ return { response: 'Je construis X et Y en parallèle.' };
233
+ }
234
+ return { response: 'X et Y sont construits.' };
235
+ }
236
+ const id = taskMatch[1];
237
+ started.push(id);
238
+ await new Promise((resolve) => { release[id] = resolve; });
239
+ return { response: `${id} construit.` };
240
+ },
241
+ };
242
+ }
243
+
244
+ test('Recipe #6 — "construis X et Y": two tasks run in parallel, converge, done', async () => {
245
+ const session = baseSession();
246
+ const agent = buildTwoParallelTasksAgent();
247
+
248
+ const running = runRuntimeAgenticWorkflow(agent, session, 'construis X et Y', {
249
+ runId: 'recipe-6',
250
+ timeoutMs: 2000,
251
+ maxTurns: 3,
252
+ maxReplans: 1,
253
+ evaluate: false,
254
+ });
255
+
256
+ await waitFor(() => agent.started.includes('x') && agent.started.includes('y'));
257
+ assert.deepEqual(
258
+ session.headlessPlan.filter((step) => ['x', 'y'].includes(step.id)).map((step) => step.status),
259
+ ['running', 'running'],
260
+ );
261
+ agent.release.x();
262
+ agent.release.y();
263
+
264
+ const result = await running;
265
+
266
+ assert.equal(result.ok, true);
267
+ assert.deepEqual(session.headlessPlan.map((step) => step.status), ['done', 'done']);
268
+ });
269
+
270
+ test('Recipe #7 — "où en es-tu ?" during an active run: status in conversation, run continues, no new run', async (t) => {
271
+ const session = { workspace: 'juno', controlQueue: [] };
272
+ let runCount = 0;
273
+ let handle;
274
+ try {
275
+ handle = await startRuntimeServer({
276
+ host: '127.0.0.1',
277
+ port: 0,
278
+ store: {
279
+ dbPath: ':memory:',
280
+ getState: () => ({
281
+ status: 'running',
282
+ plan: [{ step: 1, description: 'Construire le livrable X', status: 'running' }],
283
+ queue: [],
284
+ approvals: [],
285
+ summary: null,
286
+ }),
287
+ listEvents: () => [],
288
+ },
289
+ getContext: async () => ({
290
+ workspace: 'juno',
291
+ session,
292
+ running: true,
293
+ currentAbortController: new AbortController(),
294
+ }),
295
+ run: async () => { runCount += 1; },
296
+ });
297
+ } catch (err) {
298
+ if (err?.code === 'EPERM') {
299
+ t.skip('network listen is not permitted in this sandbox');
300
+ return;
301
+ }
302
+ throw err;
303
+ }
304
+
305
+ try {
306
+ const response = await fetch(`http://127.0.0.1:${handle.port}/control?workspace=juno`, {
307
+ method: 'POST',
308
+ headers: { 'Content-Type': 'application/json' },
309
+ body: JSON.stringify({ action: 'message', input: 'où en es-tu ?' }),
310
+ });
311
+ assert.equal(response.status, 200);
312
+ const body = await response.json();
313
+ assert.equal(body.kind, 'observe');
314
+ assert.match(body.explanation, /Construire le livrable X/);
315
+ assert.equal(session.controlQueue.length, 0);
316
+ assert.equal(runCount, 0);
317
+ } finally {
318
+ await handle.close();
319
+ }
320
+ });
321
+
322
+ test('Recipe #8 — vague request gets clarification, never a job, never boilerplate', async () => {
323
+ const session = baseSession({
324
+ headlessPlan: [{ step: 1, id: 'task', description: 'Task', status: 'done' }],
325
+ llm: {
326
+ async completeWithTools({ system }) {
327
+ assert.match(system, /strict evaluator/);
328
+ return { content: '{"ok":false,"reason":"demande vague / objectif indefini","suggestedAction":"clarifier l objectif"}' };
329
+ },
330
+ },
331
+ });
332
+ const agent = {
333
+ async invoke({ session: turnSession }) {
334
+ if (turnSession.headlessPlan === null) {
335
+ dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
336
+ origin: 'tool',
337
+ payload: { steps: [{ step: 1, id: 'task', description: 'Task', status: 'done' }] },
338
+ }));
339
+ }
340
+ return { response: 'Terminé.' };
341
+ },
342
+ };
343
+
344
+ const result = await runRuntimeAgenticWorkflow(agent, session, 'dis-moi une bêtise puis stop', {
345
+ runId: 'recipe-8',
346
+ timeoutMs: 1000,
347
+ maxTurns: 1,
348
+ maxReplans: 1,
349
+ });
350
+
351
+ assert.equal(result.ok, true);
352
+ assert.equal(result.clarified, true);
353
+ const lastMessage = session.agentProjection.conversation.at(-1).content;
354
+ assert.ok(lastMessage.includes('clarifier'));
355
+ assert.doesNotMatch(lastMessage, /Donna is active|Plan is stalled/i);
356
+ const types = eventTypes(session);
357
+ assert.equal(types.includes('run_replanned'), false);
358
+ assert.equal(types.includes('tool_call_started'), false);
359
+ });
@@ -2,7 +2,7 @@ import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
2
2
  import { activityKey, sessionActivities, terminalFailures } from '../core/activity.js';
3
3
  import { runAgenticLoop, throwIfAborted } from '../core/agentLoop.js';
4
4
  import { formatPlanStatus } from '../core/plan.js';
5
- import { formatReadyTaskPrompt, readyPlanTasks } from '../core/planPatch.js';
5
+ import { formatReadyTaskPrompt, readyPlanTasks, sanitizePlanForExecution } from '../core/planPatch.js';
6
6
  import { emitRuntimeLog, pollActivitiesOnce } from './supervisor.js';
7
7
 
8
8
  const DEFAULT_MAX_REPLANS = 2;
@@ -92,6 +92,7 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
92
92
  let replansLeft = Math.max(0, Math.floor(Number(maxReplans) || 0));
93
93
 
94
94
  while (true) {
95
+ sanitizeSessionPlanForExecution(session, runId);
95
96
  const result = shouldUseParallelScheduler(session.headlessPlan)
96
97
  ? await runRuntimeParallelPlan(agent, session, input, {
97
98
  signal,
@@ -134,7 +135,9 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
134
135
  return { ok: false, result };
135
136
  }
136
137
 
137
- const evaluation = await evaluateRuntimeRun(session, input, { runId, signal, evaluate });
138
+ const evaluation = session.headlessPlan
139
+ ? await evaluateRuntimeRun(session, input, { runId, signal, evaluate })
140
+ : null;
138
141
  if (evaluation) {
139
142
  dispatchAgentEvent(session, createAgentEvent('run_evaluated', {
140
143
  origin: 'runtime',
@@ -147,6 +150,19 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
147
150
  },
148
151
  }));
149
152
  if (!evaluation.ok) {
153
+ if (isUndefinedObjectiveEvaluation(evaluation)) {
154
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', {
155
+ origin: 'runtime',
156
+ runId,
157
+ payload: { content: clarificationMessageForEvaluation(evaluation) },
158
+ }));
159
+ dispatchAgentEvent(session, createAgentEvent('run_done', {
160
+ origin: 'runtime',
161
+ runId,
162
+ payload: { runId },
163
+ }));
164
+ return { ok: true, evaluation, clarified: true };
165
+ }
150
166
  if (replansLeft > 0) {
151
167
  const trigger = {
152
168
  kind: 'evaluation',
@@ -209,6 +225,7 @@ export async function runRuntimeParallelPlan(agent, session, input, {
209
225
  previousPlanUpdate?.();
210
226
  abortCancelledActiveTasks(session, active);
211
227
  };
228
+ sanitizeSessionPlanForExecution(session, runId);
212
229
  ensurePlanProjection(session, runId);
213
230
  emitRuntimeLog(session, `scheduler: parallel plan enabled (concurrency ${limit})`);
214
231
 
@@ -264,6 +281,20 @@ export async function runRuntimeParallelPlan(agent, session, input, {
264
281
  }
265
282
  }
266
283
 
284
+ function sanitizeSessionPlanForExecution(session, runId = null) {
285
+ if (!session.headlessPlan) return;
286
+ const sanitized = sanitizePlanForExecution(session.headlessPlan);
287
+ if (sanitized.warnings.length === 0) return;
288
+ session.headlessPlan = sanitized.plan;
289
+ dispatchAgentEvent(session, createAgentEvent('runtime_log', {
290
+ origin: 'runtime',
291
+ runId,
292
+ payload: {
293
+ message: `plan warning: ${sanitized.warnings.join('; ')}`,
294
+ },
295
+ }));
296
+ }
297
+
267
298
  async function drainActive(active, locks) {
268
299
  if (active.size === 0) return;
269
300
  const entries = [...active.values()];
@@ -570,7 +601,9 @@ export async function finishRuntimeRun(session, input, {
570
601
  signal = null,
571
602
  evaluate = true,
572
603
  } = {}) {
573
- const evaluation = await evaluateRuntimeRun(session, input, { runId, signal, evaluate });
604
+ const evaluation = session.headlessPlan
605
+ ? await evaluateRuntimeRun(session, input, { runId, signal, evaluate })
606
+ : null;
574
607
  if (evaluation) {
575
608
  dispatchAgentEvent(session, createAgentEvent('run_evaluated', {
576
609
  origin: 'runtime',
@@ -583,6 +616,19 @@ export async function finishRuntimeRun(session, input, {
583
616
  },
584
617
  }));
585
618
  if (!evaluation.ok) {
619
+ if (isUndefinedObjectiveEvaluation(evaluation)) {
620
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', {
621
+ origin: 'runtime',
622
+ runId,
623
+ payload: { content: clarificationMessageForEvaluation(evaluation) },
624
+ }));
625
+ dispatchAgentEvent(session, createAgentEvent('run_done', {
626
+ origin: 'runtime',
627
+ runId,
628
+ payload: { runId },
629
+ }));
630
+ return { ok: true, evaluation, clarified: true };
631
+ }
586
632
  dispatchAgentEvent(session, createAgentEvent('run_error', {
587
633
  origin: 'runtime',
588
634
  runId,
@@ -672,6 +718,21 @@ function normalizeEvaluation(value) {
672
718
  };
673
719
  }
674
720
 
721
+ function isUndefinedObjectiveEvaluation(evaluation) {
722
+ const text = `${evaluation?.reason ?? ''} ${evaluation?.suggestedAction ?? ''}`
723
+ .toLowerCase()
724
+ .normalize('NFD')
725
+ .replace(/[\u0300-\u036f]/g, '');
726
+ return /\b(vague|undefined|indefini|unclear|clarif|ambiguous|missing objective|no objective)\b/.test(text);
727
+ }
728
+
729
+ function clarificationMessageForEvaluation(evaluation) {
730
+ const reason = String(evaluation?.reason ?? '').trim();
731
+ return reason
732
+ ? `Je dois clarifier la demande avant d'agir : ${reason}`
733
+ : "Je dois clarifier la demande avant d'agir.";
734
+ }
735
+
675
736
  function fallbackEvaluation(reason) {
676
737
  return {
677
738
  ok: true,
@@ -723,6 +784,10 @@ export async function replanRuntimeRun(session, input, trigger, {
723
784
  steps: mergedSteps,
724
785
  },
725
786
  }));
787
+ if (hasMutatingReplanStep(steps)) {
788
+ session._runApprovalRequired = true;
789
+ session._runApprovalResolved = false;
790
+ }
726
791
  return { ok: true, steps };
727
792
  } catch (err) {
728
793
  return { ok: false, reason: err instanceof Error ? err.message : String(err) };
@@ -811,6 +876,10 @@ function normalizeReplan(steps) {
811
876
  .slice(0, 12);
812
877
  }
813
878
 
879
+ function hasMutatingReplanStep(steps) {
880
+ return steps.some((step) => /\b(build|copy|ingest|import|export|polish|pipeline|write|create|delete|update|send|deploy|publish|generate|construire|copier|importer|exporter|publier|envoyer|supprimer|modifier|creer|générer|generer)\b/i.test(step));
881
+ }
882
+
814
883
  function mergeReplanWithCompleted(plan, steps) {
815
884
  const completed = (plan ?? [])
816
885
  .filter((step) => step.status === 'done')
@@ -3,6 +3,41 @@ import test from 'node:test';
3
3
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
4
4
  import { finishRuntimeRun, replanRuntimeRun, runRuntimeAgenticWorkflow, runRuntimeParallelPlan } from './runner.js';
5
5
 
6
+ test('runRuntimeAgenticWorkflow completes conversational turns without evaluation or replan', async () => {
7
+ const events = [];
8
+ const session = {
9
+ activities: {},
10
+ headlessPlan: null,
11
+ llm: {
12
+ async completeWithTools() {
13
+ assert.fail('conversation-only turn must not call evaluator or replanner');
14
+ },
15
+ },
16
+ _onAgentEvent: (event) => events.push(event),
17
+ };
18
+ const agent = {
19
+ async invoke() {
20
+ return { response: 'Salut.' };
21
+ },
22
+ };
23
+
24
+ const started = Date.now();
25
+ const result = await runRuntimeAgenticWorkflow(agent, session, 'salut', {
26
+ runId: 'run-chat',
27
+ timeoutMs: 1000,
28
+ maxTurns: 1,
29
+ maxReplans: 1,
30
+ });
31
+
32
+ assert.equal(result.ok, true);
33
+ assert.ok(Date.now() - started < 5000);
34
+ assert.equal(session.headlessPlan, null);
35
+ assert.equal(Object.keys(session.activities).length, 0);
36
+ assert.ok(events.some((event) => event.type === 'run_done'));
37
+ assert.equal(events.some((event) => event.type === 'run_evaluated'), false);
38
+ assert.equal(events.some((event) => event.type === 'run_replanned'), false);
39
+ });
40
+
6
41
  test('finishRuntimeRun emits evaluation before run_done', async () => {
7
42
  const events = [];
8
43
  const session = {
@@ -33,6 +68,47 @@ test('finishRuntimeRun emits evaluation before run_done', async () => {
33
68
  assert.equal(session.agentProjection.status, 'done');
34
69
  });
35
70
 
71
+ test('runRuntimeAgenticWorkflow clarifies vague evaluations without replan', async () => {
72
+ const events = [];
73
+ const session = {
74
+ activities: {},
75
+ headlessPlan: [{ step: 1, id: 'task', description: 'Task', status: 'done' }],
76
+ llm: {
77
+ async completeWithTools({ system }) {
78
+ assert.match(system, /strict evaluator/);
79
+ return { content: '{"ok":false,"reason":"demande vague / objectif indefini","suggestedAction":"clarifier l objectif"}' };
80
+ },
81
+ },
82
+ _onAgentEvent: (event) => events.push(event),
83
+ };
84
+ const agent = {
85
+ async invoke({ session: turnSession }) {
86
+ if (turnSession.headlessPlan === null) {
87
+ dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
88
+ origin: 'tool',
89
+ payload: { steps: [{ step: 1, id: 'task', description: 'Task', status: 'done' }] },
90
+ }));
91
+ }
92
+ return { response: 'Terminé.' };
93
+ },
94
+ };
95
+
96
+ const result = await runRuntimeAgenticWorkflow(agent, session, 'fais le truc', {
97
+ runId: 'run-vague',
98
+ timeoutMs: 1000,
99
+ maxTurns: 1,
100
+ maxReplans: 1,
101
+ });
102
+
103
+ assert.equal(result.ok, true);
104
+ assert.equal(result.clarified, true);
105
+ assert.ok(session.agentProjection.conversation.at(-1).content.includes('clarifier'));
106
+ assert.ok(events.some((event) => event.type === 'run_evaluated'));
107
+ assert.equal(events.some((event) => event.type === 'run_replanned'), false);
108
+ assert.equal(events.some((event) => event.type === 'tool_call_started'), false);
109
+ assert.equal(session.agentProjection.status, 'done');
110
+ });
111
+
36
112
  test('finishRuntimeRun turns negative evaluation into run_error', async () => {
37
113
  const events = [];
38
114
  const session = {
@@ -62,7 +138,7 @@ test('finishRuntimeRun turns negative evaluation into run_error', async () => {
62
138
  test('finishRuntimeRun falls back open when evaluator response is invalid', async () => {
63
139
  const session = {
64
140
  activities: {},
65
- headlessPlan: null,
141
+ headlessPlan: [{ step: 1, description: 'Do work', status: 'done' }],
66
142
  agentProjection: { conversation: [] },
67
143
  llm: {
68
144
  async completeWithTools() {
@@ -126,13 +202,20 @@ test('runRuntimeAgenticWorkflow replans after negative evaluation', async () =>
126
202
  const agent = {
127
203
  async invoke({ session: turnSession }) {
128
204
  turns += 1;
205
+ if (turnSession.headlessPlan === null) {
206
+ dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
207
+ origin: 'tool',
208
+ payload: { steps: [{ step: 1, id: 'initial', description: 'Initial work', status: 'done' }] },
209
+ }));
210
+ return { response: 'Initial done.' };
211
+ }
129
212
  if (turnSession.headlessPlan?.[0]?.status === 'pending') {
130
213
  dispatchAgentEvent(turnSession, createAgentEvent('plan_step_updated', {
131
214
  origin: 'tool',
132
215
  payload: { step: 1, status: 'done' },
133
216
  }));
134
217
  }
135
- return { response: turns === 1 ? 'Initial done.' : 'Export done.' };
218
+ return { response: 'Export done.' };
136
219
  },
137
220
  };
138
221
 
@@ -175,6 +258,27 @@ test('replanRuntimeRun preserves completed outputs when replacing remaining work
175
258
  assert.deepEqual(session.headlessPlan[1].dependsOn, ['export']);
176
259
  });
177
260
 
261
+ test('replanRuntimeRun requires runtime approval for mutating replanned work', async () => {
262
+ const session = {
263
+ activities: {},
264
+ headlessPlan: [{ step: 1, id: 'build', description: 'Build', status: 'failed' }],
265
+ llm: {
266
+ async completeWithTools() {
267
+ return { content: '{"steps":["Run production build"]}' };
268
+ },
269
+ },
270
+ };
271
+
272
+ const result = await replanRuntimeRun(session, 'Build deliverable', {
273
+ kind: 'evaluation',
274
+ reason: 'Build missing.',
275
+ }, { runId: 'run-replan-approval', replansLeft: 1 });
276
+
277
+ assert.equal(result.ok, true);
278
+ assert.equal(session._runApprovalRequired, true);
279
+ assert.equal(session._runApprovalResolved, false);
280
+ });
281
+
178
282
  test('runRuntimeAgenticWorkflow replans after terminal activity error', async () => {
179
283
  const originalFetch = globalThis.fetch;
180
284
  let pollAttempts = 0;
@@ -276,7 +380,13 @@ test('runRuntimeAgenticWorkflow stops after replan budget is exhausted', async (
276
380
  },
277
381
  };
278
382
  const agent = {
279
- async invoke() {
383
+ async invoke({ session: turnSession }) {
384
+ if (turnSession.headlessPlan === null) {
385
+ dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
386
+ origin: 'tool',
387
+ payload: { steps: [{ step: 1, id: 'task', description: 'Task', status: 'done' }] },
388
+ }));
389
+ }
280
390
  return { response: 'Done.' };
281
391
  },
282
392
  };