@dotdrelle/wiki-manager 0.11.4 → 0.11.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents.docker-compose.yml +3 -3
- package/package.json +2 -2
- package/src/agent/graph.js +99 -3
- package/src/agent/graph.test.js +81 -0
- package/src/cli/wiki-manager.js +9 -2
- package/src/core/agentLoop.js +20 -5
- package/src/core/agentLoop.test.js +19 -1
- package/src/core/mcp.js +1 -1
- package/src/core/planPatch.js +39 -1
- package/src/core/planPatch.test.js +23 -1
- package/src/runtime/client.js +12 -0
- package/src/runtime/donna-contract.test.js +359 -0
- package/src/runtime/runner.js +72 -3
- package/src/runtime/runner.test.js +113 -3
- package/src/runtime/server.js +27 -1
- package/src/runtime/server.test.js +38 -0
- package/src/shell/LeftPane.tsx +2 -1
- package/src/shell/StartupScreen.tsx +101 -21
- package/src/shell/repl.js +60 -15
- package/src/shell/repl.test.js +33 -1
- package/src/shell/tui.tsx +90 -75
- package/src/shell/useAgent.ts +14 -4
- package/src/shell/useSession.ts +92 -17
|
@@ -0,0 +1,359 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import test from 'node:test';
|
|
3
|
+
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
4
|
+
import { runRuntimeAgenticWorkflow } from './runner.js';
|
|
5
|
+
import { startRuntimeServer } from './server.js';
|
|
6
|
+
|
|
7
|
+
// Executable version of plan-0.11.5-hotfix-final.md §3 ("La recette — le seul
|
|
8
|
+
// critère de sortie qui compte"). Each test below is one row of that table,
|
|
9
|
+
// referenced by its recipe number. Correctifs 1-3 and 8 (runner.js's
|
|
10
|
+
// conversational short-circuit, plan sanitization, and no-replan-on-vague
|
|
11
|
+
// paths) are exercised with a mocked agent standing in for the LLM's
|
|
12
|
+
// per-turn decision, exactly like the existing runner.test.js unit tests --
|
|
13
|
+
// this file's job is traceability to the recipe, not a new test seam.
|
|
14
|
+
// Recipes 4-6 drive the real scheduler (runAgenticLoop / runRuntimeParallelPlan,
|
|
15
|
+
// imported and executed for real, never reimplemented). Recipe 7 drives the
|
|
16
|
+
// real control-lane HTTP endpoint. Per plan §4.1: this file is the CI gate --
|
|
17
|
+
// no release ships while it is red.
|
|
18
|
+
|
|
19
|
+
function baseSession(overrides = {}) {
|
|
20
|
+
return { activities: {}, headlessPlan: null, ...overrides };
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
function eventTypes(session) {
|
|
24
|
+
return (session.agentEvents ?? []).map((event) => event.type);
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
async function waitFor(predicate, timeoutMs = 500) {
|
|
28
|
+
const deadline = Date.now() + timeoutMs;
|
|
29
|
+
while (Date.now() < deadline) {
|
|
30
|
+
if (predicate()) return;
|
|
31
|
+
await new Promise((resolve) => setTimeout(resolve, 1));
|
|
32
|
+
}
|
|
33
|
+
assert.fail('condition was not met before timeout');
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
test('Recipe #1 — "salut" gets a plain reply: no plan, no activity, no job, run done', async () => {
|
|
37
|
+
const session = baseSession({
|
|
38
|
+
llm: {
|
|
39
|
+
async completeWithTools() {
|
|
40
|
+
assert.fail('a plain greeting must never reach the evaluator or replanner');
|
|
41
|
+
},
|
|
42
|
+
},
|
|
43
|
+
});
|
|
44
|
+
const agent = {
|
|
45
|
+
async invoke() {
|
|
46
|
+
return { response: 'Bonjour ! Comment puis-je vous aider ?' };
|
|
47
|
+
},
|
|
48
|
+
};
|
|
49
|
+
|
|
50
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'salut', {
|
|
51
|
+
runId: 'recipe-1',
|
|
52
|
+
timeoutMs: 1000,
|
|
53
|
+
maxTurns: 1,
|
|
54
|
+
maxReplans: 1,
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
assert.equal(result.ok, true);
|
|
58
|
+
assert.equal(session.headlessPlan, null);
|
|
59
|
+
assert.equal(Object.keys(session.activities).length, 0);
|
|
60
|
+
const types = eventTypes(session);
|
|
61
|
+
assert.ok(types.includes('run_done'));
|
|
62
|
+
assert.equal(types.includes('run_evaluated'), false);
|
|
63
|
+
assert.equal(types.includes('run_replanned'), false);
|
|
64
|
+
assert.equal(types.includes('tool_call_started'), false);
|
|
65
|
+
assert.equal(types.includes('plan_set'), false);
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
test('Recipe #2 — "quel est le profil actif ?" answers in conversation, never poses a plan', async () => {
|
|
69
|
+
const session = baseSession({
|
|
70
|
+
llm: {
|
|
71
|
+
async completeWithTools() {
|
|
72
|
+
assert.fail('a read-only config question must never reach the evaluator or replanner');
|
|
73
|
+
},
|
|
74
|
+
},
|
|
75
|
+
});
|
|
76
|
+
const agent = {
|
|
77
|
+
async invoke({ session: turnSession }) {
|
|
78
|
+
dispatchAgentEvent(turnSession, createAgentEvent('tool_call_started', {
|
|
79
|
+
origin: 'tool',
|
|
80
|
+
payload: { name: 'shell.status', args: '{}', summary: 'calling...' },
|
|
81
|
+
}));
|
|
82
|
+
dispatchAgentEvent(turnSession, createAgentEvent('tool_call_result', {
|
|
83
|
+
origin: 'tool',
|
|
84
|
+
payload: { name: 'shell.status', ok: true, result: 'profile: albert-openai (mistral-large)', summary: 'done' },
|
|
85
|
+
}));
|
|
86
|
+
return { response: 'Le profil actif est albert-openai (mistral-large).' };
|
|
87
|
+
},
|
|
88
|
+
};
|
|
89
|
+
|
|
90
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'quel est le profil actif ?', {
|
|
91
|
+
runId: 'recipe-2',
|
|
92
|
+
timeoutMs: 1000,
|
|
93
|
+
maxTurns: 1,
|
|
94
|
+
maxReplans: 1,
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
assert.equal(result.ok, true);
|
|
98
|
+
assert.equal(session.headlessPlan, null);
|
|
99
|
+
const types = eventTypes(session);
|
|
100
|
+
assert.ok(types.includes('tool_call_started'));
|
|
101
|
+
assert.ok(types.includes('run_done'));
|
|
102
|
+
assert.equal(types.includes('plan_set'), false);
|
|
103
|
+
assert.equal(types.includes('run_evaluated'), false);
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
test('Recipe #3 — "où en est le dernier build ?" reads status, starts no new production run', async () => {
|
|
107
|
+
const session = baseSession({
|
|
108
|
+
llm: {
|
|
109
|
+
async completeWithTools() {
|
|
110
|
+
assert.fail('a status question must never reach the evaluator or replanner');
|
|
111
|
+
},
|
|
112
|
+
},
|
|
113
|
+
});
|
|
114
|
+
const agent = {
|
|
115
|
+
async invoke({ session: turnSession }) {
|
|
116
|
+
dispatchAgentEvent(turnSession, createAgentEvent('tool_call_started', {
|
|
117
|
+
origin: 'tool',
|
|
118
|
+
payload: { name: 'production.production_job_status', args: '{}', summary: 'calling...' },
|
|
119
|
+
}));
|
|
120
|
+
dispatchAgentEvent(turnSession, createAgentEvent('tool_call_result', {
|
|
121
|
+
origin: 'tool',
|
|
122
|
+
payload: { name: 'production.production_job_status', ok: true, result: 'last build: done', summary: 'done' },
|
|
123
|
+
}));
|
|
124
|
+
return { response: 'Le dernier build est terminé avec succès.' };
|
|
125
|
+
},
|
|
126
|
+
};
|
|
127
|
+
|
|
128
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'où en est le dernier build ?', {
|
|
129
|
+
runId: 'recipe-3',
|
|
130
|
+
timeoutMs: 1000,
|
|
131
|
+
maxTurns: 1,
|
|
132
|
+
maxReplans: 1,
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
assert.equal(result.ok, true);
|
|
136
|
+
assert.equal(session.headlessPlan, null);
|
|
137
|
+
assert.equal(Object.keys(session.activities).length, 0);
|
|
138
|
+
const types = eventTypes(session);
|
|
139
|
+
assert.equal(types.includes('activity_upserted'), false);
|
|
140
|
+
assert.equal(types.includes('plan_set'), false);
|
|
141
|
+
});
|
|
142
|
+
|
|
143
|
+
function buildSingleTaskAgent({ taskId, description, finalResponse }) {
|
|
144
|
+
let turn = 0;
|
|
145
|
+
return {
|
|
146
|
+
async invoke({ session: turnSession }) {
|
|
147
|
+
turn += 1;
|
|
148
|
+
if (turn === 1) {
|
|
149
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
|
|
150
|
+
origin: 'tool',
|
|
151
|
+
payload: { steps: [{ step: 1, id: taskId, description, status: 'pending', dependsOn: [] }] },
|
|
152
|
+
}));
|
|
153
|
+
return { response: `Je lance ${description}.` };
|
|
154
|
+
}
|
|
155
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_step_updated', {
|
|
156
|
+
origin: 'tool',
|
|
157
|
+
payload: { step: 1, status: 'done' },
|
|
158
|
+
}));
|
|
159
|
+
return { response: finalResponse };
|
|
160
|
+
},
|
|
161
|
+
};
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
test('Recipe #4 — "lance le doctor": one-task plan, progress visible, summarized result', async () => {
|
|
165
|
+
const session = baseSession();
|
|
166
|
+
const agent = buildSingleTaskAgent({
|
|
167
|
+
taskId: 'doctor',
|
|
168
|
+
description: 'Diagnostic doctor',
|
|
169
|
+
finalResponse: 'Doctor terminé : aucun problème détecté.',
|
|
170
|
+
});
|
|
171
|
+
|
|
172
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'lance le doctor', {
|
|
173
|
+
runId: 'recipe-4',
|
|
174
|
+
timeoutMs: 1000,
|
|
175
|
+
maxTurns: 3,
|
|
176
|
+
maxReplans: 1,
|
|
177
|
+
evaluate: false,
|
|
178
|
+
});
|
|
179
|
+
|
|
180
|
+
assert.equal(result.ok, true);
|
|
181
|
+
assert.equal(session.headlessPlan.length, 1);
|
|
182
|
+
assert.equal(session.headlessPlan[0].status, 'done');
|
|
183
|
+
const types = eventTypes(session);
|
|
184
|
+
assert.deepEqual(types.filter((type) => type === 'plan_set' || type === 'plan_step_updated'), ['plan_set', 'plan_step_updated']);
|
|
185
|
+
assert.ok(types.includes('run_done'));
|
|
186
|
+
assert.equal(session.agentProjection.conversation.at(-1).content, 'Doctor terminé : aucun problème détecté.');
|
|
187
|
+
});
|
|
188
|
+
|
|
189
|
+
test('Recipe #5 — "construis le livrable X": plan posed, job runs, progress, done, summarized', async () => {
|
|
190
|
+
const session = baseSession();
|
|
191
|
+
const agent = buildSingleTaskAgent({
|
|
192
|
+
taskId: 'build-x',
|
|
193
|
+
description: 'Construire le livrable X',
|
|
194
|
+
finalResponse: 'Le livrable X est construit.',
|
|
195
|
+
});
|
|
196
|
+
|
|
197
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'construis le livrable X', {
|
|
198
|
+
runId: 'recipe-5',
|
|
199
|
+
timeoutMs: 1000,
|
|
200
|
+
maxTurns: 3,
|
|
201
|
+
maxReplans: 1,
|
|
202
|
+
evaluate: false,
|
|
203
|
+
});
|
|
204
|
+
|
|
205
|
+
assert.equal(result.ok, true);
|
|
206
|
+
assert.equal(session.headlessPlan[0].status, 'done');
|
|
207
|
+
assert.ok(eventTypes(session).includes('run_done'));
|
|
208
|
+
assert.equal(session.agentProjection.conversation.at(-1).content, 'Le livrable X est construit.');
|
|
209
|
+
});
|
|
210
|
+
|
|
211
|
+
function buildTwoParallelTasksAgent() {
|
|
212
|
+
const started = [];
|
|
213
|
+
const release = {};
|
|
214
|
+
let parentTurnDone = false;
|
|
215
|
+
return {
|
|
216
|
+
started,
|
|
217
|
+
release,
|
|
218
|
+
async invoke({ input, session: turnSession }) {
|
|
219
|
+
const taskMatch = input.match(/Task id: (\w+)/);
|
|
220
|
+
if (!taskMatch) {
|
|
221
|
+
if (!parentTurnDone) {
|
|
222
|
+
parentTurnDone = true;
|
|
223
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
|
|
224
|
+
origin: 'tool',
|
|
225
|
+
payload: {
|
|
226
|
+
steps: [
|
|
227
|
+
{ step: 1, id: 'x', description: 'Construire X', status: 'pending', dependsOn: [] },
|
|
228
|
+
{ step: 2, id: 'y', description: 'Construire Y', status: 'pending', dependsOn: [] },
|
|
229
|
+
],
|
|
230
|
+
},
|
|
231
|
+
}));
|
|
232
|
+
return { response: 'Je construis X et Y en parallèle.' };
|
|
233
|
+
}
|
|
234
|
+
return { response: 'X et Y sont construits.' };
|
|
235
|
+
}
|
|
236
|
+
const id = taskMatch[1];
|
|
237
|
+
started.push(id);
|
|
238
|
+
await new Promise((resolve) => { release[id] = resolve; });
|
|
239
|
+
return { response: `${id} construit.` };
|
|
240
|
+
},
|
|
241
|
+
};
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
test('Recipe #6 — "construis X et Y": two tasks run in parallel, converge, done', async () => {
|
|
245
|
+
const session = baseSession();
|
|
246
|
+
const agent = buildTwoParallelTasksAgent();
|
|
247
|
+
|
|
248
|
+
const running = runRuntimeAgenticWorkflow(agent, session, 'construis X et Y', {
|
|
249
|
+
runId: 'recipe-6',
|
|
250
|
+
timeoutMs: 2000,
|
|
251
|
+
maxTurns: 3,
|
|
252
|
+
maxReplans: 1,
|
|
253
|
+
evaluate: false,
|
|
254
|
+
});
|
|
255
|
+
|
|
256
|
+
await waitFor(() => agent.started.includes('x') && agent.started.includes('y'));
|
|
257
|
+
assert.deepEqual(
|
|
258
|
+
session.headlessPlan.filter((step) => ['x', 'y'].includes(step.id)).map((step) => step.status),
|
|
259
|
+
['running', 'running'],
|
|
260
|
+
);
|
|
261
|
+
agent.release.x();
|
|
262
|
+
agent.release.y();
|
|
263
|
+
|
|
264
|
+
const result = await running;
|
|
265
|
+
|
|
266
|
+
assert.equal(result.ok, true);
|
|
267
|
+
assert.deepEqual(session.headlessPlan.map((step) => step.status), ['done', 'done']);
|
|
268
|
+
});
|
|
269
|
+
|
|
270
|
+
test('Recipe #7 — "où en es-tu ?" during an active run: status in conversation, run continues, no new run', async (t) => {
|
|
271
|
+
const session = { workspace: 'juno', controlQueue: [] };
|
|
272
|
+
let runCount = 0;
|
|
273
|
+
let handle;
|
|
274
|
+
try {
|
|
275
|
+
handle = await startRuntimeServer({
|
|
276
|
+
host: '127.0.0.1',
|
|
277
|
+
port: 0,
|
|
278
|
+
store: {
|
|
279
|
+
dbPath: ':memory:',
|
|
280
|
+
getState: () => ({
|
|
281
|
+
status: 'running',
|
|
282
|
+
plan: [{ step: 1, description: 'Construire le livrable X', status: 'running' }],
|
|
283
|
+
queue: [],
|
|
284
|
+
approvals: [],
|
|
285
|
+
summary: null,
|
|
286
|
+
}),
|
|
287
|
+
listEvents: () => [],
|
|
288
|
+
},
|
|
289
|
+
getContext: async () => ({
|
|
290
|
+
workspace: 'juno',
|
|
291
|
+
session,
|
|
292
|
+
running: true,
|
|
293
|
+
currentAbortController: new AbortController(),
|
|
294
|
+
}),
|
|
295
|
+
run: async () => { runCount += 1; },
|
|
296
|
+
});
|
|
297
|
+
} catch (err) {
|
|
298
|
+
if (err?.code === 'EPERM') {
|
|
299
|
+
t.skip('network listen is not permitted in this sandbox');
|
|
300
|
+
return;
|
|
301
|
+
}
|
|
302
|
+
throw err;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
try {
|
|
306
|
+
const response = await fetch(`http://127.0.0.1:${handle.port}/control?workspace=juno`, {
|
|
307
|
+
method: 'POST',
|
|
308
|
+
headers: { 'Content-Type': 'application/json' },
|
|
309
|
+
body: JSON.stringify({ action: 'message', input: 'où en es-tu ?' }),
|
|
310
|
+
});
|
|
311
|
+
assert.equal(response.status, 200);
|
|
312
|
+
const body = await response.json();
|
|
313
|
+
assert.equal(body.kind, 'observe');
|
|
314
|
+
assert.match(body.explanation, /Construire le livrable X/);
|
|
315
|
+
assert.equal(session.controlQueue.length, 0);
|
|
316
|
+
assert.equal(runCount, 0);
|
|
317
|
+
} finally {
|
|
318
|
+
await handle.close();
|
|
319
|
+
}
|
|
320
|
+
});
|
|
321
|
+
|
|
322
|
+
test('Recipe #8 — vague request gets clarification, never a job, never boilerplate', async () => {
|
|
323
|
+
const session = baseSession({
|
|
324
|
+
headlessPlan: [{ step: 1, id: 'task', description: 'Task', status: 'done' }],
|
|
325
|
+
llm: {
|
|
326
|
+
async completeWithTools({ system }) {
|
|
327
|
+
assert.match(system, /strict evaluator/);
|
|
328
|
+
return { content: '{"ok":false,"reason":"demande vague / objectif indefini","suggestedAction":"clarifier l objectif"}' };
|
|
329
|
+
},
|
|
330
|
+
},
|
|
331
|
+
});
|
|
332
|
+
const agent = {
|
|
333
|
+
async invoke({ session: turnSession }) {
|
|
334
|
+
if (turnSession.headlessPlan === null) {
|
|
335
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
|
|
336
|
+
origin: 'tool',
|
|
337
|
+
payload: { steps: [{ step: 1, id: 'task', description: 'Task', status: 'done' }] },
|
|
338
|
+
}));
|
|
339
|
+
}
|
|
340
|
+
return { response: 'Terminé.' };
|
|
341
|
+
},
|
|
342
|
+
};
|
|
343
|
+
|
|
344
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'dis-moi une bêtise puis stop', {
|
|
345
|
+
runId: 'recipe-8',
|
|
346
|
+
timeoutMs: 1000,
|
|
347
|
+
maxTurns: 1,
|
|
348
|
+
maxReplans: 1,
|
|
349
|
+
});
|
|
350
|
+
|
|
351
|
+
assert.equal(result.ok, true);
|
|
352
|
+
assert.equal(result.clarified, true);
|
|
353
|
+
const lastMessage = session.agentProjection.conversation.at(-1).content;
|
|
354
|
+
assert.ok(lastMessage.includes('clarifier'));
|
|
355
|
+
assert.doesNotMatch(lastMessage, /Donna is active|Plan is stalled/i);
|
|
356
|
+
const types = eventTypes(session);
|
|
357
|
+
assert.equal(types.includes('run_replanned'), false);
|
|
358
|
+
assert.equal(types.includes('tool_call_started'), false);
|
|
359
|
+
});
|
package/src/runtime/runner.js
CHANGED
|
@@ -2,7 +2,7 @@ import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
|
2
2
|
import { activityKey, sessionActivities, terminalFailures } from '../core/activity.js';
|
|
3
3
|
import { runAgenticLoop, throwIfAborted } from '../core/agentLoop.js';
|
|
4
4
|
import { formatPlanStatus } from '../core/plan.js';
|
|
5
|
-
import { formatReadyTaskPrompt, readyPlanTasks } from '../core/planPatch.js';
|
|
5
|
+
import { formatReadyTaskPrompt, readyPlanTasks, sanitizePlanForExecution } from '../core/planPatch.js';
|
|
6
6
|
import { emitRuntimeLog, pollActivitiesOnce } from './supervisor.js';
|
|
7
7
|
|
|
8
8
|
const DEFAULT_MAX_REPLANS = 2;
|
|
@@ -92,6 +92,7 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
|
|
|
92
92
|
let replansLeft = Math.max(0, Math.floor(Number(maxReplans) || 0));
|
|
93
93
|
|
|
94
94
|
while (true) {
|
|
95
|
+
sanitizeSessionPlanForExecution(session, runId);
|
|
95
96
|
const result = shouldUseParallelScheduler(session.headlessPlan)
|
|
96
97
|
? await runRuntimeParallelPlan(agent, session, input, {
|
|
97
98
|
signal,
|
|
@@ -134,7 +135,9 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
|
|
|
134
135
|
return { ok: false, result };
|
|
135
136
|
}
|
|
136
137
|
|
|
137
|
-
const evaluation =
|
|
138
|
+
const evaluation = session.headlessPlan
|
|
139
|
+
? await evaluateRuntimeRun(session, input, { runId, signal, evaluate })
|
|
140
|
+
: null;
|
|
138
141
|
if (evaluation) {
|
|
139
142
|
dispatchAgentEvent(session, createAgentEvent('run_evaluated', {
|
|
140
143
|
origin: 'runtime',
|
|
@@ -147,6 +150,19 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
|
|
|
147
150
|
},
|
|
148
151
|
}));
|
|
149
152
|
if (!evaluation.ok) {
|
|
153
|
+
if (isUndefinedObjectiveEvaluation(evaluation)) {
|
|
154
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_message', {
|
|
155
|
+
origin: 'runtime',
|
|
156
|
+
runId,
|
|
157
|
+
payload: { content: clarificationMessageForEvaluation(evaluation) },
|
|
158
|
+
}));
|
|
159
|
+
dispatchAgentEvent(session, createAgentEvent('run_done', {
|
|
160
|
+
origin: 'runtime',
|
|
161
|
+
runId,
|
|
162
|
+
payload: { runId },
|
|
163
|
+
}));
|
|
164
|
+
return { ok: true, evaluation, clarified: true };
|
|
165
|
+
}
|
|
150
166
|
if (replansLeft > 0) {
|
|
151
167
|
const trigger = {
|
|
152
168
|
kind: 'evaluation',
|
|
@@ -209,6 +225,7 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
209
225
|
previousPlanUpdate?.();
|
|
210
226
|
abortCancelledActiveTasks(session, active);
|
|
211
227
|
};
|
|
228
|
+
sanitizeSessionPlanForExecution(session, runId);
|
|
212
229
|
ensurePlanProjection(session, runId);
|
|
213
230
|
emitRuntimeLog(session, `scheduler: parallel plan enabled (concurrency ${limit})`);
|
|
214
231
|
|
|
@@ -264,6 +281,20 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
264
281
|
}
|
|
265
282
|
}
|
|
266
283
|
|
|
284
|
+
function sanitizeSessionPlanForExecution(session, runId = null) {
|
|
285
|
+
if (!session.headlessPlan) return;
|
|
286
|
+
const sanitized = sanitizePlanForExecution(session.headlessPlan);
|
|
287
|
+
if (sanitized.warnings.length === 0) return;
|
|
288
|
+
session.headlessPlan = sanitized.plan;
|
|
289
|
+
dispatchAgentEvent(session, createAgentEvent('runtime_log', {
|
|
290
|
+
origin: 'runtime',
|
|
291
|
+
runId,
|
|
292
|
+
payload: {
|
|
293
|
+
message: `plan warning: ${sanitized.warnings.join('; ')}`,
|
|
294
|
+
},
|
|
295
|
+
}));
|
|
296
|
+
}
|
|
297
|
+
|
|
267
298
|
async function drainActive(active, locks) {
|
|
268
299
|
if (active.size === 0) return;
|
|
269
300
|
const entries = [...active.values()];
|
|
@@ -570,7 +601,9 @@ export async function finishRuntimeRun(session, input, {
|
|
|
570
601
|
signal = null,
|
|
571
602
|
evaluate = true,
|
|
572
603
|
} = {}) {
|
|
573
|
-
const evaluation =
|
|
604
|
+
const evaluation = session.headlessPlan
|
|
605
|
+
? await evaluateRuntimeRun(session, input, { runId, signal, evaluate })
|
|
606
|
+
: null;
|
|
574
607
|
if (evaluation) {
|
|
575
608
|
dispatchAgentEvent(session, createAgentEvent('run_evaluated', {
|
|
576
609
|
origin: 'runtime',
|
|
@@ -583,6 +616,19 @@ export async function finishRuntimeRun(session, input, {
|
|
|
583
616
|
},
|
|
584
617
|
}));
|
|
585
618
|
if (!evaluation.ok) {
|
|
619
|
+
if (isUndefinedObjectiveEvaluation(evaluation)) {
|
|
620
|
+
dispatchAgentEvent(session, createAgentEvent('assistant_message', {
|
|
621
|
+
origin: 'runtime',
|
|
622
|
+
runId,
|
|
623
|
+
payload: { content: clarificationMessageForEvaluation(evaluation) },
|
|
624
|
+
}));
|
|
625
|
+
dispatchAgentEvent(session, createAgentEvent('run_done', {
|
|
626
|
+
origin: 'runtime',
|
|
627
|
+
runId,
|
|
628
|
+
payload: { runId },
|
|
629
|
+
}));
|
|
630
|
+
return { ok: true, evaluation, clarified: true };
|
|
631
|
+
}
|
|
586
632
|
dispatchAgentEvent(session, createAgentEvent('run_error', {
|
|
587
633
|
origin: 'runtime',
|
|
588
634
|
runId,
|
|
@@ -672,6 +718,21 @@ function normalizeEvaluation(value) {
|
|
|
672
718
|
};
|
|
673
719
|
}
|
|
674
720
|
|
|
721
|
+
function isUndefinedObjectiveEvaluation(evaluation) {
|
|
722
|
+
const text = `${evaluation?.reason ?? ''} ${evaluation?.suggestedAction ?? ''}`
|
|
723
|
+
.toLowerCase()
|
|
724
|
+
.normalize('NFD')
|
|
725
|
+
.replace(/[\u0300-\u036f]/g, '');
|
|
726
|
+
return /\b(vague|undefined|indefini|unclear|clarif|ambiguous|missing objective|no objective)\b/.test(text);
|
|
727
|
+
}
|
|
728
|
+
|
|
729
|
+
function clarificationMessageForEvaluation(evaluation) {
|
|
730
|
+
const reason = String(evaluation?.reason ?? '').trim();
|
|
731
|
+
return reason
|
|
732
|
+
? `Je dois clarifier la demande avant d'agir : ${reason}`
|
|
733
|
+
: "Je dois clarifier la demande avant d'agir.";
|
|
734
|
+
}
|
|
735
|
+
|
|
675
736
|
function fallbackEvaluation(reason) {
|
|
676
737
|
return {
|
|
677
738
|
ok: true,
|
|
@@ -723,6 +784,10 @@ export async function replanRuntimeRun(session, input, trigger, {
|
|
|
723
784
|
steps: mergedSteps,
|
|
724
785
|
},
|
|
725
786
|
}));
|
|
787
|
+
if (hasMutatingReplanStep(steps)) {
|
|
788
|
+
session._runApprovalRequired = true;
|
|
789
|
+
session._runApprovalResolved = false;
|
|
790
|
+
}
|
|
726
791
|
return { ok: true, steps };
|
|
727
792
|
} catch (err) {
|
|
728
793
|
return { ok: false, reason: err instanceof Error ? err.message : String(err) };
|
|
@@ -811,6 +876,10 @@ function normalizeReplan(steps) {
|
|
|
811
876
|
.slice(0, 12);
|
|
812
877
|
}
|
|
813
878
|
|
|
879
|
+
function hasMutatingReplanStep(steps) {
|
|
880
|
+
return steps.some((step) => /\b(build|copy|ingest|import|export|polish|pipeline|write|create|delete|update|send|deploy|publish|generate|construire|copier|importer|exporter|publier|envoyer|supprimer|modifier|creer|générer|generer)\b/i.test(step));
|
|
881
|
+
}
|
|
882
|
+
|
|
814
883
|
function mergeReplanWithCompleted(plan, steps) {
|
|
815
884
|
const completed = (plan ?? [])
|
|
816
885
|
.filter((step) => step.status === 'done')
|
|
@@ -3,6 +3,41 @@ import test from 'node:test';
|
|
|
3
3
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
4
4
|
import { finishRuntimeRun, replanRuntimeRun, runRuntimeAgenticWorkflow, runRuntimeParallelPlan } from './runner.js';
|
|
5
5
|
|
|
6
|
+
test('runRuntimeAgenticWorkflow completes conversational turns without evaluation or replan', async () => {
|
|
7
|
+
const events = [];
|
|
8
|
+
const session = {
|
|
9
|
+
activities: {},
|
|
10
|
+
headlessPlan: null,
|
|
11
|
+
llm: {
|
|
12
|
+
async completeWithTools() {
|
|
13
|
+
assert.fail('conversation-only turn must not call evaluator or replanner');
|
|
14
|
+
},
|
|
15
|
+
},
|
|
16
|
+
_onAgentEvent: (event) => events.push(event),
|
|
17
|
+
};
|
|
18
|
+
const agent = {
|
|
19
|
+
async invoke() {
|
|
20
|
+
return { response: 'Salut.' };
|
|
21
|
+
},
|
|
22
|
+
};
|
|
23
|
+
|
|
24
|
+
const started = Date.now();
|
|
25
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'salut', {
|
|
26
|
+
runId: 'run-chat',
|
|
27
|
+
timeoutMs: 1000,
|
|
28
|
+
maxTurns: 1,
|
|
29
|
+
maxReplans: 1,
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
assert.equal(result.ok, true);
|
|
33
|
+
assert.ok(Date.now() - started < 5000);
|
|
34
|
+
assert.equal(session.headlessPlan, null);
|
|
35
|
+
assert.equal(Object.keys(session.activities).length, 0);
|
|
36
|
+
assert.ok(events.some((event) => event.type === 'run_done'));
|
|
37
|
+
assert.equal(events.some((event) => event.type === 'run_evaluated'), false);
|
|
38
|
+
assert.equal(events.some((event) => event.type === 'run_replanned'), false);
|
|
39
|
+
});
|
|
40
|
+
|
|
6
41
|
test('finishRuntimeRun emits evaluation before run_done', async () => {
|
|
7
42
|
const events = [];
|
|
8
43
|
const session = {
|
|
@@ -33,6 +68,47 @@ test('finishRuntimeRun emits evaluation before run_done', async () => {
|
|
|
33
68
|
assert.equal(session.agentProjection.status, 'done');
|
|
34
69
|
});
|
|
35
70
|
|
|
71
|
+
test('runRuntimeAgenticWorkflow clarifies vague evaluations without replan', async () => {
|
|
72
|
+
const events = [];
|
|
73
|
+
const session = {
|
|
74
|
+
activities: {},
|
|
75
|
+
headlessPlan: [{ step: 1, id: 'task', description: 'Task', status: 'done' }],
|
|
76
|
+
llm: {
|
|
77
|
+
async completeWithTools({ system }) {
|
|
78
|
+
assert.match(system, /strict evaluator/);
|
|
79
|
+
return { content: '{"ok":false,"reason":"demande vague / objectif indefini","suggestedAction":"clarifier l objectif"}' };
|
|
80
|
+
},
|
|
81
|
+
},
|
|
82
|
+
_onAgentEvent: (event) => events.push(event),
|
|
83
|
+
};
|
|
84
|
+
const agent = {
|
|
85
|
+
async invoke({ session: turnSession }) {
|
|
86
|
+
if (turnSession.headlessPlan === null) {
|
|
87
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
|
|
88
|
+
origin: 'tool',
|
|
89
|
+
payload: { steps: [{ step: 1, id: 'task', description: 'Task', status: 'done' }] },
|
|
90
|
+
}));
|
|
91
|
+
}
|
|
92
|
+
return { response: 'Terminé.' };
|
|
93
|
+
},
|
|
94
|
+
};
|
|
95
|
+
|
|
96
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'fais le truc', {
|
|
97
|
+
runId: 'run-vague',
|
|
98
|
+
timeoutMs: 1000,
|
|
99
|
+
maxTurns: 1,
|
|
100
|
+
maxReplans: 1,
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
assert.equal(result.ok, true);
|
|
104
|
+
assert.equal(result.clarified, true);
|
|
105
|
+
assert.ok(session.agentProjection.conversation.at(-1).content.includes('clarifier'));
|
|
106
|
+
assert.ok(events.some((event) => event.type === 'run_evaluated'));
|
|
107
|
+
assert.equal(events.some((event) => event.type === 'run_replanned'), false);
|
|
108
|
+
assert.equal(events.some((event) => event.type === 'tool_call_started'), false);
|
|
109
|
+
assert.equal(session.agentProjection.status, 'done');
|
|
110
|
+
});
|
|
111
|
+
|
|
36
112
|
test('finishRuntimeRun turns negative evaluation into run_error', async () => {
|
|
37
113
|
const events = [];
|
|
38
114
|
const session = {
|
|
@@ -62,7 +138,7 @@ test('finishRuntimeRun turns negative evaluation into run_error', async () => {
|
|
|
62
138
|
test('finishRuntimeRun falls back open when evaluator response is invalid', async () => {
|
|
63
139
|
const session = {
|
|
64
140
|
activities: {},
|
|
65
|
-
headlessPlan:
|
|
141
|
+
headlessPlan: [{ step: 1, description: 'Do work', status: 'done' }],
|
|
66
142
|
agentProjection: { conversation: [] },
|
|
67
143
|
llm: {
|
|
68
144
|
async completeWithTools() {
|
|
@@ -126,13 +202,20 @@ test('runRuntimeAgenticWorkflow replans after negative evaluation', async () =>
|
|
|
126
202
|
const agent = {
|
|
127
203
|
async invoke({ session: turnSession }) {
|
|
128
204
|
turns += 1;
|
|
205
|
+
if (turnSession.headlessPlan === null) {
|
|
206
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
|
|
207
|
+
origin: 'tool',
|
|
208
|
+
payload: { steps: [{ step: 1, id: 'initial', description: 'Initial work', status: 'done' }] },
|
|
209
|
+
}));
|
|
210
|
+
return { response: 'Initial done.' };
|
|
211
|
+
}
|
|
129
212
|
if (turnSession.headlessPlan?.[0]?.status === 'pending') {
|
|
130
213
|
dispatchAgentEvent(turnSession, createAgentEvent('plan_step_updated', {
|
|
131
214
|
origin: 'tool',
|
|
132
215
|
payload: { step: 1, status: 'done' },
|
|
133
216
|
}));
|
|
134
217
|
}
|
|
135
|
-
return { response:
|
|
218
|
+
return { response: 'Export done.' };
|
|
136
219
|
},
|
|
137
220
|
};
|
|
138
221
|
|
|
@@ -175,6 +258,27 @@ test('replanRuntimeRun preserves completed outputs when replacing remaining work
|
|
|
175
258
|
assert.deepEqual(session.headlessPlan[1].dependsOn, ['export']);
|
|
176
259
|
});
|
|
177
260
|
|
|
261
|
+
test('replanRuntimeRun requires runtime approval for mutating replanned work', async () => {
|
|
262
|
+
const session = {
|
|
263
|
+
activities: {},
|
|
264
|
+
headlessPlan: [{ step: 1, id: 'build', description: 'Build', status: 'failed' }],
|
|
265
|
+
llm: {
|
|
266
|
+
async completeWithTools() {
|
|
267
|
+
return { content: '{"steps":["Run production build"]}' };
|
|
268
|
+
},
|
|
269
|
+
},
|
|
270
|
+
};
|
|
271
|
+
|
|
272
|
+
const result = await replanRuntimeRun(session, 'Build deliverable', {
|
|
273
|
+
kind: 'evaluation',
|
|
274
|
+
reason: 'Build missing.',
|
|
275
|
+
}, { runId: 'run-replan-approval', replansLeft: 1 });
|
|
276
|
+
|
|
277
|
+
assert.equal(result.ok, true);
|
|
278
|
+
assert.equal(session._runApprovalRequired, true);
|
|
279
|
+
assert.equal(session._runApprovalResolved, false);
|
|
280
|
+
});
|
|
281
|
+
|
|
178
282
|
test('runRuntimeAgenticWorkflow replans after terminal activity error', async () => {
|
|
179
283
|
const originalFetch = globalThis.fetch;
|
|
180
284
|
let pollAttempts = 0;
|
|
@@ -276,7 +380,13 @@ test('runRuntimeAgenticWorkflow stops after replan budget is exhausted', async (
|
|
|
276
380
|
},
|
|
277
381
|
};
|
|
278
382
|
const agent = {
|
|
279
|
-
async invoke() {
|
|
383
|
+
async invoke({ session: turnSession }) {
|
|
384
|
+
if (turnSession.headlessPlan === null) {
|
|
385
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
|
|
386
|
+
origin: 'tool',
|
|
387
|
+
payload: { steps: [{ step: 1, id: 'task', description: 'Task', status: 'done' }] },
|
|
388
|
+
}));
|
|
389
|
+
}
|
|
280
390
|
return { response: 'Done.' };
|
|
281
391
|
},
|
|
282
392
|
};
|