@dotdrelle/wiki-manager 0.11.4 → 0.11.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents.docker-compose.yml +3 -3
- package/package.json +2 -2
- package/src/agent/graph.js +118 -5
- package/src/agent/graph.test.js +158 -1
- package/src/cli/wiki-manager.js +9 -2
- package/src/commands/slash.js +36 -11
- package/src/core/agentLoop.js +20 -5
- package/src/core/agentLoop.test.js +19 -1
- package/src/core/mcp.js +1 -1
- package/src/core/planPatch.js +39 -1
- package/src/core/planPatch.test.js +23 -1
- package/src/core/profile.js +111 -0
- package/src/core/skills.js +1 -1
- package/src/runtime/client.js +12 -0
- package/src/runtime/donna-contract.test.js +359 -0
- package/src/runtime/runner.js +72 -3
- package/src/runtime/runner.test.js +113 -3
- package/src/runtime/server.js +27 -1
- package/src/runtime/server.test.js +38 -0
- package/src/shell/LeftPane.tsx +2 -1
- package/src/shell/StartupScreen.tsx +118 -23
- package/src/shell/repl.js +95 -42
- package/src/shell/repl.test.js +59 -1
- package/src/shell/tui.tsx +90 -75
- package/src/shell/useAgent.ts +14 -4
- package/src/shell/useSession.ts +92 -17
package/src/core/planPatch.js
CHANGED
|
@@ -86,7 +86,7 @@ export function rebasePlanPatch(patch, { currentRevision = 0 } = {}) {
|
|
|
86
86
|
}
|
|
87
87
|
|
|
88
88
|
export function readyPlanTasks(plan) {
|
|
89
|
-
const
|
|
89
|
+
const { plan: tasks } = sanitizePlanForExecution(plan);
|
|
90
90
|
const done = new Set(tasks.filter((task) => task.status === 'done').map(taskId));
|
|
91
91
|
return tasks
|
|
92
92
|
.filter((task) => task.status === 'pending')
|
|
@@ -125,6 +125,36 @@ export function normalizeTask(raw, index = 0) {
|
|
|
125
125
|
};
|
|
126
126
|
}
|
|
127
127
|
|
|
128
|
+
export function sanitizePlanForExecution(plan) {
|
|
129
|
+
const tasks = (plan ?? []).map((step, index) => normalizeTask(step, index));
|
|
130
|
+
const warnings = [];
|
|
131
|
+
const ids = new Set(tasks.map(taskId));
|
|
132
|
+
for (const task of tasks) {
|
|
133
|
+
const before = task.dependsOn;
|
|
134
|
+
task.dependsOn = before.filter((dep) => ids.has(String(dep)));
|
|
135
|
+
if (task.dependsOn.length !== before.length) {
|
|
136
|
+
warnings.push(`unknown dependency removed from task ${taskId(task)}`);
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
const pending = tasks.filter((task) => task.status === 'pending');
|
|
140
|
+
const done = new Set(tasks.filter((task) => task.status === 'done').map(taskId));
|
|
141
|
+
const terminalBlocked = new Set(
|
|
142
|
+
tasks
|
|
143
|
+
.filter((task) => ['failed', 'cancelled', 'canceled'].includes(String(task.status).toLowerCase()))
|
|
144
|
+
.map(taskId),
|
|
145
|
+
);
|
|
146
|
+
const hasReady = pending.some((task) => task.dependsOn.every((dep) => done.has(String(dep))));
|
|
147
|
+
if (hasDependencyCycle(tasks)) {
|
|
148
|
+
warnings.push('dependency cycle broken by sequential fallback');
|
|
149
|
+
return { plan: sequentialize(tasks), warnings };
|
|
150
|
+
}
|
|
151
|
+
if (pending.length > 0 && !hasReady && !pending.some((task) => task.dependsOn.some((dep) => terminalBlocked.has(String(dep))))) {
|
|
152
|
+
warnings.push('no ready task after dependency cleanup; using sequential fallback');
|
|
153
|
+
return { plan: sequentialize(tasks), warnings };
|
|
154
|
+
}
|
|
155
|
+
return { plan: tasks, warnings };
|
|
156
|
+
}
|
|
157
|
+
|
|
128
158
|
function normalizePatchOperation(raw) {
|
|
129
159
|
if (!raw || typeof raw !== 'object') return null;
|
|
130
160
|
const op = String(raw.op ?? '');
|
|
@@ -198,6 +228,14 @@ function resequence(plan) {
|
|
|
198
228
|
});
|
|
199
229
|
}
|
|
200
230
|
|
|
231
|
+
function sequentialize(plan) {
|
|
232
|
+
const next = plan.map((task) => ({ ...task, dependsOn: [] }));
|
|
233
|
+
for (let index = 1; index < next.length; index += 1) {
|
|
234
|
+
next[index].dependsOn = [taskId(next[index - 1])];
|
|
235
|
+
}
|
|
236
|
+
return next;
|
|
237
|
+
}
|
|
238
|
+
|
|
201
239
|
function hasDependencyCycle(plan) {
|
|
202
240
|
const tasks = new Map(plan.map((task) => [taskId(task), task]));
|
|
203
241
|
const visiting = new Set();
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import assert from 'node:assert/strict';
|
|
2
2
|
import test from 'node:test';
|
|
3
|
-
import { applyPlanPatch, nextReadyPlanTask, readyPlanTasks, rebasePlanPatch } from './planPatch.js';
|
|
3
|
+
import { applyPlanPatch, nextReadyPlanTask, readyPlanTasks, rebasePlanPatch, sanitizePlanForExecution } from './planPatch.js';
|
|
4
4
|
|
|
5
5
|
test('applyPlanPatch adds a task and increments the plan revision', () => {
|
|
6
6
|
const result = applyPlanPatch([
|
|
@@ -61,3 +61,25 @@ test('applyPlanPatch rejects dependency cycles', () => {
|
|
|
61
61
|
assert.equal(result.ok, false);
|
|
62
62
|
assert.equal(result.reason, 'dependency_cycle');
|
|
63
63
|
});
|
|
64
|
+
|
|
65
|
+
test('sanitizePlanForExecution removes unknown dependencies and keeps execution sequential', () => {
|
|
66
|
+
const result = sanitizePlanForExecution([
|
|
67
|
+
{ step: 1, id: 'a', description: 'A', status: 'pending', dependsOn: ['missing'] },
|
|
68
|
+
{ step: 2, id: 'b', description: 'B', status: 'pending', dependsOn: ['a'] },
|
|
69
|
+
]);
|
|
70
|
+
|
|
71
|
+
assert.deepEqual(result.plan.map((task) => task.dependsOn), [[], ['a']]);
|
|
72
|
+
assert.match(result.warnings.join('\n'), /unknown dependency/);
|
|
73
|
+
assert.deepEqual(readyPlanTasks(result.plan).map((task) => task.id), ['a']);
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
test('sanitizePlanForExecution breaks cycles with declaration-order fallback', () => {
|
|
77
|
+
const result = sanitizePlanForExecution([
|
|
78
|
+
{ step: 1, id: 'a', description: 'A', status: 'pending', dependsOn: ['b'] },
|
|
79
|
+
{ step: 2, id: 'b', description: 'B', status: 'pending', dependsOn: ['a'] },
|
|
80
|
+
]);
|
|
81
|
+
|
|
82
|
+
assert.deepEqual(result.plan.map((task) => task.dependsOn), [[], ['a']]);
|
|
83
|
+
assert.match(result.warnings.join('\n'), /cycle/);
|
|
84
|
+
assert.deepEqual(readyPlanTasks(result.plan).map((task) => task.id), ['a']);
|
|
85
|
+
});
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
import { mkdir, readFile, writeFile } from 'node:fs/promises';
|
|
2
|
+
import { join } from 'node:path';
|
|
3
|
+
|
|
4
|
+
const DEFAULT_PROFILE = `# Workspace Profile
|
|
5
|
+
|
|
6
|
+
## Summary
|
|
7
|
+
|
|
8
|
+
No profile summary yet.
|
|
9
|
+
|
|
10
|
+
## User Preferences
|
|
11
|
+
|
|
12
|
+
## Working Style
|
|
13
|
+
|
|
14
|
+
## Project Context
|
|
15
|
+
|
|
16
|
+
## Maintenance Notes
|
|
17
|
+
|
|
18
|
+
Keep this file concise. Do not store secrets, tokens, passwords, API keys, or temporary information.
|
|
19
|
+
`;
|
|
20
|
+
|
|
21
|
+
function profilePathForWorkspace(workspacePath) {
|
|
22
|
+
return join(workspacePath, '.wiki', 'profile.md');
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
function formatPreference(preference) {
|
|
26
|
+
const clean = String(preference ?? '').trim();
|
|
27
|
+
if (!clean) return '';
|
|
28
|
+
return clean.charAt(0).toUpperCase() + clean.slice(1);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
function normalizeForCompare(value) {
|
|
32
|
+
return String(value ?? '')
|
|
33
|
+
.normalize('NFD')
|
|
34
|
+
.replace(/[\u0300-\u036f]/g, '')
|
|
35
|
+
.toLowerCase()
|
|
36
|
+
.replace(/[^a-z0-9]+/g, ' ')
|
|
37
|
+
.trim();
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function insertPreference(content, preference) {
|
|
41
|
+
const line = `- ${formatPreference(preference)}`;
|
|
42
|
+
const normalizedLine = normalizeForCompare(line);
|
|
43
|
+
const hasDuplicate = content
|
|
44
|
+
.split('\n')
|
|
45
|
+
.some((existing) => normalizeForCompare(existing) === normalizedLine);
|
|
46
|
+
if (hasDuplicate) {
|
|
47
|
+
return { content, changed: false, line };
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const heading = '## User Preferences';
|
|
51
|
+
const index = content.indexOf(heading);
|
|
52
|
+
if (index === -1) {
|
|
53
|
+
const next = content.trimEnd();
|
|
54
|
+
return {
|
|
55
|
+
content: `${next}${next ? '\n\n' : ''}${heading}\n\n${line}\n`,
|
|
56
|
+
changed: true,
|
|
57
|
+
line,
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const afterHeading = index + heading.length;
|
|
62
|
+
const nextHeading = content.slice(afterHeading).search(/\n##\s+/);
|
|
63
|
+
if (nextHeading === -1) {
|
|
64
|
+
const prefix = content.slice(0, afterHeading).trimEnd();
|
|
65
|
+
const suffix = content.slice(afterHeading).trim();
|
|
66
|
+
return {
|
|
67
|
+
content: `${prefix}\n\n${suffix ? `${suffix}\n` : ''}${line}\n`,
|
|
68
|
+
changed: true,
|
|
69
|
+
line,
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const insertAt = afterHeading + nextHeading;
|
|
74
|
+
const before = content.slice(0, insertAt).trimEnd();
|
|
75
|
+
const after = content.slice(insertAt).replace(/^\n+/, '\n');
|
|
76
|
+
return {
|
|
77
|
+
content: `${before}\n${line}\n${after}`,
|
|
78
|
+
changed: true,
|
|
79
|
+
line,
|
|
80
|
+
};
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export async function updateWorkspaceProfilePreference(session, preference) {
|
|
84
|
+
const workspacePath = session?.workspacePath;
|
|
85
|
+
if (!workspacePath) {
|
|
86
|
+
return {
|
|
87
|
+
ok: false,
|
|
88
|
+
message: 'Profil non modifié : aucun workspace chargé. Utilise /use <workspace>.',
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
const profilePath = profilePathForWorkspace(workspacePath);
|
|
92
|
+
await mkdir(join(workspacePath, '.wiki'), { recursive: true });
|
|
93
|
+
let before = DEFAULT_PROFILE;
|
|
94
|
+
try {
|
|
95
|
+
before = await readFile(profilePath, 'utf8');
|
|
96
|
+
} catch (err) {
|
|
97
|
+
if (err?.code !== 'ENOENT') throw err;
|
|
98
|
+
}
|
|
99
|
+
const inserted = insertPreference(before, preference);
|
|
100
|
+
if (inserted.changed) {
|
|
101
|
+
await writeFile(profilePath, inserted.content, 'utf8');
|
|
102
|
+
}
|
|
103
|
+
return {
|
|
104
|
+
ok: true,
|
|
105
|
+
changed: inserted.changed,
|
|
106
|
+
preference: inserted.line.replace(/^- /, ''),
|
|
107
|
+
message: inserted.changed
|
|
108
|
+
? `Profil mis à jour : ${inserted.line.replace(/^- /, '')}`
|
|
109
|
+
: `Profil déjà à jour : ${inserted.line.replace(/^- /, '')}`,
|
|
110
|
+
};
|
|
111
|
+
}
|
package/src/core/skills.js
CHANGED
package/src/runtime/client.js
CHANGED
|
@@ -97,6 +97,18 @@ export async function postRuntimeCancel({
|
|
|
97
97
|
return response.json();
|
|
98
98
|
}
|
|
99
99
|
|
|
100
|
+
export async function postRuntimeShutdown({
|
|
101
|
+
url = runtimeUrlFromEnv(),
|
|
102
|
+
token = runtimeToken(),
|
|
103
|
+
} = {}) {
|
|
104
|
+
const response = await fetch(runtimeEndpoint(url, '/shutdown'), {
|
|
105
|
+
method: 'POST',
|
|
106
|
+
headers: runtimeHeaders(token),
|
|
107
|
+
});
|
|
108
|
+
if (!response.ok) throw new Error(`Runtime shutdown failed: HTTP ${response.status}`);
|
|
109
|
+
return response.json();
|
|
110
|
+
}
|
|
111
|
+
|
|
100
112
|
export async function postRuntimeResume({
|
|
101
113
|
url = runtimeUrlFromEnv(),
|
|
102
114
|
token = runtimeToken(),
|
|
@@ -0,0 +1,359 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import test from 'node:test';
|
|
3
|
+
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
4
|
+
import { runRuntimeAgenticWorkflow } from './runner.js';
|
|
5
|
+
import { startRuntimeServer } from './server.js';
|
|
6
|
+
|
|
7
|
+
// Executable version of plan-0.11.5-hotfix-final.md §3 ("La recette — le seul
|
|
8
|
+
// critère de sortie qui compte"). Each test below is one row of that table,
|
|
9
|
+
// referenced by its recipe number. Correctifs 1-3 and 8 (runner.js's
|
|
10
|
+
// conversational short-circuit, plan sanitization, and no-replan-on-vague
|
|
11
|
+
// paths) are exercised with a mocked agent standing in for the LLM's
|
|
12
|
+
// per-turn decision, exactly like the existing runner.test.js unit tests --
|
|
13
|
+
// this file's job is traceability to the recipe, not a new test seam.
|
|
14
|
+
// Recipes 4-6 drive the real scheduler (runAgenticLoop / runRuntimeParallelPlan,
|
|
15
|
+
// imported and executed for real, never reimplemented). Recipe 7 drives the
|
|
16
|
+
// real control-lane HTTP endpoint. Per plan §4.1: this file is the CI gate --
|
|
17
|
+
// no release ships while it is red.
|
|
18
|
+
|
|
19
|
+
function baseSession(overrides = {}) {
|
|
20
|
+
return { activities: {}, headlessPlan: null, ...overrides };
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
function eventTypes(session) {
|
|
24
|
+
return (session.agentEvents ?? []).map((event) => event.type);
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
async function waitFor(predicate, timeoutMs = 500) {
|
|
28
|
+
const deadline = Date.now() + timeoutMs;
|
|
29
|
+
while (Date.now() < deadline) {
|
|
30
|
+
if (predicate()) return;
|
|
31
|
+
await new Promise((resolve) => setTimeout(resolve, 1));
|
|
32
|
+
}
|
|
33
|
+
assert.fail('condition was not met before timeout');
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
test('Recipe #1 — "salut" gets a plain reply: no plan, no activity, no job, run done', async () => {
|
|
37
|
+
const session = baseSession({
|
|
38
|
+
llm: {
|
|
39
|
+
async completeWithTools() {
|
|
40
|
+
assert.fail('a plain greeting must never reach the evaluator or replanner');
|
|
41
|
+
},
|
|
42
|
+
},
|
|
43
|
+
});
|
|
44
|
+
const agent = {
|
|
45
|
+
async invoke() {
|
|
46
|
+
return { response: 'Bonjour ! Comment puis-je vous aider ?' };
|
|
47
|
+
},
|
|
48
|
+
};
|
|
49
|
+
|
|
50
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'salut', {
|
|
51
|
+
runId: 'recipe-1',
|
|
52
|
+
timeoutMs: 1000,
|
|
53
|
+
maxTurns: 1,
|
|
54
|
+
maxReplans: 1,
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
assert.equal(result.ok, true);
|
|
58
|
+
assert.equal(session.headlessPlan, null);
|
|
59
|
+
assert.equal(Object.keys(session.activities).length, 0);
|
|
60
|
+
const types = eventTypes(session);
|
|
61
|
+
assert.ok(types.includes('run_done'));
|
|
62
|
+
assert.equal(types.includes('run_evaluated'), false);
|
|
63
|
+
assert.equal(types.includes('run_replanned'), false);
|
|
64
|
+
assert.equal(types.includes('tool_call_started'), false);
|
|
65
|
+
assert.equal(types.includes('plan_set'), false);
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
test('Recipe #2 — "quel est le profil actif ?" answers in conversation, never poses a plan', async () => {
|
|
69
|
+
const session = baseSession({
|
|
70
|
+
llm: {
|
|
71
|
+
async completeWithTools() {
|
|
72
|
+
assert.fail('a read-only config question must never reach the evaluator or replanner');
|
|
73
|
+
},
|
|
74
|
+
},
|
|
75
|
+
});
|
|
76
|
+
const agent = {
|
|
77
|
+
async invoke({ session: turnSession }) {
|
|
78
|
+
dispatchAgentEvent(turnSession, createAgentEvent('tool_call_started', {
|
|
79
|
+
origin: 'tool',
|
|
80
|
+
payload: { name: 'shell.status', args: '{}', summary: 'calling...' },
|
|
81
|
+
}));
|
|
82
|
+
dispatchAgentEvent(turnSession, createAgentEvent('tool_call_result', {
|
|
83
|
+
origin: 'tool',
|
|
84
|
+
payload: { name: 'shell.status', ok: true, result: 'profile: albert-openai (mistral-large)', summary: 'done' },
|
|
85
|
+
}));
|
|
86
|
+
return { response: 'Le profil actif est albert-openai (mistral-large).' };
|
|
87
|
+
},
|
|
88
|
+
};
|
|
89
|
+
|
|
90
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'quel est le profil actif ?', {
|
|
91
|
+
runId: 'recipe-2',
|
|
92
|
+
timeoutMs: 1000,
|
|
93
|
+
maxTurns: 1,
|
|
94
|
+
maxReplans: 1,
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
assert.equal(result.ok, true);
|
|
98
|
+
assert.equal(session.headlessPlan, null);
|
|
99
|
+
const types = eventTypes(session);
|
|
100
|
+
assert.ok(types.includes('tool_call_started'));
|
|
101
|
+
assert.ok(types.includes('run_done'));
|
|
102
|
+
assert.equal(types.includes('plan_set'), false);
|
|
103
|
+
assert.equal(types.includes('run_evaluated'), false);
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
test('Recipe #3 — "où en est le dernier build ?" reads status, starts no new production run', async () => {
|
|
107
|
+
const session = baseSession({
|
|
108
|
+
llm: {
|
|
109
|
+
async completeWithTools() {
|
|
110
|
+
assert.fail('a status question must never reach the evaluator or replanner');
|
|
111
|
+
},
|
|
112
|
+
},
|
|
113
|
+
});
|
|
114
|
+
const agent = {
|
|
115
|
+
async invoke({ session: turnSession }) {
|
|
116
|
+
dispatchAgentEvent(turnSession, createAgentEvent('tool_call_started', {
|
|
117
|
+
origin: 'tool',
|
|
118
|
+
payload: { name: 'production.production_job_status', args: '{}', summary: 'calling...' },
|
|
119
|
+
}));
|
|
120
|
+
dispatchAgentEvent(turnSession, createAgentEvent('tool_call_result', {
|
|
121
|
+
origin: 'tool',
|
|
122
|
+
payload: { name: 'production.production_job_status', ok: true, result: 'last build: done', summary: 'done' },
|
|
123
|
+
}));
|
|
124
|
+
return { response: 'Le dernier build est terminé avec succès.' };
|
|
125
|
+
},
|
|
126
|
+
};
|
|
127
|
+
|
|
128
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'où en est le dernier build ?', {
|
|
129
|
+
runId: 'recipe-3',
|
|
130
|
+
timeoutMs: 1000,
|
|
131
|
+
maxTurns: 1,
|
|
132
|
+
maxReplans: 1,
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
assert.equal(result.ok, true);
|
|
136
|
+
assert.equal(session.headlessPlan, null);
|
|
137
|
+
assert.equal(Object.keys(session.activities).length, 0);
|
|
138
|
+
const types = eventTypes(session);
|
|
139
|
+
assert.equal(types.includes('activity_upserted'), false);
|
|
140
|
+
assert.equal(types.includes('plan_set'), false);
|
|
141
|
+
});
|
|
142
|
+
|
|
143
|
+
function buildSingleTaskAgent({ taskId, description, finalResponse }) {
|
|
144
|
+
let turn = 0;
|
|
145
|
+
return {
|
|
146
|
+
async invoke({ session: turnSession }) {
|
|
147
|
+
turn += 1;
|
|
148
|
+
if (turn === 1) {
|
|
149
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
|
|
150
|
+
origin: 'tool',
|
|
151
|
+
payload: { steps: [{ step: 1, id: taskId, description, status: 'pending', dependsOn: [] }] },
|
|
152
|
+
}));
|
|
153
|
+
return { response: `Je lance ${description}.` };
|
|
154
|
+
}
|
|
155
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_step_updated', {
|
|
156
|
+
origin: 'tool',
|
|
157
|
+
payload: { step: 1, status: 'done' },
|
|
158
|
+
}));
|
|
159
|
+
return { response: finalResponse };
|
|
160
|
+
},
|
|
161
|
+
};
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
test('Recipe #4 — "lance le doctor": one-task plan, progress visible, summarized result', async () => {
|
|
165
|
+
const session = baseSession();
|
|
166
|
+
const agent = buildSingleTaskAgent({
|
|
167
|
+
taskId: 'doctor',
|
|
168
|
+
description: 'Diagnostic doctor',
|
|
169
|
+
finalResponse: 'Doctor terminé : aucun problème détecté.',
|
|
170
|
+
});
|
|
171
|
+
|
|
172
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'lance le doctor', {
|
|
173
|
+
runId: 'recipe-4',
|
|
174
|
+
timeoutMs: 1000,
|
|
175
|
+
maxTurns: 3,
|
|
176
|
+
maxReplans: 1,
|
|
177
|
+
evaluate: false,
|
|
178
|
+
});
|
|
179
|
+
|
|
180
|
+
assert.equal(result.ok, true);
|
|
181
|
+
assert.equal(session.headlessPlan.length, 1);
|
|
182
|
+
assert.equal(session.headlessPlan[0].status, 'done');
|
|
183
|
+
const types = eventTypes(session);
|
|
184
|
+
assert.deepEqual(types.filter((type) => type === 'plan_set' || type === 'plan_step_updated'), ['plan_set', 'plan_step_updated']);
|
|
185
|
+
assert.ok(types.includes('run_done'));
|
|
186
|
+
assert.equal(session.agentProjection.conversation.at(-1).content, 'Doctor terminé : aucun problème détecté.');
|
|
187
|
+
});
|
|
188
|
+
|
|
189
|
+
test('Recipe #5 — "construis le livrable X": plan posed, job runs, progress, done, summarized', async () => {
|
|
190
|
+
const session = baseSession();
|
|
191
|
+
const agent = buildSingleTaskAgent({
|
|
192
|
+
taskId: 'build-x',
|
|
193
|
+
description: 'Construire le livrable X',
|
|
194
|
+
finalResponse: 'Le livrable X est construit.',
|
|
195
|
+
});
|
|
196
|
+
|
|
197
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'construis le livrable X', {
|
|
198
|
+
runId: 'recipe-5',
|
|
199
|
+
timeoutMs: 1000,
|
|
200
|
+
maxTurns: 3,
|
|
201
|
+
maxReplans: 1,
|
|
202
|
+
evaluate: false,
|
|
203
|
+
});
|
|
204
|
+
|
|
205
|
+
assert.equal(result.ok, true);
|
|
206
|
+
assert.equal(session.headlessPlan[0].status, 'done');
|
|
207
|
+
assert.ok(eventTypes(session).includes('run_done'));
|
|
208
|
+
assert.equal(session.agentProjection.conversation.at(-1).content, 'Le livrable X est construit.');
|
|
209
|
+
});
|
|
210
|
+
|
|
211
|
+
function buildTwoParallelTasksAgent() {
|
|
212
|
+
const started = [];
|
|
213
|
+
const release = {};
|
|
214
|
+
let parentTurnDone = false;
|
|
215
|
+
return {
|
|
216
|
+
started,
|
|
217
|
+
release,
|
|
218
|
+
async invoke({ input, session: turnSession }) {
|
|
219
|
+
const taskMatch = input.match(/Task id: (\w+)/);
|
|
220
|
+
if (!taskMatch) {
|
|
221
|
+
if (!parentTurnDone) {
|
|
222
|
+
parentTurnDone = true;
|
|
223
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
|
|
224
|
+
origin: 'tool',
|
|
225
|
+
payload: {
|
|
226
|
+
steps: [
|
|
227
|
+
{ step: 1, id: 'x', description: 'Construire X', status: 'pending', dependsOn: [] },
|
|
228
|
+
{ step: 2, id: 'y', description: 'Construire Y', status: 'pending', dependsOn: [] },
|
|
229
|
+
],
|
|
230
|
+
},
|
|
231
|
+
}));
|
|
232
|
+
return { response: 'Je construis X et Y en parallèle.' };
|
|
233
|
+
}
|
|
234
|
+
return { response: 'X et Y sont construits.' };
|
|
235
|
+
}
|
|
236
|
+
const id = taskMatch[1];
|
|
237
|
+
started.push(id);
|
|
238
|
+
await new Promise((resolve) => { release[id] = resolve; });
|
|
239
|
+
return { response: `${id} construit.` };
|
|
240
|
+
},
|
|
241
|
+
};
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
test('Recipe #6 — "construis X et Y": two tasks run in parallel, converge, done', async () => {
|
|
245
|
+
const session = baseSession();
|
|
246
|
+
const agent = buildTwoParallelTasksAgent();
|
|
247
|
+
|
|
248
|
+
const running = runRuntimeAgenticWorkflow(agent, session, 'construis X et Y', {
|
|
249
|
+
runId: 'recipe-6',
|
|
250
|
+
timeoutMs: 2000,
|
|
251
|
+
maxTurns: 3,
|
|
252
|
+
maxReplans: 1,
|
|
253
|
+
evaluate: false,
|
|
254
|
+
});
|
|
255
|
+
|
|
256
|
+
await waitFor(() => agent.started.includes('x') && agent.started.includes('y'));
|
|
257
|
+
assert.deepEqual(
|
|
258
|
+
session.headlessPlan.filter((step) => ['x', 'y'].includes(step.id)).map((step) => step.status),
|
|
259
|
+
['running', 'running'],
|
|
260
|
+
);
|
|
261
|
+
agent.release.x();
|
|
262
|
+
agent.release.y();
|
|
263
|
+
|
|
264
|
+
const result = await running;
|
|
265
|
+
|
|
266
|
+
assert.equal(result.ok, true);
|
|
267
|
+
assert.deepEqual(session.headlessPlan.map((step) => step.status), ['done', 'done']);
|
|
268
|
+
});
|
|
269
|
+
|
|
270
|
+
test('Recipe #7 — "où en es-tu ?" during an active run: status in conversation, run continues, no new run', async (t) => {
|
|
271
|
+
const session = { workspace: 'juno', controlQueue: [] };
|
|
272
|
+
let runCount = 0;
|
|
273
|
+
let handle;
|
|
274
|
+
try {
|
|
275
|
+
handle = await startRuntimeServer({
|
|
276
|
+
host: '127.0.0.1',
|
|
277
|
+
port: 0,
|
|
278
|
+
store: {
|
|
279
|
+
dbPath: ':memory:',
|
|
280
|
+
getState: () => ({
|
|
281
|
+
status: 'running',
|
|
282
|
+
plan: [{ step: 1, description: 'Construire le livrable X', status: 'running' }],
|
|
283
|
+
queue: [],
|
|
284
|
+
approvals: [],
|
|
285
|
+
summary: null,
|
|
286
|
+
}),
|
|
287
|
+
listEvents: () => [],
|
|
288
|
+
},
|
|
289
|
+
getContext: async () => ({
|
|
290
|
+
workspace: 'juno',
|
|
291
|
+
session,
|
|
292
|
+
running: true,
|
|
293
|
+
currentAbortController: new AbortController(),
|
|
294
|
+
}),
|
|
295
|
+
run: async () => { runCount += 1; },
|
|
296
|
+
});
|
|
297
|
+
} catch (err) {
|
|
298
|
+
if (err?.code === 'EPERM') {
|
|
299
|
+
t.skip('network listen is not permitted in this sandbox');
|
|
300
|
+
return;
|
|
301
|
+
}
|
|
302
|
+
throw err;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
try {
|
|
306
|
+
const response = await fetch(`http://127.0.0.1:${handle.port}/control?workspace=juno`, {
|
|
307
|
+
method: 'POST',
|
|
308
|
+
headers: { 'Content-Type': 'application/json' },
|
|
309
|
+
body: JSON.stringify({ action: 'message', input: 'où en es-tu ?' }),
|
|
310
|
+
});
|
|
311
|
+
assert.equal(response.status, 200);
|
|
312
|
+
const body = await response.json();
|
|
313
|
+
assert.equal(body.kind, 'observe');
|
|
314
|
+
assert.match(body.explanation, /Construire le livrable X/);
|
|
315
|
+
assert.equal(session.controlQueue.length, 0);
|
|
316
|
+
assert.equal(runCount, 0);
|
|
317
|
+
} finally {
|
|
318
|
+
await handle.close();
|
|
319
|
+
}
|
|
320
|
+
});
|
|
321
|
+
|
|
322
|
+
test('Recipe #8 — vague request gets clarification, never a job, never boilerplate', async () => {
|
|
323
|
+
const session = baseSession({
|
|
324
|
+
headlessPlan: [{ step: 1, id: 'task', description: 'Task', status: 'done' }],
|
|
325
|
+
llm: {
|
|
326
|
+
async completeWithTools({ system }) {
|
|
327
|
+
assert.match(system, /strict evaluator/);
|
|
328
|
+
return { content: '{"ok":false,"reason":"demande vague / objectif indefini","suggestedAction":"clarifier l objectif"}' };
|
|
329
|
+
},
|
|
330
|
+
},
|
|
331
|
+
});
|
|
332
|
+
const agent = {
|
|
333
|
+
async invoke({ session: turnSession }) {
|
|
334
|
+
if (turnSession.headlessPlan === null) {
|
|
335
|
+
dispatchAgentEvent(turnSession, createAgentEvent('plan_set', {
|
|
336
|
+
origin: 'tool',
|
|
337
|
+
payload: { steps: [{ step: 1, id: 'task', description: 'Task', status: 'done' }] },
|
|
338
|
+
}));
|
|
339
|
+
}
|
|
340
|
+
return { response: 'Terminé.' };
|
|
341
|
+
},
|
|
342
|
+
};
|
|
343
|
+
|
|
344
|
+
const result = await runRuntimeAgenticWorkflow(agent, session, 'dis-moi une bêtise puis stop', {
|
|
345
|
+
runId: 'recipe-8',
|
|
346
|
+
timeoutMs: 1000,
|
|
347
|
+
maxTurns: 1,
|
|
348
|
+
maxReplans: 1,
|
|
349
|
+
});
|
|
350
|
+
|
|
351
|
+
assert.equal(result.ok, true);
|
|
352
|
+
assert.equal(result.clarified, true);
|
|
353
|
+
const lastMessage = session.agentProjection.conversation.at(-1).content;
|
|
354
|
+
assert.ok(lastMessage.includes('clarifier'));
|
|
355
|
+
assert.doesNotMatch(lastMessage, /Donna is active|Plan is stalled/i);
|
|
356
|
+
const types = eventTypes(session);
|
|
357
|
+
assert.equal(types.includes('run_replanned'), false);
|
|
358
|
+
assert.equal(types.includes('tool_call_started'), false);
|
|
359
|
+
});
|