@dotdrelle/wiki-manager 0.12.11 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +6 -0
- package/docker-compose.yml +1 -1
- package/package.json +1 -1
- package/src/agent/graph.js +377 -142
- package/src/agent/graph.test.js +576 -34
- package/src/agent/llm.js +5 -5
- package/src/cli/wiki-manager.js +294 -9
- package/src/cli/wiki-manager.test.js +28 -0
- package/src/commands/slash.js +80 -13
- package/src/commands/slash.test.js +9 -1
- package/src/contracts/schemas.js +33 -0
- package/src/contracts/schemas.test.js +14 -0
- package/src/core/agentEvents.js +6 -0
- package/src/core/agentEvents.test.js +26 -0
- package/src/core/agentLoop.js +15 -16
- package/src/core/agentLoop.test.js +9 -7
- package/src/core/buildInfo.json +2 -2
- package/src/core/mcp.js +13 -6
- package/src/core/mcp.test.js +0 -12
- package/src/core/skills.js +0 -28
- package/src/orchestrator/capabilityRegistry.js +14 -0
- package/src/orchestrator/capabilityRegistry.test.js +12 -1
- package/src/orchestrator/dependencyResolver.js +10 -1
- package/src/orchestrator/dispatcher.js +34 -3
- package/src/orchestrator/dispatcher.test.js +34 -0
- package/src/orchestrator/objectiveResolver.js +79 -0
- package/src/orchestrator/objectiveResolver.test.js +50 -0
- package/src/orchestrator/scheduler.js +24 -0
- package/src/orchestrator/scheduler.test.js +65 -1
- package/src/runtime/client.js +34 -2
- package/src/runtime/lifecycle.js +1 -1
- package/src/runtime/recoveryManager.js +14 -7
- package/src/runtime/runner.js +214 -14
- package/src/runtime/runner.test.js +100 -2
- package/src/runtime/server.js +43 -3
- package/src/runtime/supervisor.js +65 -1
- package/src/runtime/supervisor.test.js +80 -0
- package/src/shell/LeftPane.tsx +9 -2
- package/src/shell/repl.js +57 -42
- package/src/shell/repl.test.js +81 -12
- package/src/shell/tui.tsx +26 -3
- package/src/shell/useSession.ts +15 -3
package/src/agent/graph.test.js
CHANGED
|
@@ -3,7 +3,57 @@ import test from 'node:test';
|
|
|
3
3
|
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs';
|
|
4
4
|
import { tmpdir } from 'node:os';
|
|
5
5
|
import { join } from 'node:path';
|
|
6
|
-
import { buildAgentSystemPrompt,
|
|
6
|
+
import { buildAgentSystemPrompt, createAgentGraph, invalidSuggestedSlashCommands, invalidUserFacingToolNames, knownCapabilityIds, normalizeToolArgumentsFromSchema } from './graph.js';
|
|
7
|
+
|
|
8
|
+
test('user-facing response guard hides MCP identifiers generically', () => {
|
|
9
|
+
const session = sessionBase();
|
|
10
|
+
assert.deepEqual(
|
|
11
|
+
invalidUserFacingToolNames('Utilisez production__production_start_job.', session),
|
|
12
|
+
['production__production_start_job'],
|
|
13
|
+
);
|
|
14
|
+
});
|
|
15
|
+
|
|
16
|
+
test('Donna cannot answer an explicit action with manual instructions instead of delegating', async () => {
|
|
17
|
+
const originalFetch = globalThis.fetch;
|
|
18
|
+
let delegated = false;
|
|
19
|
+
globalThis.fetch = async (url) => {
|
|
20
|
+
delegated = String(url).includes('/delegate');
|
|
21
|
+
return { ok: true, status: 202, json: async () => ({ accepted: true, runId: 'run-action', delegation: { tasks: 2, agent: 'production' } }) };
|
|
22
|
+
};
|
|
23
|
+
let mainCalls = 0;
|
|
24
|
+
const session = sessionBase({
|
|
25
|
+
runtime: { url: 'http://runtime.test' },
|
|
26
|
+
llm: {
|
|
27
|
+
async completeWithTools({ tools }) {
|
|
28
|
+
if (tools.length === 0) return { content: '{"action":true}', message: { role: 'assistant', content: '{"action":true}' }, tool_calls: null };
|
|
29
|
+
mainCalls += 1;
|
|
30
|
+
if (mainCalls === 1) {
|
|
31
|
+
return {
|
|
32
|
+
content: 'Déplacez raw/untracked/demo.md vers raw/ puis utilisez wiki__wiki_workspace_status.',
|
|
33
|
+
message: { role: 'assistant', content: 'instructions manuelles' },
|
|
34
|
+
tool_calls: null,
|
|
35
|
+
};
|
|
36
|
+
}
|
|
37
|
+
if (mainCalls === 2) {
|
|
38
|
+
return {
|
|
39
|
+
content: null,
|
|
40
|
+
message: { role: 'assistant', content: null },
|
|
41
|
+
tool_calls: [{ id: 'delegate', type: 'function', function: { name: 'runtime__delegate', arguments: '{"objective":"Lance ingestion"}' } }],
|
|
42
|
+
};
|
|
43
|
+
}
|
|
44
|
+
return { content: 'Plan soumis.', message: { role: 'assistant', content: 'Plan soumis.' }, tool_calls: null };
|
|
45
|
+
},
|
|
46
|
+
},
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
try {
|
|
50
|
+
const result = await createAgentGraph().invoke({ input: 'lance ingestion', session });
|
|
51
|
+
assert.equal(delegated, true);
|
|
52
|
+
assert.equal(result.response, 'Plan soumis.');
|
|
53
|
+
} finally {
|
|
54
|
+
globalThis.fetch = originalFetch;
|
|
55
|
+
}
|
|
56
|
+
});
|
|
7
57
|
|
|
8
58
|
function sessionBase(overrides = {}) {
|
|
9
59
|
return {
|
|
@@ -183,6 +233,258 @@ test('agent graph does not pre-filter mutating MCP tools for config questions',
|
|
|
183
233
|
assert.ok(seenTools.includes('shell__run_command'));
|
|
184
234
|
});
|
|
185
235
|
|
|
236
|
+
test('interactive Donna delegates provider execution to the runtime orchestrator', async () => {
|
|
237
|
+
const seenTools = [];
|
|
238
|
+
const session = sessionBase({
|
|
239
|
+
runtime: { url: 'http://runtime.test' },
|
|
240
|
+
agentRegistrySnapshot: ingestAgentSnapshot(),
|
|
241
|
+
mcp: {
|
|
242
|
+
production: {
|
|
243
|
+
status: 'connected',
|
|
244
|
+
url: 'http://127.0.0.1:3000/mcp/',
|
|
245
|
+
tools: [
|
|
246
|
+
{ name: 'agent_plan', description: 'Plan tasks', inputSchema: { type: 'object', properties: {} } },
|
|
247
|
+
{ name: 'agent_execute', description: 'Execute a task', inputSchema: { type: 'object', properties: {} } },
|
|
248
|
+
{ name: 'agent_status', description: 'Read task status', inputSchema: { type: 'object', properties: {} } },
|
|
249
|
+
{ name: 'production_start_job', description: 'Legacy job start', inputSchema: { type: 'object', properties: {} } },
|
|
250
|
+
{ name: 'production_status', description: 'Read production status', inputSchema: { type: 'object', properties: {} } },
|
|
251
|
+
],
|
|
252
|
+
},
|
|
253
|
+
},
|
|
254
|
+
llm: {
|
|
255
|
+
async completeWithTools({ tools }) {
|
|
256
|
+
seenTools.push(...tools.map((tool) => tool.function.name));
|
|
257
|
+
return { content: 'Prêt.', message: { role: 'assistant', content: 'Prêt.' }, tool_calls: null };
|
|
258
|
+
},
|
|
259
|
+
},
|
|
260
|
+
});
|
|
261
|
+
|
|
262
|
+
await createAgentGraph().invoke({ input: 'bonjour', session });
|
|
263
|
+
|
|
264
|
+
assert.ok(seenTools.includes('runtime__delegate'));
|
|
265
|
+
assert.ok(seenTools.includes('production__agent_status'));
|
|
266
|
+
assert.ok(seenTools.includes('production__production_status'));
|
|
267
|
+
assert.ok(!seenTools.includes('production__agent_plan'));
|
|
268
|
+
assert.ok(!seenTools.includes('production__agent_execute'));
|
|
269
|
+
assert.ok(!seenTools.includes('production__production_start_job'));
|
|
270
|
+
assert.ok(!seenTools.includes('wiki__plan_set'));
|
|
271
|
+
assert.ok(!seenTools.includes('wiki__plan_done'));
|
|
272
|
+
});
|
|
273
|
+
|
|
274
|
+
test('interactive Donna can delegate while the shell capability snapshot is still empty', async () => {
|
|
275
|
+
const seenTools = [];
|
|
276
|
+
const session = sessionBase({
|
|
277
|
+
runtime: { url: 'http://runtime.test' },
|
|
278
|
+
agentRegistrySnapshot: [],
|
|
279
|
+
llm: {
|
|
280
|
+
async completeWithTools({ tools }) {
|
|
281
|
+
seenTools.push(...tools.map((tool) => tool.function.name));
|
|
282
|
+
return { content: 'Prêt.', message: { role: 'assistant', content: 'Prêt.' }, tool_calls: null };
|
|
283
|
+
},
|
|
284
|
+
},
|
|
285
|
+
});
|
|
286
|
+
|
|
287
|
+
await createAgentGraph().invoke({ input: 'bonjour', session });
|
|
288
|
+
|
|
289
|
+
assert.ok(seenTools.includes('runtime__delegate'));
|
|
290
|
+
});
|
|
291
|
+
|
|
292
|
+
function ingestAgentSnapshot() {
|
|
293
|
+
return [{
|
|
294
|
+
agentInstanceId: 'production-ingest',
|
|
295
|
+
serverName: 'production',
|
|
296
|
+
health: 'available',
|
|
297
|
+
description: {
|
|
298
|
+
contractVersion: '1',
|
|
299
|
+
agentType: 'production',
|
|
300
|
+
displayName: 'Production',
|
|
301
|
+
capabilities: [{
|
|
302
|
+
id: 'knowledge.update',
|
|
303
|
+
version: '1',
|
|
304
|
+
supportedOperations: ['ingest', 'ingest_plan', 'ingest_apply'],
|
|
305
|
+
}],
|
|
306
|
+
},
|
|
307
|
+
}];
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
test('runtime delegation tool declares only its canonical natural-language objective', async () => {
|
|
311
|
+
let delegationTool = null;
|
|
312
|
+
const session = sessionBase({
|
|
313
|
+
runtime: { url: 'http://runtime.test' },
|
|
314
|
+
agentRegistrySnapshot: ingestAgentSnapshot(),
|
|
315
|
+
llm: {
|
|
316
|
+
async completeWithTools({ tools }) {
|
|
317
|
+
delegationTool ??= tools.find((tool) => tool.function.name === 'runtime__delegate');
|
|
318
|
+
return { content: 'Prêt.', message: { role: 'assistant', content: 'Prêt.' }, tool_calls: null };
|
|
319
|
+
},
|
|
320
|
+
},
|
|
321
|
+
});
|
|
322
|
+
|
|
323
|
+
await createAgentGraph().invoke({ input: 'bonjour', session });
|
|
324
|
+
|
|
325
|
+
assert.deepEqual(Object.keys(delegationTool.function.parameters.properties), ['objective']);
|
|
326
|
+
assert.deepEqual(delegationTool.function.parameters.required, ['objective']);
|
|
327
|
+
});
|
|
328
|
+
|
|
329
|
+
test('tool argument normalization repairs only an unambiguous schema-compatible field name', () => {
|
|
330
|
+
const schema = {
|
|
331
|
+
type: 'object',
|
|
332
|
+
properties: { objective: { type: 'string' } },
|
|
333
|
+
required: ['objective'],
|
|
334
|
+
additionalProperties: false,
|
|
335
|
+
};
|
|
336
|
+
assert.deepEqual(
|
|
337
|
+
normalizeToolArgumentsFromSchema({ input: 'Ingérer les fichiers' }, schema),
|
|
338
|
+
{ objective: 'Ingérer les fichiers' },
|
|
339
|
+
);
|
|
340
|
+
assert.deepEqual(
|
|
341
|
+
normalizeToolArgumentsFromSchema({ input: 'x', other: 'y' }, schema),
|
|
342
|
+
{ input: 'x', other: 'y' },
|
|
343
|
+
);
|
|
344
|
+
assert.deepEqual(
|
|
345
|
+
normalizeToolArgumentsFromSchema({ input: 42 }, schema),
|
|
346
|
+
{ input: 42 },
|
|
347
|
+
);
|
|
348
|
+
});
|
|
349
|
+
|
|
350
|
+
test('Donna delegates the objective without choosing technical identifiers', async () => {
|
|
351
|
+
const originalFetch = globalThis.fetch;
|
|
352
|
+
let request = null;
|
|
353
|
+
globalThis.fetch = async (url, options) => {
|
|
354
|
+
request = { url: String(url), body: JSON.parse(options.body) };
|
|
355
|
+
return { ok: true, status: 202, json: async () => ({ accepted: true, runId: 'run-1', delegation: { tasks: 5, agent: 'production' } }) };
|
|
356
|
+
};
|
|
357
|
+
let calls = 0;
|
|
358
|
+
const session = sessionBase({
|
|
359
|
+
runtime: { url: 'http://runtime.test' },
|
|
360
|
+
agentRegistrySnapshot: ingestAgentSnapshot(),
|
|
361
|
+
llm: {
|
|
362
|
+
async completeWithTools({ messages }) {
|
|
363
|
+
calls += 1;
|
|
364
|
+
if (calls === 1) {
|
|
365
|
+
return {
|
|
366
|
+
content: null,
|
|
367
|
+
message: { role: 'assistant', content: null },
|
|
368
|
+
tool_calls: [{
|
|
369
|
+
id: 'delegate',
|
|
370
|
+
type: 'function',
|
|
371
|
+
function: { name: 'runtime__delegate', arguments: '{"input":"Ingérer tous les fichiers en attente"}' },
|
|
372
|
+
}],
|
|
373
|
+
};
|
|
374
|
+
}
|
|
375
|
+
return { content: 'Plan validé.', message: { role: 'assistant', content: 'Plan validé.' }, tool_calls: null };
|
|
376
|
+
},
|
|
377
|
+
},
|
|
378
|
+
});
|
|
379
|
+
|
|
380
|
+
try {
|
|
381
|
+
await createAgentGraph().invoke({ input: 'ingère tout', session });
|
|
382
|
+
assert.match(request.url, /\/delegate/);
|
|
383
|
+
assert.deepEqual(request.body, { objective: 'Ingérer tous les fichiers en attente', workspace: 'docs' });
|
|
384
|
+
assert.equal('capability' in request.body, false);
|
|
385
|
+
assert.equal('operation' in request.body, false);
|
|
386
|
+
} finally {
|
|
387
|
+
globalThis.fetch = originalFetch;
|
|
388
|
+
}
|
|
389
|
+
});
|
|
390
|
+
|
|
391
|
+
test('runtime status does not manufacture a plan', async () => {
|
|
392
|
+
const originalFetch = globalThis.fetch;
|
|
393
|
+
globalThis.fetch = async () => ({
|
|
394
|
+
ok: true,
|
|
395
|
+
status: 200,
|
|
396
|
+
json: async () => ({ status: 'idle', running: false, plan: [], queue: [], controlQueue: [], approvals: [] }),
|
|
397
|
+
});
|
|
398
|
+
let calls = 0;
|
|
399
|
+
const session = sessionBase({
|
|
400
|
+
runtime: { url: 'http://runtime.test' },
|
|
401
|
+
llm: {
|
|
402
|
+
async completeWithTools() {
|
|
403
|
+
calls += 1;
|
|
404
|
+
if (calls === 1) {
|
|
405
|
+
return {
|
|
406
|
+
content: null,
|
|
407
|
+
message: { role: 'assistant', content: null },
|
|
408
|
+
tool_calls: [{ id: 'status', type: 'function', function: { name: 'runtime__status', arguments: '{}' } }],
|
|
409
|
+
};
|
|
410
|
+
}
|
|
411
|
+
return { content: 'Aucun run actif.', message: { role: 'assistant', content: 'Aucun run actif.' }, tool_calls: null };
|
|
412
|
+
},
|
|
413
|
+
},
|
|
414
|
+
});
|
|
415
|
+
|
|
416
|
+
try {
|
|
417
|
+
const result = await createAgentGraph().invoke({ input: 'où en est le travail ?', session });
|
|
418
|
+
assert.equal(result.response, 'Aucun run actif.');
|
|
419
|
+
assert.equal(session.headlessPlan ?? null, null);
|
|
420
|
+
} finally {
|
|
421
|
+
globalThis.fetch = originalFetch;
|
|
422
|
+
}
|
|
423
|
+
});
|
|
424
|
+
|
|
425
|
+
test('one Donna approval grants the complete validated run revision', async () => {
|
|
426
|
+
const originalFetch = globalThis.fetch;
|
|
427
|
+
const requests = [];
|
|
428
|
+
globalThis.fetch = async (url, options = {}) => {
|
|
429
|
+
requests.push({ url: String(url), method: options.method ?? 'GET', body: options.body ? JSON.parse(options.body) : null });
|
|
430
|
+
if ((options.method ?? 'GET') === 'GET') {
|
|
431
|
+
return {
|
|
432
|
+
ok: true,
|
|
433
|
+
status: 200,
|
|
434
|
+
json: async () => ({
|
|
435
|
+
running: true,
|
|
436
|
+
runId: 'run-approval',
|
|
437
|
+
planRevision: 3,
|
|
438
|
+
approvals: [
|
|
439
|
+
{ status: 'pending_approval', approvalClasses: ['workspace'] },
|
|
440
|
+
{ status: 'pending_approval', approvalClasses: ['workspace'] },
|
|
441
|
+
],
|
|
442
|
+
}),
|
|
443
|
+
};
|
|
444
|
+
}
|
|
445
|
+
return { ok: true, status: 202, json: async () => ({ approved: true, runId: 'run-approval' }) };
|
|
446
|
+
};
|
|
447
|
+
let calls = 0;
|
|
448
|
+
const session = sessionBase({
|
|
449
|
+
runtime: { url: 'http://runtime.test' },
|
|
450
|
+
agentProjection: { status: 'running', conversation: [], activities: [] },
|
|
451
|
+
llm: {
|
|
452
|
+
async completeWithTools() {
|
|
453
|
+
calls += 1;
|
|
454
|
+
if (calls === 1) {
|
|
455
|
+
return {
|
|
456
|
+
content: null,
|
|
457
|
+
message: { role: 'assistant', content: null },
|
|
458
|
+
tool_calls: [{
|
|
459
|
+
id: 'approve-run',
|
|
460
|
+
type: 'function',
|
|
461
|
+
function: { name: 'runtime__approve', arguments: '{}' },
|
|
462
|
+
}],
|
|
463
|
+
};
|
|
464
|
+
}
|
|
465
|
+
return { content: 'Plan approuvé.', message: { role: 'assistant', content: 'Plan approuvé.' }, tool_calls: null };
|
|
466
|
+
},
|
|
467
|
+
},
|
|
468
|
+
});
|
|
469
|
+
|
|
470
|
+
try {
|
|
471
|
+
const result = await createAgentGraph().invoke({ input: 'oui', session });
|
|
472
|
+
assert.equal(result.response, 'Plan approuvé.');
|
|
473
|
+
const approval = requests.find((request) => request.url.includes('/approve'));
|
|
474
|
+
assert.deepEqual(approval.body, {
|
|
475
|
+
workspace: 'docs',
|
|
476
|
+
runId: 'run-approval',
|
|
477
|
+
itemId: null,
|
|
478
|
+
approvalId: null,
|
|
479
|
+
scope: 'run',
|
|
480
|
+
planRevision: 3,
|
|
481
|
+
approvalClasses: ['workspace'],
|
|
482
|
+
});
|
|
483
|
+
} finally {
|
|
484
|
+
globalThis.fetch = originalFetch;
|
|
485
|
+
}
|
|
486
|
+
});
|
|
487
|
+
|
|
186
488
|
test('agent graph binds the full toolset for a "remember my preference" request, not just read-only tools', async () => {
|
|
187
489
|
const seenTools = [];
|
|
188
490
|
const session = sessionBase({
|
|
@@ -388,6 +690,156 @@ test('buildAgentSystemPrompt forbids inventing slash commands or arguments', ()
|
|
|
388
690
|
assert.doesNotMatch(prompt, /executor:"/);
|
|
389
691
|
});
|
|
390
692
|
|
|
693
|
+
test('workspace package manifest is not exposed as an executable skill', () => {
|
|
694
|
+
const root = mkdtempSync(join(tmpdir(), 'wiki-manager-skills-'));
|
|
695
|
+
try {
|
|
696
|
+
mkdirSync(join(root, '.wiki', 'skills'), { recursive: true });
|
|
697
|
+
writeFileSync(join(root, 'skill.yaml'), [
|
|
698
|
+
'name: basic',
|
|
699
|
+
'description: Demo workspace package',
|
|
700
|
+
'entrypoints:',
|
|
701
|
+
' uiSkillDir: .wiki/skills',
|
|
702
|
+
].join('\n'));
|
|
703
|
+
writeFileSync(join(root, '.wiki', 'skills', 'ingest.md'), [
|
|
704
|
+
'---',
|
|
705
|
+
'name: ingest',
|
|
706
|
+
'description: Ingest pending sources',
|
|
707
|
+
'---',
|
|
708
|
+
'Use the production capability.',
|
|
709
|
+
].join('\n'));
|
|
710
|
+
|
|
711
|
+
const prompt = buildAgentSystemPrompt({ session: sessionBase({ workspacePath: root }) });
|
|
712
|
+
assert.doesNotMatch(prompt, /\/basic:/);
|
|
713
|
+
assert.match(prompt, /\/ingest: Ingest pending sources/);
|
|
714
|
+
} finally {
|
|
715
|
+
rmSync(root, { recursive: true, force: true });
|
|
716
|
+
}
|
|
717
|
+
});
|
|
718
|
+
|
|
719
|
+
test('system prompt forbids unsolicited next-step sections', () => {
|
|
720
|
+
const prompt = buildAgentSystemPrompt({ session: sessionBase() });
|
|
721
|
+
assert.match(prompt, /Never add a "Next steps", "Prochaines étapes", "À suivre"/);
|
|
722
|
+
assert.match(prompt, /unless the user explicitly asks what to do next/);
|
|
723
|
+
assert.doesNotMatch(prompt, /list the suggested follow-ups/);
|
|
724
|
+
});
|
|
725
|
+
|
|
726
|
+
test('system prompt requests synthetic responses capped at about twenty lines without internal narration', () => {
|
|
727
|
+
const prompt = buildAgentSystemPrompt({ session: sessionBase() });
|
|
728
|
+
assert.match(prompt, /never exceed roughly 15 to 20 short lines/);
|
|
729
|
+
assert.match(prompt, /Prioritize the result, essential facts, concrete errors, and actual outputs/);
|
|
730
|
+
assert.match(prompt, /Never expose internal reasoning, repeated checks, tool-selection commentary, or a chronological diary/);
|
|
731
|
+
});
|
|
732
|
+
|
|
733
|
+
test('slash-command output guard rejects commands outside the real agent command set', () => {
|
|
734
|
+
const session = sessionBase({ commands: ['status', 'wiki', 'openui'] });
|
|
735
|
+
assert.deepEqual(
|
|
736
|
+
invalidSuggestedSlashCommands('Vérifiez avec :\n```bash\n/wiki list_pages\n```', session),
|
|
737
|
+
['wiki'],
|
|
738
|
+
);
|
|
739
|
+
assert.deepEqual(invalidSuggestedSlashCommands('Le résultat est disponible dans `/openui`.', session), []);
|
|
740
|
+
});
|
|
741
|
+
|
|
742
|
+
test('Donna retries instead of displaying an invented slash command', async () => {
|
|
743
|
+
let calls = 0;
|
|
744
|
+
const session = sessionBase({
|
|
745
|
+
commands: ['status', 'openui'],
|
|
746
|
+
llm: {
|
|
747
|
+
async completeWithTools() {
|
|
748
|
+
calls += 1;
|
|
749
|
+
if (calls === 1) {
|
|
750
|
+
const content = 'Vérifiez ensuite avec :\n```bash\n/wiki list_pages\n```';
|
|
751
|
+
return { content, message: { role: 'assistant', content }, tool_calls: null };
|
|
752
|
+
}
|
|
753
|
+
const content = 'Ingestion terminée. Le résultat est disponible dans `/openui`.';
|
|
754
|
+
return { content, message: { role: 'assistant', content }, tool_calls: null };
|
|
755
|
+
},
|
|
756
|
+
},
|
|
757
|
+
});
|
|
758
|
+
|
|
759
|
+
const result = await createAgentGraph().invoke({ input: 'résume le résultat', session });
|
|
760
|
+
assert.equal(calls, 2);
|
|
761
|
+
assert.equal(result.response, 'Ingestion terminée. Le résultat est disponible dans `/openui`.');
|
|
762
|
+
assert.doesNotMatch(result.response, /wiki list_pages/);
|
|
763
|
+
});
|
|
764
|
+
|
|
765
|
+
test('Donna retries a malformed tool call without reinjecting its broken JSON', async () => {
|
|
766
|
+
let calls = 0;
|
|
767
|
+
const session = sessionBase({
|
|
768
|
+
runtime: { url: 'http://runtime.test' },
|
|
769
|
+
llm: {
|
|
770
|
+
async completeWithTools({ messages }) {
|
|
771
|
+
calls += 1;
|
|
772
|
+
if (calls === 1) {
|
|
773
|
+
return {
|
|
774
|
+
content: null,
|
|
775
|
+
message: { role: 'assistant', content: null },
|
|
776
|
+
tool_calls: [{
|
|
777
|
+
id: 'broken',
|
|
778
|
+
type: 'function',
|
|
779
|
+
function: { name: 'runtime__delegate', arguments: '{"objective":"Ingère' },
|
|
780
|
+
}],
|
|
781
|
+
};
|
|
782
|
+
}
|
|
783
|
+
assert.doesNotMatch(JSON.stringify(messages), /arguments.*Ingère/);
|
|
784
|
+
return { content: 'Appel reformulé.', message: { role: 'assistant', content: 'Appel reformulé.' }, tool_calls: null };
|
|
785
|
+
},
|
|
786
|
+
},
|
|
787
|
+
});
|
|
788
|
+
|
|
789
|
+
await createAgentGraph().invoke({ input: 'ingère les fichiers', session });
|
|
790
|
+
assert.equal(calls, 2);
|
|
791
|
+
});
|
|
792
|
+
|
|
793
|
+
test('forced delegation is cleared after one valid tool call and does not loop', async () => {
|
|
794
|
+
const originalFetch = globalThis.fetch;
|
|
795
|
+
globalThis.fetch = async () => ({
|
|
796
|
+
ok: true,
|
|
797
|
+
status: 202,
|
|
798
|
+
json: async () => ({ accepted: true, runId: 'run-once' }),
|
|
799
|
+
});
|
|
800
|
+
const choices = [];
|
|
801
|
+
let calls = 0;
|
|
802
|
+
const session = sessionBase({
|
|
803
|
+
runtime: { url: 'http://runtime.test' },
|
|
804
|
+
commands: ['status'],
|
|
805
|
+
llm: {
|
|
806
|
+
async completeWithTools({ toolChoice, tools }) {
|
|
807
|
+
if (tools.length === 0) {
|
|
808
|
+
return { content: '{"action":true}', message: { role: 'assistant', content: '{"action":true}' }, tool_calls: null };
|
|
809
|
+
}
|
|
810
|
+
calls += 1;
|
|
811
|
+
choices.push(toolChoice);
|
|
812
|
+
if (calls === 1) {
|
|
813
|
+
const content = 'Utilise cette commande :\n/pipeline';
|
|
814
|
+
return { content, message: { role: 'assistant', content }, tool_calls: null };
|
|
815
|
+
}
|
|
816
|
+
if (calls === 2) {
|
|
817
|
+
return {
|
|
818
|
+
content: null,
|
|
819
|
+
message: { role: 'assistant', content: null },
|
|
820
|
+
tool_calls: [{
|
|
821
|
+
id: 'delegate-once',
|
|
822
|
+
type: 'function',
|
|
823
|
+
function: { name: 'runtime__delegate', arguments: '{"objective":"Ingérer les fichiers"}' },
|
|
824
|
+
}],
|
|
825
|
+
};
|
|
826
|
+
}
|
|
827
|
+
return { content: 'Ingestion déléguée.', message: { role: 'assistant', content: 'Ingestion déléguée.' }, tool_calls: null };
|
|
828
|
+
},
|
|
829
|
+
},
|
|
830
|
+
});
|
|
831
|
+
|
|
832
|
+
try {
|
|
833
|
+
const result = await createAgentGraph().invoke({ input: 'lance ingestion', session });
|
|
834
|
+
assert.equal(calls, 3);
|
|
835
|
+
assert.deepEqual(choices[1], { type: 'function', function: { name: 'runtime__delegate' } });
|
|
836
|
+
assert.equal(choices[2], 'auto');
|
|
837
|
+
assert.equal(result.response, 'Ingestion déléguée.');
|
|
838
|
+
} finally {
|
|
839
|
+
globalThis.fetch = originalFetch;
|
|
840
|
+
}
|
|
841
|
+
});
|
|
842
|
+
|
|
391
843
|
// Guard: the system prompt must never show a connected tool's bare name
|
|
392
844
|
// outside its qualified server__tool form. Bare mentions are what teach the
|
|
393
845
|
// model to emit unqualified tool calls (the cme_status incident). The bare
|
|
@@ -435,33 +887,6 @@ test('buildAgentSystemPrompt contains no unqualified tool names for connected se
|
|
|
435
887
|
assert.deepEqual(offenders, [], `Unqualified tool names found in system prompt: ${offenders.join(', ')}`);
|
|
436
888
|
});
|
|
437
889
|
|
|
438
|
-
test('classifyAgentInput routes information requests to observe, not runs', () => {
|
|
439
|
-
const session = sessionBase();
|
|
440
|
-
// The original incident: a config question must never become a run.
|
|
441
|
-
assert.equal(classifyAgentInput('donne moi la config du cme', session).kind, 'observe');
|
|
442
|
-
assert.equal(classifyAgentInput('montre la configuration du workspace', session).kind, 'observe');
|
|
443
|
-
assert.equal(classifyAgentInput('où en est le run', session).kind, 'observe');
|
|
444
|
-
assert.equal(classifyAgentInput('explique le build', session).kind, 'observe');
|
|
445
|
-
assert.equal(classifyAgentInput("qu'est-ce que le pipeline polish ?", session).kind, 'observe');
|
|
446
|
-
});
|
|
447
|
-
|
|
448
|
-
test('classifyAgentInput routes action requests to start_run without an active run', () => {
|
|
449
|
-
const session = sessionBase();
|
|
450
|
-
assert.equal(classifyAgentInput('lance le pipeline complet', session).kind, 'start_run');
|
|
451
|
-
assert.equal(classifyAgentInput('configure le cme avec ce token', session).kind, 'start_run');
|
|
452
|
-
assert.equal(classifyAgentInput('lance le run', session).kind, 'start_run');
|
|
453
|
-
assert.equal(classifyAgentInput('exporte les deliverables', session).kind, 'start_run');
|
|
454
|
-
});
|
|
455
|
-
|
|
456
|
-
test('classifyAgentInput keeps small talk as converse and active-run branches intact', () => {
|
|
457
|
-
const session = sessionBase();
|
|
458
|
-
assert.equal(classifyAgentInput('bonjour', session).kind, 'converse');
|
|
459
|
-
assert.equal(classifyAgentInput('merci beaucoup', session).kind, 'converse');
|
|
460
|
-
const activeSession = sessionBase({ agentProjection: { status: 'running' } });
|
|
461
|
-
assert.equal(classifyAgentInput('lance un build', activeSession).kind, 'ambiguous');
|
|
462
|
-
assert.equal(classifyAgentInput('annule tout', activeSession).kind, 'cancel');
|
|
463
|
-
});
|
|
464
|
-
|
|
465
890
|
function orchestrableAgentSnapshot() {
|
|
466
891
|
return [{
|
|
467
892
|
agentInstanceId: 'production-1',
|
|
@@ -572,13 +997,11 @@ test('agent graph accepts plan steps with known capabilities and null capability
|
|
|
572
997
|
assert.deepEqual(session.headlessPlan.map((step) => step.id), ['export', 'report']);
|
|
573
998
|
});
|
|
574
999
|
|
|
575
|
-
test('buildAgentSystemPrompt
|
|
1000
|
+
test('buildAgentSystemPrompt assigns capability resolution exclusively to the runtime', () => {
|
|
576
1001
|
const withAgents = buildAgentSystemPrompt({ session: sessionBase({ agentRegistrySnapshot: orchestrableAgentSnapshot() }) });
|
|
577
|
-
assert.match(withAgents, /
|
|
578
|
-
assert.match(withAgents, /Never
|
|
579
|
-
|
|
580
|
-
const withoutAgents = buildAgentSystemPrompt({ session: sessionBase() });
|
|
581
|
-
assert.match(withoutAgents, /No orchestration capabilities discovered yet/);
|
|
1002
|
+
assert.match(withAgents, /call runtime__delegate with the user objective only/);
|
|
1003
|
+
assert.match(withAgents, /Never choose a capability, operation, agent, plan, or implementation yourself/);
|
|
1004
|
+
assert.doesNotMatch(withAgents, /ONLY values allowed in requiredCapability/);
|
|
582
1005
|
});
|
|
583
1006
|
|
|
584
1007
|
test('agent graph executes action inputs inside a runtime run instead of asking for clarification', async () => {
|
|
@@ -845,3 +1268,122 @@ test('Donna interprets a cleanup request and calls runtime__kill herself', async
|
|
|
845
1268
|
globalThis.fetch = originalFetch;
|
|
846
1269
|
}
|
|
847
1270
|
});
|
|
1271
|
+
|
|
1272
|
+
test('Donna refuses a direct mutating provider tool in interactive mode and is steered to delegation', async () => {
|
|
1273
|
+
const fetchedHosts = [];
|
|
1274
|
+
const originalFetch = globalThis.fetch;
|
|
1275
|
+
globalThis.fetch = async (url) => {
|
|
1276
|
+
fetchedHosts.push(new URL(String(url)).host);
|
|
1277
|
+
return { ok: true, status: 200, json: async () => ({}), text: async () => '{}', headers: { get: () => null } };
|
|
1278
|
+
};
|
|
1279
|
+
let calls = 0;
|
|
1280
|
+
const session = sessionBase({
|
|
1281
|
+
runtime: { url: 'http://runtime.test' },
|
|
1282
|
+
llm: {
|
|
1283
|
+
async completeWithTools({ tools, messages }) {
|
|
1284
|
+
calls += 1;
|
|
1285
|
+
if (calls === 1) {
|
|
1286
|
+
const names = tools.map((tool) => tool.function.name);
|
|
1287
|
+
assert.ok(!names.includes('production__production_start_job'), 'a mutating provider tool must not be offered in interactive mode');
|
|
1288
|
+
assert.ok(names.includes('runtime__delegate'), 'delegate must be offered in interactive mode');
|
|
1289
|
+
return {
|
|
1290
|
+
content: null,
|
|
1291
|
+
message: { role: 'assistant', content: null },
|
|
1292
|
+
tool_calls: [{ id: 'direct-call', type: 'function', function: { name: 'production__production_start_job', arguments: '{"type":"ingest"}' } }],
|
|
1293
|
+
};
|
|
1294
|
+
}
|
|
1295
|
+
const lastTool = (messages ?? []).filter((message) => message.role === 'tool').at(-1);
|
|
1296
|
+
assert.match(String(lastTool?.content ?? ''), /not available in interactive mode/);
|
|
1297
|
+
return { content: 'Objectif transmis au runtime.', message: { role: 'assistant', content: 'Objectif transmis au runtime.' }, tool_calls: null };
|
|
1298
|
+
},
|
|
1299
|
+
},
|
|
1300
|
+
});
|
|
1301
|
+
|
|
1302
|
+
try {
|
|
1303
|
+
const agent = createAgentGraph();
|
|
1304
|
+
const result = await agent.invoke({ input: 'lance une ingestion', session });
|
|
1305
|
+
assert.equal(calls, 2, 'the model must get a second turn after the refusal');
|
|
1306
|
+
assert.equal(result.response, 'Objectif transmis au runtime.');
|
|
1307
|
+
assert.ok(!fetchedHosts.includes('127.0.0.1:3000'), 'the refused provider tool must never be executed');
|
|
1308
|
+
} finally {
|
|
1309
|
+
globalThis.fetch = originalFetch;
|
|
1310
|
+
}
|
|
1311
|
+
});
|
|
1312
|
+
|
|
1313
|
+
test('buildAgentSystemPrompt uses the canonical wiki workspace status without filesystem fallback', () => {
|
|
1314
|
+
const workspacePath = mkdtempSync(join(tmpdir(), 'facts-'));
|
|
1315
|
+
try {
|
|
1316
|
+
mkdirSync(join(workspacePath, 'raw', 'untracked'), { recursive: true });
|
|
1317
|
+
writeFileSync(join(workspacePath, 'raw', 'untracked', 'note-a.md'), '# a\n');
|
|
1318
|
+
const prompt = buildAgentSystemPrompt({ session: sessionBase({ workspacePath }) });
|
|
1319
|
+
assert.match(prompt, /call wiki__wiki_workspace_status first/);
|
|
1320
|
+
assert.match(prompt, /canonical read-only workspace state/);
|
|
1321
|
+
assert.doesNotMatch(prompt, /note-a\.md/);
|
|
1322
|
+
assert.doesNotMatch(prompt, /Workspace facts:/);
|
|
1323
|
+
} finally {
|
|
1324
|
+
rmSync(workspacePath, { recursive: true, force: true });
|
|
1325
|
+
}
|
|
1326
|
+
});
|
|
1327
|
+
|
|
1328
|
+
test('Donna reads workspace inventory from the canonical wiki status tool', async () => {
|
|
1329
|
+
const originalFetch = globalThis.fetch;
|
|
1330
|
+
let requestedArguments = null;
|
|
1331
|
+
globalThis.fetch = async (_url, options) => {
|
|
1332
|
+
const request = JSON.parse(options.body);
|
|
1333
|
+
requestedArguments = request.params.arguments;
|
|
1334
|
+
return {
|
|
1335
|
+
ok: true,
|
|
1336
|
+
status: 200,
|
|
1337
|
+
headers: { get: () => null },
|
|
1338
|
+
text: async () => JSON.stringify({ result: { content: [{
|
|
1339
|
+
type: 'text',
|
|
1340
|
+
text: JSON.stringify({
|
|
1341
|
+
pendingSources: { count: 2, files: ['raw/untracked/a.md', 'raw/untracked/b.md'] },
|
|
1342
|
+
templates: { count: 1, files: ['templates/report.md'] },
|
|
1343
|
+
deliverables: { count: 0, files: [] },
|
|
1344
|
+
}),
|
|
1345
|
+
}] } }),
|
|
1346
|
+
};
|
|
1347
|
+
};
|
|
1348
|
+
let calls = 0;
|
|
1349
|
+
const session = sessionBase({
|
|
1350
|
+
mcp: {
|
|
1351
|
+
wiki: {
|
|
1352
|
+
status: 'connected',
|
|
1353
|
+
url: 'http://127.0.0.1:3000/mcp/',
|
|
1354
|
+
tools: [{
|
|
1355
|
+
name: 'wiki_workspace_status',
|
|
1356
|
+
description: 'Read the canonical local workspace inventory.',
|
|
1357
|
+
inputSchema: { type: 'object', properties: {} },
|
|
1358
|
+
}],
|
|
1359
|
+
},
|
|
1360
|
+
},
|
|
1361
|
+
llm: {
|
|
1362
|
+
async completeWithTools({ messages }) {
|
|
1363
|
+
calls += 1;
|
|
1364
|
+
if (calls === 1) {
|
|
1365
|
+
return {
|
|
1366
|
+
content: null,
|
|
1367
|
+
message: { role: 'assistant', content: null },
|
|
1368
|
+
tool_calls: [{
|
|
1369
|
+
id: 'status-call',
|
|
1370
|
+
type: 'function',
|
|
1371
|
+
function: { name: 'wiki__wiki_workspace_status', arguments: '{}' },
|
|
1372
|
+
}],
|
|
1373
|
+
};
|
|
1374
|
+
}
|
|
1375
|
+
assert.match(JSON.stringify(messages), /raw\/untracked\/a\.md/);
|
|
1376
|
+
return { content: 'Deux fichiers sont en attente : a.md et b.md.', message: { role: 'assistant', content: 'Deux fichiers sont en attente : a.md et b.md.' }, tool_calls: null };
|
|
1377
|
+
},
|
|
1378
|
+
},
|
|
1379
|
+
});
|
|
1380
|
+
|
|
1381
|
+
try {
|
|
1382
|
+
const result = await createAgentGraph().invoke({ input: 'as ton des fichier en attente d ingestion', session });
|
|
1383
|
+
assert.equal(result.response, 'Deux fichiers sont en attente : a.md et b.md.');
|
|
1384
|
+
assert.deepEqual(requestedArguments, {});
|
|
1385
|
+
assert.equal(session.headlessPlan, null, 'a read-only workspace inventory must not create a plan');
|
|
1386
|
+
} finally {
|
|
1387
|
+
globalThis.fetch = originalFetch;
|
|
1388
|
+
}
|
|
1389
|
+
});
|
package/src/agent/llm.js
CHANGED
|
@@ -55,7 +55,7 @@ export function createLlmClientFromWikiConfig(config) {
|
|
|
55
55
|
}
|
|
56
56
|
return content;
|
|
57
57
|
},
|
|
58
|
-
async completeWithTools({ system, tools = [], messages = [], signal }) {
|
|
58
|
+
async completeWithTools({ system, tools = [], messages = [], toolChoice = 'auto', signal }) {
|
|
59
59
|
const allMessages = [
|
|
60
60
|
{ role: 'system', content: system },
|
|
61
61
|
...messages,
|
|
@@ -67,7 +67,7 @@ export function createLlmClientFromWikiConfig(config) {
|
|
|
67
67
|
};
|
|
68
68
|
if (tools.length > 0) {
|
|
69
69
|
body.tools = tools;
|
|
70
|
-
body.tool_choice =
|
|
70
|
+
body.tool_choice = toolChoice;
|
|
71
71
|
}
|
|
72
72
|
const response = await fetch(`${baseUrl}/chat/completions`, {
|
|
73
73
|
method: 'POST',
|
|
@@ -90,7 +90,7 @@ export function createLlmClientFromWikiConfig(config) {
|
|
|
90
90
|
message: { role: 'assistant', content: msg?.content ?? null, tool_calls: msg?.tool_calls },
|
|
91
91
|
};
|
|
92
92
|
},
|
|
93
|
-
async streamWithTools({ system, tools = [], messages = [], onTextDelta, signal }) {
|
|
93
|
+
async streamWithTools({ system, tools = [], messages = [], toolChoice = 'auto', onTextDelta, signal }) {
|
|
94
94
|
const allMessages = [
|
|
95
95
|
{ role: 'system', content: system },
|
|
96
96
|
...messages,
|
|
@@ -103,7 +103,7 @@ export function createLlmClientFromWikiConfig(config) {
|
|
|
103
103
|
};
|
|
104
104
|
if (tools.length > 0) {
|
|
105
105
|
body.tools = tools;
|
|
106
|
-
body.tool_choice =
|
|
106
|
+
body.tool_choice = toolChoice;
|
|
107
107
|
}
|
|
108
108
|
const response = await fetch(`${baseUrl}/chat/completions`, {
|
|
109
109
|
method: 'POST',
|
|
@@ -119,7 +119,7 @@ export function createLlmClientFromWikiConfig(config) {
|
|
|
119
119
|
throw new Error(`HTTP ${response.status} ${text.slice(0, 240)}`);
|
|
120
120
|
}
|
|
121
121
|
if (!response.body) {
|
|
122
|
-
const result = await this.completeWithTools({ system, tools, messages, signal });
|
|
122
|
+
const result = await this.completeWithTools({ system, tools, messages, toolChoice, signal });
|
|
123
123
|
if (result.content) onTextDelta?.(result.content);
|
|
124
124
|
return result;
|
|
125
125
|
}
|