@dotdrelle/wiki-manager 0.12.12 → 0.14.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docker-compose.yml +1 -1
- package/mcp.endpoints.example.json +7 -0
- package/package.json +2 -2
- package/src/agent/graph.js +354 -143
- package/src/agent/graph.test.js +516 -54
- package/src/agent/llm.js +5 -5
- package/src/cli/wiki-manager.js +234 -6
- package/src/cli/wiki-manager.test.js +28 -0
- package/src/commands/slash.js +32 -11
- package/src/commands/slash.test.js +9 -1
- package/src/core/agentEvents.js +7 -1
- package/src/core/agentEvents.test.js +13 -1
- package/src/core/buildInfo.json +2 -2
- package/src/core/mcp.js +46 -4
- package/src/core/skills.js +0 -28
- package/src/core/toolLoop.js +56 -0
- package/src/core/toolLoop.test.js +88 -0
- package/src/orchestrator/capabilityRegistry.js +14 -0
- package/src/orchestrator/capabilityRegistry.test.js +12 -1
- package/src/orchestrator/dependencyResolver.js +10 -1
- package/src/orchestrator/dispatcher.js +34 -3
- package/src/orchestrator/dispatcher.test.js +34 -0
- package/src/orchestrator/objectiveResolver.js +79 -0
- package/src/orchestrator/objectiveResolver.test.js +50 -0
- package/src/orchestrator/scheduler.test.js +25 -0
- package/src/runtime/client.js +32 -1
- package/src/runtime/lifecycle.js +32 -2
- package/src/runtime/recoveryManager.js +14 -7
- package/src/runtime/runner.js +112 -13
- package/src/runtime/runner.test.js +64 -1
- package/src/runtime/server.js +47 -3
- package/src/runtime/supervisor.js +4 -1
- package/src/runtime/supervisor.test.js +49 -0
- package/src/shell/repl.js +134 -55
- package/src/shell/repl.test.js +151 -12
- package/src/shell/useSession.ts +15 -3
package/src/agent/graph.test.js
CHANGED
|
@@ -3,7 +3,57 @@ import test from 'node:test';
|
|
|
3
3
|
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs';
|
|
4
4
|
import { tmpdir } from 'node:os';
|
|
5
5
|
import { join } from 'node:path';
|
|
6
|
-
import { buildAgentSystemPrompt,
|
|
6
|
+
import { buildAgentSystemPrompt, createAgentGraph, invalidSuggestedSlashCommands, invalidUserFacingToolNames, knownCapabilityIds, normalizeToolArgumentsFromSchema } from './graph.js';
|
|
7
|
+
|
|
8
|
+
test('user-facing response guard hides MCP identifiers generically', () => {
|
|
9
|
+
const session = sessionBase();
|
|
10
|
+
assert.deepEqual(
|
|
11
|
+
invalidUserFacingToolNames('Utilisez production__production_start_job.', session),
|
|
12
|
+
['production__production_start_job'],
|
|
13
|
+
);
|
|
14
|
+
});
|
|
15
|
+
|
|
16
|
+
test('Donna cannot answer an explicit action with manual instructions instead of delegating', async () => {
|
|
17
|
+
const originalFetch = globalThis.fetch;
|
|
18
|
+
let delegated = false;
|
|
19
|
+
globalThis.fetch = async (url) => {
|
|
20
|
+
delegated = String(url).includes('/delegate');
|
|
21
|
+
return { ok: true, status: 202, json: async () => ({ accepted: true, runId: 'run-action', delegation: { tasks: 2, agent: 'production' } }) };
|
|
22
|
+
};
|
|
23
|
+
let mainCalls = 0;
|
|
24
|
+
const session = sessionBase({
|
|
25
|
+
runtime: { url: 'http://runtime.test' },
|
|
26
|
+
llm: {
|
|
27
|
+
async completeWithTools({ tools }) {
|
|
28
|
+
if (tools.length === 0) return { content: '{"action":true}', message: { role: 'assistant', content: '{"action":true}' }, tool_calls: null };
|
|
29
|
+
mainCalls += 1;
|
|
30
|
+
if (mainCalls === 1) {
|
|
31
|
+
return {
|
|
32
|
+
content: 'Déplacez raw/untracked/demo.md vers raw/ puis utilisez wiki__wiki_workspace_status.',
|
|
33
|
+
message: { role: 'assistant', content: 'instructions manuelles' },
|
|
34
|
+
tool_calls: null,
|
|
35
|
+
};
|
|
36
|
+
}
|
|
37
|
+
if (mainCalls === 2) {
|
|
38
|
+
return {
|
|
39
|
+
content: null,
|
|
40
|
+
message: { role: 'assistant', content: null },
|
|
41
|
+
tool_calls: [{ id: 'delegate', type: 'function', function: { name: 'runtime__delegate', arguments: '{"objective":"Lance ingestion"}' } }],
|
|
42
|
+
};
|
|
43
|
+
}
|
|
44
|
+
return { content: 'Plan soumis.', message: { role: 'assistant', content: 'Plan soumis.' }, tool_calls: null };
|
|
45
|
+
},
|
|
46
|
+
},
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
try {
|
|
50
|
+
const result = await createAgentGraph().invoke({ input: 'lance ingestion', session });
|
|
51
|
+
assert.equal(delegated, true);
|
|
52
|
+
assert.equal(result.response, 'Plan soumis.');
|
|
53
|
+
} finally {
|
|
54
|
+
globalThis.fetch = originalFetch;
|
|
55
|
+
}
|
|
56
|
+
});
|
|
7
57
|
|
|
8
58
|
function sessionBase(overrides = {}) {
|
|
9
59
|
return {
|
|
@@ -183,6 +233,258 @@ test('agent graph does not pre-filter mutating MCP tools for config questions',
|
|
|
183
233
|
assert.ok(seenTools.includes('shell__run_command'));
|
|
184
234
|
});
|
|
185
235
|
|
|
236
|
+
test('interactive Donna delegates provider execution to the runtime orchestrator', async () => {
|
|
237
|
+
const seenTools = [];
|
|
238
|
+
const session = sessionBase({
|
|
239
|
+
runtime: { url: 'http://runtime.test' },
|
|
240
|
+
agentRegistrySnapshot: ingestAgentSnapshot(),
|
|
241
|
+
mcp: {
|
|
242
|
+
production: {
|
|
243
|
+
status: 'connected',
|
|
244
|
+
url: 'http://127.0.0.1:3000/mcp/',
|
|
245
|
+
tools: [
|
|
246
|
+
{ name: 'agent_plan', description: 'Plan tasks', inputSchema: { type: 'object', properties: {} } },
|
|
247
|
+
{ name: 'agent_execute', description: 'Execute a task', inputSchema: { type: 'object', properties: {} } },
|
|
248
|
+
{ name: 'agent_status', description: 'Read task status', inputSchema: { type: 'object', properties: {} } },
|
|
249
|
+
{ name: 'production_start_job', description: 'Legacy job start', inputSchema: { type: 'object', properties: {} } },
|
|
250
|
+
{ name: 'production_status', description: 'Read production status', inputSchema: { type: 'object', properties: {} } },
|
|
251
|
+
],
|
|
252
|
+
},
|
|
253
|
+
},
|
|
254
|
+
llm: {
|
|
255
|
+
async completeWithTools({ tools }) {
|
|
256
|
+
seenTools.push(...tools.map((tool) => tool.function.name));
|
|
257
|
+
return { content: 'Prêt.', message: { role: 'assistant', content: 'Prêt.' }, tool_calls: null };
|
|
258
|
+
},
|
|
259
|
+
},
|
|
260
|
+
});
|
|
261
|
+
|
|
262
|
+
await createAgentGraph().invoke({ input: 'bonjour', session });
|
|
263
|
+
|
|
264
|
+
assert.ok(seenTools.includes('runtime__delegate'));
|
|
265
|
+
assert.ok(seenTools.includes('production__agent_status'));
|
|
266
|
+
assert.ok(seenTools.includes('production__production_status'));
|
|
267
|
+
assert.ok(!seenTools.includes('production__agent_plan'));
|
|
268
|
+
assert.ok(!seenTools.includes('production__agent_execute'));
|
|
269
|
+
assert.ok(!seenTools.includes('production__production_start_job'));
|
|
270
|
+
assert.ok(!seenTools.includes('wiki__plan_set'));
|
|
271
|
+
assert.ok(!seenTools.includes('wiki__plan_done'));
|
|
272
|
+
});
|
|
273
|
+
|
|
274
|
+
test('interactive Donna can delegate while the shell capability snapshot is still empty', async () => {
|
|
275
|
+
const seenTools = [];
|
|
276
|
+
const session = sessionBase({
|
|
277
|
+
runtime: { url: 'http://runtime.test' },
|
|
278
|
+
agentRegistrySnapshot: [],
|
|
279
|
+
llm: {
|
|
280
|
+
async completeWithTools({ tools }) {
|
|
281
|
+
seenTools.push(...tools.map((tool) => tool.function.name));
|
|
282
|
+
return { content: 'Prêt.', message: { role: 'assistant', content: 'Prêt.' }, tool_calls: null };
|
|
283
|
+
},
|
|
284
|
+
},
|
|
285
|
+
});
|
|
286
|
+
|
|
287
|
+
await createAgentGraph().invoke({ input: 'bonjour', session });
|
|
288
|
+
|
|
289
|
+
assert.ok(seenTools.includes('runtime__delegate'));
|
|
290
|
+
});
|
|
291
|
+
|
|
292
|
+
function ingestAgentSnapshot() {
|
|
293
|
+
return [{
|
|
294
|
+
agentInstanceId: 'production-ingest',
|
|
295
|
+
serverName: 'production',
|
|
296
|
+
health: 'available',
|
|
297
|
+
description: {
|
|
298
|
+
contractVersion: '1',
|
|
299
|
+
agentType: 'production',
|
|
300
|
+
displayName: 'Production',
|
|
301
|
+
capabilities: [{
|
|
302
|
+
id: 'knowledge.update',
|
|
303
|
+
version: '1',
|
|
304
|
+
supportedOperations: ['ingest', 'ingest_plan', 'ingest_apply'],
|
|
305
|
+
}],
|
|
306
|
+
},
|
|
307
|
+
}];
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
test('runtime delegation tool declares only its canonical natural-language objective', async () => {
|
|
311
|
+
let delegationTool = null;
|
|
312
|
+
const session = sessionBase({
|
|
313
|
+
runtime: { url: 'http://runtime.test' },
|
|
314
|
+
agentRegistrySnapshot: ingestAgentSnapshot(),
|
|
315
|
+
llm: {
|
|
316
|
+
async completeWithTools({ tools }) {
|
|
317
|
+
delegationTool ??= tools.find((tool) => tool.function.name === 'runtime__delegate');
|
|
318
|
+
return { content: 'Prêt.', message: { role: 'assistant', content: 'Prêt.' }, tool_calls: null };
|
|
319
|
+
},
|
|
320
|
+
},
|
|
321
|
+
});
|
|
322
|
+
|
|
323
|
+
await createAgentGraph().invoke({ input: 'bonjour', session });
|
|
324
|
+
|
|
325
|
+
assert.deepEqual(Object.keys(delegationTool.function.parameters.properties), ['objective']);
|
|
326
|
+
assert.deepEqual(delegationTool.function.parameters.required, ['objective']);
|
|
327
|
+
});
|
|
328
|
+
|
|
329
|
+
test('tool argument normalization repairs only an unambiguous schema-compatible field name', () => {
|
|
330
|
+
const schema = {
|
|
331
|
+
type: 'object',
|
|
332
|
+
properties: { objective: { type: 'string' } },
|
|
333
|
+
required: ['objective'],
|
|
334
|
+
additionalProperties: false,
|
|
335
|
+
};
|
|
336
|
+
assert.deepEqual(
|
|
337
|
+
normalizeToolArgumentsFromSchema({ input: 'Ingérer les fichiers' }, schema),
|
|
338
|
+
{ objective: 'Ingérer les fichiers' },
|
|
339
|
+
);
|
|
340
|
+
assert.deepEqual(
|
|
341
|
+
normalizeToolArgumentsFromSchema({ input: 'x', other: 'y' }, schema),
|
|
342
|
+
{ input: 'x', other: 'y' },
|
|
343
|
+
);
|
|
344
|
+
assert.deepEqual(
|
|
345
|
+
normalizeToolArgumentsFromSchema({ input: 42 }, schema),
|
|
346
|
+
{ input: 42 },
|
|
347
|
+
);
|
|
348
|
+
});
|
|
349
|
+
|
|
350
|
+
test('Donna delegates the objective without choosing technical identifiers', async () => {
|
|
351
|
+
const originalFetch = globalThis.fetch;
|
|
352
|
+
let request = null;
|
|
353
|
+
globalThis.fetch = async (url, options) => {
|
|
354
|
+
request = { url: String(url), body: JSON.parse(options.body) };
|
|
355
|
+
return { ok: true, status: 202, json: async () => ({ accepted: true, runId: 'run-1', delegation: { tasks: 5, agent: 'production' } }) };
|
|
356
|
+
};
|
|
357
|
+
let calls = 0;
|
|
358
|
+
const session = sessionBase({
|
|
359
|
+
runtime: { url: 'http://runtime.test' },
|
|
360
|
+
agentRegistrySnapshot: ingestAgentSnapshot(),
|
|
361
|
+
llm: {
|
|
362
|
+
async completeWithTools({ messages }) {
|
|
363
|
+
calls += 1;
|
|
364
|
+
if (calls === 1) {
|
|
365
|
+
return {
|
|
366
|
+
content: null,
|
|
367
|
+
message: { role: 'assistant', content: null },
|
|
368
|
+
tool_calls: [{
|
|
369
|
+
id: 'delegate',
|
|
370
|
+
type: 'function',
|
|
371
|
+
function: { name: 'runtime__delegate', arguments: '{"input":"Ingérer tous les fichiers en attente"}' },
|
|
372
|
+
}],
|
|
373
|
+
};
|
|
374
|
+
}
|
|
375
|
+
return { content: 'Plan validé.', message: { role: 'assistant', content: 'Plan validé.' }, tool_calls: null };
|
|
376
|
+
},
|
|
377
|
+
},
|
|
378
|
+
});
|
|
379
|
+
|
|
380
|
+
try {
|
|
381
|
+
await createAgentGraph().invoke({ input: 'ingère tout', session });
|
|
382
|
+
assert.match(request.url, /\/delegate/);
|
|
383
|
+
assert.deepEqual(request.body, { objective: 'Ingérer tous les fichiers en attente', workspace: 'docs' });
|
|
384
|
+
assert.equal('capability' in request.body, false);
|
|
385
|
+
assert.equal('operation' in request.body, false);
|
|
386
|
+
} finally {
|
|
387
|
+
globalThis.fetch = originalFetch;
|
|
388
|
+
}
|
|
389
|
+
});
|
|
390
|
+
|
|
391
|
+
test('runtime status does not manufacture a plan', async () => {
|
|
392
|
+
const originalFetch = globalThis.fetch;
|
|
393
|
+
globalThis.fetch = async () => ({
|
|
394
|
+
ok: true,
|
|
395
|
+
status: 200,
|
|
396
|
+
json: async () => ({ status: 'idle', running: false, plan: [], queue: [], controlQueue: [], approvals: [] }),
|
|
397
|
+
});
|
|
398
|
+
let calls = 0;
|
|
399
|
+
const session = sessionBase({
|
|
400
|
+
runtime: { url: 'http://runtime.test' },
|
|
401
|
+
llm: {
|
|
402
|
+
async completeWithTools() {
|
|
403
|
+
calls += 1;
|
|
404
|
+
if (calls === 1) {
|
|
405
|
+
return {
|
|
406
|
+
content: null,
|
|
407
|
+
message: { role: 'assistant', content: null },
|
|
408
|
+
tool_calls: [{ id: 'status', type: 'function', function: { name: 'runtime__status', arguments: '{}' } }],
|
|
409
|
+
};
|
|
410
|
+
}
|
|
411
|
+
return { content: 'Aucun run actif.', message: { role: 'assistant', content: 'Aucun run actif.' }, tool_calls: null };
|
|
412
|
+
},
|
|
413
|
+
},
|
|
414
|
+
});
|
|
415
|
+
|
|
416
|
+
try {
|
|
417
|
+
const result = await createAgentGraph().invoke({ input: 'où en est le travail ?', session });
|
|
418
|
+
assert.equal(result.response, 'Aucun run actif.');
|
|
419
|
+
assert.equal(session.headlessPlan ?? null, null);
|
|
420
|
+
} finally {
|
|
421
|
+
globalThis.fetch = originalFetch;
|
|
422
|
+
}
|
|
423
|
+
});
|
|
424
|
+
|
|
425
|
+
test('one Donna approval grants the complete validated run revision', async () => {
|
|
426
|
+
const originalFetch = globalThis.fetch;
|
|
427
|
+
const requests = [];
|
|
428
|
+
globalThis.fetch = async (url, options = {}) => {
|
|
429
|
+
requests.push({ url: String(url), method: options.method ?? 'GET', body: options.body ? JSON.parse(options.body) : null });
|
|
430
|
+
if ((options.method ?? 'GET') === 'GET') {
|
|
431
|
+
return {
|
|
432
|
+
ok: true,
|
|
433
|
+
status: 200,
|
|
434
|
+
json: async () => ({
|
|
435
|
+
running: true,
|
|
436
|
+
runId: 'run-approval',
|
|
437
|
+
planRevision: 3,
|
|
438
|
+
approvals: [
|
|
439
|
+
{ status: 'pending_approval', approvalClasses: ['workspace'] },
|
|
440
|
+
{ status: 'pending_approval', approvalClasses: ['workspace'] },
|
|
441
|
+
],
|
|
442
|
+
}),
|
|
443
|
+
};
|
|
444
|
+
}
|
|
445
|
+
return { ok: true, status: 202, json: async () => ({ approved: true, runId: 'run-approval' }) };
|
|
446
|
+
};
|
|
447
|
+
let calls = 0;
|
|
448
|
+
const session = sessionBase({
|
|
449
|
+
runtime: { url: 'http://runtime.test' },
|
|
450
|
+
agentProjection: { status: 'running', conversation: [], activities: [] },
|
|
451
|
+
llm: {
|
|
452
|
+
async completeWithTools() {
|
|
453
|
+
calls += 1;
|
|
454
|
+
if (calls === 1) {
|
|
455
|
+
return {
|
|
456
|
+
content: null,
|
|
457
|
+
message: { role: 'assistant', content: null },
|
|
458
|
+
tool_calls: [{
|
|
459
|
+
id: 'approve-run',
|
|
460
|
+
type: 'function',
|
|
461
|
+
function: { name: 'runtime__approve', arguments: '{}' },
|
|
462
|
+
}],
|
|
463
|
+
};
|
|
464
|
+
}
|
|
465
|
+
return { content: 'Plan approuvé.', message: { role: 'assistant', content: 'Plan approuvé.' }, tool_calls: null };
|
|
466
|
+
},
|
|
467
|
+
},
|
|
468
|
+
});
|
|
469
|
+
|
|
470
|
+
try {
|
|
471
|
+
const result = await createAgentGraph().invoke({ input: 'oui', session });
|
|
472
|
+
assert.equal(result.response, 'Plan approuvé.');
|
|
473
|
+
const approval = requests.find((request) => request.url.includes('/approve'));
|
|
474
|
+
assert.deepEqual(approval.body, {
|
|
475
|
+
workspace: 'docs',
|
|
476
|
+
runId: 'run-approval',
|
|
477
|
+
itemId: null,
|
|
478
|
+
approvalId: null,
|
|
479
|
+
scope: 'run',
|
|
480
|
+
planRevision: 3,
|
|
481
|
+
approvalClasses: ['workspace'],
|
|
482
|
+
});
|
|
483
|
+
} finally {
|
|
484
|
+
globalThis.fetch = originalFetch;
|
|
485
|
+
}
|
|
486
|
+
});
|
|
487
|
+
|
|
186
488
|
test('agent graph binds the full toolset for a "remember my preference" request, not just read-only tools', async () => {
|
|
187
489
|
const seenTools = [];
|
|
188
490
|
const session = sessionBase({
|
|
@@ -388,6 +690,156 @@ test('buildAgentSystemPrompt forbids inventing slash commands or arguments', ()
|
|
|
388
690
|
assert.doesNotMatch(prompt, /executor:"/);
|
|
389
691
|
});
|
|
390
692
|
|
|
693
|
+
test('workspace package manifest is not exposed as an executable skill', () => {
|
|
694
|
+
const root = mkdtempSync(join(tmpdir(), 'wiki-manager-skills-'));
|
|
695
|
+
try {
|
|
696
|
+
mkdirSync(join(root, '.wiki', 'skills'), { recursive: true });
|
|
697
|
+
writeFileSync(join(root, 'skill.yaml'), [
|
|
698
|
+
'name: basic',
|
|
699
|
+
'description: Demo workspace package',
|
|
700
|
+
'entrypoints:',
|
|
701
|
+
' uiSkillDir: .wiki/skills',
|
|
702
|
+
].join('\n'));
|
|
703
|
+
writeFileSync(join(root, '.wiki', 'skills', 'ingest.md'), [
|
|
704
|
+
'---',
|
|
705
|
+
'name: ingest',
|
|
706
|
+
'description: Ingest pending sources',
|
|
707
|
+
'---',
|
|
708
|
+
'Use the production capability.',
|
|
709
|
+
].join('\n'));
|
|
710
|
+
|
|
711
|
+
const prompt = buildAgentSystemPrompt({ session: sessionBase({ workspacePath: root }) });
|
|
712
|
+
assert.doesNotMatch(prompt, /\/basic:/);
|
|
713
|
+
assert.match(prompt, /\/ingest: Ingest pending sources/);
|
|
714
|
+
} finally {
|
|
715
|
+
rmSync(root, { recursive: true, force: true });
|
|
716
|
+
}
|
|
717
|
+
});
|
|
718
|
+
|
|
719
|
+
test('system prompt forbids unsolicited next-step sections', () => {
|
|
720
|
+
const prompt = buildAgentSystemPrompt({ session: sessionBase() });
|
|
721
|
+
assert.match(prompt, /Never add a "Next steps", "Prochaines étapes", "À suivre"/);
|
|
722
|
+
assert.match(prompt, /unless the user explicitly asks what to do next/);
|
|
723
|
+
assert.doesNotMatch(prompt, /list the suggested follow-ups/);
|
|
724
|
+
});
|
|
725
|
+
|
|
726
|
+
test('system prompt requests synthetic responses capped at about twenty lines without internal narration', () => {
|
|
727
|
+
const prompt = buildAgentSystemPrompt({ session: sessionBase() });
|
|
728
|
+
assert.match(prompt, /never exceed roughly 15 to 20 short lines/);
|
|
729
|
+
assert.match(prompt, /Prioritize the result, essential facts, concrete errors, and actual outputs/);
|
|
730
|
+
assert.match(prompt, /Never expose internal reasoning, repeated checks, tool-selection commentary, or a chronological diary/);
|
|
731
|
+
});
|
|
732
|
+
|
|
733
|
+
test('slash-command output guard rejects commands outside the real agent command set', () => {
|
|
734
|
+
const session = sessionBase({ commands: ['status', 'wiki', 'openui'] });
|
|
735
|
+
assert.deepEqual(
|
|
736
|
+
invalidSuggestedSlashCommands('Vérifiez avec :\n```bash\n/wiki list_pages\n```', session),
|
|
737
|
+
['wiki'],
|
|
738
|
+
);
|
|
739
|
+
assert.deepEqual(invalidSuggestedSlashCommands('Le résultat est disponible dans `/openui`.', session), []);
|
|
740
|
+
});
|
|
741
|
+
|
|
742
|
+
test('Donna retries instead of displaying an invented slash command', async () => {
|
|
743
|
+
let calls = 0;
|
|
744
|
+
const session = sessionBase({
|
|
745
|
+
commands: ['status', 'openui'],
|
|
746
|
+
llm: {
|
|
747
|
+
async completeWithTools() {
|
|
748
|
+
calls += 1;
|
|
749
|
+
if (calls === 1) {
|
|
750
|
+
const content = 'Vérifiez ensuite avec :\n```bash\n/wiki list_pages\n```';
|
|
751
|
+
return { content, message: { role: 'assistant', content }, tool_calls: null };
|
|
752
|
+
}
|
|
753
|
+
const content = 'Ingestion terminée. Le résultat est disponible dans `/openui`.';
|
|
754
|
+
return { content, message: { role: 'assistant', content }, tool_calls: null };
|
|
755
|
+
},
|
|
756
|
+
},
|
|
757
|
+
});
|
|
758
|
+
|
|
759
|
+
const result = await createAgentGraph().invoke({ input: 'résume le résultat', session });
|
|
760
|
+
assert.equal(calls, 2);
|
|
761
|
+
assert.equal(result.response, 'Ingestion terminée. Le résultat est disponible dans `/openui`.');
|
|
762
|
+
assert.doesNotMatch(result.response, /wiki list_pages/);
|
|
763
|
+
});
|
|
764
|
+
|
|
765
|
+
test('Donna retries a malformed tool call without reinjecting its broken JSON', async () => {
|
|
766
|
+
let calls = 0;
|
|
767
|
+
const session = sessionBase({
|
|
768
|
+
runtime: { url: 'http://runtime.test' },
|
|
769
|
+
llm: {
|
|
770
|
+
async completeWithTools({ messages }) {
|
|
771
|
+
calls += 1;
|
|
772
|
+
if (calls === 1) {
|
|
773
|
+
return {
|
|
774
|
+
content: null,
|
|
775
|
+
message: { role: 'assistant', content: null },
|
|
776
|
+
tool_calls: [{
|
|
777
|
+
id: 'broken',
|
|
778
|
+
type: 'function',
|
|
779
|
+
function: { name: 'runtime__delegate', arguments: '{"objective":"Ingère' },
|
|
780
|
+
}],
|
|
781
|
+
};
|
|
782
|
+
}
|
|
783
|
+
assert.doesNotMatch(JSON.stringify(messages), /arguments.*Ingère/);
|
|
784
|
+
return { content: 'Appel reformulé.', message: { role: 'assistant', content: 'Appel reformulé.' }, tool_calls: null };
|
|
785
|
+
},
|
|
786
|
+
},
|
|
787
|
+
});
|
|
788
|
+
|
|
789
|
+
await createAgentGraph().invoke({ input: 'ingère les fichiers', session });
|
|
790
|
+
assert.equal(calls, 2);
|
|
791
|
+
});
|
|
792
|
+
|
|
793
|
+
test('forced delegation is cleared after one valid tool call and does not loop', async () => {
|
|
794
|
+
const originalFetch = globalThis.fetch;
|
|
795
|
+
globalThis.fetch = async () => ({
|
|
796
|
+
ok: true,
|
|
797
|
+
status: 202,
|
|
798
|
+
json: async () => ({ accepted: true, runId: 'run-once' }),
|
|
799
|
+
});
|
|
800
|
+
const choices = [];
|
|
801
|
+
let calls = 0;
|
|
802
|
+
const session = sessionBase({
|
|
803
|
+
runtime: { url: 'http://runtime.test' },
|
|
804
|
+
commands: ['status'],
|
|
805
|
+
llm: {
|
|
806
|
+
async completeWithTools({ toolChoice, tools }) {
|
|
807
|
+
if (tools.length === 0) {
|
|
808
|
+
return { content: '{"action":true}', message: { role: 'assistant', content: '{"action":true}' }, tool_calls: null };
|
|
809
|
+
}
|
|
810
|
+
calls += 1;
|
|
811
|
+
choices.push(toolChoice);
|
|
812
|
+
if (calls === 1) {
|
|
813
|
+
const content = 'Utilise cette commande :\n/pipeline';
|
|
814
|
+
return { content, message: { role: 'assistant', content }, tool_calls: null };
|
|
815
|
+
}
|
|
816
|
+
if (calls === 2) {
|
|
817
|
+
return {
|
|
818
|
+
content: null,
|
|
819
|
+
message: { role: 'assistant', content: null },
|
|
820
|
+
tool_calls: [{
|
|
821
|
+
id: 'delegate-once',
|
|
822
|
+
type: 'function',
|
|
823
|
+
function: { name: 'runtime__delegate', arguments: '{"objective":"Ingérer les fichiers"}' },
|
|
824
|
+
}],
|
|
825
|
+
};
|
|
826
|
+
}
|
|
827
|
+
return { content: 'Ingestion déléguée.', message: { role: 'assistant', content: 'Ingestion déléguée.' }, tool_calls: null };
|
|
828
|
+
},
|
|
829
|
+
},
|
|
830
|
+
});
|
|
831
|
+
|
|
832
|
+
try {
|
|
833
|
+
const result = await createAgentGraph().invoke({ input: 'lance ingestion', session });
|
|
834
|
+
assert.equal(calls, 3);
|
|
835
|
+
assert.deepEqual(choices[1], { type: 'function', function: { name: 'runtime__delegate' } });
|
|
836
|
+
assert.equal(choices[2], 'auto');
|
|
837
|
+
assert.equal(result.response, 'Ingestion déléguée.');
|
|
838
|
+
} finally {
|
|
839
|
+
globalThis.fetch = originalFetch;
|
|
840
|
+
}
|
|
841
|
+
});
|
|
842
|
+
|
|
391
843
|
// Guard: the system prompt must never show a connected tool's bare name
|
|
392
844
|
// outside its qualified server__tool form. Bare mentions are what teach the
|
|
393
845
|
// model to emit unqualified tool calls (the cme_status incident). The bare
|
|
@@ -435,33 +887,6 @@ test('buildAgentSystemPrompt contains no unqualified tool names for connected se
|
|
|
435
887
|
assert.deepEqual(offenders, [], `Unqualified tool names found in system prompt: ${offenders.join(', ')}`);
|
|
436
888
|
});
|
|
437
889
|
|
|
438
|
-
test('classifyAgentInput routes information requests to observe, not runs', () => {
|
|
439
|
-
const session = sessionBase();
|
|
440
|
-
// The original incident: a config question must never become a run.
|
|
441
|
-
assert.equal(classifyAgentInput('donne moi la config du cme', session).kind, 'observe');
|
|
442
|
-
assert.equal(classifyAgentInput('montre la configuration du workspace', session).kind, 'observe');
|
|
443
|
-
assert.equal(classifyAgentInput('où en est le run', session).kind, 'observe');
|
|
444
|
-
assert.equal(classifyAgentInput('explique le build', session).kind, 'observe');
|
|
445
|
-
assert.equal(classifyAgentInput("qu'est-ce que le pipeline polish ?", session).kind, 'observe');
|
|
446
|
-
});
|
|
447
|
-
|
|
448
|
-
test('classifyAgentInput routes action requests to start_run without an active run', () => {
|
|
449
|
-
const session = sessionBase();
|
|
450
|
-
assert.equal(classifyAgentInput('lance le pipeline complet', session).kind, 'start_run');
|
|
451
|
-
assert.equal(classifyAgentInput('configure le cme avec ce token', session).kind, 'start_run');
|
|
452
|
-
assert.equal(classifyAgentInput('lance le run', session).kind, 'start_run');
|
|
453
|
-
assert.equal(classifyAgentInput('exporte les deliverables', session).kind, 'start_run');
|
|
454
|
-
});
|
|
455
|
-
|
|
456
|
-
test('classifyAgentInput keeps small talk as converse and active-run branches intact', () => {
|
|
457
|
-
const session = sessionBase();
|
|
458
|
-
assert.equal(classifyAgentInput('bonjour', session).kind, 'converse');
|
|
459
|
-
assert.equal(classifyAgentInput('merci beaucoup', session).kind, 'converse');
|
|
460
|
-
const activeSession = sessionBase({ agentProjection: { status: 'running' } });
|
|
461
|
-
assert.equal(classifyAgentInput('lance un build', activeSession).kind, 'ambiguous');
|
|
462
|
-
assert.equal(classifyAgentInput('annule tout', activeSession).kind, 'cancel');
|
|
463
|
-
});
|
|
464
|
-
|
|
465
890
|
function orchestrableAgentSnapshot() {
|
|
466
891
|
return [{
|
|
467
892
|
agentInstanceId: 'production-1',
|
|
@@ -572,13 +997,11 @@ test('agent graph accepts plan steps with known capabilities and null capability
|
|
|
572
997
|
assert.deepEqual(session.headlessPlan.map((step) => step.id), ['export', 'report']);
|
|
573
998
|
});
|
|
574
999
|
|
|
575
|
-
test('buildAgentSystemPrompt
|
|
1000
|
+
test('buildAgentSystemPrompt assigns capability resolution exclusively to the runtime', () => {
|
|
576
1001
|
const withAgents = buildAgentSystemPrompt({ session: sessionBase({ agentRegistrySnapshot: orchestrableAgentSnapshot() }) });
|
|
577
|
-
assert.match(withAgents, /
|
|
578
|
-
assert.match(withAgents, /Never
|
|
579
|
-
|
|
580
|
-
const withoutAgents = buildAgentSystemPrompt({ session: sessionBase() });
|
|
581
|
-
assert.match(withoutAgents, /No orchestration capabilities discovered yet/);
|
|
1002
|
+
assert.match(withAgents, /call runtime__delegate with the user objective only/);
|
|
1003
|
+
assert.match(withAgents, /Never choose a capability, operation, agent, plan, or implementation yourself/);
|
|
1004
|
+
assert.doesNotMatch(withAgents, /ONLY values allowed in requiredCapability/);
|
|
582
1005
|
});
|
|
583
1006
|
|
|
584
1007
|
test('agent graph executes action inputs inside a runtime run instead of asking for clarification', async () => {
|
|
@@ -846,14 +1269,55 @@ test('Donna interprets a cleanup request and calls runtime__kill herself', async
|
|
|
846
1269
|
}
|
|
847
1270
|
});
|
|
848
1271
|
|
|
849
|
-
test('
|
|
1272
|
+
test('Donna refuses a direct mutating provider tool in interactive mode and is steered to delegation', async () => {
|
|
1273
|
+
const fetchedHosts = [];
|
|
1274
|
+
const originalFetch = globalThis.fetch;
|
|
1275
|
+
globalThis.fetch = async (url) => {
|
|
1276
|
+
fetchedHosts.push(new URL(String(url)).host);
|
|
1277
|
+
return { ok: true, status: 200, json: async () => ({}), text: async () => '{}', headers: { get: () => null } };
|
|
1278
|
+
};
|
|
1279
|
+
let calls = 0;
|
|
1280
|
+
const session = sessionBase({
|
|
1281
|
+
runtime: { url: 'http://runtime.test' },
|
|
1282
|
+
llm: {
|
|
1283
|
+
async completeWithTools({ tools, messages }) {
|
|
1284
|
+
calls += 1;
|
|
1285
|
+
if (calls === 1) {
|
|
1286
|
+
const names = tools.map((tool) => tool.function.name);
|
|
1287
|
+
assert.ok(!names.includes('production__production_start_job'), 'a mutating provider tool must not be offered in interactive mode');
|
|
1288
|
+
assert.ok(names.includes('runtime__delegate'), 'delegate must be offered in interactive mode');
|
|
1289
|
+
return {
|
|
1290
|
+
content: null,
|
|
1291
|
+
message: { role: 'assistant', content: null },
|
|
1292
|
+
tool_calls: [{ id: 'direct-call', type: 'function', function: { name: 'production__production_start_job', arguments: '{"type":"ingest"}' } }],
|
|
1293
|
+
};
|
|
1294
|
+
}
|
|
1295
|
+
const lastTool = (messages ?? []).filter((message) => message.role === 'tool').at(-1);
|
|
1296
|
+
assert.match(String(lastTool?.content ?? ''), /not available in interactive mode/);
|
|
1297
|
+
return { content: 'Objectif transmis au runtime.', message: { role: 'assistant', content: 'Objectif transmis au runtime.' }, tool_calls: null };
|
|
1298
|
+
},
|
|
1299
|
+
},
|
|
1300
|
+
});
|
|
1301
|
+
|
|
1302
|
+
try {
|
|
1303
|
+
const agent = createAgentGraph();
|
|
1304
|
+
const result = await agent.invoke({ input: 'lance une ingestion', session });
|
|
1305
|
+
assert.equal(calls, 2, 'the model must get a second turn after the refusal');
|
|
1306
|
+
assert.equal(result.response, 'Objectif transmis au runtime.');
|
|
1307
|
+
assert.ok(!fetchedHosts.includes('127.0.0.1:3000'), 'the refused provider tool must never be executed');
|
|
1308
|
+
} finally {
|
|
1309
|
+
globalThis.fetch = originalFetch;
|
|
1310
|
+
}
|
|
1311
|
+
});
|
|
1312
|
+
|
|
1313
|
+
test('buildAgentSystemPrompt uses the canonical wiki workspace status without filesystem fallback', () => {
|
|
850
1314
|
const workspacePath = mkdtempSync(join(tmpdir(), 'facts-'));
|
|
851
1315
|
try {
|
|
852
1316
|
mkdirSync(join(workspacePath, 'raw', 'untracked'), { recursive: true });
|
|
853
1317
|
writeFileSync(join(workspacePath, 'raw', 'untracked', 'note-a.md'), '# a\n');
|
|
854
1318
|
const prompt = buildAgentSystemPrompt({ session: sessionBase({ workspacePath }) });
|
|
855
|
-
assert.match(prompt, /call
|
|
856
|
-
assert.match(prompt, /
|
|
1319
|
+
assert.match(prompt, /call wiki__wiki_workspace_status first/);
|
|
1320
|
+
assert.match(prompt, /canonical read-only workspace state/);
|
|
857
1321
|
assert.doesNotMatch(prompt, /note-a\.md/);
|
|
858
1322
|
assert.doesNotMatch(prompt, /Workspace facts:/);
|
|
859
1323
|
} finally {
|
|
@@ -861,7 +1325,7 @@ test('buildAgentSystemPrompt delegates pending inputs to the capability provider
|
|
|
861
1325
|
}
|
|
862
1326
|
});
|
|
863
1327
|
|
|
864
|
-
test('Donna reads
|
|
1328
|
+
test('Donna reads workspace inventory from the canonical wiki status tool', async () => {
|
|
865
1329
|
const originalFetch = globalThis.fetch;
|
|
866
1330
|
let requestedArguments = null;
|
|
867
1331
|
globalThis.fetch = async (_url, options) => {
|
|
@@ -874,12 +1338,9 @@ test('Donna reads pending inputs from the qualified capability provider status t
|
|
|
874
1338
|
text: async () => JSON.stringify({ result: { content: [{
|
|
875
1339
|
type: 'text',
|
|
876
1340
|
text: JSON.stringify({
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
operation: 'ingest',
|
|
881
|
-
available: true,
|
|
882
|
-
pendingInputs: [{ type: 'file', ref: 'provider/source-a', label: 'source-a.md', mediaType: 'text/markdown' }],
|
|
1341
|
+
pendingSources: { count: 2, files: ['raw/untracked/a.md', 'raw/untracked/b.md'] },
|
|
1342
|
+
templates: { count: 1, files: ['templates/report.md'] },
|
|
1343
|
+
deliverables: { count: 0, files: [] },
|
|
883
1344
|
}),
|
|
884
1345
|
}] } }),
|
|
885
1346
|
};
|
|
@@ -887,13 +1348,13 @@ test('Donna reads pending inputs from the qualified capability provider status t
|
|
|
887
1348
|
let calls = 0;
|
|
888
1349
|
const session = sessionBase({
|
|
889
1350
|
mcp: {
|
|
890
|
-
|
|
1351
|
+
wiki: {
|
|
891
1352
|
status: 'connected',
|
|
892
1353
|
url: 'http://127.0.0.1:3000/mcp/',
|
|
893
1354
|
tools: [{
|
|
894
|
-
name: '
|
|
895
|
-
description: 'Read
|
|
896
|
-
inputSchema: { type: 'object', properties: {
|
|
1355
|
+
name: 'wiki_workspace_status',
|
|
1356
|
+
description: 'Read the canonical local workspace inventory.',
|
|
1357
|
+
inputSchema: { type: 'object', properties: {} },
|
|
897
1358
|
}],
|
|
898
1359
|
},
|
|
899
1360
|
},
|
|
@@ -907,20 +1368,21 @@ test('Donna reads pending inputs from the qualified capability provider status t
|
|
|
907
1368
|
tool_calls: [{
|
|
908
1369
|
id: 'status-call',
|
|
909
1370
|
type: 'function',
|
|
910
|
-
function: { name: '
|
|
1371
|
+
function: { name: 'wiki__wiki_workspace_status', arguments: '{}' },
|
|
911
1372
|
}],
|
|
912
1373
|
};
|
|
913
1374
|
}
|
|
914
|
-
assert.match(JSON.stringify(messages), /
|
|
915
|
-
return { content: '
|
|
1375
|
+
assert.match(JSON.stringify(messages), /raw\/untracked\/a\.md/);
|
|
1376
|
+
return { content: 'Deux fichiers sont en attente : a.md et b.md.', message: { role: 'assistant', content: 'Deux fichiers sont en attente : a.md et b.md.' }, tool_calls: null };
|
|
916
1377
|
},
|
|
917
1378
|
},
|
|
918
1379
|
});
|
|
919
1380
|
|
|
920
1381
|
try {
|
|
921
|
-
const result = await createAgentGraph().invoke({ input: '
|
|
922
|
-
assert.equal(result.response, '
|
|
923
|
-
assert.deepEqual(requestedArguments, {
|
|
1382
|
+
const result = await createAgentGraph().invoke({ input: 'as ton des fichier en attente d ingestion', session });
|
|
1383
|
+
assert.equal(result.response, 'Deux fichiers sont en attente : a.md et b.md.');
|
|
1384
|
+
assert.deepEqual(requestedArguments, {});
|
|
1385
|
+
assert.equal(session.headlessPlan, null, 'a read-only workspace inventory must not create a plan');
|
|
924
1386
|
} finally {
|
|
925
1387
|
globalThis.fetch = originalFetch;
|
|
926
1388
|
}
|