@dotdrelle/wiki-manager 0.12.12 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docker-compose.yml +1 -1
- package/package.json +1 -1
- package/src/agent/graph.js +331 -143
- package/src/agent/graph.test.js +516 -54
- package/src/agent/llm.js +5 -5
- package/src/cli/wiki-manager.js +225 -6
- package/src/cli/wiki-manager.test.js +28 -0
- package/src/commands/slash.js +32 -11
- package/src/commands/slash.test.js +9 -1
- package/src/core/buildInfo.json +2 -2
- package/src/core/mcp.js +12 -4
- package/src/core/skills.js +0 -28
- package/src/orchestrator/capabilityRegistry.js +14 -0
- package/src/orchestrator/capabilityRegistry.test.js +12 -1
- package/src/orchestrator/dependencyResolver.js +10 -1
- package/src/orchestrator/dispatcher.js +34 -3
- package/src/orchestrator/dispatcher.test.js +34 -0
- package/src/orchestrator/objectiveResolver.js +79 -0
- package/src/orchestrator/objectiveResolver.test.js +50 -0
- package/src/orchestrator/scheduler.test.js +25 -0
- package/src/runtime/client.js +32 -1
- package/src/runtime/recoveryManager.js +14 -7
- package/src/runtime/runner.js +112 -13
- package/src/runtime/runner.test.js +64 -1
- package/src/runtime/server.js +43 -3
- package/src/runtime/supervisor.js +4 -1
- package/src/runtime/supervisor.test.js +49 -0
- package/src/shell/repl.js +45 -38
- package/src/shell/repl.test.js +81 -12
- package/src/shell/useSession.ts +15 -3
package/src/agent/graph.js
CHANGED
|
@@ -14,8 +14,8 @@ import { extractActivity, formatActivitySummary, parseJsonText, sessionActivitie
|
|
|
14
14
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
15
15
|
import { enqueueProductionJob, ensureJobQueue, formatQueue, productionLockBusy } from '../core/jobQueue.js';
|
|
16
16
|
import { updateWorkspaceProfilePreference } from '../core/profile.js';
|
|
17
|
-
import {
|
|
18
|
-
import { fetchRuntimeState, postRuntimeCancel, postRuntimeControl, postRuntimeKill } from '../runtime/client.js';
|
|
17
|
+
import { capabilityRegistryForSession } from '../orchestrator/capabilityRegistry.js';
|
|
18
|
+
import { fetchRuntimeState, postRuntimeApprove, postRuntimeCancel, postRuntimeControl, postRuntimeDelegate, postRuntimeKill } from '../runtime/client.js';
|
|
19
19
|
|
|
20
20
|
const MAX_TOOL_ITERATIONS = 80;
|
|
21
21
|
const MAX_SPINNER_ARG_LENGTH = 96;
|
|
@@ -27,7 +27,7 @@ const MAX_PROFILE_CHARS = 4000;
|
|
|
27
27
|
const INTERNAL_TOOL_SERVERS = {
|
|
28
28
|
wiki: ['plan_set', 'plan_done'],
|
|
29
29
|
shell: ['run_command', 'read_command', 'profile_update'],
|
|
30
|
-
runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue', '
|
|
30
|
+
runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue', 'delegate'],
|
|
31
31
|
};
|
|
32
32
|
|
|
33
33
|
const AGENT_SLASH_COMMANDS = new Set([
|
|
@@ -40,8 +40,8 @@ const AGENT_SLASH_COMMANDS = new Set([
|
|
|
40
40
|
'services',
|
|
41
41
|
'skills',
|
|
42
42
|
'upload',
|
|
43
|
-
'uploads',
|
|
44
43
|
'queue',
|
|
44
|
+
'openui',
|
|
45
45
|
]);
|
|
46
46
|
|
|
47
47
|
const SHELL_RUN_COMMAND_TOOL = {
|
|
@@ -50,7 +50,7 @@ const SHELL_RUN_COMMAND_TOOL = {
|
|
|
50
50
|
name: 'shell__run_command',
|
|
51
51
|
description: [
|
|
52
52
|
'Run a deterministic wiki-manager slash command inside the current shell session.',
|
|
53
|
-
'Allowed commands: /workspace list, /workspace init <name> [path], /use <workspace>, /config, /status, /services, /skills, /skills show <name>, /skills run <name>, /upload <path>, /upload convert <id|pending
|
|
53
|
+
'Allowed commands: /workspace list, /workspace init <name> [path], /use <workspace>, /config, /status, /services, /skills, /skills show <name>, /skills run <name>, /upload <path>, /upload convert <id|pending>.',
|
|
54
54
|
'Do not use for arbitrary system shell commands, /workspace delete, /mcp call, /wiki run, /start, /stop, /logs, or /exit.',
|
|
55
55
|
].join(' '),
|
|
56
56
|
parameters: {
|
|
@@ -73,7 +73,7 @@ const SHELL_READ_COMMAND_TOOL = {
|
|
|
73
73
|
name: 'shell__read_command',
|
|
74
74
|
description: [
|
|
75
75
|
'Run a read-only deterministic wiki-manager slash command inside the current shell session.',
|
|
76
|
-
'Allowed commands: /help, /version, /config, /config list, /config status, /status, /services, /skills, /skills list, /skills show <name>, /
|
|
76
|
+
'Allowed commands: /help, /version, /config, /config list, /config status, /status, /services, /skills, /skills list, /skills show <name>, /queue.',
|
|
77
77
|
'Do not use for workspace creation/deletion, uploads conversion, service start/stop, MCP calls, wiki runs, or any mutation.',
|
|
78
78
|
].join(' '),
|
|
79
79
|
parameters: {
|
|
@@ -165,20 +165,18 @@ const RUNTIME_ENQUEUE_TOOL = {
|
|
|
165
165
|
},
|
|
166
166
|
};
|
|
167
167
|
|
|
168
|
-
const
|
|
168
|
+
const RUNTIME_DELEGATE_TOOL = {
|
|
169
169
|
type: 'function',
|
|
170
170
|
function: {
|
|
171
|
-
name: '
|
|
172
|
-
description: '
|
|
171
|
+
name: 'runtime__delegate',
|
|
172
|
+
description: 'Delegate the user objective to the runtime. Pass the objective in natural language without choosing a capability, operation, agent, plan, file list, or implementation. The runtime resolves the agent, obtains and validates the real plan before accepting the run.',
|
|
173
173
|
parameters: {
|
|
174
174
|
type: 'object',
|
|
175
175
|
additionalProperties: false,
|
|
176
176
|
properties: {
|
|
177
|
-
|
|
178
|
-
operation: { type: 'string', description: 'Operation supported by the capability (e.g. ingest, build).' },
|
|
179
|
-
inputs: { type: 'array', items: { type: 'string' }, description: 'Optional file subset; omit to cover everything pending.' },
|
|
177
|
+
objective: { type: 'string', description: 'The complete user objective, preserving scope and constraints but containing no invented technical identifiers.' },
|
|
180
178
|
},
|
|
181
|
-
required: ['
|
|
179
|
+
required: ['objective'],
|
|
182
180
|
},
|
|
183
181
|
},
|
|
184
182
|
};
|
|
@@ -264,15 +262,124 @@ const AgentState = Annotation.Root({
|
|
|
264
262
|
}),
|
|
265
263
|
toolIterations: Annotation({ default: () => 0 }),
|
|
266
264
|
pendingToolCalls: Annotation(),
|
|
265
|
+
allowedToolNames: Annotation(),
|
|
267
266
|
inputClassification: Annotation(),
|
|
268
267
|
readyToStream: Annotation(),
|
|
269
268
|
streamContext: Annotation(),
|
|
270
269
|
streamedInline: Annotation(),
|
|
271
270
|
retryWithoutTool: Annotation({ default: () => false }),
|
|
271
|
+
invalidResponseRetries: Annotation({ default: () => 0 }),
|
|
272
|
+
invalidToolCallRetries: Annotation({ default: () => 0 }),
|
|
273
|
+
forceDelegation: Annotation({ default: () => false }),
|
|
272
274
|
});
|
|
273
275
|
|
|
276
|
+
function invalidToolCalls(toolCalls) {
|
|
277
|
+
if (!Array.isArray(toolCalls)) return [];
|
|
278
|
+
return toolCalls.filter((call) => {
|
|
279
|
+
if (!call?.id || !call?.function?.name) return true;
|
|
280
|
+
try {
|
|
281
|
+
const args = JSON.parse(call.function.arguments || '{}');
|
|
282
|
+
return !args || typeof args !== 'object' || Array.isArray(args);
|
|
283
|
+
} catch {
|
|
284
|
+
return true;
|
|
285
|
+
}
|
|
286
|
+
});
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
export function normalizeToolArgumentsFromSchema(args, parameters) {
|
|
290
|
+
if (!args || typeof args !== 'object' || Array.isArray(args)) return args;
|
|
291
|
+
const schema = parameters && typeof parameters === 'object' ? parameters : {};
|
|
292
|
+
const properties = schema.properties && typeof schema.properties === 'object' ? schema.properties : {};
|
|
293
|
+
const required = Array.isArray(schema.required) ? schema.required.filter((key) => typeof key === 'string') : [];
|
|
294
|
+
const missing = required.filter((key) => args[key] === undefined);
|
|
295
|
+
const unknown = Object.keys(args).filter((key) => properties[key] === undefined);
|
|
296
|
+
if (missing.length !== 1 || unknown.length !== 1) return args;
|
|
297
|
+
const target = missing[0];
|
|
298
|
+
const source = unknown[0];
|
|
299
|
+
if (!schemaValueMatches(args[source], properties[target])) return args;
|
|
300
|
+
const normalized = { ...args, [target]: args[source] };
|
|
301
|
+
delete normalized[source];
|
|
302
|
+
return normalized;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
function schemaValueMatches(value, propertySchema) {
|
|
306
|
+
const types = Array.isArray(propertySchema?.type) ? propertySchema.type : [propertySchema?.type];
|
|
307
|
+
if (types.includes(undefined) || types.includes(null)) return true;
|
|
308
|
+
return types.some((type) => {
|
|
309
|
+
if (type === 'array') return Array.isArray(value);
|
|
310
|
+
if (type === 'object') return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
311
|
+
if (type === 'integer') return Number.isInteger(value);
|
|
312
|
+
if (type === 'number') return typeof value === 'number' && Number.isFinite(value);
|
|
313
|
+
if (type === 'null') return value === null;
|
|
314
|
+
return typeof value === type;
|
|
315
|
+
});
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
function toolDefinitionForCall(session, callName) {
|
|
319
|
+
const internal = [
|
|
320
|
+
SHELL_RUN_COMMAND_TOOL,
|
|
321
|
+
SHELL_READ_COMMAND_TOOL,
|
|
322
|
+
SHELL_PROFILE_UPDATE_TOOL,
|
|
323
|
+
RUNTIME_STATUS_TOOL,
|
|
324
|
+
RUNTIME_CANCEL_TOOL,
|
|
325
|
+
RUNTIME_KILL_TOOL,
|
|
326
|
+
RUNTIME_APPROVE_TOOL,
|
|
327
|
+
RUNTIME_ENQUEUE_TOOL,
|
|
328
|
+
RUNTIME_DELEGATE_TOOL,
|
|
329
|
+
WIKI_PLAN_SET_TOOL,
|
|
330
|
+
WIKI_PLAN_DONE_TOOL,
|
|
331
|
+
];
|
|
332
|
+
return [...internal, ...buildLlmTools(session?.mcp)]
|
|
333
|
+
.find((item) => item?.function?.name === callName) ?? null;
|
|
334
|
+
}
|
|
335
|
+
|
|
274
336
|
function commandList(session) {
|
|
275
|
-
return session.commands
|
|
337
|
+
return session.commands
|
|
338
|
+
.filter((command) => AGENT_SLASH_COMMANDS.has(command))
|
|
339
|
+
.map((command) => `/${command}`)
|
|
340
|
+
.join(', ');
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
export function invalidSuggestedSlashCommands(content, session) {
|
|
344
|
+
const allowed = new Set((session?.commands ?? []).filter((command) => AGENT_SLASH_COMMANDS.has(command)));
|
|
345
|
+
const candidates = new Set();
|
|
346
|
+
for (const line of String(content ?? '').split(/\r?\n/)) {
|
|
347
|
+
const trimmed = line.trim();
|
|
348
|
+
const standalone = trimmed.match(/^\/([a-z][\w-]*)\b/i);
|
|
349
|
+
if (standalone) candidates.add(standalone[1].toLowerCase());
|
|
350
|
+
for (const match of line.matchAll(/`\/([a-z][\w-]*)\b/gi)) candidates.add(match[1].toLowerCase());
|
|
351
|
+
}
|
|
352
|
+
return [...candidates].filter((command) => !allowed.has(command)).sort();
|
|
353
|
+
}
|
|
354
|
+
|
|
355
|
+
export function invalidUserFacingToolNames(content, session) {
|
|
356
|
+
const text = String(content ?? '');
|
|
357
|
+
const connected = buildLlmTools(session?.mcp)
|
|
358
|
+
.map((item) => item?.function?.name)
|
|
359
|
+
.filter(Boolean)
|
|
360
|
+
.filter((name) => text.includes(name));
|
|
361
|
+
const syntactic = [...text.matchAll(/\b[a-z][a-z0-9_-]*__[a-z][a-z0-9_-]*\b/gi)].map((match) => match[0]);
|
|
362
|
+
return [...new Set([...connected, ...syntactic])].sort();
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
async function classifyRequestedAction(llm, input, signal) {
|
|
366
|
+
try {
|
|
367
|
+
const result = await llm.completeWithTools({
|
|
368
|
+
system: [
|
|
369
|
+
'Classify whether the user explicitly requests a real state-changing action now.',
|
|
370
|
+
'Actions include starting, stopping, importing, ingesting, building, exporting, configuring, writing, deleting, or sending.',
|
|
371
|
+
'Questions, explanations, status questions, greetings, and hypothetical discussions are not actions.',
|
|
372
|
+
'Return JSON only: {"action":true} or {"action":false}.',
|
|
373
|
+
].join('\n'),
|
|
374
|
+
tools: [],
|
|
375
|
+
messages: [{ role: 'user', content: String(input ?? '') }],
|
|
376
|
+
signal,
|
|
377
|
+
});
|
|
378
|
+
const text = String(result?.content ?? '').trim().replace(/^```(?:json)?\s*/i, '').replace(/\s*```$/, '');
|
|
379
|
+
return JSON.parse(text)?.action === true;
|
|
380
|
+
} catch {
|
|
381
|
+
return false;
|
|
382
|
+
}
|
|
276
383
|
}
|
|
277
384
|
|
|
278
385
|
function summarizeToolArguments(rawArguments) {
|
|
@@ -384,8 +491,7 @@ function assertAgentReadSlashCommandAllowed(commandLine) {
|
|
|
384
491
|
command === 'services' ||
|
|
385
492
|
command === 'queue' ||
|
|
386
493
|
(command === 'config' && ['', 'list', 'status'].includes(subcommand)) ||
|
|
387
|
-
(command === 'skills' && ['', 'list', 'show'].includes(subcommand))
|
|
388
|
-
(command === 'uploads' && ['', 'list'].includes(subcommand));
|
|
494
|
+
(command === 'skills' && ['', 'list', 'show'].includes(subcommand));
|
|
389
495
|
if (!allowed) {
|
|
390
496
|
throw new Error(`Read-only command is not available to the agent: /${parts.join(' ')}`);
|
|
391
497
|
}
|
|
@@ -530,9 +636,7 @@ function emitAgentEvent(session, type, origin, payload = {}) {
|
|
|
530
636
|
// agents. This is the live registry the dispatcher will resolve against —
|
|
531
637
|
// a plan declaring anything outside this set can only stall forever.
|
|
532
638
|
export function knownCapabilityIds(session) {
|
|
533
|
-
const registry = session
|
|
534
|
-
agents: session?.agentRegistrySnapshot ?? session?.agents ?? [],
|
|
535
|
-
});
|
|
639
|
+
const registry = capabilityRegistryForSession(session);
|
|
536
640
|
const snapshot = typeof registry.snapshot === 'function' ? registry.snapshot() : registry;
|
|
537
641
|
return [...new Set(Object.keys(snapshot ?? {}).map((key) => {
|
|
538
642
|
const index = key.lastIndexOf('@');
|
|
@@ -581,25 +685,34 @@ async function handleRuntimeControlTool(session, tool, args = {}) {
|
|
|
581
685
|
return result.cancelled ? 'Runtime run cancellation requested.' : `No active run to cancel${result.reason ? ` (${result.reason})` : ''}.`;
|
|
582
686
|
}
|
|
583
687
|
if (tool === 'approve') {
|
|
584
|
-
const
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
688
|
+
const state = await fetchRuntimeState({ url, workspace });
|
|
689
|
+
const pending = (Array.isArray(state?.approvals) ? state.approvals : [])
|
|
690
|
+
.filter((approval) => approval.status === 'pending_approval');
|
|
691
|
+
const runId = state?.runId
|
|
692
|
+
?? state?.runs?.find((run) => ['running', 'pending_approval'].includes(run.status))?.id
|
|
693
|
+
?? null;
|
|
694
|
+
if (!runId || pending.length === 0) return 'No pending approval found.';
|
|
695
|
+
const approvalClasses = [...new Set(pending.flatMap((approval) => {
|
|
696
|
+
const value = approval.approvalClasses ?? approval.approvalClass ?? [];
|
|
697
|
+
return Array.isArray(value) ? value : [value];
|
|
698
|
+
}).map(String).filter(Boolean))];
|
|
699
|
+
const result = await postRuntimeApprove({
|
|
590
700
|
url,
|
|
591
701
|
workspace,
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
// No concurrency dictated here: the runtime reads the provider's
|
|
597
|
-
// own declared capacity (agent_describe limits).
|
|
598
|
-
},
|
|
702
|
+
runId,
|
|
703
|
+
scope: 'run',
|
|
704
|
+
planRevision: state?.planRevision ?? null,
|
|
705
|
+
approvalClasses: approvalClasses.length > 0 ? approvalClasses : ['default'],
|
|
599
706
|
});
|
|
707
|
+
return result?.approved ? 'Current validated plan approved.' : 'No pending approval found.';
|
|
708
|
+
}
|
|
709
|
+
if (tool === 'delegate') {
|
|
710
|
+
const objective = String(args.objective ?? '').trim();
|
|
711
|
+
if (!objective) return 'Delegation rejected: missing objective.';
|
|
712
|
+
const result = await postRuntimeDelegate(objective, { url, workspace });
|
|
600
713
|
return result?.runId
|
|
601
|
-
? `
|
|
602
|
-
: `
|
|
714
|
+
? `Délégation acceptée (${String(result.runId).slice(0, 8)}) après validation du plan réel : ${result.delegation?.tasks ?? 0} tâche(s), ${result.delegation?.agent ?? 'agent résolu'}. L'approbation porte sur ce plan.`
|
|
715
|
+
: `Délégation refusée : ${result?.error ?? JSON.stringify(result)}`;
|
|
603
716
|
}
|
|
604
717
|
if (tool === 'enqueue') {
|
|
605
718
|
const result = await postRuntimeControl('message', { url, workspace, input: String(args.input ?? ''), intent: 'enqueue' });
|
|
@@ -728,7 +841,15 @@ export function buildAgentSystemPrompt(state) {
|
|
|
728
841
|
const workspace = state.session.workspace ?? 'no workspace selected';
|
|
729
842
|
const wikirc = state.session.wikirc?.profile ?? 'no profile loaded';
|
|
730
843
|
const language = state.session.language ?? 'en-US';
|
|
731
|
-
|
|
844
|
+
// Advertise only the read-only tools Donna may call directly. Listing
|
|
845
|
+
// mutating provider tools (e.g. production__production_start_job) here teaches
|
|
846
|
+
// a capable model to invoke them directly and bypass runtime__delegate.
|
|
847
|
+
const mcpTools = formatMcpToolsForAgent(state.session.mcp, {
|
|
848
|
+
include: (qualifiedName, tool) => isDonnaReadTool({
|
|
849
|
+
function: { name: qualifiedName },
|
|
850
|
+
readOnly: tool?.annotations?.readOnlyHint === true,
|
|
851
|
+
}),
|
|
852
|
+
});
|
|
732
853
|
const skills = formatSkillsForAgent(state.session);
|
|
733
854
|
const customPrompt = state.session.systemPrompt ?? null;
|
|
734
855
|
const workspaceProfile = loadWorkspaceProfile(state.session.workspacePath);
|
|
@@ -743,74 +864,39 @@ export function buildAgentSystemPrompt(state) {
|
|
|
743
864
|
`Current wikirc profile: ${wikirc}.`,
|
|
744
865
|
`Available primitives: ${commandList(state.session)}.`,
|
|
745
866
|
'Only announce or call slash commands that appear exactly in Available primitives. Do not invent command names, subcommands, or arguments.',
|
|
746
|
-
'Connected MCP tools
|
|
867
|
+
'Connected read-only MCP tools you may call directly to answer questions (server__tool naming convention). Every mutation or action — ingest, build, export, configure, send, write — goes through runtime__delegate, never a direct provider tool call:',
|
|
747
868
|
mcpTools,
|
|
748
869
|
'Current local MCP job queue:',
|
|
749
870
|
formatQueue(state.session),
|
|
750
871
|
'Available skills:',
|
|
751
872
|
skills,
|
|
752
|
-
'
|
|
873
|
+
'In interactive agent mode you may call only the read-only tools and runtime control/delegation tools actually provided to you.',
|
|
753
874
|
'When the user asks for an action that can be performed with connected MCP tools or safe primitives, do not answer with future intent such as "I will call...", "I am going to run...", or "launching..." unless you also call the tool in the same turn. Either call the tool now, ask for the exact missing required arguments, or explain the concrete blocker.',
|
|
754
875
|
'Execution truthfulness: never invent a job id, status, percentage, duration, generated file, file content, URL, command, or tool result. An action is executed only when you call an available tool and receive its result. Examples and placeholders are forbidden in execution reports.',
|
|
755
876
|
'After any completed action, give a short factual summary based only on the tool result: outcome and concrete outputs or references actually returned. Mention a viewing primitive only when it exists in Available primitives and is relevant. Do not interpret generated content, propose verification checklists, invent next steps, or suggest commands unless the user explicitly asks.',
|
|
877
|
+
'Never add a "Next steps", "Prochaines étapes", "À suivre", options, or suggestions section unless the user explicitly asks what to do next. End after the requested result or the concrete error.',
|
|
756
878
|
'When calling a tool, emit no preliminary narration. Call it directly; the PLAN and Activity panels show progress. After completion, keep the final response concise and proportional to the result.',
|
|
757
|
-
'
|
|
758
|
-
'
|
|
759
|
-
'
|
|
760
|
-
'
|
|
761
|
-
'
|
|
762
|
-
|
|
763
|
-
state.session.headless ? 'HEADLESS MODE ACTIVE. Execute the requested skill or task autonomously using available safe primitives and MCP tools. Do not ask for interactive confirmation unless the request is genuinely ambiguous or outside the loaded workspace.' : null,
|
|
764
|
-
'',
|
|
765
|
-
'You have two internal planning tools: wiki__plan_set and wiki__plan_done.',
|
|
766
|
-
'Prefer MCP tools that declare their own plan via _activity.plan.steps — when such a tool returns _activity, the shell creates and tracks the plan automatically without requiring wiki__plan_set.',
|
|
767
|
-
'Use wiki__plan_set when the MCP tool cannot declare its own plan or when the task spans multiple independent tools (e.g. CME export then email report). For a single self-describing async job, wiki__plan_set is optional.',
|
|
768
|
-
'',
|
|
769
|
-
(() => {
|
|
770
|
-
const capabilityIds = knownCapabilityIds(state.session);
|
|
771
|
-
return capabilityIds.length > 0
|
|
772
|
-
? `Known orchestration capabilities — the ONLY values allowed in requiredCapability: ${capabilityIds.join(', ')}. Never invent capability names; a plan declaring an unknown capability will be rejected. A step you execute yourself directly takes requiredCapability: null.`
|
|
773
|
-
: 'No orchestration capabilities discovered yet: declare plan steps with requiredCapability: null and execute them yourself with the connected MCP tools.';
|
|
774
|
-
})(),
|
|
775
|
-
'',
|
|
776
|
-
'Task startup:',
|
|
777
|
-
' 1. If the next MCP tool returns _activity.plan.steps, call that tool directly; the shell will create the visible plan from the returned activity.',
|
|
778
|
-
' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, operation, arguments, dependsOn, outputRefs}; capability steps need operation+arguments for the dispatcher to execute them; a legacy list of strings is still accepted.',
|
|
779
|
-
' Multi-tool example: wiki__plan_set(steps=[{id:"cme-export",description:"CME export",requiredCapability:"external-source.export",dependsOn:[],outputRefs:["raw/untracked"]},{id:"production",description:"Production pipeline",requiredCapability:"knowledge.pipeline",dependsOn:["cme-export"],outputRefs:["deliverables"]}])',
|
|
780
|
-
' 3. Immediately execute the first step using the appropriate MCP tool. Do not start step 2 in the same turn unless one async pipeline tool owns and declares the whole sequence.',
|
|
781
|
-
' For synchronous steps (result is immediate, no _activity polling), call wiki__plan_done(step=1) after confirming success.',
|
|
782
|
-
' For async MCP jobs (returns _activity with poll), the orchestrator tracks completion automatically.',
|
|
783
|
-
'',
|
|
784
|
-
state.session.headless ? [
|
|
785
|
-
'Headless follow-up turns — the orchestrator re-invokes you with:',
|
|
786
|
-
' (a) the original task,',
|
|
787
|
-
' (b) the current plan status — [✓] done / [✗] failed / [ ] pending,',
|
|
788
|
-
' (c) the just-completed activities.',
|
|
789
|
-
' Read the plan status. Find the first [ ] pending step. Execute it only.',
|
|
790
|
-
' Never re-execute a [✓] or [✗] step. Never skip a [ ] step.',
|
|
791
|
-
'',
|
|
792
|
-
'Final turn — when all steps are [✓] or [✗]: respond with a concise summary. Do not start new actions.',
|
|
793
|
-
].join('\n') : null,
|
|
794
|
-
'',
|
|
795
|
-
'On failure: if a completed activity is failed/error/cancelled, call wiki__plan_done(step=N, status="failed") then stop with a clear error report.',
|
|
796
|
-
].filter(Boolean).join('\n'),
|
|
879
|
+
'Keep every response synthetic and information-dense. Use only the lines needed, and never exceed roughly 15 to 20 short lines even for a detailed answer. Never expose internal reasoning, repeated checks, tool-selection commentary, or a chronological diary. Prioritize the result, essential facts, concrete errors, and actual outputs.',
|
|
880
|
+
'Configuration, connector, import, export, conversion, generation, and every other mutation are actions: delegate the objective to the runtime instead of calling an external tool directly.',
|
|
881
|
+
'For any question about the current workspace inventory or what is waiting there, call wiki__wiki_workspace_status first and answer only from its result. This is the canonical read-only workspace state; do not reconstruct it from upload, connector, or production tools.',
|
|
882
|
+
'Tool identifiers are private implementation details. Never print MCP tool names such as server__tool in a user-facing answer. Describe the human result instead.',
|
|
883
|
+
'Never suggest a manual filesystem command or implementation workaround unless the user explicitly asks for manual instructions. For an action request, delegate the objective and let the specialized agent determine paths and operations from its live contract.',
|
|
884
|
+
'Skills are documentation only in this stabilized version. Never execute a skill from conversation; delegate the user objective.',
|
|
797
885
|
'For service actions, recommend only available service primitives from Available primitives, with the exact service name when the primitive supports one.',
|
|
798
|
-
'Scope discipline: execute ONLY the action(s) the user explicitly requested. Never chain additional mutating operations (ingest, build, export, polish, delete, send…) that the user did not ask for — even when diagnostics or recommendations suggest them. Finish the requested
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
'For
|
|
803
|
-
'
|
|
804
|
-
'
|
|
805
|
-
'
|
|
806
|
-
'If production__production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
|
|
807
|
-
'For diagnostics (doctor), use production__production_start_job with type="doctor" like any other production job; /wiki run doctor is only the fallback when the production MCP is not connected. Use /workspace init <name> [path] for low-level non-interactive workspace creation. In the interactive TUI, /new <name> opens the setup wizard. Use /wiki for index, or /wiki run index through the explicit backup hatch. Use /wiki run init only for explicit current-workspace llm-wiki init.',
|
|
886
|
+
'Scope discipline: execute ONLY the action(s) the user explicitly requested. Never chain additional mutating operations (ingest, build, export, polish, delete, send…) that the user did not ask for — even when diagnostics or recommendations suggest them. Finish with the requested result and stop. Example: "applique les recommandations de config" means apply the config; it does NOT authorize launching the ingest those recommendations mention.',
|
|
887
|
+
state.session.runtime?.url
|
|
888
|
+
? 'The runtime is connected and runtime__delegate is bound and available to you right now — it is a tool you call directly, not a slash command or a missing primitive. It is the ONLY way to execute an action (ingest, build, export, configure, send…). Never tell the user that delegation or the runtime is unavailable while it is connected; call runtime__delegate instead.'
|
|
889
|
+
: 'No runtime is connected, so you cannot execute actions. State that plainly and name the runtime connection as the missing capability — do not invent a workaround.',
|
|
890
|
+
'For any requested action, call runtime__delegate with the user objective only. Never choose a capability, operation, agent, plan, or implementation yourself. The runtime resolves the registry and validates the provider plan before accepting. Never call <provider>__agent_plan, <provider>__agent_execute, legacy production__production_start_job, wiki__plan_set, or wiki__plan_done from interactive chat.',
|
|
891
|
+
'Do not ask the user which sources, files, connectors, or templates to use for an ingest, build, or export: the specialized agent discovers them from the workspace. When the objective is clear (e.g. "lance une ingestion"), delegate it as stated, without a clarifying question.',
|
|
892
|
+
'If runtime__delegate returns a blocker or no specialized provider is available, report only that concrete blocker concisely. Never replace the missing execution path with a suggested slash command, skill, MCP tool name, manual file move, administrator escalation, or alternative workflow unless the user explicitly asks for alternatives.',
|
|
893
|
+
'For workspace inventory and page listings, use the connected wiki MCP read tools. Never invent or call a /wiki shell command through shell__run_command. Use /workspace init <name> [path] for low-level non-interactive workspace creation; in the interactive TUI, /new <name> opens the setup wizard.',
|
|
808
894
|
'If an action requires tools or skills not available yet, explain the limitation and name the expected primitive.',
|
|
809
895
|
workspaceProfile
|
|
810
896
|
? `Workspace profile (.wiki/profile.md) — durable user preferences, apply these to every reply (tone, tutoiement/vouvoiement, formatting, etc.):\n${workspaceProfile}`
|
|
811
897
|
: null,
|
|
812
898
|
'Runtime control: you have runtime__status, runtime__cancel, runtime__kill, runtime__approve and runtime__enqueue. When the user asks to stop, remove, clean or kill the current run, its jobs or the queue ("supprime le job et la queue", "arr\u00eate tout"), call runtime__kill (or runtime__cancel for a soft stop of just the run) and confirm what was stopped. For questions about what is running or queued, call runtime__status and answer from its data. When the user consents to a pending approval in any phrasing ("vas-y", "ok pour l\'export"), call runtime__approve. When the user asks for a NEW action while a run is active, do not execute it: propose runtime__enqueue (run it after) or, if they insist it replaces the current work, runtime__kill then the new action.',
|
|
813
|
-
'
|
|
899
|
+
'Durable profile updates are actions in this stabilized version: delegate them instead of writing directly.',
|
|
814
900
|
].filter(Boolean).join('\n');
|
|
815
901
|
|
|
816
902
|
return customPrompt ? `${customPrompt}\n\n${agentContext}` : agentContext;
|
|
@@ -840,66 +926,50 @@ export function formatLlmUnavailableMessage(reason) {
|
|
|
840
926
|
return `⚠ LLM injoignable : ${clean || 'raison inconnue'}`;
|
|
841
927
|
}
|
|
842
928
|
|
|
843
|
-
// Verbs that clearly request work (a runtime run), in French and English.
|
|
844
|
-
// "configure/configurer" is an action; the nouns "config/configuration" are
|
|
845
|
-
// NOT matched here — asking for a config is an observe request.
|
|
846
|
-
const ACTION_REQUEST_PATTERN = /\b(lance|relance|d[eé]marre|start|ex[eé]cute|execute|g[eé]n[eè]re|generate|build|construis|exporte?|ingest\w*|ing[eè]re|importe?|convert(?:is|it|s)?|cr[eé]e|create|polish|publie|publish|d[eé]ploie|deploy|envoie|send|configure[rsz]?|setup|installe|update|mets? [aà] jour|supprime|delete|efface|nettoie|clean|r[eé]pare|fix|corrige)\b/i;
|
|
847
|
-
|
|
848
|
-
// Explicit explanation/question markers dominate action verbs: "explique le
|
|
849
|
-
// build" is a question about the build, not a request to build.
|
|
850
|
-
const EXPLANATION_REQUEST_PATTERN = /\b(explique|explain|pourquoi|why|comment|how|c'est quoi|qu'est[- ]ce)\b/i;
|
|
851
|
-
|
|
852
|
-
export function classifyAgentInput(input, session) {
|
|
853
|
-
const lower = String(input ?? '').toLowerCase();
|
|
854
|
-
const hasActiveRun = session?.agentProjection?.status === 'running'
|
|
855
|
-
|| sessionActivities(session).some((activity) => !activity.terminal);
|
|
856
|
-
if (/\b(valide tout|approve all|approve|approuve|valid[eé]|ok pour tout|go pour tout)\b/i.test(lower)) {
|
|
857
|
-
return { kind: 'approve', confidence: 0.86, reason: 'approval_request', activeRun: hasActiveRun };
|
|
858
|
-
}
|
|
859
|
-
if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort|supprime|kill|tue|purge|vide la (file|queue)|nettoie la (file|queue))\b/i.test(lower)) {
|
|
860
|
-
return { kind: 'cancel', confidence: 0.86, reason: 'cancel_request', activeRun: hasActiveRun };
|
|
861
|
-
}
|
|
862
|
-
if (/\b(plus tard|later|ensuite|apr[eè]s ce run|enqueue|mets en file|met en file|futur|next run|future run)\b/i.test(lower)) {
|
|
863
|
-
return { kind: 'enqueue_run', confidence: 0.8, reason: 'future_run_request', activeRun: hasActiveRun };
|
|
864
|
-
}
|
|
865
|
-
if (EXPLANATION_REQUEST_PATTERN.test(lower)) {
|
|
866
|
-
return { kind: 'observe', confidence: 0.86, reason: 'explanation_request', activeRun: hasActiveRun };
|
|
867
|
-
}
|
|
868
|
-
// Observe markers only win when no action verb is present: "où en est le
|
|
869
|
-
// run" is observe, "lance le run" is an action request.
|
|
870
|
-
if (/\b(o[uù] en es[t-]|status|statut|progress|progression|run|job|queue|logs?|inspect|show|montre|affiche|donne|liste|list|quel(?:le)?s?|combien|config(?:uration)?|quoi de neuf)\b/i.test(lower)
|
|
871
|
-
&& !ACTION_REQUEST_PATTERN.test(lower)) {
|
|
872
|
-
return { kind: 'observe', confidence: 0.86, reason: 'status_or_explanation_request', activeRun: hasActiveRun };
|
|
873
|
-
}
|
|
874
|
-
if (hasActiveRun && /\b(ajoute|add|change|modifie|modify|remplace|replace|retire|remove|skip|ignore|plan|step|t[aâ]che)\b/i.test(lower)) {
|
|
875
|
-
return { kind: 'modify_run', confidence: 0.78, reason: 'active_run_change_request', activeRun: hasActiveRun };
|
|
876
|
-
}
|
|
877
|
-
if (hasActiveRun && /\b(lance|run|g[eé]n[eè]re|build|export|cr[eé]e|create|send|envoie|ingest|convert|importe|import)\b/i.test(lower)) {
|
|
878
|
-
return { kind: 'ambiguous', confidence: 0.45, reason: 'active_run_action_is_ambiguous', activeRun: hasActiveRun };
|
|
879
|
-
}
|
|
880
|
-
if (ACTION_REQUEST_PATTERN.test(lower)) {
|
|
881
|
-
return { kind: 'start_run', confidence: 0.8, reason: 'action_request', activeRun: hasActiveRun };
|
|
882
|
-
}
|
|
883
|
-
return { kind: 'converse', confidence: 0.62, reason: 'plain_conversation', activeRun: hasActiveRun };
|
|
884
|
-
}
|
|
885
|
-
|
|
886
929
|
function toolsForClassification(classification, writeTools, session = null) {
|
|
887
930
|
const controlTools = session?.runtime?.url
|
|
888
931
|
? [RUNTIME_STATUS_TOOL, RUNTIME_CANCEL_TOOL, RUNTIME_KILL_TOOL, RUNTIME_APPROVE_TOOL, RUNTIME_ENQUEUE_TOOL]
|
|
889
932
|
: [];
|
|
933
|
+
// Provider discovery and validation belong to the runtime. Hiding
|
|
934
|
+
// delegation while the shell snapshot is temporarily empty forced Donna
|
|
935
|
+
// to invent commands instead of submitting the objective.
|
|
890
936
|
const capabilityRunTools = session?.runtime?.url && !classification.activeRun
|
|
891
|
-
? [
|
|
937
|
+
? [RUNTIME_DELEGATE_TOOL]
|
|
892
938
|
: [];
|
|
893
|
-
if (classification.activeRun
|
|
939
|
+
if (classification.activeRun) {
|
|
894
940
|
// During an active run Donna gets read + profile + the runtime control
|
|
895
941
|
// suite: she can answer, approve, enqueue for later, soft-cancel or
|
|
896
942
|
// kill — but she must not fire new MCP jobs alongside the run (that is
|
|
897
943
|
// what runtime__enqueue is for). No canned regex answers anywhere.
|
|
898
|
-
return [SHELL_READ_COMMAND_TOOL,
|
|
944
|
+
return [SHELL_READ_COMMAND_TOOL, ...controlTools];
|
|
945
|
+
}
|
|
946
|
+
if (session?.runtime?.url) {
|
|
947
|
+
const readTools = writeTools.filter(isDonnaReadTool);
|
|
948
|
+
return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...readTools];
|
|
899
949
|
}
|
|
900
950
|
return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...writeTools];
|
|
901
951
|
}
|
|
902
952
|
|
|
953
|
+
function isDonnaReadTool(item) {
|
|
954
|
+
const name = String(item?.function?.name ?? '');
|
|
955
|
+
if (!name || name.startsWith('shell__') || name === 'wiki__plan_set' || name === 'wiki__plan_done') return false;
|
|
956
|
+
if (item?.readOnly === true) return true;
|
|
957
|
+
const tool = name.includes('__') ? name.slice(name.indexOf('__') + 2) : name;
|
|
958
|
+
return tool === 'wiki_workspace_status'
|
|
959
|
+
|| tool === 'agent_describe'
|
|
960
|
+
|| tool === 'agent_status'
|
|
961
|
+
|| /(?:^|_)(?:status|list|search|read|get)$/.test(tool);
|
|
962
|
+
}
|
|
963
|
+
|
|
964
|
+
function isReadOnlyMcpCall(session, server, tool) {
|
|
965
|
+
const descriptor = (session?.mcp?.[server]?.tools ?? [])
|
|
966
|
+
.find((item) => String(item?.name ?? '') === tool || String(item?.name ?? '').endsWith(`__${tool}`));
|
|
967
|
+
return isDonnaReadTool({
|
|
968
|
+
function: { name: `${server}__${tool}` },
|
|
969
|
+
readOnly: descriptor?.readOnly === true,
|
|
970
|
+
});
|
|
971
|
+
}
|
|
972
|
+
|
|
903
973
|
export function createAgentGraph(options = {}) {
|
|
904
974
|
async function orchestratorNode(state) {
|
|
905
975
|
const llm = state.session.llm ?? options.llm ?? null;
|
|
@@ -933,7 +1003,13 @@ export function createAgentGraph(options = {}) {
|
|
|
933
1003
|
const classification = iterations === 0
|
|
934
1004
|
? (runtimeExecution
|
|
935
1005
|
? { kind: 'execute_run', confidence: 1, reason: 'runtime_run_execution', activeRun: true }
|
|
936
|
-
:
|
|
1006
|
+
: {
|
|
1007
|
+
kind: 'agent_turn',
|
|
1008
|
+
confidence: 1,
|
|
1009
|
+
reason: 'agent_mode_llm_decision',
|
|
1010
|
+
activeRun: state.session?.agentProjection?.status === 'running'
|
|
1011
|
+
|| sessionActivities(state.session).some((activity) => !activity.terminal),
|
|
1012
|
+
})
|
|
937
1013
|
: (state.inputClassification ?? { kind: 'modify_run', confidence: 1, reason: 'tool_iteration' });
|
|
938
1014
|
if (iterations === 0) {
|
|
939
1015
|
state.session._onStep?.(`Agent: classified input as ${classification.kind}`);
|
|
@@ -962,28 +1038,54 @@ export function createAgentGraph(options = {}) {
|
|
|
962
1038
|
|
|
963
1039
|
try {
|
|
964
1040
|
const useStreamWithTools = typeof llm.streamWithTools === 'function';
|
|
965
|
-
const
|
|
1041
|
+
const toolChoice = state.forceDelegation
|
|
1042
|
+
? { type: 'function', function: { name: 'runtime__delegate' } }
|
|
1043
|
+
: 'auto';
|
|
966
1044
|
const result = useStreamWithTools
|
|
967
1045
|
? await llm.streamWithTools({
|
|
968
1046
|
system,
|
|
969
1047
|
tools,
|
|
970
1048
|
messages: conversationMessages,
|
|
971
|
-
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
},
|
|
1049
|
+
toolChoice,
|
|
1050
|
+
// Buffer until validation. Invalid commands and malformed tool
|
|
1051
|
+
// calls must never flash hundreds of lines before disappearing.
|
|
1052
|
+
onTextDelta: () => {},
|
|
976
1053
|
signal: state.session._abortSignal,
|
|
977
1054
|
})
|
|
978
1055
|
: await llm.completeWithTools({
|
|
979
1056
|
system,
|
|
980
1057
|
tools,
|
|
981
1058
|
messages: conversationMessages,
|
|
1059
|
+
toolChoice,
|
|
982
1060
|
signal: state.session._abortSignal,
|
|
983
1061
|
});
|
|
984
1062
|
|
|
985
1063
|
if (result.tool_calls?.length > 0) {
|
|
986
1064
|
state.session._onStreamReset?.();
|
|
1065
|
+
const malformed = invalidToolCalls(result.tool_calls);
|
|
1066
|
+
if (malformed.length > 0) {
|
|
1067
|
+
const retries = Number(state.invalidToolCallRetries ?? 0);
|
|
1068
|
+
if (retries < 2) {
|
|
1069
|
+
state.session._onStep?.('Agent: malformed tool call rejected; retrying…');
|
|
1070
|
+
return {
|
|
1071
|
+
pendingToolCalls: null,
|
|
1072
|
+
messages: [
|
|
1073
|
+
...(iterations === 0 ? [{ role: 'user', content: state.input }] : []),
|
|
1074
|
+
{
|
|
1075
|
+
role: 'user',
|
|
1076
|
+
content: 'Your previous tool call was incomplete or contained invalid JSON arguments. Call the appropriate available tool again with one complete valid JSON object. Do not narrate or reproduce the broken call.',
|
|
1077
|
+
},
|
|
1078
|
+
],
|
|
1079
|
+
toolIterations: iterations + 1,
|
|
1080
|
+
readyToStream: false,
|
|
1081
|
+
inputClassification: classification,
|
|
1082
|
+
invalidToolCallRetries: retries + 1,
|
|
1083
|
+
};
|
|
1084
|
+
}
|
|
1085
|
+
const failure = 'Action non exécutée : l’appel d’outil généré par le modèle était incomplet.';
|
|
1086
|
+
emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: failure });
|
|
1087
|
+
return { response: failure, pendingToolCalls: null, readyToStream: false };
|
|
1088
|
+
}
|
|
987
1089
|
// Close the streaming conversation entry now: the text streamed so
|
|
988
1090
|
// far is this iteration's narration. Without this, the next
|
|
989
1091
|
// iteration's deltas append to the SAME entry with no separator and
|
|
@@ -998,6 +1100,7 @@ export function createAgentGraph(options = {}) {
|
|
|
998
1100
|
: [result.message];
|
|
999
1101
|
return {
|
|
1000
1102
|
pendingToolCalls: result.tool_calls,
|
|
1103
|
+
allowedToolNames: tools.map((item) => item?.function?.name).filter(Boolean),
|
|
1001
1104
|
messages: newMessages,
|
|
1002
1105
|
toolIterations: iterations + 1,
|
|
1003
1106
|
readyToStream: false,
|
|
@@ -1026,6 +1129,26 @@ export function createAgentGraph(options = {}) {
|
|
|
1026
1129
|
};
|
|
1027
1130
|
}
|
|
1028
1131
|
|
|
1132
|
+
const canDelegate = tools.some((item) => item?.function?.name === 'runtime__delegate');
|
|
1133
|
+
if (!runtimeExecution && iterations === 0 && canDelegate && !state.retryWithoutTool
|
|
1134
|
+
&& await classifyRequestedAction(llm, state.input, state.session._abortSignal)) {
|
|
1135
|
+
state.session._onStreamReset?.();
|
|
1136
|
+
state.session._onStep?.('Agent: action response rejected — delegation required; retrying…');
|
|
1137
|
+
return {
|
|
1138
|
+
pendingToolCalls: null,
|
|
1139
|
+
messages: [
|
|
1140
|
+
{ role: 'user', content: state.input },
|
|
1141
|
+
result.message ?? { role: 'assistant', content: result.content ?? '' },
|
|
1142
|
+
{ role: 'user', content: 'This is an action request. Call runtime__delegate now with the original objective only. Do not provide instructions or narration.' },
|
|
1143
|
+
],
|
|
1144
|
+
toolIterations: 1,
|
|
1145
|
+
readyToStream: false,
|
|
1146
|
+
inputClassification: classification,
|
|
1147
|
+
retryWithoutTool: true,
|
|
1148
|
+
forceDelegation: true,
|
|
1149
|
+
};
|
|
1150
|
+
}
|
|
1151
|
+
|
|
1029
1152
|
if (runtimeExecution && state.retryWithoutTool) {
|
|
1030
1153
|
state.session._onStreamReset?.();
|
|
1031
1154
|
const failure = 'Action non exécutée : Donna n’a appelé aucun outil disponible. Aucun job ni résultat n’a été créé.';
|
|
@@ -1038,7 +1161,43 @@ export function createAgentGraph(options = {}) {
|
|
|
1038
1161
|
};
|
|
1039
1162
|
}
|
|
1040
1163
|
|
|
1164
|
+
const invalidCommands = invalidSuggestedSlashCommands(result.content, state.session);
|
|
1165
|
+
const leakedTools = invalidUserFacingToolNames(result.content, state.session);
|
|
1166
|
+
if (invalidCommands.length > 0 || leakedTools.length > 0) {
|
|
1167
|
+
state.session._onStreamReset?.();
|
|
1168
|
+
const retries = Number(state.invalidResponseRetries ?? 0);
|
|
1169
|
+
if (retries < 2) {
|
|
1170
|
+
const canDelegate = tools.some((item) => item?.function?.name === 'runtime__delegate');
|
|
1171
|
+
state.session._onStep?.('Agent: invalid user-facing implementation detail rejected; retrying…');
|
|
1172
|
+
return {
|
|
1173
|
+
pendingToolCalls: null,
|
|
1174
|
+
messages: [
|
|
1175
|
+
...(iterations === 0 ? [{ role: 'user', content: state.input }] : []),
|
|
1176
|
+
result.message ?? { role: 'assistant', content: result.content ?? '' },
|
|
1177
|
+
{
|
|
1178
|
+
role: 'user',
|
|
1179
|
+
content: [
|
|
1180
|
+
'Rewrite the answer for the end user without internal MCP tool identifiers or unsolicited shell commands.',
|
|
1181
|
+
invalidCommands.length > 0 ? `Unavailable slash commands: /${invalidCommands.join(', /')}.` : null,
|
|
1182
|
+
leakedTools.length > 0 ? 'Do not print tool names; use them internally if needed.' : null,
|
|
1183
|
+
'If the user requested an action and runtime delegation is available, call runtime__delegate instead of giving manual instructions.',
|
|
1184
|
+
].filter(Boolean).join(' '),
|
|
1185
|
+
},
|
|
1186
|
+
],
|
|
1187
|
+
toolIterations: iterations + 1,
|
|
1188
|
+
readyToStream: false,
|
|
1189
|
+
inputClassification: classification,
|
|
1190
|
+
invalidResponseRetries: retries + 1,
|
|
1191
|
+
forceDelegation: canDelegate,
|
|
1192
|
+
};
|
|
1193
|
+
}
|
|
1194
|
+
const failure = 'Réponse rejetée : Donna a exposé une instruction interne ou une procédure manuelle incorrecte.';
|
|
1195
|
+
emitAgentEvent(state.session, 'assistant_message', 'agent_guard', { content: failure });
|
|
1196
|
+
return { response: failure, pendingToolCalls: null, readyToStream: false };
|
|
1197
|
+
}
|
|
1198
|
+
|
|
1041
1199
|
if (useStreamWithTools) {
|
|
1200
|
+
if (result.content) state.session._onStream?.(result.content);
|
|
1042
1201
|
emitAgentEvent(state.session, 'assistant_message', 'llm', { content: result.content ?? '' });
|
|
1043
1202
|
// Text was streamed inline via session._onStream — no second LLM call needed.
|
|
1044
1203
|
const newMessages = iterations === 0
|
|
@@ -1088,6 +1247,26 @@ export function createAgentGraph(options = {}) {
|
|
|
1088
1247
|
const isInternalWikiTool = server === 'wiki' && (tool === 'plan_set' || tool === 'plan_done');
|
|
1089
1248
|
const serverLabel = server === 'shell' ? 'Shell' : isInternalWikiTool ? 'Plan' : 'MCP';
|
|
1090
1249
|
const toolName = server ? `${server}.${tool}` : call.function.name;
|
|
1250
|
+
// Hard guardrail: only execute tools that were actually offered this turn
|
|
1251
|
+
// (read-only tools + runtime controls + delegate). A capable model that
|
|
1252
|
+
// spots a mutating provider tool in the prompt and calls it directly must
|
|
1253
|
+
// be refused and steered back to runtime__delegate — this is what keeps
|
|
1254
|
+
// the orchestration capability-driven regardless of model strength.
|
|
1255
|
+
// Only interactive turns are constrained. Inside a runtime run
|
|
1256
|
+
// (_currentRunIdentity set) the graph legitimately executes the
|
|
1257
|
+
// already-validated, already-approved delegated task via provider tools.
|
|
1258
|
+
const runtimeExecutionTurn = Boolean(state.session._currentRunIdentity);
|
|
1259
|
+
const allowedNames = !runtimeExecutionTurn && Array.isArray(state.allowedToolNames) ? state.allowedToolNames : null;
|
|
1260
|
+
const isInternalCall = server === 'shell' || server === 'runtime' || isInternalWikiTool;
|
|
1261
|
+
if (allowedNames && server && !isInternalCall && !allowedNames.includes(`${server}__${tool}`)) {
|
|
1262
|
+
const refusal = `${server}__${tool} is not available in interactive mode. Do not call provider tools directly. For any action or mutation, call runtime__delegate with the user objective; only read-only tools and runtime controls may be called directly.`;
|
|
1263
|
+
state.session._onStep?.(`tool call refused (not offered): ${server}__${tool}`);
|
|
1264
|
+
emitAgentEvent(state.session, 'tool_call_result', 'tool', {
|
|
1265
|
+
callId: call.id, name: toolName, ok: false, result: refusal, summary: 'refused',
|
|
1266
|
+
});
|
|
1267
|
+
toolResultMessages.push({ role: 'tool', tool_call_id: call.id, content: refusal });
|
|
1268
|
+
continue;
|
|
1269
|
+
}
|
|
1091
1270
|
if (resolved.normalized) {
|
|
1092
1271
|
// Keep normalizations visible: the defensive routing must not hide
|
|
1093
1272
|
// prompt/skill regressions that reintroduce unqualified names.
|
|
@@ -1104,9 +1283,11 @@ export function createAgentGraph(options = {}) {
|
|
|
1104
1283
|
args: call.function.arguments ?? '{}',
|
|
1105
1284
|
summary: argsSummary || 'calling...',
|
|
1106
1285
|
});
|
|
1107
|
-
//
|
|
1286
|
+
// A plan represents work, never observation. Read-only inventory/status
|
|
1287
|
+
// calls stay out of Plan even when Donna uses them to answer a question.
|
|
1108
1288
|
let minimalPlanActive = false;
|
|
1109
|
-
if (!isInternalWikiTool && server !== 'shell' &&
|
|
1289
|
+
if (!isInternalWikiTool && server !== 'shell' && server !== 'runtime'
|
|
1290
|
+
&& !isReadOnlyMcpCall(state.session, server, tool) && !state.session.headlessPlan) {
|
|
1110
1291
|
minimalPlanActive = true;
|
|
1111
1292
|
emitAgentEvent(state.session, 'plan_set', 'tool', {
|
|
1112
1293
|
steps: [{ step: 1, id: null, description: toolName, status: 'running', _activityKey: null }],
|
|
@@ -1128,6 +1309,8 @@ export function createAgentGraph(options = {}) {
|
|
|
1128
1309
|
);
|
|
1129
1310
|
}
|
|
1130
1311
|
let args = JSON.parse(call.function.arguments ?? '{}');
|
|
1312
|
+
const definition = toolDefinitionForCall(state.session, call.function.name);
|
|
1313
|
+
args = normalizeToolArgumentsFromSchema(args, definition?.function?.parameters);
|
|
1131
1314
|
if (server === 'production' && tool === 'production_start_job' && state.session.workspace && !args.callerLabel) {
|
|
1132
1315
|
args = { ...args, callerLabel: `${state.session.workspace}/wiki-manager` };
|
|
1133
1316
|
}
|
|
@@ -1244,12 +1427,17 @@ export function createAgentGraph(options = {}) {
|
|
|
1244
1427
|
return {
|
|
1245
1428
|
messages: toolResultMessages,
|
|
1246
1429
|
pendingToolCalls: null,
|
|
1430
|
+
forceDelegation: false,
|
|
1431
|
+
invalidToolCallRetries: 0,
|
|
1432
|
+
invalidResponseRetries: 0,
|
|
1247
1433
|
};
|
|
1248
1434
|
}
|
|
1249
1435
|
|
|
1250
1436
|
function routeOrchestrator(state) {
|
|
1251
1437
|
if (state.pendingToolCalls?.length > 0) return 'tool_executor';
|
|
1252
1438
|
if (state.retryWithoutTool) return 'orchestrator';
|
|
1439
|
+
if (state.invalidToolCallRetries > 0 && state.response == null && !state.streamedInline) return 'orchestrator';
|
|
1440
|
+
if (state.invalidResponseRetries > 0 && state.response == null && !state.streamedInline) return 'orchestrator';
|
|
1253
1441
|
return END;
|
|
1254
1442
|
}
|
|
1255
1443
|
|