@dotdrelle/wiki-manager 0.12.0 → 0.12.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/src/activity/activityAggregator.js +34 -3
- package/src/activity/activityAggregator.test.js +32 -0
- package/src/agent/graph.js +307 -29
- package/src/agent/graph.test.js +404 -1
- package/src/cli/wiki-manager.js +24 -1
- package/src/commands/slash.js +74 -1
- package/src/commands/slash.test.js +36 -0
- package/src/contracts/schemas.test.js +2 -2
- package/src/core/activity.js +4 -0
- package/src/core/activity.test.js +9 -1
- package/src/core/agentEvents.js +30 -0
- package/src/core/agentEvents.test.js +42 -1
- package/src/core/buildInfo.js +58 -0
- package/src/core/buildInfo.json +4 -0
- package/src/core/buildInfo.test.js +14 -0
- package/src/core/mcp.js +50 -3
- package/src/core/mcp.test.js +94 -1
- package/src/core/runtimeLog.js +6 -1
- package/src/core/runtimeLog.test.js +3 -1
- package/src/runtime/client.js +22 -0
- package/src/runtime/controlMessages.js +43 -0
- package/src/runtime/controlMessages.test.js +21 -0
- package/src/runtime/donna-contract.test.js +3 -3
- package/src/runtime/recoveryManager.js +54 -0
- package/src/runtime/recoveryManager.test.js +73 -0
- package/src/runtime/runner.js +49 -13
- package/src/runtime/runner.test.js +196 -0
- package/src/runtime/server.js +73 -13
- package/src/runtime/server.test.js +284 -67
- package/src/runtime/store.js +48 -3
- package/src/runtime/store.test.js +56 -24
- package/src/runtime/supervisor.js +77 -1
- package/src/shell/RightPane.tsx +124 -31
- package/src/shell/SetupWizard.tsx +13 -1
- package/src/shell/repl.js +81 -11
- package/src/shell/repl.test.js +115 -2
- package/src/shell/tui.tsx +4 -1
- package/src/shell/useAgent.ts +37 -4
- package/src/shell/useSession.ts +51 -10
package/src/agent/graph.js
CHANGED
|
@@ -5,7 +5,8 @@ import {
|
|
|
5
5
|
callMcpTool,
|
|
6
6
|
formatMcpToolResult,
|
|
7
7
|
formatMcpToolsForAgent,
|
|
8
|
-
|
|
8
|
+
resolveToolCallName,
|
|
9
|
+
truncateToolResult,
|
|
9
10
|
} from '../core/mcp.js';
|
|
10
11
|
import { formatSkillsForAgent, readOptionalText } from '../core/skills.js';
|
|
11
12
|
import { handleSlashCommand } from '../commands/slash.js';
|
|
@@ -13,11 +14,22 @@ import { extractActivity, formatActivitySummary, parseJsonText, sessionActivitie
|
|
|
13
14
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
14
15
|
import { enqueueProductionJob, ensureJobQueue, formatQueue, productionLockBusy } from '../core/jobQueue.js';
|
|
15
16
|
import { updateWorkspaceProfilePreference } from '../core/profile.js';
|
|
17
|
+
import { createCapabilityRegistry } from '../orchestrator/capabilityRegistry.js';
|
|
18
|
+
import { fetchRuntimeState, postRuntimeCancel, postRuntimeControl, postRuntimeKill } from '../runtime/client.js';
|
|
16
19
|
|
|
17
20
|
const MAX_TOOL_ITERATIONS = 80;
|
|
18
21
|
const MAX_SPINNER_ARG_LENGTH = 96;
|
|
19
22
|
const MAX_PROFILE_CHARS = 4000;
|
|
20
23
|
|
|
24
|
+
// Pseudo-servers handled directly by the tool executor (not present in
|
|
25
|
+
// session.mcp). Listed so unqualified names like "plan_set" resolve the same
|
|
26
|
+
// way as MCP tools in resolveToolCallName.
|
|
27
|
+
const INTERNAL_TOOL_SERVERS = {
|
|
28
|
+
wiki: ['plan_set', 'plan_done'],
|
|
29
|
+
shell: ['run_command', 'read_command', 'profile_update'],
|
|
30
|
+
runtime: ['kill', 'cancel', 'status', 'approve', 'enqueue'],
|
|
31
|
+
};
|
|
32
|
+
|
|
21
33
|
const AGENT_SLASH_COMMANDS = new Set([
|
|
22
34
|
'help',
|
|
23
35
|
'version',
|
|
@@ -100,6 +112,59 @@ const SHELL_PROFILE_UPDATE_TOOL = {
|
|
|
100
112
|
},
|
|
101
113
|
};
|
|
102
114
|
|
|
115
|
+
// Runtime control tools: Donna interprets the user's intent ("supprime le
|
|
116
|
+
// job et la queue", "arrête tout", "où en est le run") and ACTS through
|
|
117
|
+
// these, instead of a hardcoded regex classifier answering with canned text.
|
|
118
|
+
const RUNTIME_KILL_TOOL = {
|
|
119
|
+
type: 'function',
|
|
120
|
+
function: {
|
|
121
|
+
name: 'runtime__kill',
|
|
122
|
+
description: 'Hard-stop the workspace runtime: abort the active run, cancel its agent jobs, mark persisted runs interrupted and purge the control queue. Use when the user asks to remove/kill/clean the current run, its jobs or the queue.',
|
|
123
|
+
parameters: { type: 'object', additionalProperties: false, properties: { runId: { type: 'string', description: 'Optional specific run id; omit to kill everything active in the workspace.' } } },
|
|
124
|
+
},
|
|
125
|
+
};
|
|
126
|
+
|
|
127
|
+
const RUNTIME_CANCEL_TOOL = {
|
|
128
|
+
type: 'function',
|
|
129
|
+
function: {
|
|
130
|
+
name: 'runtime__cancel',
|
|
131
|
+
description: 'Soft-cancel the active runtime run (graceful abort, no queue purge). Use for "annule le run" when the user does not ask to clean everything.',
|
|
132
|
+
parameters: { type: 'object', additionalProperties: false, properties: {} },
|
|
133
|
+
},
|
|
134
|
+
};
|
|
135
|
+
|
|
136
|
+
const RUNTIME_STATUS_TOOL = {
|
|
137
|
+
type: 'function',
|
|
138
|
+
function: {
|
|
139
|
+
name: 'runtime__status',
|
|
140
|
+
description: 'Read the runtime state: active run, plan steps, queue items, approvals. Use to answer questions about what is currently running or queued.',
|
|
141
|
+
parameters: { type: 'object', additionalProperties: false, properties: {} },
|
|
142
|
+
},
|
|
143
|
+
};
|
|
144
|
+
|
|
145
|
+
const RUNTIME_APPROVE_TOOL = {
|
|
146
|
+
type: 'function',
|
|
147
|
+
function: {
|
|
148
|
+
name: 'runtime__approve',
|
|
149
|
+
description: 'Grant the pending approval of the active runtime run (mutating tasks wait on it). Use when the user consents in ANY phrasing: "vas-y", "ok pour l\'export", "approuve", "valide". Confirm what was approved.',
|
|
150
|
+
parameters: { type: 'object', additionalProperties: false, properties: {} },
|
|
151
|
+
},
|
|
152
|
+
};
|
|
153
|
+
|
|
154
|
+
const RUNTIME_ENQUEUE_TOOL = {
|
|
155
|
+
type: 'function',
|
|
156
|
+
function: {
|
|
157
|
+
name: 'runtime__enqueue',
|
|
158
|
+
description: 'Queue a request to run AFTER the currently active runtime run finishes. Use when the user asks for a new action while a run is active and wants it done afterwards.',
|
|
159
|
+
parameters: {
|
|
160
|
+
type: 'object',
|
|
161
|
+
additionalProperties: false,
|
|
162
|
+
properties: { input: { type: 'string', description: 'The request to execute after the current run, phrased as a complete instruction.' } },
|
|
163
|
+
required: ['input'],
|
|
164
|
+
},
|
|
165
|
+
},
|
|
166
|
+
};
|
|
167
|
+
|
|
103
168
|
const WIKI_PLAN_SET_TOOL = {
|
|
104
169
|
type: 'function',
|
|
105
170
|
function: {
|
|
@@ -128,6 +193,8 @@ const WIKI_PLAN_SET_TOOL = {
|
|
|
128
193
|
status: { type: 'string', enum: ['pending', 'queued', 'running', 'waiting', 'pending_approval', 'done', 'failed', 'cancelled', 'stalled', 'added_during_run'] },
|
|
129
194
|
dependsOn: { type: 'array', items: { type: 'string' } },
|
|
130
195
|
outputRefs: { type: 'array', items: { type: 'string' } },
|
|
196
|
+
operation: { type: ['string', 'null'], description: 'Operation for the capability provider (e.g. ingest_plan, build).' },
|
|
197
|
+
arguments: { type: 'object', description: 'Arguments passed to the provider agent_execute for this step.' },
|
|
131
198
|
},
|
|
132
199
|
required: ['description'],
|
|
133
200
|
},
|
|
@@ -440,9 +507,87 @@ function emitAgentEvent(session, type, origin, payload = {}) {
|
|
|
440
507
|
dispatchAgentEvent(session, createAgentEvent(type, { origin, payload }));
|
|
441
508
|
}
|
|
442
509
|
|
|
510
|
+
// Capability ids currently provided by discovered, orchestrable, healthy
|
|
511
|
+
// agents. This is the live registry the dispatcher will resolve against —
|
|
512
|
+
// a plan declaring anything outside this set can only stall forever.
|
|
513
|
+
export function knownCapabilityIds(session) {
|
|
514
|
+
const registry = session?.capabilityRegistry ?? createCapabilityRegistry({
|
|
515
|
+
agents: session?.agentRegistrySnapshot ?? session?.agents ?? [],
|
|
516
|
+
});
|
|
517
|
+
const snapshot = typeof registry.snapshot === 'function' ? registry.snapshot() : registry;
|
|
518
|
+
return [...new Set(Object.keys(snapshot ?? {}).map((key) => {
|
|
519
|
+
const index = key.lastIndexOf('@');
|
|
520
|
+
return index > 0 ? key.slice(0, index) : key;
|
|
521
|
+
}))].sort();
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
async function handleRuntimeControlTool(session, tool, args = {}) {
|
|
525
|
+
const url = session.runtime?.url ?? null;
|
|
526
|
+
if (!url) return 'Runtime not connected: no runtime URL available in this session.';
|
|
527
|
+
const workspace = session.workspace ?? null;
|
|
528
|
+
try {
|
|
529
|
+
if (tool === 'kill') {
|
|
530
|
+
const result = await postRuntimeKill({ url, workspace, runId: args.runId ?? null });
|
|
531
|
+
return `Runtime killed: ${result.runs ?? 0} run(s) interrupted, ${result.tasks ?? 0} task(s) cancelled, ${result.queued ?? 0} queued control request(s) purged.`;
|
|
532
|
+
}
|
|
533
|
+
if (tool === 'cancel') {
|
|
534
|
+
const result = await postRuntimeCancel({ url, workspace });
|
|
535
|
+
return result.cancelled ? 'Runtime run cancellation requested.' : `No active run to cancel${result.reason ? ` (${result.reason})` : ''}.`;
|
|
536
|
+
}
|
|
537
|
+
if (tool === 'approve') {
|
|
538
|
+
const result = await postRuntimeControl('message', { url, workspace, input: 'approve', intent: 'approve' });
|
|
539
|
+
return String(result?.explanation ?? (result?.accepted ? 'Approval granted.' : 'No pending approval found.'));
|
|
540
|
+
}
|
|
541
|
+
if (tool === 'enqueue') {
|
|
542
|
+
const result = await postRuntimeControl('message', { url, workspace, input: String(args.input ?? ''), intent: 'enqueue' });
|
|
543
|
+
return String(result?.explanation ?? 'Request queued for after the current run.');
|
|
544
|
+
}
|
|
545
|
+
if (tool === 'status') {
|
|
546
|
+
const state = await fetchRuntimeState({ url, workspace });
|
|
547
|
+
const plan = Array.isArray(state?.plan) ? state.plan : [];
|
|
548
|
+
const queue = Array.isArray(state?.queue) ? state.queue : [];
|
|
549
|
+
const controlQueue = Array.isArray(state?.controlQueue) ? state.controlQueue : [];
|
|
550
|
+
return JSON.stringify({
|
|
551
|
+
status: state?.status ?? 'unknown',
|
|
552
|
+
running: Boolean(state?.running),
|
|
553
|
+
runId: state?.runId ?? null,
|
|
554
|
+
plan: plan.map((step) => ({ id: step.id ?? step.step, description: step.description, status: step.status })),
|
|
555
|
+
queue: queue.map((item) => ({ id: item.id, status: item.status, tool: item.tool ?? item.type ?? null })),
|
|
556
|
+
controlQueue: controlQueue.filter((item) => item.status === 'queued').map((item) => ({ id: item.id, input: item.input })),
|
|
557
|
+
pendingApprovals: (Array.isArray(state?.approvals) ? state.approvals : [])
|
|
558
|
+
.filter((approval) => approval.status === 'pending_approval')
|
|
559
|
+
.map((approval) => ({ id: approval.id, reason: approval.reason ?? null })),
|
|
560
|
+
}, null, 2);
|
|
561
|
+
}
|
|
562
|
+
return `Unknown runtime tool: ${tool}`;
|
|
563
|
+
} catch (err) {
|
|
564
|
+
return `Runtime control error (${tool}): ${err instanceof Error ? err.message : String(err)}`;
|
|
565
|
+
}
|
|
566
|
+
}
|
|
567
|
+
|
|
443
568
|
function handleWikiTool(session, tool, args) {
|
|
444
569
|
if (tool === 'plan_set') {
|
|
445
570
|
const steps = Array.isArray(args.steps) ? args.steps : [];
|
|
571
|
+
// Reject fantasy capabilities BEFORE the plan exists: once registered,
|
|
572
|
+
// unresolvable steps become tasks that wait forever and flood the queue.
|
|
573
|
+
// A tool-level error (not an exception) lets the LLM correct itself in
|
|
574
|
+
// the same turn. An empty registry (discovery not done yet) skips the
|
|
575
|
+
// check rather than blocking legitimate early plans.
|
|
576
|
+
const known = knownCapabilityIds(session);
|
|
577
|
+
if (known.length > 0) {
|
|
578
|
+
const unknown = [...new Set(steps
|
|
579
|
+
.map((step) => (step && typeof step === 'object' ? step.requiredCapability : null))
|
|
580
|
+
.filter(Boolean)
|
|
581
|
+
.map(String)
|
|
582
|
+
.filter((capability) => !known.includes(capability.includes('@') ? capability.slice(0, capability.lastIndexOf('@')) : capability)))];
|
|
583
|
+
if (unknown.length > 0) {
|
|
584
|
+
return `Plan rejected: unknown capabilities [${unknown.join(', ')}]. `
|
|
585
|
+
+ `Available capabilities: ${known.join(', ')}. `
|
|
586
|
+
+ 'Redeclare the plan using only available capabilities, or use requiredCapability: null for a step you execute yourself.';
|
|
587
|
+
}
|
|
588
|
+
} else if (steps.some((step) => step && typeof step === 'object' && step.requiredCapability)) {
|
|
589
|
+
session._onStep?.('plan_set: capability registry empty, validation skipped');
|
|
590
|
+
}
|
|
446
591
|
emitAgentEvent(session, 'plan_set', 'tool', {
|
|
447
592
|
steps: steps.map((raw, i) => normalizeDeclaredPlanStep(raw, i, session)),
|
|
448
593
|
});
|
|
@@ -475,6 +620,21 @@ function normalizeDeclaredPlanStep(raw, index) {
|
|
|
475
620
|
executor: null,
|
|
476
621
|
executorQuery: null,
|
|
477
622
|
outputRefs: Array.isArray(item.outputRefs) ? item.outputRefs.map(String) : [],
|
|
623
|
+
// Execution fields the deterministic dispatcher consumes (agent_execute):
|
|
624
|
+
// without them a capability step cannot actually run.
|
|
625
|
+
...(item.operation != null ? { operation: String(item.operation) } : {}),
|
|
626
|
+
...(item.arguments && typeof item.arguments === 'object' ? { arguments: item.arguments } : {}),
|
|
627
|
+
...(item.groupId != null ? { groupId: String(item.groupId) } : {}),
|
|
628
|
+
...(item.dependsOnGroup != null ? { dependsOnGroup: String(item.dependsOnGroup) } : {}),
|
|
629
|
+
...(item.parallelizable != null ? { parallelizable: Boolean(item.parallelizable) } : {}),
|
|
630
|
+
...(item.barrier ? { barrier: true } : {}),
|
|
631
|
+
...(item.locks ? { locks: item.locks } : {}),
|
|
632
|
+
...(item.requiresApproval != null ? { requiresApproval: Boolean(item.requiresApproval) } : {}),
|
|
633
|
+
...(item.approvalClass ? { approvalClass: String(item.approvalClass) } : {}),
|
|
634
|
+
...(item.approvalSummary ? { approvalSummary: String(item.approvalSummary) } : {}),
|
|
635
|
+
...(item.idempotencyKey ? { idempotencyKey: String(item.idempotencyKey) } : {}),
|
|
636
|
+
...(item.progressWeight != null ? { progressWeight: Number(item.progressWeight) } : {}),
|
|
637
|
+
...(item.recommendedConcurrency != null ? { recommendedConcurrency: Number(item.recommendedConcurrency) } : {}),
|
|
478
638
|
};
|
|
479
639
|
}
|
|
480
640
|
|
|
@@ -539,9 +699,16 @@ export function buildAgentSystemPrompt(state) {
|
|
|
539
699
|
'Prefer MCP tools that declare their own plan via _activity.plan.steps — when such a tool returns _activity, the shell creates and tracks the plan automatically without requiring wiki__plan_set.',
|
|
540
700
|
'Use wiki__plan_set when the MCP tool cannot declare its own plan or when the task spans multiple independent tools (e.g. CME export then email report). For a single self-describing async job, wiki__plan_set is optional.',
|
|
541
701
|
'',
|
|
702
|
+
(() => {
|
|
703
|
+
const capabilityIds = knownCapabilityIds(state.session);
|
|
704
|
+
return capabilityIds.length > 0
|
|
705
|
+
? `Known orchestration capabilities — the ONLY values allowed in requiredCapability: ${capabilityIds.join(', ')}. Never invent capability names; a plan declaring an unknown capability will be rejected. A step you execute yourself directly takes requiredCapability: null.`
|
|
706
|
+
: 'No orchestration capabilities discovered yet: declare plan steps with requiredCapability: null and execute them yourself with the connected MCP tools.';
|
|
707
|
+
})(),
|
|
708
|
+
'',
|
|
542
709
|
'Task startup:',
|
|
543
710
|
' 1. If the next MCP tool returns _activity.plan.steps, call that tool directly; the shell will create the visible plan from the returned activity.',
|
|
544
|
-
' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, dependsOn, outputRefs}; a legacy list of strings is still accepted.',
|
|
711
|
+
' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, operation, arguments, dependsOn, outputRefs}; capability steps need operation+arguments for the dispatcher to execute them; a legacy list of strings is still accepted.',
|
|
545
712
|
' Multi-tool example: wiki__plan_set(steps=[{id:"cme-export",description:"CME export",requiredCapability:"external-source.export",dependsOn:[],outputRefs:["raw/untracked"]},{id:"production",description:"Production pipeline",requiredCapability:"knowledge.pipeline",dependsOn:["cme-export"],outputRefs:["deliverables"]}])',
|
|
546
713
|
' 3. Immediately execute the first step using the appropriate MCP tool. Do not start step 2 in the same turn unless one async pipeline tool owns and declares the whole sequence.',
|
|
547
714
|
' For synchronous steps (result is immediate, no _activity polling), call wiki__plan_done(step=1) after confirming success.',
|
|
@@ -561,17 +728,21 @@ export function buildAgentSystemPrompt(state) {
|
|
|
561
728
|
'On failure: if a completed activity is failed/error/cancelled, call wiki__plan_done(step=N, status="failed") then stop with a clear error report.',
|
|
562
729
|
].filter(Boolean).join('\n'),
|
|
563
730
|
'For service actions, recommend only available service primitives from Available primitives, with the exact service name when the primitive supports one.',
|
|
731
|
+
'Scope discipline: execute ONLY the action(s) the user explicitly requested. Never chain additional mutating operations (ingest, build, export, polish, delete, send…) that the user did not ask for — even when diagnostics or recommendations suggest them. Finish the requested work, then list the suggested follow-ups in your final answer and stop. Example: "applique les recommandations de config" means apply the config; it does NOT authorize launching the ingest those recommendations mention.',
|
|
564
732
|
'Disambiguate export requests carefully.',
|
|
565
|
-
'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`
|
|
566
|
-
'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`
|
|
567
|
-
'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts.
|
|
733
|
+
'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`cme__cme_export_run`, then `cme__cme_export_status`). Never use production `type=export` for Confluence source export.',
|
|
734
|
+
'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`production__production_start_job` with `type:"export"` or pipeline steps). Require the deliverable path when exporting deliverables.',
|
|
735
|
+
'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts.',
|
|
736
|
+
'MULTI-DOCUMENT ingest (more than 2 files, or "ingest everything pending"): call production__agent_plan first, e.g. {capability:"knowledge.update", operation:"ingest", constraints:{maxConcurrency:3, requireApprovalForMutations:true}}. The shell integrates the returned task graph as the plan automatically and the orchestrator dispatches the per-document tasks IN PARALLEL with an approval gate. Do not call production__production_start_job for multi-document ingest — that creates one monolithic sequential job.',
|
|
737
|
+
'Single-document ingest or one-off jobs (doctor, one build, one export): production__production_start_job is fine. To chain sequential steps (e.g. build then polish) use ONE call with type="pipeline" and steps=["build","polish"] — never separate jobs (the first is asynchronous). For existing deliverables where content stability matters, pass stabilize:true. Do not ask the user to confirm between steps.',
|
|
568
738
|
'Long-running MCP jobs: do not call the same status tool more than once consecutively. When chaining jobs sequentially: (1) start the job, report job/activity id and status; (2) check status once — if done, proceed to the next step immediately; (3) if still running, report status, list the remaining steps, and return control; (4) when re-invoked, check status first, then proceed. Do not spin-poll (status → status → status with no new action between). The shell activity panel monitors non-terminal jobs automatically.',
|
|
569
|
-
'If
|
|
570
|
-
'For diagnostics, use /wiki run doctor when the
|
|
739
|
+
'If production__production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
|
|
740
|
+
'For diagnostics (doctor), use production__production_start_job with type="doctor" like any other production job; /wiki run doctor is only the fallback when the production MCP is not connected. Use /workspace init <name> [path] for low-level non-interactive workspace creation. In the interactive TUI, /new <name> opens the setup wizard. Use /wiki for index, or /wiki run index through the explicit backup hatch. Use /wiki run init only for explicit current-workspace llm-wiki init.',
|
|
571
741
|
'If an action requires tools or skills not available yet, explain the limitation and name the expected primitive.',
|
|
572
742
|
workspaceProfile
|
|
573
743
|
? `Workspace profile (.wiki/profile.md) — durable user preferences, apply these to every reply (tone, tutoiement/vouvoiement, formatting, etc.):\n${workspaceProfile}`
|
|
574
744
|
: null,
|
|
745
|
+
'Runtime control: you have runtime__status, runtime__cancel, runtime__kill, runtime__approve and runtime__enqueue. When the user asks to stop, remove, clean or kill the current run, its jobs or the queue ("supprime le job et la queue", "arr\u00eate tout"), call runtime__kill (or runtime__cancel for a soft stop of just the run) and confirm what was stopped. For questions about what is running or queued, call runtime__status and answer from its data. When the user consents to a pending approval in any phrasing ("vas-y", "ok pour l\'export"), call runtime__approve. When the user asks for a NEW action while a run is active, do not execute it: propose runtime__enqueue (run it after) or, if they insist it replaces the current work, runtime__kill then the new action.',
|
|
575
746
|
'When the user explicitly asks you to remember, persist, or update durable preference/profile information, call wiki__profile_update when it is available; otherwise call shell__profile_update. Do not just acknowledge in text without calling a profile update tool.',
|
|
576
747
|
].filter(Boolean).join('\n');
|
|
577
748
|
|
|
@@ -602,20 +773,35 @@ export function formatLlmUnavailableMessage(reason) {
|
|
|
602
773
|
return `⚠ LLM injoignable : ${clean || 'raison inconnue'}`;
|
|
603
774
|
}
|
|
604
775
|
|
|
605
|
-
|
|
776
|
+
// Verbs that clearly request work (a runtime run), in French and English.
|
|
777
|
+
// "configure/configurer" is an action; the nouns "config/configuration" are
|
|
778
|
+
// NOT matched here — asking for a config is an observe request.
|
|
779
|
+
const ACTION_REQUEST_PATTERN = /\b(lance|relance|d[eé]marre|start|ex[eé]cute|execute|g[eé]n[eè]re|generate|build|construis|exporte?|ingest\w*|ing[eè]re|importe?|convert(?:is|it|s)?|cr[eé]e|create|polish|publie|publish|d[eé]ploie|deploy|envoie|send|configure[rsz]?|setup|installe|update|mets? [aà] jour|supprime|delete|efface|nettoie|clean|r[eé]pare|fix|corrige)\b/i;
|
|
780
|
+
|
|
781
|
+
// Explicit explanation/question markers dominate action verbs: "explique le
|
|
782
|
+
// build" is a question about the build, not a request to build.
|
|
783
|
+
const EXPLANATION_REQUEST_PATTERN = /\b(explique|explain|pourquoi|why|comment|how|c'est quoi|qu'est[- ]ce)\b/i;
|
|
784
|
+
|
|
785
|
+
export function classifyAgentInput(input, session) {
|
|
606
786
|
const lower = String(input ?? '').toLowerCase();
|
|
607
787
|
const hasActiveRun = session?.agentProjection?.status === 'running'
|
|
608
788
|
|| sessionActivities(session).some((activity) => !activity.terminal);
|
|
609
789
|
if (/\b(valide tout|approve all|approve|approuve|valid[eé]|ok pour tout|go pour tout)\b/i.test(lower)) {
|
|
610
790
|
return { kind: 'approve', confidence: 0.86, reason: 'approval_request', activeRun: hasActiveRun };
|
|
611
791
|
}
|
|
612
|
-
if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort)\b/i.test(lower)) {
|
|
792
|
+
if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort|supprime|kill|tue|purge|vide la (file|queue)|nettoie la (file|queue))\b/i.test(lower)) {
|
|
613
793
|
return { kind: 'cancel', confidence: 0.86, reason: 'cancel_request', activeRun: hasActiveRun };
|
|
614
794
|
}
|
|
615
795
|
if (/\b(plus tard|later|ensuite|apr[eè]s ce run|enqueue|mets en file|met en file|futur|next run|future run)\b/i.test(lower)) {
|
|
616
796
|
return { kind: 'enqueue_run', confidence: 0.8, reason: 'future_run_request', activeRun: hasActiveRun };
|
|
617
797
|
}
|
|
618
|
-
if (
|
|
798
|
+
if (EXPLANATION_REQUEST_PATTERN.test(lower)) {
|
|
799
|
+
return { kind: 'observe', confidence: 0.86, reason: 'explanation_request', activeRun: hasActiveRun };
|
|
800
|
+
}
|
|
801
|
+
// Observe markers only win when no action verb is present: "où en est le
|
|
802
|
+
// run" is observe, "lance le run" is an action request.
|
|
803
|
+
if (/\b(o[uù] en es[t-]|status|statut|progress|progression|run|job|queue|logs?|inspect|show|montre|affiche|donne|liste|list|quel(?:le)?s?|combien|config(?:uration)?|quoi de neuf)\b/i.test(lower)
|
|
804
|
+
&& !ACTION_REQUEST_PATTERN.test(lower)) {
|
|
619
805
|
return { kind: 'observe', confidence: 0.86, reason: 'status_or_explanation_request', activeRun: hasActiveRun };
|
|
620
806
|
}
|
|
621
807
|
if (hasActiveRun && /\b(ajoute|add|change|modifie|modify|remplace|replace|retire|remove|skip|ignore|plan|step|t[aâ]che)\b/i.test(lower)) {
|
|
@@ -624,12 +810,24 @@ function classifyAgentInput(input, session) {
|
|
|
624
810
|
if (hasActiveRun && /\b(lance|run|g[eé]n[eè]re|build|export|cr[eé]e|create|send|envoie|ingest|convert|importe|import)\b/i.test(lower)) {
|
|
625
811
|
return { kind: 'ambiguous', confidence: 0.45, reason: 'active_run_action_is_ambiguous', activeRun: hasActiveRun };
|
|
626
812
|
}
|
|
813
|
+
if (ACTION_REQUEST_PATTERN.test(lower)) {
|
|
814
|
+
return { kind: 'start_run', confidence: 0.8, reason: 'action_request', activeRun: hasActiveRun };
|
|
815
|
+
}
|
|
627
816
|
return { kind: 'converse', confidence: 0.62, reason: 'plain_conversation', activeRun: hasActiveRun };
|
|
628
817
|
}
|
|
629
818
|
|
|
630
|
-
function toolsForClassification(classification, writeTools) {
|
|
631
|
-
|
|
632
|
-
|
|
819
|
+
function toolsForClassification(classification, writeTools, session = null) {
|
|
820
|
+
const controlTools = session?.runtime?.url
|
|
821
|
+
? [RUNTIME_STATUS_TOOL, RUNTIME_CANCEL_TOOL, RUNTIME_KILL_TOOL, RUNTIME_APPROVE_TOOL, RUNTIME_ENQUEUE_TOOL]
|
|
822
|
+
: [];
|
|
823
|
+
if (classification.activeRun && ['converse', 'observe', 'ambiguous', 'approve', 'cancel', 'enqueue_run'].includes(classification.kind)) {
|
|
824
|
+
// During an active run Donna gets read + profile + the runtime control
|
|
825
|
+
// suite: she can answer, approve, enqueue for later, soft-cancel or
|
|
826
|
+
// kill — but she must not fire new MCP jobs alongside the run (that is
|
|
827
|
+
// what runtime__enqueue is for). No canned regex answers anywhere.
|
|
828
|
+
return [SHELL_READ_COMMAND_TOOL, SHELL_PROFILE_UPDATE_TOOL, ...controlTools];
|
|
829
|
+
}
|
|
830
|
+
return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...writeTools];
|
|
633
831
|
}
|
|
634
832
|
|
|
635
833
|
export function createAgentGraph(options = {}) {
|
|
@@ -655,8 +853,17 @@ export function createAgentGraph(options = {}) {
|
|
|
655
853
|
state.session._onStep?.('Agent: planning next action…');
|
|
656
854
|
}
|
|
657
855
|
|
|
856
|
+
// Inside a runtime run the input IS the task to execute (the runtime
|
|
857
|
+
// already accepted it as a run): the interactive control-message
|
|
858
|
+
// classifier must not apply. Without this, agentProjection.status is
|
|
859
|
+
// 'running' during every run, so any action verb ("lance l'ingestion")
|
|
860
|
+
// matched the active-run 'ambiguous' branch and returned a canned
|
|
861
|
+
// clarification instead of executing — the run ended silently.
|
|
862
|
+
const runtimeExecution = Boolean(state.session._currentRunIdentity);
|
|
658
863
|
const classification = iterations === 0
|
|
659
|
-
?
|
|
864
|
+
? (runtimeExecution
|
|
865
|
+
? { kind: 'execute_run', confidence: 1, reason: 'runtime_run_execution', activeRun: true }
|
|
866
|
+
: classifyAgentInput(state.input, state.session))
|
|
660
867
|
: (state.inputClassification ?? { kind: 'modify_run', confidence: 1, reason: 'tool_iteration' });
|
|
661
868
|
if (iterations === 0) {
|
|
662
869
|
state.session._onStep?.(`Agent: classified input as ${classification.kind}`);
|
|
@@ -665,14 +872,6 @@ export function createAgentGraph(options = {}) {
|
|
|
665
872
|
classification,
|
|
666
873
|
});
|
|
667
874
|
}
|
|
668
|
-
if (iterations === 0 && classification.kind === 'ambiguous') {
|
|
669
|
-
return {
|
|
670
|
-
response: 'Je peux répondre sur le run en cours, modifier ce run, mettre une nouvelle demande en file, approuver, ou annuler. Peux-tu préciser ce que tu veux faire ?',
|
|
671
|
-
pendingToolCalls: null,
|
|
672
|
-
readyToStream: false,
|
|
673
|
-
inputClassification: classification,
|
|
674
|
-
};
|
|
675
|
-
}
|
|
676
875
|
|
|
677
876
|
const writeTools = [
|
|
678
877
|
SHELL_RUN_COMMAND_TOOL,
|
|
@@ -681,7 +880,7 @@ export function createAgentGraph(options = {}) {
|
|
|
681
880
|
WIKI_PLAN_DONE_TOOL,
|
|
682
881
|
...buildLlmTools(state.session.mcp),
|
|
683
882
|
];
|
|
684
|
-
const tools = toolsForClassification(classification, writeTools);
|
|
883
|
+
const tools = toolsForClassification(classification, writeTools, state.session);
|
|
685
884
|
const system = buildAgentSystemPrompt(state);
|
|
686
885
|
|
|
687
886
|
// On iteration 0: prior history is in state.messages, user input must be appended.
|
|
@@ -713,6 +912,13 @@ export function createAgentGraph(options = {}) {
|
|
|
713
912
|
|
|
714
913
|
if (result.tool_calls?.length > 0) {
|
|
715
914
|
state.session._onStreamReset?.();
|
|
915
|
+
// Close the streaming conversation entry now: the text streamed so
|
|
916
|
+
// far is this iteration's narration. Without this, the next
|
|
917
|
+
// iteration's deltas append to the SAME entry with no separator and
|
|
918
|
+
// the chat becomes one glued wall of text ("…de la config.Voyons…").
|
|
919
|
+
// An empty finalize keeps the accumulated content and just drops the
|
|
920
|
+
// streaming flag; it is a no-op when nothing was streamed.
|
|
921
|
+
emitAgentEvent(state.session, 'assistant_message', 'llm', { content: '' });
|
|
716
922
|
state.session._onStep?.(`[${iterations + 1}/${MAX_TOOL_ITERATIONS}] ${result.tool_calls.length} MCP action${result.tool_calls.length > 1 ? 's' : ''} queued…`);
|
|
717
923
|
// On iteration 0 persist the user message too so it survives the loop.
|
|
718
924
|
const newMessages = iterations === 0
|
|
@@ -771,11 +977,19 @@ export function createAgentGraph(options = {}) {
|
|
|
771
977
|
const toolResultMessages = [];
|
|
772
978
|
|
|
773
979
|
for (const call of toolCalls) {
|
|
774
|
-
const
|
|
980
|
+
const resolved = resolveToolCallName(state.session.mcp, call.function.name, INTERNAL_TOOL_SERVERS);
|
|
981
|
+
const { server, tool } = resolved;
|
|
775
982
|
const argsSummary = summarizeToolArguments(call.function.arguments);
|
|
776
983
|
const isInternalWikiTool = server === 'wiki' && (tool === 'plan_set' || tool === 'plan_done');
|
|
777
984
|
const serverLabel = server === 'shell' ? 'Shell' : isInternalWikiTool ? 'Plan' : 'MCP';
|
|
778
|
-
const toolName = `${server}.${tool}
|
|
985
|
+
const toolName = server ? `${server}.${tool}` : call.function.name;
|
|
986
|
+
if (resolved.normalized) {
|
|
987
|
+
// Keep normalizations visible: the defensive routing must not hide
|
|
988
|
+
// prompt/skill regressions that reintroduce unqualified names.
|
|
989
|
+
state.session._onStep?.(
|
|
990
|
+
`tool name normalized: ${call.function.name} -> ${server}__${tool}`,
|
|
991
|
+
);
|
|
992
|
+
}
|
|
779
993
|
state.session._onStep?.(
|
|
780
994
|
`[${state.toolIterations}/${MAX_TOOL_ITERATIONS}] ${serverLabel} ${toolName}${argsSummary ? ` (${argsSummary})` : ''}`,
|
|
781
995
|
);
|
|
@@ -796,6 +1010,18 @@ export function createAgentGraph(options = {}) {
|
|
|
796
1010
|
let resultText;
|
|
797
1011
|
let ok = true;
|
|
798
1012
|
try {
|
|
1013
|
+
if (!server) {
|
|
1014
|
+
if (resolved.candidates.length > 1) {
|
|
1015
|
+
throw new Error(
|
|
1016
|
+
`Ambiguous unqualified tool name "${call.function.name}": several connected servers expose it. `
|
|
1017
|
+
+ `Use the <server>__<tool> form: ${resolved.candidates.map((s) => `${s}__${tool}`).join(', ')}.`,
|
|
1018
|
+
);
|
|
1019
|
+
}
|
|
1020
|
+
throw new Error(
|
|
1021
|
+
`Unqualified tool call name "${call.function.name}". Tool calls must use the <server>__<tool> `
|
|
1022
|
+
+ `naming convention (e.g. production__production_start_job); no connected server exposes a tool named "${tool}".`,
|
|
1023
|
+
);
|
|
1024
|
+
}
|
|
799
1025
|
let args = JSON.parse(call.function.arguments ?? '{}');
|
|
800
1026
|
if (server === 'production' && tool === 'production_start_job' && state.session.workspace && !args.callerLabel) {
|
|
801
1027
|
args = { ...args, callerLabel: `${state.session.workspace}/wiki-manager` };
|
|
@@ -811,6 +1037,8 @@ export function createAgentGraph(options = {}) {
|
|
|
811
1037
|
} else if (server === 'shell' && tool === 'profile_update') {
|
|
812
1038
|
const result = await updateWorkspaceProfilePreference(state.session, args.preference);
|
|
813
1039
|
resultText = JSON.stringify(result, null, 2);
|
|
1040
|
+
} else if (server === 'runtime') {
|
|
1041
|
+
resultText = await handleRuntimeControlTool(state.session, tool, args);
|
|
814
1042
|
} else if (server !== 'shell') {
|
|
815
1043
|
await awaitRunApproval(state.session, { runId, tool: toolName });
|
|
816
1044
|
await awaitToolApproval(state.session, {
|
|
@@ -833,6 +1061,41 @@ export function createAgentGraph(options = {}) {
|
|
|
833
1061
|
resultText = formatMcpToolResult(result);
|
|
834
1062
|
}
|
|
835
1063
|
}
|
|
1064
|
+
{
|
|
1065
|
+
const payload = parseJsonText(resultText);
|
|
1066
|
+
// agent_plan (any provider) returned a task-graph fragment: declare it as
|
|
1067
|
+
// the plan DETERMINISTICALLY. Asking the LLM to copy N tasks into
|
|
1068
|
+
// wiki__plan_set would lose fields (a small local model dropped
|
|
1069
|
+
// arguments/operations in testing) — the shell does the mapping.
|
|
1070
|
+
if (/(^|__)agent_plan$/.test(tool) && Array.isArray(payload?.tasks) && payload.tasks.length > 0) {
|
|
1071
|
+
const steps = payload.tasks.map((task, index) => normalizeDeclaredPlanStep({
|
|
1072
|
+
id: task.id,
|
|
1073
|
+
description: task.label ?? task.id ?? `Task ${index + 1}`,
|
|
1074
|
+
requiredCapability: task.requiredCapability ?? payload.capability ?? null,
|
|
1075
|
+
operation: task.operation ?? null,
|
|
1076
|
+
arguments: task.arguments ?? {},
|
|
1077
|
+
dependsOn: task.dependsOn ?? [],
|
|
1078
|
+
outputRefs: (task.expectedOutputRefs ?? []).map((ref) => (ref && typeof ref === 'object' ? String(ref.ref ?? '') : String(ref))).filter(Boolean),
|
|
1079
|
+
groupId: task.groupId ?? null,
|
|
1080
|
+
dependsOnGroup: task.dependsOnGroup ?? null,
|
|
1081
|
+
parallelizable: task.parallelizable,
|
|
1082
|
+
barrier: task.barrier,
|
|
1083
|
+
locks: task.locks,
|
|
1084
|
+
requiresApproval: task.requiresApproval,
|
|
1085
|
+
approvalClass: task.approvalClass,
|
|
1086
|
+
approvalSummary: task.approvalSummary,
|
|
1087
|
+
idempotencyKey: task.idempotencyKey,
|
|
1088
|
+
progressWeight: task.progressWeight,
|
|
1089
|
+
recommendedConcurrency: task.recommendedConcurrency,
|
|
1090
|
+
}, index, state.session));
|
|
1091
|
+
emitAgentEvent(state.session, 'plan_set', 'tool', { steps });
|
|
1092
|
+
state.session._onStep?.(`Plan: ${steps.length} task(s) declared from ${server} fragment`);
|
|
1093
|
+
resultText = `Task-graph fragment integrated as the current plan (${steps.length} task(s), groups: ${[...new Set(payload.tasks.map((task) => task.groupId).filter(Boolean))].join(', ') || 'none'}). The orchestrator will dispatch these tasks — do NOT call production tools for them yourself. Reply with a short summary and wait.`;
|
|
1094
|
+
}
|
|
1095
|
+
if (/(^|__)agent_plan$/.test(tool) && Array.isArray(payload?.tasks) && payload.tasks.length > 0) {
|
|
1096
|
+
minimalPlanActive = false;
|
|
1097
|
+
}
|
|
1098
|
+
}
|
|
836
1099
|
if (server === 'production') {
|
|
837
1100
|
let payload = parseJsonText(resultText);
|
|
838
1101
|
if (tool === 'production_start_job' && payload?.ok === false && payload?.error === 'workspace_busy') {
|
|
@@ -869,22 +1132,26 @@ export function createAgentGraph(options = {}) {
|
|
|
869
1132
|
err.name === 'ApprovalError'
|
|
870
1133
|
) throw err;
|
|
871
1134
|
ok = false;
|
|
872
|
-
resultText = `Error [${
|
|
1135
|
+
resultText = `Error [${toolName}]: ${err instanceof Error ? err.message : String(err)}`;
|
|
873
1136
|
if (minimalPlanActive && state.session.headlessPlan?.[0]?._activityKey === null) {
|
|
874
1137
|
emitAgentEvent(state.session, 'plan_step_updated', 'tool', { step: 1, status: 'failed' });
|
|
875
1138
|
}
|
|
876
1139
|
}
|
|
1140
|
+
// Bound the result at its two exit points only (LLM context + display).
|
|
1141
|
+
// The full resultText above was already used for payload parsing and
|
|
1142
|
+
// _activity extraction, which must never see a truncated document.
|
|
1143
|
+
const boundedResult = truncateToolResult(resultText);
|
|
877
1144
|
emitAgentEvent(state.session, 'tool_call_result', 'tool', {
|
|
878
1145
|
callId: call.id,
|
|
879
1146
|
name: toolName,
|
|
880
1147
|
ok,
|
|
881
|
-
result:
|
|
1148
|
+
result: boundedResult,
|
|
882
1149
|
summary: ok ? 'done' : 'failed',
|
|
883
1150
|
});
|
|
884
1151
|
toolResultMessages.push({
|
|
885
1152
|
role: 'tool',
|
|
886
1153
|
tool_call_id: call.id,
|
|
887
|
-
content:
|
|
1154
|
+
content: boundedResult,
|
|
888
1155
|
});
|
|
889
1156
|
}
|
|
890
1157
|
|
|
@@ -899,11 +1166,22 @@ export function createAgentGraph(options = {}) {
|
|
|
899
1166
|
return END;
|
|
900
1167
|
}
|
|
901
1168
|
|
|
902
|
-
|
|
1169
|
+
const compiled = new StateGraph(AgentState)
|
|
903
1170
|
.addNode('orchestrator', orchestratorNode)
|
|
904
1171
|
.addNode('tool_executor', toolExecutorNode)
|
|
905
1172
|
.addEdge(START, 'orchestrator')
|
|
906
1173
|
.addConditionalEdges('orchestrator', routeOrchestrator)
|
|
907
1174
|
.addEdge('tool_executor', 'orchestrator')
|
|
908
1175
|
.compile();
|
|
1176
|
+
|
|
1177
|
+
// LangGraph's default recursionLimit is 25 super-steps. Each tool round
|
|
1178
|
+
// costs two of them (orchestrator + tool_executor), so runs died with
|
|
1179
|
+
// GRAPH_RECURSION_LIMIT around iteration 12 — far below the intended
|
|
1180
|
+
// MAX_TOOL_ITERATIONS budget — before finishing their work (observed as
|
|
1181
|
+
// `knowledge.update — error 0%` with no production job ever created).
|
|
1182
|
+
// Bake a limit matching the iteration budget into every invocation.
|
|
1183
|
+
const recursionLimit = MAX_TOOL_ITERATIONS * 2 + 10;
|
|
1184
|
+
return {
|
|
1185
|
+
invoke: (state, config = {}) => compiled.invoke(state, { recursionLimit, ...config }),
|
|
1186
|
+
};
|
|
909
1187
|
}
|