@dotdrelle/wiki-manager 0.14.0 → 0.14.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/mcp.endpoints.example.json +7 -0
- package/package.json +2 -2
- package/src/agent/graph.js +40 -14
- package/src/agent/graph.test.js +5 -3
- package/src/cli/wiki-manager.js +9 -0
- package/src/commands/slash.js +48 -2
- package/src/core/agentEvents.js +22 -1
- package/src/core/agentEvents.test.js +13 -1
- package/src/core/buildInfo.json +2 -2
- package/src/core/mcp.js +43 -16
- package/src/core/toolLoop.js +56 -0
- package/src/core/toolLoop.test.js +88 -0
- package/src/runtime/client.js +2 -1
- package/src/runtime/donna-contract.test.js +3 -0
- package/src/runtime/lifecycle.js +32 -2
- package/src/runtime/runner.js +22 -3
- package/src/runtime/server.js +20 -4
- package/src/runtime/store.js +40 -0
- package/src/shell/repl.js +92 -19
- package/src/shell/repl.test.js +70 -0
|
@@ -25,5 +25,12 @@
|
|
|
25
25
|
"Authorization": "Bearer ${DOCUMENTS_MCP_AUTH_TOKEN}"
|
|
26
26
|
}
|
|
27
27
|
}
|
|
28
|
+
},
|
|
29
|
+
"chatAccess": {
|
|
30
|
+
"maxToolIterations": 6,
|
|
31
|
+
"servers": {
|
|
32
|
+
"production": { "allow": ["production_job_status", "production_jobs_list"] },
|
|
33
|
+
"cme": { "allow": ["cme_status", "cme_sources_list", "cme_export_status"] }
|
|
34
|
+
}
|
|
28
35
|
}
|
|
29
36
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@dotdrelle/wiki-manager",
|
|
3
|
-
"version": "0.14.
|
|
3
|
+
"version": "0.14.2",
|
|
4
4
|
"description": "Agentic shell and orchestration cockpit for llm-wiki workspaces.",
|
|
5
5
|
"license": "PolyForm-Noncommercial-1.0.0",
|
|
6
6
|
"author": "dotrelle",
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
},
|
|
12
12
|
"scripts": {
|
|
13
13
|
"start": "bun ./bin/wiki-manager.js",
|
|
14
|
-
"test": "node --test src/agent/graph.test.js src/contracts/schemas.test.js src/core/activity.test.js src/core/buildInfo.test.js src/core/agentEvents.test.js src/core/runtimeLog.test.js src/activity/activityAggregator.test.js src/graph/runGraphProjector.test.js src/core/workflow.test.js src/core/planPatch.test.js src/core/agentLoop.test.js src/core/plan.test.js src/core/mcp.test.js src/core/documentIntake.test.js src/core/dockerCompose.test.js src/core/wikiWorkspace.test.js src/core/wikirc.test.js src/core/modelFetch.test.js src/core/startupCheck.test.js src/core/queueStore.test.js src/orchestrator/agentRegistry.test.js src/orchestrator/capabilityRegistry.test.js src/orchestrator/capabilityResolver.test.js src/orchestrator/planValidator.test.js src/orchestrator/planIntegrator.test.js src/orchestrator/scheduler.test.js src/orchestrator/attemptManager.test.js src/orchestrator/resultAggregator.test.js src/orchestrator/approvalPolicy.test.js src/commands/slash.test.js src/shell/repl.test.js src/runtime/store.test.js src/runtime/controlMessages.test.js src/runtime/recoveryManager.test.js src/runtime/server.test.js src/runtime/supervisor.test.js src/runtime/runner.test.js src/runtime/runner.e2e.test.js src/runtime/donna-contract.test.js src/runtime/auth.test.js",
|
|
14
|
+
"test": "node --test src/agent/graph.test.js src/contracts/schemas.test.js src/core/activity.test.js src/core/buildInfo.test.js src/core/agentEvents.test.js src/core/runtimeLog.test.js src/activity/activityAggregator.test.js src/graph/runGraphProjector.test.js src/core/workflow.test.js src/core/planPatch.test.js src/core/agentLoop.test.js src/core/plan.test.js src/core/mcp.test.js src/core/toolLoop.test.js src/core/documentIntake.test.js src/core/dockerCompose.test.js src/core/wikiWorkspace.test.js src/core/wikirc.test.js src/core/modelFetch.test.js src/core/startupCheck.test.js src/core/queueStore.test.js src/orchestrator/agentRegistry.test.js src/orchestrator/capabilityRegistry.test.js src/orchestrator/capabilityResolver.test.js src/orchestrator/planValidator.test.js src/orchestrator/planIntegrator.test.js src/orchestrator/scheduler.test.js src/orchestrator/attemptManager.test.js src/orchestrator/resultAggregator.test.js src/orchestrator/approvalPolicy.test.js src/commands/slash.test.js src/shell/repl.test.js src/runtime/store.test.js src/runtime/controlMessages.test.js src/runtime/recoveryManager.test.js src/runtime/server.test.js src/runtime/supervisor.test.js src/runtime/runner.test.js src/runtime/runner.e2e.test.js src/runtime/donna-contract.test.js src/runtime/auth.test.js",
|
|
15
15
|
"check-versions": "node scripts/check-versions.js",
|
|
16
16
|
"prepack": "node scripts/check-versions.js",
|
|
17
17
|
"prepublishOnly": "node scripts/check-versions.js",
|
package/src/agent/graph.js
CHANGED
|
@@ -711,7 +711,7 @@ async function handleRuntimeControlTool(session, tool, args = {}) {
|
|
|
711
711
|
if (!objective) return 'Delegation rejected: missing objective.';
|
|
712
712
|
const result = await postRuntimeDelegate(objective, { url, workspace });
|
|
713
713
|
return result?.runId
|
|
714
|
-
? `
|
|
714
|
+
? `Action lancée (${String(result.runId).slice(0, 8)}) après validation du plan réel : ${result.delegation?.tasks ?? 0} tâche(s), ${result.delegation?.agent ?? 'agent résolu'}. Exécution en cours.`
|
|
715
715
|
: `Délégation refusée : ${result?.error ?? JSON.stringify(result)}`;
|
|
716
716
|
}
|
|
717
717
|
if (tool === 'enqueue') {
|
|
@@ -845,17 +845,14 @@ export function buildAgentSystemPrompt(state) {
|
|
|
845
845
|
// mutating provider tools (e.g. production__production_start_job) here teaches
|
|
846
846
|
// a capable model to invoke them directly and bypass runtime__delegate.
|
|
847
847
|
const mcpTools = formatMcpToolsForAgent(state.session.mcp, {
|
|
848
|
-
include: (qualifiedName
|
|
849
|
-
function: { name: qualifiedName },
|
|
850
|
-
readOnly: tool?.annotations?.readOnlyHint === true,
|
|
851
|
-
}),
|
|
848
|
+
include: (qualifiedName) => !isOrchestrationBypassTool(qualifiedName),
|
|
852
849
|
});
|
|
853
850
|
const skills = formatSkillsForAgent(state.session);
|
|
854
851
|
const customPrompt = state.session.systemPrompt ?? null;
|
|
855
852
|
const workspaceProfile = loadWorkspaceProfile(state.session.workspacePath);
|
|
856
853
|
|
|
857
854
|
const agentContext = [
|
|
858
|
-
'You are Donna
|
|
855
|
+
'You are Donna: first and foremost a warm, helpful assistant for the llm-wiki-manager team, who also happens to orchestrate the workspace behind the scenes. Orchestration is how you help — it is not your personality. Speak like an attentive human colleague: natural, friendly, plain-spoken. Never sound like a raw status dump or a machine reciting fields.',
|
|
859
856
|
'The shell is agent-first: every input without a leading slash is routed to you.',
|
|
860
857
|
'Default to a plain conversational reply with no tool call. Only call a tool, create a plan, or start a job when the user\'s message clearly requests an action (ingest, build, export, configure, run a skill, check a concrete status, etc.). Greetings, small talk, thanks, and general questions do not warrant starting a job or calling a tool — just answer in text.',
|
|
861
858
|
'Commands starting with / are deterministic primitives. You may run a safe subset through shell__run_command.',
|
|
@@ -864,7 +861,7 @@ export function buildAgentSystemPrompt(state) {
|
|
|
864
861
|
`Current wikirc profile: ${wikirc}.`,
|
|
865
862
|
`Available primitives: ${commandList(state.session)}.`,
|
|
866
863
|
'Only announce or call slash commands that appear exactly in Available primitives. Do not invent command names, subcommands, or arguments.',
|
|
867
|
-
'Connected
|
|
864
|
+
'Connected MCP tools you may call directly (server__tool naming convention) — reads AND single-step actions like configuring or adding a connector source, converting a document, sending, or searching. Only the heavy multi-step operations (ingest, build, export, polish, pipeline) go through runtime__delegate to get their parallel plan. Everything listed below is directly callable:',
|
|
868
865
|
mcpTools,
|
|
869
866
|
'Current local MCP job queue:',
|
|
870
867
|
formatQueue(state.session),
|
|
@@ -873,13 +870,14 @@ export function buildAgentSystemPrompt(state) {
|
|
|
873
870
|
'In interactive agent mode you may call only the read-only tools and runtime control/delegation tools actually provided to you.',
|
|
874
871
|
'When the user asks for an action that can be performed with connected MCP tools or safe primitives, do not answer with future intent such as "I will call...", "I am going to run...", or "launching..." unless you also call the tool in the same turn. Either call the tool now, ask for the exact missing required arguments, or explain the concrete blocker.',
|
|
875
872
|
'Execution truthfulness: never invent a job id, status, percentage, duration, generated file, file content, URL, command, or tool result. An action is executed only when you call an available tool and receive its result. Examples and placeholders are forbidden in execution reports.',
|
|
876
|
-
'After any completed action, give a short factual summary based only on the tool result: outcome and concrete outputs or references actually returned. Mention a viewing primitive only when it exists in Available primitives and is relevant.
|
|
877
|
-
'
|
|
873
|
+
'After any completed action, give a short factual summary based only on the tool result: outcome and concrete outputs or references actually returned. Mention a viewing primitive only when it exists in Available primitives and is relevant. Never invent results, interpret generated content beyond what the tool returned, or fabricate a verification checklist.',
|
|
874
|
+
'You are in AGENT mode, so you can actually act. You MAY close with ONE short, natural follow-up — a single sentence phrased as an offer, and only when it genuinely helps and is an action you can perform right here (delegate it or call a tool), e.g. "Want me to start ingesting these pages?" (phrased in the reply language). This is what makes you feel like an assistant rather than a readout. Only offer what you can truly do in agent mode — never an offer that would require another mode. Keep it to that one line: never produce a "Next steps"/"Prochaines étapes"/"À suivre" list, a checklist, an options menu, or commands for the user to type. If nothing useful naturally follows, simply stop after the answer — do not pad.',
|
|
878
875
|
'When calling a tool, emit no preliminary narration. Call it directly; the PLAN and Activity panels show progress. After completion, keep the final response concise and proportional to the result.',
|
|
879
|
-
'
|
|
880
|
-
'
|
|
876
|
+
'Write the way a thoughtful colleague speaks: warm, plain, and to the point. For a simple factual question, 1 to 3 sentences is the sweet spot. Stay synthetic and information-dense — use only the lines needed, and never exceed roughly 15 to 20 short lines even for a detailed answer. Never expose internal reasoning, repeated checks, tool-selection commentary, or a chronological diary. Prioritize the result, essential facts, concrete errors, and actual outputs — but say them in human language, not as a field dump.',
|
|
877
|
+
'Only the heavy multi-step operations — ingest, build, export, polish, pipeline — are delegated via runtime__delegate (for their DAG and parallelism). Single-step actions — configuring or adding a connector source, converting a document, sending, searching — are called directly on the connected tool. Never call an agent orchestration-contract or plan tool directly.',
|
|
881
878
|
'For any question about the current workspace inventory or what is waiting there, call wiki__wiki_workspace_status first and answer only from its result. This is the canonical read-only workspace state; do not reconstruct it from upload, connector, or production tools.',
|
|
882
879
|
'Tool identifiers are private implementation details. Never print MCP tool names such as server__tool in a user-facing answer. Describe the human result instead.',
|
|
880
|
+
'Internal data shapes are private too. Never quote raw JSON field names (e.g. pendingSources.files), internal directory paths (e.g. raw/untracked/), or config keys in a user-facing answer — translate them into plain language. Say "36 pages sources sont en attente d\'ingestion", not the field or path they came from.',
|
|
883
881
|
'Never suggest a manual filesystem command or implementation workaround unless the user explicitly asks for manual instructions. For an action request, delegate the objective and let the specialized agent determine paths and operations from its live contract.',
|
|
884
882
|
'Skills are documentation only in this stabilized version. Never execute a skill from conversation; delegate the user objective.',
|
|
885
883
|
'For service actions, recommend only available service primitives from Available primitives, with the exact service name when the primitive supports one.',
|
|
@@ -887,8 +885,10 @@ export function buildAgentSystemPrompt(state) {
|
|
|
887
885
|
state.session.runtime?.url
|
|
888
886
|
? 'The runtime is connected and runtime__delegate is bound and available to you right now — it is a tool you call directly, not a slash command or a missing primitive. It is the ONLY way to execute an action (ingest, build, export, configure, send…). Never tell the user that delegation or the runtime is unavailable while it is connected; call runtime__delegate instead.'
|
|
889
887
|
: 'No runtime is connected, so you cannot execute actions. State that plainly and name the runtime connection as the missing capability — do not invent a workaround.',
|
|
888
|
+
'If the connector or service needed for a requested read or action is absent from the Connected MCP tools above (its service is not running — e.g. CME, documents, or production), say plainly that this service is not connected and name it as the missing capability. Never redirect a simple read (e.g. "give me the CME config") to an "agent action", never invent its result, and never propose a workaround. Only requests you can actually serve with a listed tool are answered with data.',
|
|
890
889
|
'For any requested action, call runtime__delegate with the user objective only. Never choose a capability, operation, agent, plan, or implementation yourself. The runtime resolves the registry and validates the provider plan before accepting. Never call <provider>__agent_plan, <provider>__agent_execute, legacy production__production_start_job, wiki__plan_set, or wiki__plan_done from interactive chat.',
|
|
891
890
|
'Do not ask the user which sources, files, connectors, or templates to use for an ingest, build, or export: the specialized agent discovers them from the workspace. When the objective is clear (e.g. "lance une ingestion"), delegate it as stated, without a clarifying question.',
|
|
891
|
+
'Promise only what the resolved capability actually exposes in its declared contract (the input schema the specialized agent publishes for that capability). When the user requests an execution parameter — a batch or chunk size, a count "N at a time", concurrency, ordering, priority, or any tuning knob — apply it only if that parameter exists in the target capability\'s published input schema. Otherwise do not confirm or promise it: delegate the objective, and if the user explicitly asked for that parameter, say plainly in one line that you started the work but do not control that aspect (the runtime and the specialized agent decide it). Never state or imply a parameter was applied when the agent contract cannot enforce it.',
|
|
892
892
|
'If runtime__delegate returns a blocker or no specialized provider is available, report only that concrete blocker concisely. Never replace the missing execution path with a suggested slash command, skill, MCP tool name, manual file move, administrator escalation, or alternative workflow unless the user explicitly asks for alternatives.',
|
|
893
893
|
'For workspace inventory and page listings, use the connected wiki MCP read tools. Never invent or call a /wiki shell command through shell__run_command. Use /workspace init <name> [path] for low-level non-interactive workspace creation; in the interactive TUI, /new <name> opens the setup wizard.',
|
|
894
894
|
'If an action requires tools or skills not available yet, explain the limitation and name the expected primitive.',
|
|
@@ -896,6 +896,7 @@ export function buildAgentSystemPrompt(state) {
|
|
|
896
896
|
? `Workspace profile (.wiki/profile.md) — durable user preferences, apply these to every reply (tone, tutoiement/vouvoiement, formatting, etc.):\n${workspaceProfile}`
|
|
897
897
|
: null,
|
|
898
898
|
'Runtime control: you have runtime__status, runtime__cancel, runtime__kill, runtime__approve and runtime__enqueue. When the user asks to stop, remove, clean or kill the current run, its jobs or the queue ("supprime le job et la queue", "arr\u00eate tout"), call runtime__kill (or runtime__cancel for a soft stop of just the run) and confirm what was stopped. For questions about what is running or queued, call runtime__status and answer from its data. When the user consents to a pending approval in any phrasing ("vas-y", "ok pour l\'export"), call runtime__approve. When the user asks for a NEW action while a run is active, do not execute it: propose runtime__enqueue (run it after) or, if they insist it replaces the current work, runtime__kill then the new action.',
|
|
899
|
+
'Report every runtime control outcome exactly as the tool returned it \u2014 never embellish. If runtime__kill reports 0 run(s)/0 task(s)/0 purged, say there was nothing active to stop or purge; do NOT claim a run, plan, pending approval or queue item was removed. If runtime__status returns an error or could not be read, say the runtime state could not be retrieved and do not describe a state you never obtained. Never assert that something was cleaned, cancelled, approved or purged unless that specific tool result confirms it.',
|
|
899
900
|
'Durable profile updates are actions in this stabilized version: delegate them instead of writing directly.',
|
|
900
901
|
].filter(Boolean).join('\n');
|
|
901
902
|
|
|
@@ -944,13 +945,20 @@ function toolsForClassification(classification, writeTools, session = null) {
|
|
|
944
945
|
return [SHELL_READ_COMMAND_TOOL, ...controlTools];
|
|
945
946
|
}
|
|
946
947
|
if (session?.runtime?.url) {
|
|
947
|
-
|
|
948
|
-
|
|
948
|
+
// Offer every connected tool directly EXCEPT orchestration-bypass tools
|
|
949
|
+
// and raw shell write/profile mutation. Reads, configuration, connector
|
|
950
|
+
// setup — and any newly added MCP's tools — stay directly callable.
|
|
951
|
+
const directTools = writeTools.filter((item) => {
|
|
952
|
+
const name = item?.function?.name;
|
|
953
|
+
if (!name || name === 'shell__run_command' || name === 'shell__profile_update') return false;
|
|
954
|
+
return !isOrchestrationBypassTool(name);
|
|
955
|
+
});
|
|
956
|
+
return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...directTools];
|
|
949
957
|
}
|
|
950
958
|
return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...writeTools];
|
|
951
959
|
}
|
|
952
960
|
|
|
953
|
-
function isDonnaReadTool(item) {
|
|
961
|
+
export function isDonnaReadTool(item) {
|
|
954
962
|
const name = String(item?.function?.name ?? '');
|
|
955
963
|
if (!name || name.startsWith('shell__') || name === 'wiki__plan_set' || name === 'wiki__plan_done') return false;
|
|
956
964
|
if (item?.readOnly === true) return true;
|
|
@@ -961,6 +969,24 @@ function isDonnaReadTool(item) {
|
|
|
961
969
|
|| /(?:^|_)(?:status|list|search|read|get)$/.test(tool);
|
|
962
970
|
}
|
|
963
971
|
|
|
972
|
+
// Two-tier tool policy. Donna may call any connected MCP tool directly
|
|
973
|
+
// (reads AND plain writes: cme_setup, connector setup, document conversion,
|
|
974
|
+
// send, search, and anything a newly added MCP exposes) EXCEPT the small set
|
|
975
|
+
// that must go through the runtime's orchestration: the universal five-tool
|
|
976
|
+
// contract executors (agent_plan/agent_execute), the legacy job starter, and
|
|
977
|
+
// direct plan mutation. Heavy multi-step work (ingest/build/export via the
|
|
978
|
+
// production agent) is delegated for its DAG/parallelism; plain single-step
|
|
979
|
+
// tools are called directly. This is a blocklist, not a whitelist, so adding a
|
|
980
|
+
// new MCP never silently disables its tools.
|
|
981
|
+
function isOrchestrationBypassTool(name) {
|
|
982
|
+
const full = String(name ?? '');
|
|
983
|
+
if (!full) return true;
|
|
984
|
+
if (full === 'wiki__plan_set' || full === 'wiki__plan_done') return true;
|
|
985
|
+
const sep = full.indexOf('__');
|
|
986
|
+
const tool = sep === -1 ? full : full.slice(sep + 2);
|
|
987
|
+
return tool === 'agent_plan' || tool === 'agent_execute' || tool === 'production_start_job';
|
|
988
|
+
}
|
|
989
|
+
|
|
964
990
|
function isReadOnlyMcpCall(session, server, tool) {
|
|
965
991
|
const descriptor = (session?.mcp?.[server]?.tools ?? [])
|
|
966
992
|
.find((item) => String(item?.name ?? '') === tool || String(item?.name ?? '').endsWith(`__${tool}`));
|
package/src/agent/graph.test.js
CHANGED
|
@@ -716,10 +716,12 @@ test('workspace package manifest is not exposed as an executable skill', () => {
|
|
|
716
716
|
}
|
|
717
717
|
});
|
|
718
718
|
|
|
719
|
-
test('system prompt forbids
|
|
719
|
+
test('system prompt allows one follow-up line but forbids next-step sections', () => {
|
|
720
720
|
const prompt = buildAgentSystemPrompt({ session: sessionBase() });
|
|
721
|
-
|
|
722
|
-
assert.match(prompt, /
|
|
721
|
+
// A single natural follow-up offer is allowed (assistant feel), ...
|
|
722
|
+
assert.match(prompt, /ONE short, natural follow-up/);
|
|
723
|
+
// ... but multi-item next-step sections / checklists / option menus stay banned.
|
|
724
|
+
assert.match(prompt, /never produce a "Next steps"\/"Prochaines étapes"\/"À suivre" list/);
|
|
723
725
|
assert.doesNotMatch(prompt, /list the suggested follow-ups/);
|
|
724
726
|
});
|
|
725
727
|
|
package/src/cli/wiki-manager.js
CHANGED
|
@@ -847,6 +847,15 @@ async function runRuntime(argv, agent) {
|
|
|
847
847
|
throw new Error(`Delegated plan integration failed: ${(integrated.errors ?? []).map((error) => error.message ?? error.code ?? String(error)).join('; ')}`);
|
|
848
848
|
}
|
|
849
849
|
emitRuntimeLog(session, `delegation: ${prepared.fragment.tasks.length} validated task(s) integrated from ${prepared.provider.serverName}.agent_plan (${prepared.capability}/${prepared.operation})`);
|
|
850
|
+
// Demandé = consenti: a directly-delegated run carries the user's
|
|
851
|
+
// explicit consent, so auto-approve its initial plan. Persisting a
|
|
852
|
+
// run-scope grant (via the approval manager) makes the scheduler's
|
|
853
|
+
// readyTasks approval check pass, so the tasks run without re-prompting.
|
|
854
|
+
// Replanned tasks are integrated later without a fresh grant.
|
|
855
|
+
if (context.approvalManager?.approve) {
|
|
856
|
+
context.approvalManager.approve({ scope: 'run', runId });
|
|
857
|
+
emitRuntimeLog(session, `approval: run ${runId} auto-approved (user-requested action)`);
|
|
858
|
+
}
|
|
850
859
|
body._planReady?.resolve?.({ runId, planRevision: session.agentProjection?.planRevision ?? 0 });
|
|
851
860
|
}
|
|
852
861
|
// Deterministic capability run (/ingest): ask the capable agent for its
|
package/src/commands/slash.js
CHANGED
|
@@ -634,7 +634,8 @@ ${helpPair('/cancel', 'Cancel active run', '', '')}
|
|
|
634
634
|
${helpPair('/run cancel', 'Cancel active run', '', '')}
|
|
635
635
|
${helpPair('/queue', 'MCP job queue', '/queue clear', 'Clear finished')}
|
|
636
636
|
${helpPair('/queue cancel <id>', 'Cancel queued/running', '', '')}
|
|
637
|
-
${helpPair('/clear', 'Clear screen', '/
|
|
637
|
+
${helpPair('/clear', 'Clear screen', '/clear --all', 'Reset run+plan+queue+logs')}
|
|
638
|
+
${helpPair('/exit', 'Exit', '', '')}
|
|
638
639
|
${helpPair('Ctrl+Y', 'Copy last reply', '', '')}
|
|
639
640
|
${helpPair('PgUp/PgDn', 'Scroll thread', 'Ctrl+C Ctrl+C', 'Exit')}
|
|
640
641
|
|
|
@@ -1279,7 +1280,52 @@ export async function handleSlashCommand(line, context) {
|
|
|
1279
1280
|
case 'clear': {
|
|
1280
1281
|
const key = context.session.workspace || '__global__';
|
|
1281
1282
|
context.session.conversations[key] = [];
|
|
1282
|
-
|
|
1283
|
+
const wantsAll = args.slice(1).some((arg) => /^--?all$/i.test(String(arg)));
|
|
1284
|
+
if (!wantsAll) return { output: null };
|
|
1285
|
+
|
|
1286
|
+
// /clear --all is a full reset, not just a screen wipe: it purges the
|
|
1287
|
+
// persisted runtime runs (interrupted runs are terminal and never
|
|
1288
|
+
// recovered at reboot, so this is what actually removes a zombie run),
|
|
1289
|
+
// clears the local MCP job queue, and empties the local projection
|
|
1290
|
+
// (plan, activities, logs, workflow) so the UI clears immediately
|
|
1291
|
+
// instead of waiting for the next SSE sync.
|
|
1292
|
+
const runtime = context.runtime ?? {};
|
|
1293
|
+
const workspace = context.session.workspace ?? null;
|
|
1294
|
+
const parts = [];
|
|
1295
|
+
if (runtime.url) {
|
|
1296
|
+
try {
|
|
1297
|
+
const killed = await postRuntimeKill({ url: runtime.url, workspace, runId: null, purge: true });
|
|
1298
|
+
const purged = killed.purged ?? { runs: 0, events: 0, queue: 0 };
|
|
1299
|
+
parts.push(`runtime : ${killed.runs ?? 0} run(s) interrompu(s), ${killed.tasks ?? 0} tâche(s), ${killed.queued ?? 0} requête(s)`);
|
|
1300
|
+
parts.push(`store purgé : ${purged.runs ?? 0} run(s), ${purged.events ?? 0} événement(s), ${purged.queue ?? 0} item(s) de file`);
|
|
1301
|
+
} catch (err) {
|
|
1302
|
+
parts.push(`runtime kill échoué : ${err instanceof Error ? err.message : String(err)}`);
|
|
1303
|
+
}
|
|
1304
|
+
} else {
|
|
1305
|
+
parts.push('runtime non connecté (rien à purger côté serveur)');
|
|
1306
|
+
}
|
|
1307
|
+
|
|
1308
|
+
const clearedQueue = clearFinishedQueueItems(context.session);
|
|
1309
|
+
context.session.agentProjection = {
|
|
1310
|
+
conversation: [],
|
|
1311
|
+
chain: [],
|
|
1312
|
+
plan: null,
|
|
1313
|
+
activities: [],
|
|
1314
|
+
logs: [],
|
|
1315
|
+
summary: null,
|
|
1316
|
+
status: 'idle',
|
|
1317
|
+
planRevision: 0,
|
|
1318
|
+
planPatches: [],
|
|
1319
|
+
};
|
|
1320
|
+
context.session.headlessPlan = null;
|
|
1321
|
+
context.session.activities = {};
|
|
1322
|
+
context.session.controlQueue = [];
|
|
1323
|
+
context.session.workflow = null;
|
|
1324
|
+
context.session.jobQueue = [];
|
|
1325
|
+
context.session.productionActivity = null;
|
|
1326
|
+
parts.push(`file locale : ${clearedQueue} item(s) terminés nettoyés`);
|
|
1327
|
+
|
|
1328
|
+
return { output: `Interface réinitialisée (--all) — ${parts.join(' · ')}.` };
|
|
1283
1329
|
}
|
|
1284
1330
|
case 'exit':
|
|
1285
1331
|
case 'quit':
|
package/src/core/agentEvents.js
CHANGED
|
@@ -105,6 +105,21 @@ export function dispatchAgentEvent(session, event) {
|
|
|
105
105
|
return normalized;
|
|
106
106
|
}
|
|
107
107
|
|
|
108
|
+
// Full in-memory projection reset for a session. The runtime keeps the live
|
|
109
|
+
// projection in memory (session.agentProjection) and serves it from /state, so
|
|
110
|
+
// interrupting runs is not enough to clear the PLAN/ACTIVITY/LOGS panels — the
|
|
111
|
+
// projection has to be emptied here too. Pair with store.clearWorkspaceState so
|
|
112
|
+
// the reset also survives a reboot (otherwise hydrateSession replays it back).
|
|
113
|
+
export function resetSessionProjection(session) {
|
|
114
|
+
if (!session || typeof session !== 'object') return;
|
|
115
|
+
session.agentEvents = [];
|
|
116
|
+
session._agentProjectionState = createProjectionState();
|
|
117
|
+
session.agentProjection = publicProjection(session._agentProjectionState);
|
|
118
|
+
applyAgentProjectionToSession(session, session.agentProjection);
|
|
119
|
+
session.jobQueue = [];
|
|
120
|
+
session._onPlanUpdate?.();
|
|
121
|
+
}
|
|
122
|
+
|
|
108
123
|
function withSessionRunIdentity(event, session) {
|
|
109
124
|
const identity = session?._currentRunIdentity;
|
|
110
125
|
if (!identity) return event;
|
|
@@ -196,7 +211,13 @@ export function applyAgentProjectionToSession(session, projection) {
|
|
|
196
211
|
function applyEvent(state, event) {
|
|
197
212
|
switch (event.type) {
|
|
198
213
|
case 'run_started':
|
|
199
|
-
|
|
214
|
+
// Only a real runtime run (origin 'runtime') marks the projection
|
|
215
|
+
// 'running'. An interactive turn (origin 'user') is NOT a run: forcing
|
|
216
|
+
// 'running' here made the graph classify activeRun=true and hide Donna's
|
|
217
|
+
// MCP read tools (cme_status, wiki_workspace_status…), so questions about
|
|
218
|
+
// MCP state failed. Preserve the existing status (idle, or a genuinely
|
|
219
|
+
// active runtime run synced from the runtime) for interactive turns.
|
|
220
|
+
if (event.origin === 'runtime') state.status = 'running';
|
|
200
221
|
state.plan = null;
|
|
201
222
|
state.chain = [];
|
|
202
223
|
state.activities = {};
|
|
@@ -20,13 +20,25 @@ test('reduceAgentEvents: run_started clears stale plan', () => {
|
|
|
20
20
|
origin: 'tool',
|
|
21
21
|
payload: { steps: ['Old action'] },
|
|
22
22
|
}),
|
|
23
|
-
createAgentEvent('run_started', { origin: '
|
|
23
|
+
createAgentEvent('run_started', { origin: 'runtime' }),
|
|
24
24
|
]);
|
|
25
25
|
assert.equal(projection.plan, null);
|
|
26
26
|
assert.equal(projection.activities.length, 0);
|
|
27
27
|
assert.equal(projection.status, 'running');
|
|
28
28
|
});
|
|
29
29
|
|
|
30
|
+
test('reduceAgentEvents: interactive (user) run_started clears state but is not a running run', () => {
|
|
31
|
+
const projection = reduceAgentEvents([
|
|
32
|
+
createAgentEvent('plan_set', { origin: 'tool', payload: { steps: ['Old action'] } }),
|
|
33
|
+
createAgentEvent('run_started', { origin: 'user' }),
|
|
34
|
+
]);
|
|
35
|
+
// An interactive turn clears stale plan/activities but must NOT mark the
|
|
36
|
+
// projection 'running' — otherwise the graph classifies activeRun=true and
|
|
37
|
+
// hides Donna's MCP read tools.
|
|
38
|
+
assert.equal(projection.plan, null);
|
|
39
|
+
assert.notEqual(projection.status, 'running');
|
|
40
|
+
});
|
|
41
|
+
|
|
30
42
|
test('reduceAgentEvents: tracks manual plan and step updates', () => {
|
|
31
43
|
const projection = reduceAgentEvents([
|
|
32
44
|
createAgentEvent('plan_set', {
|
package/src/core/buildInfo.json
CHANGED
package/src/core/mcp.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { existsSync, readFileSync } from 'node:fs';
|
|
2
2
|
import { managerEnvFile, managerMcpEndpointsFile, readEnvFile } from './env.js';
|
|
3
3
|
|
|
4
|
-
const WIKI_MANAGER_VERSION = '0.14.
|
|
4
|
+
const WIKI_MANAGER_VERSION = '0.14.2';
|
|
5
5
|
|
|
6
6
|
function envValue(key) {
|
|
7
7
|
const filePath = managerEnvFile();
|
|
@@ -43,6 +43,29 @@ function normalizeExternalUrlForRuntime(url) {
|
|
|
43
43
|
return url;
|
|
44
44
|
}
|
|
45
45
|
|
|
46
|
+
// Config-driven policy for the /chat read-only toolset — NOT /agent, which has
|
|
47
|
+
// the full toolset and ignores this. The endpoints file's "chatAccess" block
|
|
48
|
+
// declares, per server, which tools /chat may call ("*" or a list), plus a
|
|
49
|
+
// maxToolIterations budget. Operator-owned, agnostic allow-list. Returns null
|
|
50
|
+
// when not configured — then /chat stays a plain, tool-less conversation.
|
|
51
|
+
export function readChatAccessConfig() {
|
|
52
|
+
const filePath = managerMcpEndpointsFile();
|
|
53
|
+
if (!existsSync(filePath)) return null;
|
|
54
|
+
let raw;
|
|
55
|
+
try { raw = JSON.parse(readFileSync(filePath, 'utf8')); } catch { return null; }
|
|
56
|
+
const chatAccess = raw?.chatAccess;
|
|
57
|
+
if (!chatAccess || typeof chatAccess !== 'object' || Array.isArray(chatAccess)) return null;
|
|
58
|
+
const servers = {};
|
|
59
|
+
for (const [name, entry] of Object.entries(chatAccess.servers ?? {})) {
|
|
60
|
+
if (entry?.allow === '*') servers[name] = { allow: '*' };
|
|
61
|
+
else if (Array.isArray(entry?.allow)) servers[name] = { allow: entry.allow.map(String).filter(Boolean) };
|
|
62
|
+
}
|
|
63
|
+
const maxToolIterations = Number.isFinite(Number(chatAccess.maxToolIterations)) && Number(chatAccess.maxToolIterations) > 0
|
|
64
|
+
? Math.floor(Number(chatAccess.maxToolIterations))
|
|
65
|
+
: null;
|
|
66
|
+
return { maxToolIterations, servers };
|
|
67
|
+
}
|
|
68
|
+
|
|
46
69
|
function readExternalMcpEndpoints() {
|
|
47
70
|
const filePath = managerMcpEndpointsFile();
|
|
48
71
|
if (!existsSync(filePath)) return {};
|
|
@@ -59,6 +82,14 @@ function readExternalMcpEndpoints() {
|
|
|
59
82
|
url: normalizeExternalUrlForRuntime(interpolateEnv(String(endpoint.url))),
|
|
60
83
|
configuredUrl: interpolateEnv(String(endpoint.url)),
|
|
61
84
|
headers: normalizeHeaders(endpoint.headers),
|
|
85
|
+
// Tools the endpoint marks approval-gated: Donna may still call them
|
|
86
|
+
// directly (they are single-step tools), but toolRequiresApproval
|
|
87
|
+
// makes the call wait for the user's confirmation first (e.g. a
|
|
88
|
+
// destructive cme_export_run). Agent/operator owned — no hard-coded
|
|
89
|
+
// business name in the manager.
|
|
90
|
+
requireApproval: Array.isArray(endpoint.requireApproval)
|
|
91
|
+
? endpoint.requireApproval.map(String).filter(Boolean)
|
|
92
|
+
: undefined,
|
|
62
93
|
external: true,
|
|
63
94
|
},
|
|
64
95
|
]),
|
|
@@ -95,6 +126,9 @@ const DEFAULT_MCP_RETRY_POLICY = {
|
|
|
95
126
|
};
|
|
96
127
|
|
|
97
128
|
export function buildMcpStatus(session) {
|
|
129
|
+
// Attach the /chat read-tool policy to the session alongside MCP status.
|
|
130
|
+
// Only /chat (repl.js) reads session.chatAccess; /agent ignores it.
|
|
131
|
+
if (session) session.chatAccess = readChatAccessConfig();
|
|
98
132
|
const workspaceEnv = session.workspaceEnv ?? {};
|
|
99
133
|
const wikiMcpToken = session.wikircConfig?.mcp?.accessKey;
|
|
100
134
|
const wikiMcpDetail = workspaceEnv.WIKI_MCP_PORT
|
|
@@ -163,21 +197,14 @@ function compactDescription(value) {
|
|
|
163
197
|
return text.length > 420 ? `${text.slice(0, 417)}...` : text;
|
|
164
198
|
}
|
|
165
199
|
|
|
166
|
-
function clarifyToolDescription(
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
if (serverName === 'production' && toolName === 'production_start_job') {
|
|
175
|
-
return compactDescription([
|
|
176
|
-
base,
|
|
177
|
-
'Production export means wiki deliverable/publication export only. Do not use type=export for Confluence/CME/source export; use cme__cme_export_run instead.',
|
|
178
|
-
].filter(Boolean).join(' '));
|
|
179
|
-
}
|
|
180
|
-
return base;
|
|
200
|
+
function clarifyToolDescription(_serverName, _toolName, description) {
|
|
201
|
+
// Agnostic by design: the orchestrator does NOT inject per-agent knowledge
|
|
202
|
+
// here. Tool meaning — including disambiguation like "this export publishes a
|
|
203
|
+
// wiki deliverable, not a Confluence source export" — must live in each
|
|
204
|
+
// agent's own MCP tool description, so any operator (our orchestrator or a
|
|
205
|
+
// third-party host such as Claude) gets the same self-sufficient contract.
|
|
206
|
+
// We only normalize whitespace; we never rewrite what the agent published.
|
|
207
|
+
return compactDescription(description ?? '');
|
|
181
208
|
}
|
|
182
209
|
|
|
183
210
|
async function listMcpTools(endpoint) {
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
// Minimal, side-effect-free bounded tool-use loop.
|
|
2
|
+
//
|
|
3
|
+
// This is the shared mechanic of "ask the LLM with a tool set, run the tool
|
|
4
|
+
// calls it emits, feed results back, repeat up to a cap". The caller injects
|
|
5
|
+
// the ONLY policy that varies: `executeCall(call) -> string` decides whether a
|
|
6
|
+
// requested tool is allowed and produces its textual result (allow-list check,
|
|
7
|
+
// MCP dispatch, error formatting). The loop itself owns no plan, no delegation,
|
|
8
|
+
// no run identity and no agent events — deliberately unlike the /agent
|
|
9
|
+
// orchestration loop in createAgentGraph, which is a stateful LangGraph node
|
|
10
|
+
// graph and stays separate. Use this for stateless tool-answer turns (e.g.
|
|
11
|
+
// /chat read-only questions).
|
|
12
|
+
//
|
|
13
|
+
// `executeCall` may throw to abort the whole loop (e.g. an AbortError on
|
|
14
|
+
// cancel); anything it returns is treated as the tool result for that call.
|
|
15
|
+
export async function runBoundedToolLoop({
|
|
16
|
+
llm,
|
|
17
|
+
system,
|
|
18
|
+
messages,
|
|
19
|
+
tools,
|
|
20
|
+
executeCall,
|
|
21
|
+
maxIterations = 4,
|
|
22
|
+
signal,
|
|
23
|
+
onStep,
|
|
24
|
+
} = {}) {
|
|
25
|
+
const cap = Math.max(1, Math.floor(maxIterations) || 1);
|
|
26
|
+
const convo = [...(messages ?? [])];
|
|
27
|
+
for (let i = 0; i < cap; i += 1) {
|
|
28
|
+
onStep?.(i + 1, cap);
|
|
29
|
+
const result = await llm.completeWithTools({
|
|
30
|
+
system,
|
|
31
|
+
tools,
|
|
32
|
+
messages: convo,
|
|
33
|
+
toolChoice: 'auto',
|
|
34
|
+
signal,
|
|
35
|
+
});
|
|
36
|
+
const calls = result?.tool_calls ?? [];
|
|
37
|
+
if (calls.length === 0) {
|
|
38
|
+
return {
|
|
39
|
+
content: result?.content ?? result?.message?.content ?? '',
|
|
40
|
+
iterations: i + 1,
|
|
41
|
+
capped: false,
|
|
42
|
+
};
|
|
43
|
+
}
|
|
44
|
+
convo.push(result.message ?? { role: 'assistant', content: result.content ?? '', tool_calls: calls });
|
|
45
|
+
// Tool calls within one turn are independent: dispatch concurrently, then
|
|
46
|
+
// replay results in the model's call order so the transcript stays stable.
|
|
47
|
+
const outcomes = await Promise.all(calls.map(async (call) => ({
|
|
48
|
+
tool_call_id: call.id,
|
|
49
|
+
content: await executeCall(call),
|
|
50
|
+
})));
|
|
51
|
+
for (const outcome of outcomes) {
|
|
52
|
+
convo.push({ role: 'tool', tool_call_id: outcome.tool_call_id, content: outcome.content });
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
return { content: '', iterations: cap, capped: true };
|
|
56
|
+
}
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import test from 'node:test';
|
|
3
|
+
import { runBoundedToolLoop } from './toolLoop.js';
|
|
4
|
+
|
|
5
|
+
function toolCall(id, name, args = '{}') {
|
|
6
|
+
return { id, function: { name, arguments: args } };
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
test('returns the model answer directly when no tool is called', async () => {
|
|
10
|
+
const llm = {
|
|
11
|
+
async completeWithTools() {
|
|
12
|
+
return { content: 'plain answer', tool_calls: [] };
|
|
13
|
+
},
|
|
14
|
+
};
|
|
15
|
+
const out = await runBoundedToolLoop({ llm, tools: [], executeCall: async () => 'unused' });
|
|
16
|
+
assert.deepEqual(out, { content: 'plain answer', iterations: 1, capped: false });
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
test('dispatches a tool call, feeds the result back, then returns the final answer', async () => {
|
|
20
|
+
let round = 0;
|
|
21
|
+
const seen = [];
|
|
22
|
+
const llm = {
|
|
23
|
+
async completeWithTools({ messages }) {
|
|
24
|
+
round += 1;
|
|
25
|
+
if (round === 1) return { message: { role: 'assistant', content: '', tool_calls: [toolCall('c1', 'cme__cme_status')] }, tool_calls: [toolCall('c1', 'cme__cme_status')] };
|
|
26
|
+
seen.push(messages.find((m) => m.role === 'tool')?.content);
|
|
27
|
+
return { content: 'configured', tool_calls: [] };
|
|
28
|
+
},
|
|
29
|
+
};
|
|
30
|
+
const out = await runBoundedToolLoop({
|
|
31
|
+
llm,
|
|
32
|
+
tools: [{ function: { name: 'cme__cme_status' } }],
|
|
33
|
+
executeCall: async (call) => `RESULT(${call.function.name})`,
|
|
34
|
+
});
|
|
35
|
+
assert.equal(out.content, 'configured');
|
|
36
|
+
assert.equal(out.iterations, 2);
|
|
37
|
+
assert.equal(out.capped, false);
|
|
38
|
+
assert.deepEqual(seen, ['RESULT(cme__cme_status)']);
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
test('runs concurrent tool calls and replays results in call order', async () => {
|
|
42
|
+
let round = 0;
|
|
43
|
+
const order = [];
|
|
44
|
+
const llm = {
|
|
45
|
+
async completeWithTools({ messages }) {
|
|
46
|
+
round += 1;
|
|
47
|
+
if (round === 1) {
|
|
48
|
+
const calls = [toolCall('a', 's__list'), toolCall('b', 's__status')];
|
|
49
|
+
return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
|
|
50
|
+
}
|
|
51
|
+
order.push(...messages.filter((m) => m.role === 'tool').map((m) => m.tool_call_id));
|
|
52
|
+
return { content: 'done', tool_calls: [] };
|
|
53
|
+
},
|
|
54
|
+
};
|
|
55
|
+
const out = await runBoundedToolLoop({
|
|
56
|
+
llm,
|
|
57
|
+
tools: [],
|
|
58
|
+
executeCall: async (call) => call.id,
|
|
59
|
+
});
|
|
60
|
+
assert.equal(out.content, 'done');
|
|
61
|
+
assert.deepEqual(order, ['a', 'b']); // preserved model call order
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
test('reports capped when the model keeps calling tools past the cap', async () => {
|
|
65
|
+
const llm = {
|
|
66
|
+
async completeWithTools() {
|
|
67
|
+
const calls = [toolCall('x', 's__status')];
|
|
68
|
+
return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
|
|
69
|
+
},
|
|
70
|
+
};
|
|
71
|
+
const out = await runBoundedToolLoop({ llm, tools: [], executeCall: async () => 'r', maxIterations: 3 });
|
|
72
|
+
assert.equal(out.capped, true);
|
|
73
|
+
assert.equal(out.iterations, 3);
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
test('propagates an abort thrown by executeCall', async () => {
|
|
77
|
+
const llm = {
|
|
78
|
+
async completeWithTools() {
|
|
79
|
+
const calls = [toolCall('x', 's__status')];
|
|
80
|
+
return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
|
|
81
|
+
},
|
|
82
|
+
};
|
|
83
|
+
const abort = Object.assign(new Error('aborted'), { name: 'AbortError' });
|
|
84
|
+
await assert.rejects(
|
|
85
|
+
runBoundedToolLoop({ llm, tools: [], executeCall: async () => { throw abort; } }),
|
|
86
|
+
/aborted/,
|
|
87
|
+
);
|
|
88
|
+
});
|
package/src/runtime/client.js
CHANGED
|
@@ -130,8 +130,9 @@ export async function postRuntimeKill({
|
|
|
130
130
|
token = runtimeToken(),
|
|
131
131
|
workspace = null,
|
|
132
132
|
runId = null,
|
|
133
|
+
purge = false,
|
|
133
134
|
} = {}) {
|
|
134
|
-
const response = await fetch(runtimeEndpointWithParams(url, '/kill', { workspace, runId }), {
|
|
135
|
+
const response = await fetch(runtimeEndpointWithParams(url, '/kill', { workspace, runId, ...(purge ? { purge: 'true' } : {}) }), {
|
|
135
136
|
method: 'POST',
|
|
136
137
|
headers: runtimeHeaders(token),
|
|
137
138
|
});
|
|
@@ -227,6 +227,9 @@ test('CME export is dispatched only from an approved DAG task', async () => {
|
|
|
227
227
|
let executeCalls = 0;
|
|
228
228
|
const session = baseSession({
|
|
229
229
|
workspace: 'demo-workspace',
|
|
230
|
+
// Production runs wait indefinitely for a human approval. This contract
|
|
231
|
+
// test is headless, so give the scheduler an explicit bounded deadline.
|
|
232
|
+
_approvalTimeoutMs: 10,
|
|
230
233
|
mcp: {
|
|
231
234
|
cme: {
|
|
232
235
|
status: 'connected',
|
package/src/runtime/lifecycle.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { execFile, spawn } from 'node:child_process';
|
|
2
|
-
import {
|
|
2
|
+
import { readdirSync, statSync } from 'node:fs';
|
|
3
|
+
import { dirname, join, resolve } from 'node:path';
|
|
3
4
|
import { fileURLToPath } from 'node:url';
|
|
4
5
|
import { activeCacertPath } from '../core/cacert.js';
|
|
5
6
|
import { checkRuntimeHealth, postRuntimeShutdown, runtimeUrlFromEnv } from './client.js';
|
|
@@ -10,6 +11,26 @@ const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
|
10
11
|
const managerRoot = resolve(__dirname, '../..');
|
|
11
12
|
const binPath = resolve(managerRoot, 'bin/wiki-manager.js');
|
|
12
13
|
|
|
14
|
+
// Newest mtime (ms) of the manager's own source tree. Used to detect that the
|
|
15
|
+
// code was edited after a reused runtime started, so ensureRuntime can restart
|
|
16
|
+
// it instead of serving stale code. Returns 0 if the source tree is unreadable
|
|
17
|
+
// (e.g. running from a packed install) — in that case staleness is not checked.
|
|
18
|
+
function newestManagerSourceMtimeMs() {
|
|
19
|
+
const srcDir = join(managerRoot, 'src');
|
|
20
|
+
let newest = 0;
|
|
21
|
+
try {
|
|
22
|
+
for (const entry of readdirSync(srcDir, { recursive: true })) {
|
|
23
|
+
const name = String(entry);
|
|
24
|
+
if (!(name.endsWith('.js') || name.endsWith('.ts') || name.endsWith('.tsx'))) continue;
|
|
25
|
+
try {
|
|
26
|
+
const mtime = statSync(join(srcDir, name)).mtimeMs;
|
|
27
|
+
if (mtime > newest) newest = mtime;
|
|
28
|
+
} catch { /* file vanished mid-scan */ }
|
|
29
|
+
}
|
|
30
|
+
} catch { return 0; }
|
|
31
|
+
return newest;
|
|
32
|
+
}
|
|
33
|
+
|
|
13
34
|
export function runtimeNodeExecutable() {
|
|
14
35
|
return process.versions.bun
|
|
15
36
|
? (process.env.WIKI_MANAGER_NODE_BIN ?? 'node')
|
|
@@ -47,13 +68,22 @@ export async function ensureRuntime({
|
|
|
47
68
|
if (existing) {
|
|
48
69
|
const expectedCacertPath = activeCacertPath();
|
|
49
70
|
const actualCacertPath = existing.cacertPath ? resolve(existing.cacertPath) : null;
|
|
71
|
+
// Dev staleness: if the manager source was edited after this runtime
|
|
72
|
+
// started, the reused process would keep serving old code (the recurring
|
|
73
|
+
// "my change is not taking effect" trap). Treat it as stale and restart.
|
|
74
|
+
// Packed installs report mtime 0 (unreadable src) → never flagged stale.
|
|
75
|
+
// Opt out with WIKI_MANAGER_RUNTIME_NO_STALE_CHECK=1.
|
|
76
|
+
const startedAtMs = Number(existing.startedAtMs) || 0;
|
|
77
|
+
const sourceMtimeMs = process.env.WIKI_MANAGER_RUNTIME_NO_STALE_CHECK === '1' ? 0 : newestManagerSourceMtimeMs();
|
|
78
|
+
const stale = startedAtMs > 0 && sourceMtimeMs > startedAtMs;
|
|
50
79
|
// forceRestart: the caller knows the manager configuration just changed
|
|
51
80
|
// (e.g. mcp.endpoints.json scaffolded on first run) — a runtime started
|
|
52
81
|
// BEFORE that only knows the old endpoints and would keep answering
|
|
53
82
|
// without the agents until manually restarted.
|
|
54
|
-
if (!forceRestart && actualCacertPath === expectedCacertPath) {
|
|
83
|
+
if (!forceRestart && !stale && actualCacertPath === expectedCacertPath) {
|
|
55
84
|
return { url, started: false, health: existing, token: auth.token, tokenPath: auth.tokenPath };
|
|
56
85
|
}
|
|
86
|
+
if (stale) console.error('runtime: source changed since start — restarting for fresh code.');
|
|
57
87
|
await postRuntimeShutdown({ url, token: auth.token });
|
|
58
88
|
await waitForRuntimeShutdown(url, auth.token, 2500);
|
|
59
89
|
}
|
package/src/runtime/runner.js
CHANGED
|
@@ -214,7 +214,7 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
|
|
|
214
214
|
},
|
|
215
215
|
}));
|
|
216
216
|
if (!evaluation.ok) {
|
|
217
|
-
if (
|
|
217
|
+
if (isAwaitingUserInputEvaluation(evaluation)) {
|
|
218
218
|
dispatchAgentEvent(session, createAgentEvent('assistant_message', {
|
|
219
219
|
origin: 'runtime',
|
|
220
220
|
runId,
|
|
@@ -820,7 +820,7 @@ export async function finishRuntimeRun(session, input, {
|
|
|
820
820
|
},
|
|
821
821
|
}));
|
|
822
822
|
if (!evaluation.ok) {
|
|
823
|
-
if (
|
|
823
|
+
if (isAwaitingUserInputEvaluation(evaluation)) {
|
|
824
824
|
dispatchAgentEvent(session, createAgentEvent('assistant_message', {
|
|
825
825
|
origin: 'runtime',
|
|
826
826
|
runId,
|
|
@@ -874,8 +874,10 @@ export async function evaluateRuntimeRun(session, input, {
|
|
|
874
874
|
system: [
|
|
875
875
|
'You are a strict evaluator for an agentic runtime run.',
|
|
876
876
|
'Inspect whether the original task was accomplished using the final plan and recent conversation.',
|
|
877
|
-
'Return only JSON with this exact shape: {"ok":boolean,"reason":"...","suggestedAction":string|null}.',
|
|
877
|
+
'Return only JSON with this exact shape: {"ok":boolean,"awaitingUserInput":boolean,"reason":"...","suggestedAction":string|null}.',
|
|
878
878
|
'Use ok=false only when a concrete missing action, failed requirement, or wrong result is visible.',
|
|
879
|
+
'Set awaitingUserInput=true when the run correctly stopped to obtain information or a decision only the user can provide (e.g. credentials, choosing between options, an explicit confirmation) — this is expected behaviour, NOT a failure or a missing action. In that case still set ok=false, put the pending question in reason, and do not treat the unasked configuration as a failed requirement.',
|
|
880
|
+
'Otherwise set awaitingUserInput=false.',
|
|
879
881
|
].join('\n'),
|
|
880
882
|
tools: [],
|
|
881
883
|
messages: [{ role: 'user', content: buildEvaluationPrompt(input, session, { runId }) }],
|
|
@@ -956,6 +958,7 @@ function parseJsonFenced(content, label = 'JSON response') {
|
|
|
956
958
|
function normalizeEvaluation(value) {
|
|
957
959
|
return {
|
|
958
960
|
ok: value?.ok === true,
|
|
961
|
+
awaitingUserInput: value?.awaitingUserInput === true,
|
|
959
962
|
reason: String(value?.reason ?? '').trim() || (value?.ok === true ? 'Task completed.' : 'Evaluator rejected the run.'),
|
|
960
963
|
suggestedAction: value?.suggestedAction == null ? null : String(value.suggestedAction),
|
|
961
964
|
};
|
|
@@ -969,6 +972,22 @@ function isUndefinedObjectiveEvaluation(evaluation) {
|
|
|
969
972
|
return /\b(vague|undefined|indefini|unclear|clarif|ambiguous|missing objective|no objective)\b/.test(text);
|
|
970
973
|
}
|
|
971
974
|
|
|
975
|
+
// A run that correctly hands back to the user for required input/decision is
|
|
976
|
+
// NOT a failed run \u2014 it is the expected end of an interactive turn. Treat it
|
|
977
|
+
// like the undefined-objective case: surface the pending question in chat and
|
|
978
|
+
// close the run cleanly (ok, run_done), never replan or emit run_error. The
|
|
979
|
+
// explicit evaluator field is authoritative; the keyword fallback covers models
|
|
980
|
+
// that answer in prose without emitting the flag.
|
|
981
|
+
function isAwaitingUserInputEvaluation(evaluation) {
|
|
982
|
+
if (evaluation?.awaitingUserInput === true) return true;
|
|
983
|
+
if (isUndefinedObjectiveEvaluation(evaluation)) return true;
|
|
984
|
+
const text = `${evaluation?.reason ?? ''} ${evaluation?.suggestedAction ?? ''}`
|
|
985
|
+
.toLowerCase()
|
|
986
|
+
.normalize('NFD')
|
|
987
|
+
.replace(/[\u0300-\u036f]/g, '');
|
|
988
|
+
return /\b(awaiting user|await user|user input|user decision|user confirmation|needs? (?:the )?user|requires? (?:the )?user|what to modify|which .* (?:to|should)|ask(?:ed|ing)? the user|attend .* utilisateur|demande .* utilisateur)\b/.test(text);
|
|
989
|
+
}
|
|
990
|
+
|
|
972
991
|
function clarificationMessageForEvaluation(evaluation) {
|
|
973
992
|
const reason = String(evaluation?.reason ?? '').trim();
|
|
974
993
|
return reason
|
package/src/runtime/server.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { createServer } from 'node:http';
|
|
2
2
|
import { randomUUID, timingSafeEqual } from 'node:crypto';
|
|
3
|
-
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
3
|
+
import { createAgentEvent, dispatchAgentEvent, resetSessionProjection } from '../core/agentEvents.js';
|
|
4
4
|
import { activeCacertPath } from '../core/cacert.js';
|
|
5
5
|
import { normalizePlanPatch, rebasePlanPatch } from '../core/planPatch.js';
|
|
6
6
|
import { validateContractInDev } from '../contracts/schemas.js';
|
|
@@ -25,6 +25,9 @@ export function startRuntimeServer({
|
|
|
25
25
|
exitOnShutdown = process.env.WIKI_MANAGER_RUNTIME_CHILD === '1',
|
|
26
26
|
} = {}) {
|
|
27
27
|
const clients = new Set();
|
|
28
|
+
// When this runtime process started — used by ensureRuntime to detect that
|
|
29
|
+
// the manager source has been edited since (dev staleness) and auto-restart.
|
|
30
|
+
const runtimeStartedAtMs = Date.now();
|
|
28
31
|
const defaultContext = { workspace: null, session, running: false, currentAbortController: null, currentRunId: null };
|
|
29
32
|
const resolvedGetContext = getContext ?? (() => defaultContext);
|
|
30
33
|
|
|
@@ -62,6 +65,7 @@ export function startRuntimeServer({
|
|
|
62
65
|
status: context?.running ? 'running' : 'idle',
|
|
63
66
|
workspace: context?.workspace ?? workspace ?? null,
|
|
64
67
|
activeRuns,
|
|
68
|
+
startedAtMs: runtimeStartedAtMs,
|
|
65
69
|
dbPath: store.dbPath,
|
|
66
70
|
cacertPath: activeCacertPath(),
|
|
67
71
|
nodeExtraCaCerts: process.env.NODE_EXTRA_CA_CERTS ?? null,
|
|
@@ -318,8 +322,9 @@ export function startRuntimeServer({
|
|
|
318
322
|
const body = await readJson(request);
|
|
319
323
|
const workspace = workspaceFromBody(body) ?? workspaceFromUrl(url);
|
|
320
324
|
const runId = url.searchParams.get('runId') ?? body.runId ?? null;
|
|
325
|
+
const purge = body.purge === true || url.searchParams.get('purge') === 'true';
|
|
321
326
|
const context = await resolveContext({ workspace });
|
|
322
|
-
const result = await killRuntimeRuns(context, { workspace, runId });
|
|
327
|
+
const result = await killRuntimeRuns(context, { workspace, runId, purge });
|
|
323
328
|
publishState(context.workspace ?? workspace ?? null, context);
|
|
324
329
|
sendJson(response, 202, result);
|
|
325
330
|
return;
|
|
@@ -405,7 +410,7 @@ export function startRuntimeServer({
|
|
|
405
410
|
return { body, workspace, context };
|
|
406
411
|
}
|
|
407
412
|
|
|
408
|
-
async function killRuntimeRuns(context, { workspace = null, runId = null } = {}) {
|
|
413
|
+
async function killRuntimeRuns(context, { workspace = null, runId = null, purge = false } = {}) {
|
|
409
414
|
const targetWorkspace = context?.workspace ?? workspace ?? null;
|
|
410
415
|
const targetRunId = runId ? String(runId) : null;
|
|
411
416
|
if (!targetRunId || targetRunId === context?.currentRunId) {
|
|
@@ -419,7 +424,18 @@ export function startRuntimeServer({
|
|
|
419
424
|
? store.cancelActiveTasksForInterruptedRuns({ workspace: targetWorkspace, runId: targetRunId })
|
|
420
425
|
: 0;
|
|
421
426
|
const queued = cancelQueuedControlItems(context?.session, targetWorkspace);
|
|
422
|
-
|
|
427
|
+
// Purge (/clear --all): interrupting recoverable runs does not clear a
|
|
428
|
+
// terminal 'error' run still projected in memory, nor the persisted event
|
|
429
|
+
// log. Empty both so PLAN/ACTIVITY/LOGS disappear and stay gone after a
|
|
430
|
+
// reboot. runId-scoped kills never purge (too coarse — it is workspace-wide).
|
|
431
|
+
let purged = null;
|
|
432
|
+
if (purge && !targetRunId) {
|
|
433
|
+
resetSessionProjection(context?.session);
|
|
434
|
+
purged = typeof store.clearWorkspaceState === 'function'
|
|
435
|
+
? store.clearWorkspaceState({ workspace: targetWorkspace })
|
|
436
|
+
: { runs: 0, events: 0, queue: 0 };
|
|
437
|
+
}
|
|
438
|
+
return { killed: true, workspace: targetWorkspace, runId: targetRunId, runs, tasks, queued, ...(purged !== null ? { purged } : {}) };
|
|
423
439
|
}
|
|
424
440
|
|
|
425
441
|
function startRuntimeRun(context, body, { controlItemId = null, waitForPlan = false } = {}) {
|
package/src/runtime/store.js
CHANGED
|
@@ -670,6 +670,45 @@ export function openRuntimeStore({ stateDir = defaultRuntimeStateDir(), fileName
|
|
|
670
670
|
return tasks.length;
|
|
671
671
|
}
|
|
672
672
|
|
|
673
|
+
// Full wipe of a workspace's persisted runtime state: runs (and, via the
|
|
674
|
+
// ON DELETE CASCADE foreign keys, their task_groups/tasks/task_dependencies/
|
|
675
|
+
// plan_revisions/approval_grants), the event log, and the queue. The
|
|
676
|
+
// non-cascading task_* side tables are cleared explicitly. This is what makes
|
|
677
|
+
// /clear --all survive a reboot: without it, hydrateSession replays the events
|
|
678
|
+
// and the PLAN/ACTIVITY/LOGS come straight back. Destructive by design — only
|
|
679
|
+
// reached on an explicit purge.
|
|
680
|
+
function clearWorkspaceState({ workspace = null } = {}) {
|
|
681
|
+
const runIds = (workspace
|
|
682
|
+
? db.prepare('SELECT id FROM runs WHERE workspace = ?').all(workspace)
|
|
683
|
+
: db.prepare('SELECT id FROM runs').all()
|
|
684
|
+
).map((row) => row.id);
|
|
685
|
+
db.exec('BEGIN');
|
|
686
|
+
try {
|
|
687
|
+
const delResults = db.prepare('DELETE FROM task_results WHERE task_id IN (SELECT id FROM tasks WHERE run_id = ?)');
|
|
688
|
+
const delAssignments = db.prepare('DELETE FROM task_assignments WHERE task_id IN (SELECT id FROM tasks WHERE run_id = ?)');
|
|
689
|
+
const delAttempts = db.prepare('DELETE FROM task_attempts WHERE run_id = ?');
|
|
690
|
+
for (const runId of runIds) {
|
|
691
|
+
delResults.run(runId);
|
|
692
|
+
delAssignments.run(runId);
|
|
693
|
+
delAttempts.run(runId);
|
|
694
|
+
}
|
|
695
|
+
const events = (workspace
|
|
696
|
+
? db.prepare('DELETE FROM events WHERE workspace = ?').run(workspace)
|
|
697
|
+
: db.prepare('DELETE FROM events').run()).changes ?? 0;
|
|
698
|
+
const queue = (workspace
|
|
699
|
+
? db.prepare('DELETE FROM queue_items WHERE workspace = ?').run(workspace)
|
|
700
|
+
: db.prepare('DELETE FROM queue_items').run()).changes ?? 0;
|
|
701
|
+
const runs = (workspace
|
|
702
|
+
? db.prepare('DELETE FROM runs WHERE workspace = ?').run(workspace)
|
|
703
|
+
: db.prepare('DELETE FROM runs').run()).changes ?? 0;
|
|
704
|
+
db.exec('COMMIT');
|
|
705
|
+
return { runs, events, queue };
|
|
706
|
+
} catch (error) {
|
|
707
|
+
db.exec('ROLLBACK');
|
|
708
|
+
throw error;
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
|
|
673
712
|
function saveQueue(queue = [], { workspace = null } = {}) {
|
|
674
713
|
const items = Array.isArray(queue) ? queue : [];
|
|
675
714
|
const now = new Date().toISOString();
|
|
@@ -1166,6 +1205,7 @@ export function openRuntimeStore({ stateDir = defaultRuntimeStateDir(), fileName
|
|
|
1166
1205
|
listRecoverableWorkspaces,
|
|
1167
1206
|
interruptRuns,
|
|
1168
1207
|
cancelActiveTasksForInterruptedRuns,
|
|
1208
|
+
clearWorkspaceState,
|
|
1169
1209
|
saveQueue,
|
|
1170
1210
|
listQueue,
|
|
1171
1211
|
listTasks,
|
package/src/shell/repl.js
CHANGED
|
@@ -5,12 +5,13 @@ import { execFileSync } from 'node:child_process';
|
|
|
5
5
|
import { stdin as input, stdout as output } from 'node:process';
|
|
6
6
|
import { marked } from 'marked';
|
|
7
7
|
import { markedTerminal } from 'marked-terminal';
|
|
8
|
-
import { buildAgentSystemPrompt, formatLlmUnavailableMessage } from '../agent/graph.js';
|
|
8
|
+
import { buildAgentSystemPrompt, formatLlmUnavailableMessage, isDonnaReadTool } from '../agent/graph.js';
|
|
9
9
|
import { handleSlashCommand } from '../commands/slash.js';
|
|
10
10
|
import { serviceDescription, serviceNames as composeServiceNames } from '../core/compose.js';
|
|
11
11
|
import { extractActivity, parseJsonText, sessionActivities } from '../core/activity.js';
|
|
12
12
|
import { syncActivitiesToPlan } from '../core/plan.js';
|
|
13
|
-
import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
|
|
13
|
+
import { buildLlmTools, callMcpTool, formatMcpToolResult, parseToolCallName, resolveToolCallName } from '../core/mcp.js';
|
|
14
|
+
import { runBoundedToolLoop } from '../core/toolLoop.js';
|
|
14
15
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
15
16
|
import { listSkills } from '../core/skills.js';
|
|
16
17
|
import { listWikircProfiles } from '../core/wikirc.js';
|
|
@@ -285,15 +286,38 @@ function isDonnaRole(role) {
|
|
|
285
286
|
return role === 'donna' || role === LEGACY_DONNA_ROLE;
|
|
286
287
|
}
|
|
287
288
|
|
|
289
|
+
// Read-only MCP tools exposed to /chat. The operator declares which tools per
|
|
290
|
+
// server in the `chatAccess` config (mcp.endpoints.json). /chat is extended to
|
|
291
|
+
// those tools ONLY, filtered through the same read-only test Donna's /agent
|
|
292
|
+
// mode uses (isDonnaReadTool), and only when their server is connected — so
|
|
293
|
+
// /chat can answer live state questions ("le CME est-il configuré ?") but can
|
|
294
|
+
// never mutate or delegate. Actions still belong to /agent, which has all tools.
|
|
295
|
+
export function chatReadTools(session) {
|
|
296
|
+
const servers = session?.chatAccess?.servers;
|
|
297
|
+
if (!servers) return [];
|
|
298
|
+
const scopedMcp = Object.fromEntries(
|
|
299
|
+
Object.entries(session.mcp ?? {}).filter(([name]) => Object.hasOwn(servers, name)),
|
|
300
|
+
);
|
|
301
|
+
return buildLlmTools(scopedMcp).filter((item) => {
|
|
302
|
+
const { server, tool } = parseToolCallName(item.function.name);
|
|
303
|
+
const entry = servers[server];
|
|
304
|
+
const declared = entry.allow === '*' ? true : (Array.isArray(entry.allow) && entry.allow.includes(tool));
|
|
305
|
+
return declared && isDonnaReadTool(item);
|
|
306
|
+
});
|
|
307
|
+
}
|
|
308
|
+
|
|
288
309
|
function buildDirectChatSystemPrompt(session) {
|
|
289
310
|
const workspace = session.workspace ?? 'no workspace selected';
|
|
290
311
|
const wikirc = session.wikirc?.profile ?? 'no profile loaded';
|
|
291
312
|
const language = session.language ?? 'en-US';
|
|
292
313
|
return [
|
|
293
|
-
'You are Donna, the llm-wiki-manager chat assistant.',
|
|
294
|
-
'
|
|
314
|
+
'You are Donna, the llm-wiki-manager chat assistant: warm, plain-spoken, and helpful — like an attentive colleague, never a raw status dump.',
|
|
315
|
+
'You have a small READ-ONLY toolset — the tools provided to you for this turn, which may be none. Use them to answer questions about live state (e.g. "le CME est-il configuré", "quelles pages sont en attente"), and answer only from their results.',
|
|
316
|
+
'If no provided tool covers the request — or the request is an action or mutation (ingest, build, export, configure, send, delete…), or needs a service that is not connected — say plainly you cannot do it in chat mode and to switch to agent mode (/agent). Do not pretend to execute it and never guess.',
|
|
317
|
+
'Answer directly and concisely. Do not claim to have called tools or changed files beyond the tools actually provided.',
|
|
318
|
+
'Chat mode is READ-ONLY, so never offer to perform an action yourself here — do NOT say "want me to start the ingestion?", because you cannot. That offer belongs to agent mode. When a natural next step is an action, you may warmly hand off instead, in one short line (in the reply language): e.g. "If you want to run the ingestion, switch to agent mode with /agent." Point the way; never promise to do it.',
|
|
295
319
|
'Never add a "Next steps", "Prochaines étapes", "À suivre", options, or suggestions section unless the user explicitly asks what to do next. End after answering the question.',
|
|
296
|
-
'
|
|
320
|
+
'Never invent a tool name, command, job id, status, or result (e.g. do not fabricate names like "check_cme_configuration"). If you cannot know something with the tools you were given, say so.',
|
|
297
321
|
`Reply language: ${language}.`,
|
|
298
322
|
`Current workspace: ${workspace}.`,
|
|
299
323
|
`Current wikirc profile: ${wikirc}.`,
|
|
@@ -1164,6 +1188,49 @@ async function runAgentTurn(input, { agent, session, onUpdate, onStep, displayIn
|
|
|
1164
1188
|
return {};
|
|
1165
1189
|
}
|
|
1166
1190
|
|
|
1191
|
+
// Bounded read-only tool loop for /chat. Only the tools declared in chatAccess
|
|
1192
|
+
// (and read-only) are offered; every call goes through callMcpTool, and any
|
|
1193
|
+
// tool the model names outside the offered set is refused — /chat can never
|
|
1194
|
+
// mutate or delegate. maxToolIterations caps the loop.
|
|
1195
|
+
async function runChatReadToolLoop({ input, session, history, donnaMessage, onUpdate, onStep, readTools }) {
|
|
1196
|
+
const allowed = new Set(readTools.map((item) => item.function.name));
|
|
1197
|
+
// /chat only supplies the policy; the loop mechanic lives in runBoundedToolLoop.
|
|
1198
|
+
// executeCall enforces the allow-list and turns each call into a text result;
|
|
1199
|
+
// it never mutates and refuses anything outside the offered read set.
|
|
1200
|
+
const executeCall = async (call) => {
|
|
1201
|
+
const rawName = call.function?.name ?? '';
|
|
1202
|
+
const { server, tool } = resolveToolCallName(session.mcp, rawName);
|
|
1203
|
+
const qualified = server ? `${server}__${tool}` : null;
|
|
1204
|
+
if (!qualified || !allowed.has(qualified)) {
|
|
1205
|
+
return `Refused: "${rawName}" is not an available read-only tool in chat mode. Actions and other tools live in agent mode (/agent).`;
|
|
1206
|
+
}
|
|
1207
|
+
let args = {};
|
|
1208
|
+
try { args = call.function?.arguments ? JSON.parse(call.function.arguments) : {}; } catch { args = {}; }
|
|
1209
|
+
try {
|
|
1210
|
+
onStep?.(`Chat: read ${server} ${tool}…`);
|
|
1211
|
+
const res = await callMcpTool(session.mcp, server, tool, args, session._abortSignal);
|
|
1212
|
+
return formatMcpToolResult(res);
|
|
1213
|
+
} catch (err) {
|
|
1214
|
+
if (err.name === 'AbortError' && session._abortSignal?.aborted) throw err;
|
|
1215
|
+
return `Error [${qualified}]: ${err instanceof Error ? err.message : String(err)}`;
|
|
1216
|
+
}
|
|
1217
|
+
};
|
|
1218
|
+
const { content, capped } = await runBoundedToolLoop({
|
|
1219
|
+
llm: session.llm,
|
|
1220
|
+
system: buildDirectChatSystemPrompt(session),
|
|
1221
|
+
messages: [...history, { role: 'user', content: input }],
|
|
1222
|
+
tools: readTools,
|
|
1223
|
+
executeCall,
|
|
1224
|
+
maxIterations: Math.min(8, Number(session?.chatAccess?.maxToolIterations) || 4),
|
|
1225
|
+
signal: session._abortSignal,
|
|
1226
|
+
onStep: (i, cap) => onStep?.(`Chat: consulting… [${i}/${cap}]`),
|
|
1227
|
+
});
|
|
1228
|
+
donnaMessage.content = capped
|
|
1229
|
+
? 'Je n’ai pas pu conclure dans la limite d’itérations du mode chat. Passe en /agent si besoin.'
|
|
1230
|
+
: (stripDsmlArtifacts(content).trimEnd() || formatLlmUnavailableMessage('reponse vide'));
|
|
1231
|
+
onUpdate?.();
|
|
1232
|
+
}
|
|
1233
|
+
|
|
1167
1234
|
async function runDirectChatTurn(input, { session, onUpdate, onStep }) {
|
|
1168
1235
|
if (!session.llm?.stream) {
|
|
1169
1236
|
conversationMessages(session).push({ role: 'command', content: directChatUnavailableText(session) });
|
|
@@ -1176,24 +1243,30 @@ async function runDirectChatTurn(input, { session, onUpdate, onStep }) {
|
|
|
1176
1243
|
const donnaMessage = { role: 'donna', content: '' };
|
|
1177
1244
|
messages.push(donnaMessage);
|
|
1178
1245
|
onUpdate?.();
|
|
1246
|
+
const readTools = chatReadTools(session);
|
|
1247
|
+
const canUseReadTools = readTools.length > 0 && typeof session.llm.completeWithTools === 'function';
|
|
1179
1248
|
try {
|
|
1180
|
-
|
|
1181
|
-
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1249
|
+
if (canUseReadTools) {
|
|
1250
|
+
await runChatReadToolLoop({ input, session, history, donnaMessage, onUpdate, onStep, readTools });
|
|
1251
|
+
} else {
|
|
1252
|
+
onStep?.('Chat: streaming direct answer…');
|
|
1253
|
+
for await (const delta of session.llm.stream({
|
|
1254
|
+
system: buildDirectChatSystemPrompt(session),
|
|
1255
|
+
messages: [...history, { role: 'user', content: input }],
|
|
1256
|
+
signal: session._abortSignal,
|
|
1257
|
+
})) {
|
|
1258
|
+
const cleanDelta = stripDsmlArtifacts(delta);
|
|
1259
|
+
if (cleanDelta) {
|
|
1260
|
+
donnaMessage.content += cleanDelta;
|
|
1261
|
+
onUpdate?.();
|
|
1262
|
+
}
|
|
1263
|
+
}
|
|
1264
|
+
donnaMessage.content = stripDsmlArtifacts(donnaMessage.content).trimEnd();
|
|
1265
|
+
if (!donnaMessage.content.trim()) {
|
|
1266
|
+
donnaMessage.content = formatLlmUnavailableMessage('flux vide');
|
|
1189
1267
|
onUpdate?.();
|
|
1190
1268
|
}
|
|
1191
1269
|
}
|
|
1192
|
-
donnaMessage.content = stripDsmlArtifacts(donnaMessage.content).trimEnd();
|
|
1193
|
-
if (!donnaMessage.content.trim()) {
|
|
1194
|
-
donnaMessage.content = formatLlmUnavailableMessage('flux vide');
|
|
1195
|
-
onUpdate?.();
|
|
1196
|
-
}
|
|
1197
1270
|
} catch (err) {
|
|
1198
1271
|
if (err.name === 'AbortError') {
|
|
1199
1272
|
messages.pop();
|
package/src/shell/repl.test.js
CHANGED
|
@@ -5,6 +5,7 @@ import { tmpdir } from 'node:os';
|
|
|
5
5
|
import { join } from 'node:path';
|
|
6
6
|
import {
|
|
7
7
|
applyRuntimeStateToShellSession,
|
|
8
|
+
chatReadTools,
|
|
8
9
|
createSession,
|
|
9
10
|
conversationMessages,
|
|
10
11
|
recordRuntimeUnavailableAgentInput,
|
|
@@ -373,3 +374,72 @@ test('submitRuntimeRun sends a control message instead of /run while a run is ac
|
|
|
373
374
|
restore();
|
|
374
375
|
}
|
|
375
376
|
});
|
|
377
|
+
|
|
378
|
+
test('chatReadTools exposes only declared, read-only MCP tools to /chat', () => {
|
|
379
|
+
const session = {
|
|
380
|
+
chatAccess: {
|
|
381
|
+
servers: {
|
|
382
|
+
cme: { allow: ['cme_status', 'cme_sources_list', 'cme_export_run'] },
|
|
383
|
+
},
|
|
384
|
+
},
|
|
385
|
+
mcp: {
|
|
386
|
+
cme: {
|
|
387
|
+
status: 'connected',
|
|
388
|
+
tools: [
|
|
389
|
+
{ name: 'cme_status', inputSchema: { type: 'object', properties: {} } },
|
|
390
|
+
{ name: 'cme_sources_list', inputSchema: { type: 'object', properties: {} } },
|
|
391
|
+
{ name: 'cme_setup', inputSchema: { type: 'object', properties: {} } },
|
|
392
|
+
{ name: 'cme_export_run', inputSchema: { type: 'object', properties: {} } },
|
|
393
|
+
],
|
|
394
|
+
},
|
|
395
|
+
documents: {
|
|
396
|
+
status: 'connected',
|
|
397
|
+
tools: [{ name: 'documents_status', inputSchema: { type: 'object', properties: {} } }],
|
|
398
|
+
},
|
|
399
|
+
},
|
|
400
|
+
};
|
|
401
|
+
const names = chatReadTools(session).map((item) => item.function.name).sort();
|
|
402
|
+
// cme_setup: not declared. cme_export_run: declared but a write (excluded by
|
|
403
|
+
// the read-only guard). documents_status: server absent from chatAccess.
|
|
404
|
+
assert.deepEqual(names, ['cme__cme_sources_list', 'cme__cme_status']);
|
|
405
|
+
});
|
|
406
|
+
|
|
407
|
+
test('chatReadTools is empty when no chatAccess is configured', () => {
|
|
408
|
+
const session = { mcp: { cme: { status: 'connected', tools: [{ name: 'cme_status', inputSchema: {} }] } } };
|
|
409
|
+
assert.deepEqual(chatReadTools(session), []);
|
|
410
|
+
});
|
|
411
|
+
|
|
412
|
+
test('/chat uses the tool-capable path when read tools are declared', async () => {
|
|
413
|
+
const session = createSession();
|
|
414
|
+
session.chatMode = true;
|
|
415
|
+
session.chatAccess = { maxToolIterations: 4, servers: { cme: { allow: ['cme_status'] } } };
|
|
416
|
+
session.mcp = { cme: { status: 'connected', tools: [{ name: 'cme_status', inputSchema: { type: 'object', properties: {} } }] } };
|
|
417
|
+
let usedComplete = false;
|
|
418
|
+
session.llm = {
|
|
419
|
+
async *stream() { yield 'STREAM_FALLBACK'; },
|
|
420
|
+
async completeWithTools() {
|
|
421
|
+
usedComplete = true;
|
|
422
|
+
return { tool_calls: [], content: 'Réponse via outils.', message: { role: 'assistant', content: 'Réponse via outils.' } };
|
|
423
|
+
},
|
|
424
|
+
};
|
|
425
|
+
await runLine('le cme est-il configuré', { session, chatMode: true });
|
|
426
|
+
const last = conversationMessages(session).at(-1);
|
|
427
|
+
assert.ok(usedComplete, 'completeWithTools path was taken');
|
|
428
|
+
assert.match(last.content, /Réponse via outils/);
|
|
429
|
+
assert.doesNotMatch(last.content, /STREAM_FALLBACK/);
|
|
430
|
+
});
|
|
431
|
+
|
|
432
|
+
test('/chat falls back to the plain stream when no read tools are declared', async () => {
|
|
433
|
+
const session = createSession();
|
|
434
|
+
session.chatMode = true;
|
|
435
|
+
session.chatAccess = null;
|
|
436
|
+
session.mcp = {};
|
|
437
|
+
session.llm = {
|
|
438
|
+
async *stream() { yield 'PLAIN_STREAM'; },
|
|
439
|
+
async completeWithTools() { return { tool_calls: [], content: 'SHOULD_NOT_APPEAR' }; },
|
|
440
|
+
};
|
|
441
|
+
await runLine('bonjour', { session, chatMode: true });
|
|
442
|
+
const last = conversationMessages(session).at(-1);
|
|
443
|
+
assert.match(last.content, /PLAIN_STREAM/);
|
|
444
|
+
assert.doesNotMatch(last.content, /SHOULD_NOT_APPEAR/);
|
|
445
|
+
});
|