@dotdrelle/wiki-manager 0.14.0 → 0.14.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/mcp.endpoints.example.json +7 -0
- package/package.json +2 -2
- package/src/agent/graph.js +33 -10
- package/src/cli/wiki-manager.js +9 -0
- package/src/core/agentEvents.js +7 -1
- package/src/core/agentEvents.test.js +13 -1
- package/src/core/buildInfo.json +2 -2
- package/src/core/mcp.js +35 -1
- package/src/core/toolLoop.js +56 -0
- package/src/core/toolLoop.test.js +88 -0
- package/src/runtime/lifecycle.js +32 -2
- package/src/runtime/server.js +4 -0
- package/src/shell/repl.js +90 -18
- package/src/shell/repl.test.js +70 -0
|
@@ -25,5 +25,12 @@
|
|
|
25
25
|
"Authorization": "Bearer ${DOCUMENTS_MCP_AUTH_TOKEN}"
|
|
26
26
|
}
|
|
27
27
|
}
|
|
28
|
+
},
|
|
29
|
+
"chatAccess": {
|
|
30
|
+
"maxToolIterations": 6,
|
|
31
|
+
"servers": {
|
|
32
|
+
"production": { "allow": ["production_job_status", "production_jobs_list"] },
|
|
33
|
+
"cme": { "allow": ["cme_status", "cme_sources_list", "cme_export_status"] }
|
|
34
|
+
}
|
|
28
35
|
}
|
|
29
36
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@dotdrelle/wiki-manager",
|
|
3
|
-
"version": "0.14.
|
|
3
|
+
"version": "0.14.1",
|
|
4
4
|
"description": "Agentic shell and orchestration cockpit for llm-wiki workspaces.",
|
|
5
5
|
"license": "PolyForm-Noncommercial-1.0.0",
|
|
6
6
|
"author": "dotrelle",
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
},
|
|
12
12
|
"scripts": {
|
|
13
13
|
"start": "bun ./bin/wiki-manager.js",
|
|
14
|
-
"test": "node --test src/agent/graph.test.js src/contracts/schemas.test.js src/core/activity.test.js src/core/buildInfo.test.js src/core/agentEvents.test.js src/core/runtimeLog.test.js src/activity/activityAggregator.test.js src/graph/runGraphProjector.test.js src/core/workflow.test.js src/core/planPatch.test.js src/core/agentLoop.test.js src/core/plan.test.js src/core/mcp.test.js src/core/documentIntake.test.js src/core/dockerCompose.test.js src/core/wikiWorkspace.test.js src/core/wikirc.test.js src/core/modelFetch.test.js src/core/startupCheck.test.js src/core/queueStore.test.js src/orchestrator/agentRegistry.test.js src/orchestrator/capabilityRegistry.test.js src/orchestrator/capabilityResolver.test.js src/orchestrator/planValidator.test.js src/orchestrator/planIntegrator.test.js src/orchestrator/scheduler.test.js src/orchestrator/attemptManager.test.js src/orchestrator/resultAggregator.test.js src/orchestrator/approvalPolicy.test.js src/commands/slash.test.js src/shell/repl.test.js src/runtime/store.test.js src/runtime/controlMessages.test.js src/runtime/recoveryManager.test.js src/runtime/server.test.js src/runtime/supervisor.test.js src/runtime/runner.test.js src/runtime/runner.e2e.test.js src/runtime/donna-contract.test.js src/runtime/auth.test.js",
|
|
14
|
+
"test": "node --test src/agent/graph.test.js src/contracts/schemas.test.js src/core/activity.test.js src/core/buildInfo.test.js src/core/agentEvents.test.js src/core/runtimeLog.test.js src/activity/activityAggregator.test.js src/graph/runGraphProjector.test.js src/core/workflow.test.js src/core/planPatch.test.js src/core/agentLoop.test.js src/core/plan.test.js src/core/mcp.test.js src/core/toolLoop.test.js src/core/documentIntake.test.js src/core/dockerCompose.test.js src/core/wikiWorkspace.test.js src/core/wikirc.test.js src/core/modelFetch.test.js src/core/startupCheck.test.js src/core/queueStore.test.js src/orchestrator/agentRegistry.test.js src/orchestrator/capabilityRegistry.test.js src/orchestrator/capabilityResolver.test.js src/orchestrator/planValidator.test.js src/orchestrator/planIntegrator.test.js src/orchestrator/scheduler.test.js src/orchestrator/attemptManager.test.js src/orchestrator/resultAggregator.test.js src/orchestrator/approvalPolicy.test.js src/commands/slash.test.js src/shell/repl.test.js src/runtime/store.test.js src/runtime/controlMessages.test.js src/runtime/recoveryManager.test.js src/runtime/server.test.js src/runtime/supervisor.test.js src/runtime/runner.test.js src/runtime/runner.e2e.test.js src/runtime/donna-contract.test.js src/runtime/auth.test.js",
|
|
15
15
|
"check-versions": "node scripts/check-versions.js",
|
|
16
16
|
"prepack": "node scripts/check-versions.js",
|
|
17
17
|
"prepublishOnly": "node scripts/check-versions.js",
|
package/src/agent/graph.js
CHANGED
|
@@ -711,7 +711,7 @@ async function handleRuntimeControlTool(session, tool, args = {}) {
|
|
|
711
711
|
if (!objective) return 'Delegation rejected: missing objective.';
|
|
712
712
|
const result = await postRuntimeDelegate(objective, { url, workspace });
|
|
713
713
|
return result?.runId
|
|
714
|
-
? `
|
|
714
|
+
? `Action lancée (${String(result.runId).slice(0, 8)}) après validation du plan réel : ${result.delegation?.tasks ?? 0} tâche(s), ${result.delegation?.agent ?? 'agent résolu'}. Exécution en cours.`
|
|
715
715
|
: `Délégation refusée : ${result?.error ?? JSON.stringify(result)}`;
|
|
716
716
|
}
|
|
717
717
|
if (tool === 'enqueue') {
|
|
@@ -845,10 +845,7 @@ export function buildAgentSystemPrompt(state) {
|
|
|
845
845
|
// mutating provider tools (e.g. production__production_start_job) here teaches
|
|
846
846
|
// a capable model to invoke them directly and bypass runtime__delegate.
|
|
847
847
|
const mcpTools = formatMcpToolsForAgent(state.session.mcp, {
|
|
848
|
-
include: (qualifiedName
|
|
849
|
-
function: { name: qualifiedName },
|
|
850
|
-
readOnly: tool?.annotations?.readOnlyHint === true,
|
|
851
|
-
}),
|
|
848
|
+
include: (qualifiedName) => !isOrchestrationBypassTool(qualifiedName),
|
|
852
849
|
});
|
|
853
850
|
const skills = formatSkillsForAgent(state.session);
|
|
854
851
|
const customPrompt = state.session.systemPrompt ?? null;
|
|
@@ -864,7 +861,7 @@ export function buildAgentSystemPrompt(state) {
|
|
|
864
861
|
`Current wikirc profile: ${wikirc}.`,
|
|
865
862
|
`Available primitives: ${commandList(state.session)}.`,
|
|
866
863
|
'Only announce or call slash commands that appear exactly in Available primitives. Do not invent command names, subcommands, or arguments.',
|
|
867
|
-
'Connected
|
|
864
|
+
'Connected MCP tools you may call directly (server__tool naming convention) — reads AND single-step actions like configuring or adding a connector source, converting a document, sending, or searching. Only the heavy multi-step operations (ingest, build, export, polish, pipeline) go through runtime__delegate to get their parallel plan. Everything listed below is directly callable:',
|
|
868
865
|
mcpTools,
|
|
869
866
|
'Current local MCP job queue:',
|
|
870
867
|
formatQueue(state.session),
|
|
@@ -877,7 +874,7 @@ export function buildAgentSystemPrompt(state) {
|
|
|
877
874
|
'Never add a "Next steps", "Prochaines étapes", "À suivre", options, or suggestions section unless the user explicitly asks what to do next. End after the requested result or the concrete error.',
|
|
878
875
|
'When calling a tool, emit no preliminary narration. Call it directly; the PLAN and Activity panels show progress. After completion, keep the final response concise and proportional to the result.',
|
|
879
876
|
'Keep every response synthetic and information-dense. Use only the lines needed, and never exceed roughly 15 to 20 short lines even for a detailed answer. Never expose internal reasoning, repeated checks, tool-selection commentary, or a chronological diary. Prioritize the result, essential facts, concrete errors, and actual outputs.',
|
|
880
|
-
'
|
|
877
|
+
'Only the heavy multi-step operations — ingest, build, export, polish, pipeline — are delegated via runtime__delegate (for their DAG and parallelism). Single-step actions — configuring or adding a connector source, converting a document, sending, searching — are called directly on the connected tool. Never call an agent orchestration-contract or plan tool directly.',
|
|
881
878
|
'For any question about the current workspace inventory or what is waiting there, call wiki__wiki_workspace_status first and answer only from its result. This is the canonical read-only workspace state; do not reconstruct it from upload, connector, or production tools.',
|
|
882
879
|
'Tool identifiers are private implementation details. Never print MCP tool names such as server__tool in a user-facing answer. Describe the human result instead.',
|
|
883
880
|
'Never suggest a manual filesystem command or implementation workaround unless the user explicitly asks for manual instructions. For an action request, delegate the objective and let the specialized agent determine paths and operations from its live contract.',
|
|
@@ -887,6 +884,7 @@ export function buildAgentSystemPrompt(state) {
|
|
|
887
884
|
state.session.runtime?.url
|
|
888
885
|
? 'The runtime is connected and runtime__delegate is bound and available to you right now — it is a tool you call directly, not a slash command or a missing primitive. It is the ONLY way to execute an action (ingest, build, export, configure, send…). Never tell the user that delegation or the runtime is unavailable while it is connected; call runtime__delegate instead.'
|
|
889
886
|
: 'No runtime is connected, so you cannot execute actions. State that plainly and name the runtime connection as the missing capability — do not invent a workaround.',
|
|
887
|
+
'If the connector or service needed for a requested read or action is absent from the Connected MCP tools above (its service is not running — e.g. CME, documents, or production), say plainly that this service is not connected and name it as the missing capability. Never redirect a simple read (e.g. "give me the CME config") to an "agent action", never invent its result, and never propose a workaround. Only requests you can actually serve with a listed tool are answered with data.',
|
|
890
888
|
'For any requested action, call runtime__delegate with the user objective only. Never choose a capability, operation, agent, plan, or implementation yourself. The runtime resolves the registry and validates the provider plan before accepting. Never call <provider>__agent_plan, <provider>__agent_execute, legacy production__production_start_job, wiki__plan_set, or wiki__plan_done from interactive chat.',
|
|
891
889
|
'Do not ask the user which sources, files, connectors, or templates to use for an ingest, build, or export: the specialized agent discovers them from the workspace. When the objective is clear (e.g. "lance une ingestion"), delegate it as stated, without a clarifying question.',
|
|
892
890
|
'If runtime__delegate returns a blocker or no specialized provider is available, report only that concrete blocker concisely. Never replace the missing execution path with a suggested slash command, skill, MCP tool name, manual file move, administrator escalation, or alternative workflow unless the user explicitly asks for alternatives.',
|
|
@@ -944,13 +942,20 @@ function toolsForClassification(classification, writeTools, session = null) {
|
|
|
944
942
|
return [SHELL_READ_COMMAND_TOOL, ...controlTools];
|
|
945
943
|
}
|
|
946
944
|
if (session?.runtime?.url) {
|
|
947
|
-
|
|
948
|
-
|
|
945
|
+
// Offer every connected tool directly EXCEPT orchestration-bypass tools
|
|
946
|
+
// and raw shell write/profile mutation. Reads, configuration, connector
|
|
947
|
+
// setup — and any newly added MCP's tools — stay directly callable.
|
|
948
|
+
const directTools = writeTools.filter((item) => {
|
|
949
|
+
const name = item?.function?.name;
|
|
950
|
+
if (!name || name === 'shell__run_command' || name === 'shell__profile_update') return false;
|
|
951
|
+
return !isOrchestrationBypassTool(name);
|
|
952
|
+
});
|
|
953
|
+
return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...directTools];
|
|
949
954
|
}
|
|
950
955
|
return [SHELL_READ_COMMAND_TOOL, ...controlTools, ...capabilityRunTools, ...writeTools];
|
|
951
956
|
}
|
|
952
957
|
|
|
953
|
-
function isDonnaReadTool(item) {
|
|
958
|
+
export function isDonnaReadTool(item) {
|
|
954
959
|
const name = String(item?.function?.name ?? '');
|
|
955
960
|
if (!name || name.startsWith('shell__') || name === 'wiki__plan_set' || name === 'wiki__plan_done') return false;
|
|
956
961
|
if (item?.readOnly === true) return true;
|
|
@@ -961,6 +966,24 @@ function isDonnaReadTool(item) {
|
|
|
961
966
|
|| /(?:^|_)(?:status|list|search|read|get)$/.test(tool);
|
|
962
967
|
}
|
|
963
968
|
|
|
969
|
+
// Two-tier tool policy. Donna may call any connected MCP tool directly
|
|
970
|
+
// (reads AND plain writes: cme_setup, connector setup, document conversion,
|
|
971
|
+
// send, search, and anything a newly added MCP exposes) EXCEPT the small set
|
|
972
|
+
// that must go through the runtime's orchestration: the universal five-tool
|
|
973
|
+
// contract executors (agent_plan/agent_execute), the legacy job starter, and
|
|
974
|
+
// direct plan mutation. Heavy multi-step work (ingest/build/export via the
|
|
975
|
+
// production agent) is delegated for its DAG/parallelism; plain single-step
|
|
976
|
+
// tools are called directly. This is a blocklist, not a whitelist, so adding a
|
|
977
|
+
// new MCP never silently disables its tools.
|
|
978
|
+
function isOrchestrationBypassTool(name) {
|
|
979
|
+
const full = String(name ?? '');
|
|
980
|
+
if (!full) return true;
|
|
981
|
+
if (full === 'wiki__plan_set' || full === 'wiki__plan_done') return true;
|
|
982
|
+
const sep = full.indexOf('__');
|
|
983
|
+
const tool = sep === -1 ? full : full.slice(sep + 2);
|
|
984
|
+
return tool === 'agent_plan' || tool === 'agent_execute' || tool === 'production_start_job';
|
|
985
|
+
}
|
|
986
|
+
|
|
964
987
|
function isReadOnlyMcpCall(session, server, tool) {
|
|
965
988
|
const descriptor = (session?.mcp?.[server]?.tools ?? [])
|
|
966
989
|
.find((item) => String(item?.name ?? '') === tool || String(item?.name ?? '').endsWith(`__${tool}`));
|
package/src/cli/wiki-manager.js
CHANGED
|
@@ -847,6 +847,15 @@ async function runRuntime(argv, agent) {
|
|
|
847
847
|
throw new Error(`Delegated plan integration failed: ${(integrated.errors ?? []).map((error) => error.message ?? error.code ?? String(error)).join('; ')}`);
|
|
848
848
|
}
|
|
849
849
|
emitRuntimeLog(session, `delegation: ${prepared.fragment.tasks.length} validated task(s) integrated from ${prepared.provider.serverName}.agent_plan (${prepared.capability}/${prepared.operation})`);
|
|
850
|
+
// Demandé = consenti: a directly-delegated run carries the user's
|
|
851
|
+
// explicit consent, so auto-approve its initial plan. Persisting a
|
|
852
|
+
// run-scope grant (via the approval manager) makes the scheduler's
|
|
853
|
+
// readyTasks approval check pass, so the tasks run without re-prompting.
|
|
854
|
+
// Replanned tasks are integrated later without a fresh grant.
|
|
855
|
+
if (context.approvalManager?.approve) {
|
|
856
|
+
context.approvalManager.approve({ scope: 'run', runId });
|
|
857
|
+
emitRuntimeLog(session, `approval: run ${runId} auto-approved (user-requested action)`);
|
|
858
|
+
}
|
|
850
859
|
body._planReady?.resolve?.({ runId, planRevision: session.agentProjection?.planRevision ?? 0 });
|
|
851
860
|
}
|
|
852
861
|
// Deterministic capability run (/ingest): ask the capable agent for its
|
package/src/core/agentEvents.js
CHANGED
|
@@ -196,7 +196,13 @@ export function applyAgentProjectionToSession(session, projection) {
|
|
|
196
196
|
function applyEvent(state, event) {
|
|
197
197
|
switch (event.type) {
|
|
198
198
|
case 'run_started':
|
|
199
|
-
|
|
199
|
+
// Only a real runtime run (origin 'runtime') marks the projection
|
|
200
|
+
// 'running'. An interactive turn (origin 'user') is NOT a run: forcing
|
|
201
|
+
// 'running' here made the graph classify activeRun=true and hide Donna's
|
|
202
|
+
// MCP read tools (cme_status, wiki_workspace_status…), so questions about
|
|
203
|
+
// MCP state failed. Preserve the existing status (idle, or a genuinely
|
|
204
|
+
// active runtime run synced from the runtime) for interactive turns.
|
|
205
|
+
if (event.origin === 'runtime') state.status = 'running';
|
|
200
206
|
state.plan = null;
|
|
201
207
|
state.chain = [];
|
|
202
208
|
state.activities = {};
|
|
@@ -20,13 +20,25 @@ test('reduceAgentEvents: run_started clears stale plan', () => {
|
|
|
20
20
|
origin: 'tool',
|
|
21
21
|
payload: { steps: ['Old action'] },
|
|
22
22
|
}),
|
|
23
|
-
createAgentEvent('run_started', { origin: '
|
|
23
|
+
createAgentEvent('run_started', { origin: 'runtime' }),
|
|
24
24
|
]);
|
|
25
25
|
assert.equal(projection.plan, null);
|
|
26
26
|
assert.equal(projection.activities.length, 0);
|
|
27
27
|
assert.equal(projection.status, 'running');
|
|
28
28
|
});
|
|
29
29
|
|
|
30
|
+
test('reduceAgentEvents: interactive (user) run_started clears state but is not a running run', () => {
|
|
31
|
+
const projection = reduceAgentEvents([
|
|
32
|
+
createAgentEvent('plan_set', { origin: 'tool', payload: { steps: ['Old action'] } }),
|
|
33
|
+
createAgentEvent('run_started', { origin: 'user' }),
|
|
34
|
+
]);
|
|
35
|
+
// An interactive turn clears stale plan/activities but must NOT mark the
|
|
36
|
+
// projection 'running' — otherwise the graph classifies activeRun=true and
|
|
37
|
+
// hides Donna's MCP read tools.
|
|
38
|
+
assert.equal(projection.plan, null);
|
|
39
|
+
assert.notEqual(projection.status, 'running');
|
|
40
|
+
});
|
|
41
|
+
|
|
30
42
|
test('reduceAgentEvents: tracks manual plan and step updates', () => {
|
|
31
43
|
const projection = reduceAgentEvents([
|
|
32
44
|
createAgentEvent('plan_set', {
|
package/src/core/buildInfo.json
CHANGED
package/src/core/mcp.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { existsSync, readFileSync } from 'node:fs';
|
|
2
2
|
import { managerEnvFile, managerMcpEndpointsFile, readEnvFile } from './env.js';
|
|
3
3
|
|
|
4
|
-
const WIKI_MANAGER_VERSION = '0.14.
|
|
4
|
+
const WIKI_MANAGER_VERSION = '0.14.1';
|
|
5
5
|
|
|
6
6
|
function envValue(key) {
|
|
7
7
|
const filePath = managerEnvFile();
|
|
@@ -43,6 +43,29 @@ function normalizeExternalUrlForRuntime(url) {
|
|
|
43
43
|
return url;
|
|
44
44
|
}
|
|
45
45
|
|
|
46
|
+
// Config-driven policy for the /chat read-only toolset — NOT /agent, which has
|
|
47
|
+
// the full toolset and ignores this. The endpoints file's "chatAccess" block
|
|
48
|
+
// declares, per server, which tools /chat may call ("*" or a list), plus a
|
|
49
|
+
// maxToolIterations budget. Operator-owned, agnostic allow-list. Returns null
|
|
50
|
+
// when not configured — then /chat stays a plain, tool-less conversation.
|
|
51
|
+
export function readChatAccessConfig() {
|
|
52
|
+
const filePath = managerMcpEndpointsFile();
|
|
53
|
+
if (!existsSync(filePath)) return null;
|
|
54
|
+
let raw;
|
|
55
|
+
try { raw = JSON.parse(readFileSync(filePath, 'utf8')); } catch { return null; }
|
|
56
|
+
const chatAccess = raw?.chatAccess;
|
|
57
|
+
if (!chatAccess || typeof chatAccess !== 'object' || Array.isArray(chatAccess)) return null;
|
|
58
|
+
const servers = {};
|
|
59
|
+
for (const [name, entry] of Object.entries(chatAccess.servers ?? {})) {
|
|
60
|
+
if (entry?.allow === '*') servers[name] = { allow: '*' };
|
|
61
|
+
else if (Array.isArray(entry?.allow)) servers[name] = { allow: entry.allow.map(String).filter(Boolean) };
|
|
62
|
+
}
|
|
63
|
+
const maxToolIterations = Number.isFinite(Number(chatAccess.maxToolIterations)) && Number(chatAccess.maxToolIterations) > 0
|
|
64
|
+
? Math.floor(Number(chatAccess.maxToolIterations))
|
|
65
|
+
: null;
|
|
66
|
+
return { maxToolIterations, servers };
|
|
67
|
+
}
|
|
68
|
+
|
|
46
69
|
function readExternalMcpEndpoints() {
|
|
47
70
|
const filePath = managerMcpEndpointsFile();
|
|
48
71
|
if (!existsSync(filePath)) return {};
|
|
@@ -59,6 +82,14 @@ function readExternalMcpEndpoints() {
|
|
|
59
82
|
url: normalizeExternalUrlForRuntime(interpolateEnv(String(endpoint.url))),
|
|
60
83
|
configuredUrl: interpolateEnv(String(endpoint.url)),
|
|
61
84
|
headers: normalizeHeaders(endpoint.headers),
|
|
85
|
+
// Tools the endpoint marks approval-gated: Donna may still call them
|
|
86
|
+
// directly (they are single-step tools), but toolRequiresApproval
|
|
87
|
+
// makes the call wait for the user's confirmation first (e.g. a
|
|
88
|
+
// destructive cme_export_run). Agent/operator owned — no hard-coded
|
|
89
|
+
// business name in the manager.
|
|
90
|
+
requireApproval: Array.isArray(endpoint.requireApproval)
|
|
91
|
+
? endpoint.requireApproval.map(String).filter(Boolean)
|
|
92
|
+
: undefined,
|
|
62
93
|
external: true,
|
|
63
94
|
},
|
|
64
95
|
]),
|
|
@@ -95,6 +126,9 @@ const DEFAULT_MCP_RETRY_POLICY = {
|
|
|
95
126
|
};
|
|
96
127
|
|
|
97
128
|
export function buildMcpStatus(session) {
|
|
129
|
+
// Attach the /chat read-tool policy to the session alongside MCP status.
|
|
130
|
+
// Only /chat (repl.js) reads session.chatAccess; /agent ignores it.
|
|
131
|
+
if (session) session.chatAccess = readChatAccessConfig();
|
|
98
132
|
const workspaceEnv = session.workspaceEnv ?? {};
|
|
99
133
|
const wikiMcpToken = session.wikircConfig?.mcp?.accessKey;
|
|
100
134
|
const wikiMcpDetail = workspaceEnv.WIKI_MCP_PORT
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
// Minimal, side-effect-free bounded tool-use loop.
|
|
2
|
+
//
|
|
3
|
+
// This is the shared mechanic of "ask the LLM with a tool set, run the tool
|
|
4
|
+
// calls it emits, feed results back, repeat up to a cap". The caller injects
|
|
5
|
+
// the ONLY policy that varies: `executeCall(call) -> string` decides whether a
|
|
6
|
+
// requested tool is allowed and produces its textual result (allow-list check,
|
|
7
|
+
// MCP dispatch, error formatting). The loop itself owns no plan, no delegation,
|
|
8
|
+
// no run identity and no agent events — deliberately unlike the /agent
|
|
9
|
+
// orchestration loop in createAgentGraph, which is a stateful LangGraph node
|
|
10
|
+
// graph and stays separate. Use this for stateless tool-answer turns (e.g.
|
|
11
|
+
// /chat read-only questions).
|
|
12
|
+
//
|
|
13
|
+
// `executeCall` may throw to abort the whole loop (e.g. an AbortError on
|
|
14
|
+
// cancel); anything it returns is treated as the tool result for that call.
|
|
15
|
+
export async function runBoundedToolLoop({
|
|
16
|
+
llm,
|
|
17
|
+
system,
|
|
18
|
+
messages,
|
|
19
|
+
tools,
|
|
20
|
+
executeCall,
|
|
21
|
+
maxIterations = 4,
|
|
22
|
+
signal,
|
|
23
|
+
onStep,
|
|
24
|
+
} = {}) {
|
|
25
|
+
const cap = Math.max(1, Math.floor(maxIterations) || 1);
|
|
26
|
+
const convo = [...(messages ?? [])];
|
|
27
|
+
for (let i = 0; i < cap; i += 1) {
|
|
28
|
+
onStep?.(i + 1, cap);
|
|
29
|
+
const result = await llm.completeWithTools({
|
|
30
|
+
system,
|
|
31
|
+
tools,
|
|
32
|
+
messages: convo,
|
|
33
|
+
toolChoice: 'auto',
|
|
34
|
+
signal,
|
|
35
|
+
});
|
|
36
|
+
const calls = result?.tool_calls ?? [];
|
|
37
|
+
if (calls.length === 0) {
|
|
38
|
+
return {
|
|
39
|
+
content: result?.content ?? result?.message?.content ?? '',
|
|
40
|
+
iterations: i + 1,
|
|
41
|
+
capped: false,
|
|
42
|
+
};
|
|
43
|
+
}
|
|
44
|
+
convo.push(result.message ?? { role: 'assistant', content: result.content ?? '', tool_calls: calls });
|
|
45
|
+
// Tool calls within one turn are independent: dispatch concurrently, then
|
|
46
|
+
// replay results in the model's call order so the transcript stays stable.
|
|
47
|
+
const outcomes = await Promise.all(calls.map(async (call) => ({
|
|
48
|
+
tool_call_id: call.id,
|
|
49
|
+
content: await executeCall(call),
|
|
50
|
+
})));
|
|
51
|
+
for (const outcome of outcomes) {
|
|
52
|
+
convo.push({ role: 'tool', tool_call_id: outcome.tool_call_id, content: outcome.content });
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
return { content: '', iterations: cap, capped: true };
|
|
56
|
+
}
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import test from 'node:test';
|
|
3
|
+
import { runBoundedToolLoop } from './toolLoop.js';
|
|
4
|
+
|
|
5
|
+
function toolCall(id, name, args = '{}') {
|
|
6
|
+
return { id, function: { name, arguments: args } };
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
test('returns the model answer directly when no tool is called', async () => {
|
|
10
|
+
const llm = {
|
|
11
|
+
async completeWithTools() {
|
|
12
|
+
return { content: 'plain answer', tool_calls: [] };
|
|
13
|
+
},
|
|
14
|
+
};
|
|
15
|
+
const out = await runBoundedToolLoop({ llm, tools: [], executeCall: async () => 'unused' });
|
|
16
|
+
assert.deepEqual(out, { content: 'plain answer', iterations: 1, capped: false });
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
test('dispatches a tool call, feeds the result back, then returns the final answer', async () => {
|
|
20
|
+
let round = 0;
|
|
21
|
+
const seen = [];
|
|
22
|
+
const llm = {
|
|
23
|
+
async completeWithTools({ messages }) {
|
|
24
|
+
round += 1;
|
|
25
|
+
if (round === 1) return { message: { role: 'assistant', content: '', tool_calls: [toolCall('c1', 'cme__cme_status')] }, tool_calls: [toolCall('c1', 'cme__cme_status')] };
|
|
26
|
+
seen.push(messages.find((m) => m.role === 'tool')?.content);
|
|
27
|
+
return { content: 'configured', tool_calls: [] };
|
|
28
|
+
},
|
|
29
|
+
};
|
|
30
|
+
const out = await runBoundedToolLoop({
|
|
31
|
+
llm,
|
|
32
|
+
tools: [{ function: { name: 'cme__cme_status' } }],
|
|
33
|
+
executeCall: async (call) => `RESULT(${call.function.name})`,
|
|
34
|
+
});
|
|
35
|
+
assert.equal(out.content, 'configured');
|
|
36
|
+
assert.equal(out.iterations, 2);
|
|
37
|
+
assert.equal(out.capped, false);
|
|
38
|
+
assert.deepEqual(seen, ['RESULT(cme__cme_status)']);
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
test('runs concurrent tool calls and replays results in call order', async () => {
|
|
42
|
+
let round = 0;
|
|
43
|
+
const order = [];
|
|
44
|
+
const llm = {
|
|
45
|
+
async completeWithTools({ messages }) {
|
|
46
|
+
round += 1;
|
|
47
|
+
if (round === 1) {
|
|
48
|
+
const calls = [toolCall('a', 's__list'), toolCall('b', 's__status')];
|
|
49
|
+
return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
|
|
50
|
+
}
|
|
51
|
+
order.push(...messages.filter((m) => m.role === 'tool').map((m) => m.tool_call_id));
|
|
52
|
+
return { content: 'done', tool_calls: [] };
|
|
53
|
+
},
|
|
54
|
+
};
|
|
55
|
+
const out = await runBoundedToolLoop({
|
|
56
|
+
llm,
|
|
57
|
+
tools: [],
|
|
58
|
+
executeCall: async (call) => call.id,
|
|
59
|
+
});
|
|
60
|
+
assert.equal(out.content, 'done');
|
|
61
|
+
assert.deepEqual(order, ['a', 'b']); // preserved model call order
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
test('reports capped when the model keeps calling tools past the cap', async () => {
|
|
65
|
+
const llm = {
|
|
66
|
+
async completeWithTools() {
|
|
67
|
+
const calls = [toolCall('x', 's__status')];
|
|
68
|
+
return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
|
|
69
|
+
},
|
|
70
|
+
};
|
|
71
|
+
const out = await runBoundedToolLoop({ llm, tools: [], executeCall: async () => 'r', maxIterations: 3 });
|
|
72
|
+
assert.equal(out.capped, true);
|
|
73
|
+
assert.equal(out.iterations, 3);
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
test('propagates an abort thrown by executeCall', async () => {
|
|
77
|
+
const llm = {
|
|
78
|
+
async completeWithTools() {
|
|
79
|
+
const calls = [toolCall('x', 's__status')];
|
|
80
|
+
return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
|
|
81
|
+
},
|
|
82
|
+
};
|
|
83
|
+
const abort = Object.assign(new Error('aborted'), { name: 'AbortError' });
|
|
84
|
+
await assert.rejects(
|
|
85
|
+
runBoundedToolLoop({ llm, tools: [], executeCall: async () => { throw abort; } }),
|
|
86
|
+
/aborted/,
|
|
87
|
+
);
|
|
88
|
+
});
|
package/src/runtime/lifecycle.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { execFile, spawn } from 'node:child_process';
|
|
2
|
-
import {
|
|
2
|
+
import { readdirSync, statSync } from 'node:fs';
|
|
3
|
+
import { dirname, join, resolve } from 'node:path';
|
|
3
4
|
import { fileURLToPath } from 'node:url';
|
|
4
5
|
import { activeCacertPath } from '../core/cacert.js';
|
|
5
6
|
import { checkRuntimeHealth, postRuntimeShutdown, runtimeUrlFromEnv } from './client.js';
|
|
@@ -10,6 +11,26 @@ const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
|
10
11
|
const managerRoot = resolve(__dirname, '../..');
|
|
11
12
|
const binPath = resolve(managerRoot, 'bin/wiki-manager.js');
|
|
12
13
|
|
|
14
|
+
// Newest mtime (ms) of the manager's own source tree. Used to detect that the
|
|
15
|
+
// code was edited after a reused runtime started, so ensureRuntime can restart
|
|
16
|
+
// it instead of serving stale code. Returns 0 if the source tree is unreadable
|
|
17
|
+
// (e.g. running from a packed install) — in that case staleness is not checked.
|
|
18
|
+
function newestManagerSourceMtimeMs() {
|
|
19
|
+
const srcDir = join(managerRoot, 'src');
|
|
20
|
+
let newest = 0;
|
|
21
|
+
try {
|
|
22
|
+
for (const entry of readdirSync(srcDir, { recursive: true })) {
|
|
23
|
+
const name = String(entry);
|
|
24
|
+
if (!(name.endsWith('.js') || name.endsWith('.ts') || name.endsWith('.tsx'))) continue;
|
|
25
|
+
try {
|
|
26
|
+
const mtime = statSync(join(srcDir, name)).mtimeMs;
|
|
27
|
+
if (mtime > newest) newest = mtime;
|
|
28
|
+
} catch { /* file vanished mid-scan */ }
|
|
29
|
+
}
|
|
30
|
+
} catch { return 0; }
|
|
31
|
+
return newest;
|
|
32
|
+
}
|
|
33
|
+
|
|
13
34
|
export function runtimeNodeExecutable() {
|
|
14
35
|
return process.versions.bun
|
|
15
36
|
? (process.env.WIKI_MANAGER_NODE_BIN ?? 'node')
|
|
@@ -47,13 +68,22 @@ export async function ensureRuntime({
|
|
|
47
68
|
if (existing) {
|
|
48
69
|
const expectedCacertPath = activeCacertPath();
|
|
49
70
|
const actualCacertPath = existing.cacertPath ? resolve(existing.cacertPath) : null;
|
|
71
|
+
// Dev staleness: if the manager source was edited after this runtime
|
|
72
|
+
// started, the reused process would keep serving old code (the recurring
|
|
73
|
+
// "my change is not taking effect" trap). Treat it as stale and restart.
|
|
74
|
+
// Packed installs report mtime 0 (unreadable src) → never flagged stale.
|
|
75
|
+
// Opt out with WIKI_MANAGER_RUNTIME_NO_STALE_CHECK=1.
|
|
76
|
+
const startedAtMs = Number(existing.startedAtMs) || 0;
|
|
77
|
+
const sourceMtimeMs = process.env.WIKI_MANAGER_RUNTIME_NO_STALE_CHECK === '1' ? 0 : newestManagerSourceMtimeMs();
|
|
78
|
+
const stale = startedAtMs > 0 && sourceMtimeMs > startedAtMs;
|
|
50
79
|
// forceRestart: the caller knows the manager configuration just changed
|
|
51
80
|
// (e.g. mcp.endpoints.json scaffolded on first run) — a runtime started
|
|
52
81
|
// BEFORE that only knows the old endpoints and would keep answering
|
|
53
82
|
// without the agents until manually restarted.
|
|
54
|
-
if (!forceRestart && actualCacertPath === expectedCacertPath) {
|
|
83
|
+
if (!forceRestart && !stale && actualCacertPath === expectedCacertPath) {
|
|
55
84
|
return { url, started: false, health: existing, token: auth.token, tokenPath: auth.tokenPath };
|
|
56
85
|
}
|
|
86
|
+
if (stale) console.error('runtime: source changed since start — restarting for fresh code.');
|
|
57
87
|
await postRuntimeShutdown({ url, token: auth.token });
|
|
58
88
|
await waitForRuntimeShutdown(url, auth.token, 2500);
|
|
59
89
|
}
|
package/src/runtime/server.js
CHANGED
|
@@ -25,6 +25,9 @@ export function startRuntimeServer({
|
|
|
25
25
|
exitOnShutdown = process.env.WIKI_MANAGER_RUNTIME_CHILD === '1',
|
|
26
26
|
} = {}) {
|
|
27
27
|
const clients = new Set();
|
|
28
|
+
// When this runtime process started — used by ensureRuntime to detect that
|
|
29
|
+
// the manager source has been edited since (dev staleness) and auto-restart.
|
|
30
|
+
const runtimeStartedAtMs = Date.now();
|
|
28
31
|
const defaultContext = { workspace: null, session, running: false, currentAbortController: null, currentRunId: null };
|
|
29
32
|
const resolvedGetContext = getContext ?? (() => defaultContext);
|
|
30
33
|
|
|
@@ -62,6 +65,7 @@ export function startRuntimeServer({
|
|
|
62
65
|
status: context?.running ? 'running' : 'idle',
|
|
63
66
|
workspace: context?.workspace ?? workspace ?? null,
|
|
64
67
|
activeRuns,
|
|
68
|
+
startedAtMs: runtimeStartedAtMs,
|
|
65
69
|
dbPath: store.dbPath,
|
|
66
70
|
cacertPath: activeCacertPath(),
|
|
67
71
|
nodeExtraCaCerts: process.env.NODE_EXTRA_CA_CERTS ?? null,
|
package/src/shell/repl.js
CHANGED
|
@@ -5,12 +5,13 @@ import { execFileSync } from 'node:child_process';
|
|
|
5
5
|
import { stdin as input, stdout as output } from 'node:process';
|
|
6
6
|
import { marked } from 'marked';
|
|
7
7
|
import { markedTerminal } from 'marked-terminal';
|
|
8
|
-
import { buildAgentSystemPrompt, formatLlmUnavailableMessage } from '../agent/graph.js';
|
|
8
|
+
import { buildAgentSystemPrompt, formatLlmUnavailableMessage, isDonnaReadTool } from '../agent/graph.js';
|
|
9
9
|
import { handleSlashCommand } from '../commands/slash.js';
|
|
10
10
|
import { serviceDescription, serviceNames as composeServiceNames } from '../core/compose.js';
|
|
11
11
|
import { extractActivity, parseJsonText, sessionActivities } from '../core/activity.js';
|
|
12
12
|
import { syncActivitiesToPlan } from '../core/plan.js';
|
|
13
|
-
import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
|
|
13
|
+
import { buildLlmTools, callMcpTool, formatMcpToolResult, parseToolCallName, resolveToolCallName } from '../core/mcp.js';
|
|
14
|
+
import { runBoundedToolLoop } from '../core/toolLoop.js';
|
|
14
15
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
15
16
|
import { listSkills } from '../core/skills.js';
|
|
16
17
|
import { listWikircProfiles } from '../core/wikirc.js';
|
|
@@ -285,15 +286,37 @@ function isDonnaRole(role) {
|
|
|
285
286
|
return role === 'donna' || role === LEGACY_DONNA_ROLE;
|
|
286
287
|
}
|
|
287
288
|
|
|
289
|
+
// Read-only MCP tools exposed to /chat. The operator declares which tools per
|
|
290
|
+
// server in the `chatAccess` config (mcp.endpoints.json). /chat is extended to
|
|
291
|
+
// those tools ONLY, filtered through the same read-only test Donna's /agent
|
|
292
|
+
// mode uses (isDonnaReadTool), and only when their server is connected — so
|
|
293
|
+
// /chat can answer live state questions ("le CME est-il configuré ?") but can
|
|
294
|
+
// never mutate or delegate. Actions still belong to /agent, which has all tools.
|
|
295
|
+
export function chatReadTools(session) {
|
|
296
|
+
const servers = session?.chatAccess?.servers;
|
|
297
|
+
if (!servers) return [];
|
|
298
|
+
const scopedMcp = Object.fromEntries(
|
|
299
|
+
Object.entries(session.mcp ?? {}).filter(([name]) => Object.hasOwn(servers, name)),
|
|
300
|
+
);
|
|
301
|
+
return buildLlmTools(scopedMcp).filter((item) => {
|
|
302
|
+
const { server, tool } = parseToolCallName(item.function.name);
|
|
303
|
+
const entry = servers[server];
|
|
304
|
+
const declared = entry.allow === '*' ? true : (Array.isArray(entry.allow) && entry.allow.includes(tool));
|
|
305
|
+
return declared && isDonnaReadTool(item);
|
|
306
|
+
});
|
|
307
|
+
}
|
|
308
|
+
|
|
288
309
|
function buildDirectChatSystemPrompt(session) {
|
|
289
310
|
const workspace = session.workspace ?? 'no workspace selected';
|
|
290
311
|
const wikirc = session.wikirc?.profile ?? 'no profile loaded';
|
|
291
312
|
const language = session.language ?? 'en-US';
|
|
292
313
|
return [
|
|
293
314
|
'You are Donna, the llm-wiki-manager chat assistant.',
|
|
294
|
-
'
|
|
315
|
+
'You have a small READ-ONLY toolset — the tools provided to you for this turn, which may be none. Use them to answer questions about live state (e.g. "le CME est-il configuré", "quelles pages sont en attente"), and answer only from their results.',
|
|
316
|
+
'If no provided tool covers the request — or the request is an action or mutation (ingest, build, export, configure, send, delete…), or needs a service that is not connected — say plainly you cannot do it in chat mode and to switch to agent mode (/agent). Do not pretend to execute it and never guess.',
|
|
317
|
+
'Answer directly and concisely. Do not claim to have called tools or changed files beyond the tools actually provided.',
|
|
295
318
|
'Never add a "Next steps", "Prochaines étapes", "À suivre", options, or suggestions section unless the user explicitly asks what to do next. End after answering the question.',
|
|
296
|
-
'
|
|
319
|
+
'Never invent a tool name, command, job id, status, or result (e.g. do not fabricate names like "check_cme_configuration"). If you cannot know something with the tools you were given, say so.',
|
|
297
320
|
`Reply language: ${language}.`,
|
|
298
321
|
`Current workspace: ${workspace}.`,
|
|
299
322
|
`Current wikirc profile: ${wikirc}.`,
|
|
@@ -1164,6 +1187,49 @@ async function runAgentTurn(input, { agent, session, onUpdate, onStep, displayIn
|
|
|
1164
1187
|
return {};
|
|
1165
1188
|
}
|
|
1166
1189
|
|
|
1190
|
+
// Bounded read-only tool loop for /chat. Only the tools declared in chatAccess
|
|
1191
|
+
// (and read-only) are offered; every call goes through callMcpTool, and any
|
|
1192
|
+
// tool the model names outside the offered set is refused — /chat can never
|
|
1193
|
+
// mutate or delegate. maxToolIterations caps the loop.
|
|
1194
|
+
async function runChatReadToolLoop({ input, session, history, donnaMessage, onUpdate, onStep, readTools }) {
|
|
1195
|
+
const allowed = new Set(readTools.map((item) => item.function.name));
|
|
1196
|
+
// /chat only supplies the policy; the loop mechanic lives in runBoundedToolLoop.
|
|
1197
|
+
// executeCall enforces the allow-list and turns each call into a text result;
|
|
1198
|
+
// it never mutates and refuses anything outside the offered read set.
|
|
1199
|
+
const executeCall = async (call) => {
|
|
1200
|
+
const rawName = call.function?.name ?? '';
|
|
1201
|
+
const { server, tool } = resolveToolCallName(session.mcp, rawName);
|
|
1202
|
+
const qualified = server ? `${server}__${tool}` : null;
|
|
1203
|
+
if (!qualified || !allowed.has(qualified)) {
|
|
1204
|
+
return `Refused: "${rawName}" is not an available read-only tool in chat mode. Actions and other tools live in agent mode (/agent).`;
|
|
1205
|
+
}
|
|
1206
|
+
let args = {};
|
|
1207
|
+
try { args = call.function?.arguments ? JSON.parse(call.function.arguments) : {}; } catch { args = {}; }
|
|
1208
|
+
try {
|
|
1209
|
+
onStep?.(`Chat: read ${server} ${tool}…`);
|
|
1210
|
+
const res = await callMcpTool(session.mcp, server, tool, args, session._abortSignal);
|
|
1211
|
+
return formatMcpToolResult(res);
|
|
1212
|
+
} catch (err) {
|
|
1213
|
+
if (err.name === 'AbortError' && session._abortSignal?.aborted) throw err;
|
|
1214
|
+
return `Error [${qualified}]: ${err instanceof Error ? err.message : String(err)}`;
|
|
1215
|
+
}
|
|
1216
|
+
};
|
|
1217
|
+
const { content, capped } = await runBoundedToolLoop({
|
|
1218
|
+
llm: session.llm,
|
|
1219
|
+
system: buildDirectChatSystemPrompt(session),
|
|
1220
|
+
messages: [...history, { role: 'user', content: input }],
|
|
1221
|
+
tools: readTools,
|
|
1222
|
+
executeCall,
|
|
1223
|
+
maxIterations: Math.min(8, Number(session?.chatAccess?.maxToolIterations) || 4),
|
|
1224
|
+
signal: session._abortSignal,
|
|
1225
|
+
onStep: (i, cap) => onStep?.(`Chat: consulting… [${i}/${cap}]`),
|
|
1226
|
+
});
|
|
1227
|
+
donnaMessage.content = capped
|
|
1228
|
+
? 'Je n’ai pas pu conclure dans la limite d’itérations du mode chat. Passe en /agent si besoin.'
|
|
1229
|
+
: (stripDsmlArtifacts(content).trimEnd() || formatLlmUnavailableMessage('reponse vide'));
|
|
1230
|
+
onUpdate?.();
|
|
1231
|
+
}
|
|
1232
|
+
|
|
1167
1233
|
async function runDirectChatTurn(input, { session, onUpdate, onStep }) {
|
|
1168
1234
|
if (!session.llm?.stream) {
|
|
1169
1235
|
conversationMessages(session).push({ role: 'command', content: directChatUnavailableText(session) });
|
|
@@ -1176,24 +1242,30 @@ async function runDirectChatTurn(input, { session, onUpdate, onStep }) {
|
|
|
1176
1242
|
const donnaMessage = { role: 'donna', content: '' };
|
|
1177
1243
|
messages.push(donnaMessage);
|
|
1178
1244
|
onUpdate?.();
|
|
1245
|
+
const readTools = chatReadTools(session);
|
|
1246
|
+
const canUseReadTools = readTools.length > 0 && typeof session.llm.completeWithTools === 'function';
|
|
1179
1247
|
try {
|
|
1180
|
-
|
|
1181
|
-
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
|
|
1186
|
-
|
|
1187
|
-
|
|
1188
|
-
|
|
1248
|
+
if (canUseReadTools) {
|
|
1249
|
+
await runChatReadToolLoop({ input, session, history, donnaMessage, onUpdate, onStep, readTools });
|
|
1250
|
+
} else {
|
|
1251
|
+
onStep?.('Chat: streaming direct answer…');
|
|
1252
|
+
for await (const delta of session.llm.stream({
|
|
1253
|
+
system: buildDirectChatSystemPrompt(session),
|
|
1254
|
+
messages: [...history, { role: 'user', content: input }],
|
|
1255
|
+
signal: session._abortSignal,
|
|
1256
|
+
})) {
|
|
1257
|
+
const cleanDelta = stripDsmlArtifacts(delta);
|
|
1258
|
+
if (cleanDelta) {
|
|
1259
|
+
donnaMessage.content += cleanDelta;
|
|
1260
|
+
onUpdate?.();
|
|
1261
|
+
}
|
|
1262
|
+
}
|
|
1263
|
+
donnaMessage.content = stripDsmlArtifacts(donnaMessage.content).trimEnd();
|
|
1264
|
+
if (!donnaMessage.content.trim()) {
|
|
1265
|
+
donnaMessage.content = formatLlmUnavailableMessage('flux vide');
|
|
1189
1266
|
onUpdate?.();
|
|
1190
1267
|
}
|
|
1191
1268
|
}
|
|
1192
|
-
donnaMessage.content = stripDsmlArtifacts(donnaMessage.content).trimEnd();
|
|
1193
|
-
if (!donnaMessage.content.trim()) {
|
|
1194
|
-
donnaMessage.content = formatLlmUnavailableMessage('flux vide');
|
|
1195
|
-
onUpdate?.();
|
|
1196
|
-
}
|
|
1197
1269
|
} catch (err) {
|
|
1198
1270
|
if (err.name === 'AbortError') {
|
|
1199
1271
|
messages.pop();
|
package/src/shell/repl.test.js
CHANGED
|
@@ -5,6 +5,7 @@ import { tmpdir } from 'node:os';
|
|
|
5
5
|
import { join } from 'node:path';
|
|
6
6
|
import {
|
|
7
7
|
applyRuntimeStateToShellSession,
|
|
8
|
+
chatReadTools,
|
|
8
9
|
createSession,
|
|
9
10
|
conversationMessages,
|
|
10
11
|
recordRuntimeUnavailableAgentInput,
|
|
@@ -373,3 +374,72 @@ test('submitRuntimeRun sends a control message instead of /run while a run is ac
|
|
|
373
374
|
restore();
|
|
374
375
|
}
|
|
375
376
|
});
|
|
377
|
+
|
|
378
|
+
test('chatReadTools exposes only declared, read-only MCP tools to /chat', () => {
|
|
379
|
+
const session = {
|
|
380
|
+
chatAccess: {
|
|
381
|
+
servers: {
|
|
382
|
+
cme: { allow: ['cme_status', 'cme_sources_list', 'cme_export_run'] },
|
|
383
|
+
},
|
|
384
|
+
},
|
|
385
|
+
mcp: {
|
|
386
|
+
cme: {
|
|
387
|
+
status: 'connected',
|
|
388
|
+
tools: [
|
|
389
|
+
{ name: 'cme_status', inputSchema: { type: 'object', properties: {} } },
|
|
390
|
+
{ name: 'cme_sources_list', inputSchema: { type: 'object', properties: {} } },
|
|
391
|
+
{ name: 'cme_setup', inputSchema: { type: 'object', properties: {} } },
|
|
392
|
+
{ name: 'cme_export_run', inputSchema: { type: 'object', properties: {} } },
|
|
393
|
+
],
|
|
394
|
+
},
|
|
395
|
+
documents: {
|
|
396
|
+
status: 'connected',
|
|
397
|
+
tools: [{ name: 'documents_status', inputSchema: { type: 'object', properties: {} } }],
|
|
398
|
+
},
|
|
399
|
+
},
|
|
400
|
+
};
|
|
401
|
+
const names = chatReadTools(session).map((item) => item.function.name).sort();
|
|
402
|
+
// cme_setup: not declared. cme_export_run: declared but a write (excluded by
|
|
403
|
+
// the read-only guard). documents_status: server absent from chatAccess.
|
|
404
|
+
assert.deepEqual(names, ['cme__cme_sources_list', 'cme__cme_status']);
|
|
405
|
+
});
|
|
406
|
+
|
|
407
|
+
test('chatReadTools is empty when no chatAccess is configured', () => {
|
|
408
|
+
const session = { mcp: { cme: { status: 'connected', tools: [{ name: 'cme_status', inputSchema: {} }] } } };
|
|
409
|
+
assert.deepEqual(chatReadTools(session), []);
|
|
410
|
+
});
|
|
411
|
+
|
|
412
|
+
test('/chat uses the tool-capable path when read tools are declared', async () => {
|
|
413
|
+
const session = createSession();
|
|
414
|
+
session.chatMode = true;
|
|
415
|
+
session.chatAccess = { maxToolIterations: 4, servers: { cme: { allow: ['cme_status'] } } };
|
|
416
|
+
session.mcp = { cme: { status: 'connected', tools: [{ name: 'cme_status', inputSchema: { type: 'object', properties: {} } }] } };
|
|
417
|
+
let usedComplete = false;
|
|
418
|
+
session.llm = {
|
|
419
|
+
async *stream() { yield 'STREAM_FALLBACK'; },
|
|
420
|
+
async completeWithTools() {
|
|
421
|
+
usedComplete = true;
|
|
422
|
+
return { tool_calls: [], content: 'Réponse via outils.', message: { role: 'assistant', content: 'Réponse via outils.' } };
|
|
423
|
+
},
|
|
424
|
+
};
|
|
425
|
+
await runLine('le cme est-il configuré', { session, chatMode: true });
|
|
426
|
+
const last = conversationMessages(session).at(-1);
|
|
427
|
+
assert.ok(usedComplete, 'completeWithTools path was taken');
|
|
428
|
+
assert.match(last.content, /Réponse via outils/);
|
|
429
|
+
assert.doesNotMatch(last.content, /STREAM_FALLBACK/);
|
|
430
|
+
});
|
|
431
|
+
|
|
432
|
+
test('/chat falls back to the plain stream when no read tools are declared', async () => {
|
|
433
|
+
const session = createSession();
|
|
434
|
+
session.chatMode = true;
|
|
435
|
+
session.chatAccess = null;
|
|
436
|
+
session.mcp = {};
|
|
437
|
+
session.llm = {
|
|
438
|
+
async *stream() { yield 'PLAIN_STREAM'; },
|
|
439
|
+
async completeWithTools() { return { tool_calls: [], content: 'SHOULD_NOT_APPEAR' }; },
|
|
440
|
+
};
|
|
441
|
+
await runLine('bonjour', { session, chatMode: true });
|
|
442
|
+
const last = conversationMessages(session).at(-1);
|
|
443
|
+
assert.match(last.content, /PLAIN_STREAM/);
|
|
444
|
+
assert.doesNotMatch(last.content, /SHOULD_NOT_APPEAR/);
|
|
445
|
+
});
|