@dotdrelle/wiki-manager 0.11.10 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -7
- package/agents.docker-compose.yml +3 -0
- package/docker-compose.yml +1 -0
- package/package.json +3 -2
- package/src/activity/activityAggregator.js +109 -0
- package/src/activity/activityAggregator.test.js +60 -0
- package/src/activity/activityDeduplicator.js +50 -0
- package/src/activity/progressCalculator.js +61 -0
- package/src/activity/runSynthesis.js +15 -0
- package/src/agent/graph.js +101 -38
- package/src/agent/graph.test.js +75 -5
- package/src/cli/wiki-manager.js +22 -0
- package/src/contracts/schemas.js +249 -13
- package/src/contracts/schemas.test.js +147 -2
- package/src/core/activity.js +8 -5
- package/src/core/agentEvents.js +260 -23
- package/src/core/agentEvents.test.js +53 -0
- package/src/core/compose.js +6 -1
- package/src/core/dockerCompose.test.js +12 -0
- package/src/core/documentIntake.js +33 -38
- package/src/core/documentIntake.test.js +25 -0
- package/src/core/env.js +11 -1
- package/src/core/jobQueue.js +26 -2
- package/src/core/mcp.js +30 -3
- package/src/core/mcp.test.js +69 -1
- package/src/core/plan.js +46 -4
- package/src/core/plan.test.js +16 -1
- package/src/core/planPatch.js +64 -3
- package/src/core/planPatch.test.js +110 -1
- package/src/core/queueStore.test.js +21 -0
- package/src/core/runtimeLog.js +119 -0
- package/src/core/runtimeLog.test.js +84 -0
- package/src/core/wikirc.js +22 -0
- package/src/core/wikirc.test.js +49 -1
- package/src/core/workflow.js +14 -3
- package/src/graph/graphAggregator.js +5 -0
- package/src/graph/graphPatch.js +8 -0
- package/src/graph/graphSnapshot.js +14 -0
- package/src/graph/graphVisibilityPolicy.js +40 -0
- package/src/graph/runGraphProjector.js +87 -0
- package/src/graph/runGraphProjector.test.js +111 -0
- package/src/orchestrator/agentRegistry.js +172 -0
- package/src/orchestrator/agentRegistry.test.js +114 -0
- package/src/orchestrator/approvalPolicy.js +126 -0
- package/src/orchestrator/approvalPolicy.test.js +78 -0
- package/src/orchestrator/assignmentManager.js +80 -0
- package/src/orchestrator/attemptManager.js +168 -0
- package/src/orchestrator/attemptManager.test.js +160 -0
- package/src/orchestrator/budgetManager.js +127 -0
- package/src/orchestrator/capabilityRegistry.js +64 -0
- package/src/orchestrator/capabilityRegistry.test.js +59 -0
- package/src/orchestrator/capabilityResolver.js +151 -0
- package/src/orchestrator/capabilityResolver.test.js +127 -0
- package/src/orchestrator/dependencyResolver.js +112 -0
- package/src/orchestrator/dispatcher.js +302 -0
- package/src/orchestrator/lockManager.js +51 -0
- package/src/orchestrator/planIntegrator.js +268 -0
- package/src/orchestrator/planIntegrator.test.js +213 -0
- package/src/orchestrator/planValidator.js +534 -0
- package/src/orchestrator/planValidator.test.js +262 -0
- package/src/orchestrator/resultAggregator.js +248 -0
- package/src/orchestrator/resultAggregator.test.js +211 -0
- package/src/orchestrator/scheduler.js +103 -0
- package/src/orchestrator/scheduler.test.js +138 -0
- package/src/runtime/approvals.js +130 -1
- package/src/runtime/controlMessages.js +43 -0
- package/src/runtime/controlMessages.test.js +21 -0
- package/src/runtime/donna-contract.test.js +296 -33
- package/src/runtime/recoveryManager.js +176 -0
- package/src/runtime/recoveryManager.test.js +162 -0
- package/src/runtime/runner.e2e.test.js +98 -12
- package/src/runtime/runner.js +261 -227
- package/src/runtime/runner.test.js +249 -452
- package/src/runtime/server.js +142 -24
- package/src/runtime/server.test.js +192 -62
- package/src/runtime/store.js +837 -2
- package/src/runtime/store.test.js +269 -28
- package/src/runtime/supervisor.js +33 -1
- package/src/runtime/supervisor.test.js +47 -1
- package/src/shell/RightPane.tsx +8 -5
- package/src/shell/repl.js +7 -11
- package/src/shell/repl.test.js +26 -5
- package/src/shell/tui.tsx +1 -0
- package/src/shell/useAgent.ts +5 -1
- package/src/shell/useSession.ts +75 -19
package/src/agent/graph.js
CHANGED
|
@@ -5,11 +5,11 @@ import {
|
|
|
5
5
|
callMcpTool,
|
|
6
6
|
formatMcpToolResult,
|
|
7
7
|
formatMcpToolsForAgent,
|
|
8
|
-
|
|
8
|
+
resolveToolCallName,
|
|
9
9
|
} from '../core/mcp.js';
|
|
10
10
|
import { formatSkillsForAgent, readOptionalText } from '../core/skills.js';
|
|
11
11
|
import { handleSlashCommand } from '../commands/slash.js';
|
|
12
|
-
import { extractActivity, formatActivitySummary, parseJsonText } from '../core/activity.js';
|
|
12
|
+
import { extractActivity, formatActivitySummary, parseJsonText, sessionActivities } from '../core/activity.js';
|
|
13
13
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
14
14
|
import { enqueueProductionJob, ensureJobQueue, formatQueue, productionLockBusy } from '../core/jobQueue.js';
|
|
15
15
|
import { updateWorkspaceProfilePreference } from '../core/profile.js';
|
|
@@ -18,6 +18,14 @@ const MAX_TOOL_ITERATIONS = 80;
|
|
|
18
18
|
const MAX_SPINNER_ARG_LENGTH = 96;
|
|
19
19
|
const MAX_PROFILE_CHARS = 4000;
|
|
20
20
|
|
|
21
|
+
// Pseudo-servers handled directly by the tool executor (not present in
|
|
22
|
+
// session.mcp). Listed so unqualified names like "plan_set" resolve the same
|
|
23
|
+
// way as MCP tools in resolveToolCallName.
|
|
24
|
+
const INTERNAL_TOOL_SERVERS = {
|
|
25
|
+
wiki: ['plan_set', 'plan_done'],
|
|
26
|
+
shell: ['run_command', 'read_command', 'profile_update'],
|
|
27
|
+
};
|
|
28
|
+
|
|
21
29
|
const AGENT_SLASH_COMMANDS = new Set([
|
|
22
30
|
'help',
|
|
23
31
|
'version',
|
|
@@ -124,17 +132,16 @@ const WIKI_PLAN_SET_TOOL = {
|
|
|
124
132
|
properties: {
|
|
125
133
|
id: { type: 'string' },
|
|
126
134
|
description: { type: 'string' },
|
|
135
|
+
requiredCapability: { type: ['string', 'null'] },
|
|
127
136
|
status: { type: 'string', enum: ['pending', 'queued', 'running', 'waiting', 'pending_approval', 'done', 'failed', 'cancelled', 'stalled', 'added_during_run'] },
|
|
128
137
|
dependsOn: { type: 'array', items: { type: 'string' } },
|
|
129
|
-
executor: { type: ['string', 'null'] },
|
|
130
|
-
executorQuery: { type: ['object', 'null'], additionalProperties: true },
|
|
131
138
|
outputRefs: { type: 'array', items: { type: 'string' } },
|
|
132
139
|
},
|
|
133
140
|
required: ['description'],
|
|
134
141
|
},
|
|
135
142
|
],
|
|
136
143
|
},
|
|
137
|
-
description: 'Ordered steps. Backward-compatible strings are accepted; structured steps may include id,
|
|
144
|
+
description: 'Ordered steps. Backward-compatible strings are accepted; structured steps may include id, requiredCapability, dependsOn, outputRefs.',
|
|
138
145
|
},
|
|
139
146
|
},
|
|
140
147
|
required: ['steps'],
|
|
@@ -180,6 +187,7 @@ const AgentState = Annotation.Root({
|
|
|
180
187
|
}),
|
|
181
188
|
toolIterations: Annotation({ default: () => 0 }),
|
|
182
189
|
pendingToolCalls: Annotation(),
|
|
190
|
+
inputClassification: Annotation(),
|
|
183
191
|
readyToStream: Annotation(),
|
|
184
192
|
streamContext: Annotation(),
|
|
185
193
|
streamedInline: Annotation(),
|
|
@@ -460,7 +468,7 @@ function handleWikiTool(session, tool, args) {
|
|
|
460
468
|
return `Unknown wiki tool: ${tool}`;
|
|
461
469
|
}
|
|
462
470
|
|
|
463
|
-
function normalizeDeclaredPlanStep(raw, index
|
|
471
|
+
function normalizeDeclaredPlanStep(raw, index) {
|
|
464
472
|
const item = raw && typeof raw === 'object' && !Array.isArray(raw)
|
|
465
473
|
? raw
|
|
466
474
|
: { description: String(raw) };
|
|
@@ -471,8 +479,9 @@ function normalizeDeclaredPlanStep(raw, index, session) {
|
|
|
471
479
|
description,
|
|
472
480
|
status: item.status ?? 'pending',
|
|
473
481
|
dependsOn: Array.isArray(item.dependsOn) ? item.dependsOn.map(String) : [],
|
|
474
|
-
|
|
475
|
-
|
|
482
|
+
requiredCapability: item.requiredCapability != null ? String(item.requiredCapability) : null,
|
|
483
|
+
executor: null,
|
|
484
|
+
executorQuery: null,
|
|
476
485
|
outputRefs: Array.isArray(item.outputRefs) ? item.outputRefs.map(String) : [],
|
|
477
486
|
};
|
|
478
487
|
}
|
|
@@ -488,23 +497,6 @@ function slugStepId(description, index) {
|
|
|
488
497
|
return slug || `task-${index + 1}`;
|
|
489
498
|
}
|
|
490
499
|
|
|
491
|
-
function selectExecutorForStep(description, session) {
|
|
492
|
-
const text = String(description ?? '').toLowerCase();
|
|
493
|
-
let fallback = null;
|
|
494
|
-
for (const [serverName, value] of Object.entries(session.mcp ?? {})) {
|
|
495
|
-
if (value.status !== 'connected') continue;
|
|
496
|
-
for (const tool of value.tools ?? []) {
|
|
497
|
-
const executor = `${serverName}.${tool.name}`;
|
|
498
|
-
fallback ??= executor;
|
|
499
|
-
const haystack = `${serverName} ${tool.name} ${tool.description ?? ''}`.toLowerCase();
|
|
500
|
-
if (text.split(/[^a-z0-9]+/).filter((token) => token.length >= 4).some((token) => haystack.includes(token))) {
|
|
501
|
-
return executor;
|
|
502
|
-
}
|
|
503
|
-
}
|
|
504
|
-
}
|
|
505
|
-
return fallback;
|
|
506
|
-
}
|
|
507
|
-
|
|
508
500
|
// The manager runs on the same host filesystem as the workspace directory
|
|
509
501
|
// (this is the same local file wiki__profile_update writes to via its
|
|
510
502
|
// volume-mounted container), so read it fresh on every turn instead of
|
|
@@ -535,6 +527,7 @@ export function buildAgentSystemPrompt(state) {
|
|
|
535
527
|
`Current workspace: ${workspace}.`,
|
|
536
528
|
`Current wikirc profile: ${wikirc}.`,
|
|
537
529
|
`Available primitives: ${commandList(state.session)}.`,
|
|
530
|
+
'Only announce or call slash commands that appear exactly in Available primitives. Do not invent command names, subcommands, or arguments.',
|
|
538
531
|
'Connected MCP tools (use the server__tool naming convention for tool calls):',
|
|
539
532
|
mcpTools,
|
|
540
533
|
'Current local MCP job queue:',
|
|
@@ -556,8 +549,8 @@ export function buildAgentSystemPrompt(state) {
|
|
|
556
549
|
'',
|
|
557
550
|
'Task startup:',
|
|
558
551
|
' 1. If the next MCP tool returns _activity.plan.steps, call that tool directly; the shell will create the visible plan from the returned activity.',
|
|
559
|
-
' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description,
|
|
560
|
-
' Multi-tool example: wiki__plan_set(steps=[{id:"cme-export",description:"CME export",dependsOn:[],
|
|
552
|
+
' 2. If the tool cannot declare its own plan, call wiki__plan_set before executing the first step. Prefer structured steps: {id, description, requiredCapability, dependsOn, outputRefs}; a legacy list of strings is still accepted.',
|
|
553
|
+
' Multi-tool example: wiki__plan_set(steps=[{id:"cme-export",description:"CME export",requiredCapability:"external-source.export",dependsOn:[],outputRefs:["raw/untracked"]},{id:"production",description:"Production pipeline",requiredCapability:"knowledge.pipeline",dependsOn:["cme-export"],outputRefs:["deliverables"]}])',
|
|
561
554
|
' 3. Immediately execute the first step using the appropriate MCP tool. Do not start step 2 in the same turn unless one async pipeline tool owns and declares the whole sequence.',
|
|
562
555
|
' For synchronous steps (result is immediate, no _activity polling), call wiki__plan_done(step=1) after confirming success.',
|
|
563
556
|
' For async MCP jobs (returns _activity with poll), the orchestrator tracks completion automatically.',
|
|
@@ -575,13 +568,13 @@ export function buildAgentSystemPrompt(state) {
|
|
|
575
568
|
'',
|
|
576
569
|
'On failure: if a completed activity is failed/error/cancelled, call wiki__plan_done(step=N, status="failed") then stop with a clear error report.',
|
|
577
570
|
].filter(Boolean).join('\n'),
|
|
578
|
-
'For service actions, recommend
|
|
571
|
+
'For service actions, recommend only available service primitives from Available primitives, with the exact service name when the primitive supports one.',
|
|
579
572
|
'Disambiguate export requests carefully.',
|
|
580
|
-
'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`
|
|
581
|
-
'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`
|
|
582
|
-
'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts. To chain multiple sequential steps (e.g. build then polish), always use a single
|
|
573
|
+
'Confluence/CME/source export means exporting external Confluence sources into raw/untracked: use cme MCP tools (`cme__cme_export_run`, then `cme__cme_export_status`). Never use production `type=export` for Confluence source export.',
|
|
574
|
+
'Wiki/deliverable/publication export means exporting generated deliverables from the wiki: use production MCP tools (`production__production_start_job` with `type:"export"` or pipeline steps). Require the deliverable path when exporting deliverables.',
|
|
575
|
+
'For ingest/build/export/polish/pipeline workflows, use production MCP tools. Do not route these through direct /wiki shortcuts. To chain multiple sequential steps (e.g. build then polish), always use a single production__production_start_job call with type="pipeline" and steps=["build","polish"] — never start them as separate jobs: the first job is asynchronous and the second would run before it completes. For existing deliverables where content stability matters, pass stabilize:true so the build step preserves unchanged sections; keep polish in the pipeline when publication output is requested. Do not ask the user to confirm between steps; start the pipeline call directly.',
|
|
583
576
|
'Long-running MCP jobs: do not call the same status tool more than once consecutively. When chaining jobs sequentially: (1) start the job, report job/activity id and status; (2) check status once — if done, proceed to the next step immediately; (3) if still running, report status, list the remaining steps, and return control; (4) when re-invoked, check status first, then proceed. Do not spin-poll (status → status → status with no new action between). The shell activity panel monitors non-terminal jobs automatically.',
|
|
584
|
-
'If
|
|
577
|
+
'If production__production_start_job is returned as queued/waiting by the manager, report that it is waiting in the local queue and return control. Do not continue as if the production job has started.',
|
|
585
578
|
'For diagnostics, use /wiki run doctor when the user asks for doctor. Use /workspace init <name> [path] for low-level non-interactive workspace creation. In the interactive TUI, /new <name> opens the setup wizard. Use /wiki for index, or /wiki run index through the explicit backup hatch. Use /wiki run init only for explicit current-workspace llm-wiki init.',
|
|
586
579
|
'If an action requires tools or skills not available yet, explain the limitation and name the expected primitive.',
|
|
587
580
|
workspaceProfile
|
|
@@ -617,6 +610,36 @@ export function formatLlmUnavailableMessage(reason) {
|
|
|
617
610
|
return `⚠ LLM injoignable : ${clean || 'raison inconnue'}`;
|
|
618
611
|
}
|
|
619
612
|
|
|
613
|
+
function classifyAgentInput(input, session) {
|
|
614
|
+
const lower = String(input ?? '').toLowerCase();
|
|
615
|
+
const hasActiveRun = session?.agentProjection?.status === 'running'
|
|
616
|
+
|| sessionActivities(session).some((activity) => !activity.terminal);
|
|
617
|
+
if (/\b(valide tout|approve all|approve|approuve|valid[eé]|ok pour tout|go pour tout)\b/i.test(lower)) {
|
|
618
|
+
return { kind: 'approve', confidence: 0.86, reason: 'approval_request', activeRun: hasActiveRun };
|
|
619
|
+
}
|
|
620
|
+
if (/\b(cancel|annule|stop|arr[eê]te|interromps|abort)\b/i.test(lower)) {
|
|
621
|
+
return { kind: 'cancel', confidence: 0.86, reason: 'cancel_request', activeRun: hasActiveRun };
|
|
622
|
+
}
|
|
623
|
+
if (/\b(plus tard|later|ensuite|apr[eè]s ce run|enqueue|mets en file|met en file|futur|next run|future run)\b/i.test(lower)) {
|
|
624
|
+
return { kind: 'enqueue_run', confidence: 0.8, reason: 'future_run_request', activeRun: hasActiveRun };
|
|
625
|
+
}
|
|
626
|
+
if (/\b(o[uù] en es[t-]|status|statut|progress|progression|run|job|queue|logs?|explique|explain|inspect|show|montre|quoi de neuf)\b/i.test(lower)) {
|
|
627
|
+
return { kind: 'observe', confidence: 0.86, reason: 'status_or_explanation_request', activeRun: hasActiveRun };
|
|
628
|
+
}
|
|
629
|
+
if (hasActiveRun && /\b(ajoute|add|change|modifie|modify|remplace|replace|retire|remove|skip|ignore|plan|step|t[aâ]che)\b/i.test(lower)) {
|
|
630
|
+
return { kind: 'modify_run', confidence: 0.78, reason: 'active_run_change_request', activeRun: hasActiveRun };
|
|
631
|
+
}
|
|
632
|
+
if (hasActiveRun && /\b(lance|run|g[eé]n[eè]re|build|export|cr[eé]e|create|send|envoie|ingest|convert|importe|import)\b/i.test(lower)) {
|
|
633
|
+
return { kind: 'ambiguous', confidence: 0.45, reason: 'active_run_action_is_ambiguous', activeRun: hasActiveRun };
|
|
634
|
+
}
|
|
635
|
+
return { kind: 'converse', confidence: 0.62, reason: 'plain_conversation', activeRun: hasActiveRun };
|
|
636
|
+
}
|
|
637
|
+
|
|
638
|
+
function toolsForClassification(classification, writeTools) {
|
|
639
|
+
if (classification.activeRun && ['converse', 'observe'].includes(classification.kind)) return [SHELL_READ_COMMAND_TOOL];
|
|
640
|
+
return [SHELL_READ_COMMAND_TOOL, ...writeTools];
|
|
641
|
+
}
|
|
642
|
+
|
|
620
643
|
export function createAgentGraph(options = {}) {
|
|
621
644
|
async function orchestratorNode(state) {
|
|
622
645
|
const llm = state.session.llm ?? options.llm ?? null;
|
|
@@ -640,15 +663,33 @@ export function createAgentGraph(options = {}) {
|
|
|
640
663
|
state.session._onStep?.('Agent: planning next action…');
|
|
641
664
|
}
|
|
642
665
|
|
|
643
|
-
const
|
|
666
|
+
const classification = iterations === 0
|
|
667
|
+
? classifyAgentInput(state.input, state.session)
|
|
668
|
+
: (state.inputClassification ?? { kind: 'modify_run', confidence: 1, reason: 'tool_iteration' });
|
|
669
|
+
if (iterations === 0) {
|
|
670
|
+
state.session._onStep?.(`Agent: classified input as ${classification.kind}`);
|
|
671
|
+
emitAgentEvent(state.session, 'control_message_received', 'agent_classifier', {
|
|
672
|
+
input: state.input,
|
|
673
|
+
classification,
|
|
674
|
+
});
|
|
675
|
+
}
|
|
676
|
+
if (iterations === 0 && classification.kind === 'ambiguous') {
|
|
677
|
+
return {
|
|
678
|
+
response: 'Je peux répondre sur le run en cours, modifier ce run, mettre une nouvelle demande en file, approuver, ou annuler. Peux-tu préciser ce que tu veux faire ?',
|
|
679
|
+
pendingToolCalls: null,
|
|
680
|
+
readyToStream: false,
|
|
681
|
+
inputClassification: classification,
|
|
682
|
+
};
|
|
683
|
+
}
|
|
684
|
+
|
|
685
|
+
const writeTools = [
|
|
644
686
|
SHELL_RUN_COMMAND_TOOL,
|
|
645
|
-
SHELL_READ_COMMAND_TOOL,
|
|
646
687
|
SHELL_PROFILE_UPDATE_TOOL,
|
|
647
688
|
WIKI_PLAN_SET_TOOL,
|
|
648
689
|
WIKI_PLAN_DONE_TOOL,
|
|
649
690
|
...buildLlmTools(state.session.mcp),
|
|
650
691
|
];
|
|
651
|
-
const tools =
|
|
692
|
+
const tools = toolsForClassification(classification, writeTools);
|
|
652
693
|
const system = buildAgentSystemPrompt(state);
|
|
653
694
|
|
|
654
695
|
// On iteration 0: prior history is in state.messages, user input must be appended.
|
|
@@ -690,6 +731,7 @@ export function createAgentGraph(options = {}) {
|
|
|
690
731
|
messages: newMessages,
|
|
691
732
|
toolIterations: iterations + 1,
|
|
692
733
|
readyToStream: false,
|
|
734
|
+
inputClassification: classification,
|
|
693
735
|
};
|
|
694
736
|
}
|
|
695
737
|
|
|
@@ -705,6 +747,7 @@ export function createAgentGraph(options = {}) {
|
|
|
705
747
|
readyToStream: false,
|
|
706
748
|
streamedInline: true,
|
|
707
749
|
messages: newMessages,
|
|
750
|
+
inputClassification: classification,
|
|
708
751
|
};
|
|
709
752
|
}
|
|
710
753
|
|
|
@@ -736,11 +779,19 @@ export function createAgentGraph(options = {}) {
|
|
|
736
779
|
const toolResultMessages = [];
|
|
737
780
|
|
|
738
781
|
for (const call of toolCalls) {
|
|
739
|
-
const
|
|
782
|
+
const resolved = resolveToolCallName(state.session.mcp, call.function.name, INTERNAL_TOOL_SERVERS);
|
|
783
|
+
const { server, tool } = resolved;
|
|
740
784
|
const argsSummary = summarizeToolArguments(call.function.arguments);
|
|
741
785
|
const isInternalWikiTool = server === 'wiki' && (tool === 'plan_set' || tool === 'plan_done');
|
|
742
786
|
const serverLabel = server === 'shell' ? 'Shell' : isInternalWikiTool ? 'Plan' : 'MCP';
|
|
743
|
-
const toolName = `${server}.${tool}
|
|
787
|
+
const toolName = server ? `${server}.${tool}` : call.function.name;
|
|
788
|
+
if (resolved.normalized) {
|
|
789
|
+
// Keep normalizations visible: the defensive routing must not hide
|
|
790
|
+
// prompt/skill regressions that reintroduce unqualified names.
|
|
791
|
+
state.session._onStep?.(
|
|
792
|
+
`tool name normalized: ${call.function.name} -> ${server}__${tool}`,
|
|
793
|
+
);
|
|
794
|
+
}
|
|
744
795
|
state.session._onStep?.(
|
|
745
796
|
`[${state.toolIterations}/${MAX_TOOL_ITERATIONS}] ${serverLabel} ${toolName}${argsSummary ? ` (${argsSummary})` : ''}`,
|
|
746
797
|
);
|
|
@@ -761,6 +812,18 @@ export function createAgentGraph(options = {}) {
|
|
|
761
812
|
let resultText;
|
|
762
813
|
let ok = true;
|
|
763
814
|
try {
|
|
815
|
+
if (!server) {
|
|
816
|
+
if (resolved.candidates.length > 1) {
|
|
817
|
+
throw new Error(
|
|
818
|
+
`Ambiguous unqualified tool name "${call.function.name}": several connected servers expose it. `
|
|
819
|
+
+ `Use the <server>__<tool> form: ${resolved.candidates.map((s) => `${s}__${tool}`).join(', ')}.`,
|
|
820
|
+
);
|
|
821
|
+
}
|
|
822
|
+
throw new Error(
|
|
823
|
+
`Unqualified tool call name "${call.function.name}". Tool calls must use the <server>__<tool> `
|
|
824
|
+
+ `naming convention (e.g. production__production_start_job); no connected server exposes a tool named "${tool}".`,
|
|
825
|
+
);
|
|
826
|
+
}
|
|
764
827
|
let args = JSON.parse(call.function.arguments ?? '{}');
|
|
765
828
|
if (server === 'production' && tool === 'production_start_job' && state.session.workspace && !args.callerLabel) {
|
|
766
829
|
args = { ...args, callerLabel: `${state.session.workspace}/wiki-manager` };
|
|
@@ -834,7 +897,7 @@ export function createAgentGraph(options = {}) {
|
|
|
834
897
|
err.name === 'ApprovalError'
|
|
835
898
|
) throw err;
|
|
836
899
|
ok = false;
|
|
837
|
-
resultText = `Error [${
|
|
900
|
+
resultText = `Error [${toolName}]: ${err instanceof Error ? err.message : String(err)}`;
|
|
838
901
|
if (minimalPlanActive && state.session.headlessPlan?.[0]?._activityKey === null) {
|
|
839
902
|
emitAgentEvent(state.session, 'plan_step_updated', 'tool', { step: 1, status: 'failed' });
|
|
840
903
|
}
|
package/src/agent/graph.test.js
CHANGED
|
@@ -300,7 +300,7 @@ test('agent graph waits for tool-level approval configured on endpoint', async (
|
|
|
300
300
|
}
|
|
301
301
|
});
|
|
302
302
|
|
|
303
|
-
test('agent graph accepts structured wiki plan steps
|
|
303
|
+
test('agent graph accepts structured wiki plan steps without selecting MCP executors implicitly', async () => {
|
|
304
304
|
let calls = 0;
|
|
305
305
|
const session = sessionBase({
|
|
306
306
|
mcp: {
|
|
@@ -337,8 +337,20 @@ test('agent graph accepts structured wiki plan steps and selects MCP executors',
|
|
|
337
337
|
name: 'wiki__plan_set',
|
|
338
338
|
arguments: JSON.stringify({
|
|
339
339
|
steps: [
|
|
340
|
-
{
|
|
341
|
-
|
|
340
|
+
{
|
|
341
|
+
id: 'cme-export',
|
|
342
|
+
description: 'Export CME pages',
|
|
343
|
+
requiredCapability: 'external-source.export',
|
|
344
|
+
executor: 'cme.cme_export_run',
|
|
345
|
+
executorQuery: { capability: 'legacy export' },
|
|
346
|
+
outputRefs: ['raw/untracked'],
|
|
347
|
+
},
|
|
348
|
+
{
|
|
349
|
+
id: 'build',
|
|
350
|
+
description: 'Run production build',
|
|
351
|
+
requiredCapability: 'knowledge.pipeline',
|
|
352
|
+
dependsOn: ['cme-export'],
|
|
353
|
+
},
|
|
342
354
|
],
|
|
343
355
|
}),
|
|
344
356
|
},
|
|
@@ -359,8 +371,66 @@ test('agent graph accepts structured wiki plan steps and selects MCP executors',
|
|
|
359
371
|
|
|
360
372
|
assert.equal(result.response, 'Plan ready.');
|
|
361
373
|
assert.deepEqual(session.headlessPlan.map((step) => step.id), ['cme-export', 'build']);
|
|
362
|
-
assert.
|
|
363
|
-
assert.equal(session.headlessPlan[
|
|
374
|
+
assert.deepEqual(session.headlessPlan.map((step) => step.requiredCapability), ['external-source.export', 'knowledge.pipeline']);
|
|
375
|
+
assert.equal(session.headlessPlan[0].executor, null);
|
|
376
|
+
assert.equal(session.headlessPlan[0].executorQuery, null);
|
|
377
|
+
assert.equal(session.headlessPlan[1].executor, null);
|
|
364
378
|
assert.deepEqual(session.headlessPlan[1].dependsOn, ['cme-export']);
|
|
365
379
|
assert.deepEqual(session.headlessPlan[0].outputRefs, ['raw/untracked']);
|
|
366
380
|
});
|
|
381
|
+
|
|
382
|
+
test('buildAgentSystemPrompt forbids inventing slash commands or arguments', () => {
|
|
383
|
+
const prompt = buildAgentSystemPrompt({ session: sessionBase({ commands: ['status', 'services'] }) });
|
|
384
|
+
assert.match(prompt, /Available primitives: \/status, \/services\./);
|
|
385
|
+
assert.match(prompt, /Do not invent command names, subcommands, or arguments/);
|
|
386
|
+
assert.doesNotMatch(prompt, /\/restart serve/);
|
|
387
|
+
assert.doesNotMatch(prompt, /executorQuery/);
|
|
388
|
+
assert.doesNotMatch(prompt, /executor:"/);
|
|
389
|
+
});
|
|
390
|
+
|
|
391
|
+
// Guard: the system prompt must never show a connected tool's bare name
|
|
392
|
+
// outside its qualified server__tool form. Bare mentions are what teach the
|
|
393
|
+
// model to emit unqualified tool calls (the cme_status incident). The bare
|
|
394
|
+
// name list comes from the session's declared servers, never from a manual
|
|
395
|
+
// list (amendment A6). New prompt text or injected skill descriptions that
|
|
396
|
+
// reintroduce a bare name must fail here.
|
|
397
|
+
test('buildAgentSystemPrompt contains no unqualified tool names for connected servers', () => {
|
|
398
|
+
const session = sessionBase({
|
|
399
|
+
mcp: {
|
|
400
|
+
production: {
|
|
401
|
+
status: 'connected',
|
|
402
|
+
tools: [
|
|
403
|
+
{ name: 'production_start_job' }, { name: 'production_job_status' },
|
|
404
|
+
{ name: 'production_job_logs' }, { name: 'production_cancel_job' },
|
|
405
|
+
{ name: 'production_list_jobs' }, { name: 'production_list_templates' },
|
|
406
|
+
{ name: 'production_status' }, { name: 'agent_describe' },
|
|
407
|
+
{ name: 'agent_plan' }, { name: 'agent_execute' },
|
|
408
|
+
{ name: 'agent_status' }, { name: 'agent_cancel' },
|
|
409
|
+
],
|
|
410
|
+
},
|
|
411
|
+
cme: {
|
|
412
|
+
status: 'connected',
|
|
413
|
+
tools: [
|
|
414
|
+
{ name: 'cme_status' }, { name: 'cme_setup' },
|
|
415
|
+
{ name: 'cme_sources_list' }, { name: 'cme_source_add' },
|
|
416
|
+
{ name: 'cme_source_remove' }, { name: 'cme_export_run' },
|
|
417
|
+
{ name: 'cme_export_status' }, { name: 'cme_export_cancel' },
|
|
418
|
+
{ name: 'agent_describe' }, { name: 'agent_execute' },
|
|
419
|
+
{ name: 'agent_status' }, { name: 'agent_cancel' },
|
|
420
|
+
],
|
|
421
|
+
},
|
|
422
|
+
},
|
|
423
|
+
});
|
|
424
|
+
const prompt = buildAgentSystemPrompt({ session });
|
|
425
|
+
const offenders = [];
|
|
426
|
+
for (const [serverName, value] of Object.entries(session.mcp)) {
|
|
427
|
+
for (const tool of value.tools) {
|
|
428
|
+
// A bare occurrence is the tool name not embedded in a wider
|
|
429
|
+
// identifier: `production__production_start_job` does not match
|
|
430
|
+
// because the inner occurrence is preceded by `_`.
|
|
431
|
+
const bare = new RegExp(`(?<![\\w])${tool.name}(?![\\w])`);
|
|
432
|
+
if (bare.test(prompt)) offenders.push(`${serverName}:${tool.name}`);
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
assert.deepEqual(offenders, [], `Unqualified tool names found in system prompt: ${offenders.join(', ')}`);
|
|
436
|
+
});
|
package/src/cli/wiki-manager.js
CHANGED
|
@@ -294,6 +294,7 @@ async function runRuntime(argv, agent) {
|
|
|
294
294
|
}
|
|
295
295
|
const { defaultRuntimeStateDir, openRuntimeStore, RECOVERABLE_QUEUE_STATUSES } = await import('../runtime/store.js');
|
|
296
296
|
const { startRuntimeServer } = await import('../runtime/server.js');
|
|
297
|
+
const { recoverActiveRuns } = await import('../runtime/recoveryManager.js');
|
|
297
298
|
const { emitRuntimeLog, startActivitySupervisor } = await import('../runtime/supervisor.js');
|
|
298
299
|
const { resolveRuntimeAuthToken } = await import('../runtime/auth.js');
|
|
299
300
|
const { createSqliteQueueStore } = await import('../runtime/queueStore.js');
|
|
@@ -484,6 +485,27 @@ async function runRuntime(argv, agent) {
|
|
|
484
485
|
reason: `MCP unavailable: ${gaps.join(', ')}`,
|
|
485
486
|
};
|
|
486
487
|
}
|
|
488
|
+
const taskRecovery = await recoverActiveRuns({
|
|
489
|
+
store,
|
|
490
|
+
session: context.session,
|
|
491
|
+
workspace: context.workspace,
|
|
492
|
+
callTool: callMcpTool,
|
|
493
|
+
});
|
|
494
|
+
if (!taskRecovery.ok) {
|
|
495
|
+
const interrupted = store.interruptRuns({ workspace: context.workspace });
|
|
496
|
+
return {
|
|
497
|
+
workspace: context.workspace ?? workspace ?? null,
|
|
498
|
+
resumed: false,
|
|
499
|
+
interrupted,
|
|
500
|
+
reason: `Task recovery failed: ${taskRecovery.errors.map((item) => `${item.taskId}: ${item.error}`).join('; ')}`,
|
|
501
|
+
};
|
|
502
|
+
}
|
|
503
|
+
if (taskRecovery.recovered.length > 0 || taskRecovery.rescheduled.length > 0) {
|
|
504
|
+
emitRuntimeLog(
|
|
505
|
+
context.session,
|
|
506
|
+
`runtime: recovery attached ${taskRecovery.recovered.length} job(s), requeued ${taskRecovery.rescheduled.length} task(s)`,
|
|
507
|
+
);
|
|
508
|
+
}
|
|
487
509
|
const recoverableRuns = store.listRecoverableRuns({ workspace: context.workspace });
|
|
488
510
|
const runningRun = recoverableRuns.find((run) => run.status === 'running');
|
|
489
511
|
const activeActivities = activeNonTerminalActivities(context.session);
|