@dotdrelle/wiki-manager 0.15.101 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/mcp.endpoints.example.json +1 -1
- package/package.json +2 -2
- package/src/agent/graph.js +63 -13
- package/src/agent/graph.test.js +125 -1
- package/src/agent/llm.js +13 -4
- package/src/agent/llm.test.js +59 -0
- package/src/cli/wiki-manager.js +43 -3
- package/src/commands/slash.js +2 -0
- package/src/core/agentEvents.js +12 -2
- package/src/core/buildInfo.json +2 -2
- package/src/core/env.js +11 -2
- package/src/core/env.test.js +22 -6
- package/src/core/llmCapabilities.js +31 -0
- package/src/core/llmCapabilities.test.js +27 -0
- package/src/core/logLabel.js +9 -0
- package/src/core/logLabel.test.js +12 -0
- package/src/core/mcp.js +2 -2
- package/src/core/toolLoop.js +222 -20
- package/src/core/toolLoop.test.js +324 -0
- package/src/core/wikiPresearch.js +58 -0
- package/src/core/wikirc.js +61 -0
- package/src/core/wikirc.test.js +40 -1
- package/src/core/workflow.js +4 -1
- package/src/orchestrator/attemptManager.js +21 -5
- package/src/orchestrator/attemptManager.test.js +19 -0
- package/src/orchestrator/dispatcher.js +49 -8
- package/src/orchestrator/dispatcher.test.js +33 -1
- package/src/orchestrator/lockManager.js +40 -5
- package/src/orchestrator/resultAggregator.js +12 -1
- package/src/orchestrator/resultAggregator.test.js +29 -0
- package/src/runtime/controlClassify.test.js +85 -1
- package/src/runtime/conversationCompact.js +39 -0
- package/src/runtime/conversationCompaction.test.js +72 -0
- package/src/runtime/runner.e2e.test.js +49 -0
- package/src/runtime/runner.js +65 -1
- package/src/runtime/server.js +83 -75
- package/src/runtime/server.test.js +121 -0
- package/src/runtime/store.js +17 -1
- package/src/runtime/store.test.js +22 -0
- package/src/runtime/workspaceIsolation.test.js +21 -12
- package/src/shell/repl.js +148 -27
- package/src/shell/repl.test.js +182 -1
|
@@ -2,6 +2,7 @@ import { normalizeActivity, parseJsonText } from '../core/activity.js';
|
|
|
2
2
|
import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
|
|
3
3
|
import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
|
|
4
4
|
import { loadWorkspaceProfile } from '../core/profile.js';
|
|
5
|
+
import { supportsTemperature } from '../core/llmCapabilities.js';
|
|
5
6
|
import { containerReachableUrl } from '../core/wikiSetup.js';
|
|
6
7
|
import { mapRuntimeEvent } from '../core/runtimeEventAdapter.js';
|
|
7
8
|
import { emitRuntimeLog, pollActivitiesOnce } from '../runtime/supervisor.js';
|
|
@@ -259,7 +260,7 @@ async function executeExternalRuntime(task, assignment, {
|
|
|
259
260
|
model: activeProfileModel(session),
|
|
260
261
|
language: session?.language ?? session?.wikircConfig?.language ?? null,
|
|
261
262
|
mcp: mcpPool,
|
|
262
|
-
systemPrompt: activeRuntimeSystemPrompt(session, task, assignment),
|
|
263
|
+
systemPrompt: activeRuntimeSystemPrompt(session, task, assignment, mcpPool),
|
|
263
264
|
});
|
|
264
265
|
runtimeRunId = String(accepted?.runId ?? '');
|
|
265
266
|
if (!runtimeRunId) throw new Error('runtime.execute did not return runId.');
|
|
@@ -508,6 +509,10 @@ function activeProfileModel(session) {
|
|
|
508
509
|
...(llm.apiKey ? { apiKey: String(llm.apiKey) } : {}),
|
|
509
510
|
};
|
|
510
511
|
for (const key of ['temperature', 'maxTokens', 'topP', 'seed']) {
|
|
512
|
+
// A gpt-5-class model rejects `temperature`: forwarding the profile's value
|
|
513
|
+
// would make the external runtime's model call fail (HTTP 400), same as the
|
|
514
|
+
// manager's own client.
|
|
515
|
+
if (key === 'temperature' && !supportsTemperature(llm)) continue;
|
|
511
516
|
const value = Number(llm[key]);
|
|
512
517
|
if (Number.isFinite(value)) model[key] = value;
|
|
513
518
|
}
|
|
@@ -552,8 +557,8 @@ function isReadOnlyExternalTool(toolName) {
|
|
|
552
557
|
// The runtime's EYES, per run: the active workspace's wiki MCP (read tools
|
|
553
558
|
// only) PLUS the declared external MCP endpoints that are safe to hand over
|
|
554
559
|
// (connected, no approval-gated tools, not a workspace-mutating server) —
|
|
555
|
-
// typically web
|
|
556
|
-
// else reaches the runtime.
|
|
560
|
+
// typically a web-search connector. The allow-list here is the authority:
|
|
561
|
+
// nothing else reaches the runtime.
|
|
557
562
|
export function activeProfileMcp(session) {
|
|
558
563
|
const blocks = [];
|
|
559
564
|
const wiki = session?.mcp?.wiki;
|
|
@@ -585,8 +590,9 @@ export function activeProfileMcp(session) {
|
|
|
585
590
|
}
|
|
586
591
|
// External connectors ride along ONLY when the operator declared them safe
|
|
587
592
|
// for the runtime's eyes. A connector added from the serve panel lands here
|
|
588
|
-
// too — without this,
|
|
589
|
-
// delegated to the gateway and the Deep Agent answered it had no web
|
|
593
|
+
// too — without this, a web connector was offered in chat but the agentic
|
|
594
|
+
// path delegated to the gateway and the Deep Agent answered it had no web
|
|
595
|
+
// tools.
|
|
590
596
|
const EXCLUDED_EXTERNAL_SERVERS = new Set(['cme', 'documents', 'connectors', 'production']);
|
|
591
597
|
for (const [name, entry] of Object.entries(session?.mcp ?? {})) {
|
|
592
598
|
if (!entry?.external || entry.status !== 'connected') continue;
|
|
@@ -611,21 +617,56 @@ export function activeProfileMcp(session) {
|
|
|
611
617
|
// description), the eyes/bouche/mains boundary, the workspace profile and the
|
|
612
618
|
// reply language. Without it, the runtime falls back to deepagents' generic
|
|
613
619
|
// assistant prompt — which is exactly the "upload your project" hallucination.
|
|
614
|
-
|
|
620
|
+
// The runtime prompt must describe the pool the same dispatch actually hand
|
|
621
|
+
// over. It used to assert "READ tools only (the workspace wiki)" whatever the
|
|
622
|
+
// pool held, so a run handed an external web-search read tool still answered
|
|
623
|
+
// "je suis limité aux seules sources du wiki" and made zero tool calls
|
|
624
|
+
// (observed on acpi: "cherche sur internet" → agent.answer → refusal, gateway
|
|
625
|
+
// /metrics tools:0). The pool is the authority; the prompt reports it, never
|
|
626
|
+
// denies it.
|
|
627
|
+
function describeRuntimePool(mcpPool) {
|
|
628
|
+
const blocks = Array.isArray(mcpPool) ? mcpPool : [];
|
|
629
|
+
const wiki = blocks.find((block) => block?.name === 'wiki');
|
|
630
|
+
const hasWiki = Boolean(wiki && Array.isArray(wiki.tools) && wiki.tools.length > 0);
|
|
631
|
+
const external = blocks
|
|
632
|
+
.filter((block) => block && block.name !== 'wiki' && Array.isArray(block.tools) && block.tools.length > 0)
|
|
633
|
+
.map((block) => ({
|
|
634
|
+
name: String(block.name),
|
|
635
|
+
tools: block.tools.map((tool) => `${block.name}__${String(tool)}`),
|
|
636
|
+
}));
|
|
637
|
+
return { hasWiki, external };
|
|
638
|
+
}
|
|
639
|
+
|
|
640
|
+
export function activeRuntimeSystemPrompt(session, task, assignment, mcpPool = null) {
|
|
615
641
|
const capability = assignment?.capability ?? null;
|
|
616
642
|
const description = String(capability?.description ?? '').trim();
|
|
617
643
|
const language = session?.language ?? session?.wikircConfig?.language ?? null;
|
|
618
644
|
const profile = loadWorkspaceProfile(session?.workspacePath);
|
|
645
|
+
const { hasWiki, external } = describeRuntimePool(mcpPool);
|
|
646
|
+
const externalTools = external.flatMap((block) => block.tools);
|
|
647
|
+
// Wiki first, then the declared external read tools. A run that has a web
|
|
648
|
+
// search tool must not refuse with "I cannot search the internet": that
|
|
649
|
+
// refusal is exactly the symptom this line exists to prevent.
|
|
650
|
+
const vision = externalTools.length > 0
|
|
651
|
+
? [
|
|
652
|
+
`Beyond the wiki you also have these READ tools: ${externalTools.join(', ')}.`,
|
|
653
|
+
hasWiki
|
|
654
|
+
? 'Search the workspace wiki FIRST; then use those external/web read tools for current, public or out-of-workspace facts the wiki does not cover.'
|
|
655
|
+
: 'Use those external/web read tools for facts the workspace does not cover.',
|
|
656
|
+
'Never answer that you cannot search the internet or that you are limited to the wiki: when such a tool is listed here, call it.',
|
|
657
|
+
].join(' ')
|
|
658
|
+
: null;
|
|
619
659
|
return [
|
|
620
660
|
'You are the agentic analysis engine of a knowledge workspace (wikiLLM), executed behind the manager Donna.',
|
|
621
661
|
`Execute exactly ONE capability: ${task?.requiredCapability ?? 'unknown'}${description ? ` — ${description}` : ''}.`,
|
|
622
662
|
`Operation: ${task?.operation ?? 'run'}.`,
|
|
623
|
-
'Boundary: you have READ tools only (the
|
|
663
|
+
'Boundary: you have READ tools only (the tools this run makes available, listed in your pool). You never modify the workspace — structural changes are proposals you return in your final answer (a planExpansionRequest), the manager integrates them under human approval. Side-effects on the outside world are gated by approval.',
|
|
664
|
+
vision,
|
|
624
665
|
'Ground every claim in what the read tools return. Never invent pages, names, facts, jobs or results.',
|
|
625
666
|
'Tool discipline: discover real page paths with the list/search tools BEFORE reading. Never guess a path — a read refused for "path not allowed" means the path was invented, so list/search first, then read exactly what exists.',
|
|
626
667
|
...(language ? [`Reply in the workspace language: ${language}.`] : []),
|
|
627
668
|
...(profile ? [`Workspace preferences — apply them to every reply:\n${profile}`] : []),
|
|
628
|
-
].join('\n');
|
|
669
|
+
].filter(Boolean).join('\n');
|
|
629
670
|
}
|
|
630
671
|
|
|
631
672
|
function dispatchTaskActivity(session, task, assignment, jobId, statusTool, runId) {
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import assert from 'node:assert/strict';
|
|
2
2
|
import test from 'node:test';
|
|
3
|
-
import { activeProfileMcp, createDispatcher, normalizeTaskError, RUNTIME_SHUTDOWN_ABORT_REASON } from './dispatcher.js';
|
|
3
|
+
import { activeProfileMcp, activeRuntimeSystemPrompt, createDispatcher, normalizeTaskError, RUNTIME_SHUTDOWN_ABORT_REASON } from './dispatcher.js';
|
|
4
4
|
|
|
5
5
|
test('activeProfileMcp forwards only the read-only wiki tools to the external runtime', () => {
|
|
6
6
|
const session = {
|
|
@@ -114,6 +114,38 @@ test('activeProfileMcp rewrites loopback URLs for the container the runtime runs
|
|
|
114
114
|
assert.equal(pool.find((block) => block.name === 'hosted').url, 'https://mcp.exa.ai/mcp');
|
|
115
115
|
});
|
|
116
116
|
|
|
117
|
+
test('activeRuntimeSystemPrompt names the declared web tools and forbids "cannot search the internet"', () => {
|
|
118
|
+
const session = { workspace: 'acpi', language: 'fr-FR' };
|
|
119
|
+
const pool = [
|
|
120
|
+
{ name: 'wiki', tools: ['wiki_read_page', 'wiki_search_context'] },
|
|
121
|
+
{ name: 'exa', tools: ['web_search_exa', 'web_fetch_exa'] },
|
|
122
|
+
];
|
|
123
|
+
const prompt = activeRuntimeSystemPrompt(
|
|
124
|
+
session,
|
|
125
|
+
{ requiredCapability: 'agent.answer', operation: 'run' },
|
|
126
|
+
{ capability: { description: 'wiki and web research' } },
|
|
127
|
+
pool,
|
|
128
|
+
);
|
|
129
|
+
assert.match(prompt, /exa__web_search_exa/);
|
|
130
|
+
assert.match(prompt, /exa__web_fetch_exa/);
|
|
131
|
+
assert.match(prompt, /wiki FIRST/i);
|
|
132
|
+
assert.match(prompt, /cannot search the internet/i);
|
|
133
|
+
assert.match(prompt, /agent\.answer/);
|
|
134
|
+
});
|
|
135
|
+
|
|
136
|
+
test('activeRuntimeSystemPrompt stays wiki-only when the pool declares no external read tools', () => {
|
|
137
|
+
const session = { workspace: 'acpi' };
|
|
138
|
+
const prompt = activeRuntimeSystemPrompt(
|
|
139
|
+
session,
|
|
140
|
+
{ requiredCapability: 'agent.review' },
|
|
141
|
+
{ capability: {} },
|
|
142
|
+
[{ name: 'wiki', tools: ['wiki_read_page'] }],
|
|
143
|
+
);
|
|
144
|
+
assert.doesNotMatch(prompt, /exa__/);
|
|
145
|
+
assert.doesNotMatch(prompt, /cannot search the internet/i);
|
|
146
|
+
assert.match(prompt, /the tools this run makes available/);
|
|
147
|
+
});
|
|
148
|
+
|
|
117
149
|
test('dispatcher returns a retryable logical failure when agent_execute reports workspace_busy', async () => {
|
|
118
150
|
const session = {
|
|
119
151
|
workspace: 'test',
|
|
@@ -1,27 +1,62 @@
|
|
|
1
|
-
|
|
1
|
+
/*
|
|
2
|
+
One registry per WORKSPACE, not per run.
|
|
3
|
+
|
|
4
|
+
The registry used to be created by each run's attemptManager, so its locks
|
|
5
|
+
only ever excluded tasks of the same run. That was harmless while a
|
|
6
|
+
workspace ran one run at a time, but anything acting beside a run — a
|
|
7
|
+
direct write from a chat turn, a second run — held nothing the first one
|
|
8
|
+
could see. `workspaceLockRegistry` hangs one registry on the workspace
|
|
9
|
+
session; every run and every direct action shares it
|
|
10
|
+
(plan-demandes-pendant-run.md, lot 2). `owners` records who holds each lock
|
|
11
|
+
so a wait can name its holder instead of stalling in silence.
|
|
12
|
+
*/
|
|
13
|
+
export function workspaceLockRegistry(session) {
|
|
14
|
+
if (!session) return { locks: new Set(), owners: new Map() };
|
|
15
|
+
session._workspaceLocks ??= { locks: new Set(), owners: new Map() };
|
|
16
|
+
return session._workspaceLocks;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
export function createLockManager({ locks = new Set(), owners = new Map() } = {}) {
|
|
2
20
|
return {
|
|
3
21
|
canAcquire(taskOrLocks) {
|
|
4
22
|
return locksFor(taskOrLocks).every((lock) => !locks.has(lock));
|
|
5
23
|
},
|
|
6
|
-
acquire(taskOrLocks) {
|
|
24
|
+
acquire(taskOrLocks, owner = null) {
|
|
7
25
|
const lockNames = locksFor(taskOrLocks);
|
|
8
26
|
if (lockNames.some((lock) => locks.has(lock))) return null;
|
|
9
|
-
for (const lock of lockNames)
|
|
27
|
+
for (const lock of lockNames) {
|
|
28
|
+
locks.add(lock);
|
|
29
|
+
if (owner != null) owners.set(lock, owner);
|
|
30
|
+
}
|
|
10
31
|
let released = false;
|
|
11
32
|
return {
|
|
12
33
|
locks: lockNames,
|
|
13
34
|
release() {
|
|
14
35
|
if (released) return;
|
|
15
36
|
released = true;
|
|
16
|
-
for (const lock of lockNames)
|
|
37
|
+
for (const lock of lockNames) {
|
|
38
|
+
locks.delete(lock);
|
|
39
|
+
owners.delete(lock);
|
|
40
|
+
}
|
|
17
41
|
},
|
|
18
42
|
};
|
|
19
43
|
},
|
|
20
44
|
release(taskOrLocks) {
|
|
21
|
-
for (const lock of locksFor(taskOrLocks))
|
|
45
|
+
for (const lock of locksFor(taskOrLocks)) {
|
|
46
|
+
locks.delete(lock);
|
|
47
|
+
owners.delete(lock);
|
|
48
|
+
}
|
|
49
|
+
},
|
|
50
|
+
// Who holds the locks a task would need: [{ lock, owner }], owner null
|
|
51
|
+
// when the holder did not name itself.
|
|
52
|
+
holders(taskOrLocks) {
|
|
53
|
+
return locksFor(taskOrLocks)
|
|
54
|
+
.filter((lock) => locks.has(lock))
|
|
55
|
+
.map((lock) => ({ lock, owner: owners.get(lock) ?? null }));
|
|
22
56
|
},
|
|
23
57
|
clear() {
|
|
24
58
|
locks.clear();
|
|
59
|
+
owners.clear();
|
|
25
60
|
},
|
|
26
61
|
snapshot() {
|
|
27
62
|
return [...locks].sort();
|
|
@@ -78,6 +78,13 @@ export async function accept(result, {
|
|
|
78
78
|
taskId,
|
|
79
79
|
payload: { message: `agent-proposal: could not persist the worktree proposal for ${taskId}: ${worktreePersisted.error}` },
|
|
80
80
|
})));
|
|
81
|
+
} else if (worktreePersisted.empty) {
|
|
82
|
+
persistDispatch(store, dispatchAgentEvent(session, createAgentEvent('runtime_log', {
|
|
83
|
+
origin: 'result_aggregator',
|
|
84
|
+
runId,
|
|
85
|
+
taskId,
|
|
86
|
+
payload: { message: `⚠ agent-proposal: ${taskId} wrote no file on its branch — nothing to review; the corrections its report describes were not written` },
|
|
87
|
+
})));
|
|
81
88
|
} else if (worktreePersisted.path) {
|
|
82
89
|
persistDispatch(store, dispatchAgentEvent(session, createAgentEvent('runtime_log', {
|
|
83
90
|
origin: 'result_aggregator',
|
|
@@ -315,7 +322,11 @@ function persistWorktreeProposal(session, result, { runId, taskId }) {
|
|
|
315
322
|
?? result?.worktreeProposal;
|
|
316
323
|
if (!proposal || typeof proposal !== 'object') return { path: null };
|
|
317
324
|
const changes = Array.isArray(proposal.changes) ? proposal.changes : [];
|
|
318
|
-
|
|
325
|
+
// A curation that wrote nothing has nothing to review — but it must SAY so.
|
|
326
|
+
// Returning silently here left the review page empty while the run was
|
|
327
|
+
// reported a success, its report describing corrections on a branch that
|
|
328
|
+
// held none (observed on acpi).
|
|
329
|
+
if (changes.length === 0) return { path: null, empty: true };
|
|
319
330
|
const workspacePath = session?.workspacePath;
|
|
320
331
|
if (!workspacePath || typeof workspacePath !== 'string') {
|
|
321
332
|
return { error: 'no workspace path on the session — the proposal stays in the run result only' };
|
|
@@ -409,3 +409,32 @@ test('a gateway worktree proposal (rawStatus shape) is persisted for review', as
|
|
|
409
409
|
rmSync(workspacePath, { recursive: true, force: true });
|
|
410
410
|
}
|
|
411
411
|
});
|
|
412
|
+
|
|
413
|
+
test('a curation that wrote no file says so instead of vanishing from the review page', async () => {
|
|
414
|
+
// Observed on acpi: the Redactor described its corrections in prose on a
|
|
415
|
+
// branch it invented, wrote nothing, and the run was reported a success
|
|
416
|
+
// while /agent-proposals stayed empty with no word anywhere.
|
|
417
|
+
const workspacePath = mkdtempSync(join(tmpdir(), 'worktree-proposal-'));
|
|
418
|
+
const session = { agentEvents: [], activities: {}, workspace: 'docs', workspacePath, headlessPlan: [] };
|
|
419
|
+
try {
|
|
420
|
+
await accept({
|
|
421
|
+
ok: true,
|
|
422
|
+
taskId: 't-curate',
|
|
423
|
+
status: 'completed',
|
|
424
|
+
outputRefs: [],
|
|
425
|
+
rawStatus: {
|
|
426
|
+
runId: 'gateway-1',
|
|
427
|
+
status: 'completed',
|
|
428
|
+
result: { status: 'completed', content: 'report', worktreeProposal: { branch: 'agent/gateway-1', changedFiles: [], changes: [], diff: '' } },
|
|
429
|
+
},
|
|
430
|
+
}, {
|
|
431
|
+
session,
|
|
432
|
+
runId: 'run-curate',
|
|
433
|
+
task: { id: 't-curate', requiredCapability: 'agent.curate', operation: 'run' },
|
|
434
|
+
});
|
|
435
|
+
assert.equal(existsSync(join(workspacePath, '.wiki', 'agent-proposals')), false);
|
|
436
|
+
assert.ok(session.agentEvents.some((event) => /wrote no file on its branch — nothing to review/.test(String(event.payload?.message ?? ''))));
|
|
437
|
+
} finally {
|
|
438
|
+
rmSync(workspacePath, { recursive: true, force: true });
|
|
439
|
+
}
|
|
440
|
+
});
|
|
@@ -15,10 +15,11 @@ test('a bare confirmation during a run is a status check', async () => {
|
|
|
15
15
|
}
|
|
16
16
|
});
|
|
17
17
|
|
|
18
|
-
test('a plan change that merely opens on a yes
|
|
18
|
+
test('a plan change that merely opens on a yes reaches the model, not the status branch', async () => {
|
|
19
19
|
const result = await classifyControlMessage(
|
|
20
20
|
'oui, ajoute une étape de polish après le build',
|
|
21
21
|
running,
|
|
22
|
+
{ llm: { complete: async () => 'plan_change' } },
|
|
22
23
|
);
|
|
23
24
|
assert.equal(result.kind, 'modify_run');
|
|
24
25
|
});
|
|
@@ -29,3 +30,86 @@ test('a new task that opens on a yes reaches the model classifier, not the statu
|
|
|
29
30
|
});
|
|
30
31
|
assert.equal(result.kind, 'enqueue_run');
|
|
31
32
|
});
|
|
33
|
+
|
|
34
|
+
// The empty-chat "Curate the wiki" tile sends free-text prose. While a run was
|
|
35
|
+
// active, its "pages that disagree or repeat each other" matched the
|
|
36
|
+
// deterministic modify_run keyword "each" and the curation became an invisible
|
|
37
|
+
// plan patch instead of a queued run.
|
|
38
|
+
test('a curation objective during a run is queued, never an invisible plan patch', async () => {
|
|
39
|
+
const objective =
|
|
40
|
+
'Curate the wiki: find duplicate pages, pages that disagree or repeat each other, '
|
|
41
|
+
+ 'outdated or superseded pages, and claims with no cited source, then write the '
|
|
42
|
+
+ 'corrections on a dedicated branch.';
|
|
43
|
+
const result = await classifyControlMessage(objective, running, {
|
|
44
|
+
llm: { complete: async () => 'action' },
|
|
45
|
+
});
|
|
46
|
+
assert.equal(result.kind, 'enqueue_run');
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
test('a question about curation stays read-only', async () => {
|
|
50
|
+
const result = await classifyControlMessage('explain how curation works', running, {
|
|
51
|
+
llm: { complete: async () => 'question' },
|
|
52
|
+
});
|
|
53
|
+
assert.equal(result.kind, 'converse');
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
test('a genuine plan change is classified as modify_run by the model', async () => {
|
|
57
|
+
const result = await classifyControlMessage('ajoute une étape de polish après le build', running, {
|
|
58
|
+
llm: { complete: async () => 'plan_change' },
|
|
59
|
+
});
|
|
60
|
+
assert.equal(result.kind, 'modify_run');
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
// Measured 2026-09-25 (plan-demandes-pendant-run.md §2): keywords inside a
|
|
64
|
+
// sentence decided before the model, so questions typed during an ingest were
|
|
65
|
+
// answered with the run status, turned into patches of the running plan, or
|
|
66
|
+
// cancelled the run. No keyword inside a sentence decides any more.
|
|
67
|
+
test('a keyword inside a question never decides the triage', async () => {
|
|
68
|
+
const asked = [];
|
|
69
|
+
const llm = { complete: async ({ input }) => { asked.push(input); return 'question'; } };
|
|
70
|
+
for (const question of [
|
|
71
|
+
"montre-moi ce que dit le wiki sur l'option A",
|
|
72
|
+
'explique la différence entre les options A et B',
|
|
73
|
+
'cherche les pages qui parlent du plan de charge',
|
|
74
|
+
'que se passe-t-il après la validation dans SISBA ?',
|
|
75
|
+
"quand est-ce qu'on stop le support de CDPOM ?",
|
|
76
|
+
'que dit le wiki sur la curation des données ?',
|
|
77
|
+
"qu'est-ce qu'on fait ensuite dans le projet ?",
|
|
78
|
+
]) {
|
|
79
|
+
const result = await classifyControlMessage(question, running, { llm });
|
|
80
|
+
assert.equal(result.kind, 'converse', question);
|
|
81
|
+
}
|
|
82
|
+
assert.equal(asked.length, 7, 'every one of them reached the model');
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
test('the model categories map onto the control kinds', async () => {
|
|
86
|
+
for (const [word, kind] of [
|
|
87
|
+
['question', 'converse'], ['status', 'observe'], ['action', 'enqueue_run'],
|
|
88
|
+
['plan_change', 'modify_run'], ['cancel', 'cancel'], ['Action.', 'enqueue_run'],
|
|
89
|
+
]) {
|
|
90
|
+
const result = await classifyControlMessage('modifie la page option-a', running, {
|
|
91
|
+
llm: { complete: async () => word },
|
|
92
|
+
});
|
|
93
|
+
assert.equal(result.kind, kind, word);
|
|
94
|
+
}
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
test('only a whole-message command is decided without the model', async () => {
|
|
98
|
+
const cases = [
|
|
99
|
+
['stop', 'cancel'], ['annule', 'cancel'], ['arrête tout !', 'cancel'], ['cancel the run', 'cancel'],
|
|
100
|
+
["annule l'ingestion", 'cancel'], ['status', 'observe'], ['où en est-on ?', 'observe'],
|
|
101
|
+
['logs', 'observe'], ['quel est le statut du run ?', 'observe'],
|
|
102
|
+
["mets en file l'export", 'enqueue_run'], ['lance le build après ce run', 'enqueue_run'],
|
|
103
|
+
];
|
|
104
|
+
for (const [input, kind] of cases) {
|
|
105
|
+
const result = await classifyControlMessage(input, running);
|
|
106
|
+
assert.equal(result.kind, kind, input);
|
|
107
|
+
}
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
test('without a model every other message is read-only conversation', async () => {
|
|
111
|
+
for (const input of ['curate the wiki', 'modifie la page option-a', "quand est-ce qu'on stop le support ?"]) {
|
|
112
|
+
const result = await classifyControlMessage(input, running);
|
|
113
|
+
assert.equal(result.kind, 'converse', input);
|
|
114
|
+
}
|
|
115
|
+
});
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { emitRuntimeLog } from './supervisor.js';
|
|
2
|
+
|
|
3
|
+
const CONVERSATION_SUMMARY_TIMEOUT_MS = 20_000;
|
|
4
|
+
const CONVERSATION_SUMMARY_MAX_INPUT_CHARS = 8_000;
|
|
5
|
+
|
|
6
|
+
/*
|
|
7
|
+
A compact does not just cut older turns from conversationSeed — it replaces
|
|
8
|
+
them with a short rolling summary, so a decision made 20 messages ago is not
|
|
9
|
+
gone from Donna's grounding entirely, only condensed. Best-effort: no LLM
|
|
10
|
+
configured, an empty reply, or a call failure all fall back to keeping
|
|
11
|
+
whatever summary already existed (never worse than before this compact),
|
|
12
|
+
the same deterministic-under-failure shape as generateControlAcknowledgment.
|
|
13
|
+
*/
|
|
14
|
+
export async function summarizeCompactedConversation(session, { previousSummary, segment }) {
|
|
15
|
+
const llm = session?.llm;
|
|
16
|
+
const transcript = (Array.isArray(segment) ? segment : [])
|
|
17
|
+
.filter((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim())
|
|
18
|
+
.map((message) => `${message.role === 'user' ? 'User' : 'Assistant'}: ${String(message.content).trim()}`)
|
|
19
|
+
.join('\n')
|
|
20
|
+
.slice(0, CONVERSATION_SUMMARY_MAX_INPUT_CHARS);
|
|
21
|
+
if (!transcript) return previousSummary || null;
|
|
22
|
+
if (!(llm && typeof llm.complete === 'function')) return previousSummary || null;
|
|
23
|
+
try {
|
|
24
|
+
const reply = await llm.complete({
|
|
25
|
+
system: 'You maintain a compact working memory for Donna, a workspace assistant. You are shown an optional PREVIOUS SUMMARY and a NEW SEGMENT of conversation about to leave the assistant\'s context window. Write ONE updated summary that preserves the facts, decisions, open questions and user preferences that still matter for future turns. Be concise: well under 200 words. Return only the summary text — no preamble, no meta-commentary, no headings.',
|
|
26
|
+
input: [
|
|
27
|
+
previousSummary ? `PREVIOUS SUMMARY:\n${previousSummary}` : null,
|
|
28
|
+
`NEW SEGMENT:\n${transcript}`,
|
|
29
|
+
].filter(Boolean).join('\n\n'),
|
|
30
|
+
signal: AbortSignal.timeout(CONVERSATION_SUMMARY_TIMEOUT_MS),
|
|
31
|
+
});
|
|
32
|
+
const text = String(reply ?? '').trim();
|
|
33
|
+
if (text) return text;
|
|
34
|
+
emitRuntimeLog(session, 'conversation-compact: LLM returned an empty summary, keeping the previous one');
|
|
35
|
+
} catch (err) {
|
|
36
|
+
emitRuntimeLog(session, `conversation-compact: summary LLM call failed, keeping the previous summary — ${err instanceof Error ? err.message : String(err)}`);
|
|
37
|
+
}
|
|
38
|
+
return previousSummary || null;
|
|
39
|
+
}
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import test from 'node:test';
|
|
3
|
+
import { conversationCompactionPlan, compactionNoteForDonna, conversationSeed } from './runner.js';
|
|
4
|
+
import { createAgentEvent, reduceAgentEvents } from '../core/agentEvents.js';
|
|
5
|
+
|
|
6
|
+
const exchange = (n) => [
|
|
7
|
+
{ role: 'user', content: `question ${n}` },
|
|
8
|
+
{ role: 'assistant', content: `réponse ${n}` },
|
|
9
|
+
];
|
|
10
|
+
|
|
11
|
+
test('no compaction while the whole conversation fits the seed window', () => {
|
|
12
|
+
const conversation = [1, 2, 3].flatMap(exchange);
|
|
13
|
+
assert.equal(conversationCompactionPlan({ conversation }), null);
|
|
14
|
+
});
|
|
15
|
+
|
|
16
|
+
test('a message about to leave the window triggers a compaction that keeps the last exchanges', () => {
|
|
17
|
+
const conversation = [1, 2, 3, 4, 5, 6, 7].flatMap(exchange); // 14 > 12
|
|
18
|
+
const plan = conversationCompactionPlan({ conversation });
|
|
19
|
+
assert.equal(plan.reason, 'window');
|
|
20
|
+
assert.equal(plan.keepLast, 6);
|
|
21
|
+
assert.equal(plan.segment.length, 8);
|
|
22
|
+
assert.equal(plan.segment.at(-1).content, 'réponse 4');
|
|
23
|
+
});
|
|
24
|
+
|
|
25
|
+
test('a small-context profile compacts on the budget before the window fills', () => {
|
|
26
|
+
const conversation = [1, 2].flatMap(exchange).map((m) => ({ ...m, content: `${m.content} ${'x'.repeat(900)}` }));
|
|
27
|
+
assert.equal(conversationCompactionPlan({ conversation }, { budgetChars: 100000 }), null);
|
|
28
|
+
const plan = conversationCompactionPlan({ conversation }, { budgetChars: 4000, keepLast: 2 });
|
|
29
|
+
assert.equal(plan.reason, 'budget');
|
|
30
|
+
assert.equal(plan.segment.length, 2);
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
test('the compact boundary keeps the last exchanges verbatim in the seed', () => {
|
|
34
|
+
const events = [1, 2, 3, 4, 5, 6, 7].flatMap((n) => [
|
|
35
|
+
createAgentEvent('user_message', { payload: { content: `question ${n}` } }),
|
|
36
|
+
createAgentEvent('assistant_message', { payload: { content: `réponse ${n}` } }),
|
|
37
|
+
]);
|
|
38
|
+
events.push(createAgentEvent('conversation_reset', { payload: { summary: 'Résumé 1-4.', keepLast: 6, automatic: true } }));
|
|
39
|
+
const projection = reduceAgentEvents(events);
|
|
40
|
+
const seed = conversationSeed({ agentProjection: projection }, 'rajoute la liste');
|
|
41
|
+
assert.match(seed[0].content, /Résumé 1-4\./);
|
|
42
|
+
assert.deepEqual(seed.slice(1).map((m) => m.content), ['question 5', 'réponse 5', 'question 6', 'réponse 6', 'question 7', 'réponse 7']);
|
|
43
|
+
assert.equal(conversationCompactionPlan(projection), null, 'nothing more to compact');
|
|
44
|
+
});
|
|
45
|
+
|
|
46
|
+
test('Donna is the one who tells the user, in the reply language', () => {
|
|
47
|
+
const note = compactionNoteForDonna(8);
|
|
48
|
+
assert.match(note, /8 oldest messages/);
|
|
49
|
+
assert.match(note, /tell them so in one short sentence, in the reply language/);
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
test('the compact cut never separates a question from its answer', () => {
|
|
53
|
+
// Observed on acpi: 6 raw messages kept, the cut fell between the user's
|
|
54
|
+
// question and Donna's answer, and the summary dropped the question.
|
|
55
|
+
const conversation = [
|
|
56
|
+
...[1, 2, 3, 4, 5].flatMap(exchange),
|
|
57
|
+
{ role: 'user', content: 'comment utiliser https://github.com/gregmos/PII-Shield' },
|
|
58
|
+
{ role: 'assistant', content: 'Le wiki ne décrit pas PII-Shield.' },
|
|
59
|
+
{ role: 'user', content: 'cherche sur internet' },
|
|
60
|
+
{ role: 'user', content: 'cherche sur internet' },
|
|
61
|
+
{ role: 'assistant', content: 'Lancer la recherche.' },
|
|
62
|
+
{ role: 'assistant', content: 'Je suis limité au wiki.' },
|
|
63
|
+
{ role: 'assistant', content: 'Le plan est terminé.' },
|
|
64
|
+
];
|
|
65
|
+
const plan = conversationCompactionPlan({ conversation });
|
|
66
|
+
assert.equal(plan.segment.at(-1).content, 'réponse 5');
|
|
67
|
+
assert.equal(plan.keepLast, 7, 'moved back to the question');
|
|
68
|
+
const events = conversation.map((m) => createAgentEvent(m.role === 'user' ? 'user_message' : 'assistant_message', { payload: { content: m.content } }));
|
|
69
|
+
events.push(createAgentEvent('conversation_reset', { payload: { summary: 'Résumé.', keepLast: plan.keepLast } }));
|
|
70
|
+
const seed = conversationSeed({ agentProjection: reduceAgentEvents(events) }, 'x');
|
|
71
|
+
assert.match(seed[1].content, /PII-Shield/);
|
|
72
|
+
});
|
|
@@ -136,3 +136,52 @@ test('multi-agent scheduler beats sequential on 2 independent build tasks by the
|
|
|
136
136
|
`expected multi-agent duration (${parallelDurationMs}ms) under ${MAX_PARALLEL_TO_SEQUENTIAL_RATIO * 100}% of sequential duration (${sequentialDurationMs}ms, threshold ${thresholdMs}ms)`,
|
|
137
137
|
);
|
|
138
138
|
});
|
|
139
|
+
|
|
140
|
+
// plan-demandes-pendant-run.md, lot 2: the lock registry is the WORKSPACE's.
|
|
141
|
+
// A task whose lock another run (or a direct write) holds must wait for it and
|
|
142
|
+
// say so — before, the run declared itself stalled and stopped.
|
|
143
|
+
test('a task waits for a workspace lock held by another run, then runs', async () => {
|
|
144
|
+
const { workspaceLockRegistry, createLockManager } = await import('../orchestrator/lockManager.js');
|
|
145
|
+
const session = {
|
|
146
|
+
workspace: 'demo-workspace',
|
|
147
|
+
activities: {},
|
|
148
|
+
agentEvents: [],
|
|
149
|
+
headlessPlan: [{ ...plannedBuildTask(1, 'template-a'), locks: ['workspace-write'] }],
|
|
150
|
+
mcp: { production: { status: 'connected', tools: [{ name: 'agent_execute' }, { name: 'agent_status' }, { name: 'agent_cancel' }] } },
|
|
151
|
+
agentRegistrySnapshot: [productionAgent()],
|
|
152
|
+
wikircConfig: { capabilityRouting: {} },
|
|
153
|
+
};
|
|
154
|
+
const registry = workspaceLockRegistry(session);
|
|
155
|
+
const other = createLockManager(registry).acquire(['workspace-write'], 'run-ingest');
|
|
156
|
+
const RELEASE_AFTER_MS = 200;
|
|
157
|
+
const releasedAt = Date.now() + RELEASE_AFTER_MS;
|
|
158
|
+
setTimeout(() => other.release(), RELEASE_AFTER_MS);
|
|
159
|
+
let executedAt = null;
|
|
160
|
+
const callTool = async (_mcp, _serverName, toolName, args) => {
|
|
161
|
+
if (toolName === 'agent_execute') {
|
|
162
|
+
executedAt = Date.now();
|
|
163
|
+
return toolResult({ accepted: true, jobId: `job-${args.taskId}`, status: 'queued' });
|
|
164
|
+
}
|
|
165
|
+
if (toolName === 'agent_status') {
|
|
166
|
+
return toolResult({ jobId: args.jobId, status: 'done', result: { status: 'succeeded', outputRefs: [] } });
|
|
167
|
+
}
|
|
168
|
+
if (toolName === 'agent_cancel') return toolResult({ ok: true });
|
|
169
|
+
throw new Error(`unexpected tool: ${toolName}`);
|
|
170
|
+
};
|
|
171
|
+
const result = await runRuntimeParallelPlan({ invoke: async () => assert.fail('child Donna loop must not run') }, session, 'Build', {
|
|
172
|
+
runId: 'run-build',
|
|
173
|
+
timeoutMs: 10_000,
|
|
174
|
+
maxTurns: 1,
|
|
175
|
+
callTool,
|
|
176
|
+
dispatcherPollIntervalMs: 5,
|
|
177
|
+
});
|
|
178
|
+
assert.equal(result.ok, true, JSON.stringify(result));
|
|
179
|
+
assert.ok(executedAt >= releasedAt - 5, 'the task started only once the other run released the lock');
|
|
180
|
+
const waits = session.agentEvents
|
|
181
|
+
.filter((event) => event.type === 'runtime_log')
|
|
182
|
+
.map((event) => String(event.payload?.message ?? ''))
|
|
183
|
+
.filter((message) => message.includes('waiting for workspace lock'));
|
|
184
|
+
assert.equal(waits.length, 1, 'announced once, not on every poll');
|
|
185
|
+
assert.match(waits[0], /workspace-write \(held by run-ingest\)/);
|
|
186
|
+
assert.deepEqual(registry.locks.size, 0, 'nothing left held once both are done');
|
|
187
|
+
});
|
package/src/runtime/runner.js
CHANGED
|
@@ -5,6 +5,7 @@ import { formatPlanStatus, formatPlanStep } from '../core/plan.js';
|
|
|
5
5
|
import { readyPlanTasks, sanitizePlanForExecution } from '../core/planPatch.js';
|
|
6
6
|
import { createAssignmentManager } from '../orchestrator/assignmentManager.js';
|
|
7
7
|
import { createAttemptManager } from '../orchestrator/attemptManager.js';
|
|
8
|
+
import { workspaceLockRegistry } from '../orchestrator/lockManager.js';
|
|
8
9
|
import { createBudgetManager, BudgetExceededError } from '../orchestrator/budgetManager.js';
|
|
9
10
|
import { createDispatcher } from '../orchestrator/dispatcher.js';
|
|
10
11
|
import { approvalCovered, approvalRequestForTask } from '../orchestrator/approvalPolicy.js';
|
|
@@ -86,6 +87,43 @@ export function conversationSeed(session, currentInput, { limit = 12, maxChars =
|
|
|
86
87
|
return seed;
|
|
87
88
|
}
|
|
88
89
|
|
|
90
|
+
/**
|
|
91
|
+
* Decides whether the conversation memory must be compacted before a turn.
|
|
92
|
+
*
|
|
93
|
+
* `conversationSeed` keeps the last `limit` messages: anything older left
|
|
94
|
+
* Donna's context WITHOUT a summary and without a word — silence is the bug.
|
|
95
|
+
* So a compaction is due as soon as a message would fall out of that window,
|
|
96
|
+
* or when the seed alone takes more than a quarter of the model's input
|
|
97
|
+
* budget (a small-context profile). The last `keepLast` messages stay
|
|
98
|
+
* verbatim: a follow-up needs the previous answer word for word.
|
|
99
|
+
* Pure: returns the segment to summarize, or null.
|
|
100
|
+
*/
|
|
101
|
+
export function conversationCompactionPlan(projection, { limit = 12, keepLast = 6, maxChars = 2000, budgetChars = null } = {}) {
|
|
102
|
+
const conversation = Array.isArray(projection?.conversation) ? projection.conversation : [];
|
|
103
|
+
const seedStart = Math.max(0, Number(projection?.conversationSeedStart) || 0);
|
|
104
|
+
const live = conversation.slice(seedStart)
|
|
105
|
+
.filter((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim());
|
|
106
|
+
const seedChars = live.slice(-limit).reduce((total, message) => total + Math.min(maxChars, String(message.content).length), 0)
|
|
107
|
+
+ String(projection?.conversationSummary ?? '').length;
|
|
108
|
+
const overflow = live.length > limit;
|
|
109
|
+
const heavy = Number(budgetChars) > 0 && seedChars > Number(budgetChars) / 4;
|
|
110
|
+
if (!overflow && !heavy) return null;
|
|
111
|
+
// The cut never splits an exchange: it moves back to the user message that
|
|
112
|
+
// opened it. Observed: a cut between « comment utiliser <url> » and its
|
|
113
|
+
// answer sent the question into the summary, which dropped it — Donna then
|
|
114
|
+
// held an answer without its question and asked what to search for.
|
|
115
|
+
let cut = Math.max(seedStart, conversation.length - keepLast);
|
|
116
|
+
while (cut > seedStart && conversation[cut]?.role !== 'user') cut -= 1;
|
|
117
|
+
const segment = conversation.slice(seedStart, cut);
|
|
118
|
+
if (!segment.some((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim())) return null;
|
|
119
|
+
return { segment, keepLast: conversation.length - cut, reason: overflow ? 'window' : 'budget', previousSummary: projection?.conversationSummary ?? null };
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/** What Donna is told after an automatic compaction — she says it, not the system. */
|
|
123
|
+
export function compactionNoteForDonna(count) {
|
|
124
|
+
return `[Memory note for Donna — not written by the user] The ${count} oldest messages of this conversation were just condensed automatically into the summary above, to stay within the model's context; the messages below are verbatim. After answering the user's next question, tell them so in one short sentence, in the reply language.`;
|
|
125
|
+
}
|
|
126
|
+
|
|
89
127
|
export async function runRuntimeAgenticLoop(agent, session, initialInput, { signal, timeoutMs, maxTurns, runId, pollBusy, parallelHandoff = false, initialMessages = [] }) {
|
|
90
128
|
return runAgenticLoop(agent, session, initialInput, {
|
|
91
129
|
signal,
|
|
@@ -419,7 +457,15 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
419
457
|
cappedByCeiling: concurrencyDetail.cappedByCeiling,
|
|
420
458
|
};
|
|
421
459
|
const active = new Map();
|
|
422
|
-
|
|
460
|
+
// The workspace's lock registry, shared with every other run and direct
|
|
461
|
+
// action of this workspace; this run holds its locks under its own id.
|
|
462
|
+
const lockRegistry = workspaceLockRegistry(session);
|
|
463
|
+
const attempts = attemptManager ?? createAttemptManager({
|
|
464
|
+
locks: lockRegistry.locks,
|
|
465
|
+
owners: lockRegistry.owners,
|
|
466
|
+
owner: runId ?? 'run',
|
|
467
|
+
});
|
|
468
|
+
let announcedLockWait = '';
|
|
423
469
|
const assigner = assignmentManager ?? createAssignmentManager({ session });
|
|
424
470
|
const executor = dispatcher ?? createDispatcher({
|
|
425
471
|
session,
|
|
@@ -648,6 +694,24 @@ export async function runRuntimeParallelPlan(agent, session, input, {
|
|
|
648
694
|
// notamment derrière une barrière de groupe désormais terminale.
|
|
649
695
|
continue;
|
|
650
696
|
}
|
|
697
|
+
// A task whose only blocker is a lock held by ANOTHER run or a direct
|
|
698
|
+
// action waits for it: that holder will release it, so this is a
|
|
699
|
+
// queue, not a stall. Said once per change of holder, never silently.
|
|
700
|
+
const lockWaits = (session.headlessPlan ?? [])
|
|
701
|
+
.filter((step) => pendingSchedulerStatus(step.status))
|
|
702
|
+
.map((step) => ({ step, holders: attempts.foreignHolders?.(step) ?? [] }))
|
|
703
|
+
.filter((entry) => entry.holders.length > 0);
|
|
704
|
+
if (lockWaits.length > 0) {
|
|
705
|
+
const held = [...new Map(lockWaits.flatMap((entry) => entry.holders)
|
|
706
|
+
.map((holder) => [holder.lock, holder])).values()];
|
|
707
|
+
const summary = held.map((holder) => `${holder.lock} (held by ${holder.owner ?? 'another action'})`).join(', ');
|
|
708
|
+
if (summary !== announcedLockWait) {
|
|
709
|
+
announcedLockWait = summary;
|
|
710
|
+
emitRuntimeLog(session, `scheduler: waiting for workspace lock(s) — ${summary}; ${lockWaits.length} task(s) queued behind`);
|
|
711
|
+
}
|
|
712
|
+
await new Promise((resolveDelay) => setTimeout(resolveDelay, 500));
|
|
713
|
+
continue;
|
|
714
|
+
}
|
|
651
715
|
// Any genuine approval-only block returned above. Remaining tasks are
|
|
652
716
|
// unschedulable for another reason.
|
|
653
717
|
const reason = 'no_ready_plan_task';
|