@dotdrelle/wiki-manager 0.15.100 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/mcp.endpoints.example.json +1 -1
  2. package/package.json +2 -2
  3. package/src/agent/graph.js +63 -13
  4. package/src/agent/graph.test.js +125 -1
  5. package/src/agent/llm.js +13 -4
  6. package/src/agent/llm.test.js +59 -0
  7. package/src/cli/wiki-manager.js +43 -3
  8. package/src/commands/slash.js +2 -0
  9. package/src/core/agentEvents.js +12 -2
  10. package/src/core/buildInfo.json +2 -2
  11. package/src/core/env.js +11 -2
  12. package/src/core/env.test.js +22 -6
  13. package/src/core/llmCapabilities.js +31 -0
  14. package/src/core/llmCapabilities.test.js +27 -0
  15. package/src/core/logLabel.js +9 -0
  16. package/src/core/logLabel.test.js +12 -0
  17. package/src/core/mcp.js +2 -2
  18. package/src/core/toolLoop.js +222 -20
  19. package/src/core/toolLoop.test.js +324 -0
  20. package/src/core/wikiPresearch.js +58 -0
  21. package/src/core/wikirc.js +61 -0
  22. package/src/core/wikirc.test.js +40 -1
  23. package/src/core/workflow.js +4 -1
  24. package/src/orchestrator/attemptManager.js +21 -5
  25. package/src/orchestrator/attemptManager.test.js +19 -0
  26. package/src/orchestrator/dispatcher.js +49 -8
  27. package/src/orchestrator/dispatcher.test.js +33 -1
  28. package/src/orchestrator/lockManager.js +40 -5
  29. package/src/orchestrator/resultAggregator.js +12 -1
  30. package/src/orchestrator/resultAggregator.test.js +29 -0
  31. package/src/runtime/controlClassify.test.js +85 -1
  32. package/src/runtime/conversationCompact.js +39 -0
  33. package/src/runtime/conversationCompaction.test.js +72 -0
  34. package/src/runtime/runner.e2e.test.js +49 -0
  35. package/src/runtime/runner.js +65 -1
  36. package/src/runtime/server.js +83 -75
  37. package/src/runtime/server.test.js +121 -0
  38. package/src/runtime/store.js +17 -1
  39. package/src/runtime/store.test.js +22 -0
  40. package/src/runtime/workspaceIsolation.test.js +21 -12
  41. package/src/shell/repl.js +148 -27
  42. package/src/shell/repl.test.js +182 -1
@@ -2,6 +2,7 @@ import { normalizeActivity, parseJsonText } from '../core/activity.js';
2
2
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
3
3
  import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
4
4
  import { loadWorkspaceProfile } from '../core/profile.js';
5
+ import { supportsTemperature } from '../core/llmCapabilities.js';
5
6
  import { containerReachableUrl } from '../core/wikiSetup.js';
6
7
  import { mapRuntimeEvent } from '../core/runtimeEventAdapter.js';
7
8
  import { emitRuntimeLog, pollActivitiesOnce } from '../runtime/supervisor.js';
@@ -259,7 +260,7 @@ async function executeExternalRuntime(task, assignment, {
259
260
  model: activeProfileModel(session),
260
261
  language: session?.language ?? session?.wikircConfig?.language ?? null,
261
262
  mcp: mcpPool,
262
- systemPrompt: activeRuntimeSystemPrompt(session, task, assignment),
263
+ systemPrompt: activeRuntimeSystemPrompt(session, task, assignment, mcpPool),
263
264
  });
264
265
  runtimeRunId = String(accepted?.runId ?? '');
265
266
  if (!runtimeRunId) throw new Error('runtime.execute did not return runId.');
@@ -508,6 +509,10 @@ function activeProfileModel(session) {
508
509
  ...(llm.apiKey ? { apiKey: String(llm.apiKey) } : {}),
509
510
  };
510
511
  for (const key of ['temperature', 'maxTokens', 'topP', 'seed']) {
512
+ // A gpt-5-class model rejects `temperature`: forwarding the profile's value
513
+ // would make the external runtime's model call fail (HTTP 400), same as the
514
+ // manager's own client.
515
+ if (key === 'temperature' && !supportsTemperature(llm)) continue;
511
516
  const value = Number(llm[key]);
512
517
  if (Number.isFinite(value)) model[key] = value;
513
518
  }
@@ -552,8 +557,8 @@ function isReadOnlyExternalTool(toolName) {
552
557
  // The runtime's EYES, per run: the active workspace's wiki MCP (read tools
553
558
  // only) PLUS the declared external MCP endpoints that are safe to hand over
554
559
  // (connected, no approval-gated tools, not a workspace-mutating server) —
555
- // typically web search (exa). The allow-list here is the authority: nothing
556
- // else reaches the runtime.
560
+ // typically a web-search connector. The allow-list here is the authority:
561
+ // nothing else reaches the runtime.
557
562
  export function activeProfileMcp(session) {
558
563
  const blocks = [];
559
564
  const wiki = session?.mcp?.wiki;
@@ -585,8 +590,9 @@ export function activeProfileMcp(session) {
585
590
  }
586
591
  // External connectors ride along ONLY when the operator declared them safe
587
592
  // for the runtime's eyes. A connector added from the serve panel lands here
588
- // too — without this, exa was offered in chat but the agentic path
589
- // delegated to the gateway and the Deep Agent answered it had no web tools.
593
+ // too — without this, a web connector was offered in chat but the agentic
594
+ // path delegated to the gateway and the Deep Agent answered it had no web
595
+ // tools.
590
596
  const EXCLUDED_EXTERNAL_SERVERS = new Set(['cme', 'documents', 'connectors', 'production']);
591
597
  for (const [name, entry] of Object.entries(session?.mcp ?? {})) {
592
598
  if (!entry?.external || entry.status !== 'connected') continue;
@@ -611,21 +617,56 @@ export function activeProfileMcp(session) {
611
617
  // description), the eyes/bouche/mains boundary, the workspace profile and the
612
618
  // reply language. Without it, the runtime falls back to deepagents' generic
613
619
  // assistant prompt — which is exactly the "upload your project" hallucination.
614
- function activeRuntimeSystemPrompt(session, task, assignment) {
620
+ // The runtime prompt must describe the pool the same dispatch actually hand
621
+ // over. It used to assert "READ tools only (the workspace wiki)" whatever the
622
+ // pool held, so a run handed an external web-search read tool still answered
623
+ // "je suis limité aux seules sources du wiki" and made zero tool calls
624
+ // (observed on acpi: "cherche sur internet" → agent.answer → refusal, gateway
625
+ // /metrics tools:0). The pool is the authority; the prompt reports it, never
626
+ // denies it.
627
+ function describeRuntimePool(mcpPool) {
628
+ const blocks = Array.isArray(mcpPool) ? mcpPool : [];
629
+ const wiki = blocks.find((block) => block?.name === 'wiki');
630
+ const hasWiki = Boolean(wiki && Array.isArray(wiki.tools) && wiki.tools.length > 0);
631
+ const external = blocks
632
+ .filter((block) => block && block.name !== 'wiki' && Array.isArray(block.tools) && block.tools.length > 0)
633
+ .map((block) => ({
634
+ name: String(block.name),
635
+ tools: block.tools.map((tool) => `${block.name}__${String(tool)}`),
636
+ }));
637
+ return { hasWiki, external };
638
+ }
639
+
640
+ export function activeRuntimeSystemPrompt(session, task, assignment, mcpPool = null) {
615
641
  const capability = assignment?.capability ?? null;
616
642
  const description = String(capability?.description ?? '').trim();
617
643
  const language = session?.language ?? session?.wikircConfig?.language ?? null;
618
644
  const profile = loadWorkspaceProfile(session?.workspacePath);
645
+ const { hasWiki, external } = describeRuntimePool(mcpPool);
646
+ const externalTools = external.flatMap((block) => block.tools);
647
+ // Wiki first, then the declared external read tools. A run that has a web
648
+ // search tool must not refuse with "I cannot search the internet": that
649
+ // refusal is exactly the symptom this line exists to prevent.
650
+ const vision = externalTools.length > 0
651
+ ? [
652
+ `Beyond the wiki you also have these READ tools: ${externalTools.join(', ')}.`,
653
+ hasWiki
654
+ ? 'Search the workspace wiki FIRST; then use those external/web read tools for current, public or out-of-workspace facts the wiki does not cover.'
655
+ : 'Use those external/web read tools for facts the workspace does not cover.',
656
+ 'Never answer that you cannot search the internet or that you are limited to the wiki: when such a tool is listed here, call it.',
657
+ ].join(' ')
658
+ : null;
619
659
  return [
620
660
  'You are the agentic analysis engine of a knowledge workspace (wikiLLM), executed behind the manager Donna.',
621
661
  `Execute exactly ONE capability: ${task?.requiredCapability ?? 'unknown'}${description ? ` — ${description}` : ''}.`,
622
662
  `Operation: ${task?.operation ?? 'run'}.`,
623
- 'Boundary: you have READ tools only (the workspace wiki). You never modify the workspace — structural changes are proposals you return in your final answer (a planExpansionRequest), the manager integrates them under human approval. Side-effects on the outside world are gated by approval.',
663
+ 'Boundary: you have READ tools only (the tools this run makes available, listed in your pool). You never modify the workspace — structural changes are proposals you return in your final answer (a planExpansionRequest), the manager integrates them under human approval. Side-effects on the outside world are gated by approval.',
664
+ vision,
624
665
  'Ground every claim in what the read tools return. Never invent pages, names, facts, jobs or results.',
625
666
  'Tool discipline: discover real page paths with the list/search tools BEFORE reading. Never guess a path — a read refused for "path not allowed" means the path was invented, so list/search first, then read exactly what exists.',
626
667
  ...(language ? [`Reply in the workspace language: ${language}.`] : []),
627
668
  ...(profile ? [`Workspace preferences — apply them to every reply:\n${profile}`] : []),
628
- ].join('\n');
669
+ ].filter(Boolean).join('\n');
629
670
  }
630
671
 
631
672
  function dispatchTaskActivity(session, task, assignment, jobId, statusTool, runId) {
@@ -1,6 +1,6 @@
1
1
  import assert from 'node:assert/strict';
2
2
  import test from 'node:test';
3
- import { activeProfileMcp, createDispatcher, normalizeTaskError, RUNTIME_SHUTDOWN_ABORT_REASON } from './dispatcher.js';
3
+ import { activeProfileMcp, activeRuntimeSystemPrompt, createDispatcher, normalizeTaskError, RUNTIME_SHUTDOWN_ABORT_REASON } from './dispatcher.js';
4
4
 
5
5
  test('activeProfileMcp forwards only the read-only wiki tools to the external runtime', () => {
6
6
  const session = {
@@ -114,6 +114,38 @@ test('activeProfileMcp rewrites loopback URLs for the container the runtime runs
114
114
  assert.equal(pool.find((block) => block.name === 'hosted').url, 'https://mcp.exa.ai/mcp');
115
115
  });
116
116
 
117
+ test('activeRuntimeSystemPrompt names the declared web tools and forbids "cannot search the internet"', () => {
118
+ const session = { workspace: 'acpi', language: 'fr-FR' };
119
+ const pool = [
120
+ { name: 'wiki', tools: ['wiki_read_page', 'wiki_search_context'] },
121
+ { name: 'exa', tools: ['web_search_exa', 'web_fetch_exa'] },
122
+ ];
123
+ const prompt = activeRuntimeSystemPrompt(
124
+ session,
125
+ { requiredCapability: 'agent.answer', operation: 'run' },
126
+ { capability: { description: 'wiki and web research' } },
127
+ pool,
128
+ );
129
+ assert.match(prompt, /exa__web_search_exa/);
130
+ assert.match(prompt, /exa__web_fetch_exa/);
131
+ assert.match(prompt, /wiki FIRST/i);
132
+ assert.match(prompt, /cannot search the internet/i);
133
+ assert.match(prompt, /agent\.answer/);
134
+ });
135
+
136
+ test('activeRuntimeSystemPrompt stays wiki-only when the pool declares no external read tools', () => {
137
+ const session = { workspace: 'acpi' };
138
+ const prompt = activeRuntimeSystemPrompt(
139
+ session,
140
+ { requiredCapability: 'agent.review' },
141
+ { capability: {} },
142
+ [{ name: 'wiki', tools: ['wiki_read_page'] }],
143
+ );
144
+ assert.doesNotMatch(prompt, /exa__/);
145
+ assert.doesNotMatch(prompt, /cannot search the internet/i);
146
+ assert.match(prompt, /the tools this run makes available/);
147
+ });
148
+
117
149
  test('dispatcher returns a retryable logical failure when agent_execute reports workspace_busy', async () => {
118
150
  const session = {
119
151
  workspace: 'test',
@@ -1,27 +1,62 @@
1
- export function createLockManager({ locks = new Set() } = {}) {
1
+ /*
2
+ One registry per WORKSPACE, not per run.
3
+
4
+ The registry used to be created by each run's attemptManager, so its locks
5
+ only ever excluded tasks of the same run. That was harmless while a
6
+ workspace ran one run at a time, but anything acting beside a run — a
7
+ direct write from a chat turn, a second run — held nothing the first one
8
+ could see. `workspaceLockRegistry` hangs one registry on the workspace
9
+ session; every run and every direct action shares it
10
+ (plan-demandes-pendant-run.md, lot 2). `owners` records who holds each lock
11
+ so a wait can name its holder instead of stalling in silence.
12
+ */
13
+ export function workspaceLockRegistry(session) {
14
+ if (!session) return { locks: new Set(), owners: new Map() };
15
+ session._workspaceLocks ??= { locks: new Set(), owners: new Map() };
16
+ return session._workspaceLocks;
17
+ }
18
+
19
+ export function createLockManager({ locks = new Set(), owners = new Map() } = {}) {
2
20
  return {
3
21
  canAcquire(taskOrLocks) {
4
22
  return locksFor(taskOrLocks).every((lock) => !locks.has(lock));
5
23
  },
6
- acquire(taskOrLocks) {
24
+ acquire(taskOrLocks, owner = null) {
7
25
  const lockNames = locksFor(taskOrLocks);
8
26
  if (lockNames.some((lock) => locks.has(lock))) return null;
9
- for (const lock of lockNames) locks.add(lock);
27
+ for (const lock of lockNames) {
28
+ locks.add(lock);
29
+ if (owner != null) owners.set(lock, owner);
30
+ }
10
31
  let released = false;
11
32
  return {
12
33
  locks: lockNames,
13
34
  release() {
14
35
  if (released) return;
15
36
  released = true;
16
- for (const lock of lockNames) locks.delete(lock);
37
+ for (const lock of lockNames) {
38
+ locks.delete(lock);
39
+ owners.delete(lock);
40
+ }
17
41
  },
18
42
  };
19
43
  },
20
44
  release(taskOrLocks) {
21
- for (const lock of locksFor(taskOrLocks)) locks.delete(lock);
45
+ for (const lock of locksFor(taskOrLocks)) {
46
+ locks.delete(lock);
47
+ owners.delete(lock);
48
+ }
49
+ },
50
+ // Who holds the locks a task would need: [{ lock, owner }], owner null
51
+ // when the holder did not name itself.
52
+ holders(taskOrLocks) {
53
+ return locksFor(taskOrLocks)
54
+ .filter((lock) => locks.has(lock))
55
+ .map((lock) => ({ lock, owner: owners.get(lock) ?? null }));
22
56
  },
23
57
  clear() {
24
58
  locks.clear();
59
+ owners.clear();
25
60
  },
26
61
  snapshot() {
27
62
  return [...locks].sort();
@@ -78,6 +78,13 @@ export async function accept(result, {
78
78
  taskId,
79
79
  payload: { message: `agent-proposal: could not persist the worktree proposal for ${taskId}: ${worktreePersisted.error}` },
80
80
  })));
81
+ } else if (worktreePersisted.empty) {
82
+ persistDispatch(store, dispatchAgentEvent(session, createAgentEvent('runtime_log', {
83
+ origin: 'result_aggregator',
84
+ runId,
85
+ taskId,
86
+ payload: { message: `⚠ agent-proposal: ${taskId} wrote no file on its branch — nothing to review; the corrections its report describes were not written` },
87
+ })));
81
88
  } else if (worktreePersisted.path) {
82
89
  persistDispatch(store, dispatchAgentEvent(session, createAgentEvent('runtime_log', {
83
90
  origin: 'result_aggregator',
@@ -315,7 +322,11 @@ function persistWorktreeProposal(session, result, { runId, taskId }) {
315
322
  ?? result?.worktreeProposal;
316
323
  if (!proposal || typeof proposal !== 'object') return { path: null };
317
324
  const changes = Array.isArray(proposal.changes) ? proposal.changes : [];
318
- if (changes.length === 0) return { path: null };
325
+ // A curation that wrote nothing has nothing to review — but it must SAY so.
326
+ // Returning silently here left the review page empty while the run was
327
+ // reported a success, its report describing corrections on a branch that
328
+ // held none (observed on acpi).
329
+ if (changes.length === 0) return { path: null, empty: true };
319
330
  const workspacePath = session?.workspacePath;
320
331
  if (!workspacePath || typeof workspacePath !== 'string') {
321
332
  return { error: 'no workspace path on the session — the proposal stays in the run result only' };
@@ -409,3 +409,32 @@ test('a gateway worktree proposal (rawStatus shape) is persisted for review', as
409
409
  rmSync(workspacePath, { recursive: true, force: true });
410
410
  }
411
411
  });
412
+
413
+ test('a curation that wrote no file says so instead of vanishing from the review page', async () => {
414
+ // Observed on acpi: the Redactor described its corrections in prose on a
415
+ // branch it invented, wrote nothing, and the run was reported a success
416
+ // while /agent-proposals stayed empty with no word anywhere.
417
+ const workspacePath = mkdtempSync(join(tmpdir(), 'worktree-proposal-'));
418
+ const session = { agentEvents: [], activities: {}, workspace: 'docs', workspacePath, headlessPlan: [] };
419
+ try {
420
+ await accept({
421
+ ok: true,
422
+ taskId: 't-curate',
423
+ status: 'completed',
424
+ outputRefs: [],
425
+ rawStatus: {
426
+ runId: 'gateway-1',
427
+ status: 'completed',
428
+ result: { status: 'completed', content: 'report', worktreeProposal: { branch: 'agent/gateway-1', changedFiles: [], changes: [], diff: '' } },
429
+ },
430
+ }, {
431
+ session,
432
+ runId: 'run-curate',
433
+ task: { id: 't-curate', requiredCapability: 'agent.curate', operation: 'run' },
434
+ });
435
+ assert.equal(existsSync(join(workspacePath, '.wiki', 'agent-proposals')), false);
436
+ assert.ok(session.agentEvents.some((event) => /wrote no file on its branch — nothing to review/.test(String(event.payload?.message ?? ''))));
437
+ } finally {
438
+ rmSync(workspacePath, { recursive: true, force: true });
439
+ }
440
+ });
@@ -15,10 +15,11 @@ test('a bare confirmation during a run is a status check', async () => {
15
15
  }
16
16
  });
17
17
 
18
- test('a plan change that merely opens on a yes is still a plan change', async () => {
18
+ test('a plan change that merely opens on a yes reaches the model, not the status branch', async () => {
19
19
  const result = await classifyControlMessage(
20
20
  'oui, ajoute une étape de polish après le build',
21
21
  running,
22
+ { llm: { complete: async () => 'plan_change' } },
22
23
  );
23
24
  assert.equal(result.kind, 'modify_run');
24
25
  });
@@ -29,3 +30,86 @@ test('a new task that opens on a yes reaches the model classifier, not the statu
29
30
  });
30
31
  assert.equal(result.kind, 'enqueue_run');
31
32
  });
33
+
34
+ // The empty-chat "Curate the wiki" tile sends free-text prose. While a run was
35
+ // active, its "pages that disagree or repeat each other" matched the
36
+ // deterministic modify_run keyword "each" and the curation became an invisible
37
+ // plan patch instead of a queued run.
38
+ test('a curation objective during a run is queued, never an invisible plan patch', async () => {
39
+ const objective =
40
+ 'Curate the wiki: find duplicate pages, pages that disagree or repeat each other, '
41
+ + 'outdated or superseded pages, and claims with no cited source, then write the '
42
+ + 'corrections on a dedicated branch.';
43
+ const result = await classifyControlMessage(objective, running, {
44
+ llm: { complete: async () => 'action' },
45
+ });
46
+ assert.equal(result.kind, 'enqueue_run');
47
+ });
48
+
49
+ test('a question about curation stays read-only', async () => {
50
+ const result = await classifyControlMessage('explain how curation works', running, {
51
+ llm: { complete: async () => 'question' },
52
+ });
53
+ assert.equal(result.kind, 'converse');
54
+ });
55
+
56
+ test('a genuine plan change is classified as modify_run by the model', async () => {
57
+ const result = await classifyControlMessage('ajoute une étape de polish après le build', running, {
58
+ llm: { complete: async () => 'plan_change' },
59
+ });
60
+ assert.equal(result.kind, 'modify_run');
61
+ });
62
+
63
+ // Measured 2026-09-25 (plan-demandes-pendant-run.md §2): keywords inside a
64
+ // sentence decided before the model, so questions typed during an ingest were
65
+ // answered with the run status, turned into patches of the running plan, or
66
+ // cancelled the run. No keyword inside a sentence decides any more.
67
+ test('a keyword inside a question never decides the triage', async () => {
68
+ const asked = [];
69
+ const llm = { complete: async ({ input }) => { asked.push(input); return 'question'; } };
70
+ for (const question of [
71
+ "montre-moi ce que dit le wiki sur l'option A",
72
+ 'explique la différence entre les options A et B',
73
+ 'cherche les pages qui parlent du plan de charge',
74
+ 'que se passe-t-il après la validation dans SISBA ?',
75
+ "quand est-ce qu'on stop le support de CDPOM ?",
76
+ 'que dit le wiki sur la curation des données ?',
77
+ "qu'est-ce qu'on fait ensuite dans le projet ?",
78
+ ]) {
79
+ const result = await classifyControlMessage(question, running, { llm });
80
+ assert.equal(result.kind, 'converse', question);
81
+ }
82
+ assert.equal(asked.length, 7, 'every one of them reached the model');
83
+ });
84
+
85
+ test('the model categories map onto the control kinds', async () => {
86
+ for (const [word, kind] of [
87
+ ['question', 'converse'], ['status', 'observe'], ['action', 'enqueue_run'],
88
+ ['plan_change', 'modify_run'], ['cancel', 'cancel'], ['Action.', 'enqueue_run'],
89
+ ]) {
90
+ const result = await classifyControlMessage('modifie la page option-a', running, {
91
+ llm: { complete: async () => word },
92
+ });
93
+ assert.equal(result.kind, kind, word);
94
+ }
95
+ });
96
+
97
+ test('only a whole-message command is decided without the model', async () => {
98
+ const cases = [
99
+ ['stop', 'cancel'], ['annule', 'cancel'], ['arrête tout !', 'cancel'], ['cancel the run', 'cancel'],
100
+ ["annule l'ingestion", 'cancel'], ['status', 'observe'], ['où en est-on ?', 'observe'],
101
+ ['logs', 'observe'], ['quel est le statut du run ?', 'observe'],
102
+ ["mets en file l'export", 'enqueue_run'], ['lance le build après ce run', 'enqueue_run'],
103
+ ];
104
+ for (const [input, kind] of cases) {
105
+ const result = await classifyControlMessage(input, running);
106
+ assert.equal(result.kind, kind, input);
107
+ }
108
+ });
109
+
110
+ test('without a model every other message is read-only conversation', async () => {
111
+ for (const input of ['curate the wiki', 'modifie la page option-a', "quand est-ce qu'on stop le support ?"]) {
112
+ const result = await classifyControlMessage(input, running);
113
+ assert.equal(result.kind, 'converse', input);
114
+ }
115
+ });
@@ -0,0 +1,39 @@
1
+ import { emitRuntimeLog } from './supervisor.js';
2
+
3
+ const CONVERSATION_SUMMARY_TIMEOUT_MS = 20_000;
4
+ const CONVERSATION_SUMMARY_MAX_INPUT_CHARS = 8_000;
5
+
6
+ /*
7
+ A compact does not just cut older turns from conversationSeed — it replaces
8
+ them with a short rolling summary, so a decision made 20 messages ago is not
9
+ gone from Donna's grounding entirely, only condensed. Best-effort: no LLM
10
+ configured, an empty reply, or a call failure all fall back to keeping
11
+ whatever summary already existed (never worse than before this compact),
12
+ the same deterministic-under-failure shape as generateControlAcknowledgment.
13
+ */
14
+ export async function summarizeCompactedConversation(session, { previousSummary, segment }) {
15
+ const llm = session?.llm;
16
+ const transcript = (Array.isArray(segment) ? segment : [])
17
+ .filter((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim())
18
+ .map((message) => `${message.role === 'user' ? 'User' : 'Assistant'}: ${String(message.content).trim()}`)
19
+ .join('\n')
20
+ .slice(0, CONVERSATION_SUMMARY_MAX_INPUT_CHARS);
21
+ if (!transcript) return previousSummary || null;
22
+ if (!(llm && typeof llm.complete === 'function')) return previousSummary || null;
23
+ try {
24
+ const reply = await llm.complete({
25
+ system: 'You maintain a compact working memory for Donna, a workspace assistant. You are shown an optional PREVIOUS SUMMARY and a NEW SEGMENT of conversation about to leave the assistant\'s context window. Write ONE updated summary that preserves the facts, decisions, open questions and user preferences that still matter for future turns. Be concise: well under 200 words. Return only the summary text — no preamble, no meta-commentary, no headings.',
26
+ input: [
27
+ previousSummary ? `PREVIOUS SUMMARY:\n${previousSummary}` : null,
28
+ `NEW SEGMENT:\n${transcript}`,
29
+ ].filter(Boolean).join('\n\n'),
30
+ signal: AbortSignal.timeout(CONVERSATION_SUMMARY_TIMEOUT_MS),
31
+ });
32
+ const text = String(reply ?? '').trim();
33
+ if (text) return text;
34
+ emitRuntimeLog(session, 'conversation-compact: LLM returned an empty summary, keeping the previous one');
35
+ } catch (err) {
36
+ emitRuntimeLog(session, `conversation-compact: summary LLM call failed, keeping the previous summary — ${err instanceof Error ? err.message : String(err)}`);
37
+ }
38
+ return previousSummary || null;
39
+ }
@@ -0,0 +1,72 @@
1
+ import assert from 'node:assert/strict';
2
+ import test from 'node:test';
3
+ import { conversationCompactionPlan, compactionNoteForDonna, conversationSeed } from './runner.js';
4
+ import { createAgentEvent, reduceAgentEvents } from '../core/agentEvents.js';
5
+
6
+ const exchange = (n) => [
7
+ { role: 'user', content: `question ${n}` },
8
+ { role: 'assistant', content: `réponse ${n}` },
9
+ ];
10
+
11
+ test('no compaction while the whole conversation fits the seed window', () => {
12
+ const conversation = [1, 2, 3].flatMap(exchange);
13
+ assert.equal(conversationCompactionPlan({ conversation }), null);
14
+ });
15
+
16
+ test('a message about to leave the window triggers a compaction that keeps the last exchanges', () => {
17
+ const conversation = [1, 2, 3, 4, 5, 6, 7].flatMap(exchange); // 14 > 12
18
+ const plan = conversationCompactionPlan({ conversation });
19
+ assert.equal(plan.reason, 'window');
20
+ assert.equal(plan.keepLast, 6);
21
+ assert.equal(plan.segment.length, 8);
22
+ assert.equal(plan.segment.at(-1).content, 'réponse 4');
23
+ });
24
+
25
+ test('a small-context profile compacts on the budget before the window fills', () => {
26
+ const conversation = [1, 2].flatMap(exchange).map((m) => ({ ...m, content: `${m.content} ${'x'.repeat(900)}` }));
27
+ assert.equal(conversationCompactionPlan({ conversation }, { budgetChars: 100000 }), null);
28
+ const plan = conversationCompactionPlan({ conversation }, { budgetChars: 4000, keepLast: 2 });
29
+ assert.equal(plan.reason, 'budget');
30
+ assert.equal(plan.segment.length, 2);
31
+ });
32
+
33
+ test('the compact boundary keeps the last exchanges verbatim in the seed', () => {
34
+ const events = [1, 2, 3, 4, 5, 6, 7].flatMap((n) => [
35
+ createAgentEvent('user_message', { payload: { content: `question ${n}` } }),
36
+ createAgentEvent('assistant_message', { payload: { content: `réponse ${n}` } }),
37
+ ]);
38
+ events.push(createAgentEvent('conversation_reset', { payload: { summary: 'Résumé 1-4.', keepLast: 6, automatic: true } }));
39
+ const projection = reduceAgentEvents(events);
40
+ const seed = conversationSeed({ agentProjection: projection }, 'rajoute la liste');
41
+ assert.match(seed[0].content, /Résumé 1-4\./);
42
+ assert.deepEqual(seed.slice(1).map((m) => m.content), ['question 5', 'réponse 5', 'question 6', 'réponse 6', 'question 7', 'réponse 7']);
43
+ assert.equal(conversationCompactionPlan(projection), null, 'nothing more to compact');
44
+ });
45
+
46
+ test('Donna is the one who tells the user, in the reply language', () => {
47
+ const note = compactionNoteForDonna(8);
48
+ assert.match(note, /8 oldest messages/);
49
+ assert.match(note, /tell them so in one short sentence, in the reply language/);
50
+ });
51
+
52
+ test('the compact cut never separates a question from its answer', () => {
53
+ // Observed on acpi: 6 raw messages kept, the cut fell between the user's
54
+ // question and Donna's answer, and the summary dropped the question.
55
+ const conversation = [
56
+ ...[1, 2, 3, 4, 5].flatMap(exchange),
57
+ { role: 'user', content: 'comment utiliser https://github.com/gregmos/PII-Shield' },
58
+ { role: 'assistant', content: 'Le wiki ne décrit pas PII-Shield.' },
59
+ { role: 'user', content: 'cherche sur internet' },
60
+ { role: 'user', content: 'cherche sur internet' },
61
+ { role: 'assistant', content: 'Lancer la recherche.' },
62
+ { role: 'assistant', content: 'Je suis limité au wiki.' },
63
+ { role: 'assistant', content: 'Le plan est terminé.' },
64
+ ];
65
+ const plan = conversationCompactionPlan({ conversation });
66
+ assert.equal(plan.segment.at(-1).content, 'réponse 5');
67
+ assert.equal(plan.keepLast, 7, 'moved back to the question');
68
+ const events = conversation.map((m) => createAgentEvent(m.role === 'user' ? 'user_message' : 'assistant_message', { payload: { content: m.content } }));
69
+ events.push(createAgentEvent('conversation_reset', { payload: { summary: 'Résumé.', keepLast: plan.keepLast } }));
70
+ const seed = conversationSeed({ agentProjection: reduceAgentEvents(events) }, 'x');
71
+ assert.match(seed[1].content, /PII-Shield/);
72
+ });
@@ -136,3 +136,52 @@ test('multi-agent scheduler beats sequential on 2 independent build tasks by the
136
136
  `expected multi-agent duration (${parallelDurationMs}ms) under ${MAX_PARALLEL_TO_SEQUENTIAL_RATIO * 100}% of sequential duration (${sequentialDurationMs}ms, threshold ${thresholdMs}ms)`,
137
137
  );
138
138
  });
139
+
140
+ // plan-demandes-pendant-run.md, lot 2: the lock registry is the WORKSPACE's.
141
+ // A task whose lock another run (or a direct write) holds must wait for it and
142
+ // say so — before, the run declared itself stalled and stopped.
143
+ test('a task waits for a workspace lock held by another run, then runs', async () => {
144
+ const { workspaceLockRegistry, createLockManager } = await import('../orchestrator/lockManager.js');
145
+ const session = {
146
+ workspace: 'demo-workspace',
147
+ activities: {},
148
+ agentEvents: [],
149
+ headlessPlan: [{ ...plannedBuildTask(1, 'template-a'), locks: ['workspace-write'] }],
150
+ mcp: { production: { status: 'connected', tools: [{ name: 'agent_execute' }, { name: 'agent_status' }, { name: 'agent_cancel' }] } },
151
+ agentRegistrySnapshot: [productionAgent()],
152
+ wikircConfig: { capabilityRouting: {} },
153
+ };
154
+ const registry = workspaceLockRegistry(session);
155
+ const other = createLockManager(registry).acquire(['workspace-write'], 'run-ingest');
156
+ const RELEASE_AFTER_MS = 200;
157
+ const releasedAt = Date.now() + RELEASE_AFTER_MS;
158
+ setTimeout(() => other.release(), RELEASE_AFTER_MS);
159
+ let executedAt = null;
160
+ const callTool = async (_mcp, _serverName, toolName, args) => {
161
+ if (toolName === 'agent_execute') {
162
+ executedAt = Date.now();
163
+ return toolResult({ accepted: true, jobId: `job-${args.taskId}`, status: 'queued' });
164
+ }
165
+ if (toolName === 'agent_status') {
166
+ return toolResult({ jobId: args.jobId, status: 'done', result: { status: 'succeeded', outputRefs: [] } });
167
+ }
168
+ if (toolName === 'agent_cancel') return toolResult({ ok: true });
169
+ throw new Error(`unexpected tool: ${toolName}`);
170
+ };
171
+ const result = await runRuntimeParallelPlan({ invoke: async () => assert.fail('child Donna loop must not run') }, session, 'Build', {
172
+ runId: 'run-build',
173
+ timeoutMs: 10_000,
174
+ maxTurns: 1,
175
+ callTool,
176
+ dispatcherPollIntervalMs: 5,
177
+ });
178
+ assert.equal(result.ok, true, JSON.stringify(result));
179
+ assert.ok(executedAt >= releasedAt - 5, 'the task started only once the other run released the lock');
180
+ const waits = session.agentEvents
181
+ .filter((event) => event.type === 'runtime_log')
182
+ .map((event) => String(event.payload?.message ?? ''))
183
+ .filter((message) => message.includes('waiting for workspace lock'));
184
+ assert.equal(waits.length, 1, 'announced once, not on every poll');
185
+ assert.match(waits[0], /workspace-write \(held by run-ingest\)/);
186
+ assert.deepEqual(registry.locks.size, 0, 'nothing left held once both are done');
187
+ });
@@ -5,6 +5,7 @@ import { formatPlanStatus, formatPlanStep } from '../core/plan.js';
5
5
  import { readyPlanTasks, sanitizePlanForExecution } from '../core/planPatch.js';
6
6
  import { createAssignmentManager } from '../orchestrator/assignmentManager.js';
7
7
  import { createAttemptManager } from '../orchestrator/attemptManager.js';
8
+ import { workspaceLockRegistry } from '../orchestrator/lockManager.js';
8
9
  import { createBudgetManager, BudgetExceededError } from '../orchestrator/budgetManager.js';
9
10
  import { createDispatcher } from '../orchestrator/dispatcher.js';
10
11
  import { approvalCovered, approvalRequestForTask } from '../orchestrator/approvalPolicy.js';
@@ -86,6 +87,43 @@ export function conversationSeed(session, currentInput, { limit = 12, maxChars =
86
87
  return seed;
87
88
  }
88
89
 
90
+ /**
91
+ * Decides whether the conversation memory must be compacted before a turn.
92
+ *
93
+ * `conversationSeed` keeps the last `limit` messages: anything older left
94
+ * Donna's context WITHOUT a summary and without a word — silence is the bug.
95
+ * So a compaction is due as soon as a message would fall out of that window,
96
+ * or when the seed alone takes more than a quarter of the model's input
97
+ * budget (a small-context profile). The last `keepLast` messages stay
98
+ * verbatim: a follow-up needs the previous answer word for word.
99
+ * Pure: returns the segment to summarize, or null.
100
+ */
101
+ export function conversationCompactionPlan(projection, { limit = 12, keepLast = 6, maxChars = 2000, budgetChars = null } = {}) {
102
+ const conversation = Array.isArray(projection?.conversation) ? projection.conversation : [];
103
+ const seedStart = Math.max(0, Number(projection?.conversationSeedStart) || 0);
104
+ const live = conversation.slice(seedStart)
105
+ .filter((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim());
106
+ const seedChars = live.slice(-limit).reduce((total, message) => total + Math.min(maxChars, String(message.content).length), 0)
107
+ + String(projection?.conversationSummary ?? '').length;
108
+ const overflow = live.length > limit;
109
+ const heavy = Number(budgetChars) > 0 && seedChars > Number(budgetChars) / 4;
110
+ if (!overflow && !heavy) return null;
111
+ // The cut never splits an exchange: it moves back to the user message that
112
+ // opened it. Observed: a cut between « comment utiliser <url> » and its
113
+ // answer sent the question into the summary, which dropped it — Donna then
114
+ // held an answer without its question and asked what to search for.
115
+ let cut = Math.max(seedStart, conversation.length - keepLast);
116
+ while (cut > seedStart && conversation[cut]?.role !== 'user') cut -= 1;
117
+ const segment = conversation.slice(seedStart, cut);
118
+ if (!segment.some((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim())) return null;
119
+ return { segment, keepLast: conversation.length - cut, reason: overflow ? 'window' : 'budget', previousSummary: projection?.conversationSummary ?? null };
120
+ }
121
+
122
+ /** What Donna is told after an automatic compaction — she says it, not the system. */
123
+ export function compactionNoteForDonna(count) {
124
+ return `[Memory note for Donna — not written by the user] The ${count} oldest messages of this conversation were just condensed automatically into the summary above, to stay within the model's context; the messages below are verbatim. After answering the user's next question, tell them so in one short sentence, in the reply language.`;
125
+ }
126
+
89
127
  export async function runRuntimeAgenticLoop(agent, session, initialInput, { signal, timeoutMs, maxTurns, runId, pollBusy, parallelHandoff = false, initialMessages = [] }) {
90
128
  return runAgenticLoop(agent, session, initialInput, {
91
129
  signal,
@@ -419,7 +457,15 @@ export async function runRuntimeParallelPlan(agent, session, input, {
419
457
  cappedByCeiling: concurrencyDetail.cappedByCeiling,
420
458
  };
421
459
  const active = new Map();
422
- const attempts = attemptManager ?? createAttemptManager();
460
+ // The workspace's lock registry, shared with every other run and direct
461
+ // action of this workspace; this run holds its locks under its own id.
462
+ const lockRegistry = workspaceLockRegistry(session);
463
+ const attempts = attemptManager ?? createAttemptManager({
464
+ locks: lockRegistry.locks,
465
+ owners: lockRegistry.owners,
466
+ owner: runId ?? 'run',
467
+ });
468
+ let announcedLockWait = '';
423
469
  const assigner = assignmentManager ?? createAssignmentManager({ session });
424
470
  const executor = dispatcher ?? createDispatcher({
425
471
  session,
@@ -648,6 +694,24 @@ export async function runRuntimeParallelPlan(agent, session, input, {
648
694
  // notamment derrière une barrière de groupe désormais terminale.
649
695
  continue;
650
696
  }
697
+ // A task whose only blocker is a lock held by ANOTHER run or a direct
698
+ // action waits for it: that holder will release it, so this is a
699
+ // queue, not a stall. Said once per change of holder, never silently.
700
+ const lockWaits = (session.headlessPlan ?? [])
701
+ .filter((step) => pendingSchedulerStatus(step.status))
702
+ .map((step) => ({ step, holders: attempts.foreignHolders?.(step) ?? [] }))
703
+ .filter((entry) => entry.holders.length > 0);
704
+ if (lockWaits.length > 0) {
705
+ const held = [...new Map(lockWaits.flatMap((entry) => entry.holders)
706
+ .map((holder) => [holder.lock, holder])).values()];
707
+ const summary = held.map((holder) => `${holder.lock} (held by ${holder.owner ?? 'another action'})`).join(', ');
708
+ if (summary !== announcedLockWait) {
709
+ announcedLockWait = summary;
710
+ emitRuntimeLog(session, `scheduler: waiting for workspace lock(s) — ${summary}; ${lockWaits.length} task(s) queued behind`);
711
+ }
712
+ await new Promise((resolveDelay) => setTimeout(resolveDelay, 500));
713
+ continue;
714
+ }
651
715
  // Any genuine approval-only block returned above. Remaining tasks are
652
716
  // unschedulable for another reason.
653
717
  const reason = 'no_ready_plan_task';