@dotdrelle/wiki-manager 0.15.94 → 0.15.97

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,6 +7,8 @@
7
7
  * which is why the shaping lives here and not in either renderer.
8
8
  */
9
9
 
10
+ import { TERMINAL_STATUS_SET } from '../orchestrator/taskStatuses.js';
11
+
10
12
  const SYMBOLS = {
11
13
  done: '✓',
12
14
  running: '●',
@@ -29,7 +31,7 @@ export function selectionKindLabel(selectionKind) {
29
31
  return SELECTION_KIND_LABELS[selectionKind] ?? selectionKind ?? null;
30
32
  }
31
33
 
32
- export const TERMINAL = new Set(['done', 'failed', 'cancelled', 'skipped']);
34
+ export const TERMINAL = TERMINAL_STATUS_SET;
33
35
 
34
36
  // Objectives are whole paragraphs; a chain view needs a line. Keep the first
35
37
  // sentence, drop the parameter block the compiler appends, and never cut a word
@@ -38,7 +38,15 @@ export async function runBoundedToolLoop({
38
38
  // livre le texte au fil de l'eau. Sans lui, la réponse finale n'apparaissait
39
39
  // qu'une fois complète — le tour paraissait figé pendant toute sa durée.
40
40
  const canStream = typeof onTextDelta === 'function' && typeof llm?.streamWithTools === 'function';
41
+ // The exact same tool + arguments called again is a loop, not progress: a
42
+ // model that keeps re-issuing `search("x")` will never finish, and burning
43
+ // the whole iteration cap on it only produced "could not finish". Track the
44
+ // signatures and stop as soon as a turn repeats one already executed.
45
+ const seen = new Set();
46
+ const signature = (call) => `${call?.function?.name ?? ''}\u0000${String(call?.function?.arguments ?? '')}`;
47
+ let iterations = 0;
41
48
  for (let i = 0; i < cap; i += 1) {
49
+ iterations = i + 1;
42
50
  onStep?.(i + 1, cap);
43
51
  let streamedText = false;
44
52
  const result = canStream
@@ -62,10 +70,12 @@ export async function runBoundedToolLoop({
62
70
  if (calls.length === 0) {
63
71
  return {
64
72
  content: result?.content ?? result?.message?.content ?? '',
65
- iterations: i + 1,
73
+ iterations,
66
74
  capped: false,
67
75
  };
68
76
  }
77
+ if (calls.every((call) => seen.has(signature(call)))) break;
78
+ for (const call of calls) seen.add(signature(call));
69
79
  convo.push(result.message ?? { role: 'assistant', content: result.content ?? '', tool_calls: calls });
70
80
  // Tool calls within one turn are independent: dispatch concurrently, then
71
81
  // replay results in the model's call order so the transcript stays stable.
@@ -77,5 +87,49 @@ export async function runBoundedToolLoop({
77
87
  convo.push({ role: 'tool', tool_call_id: outcome.tool_call_id, content: outcome.content });
78
88
  }
79
89
  }
80
- return { content: '', iterations: cap, capped: true };
90
+ // Cap reached or a loop detected: ask once more WITHOUT tools for the best
91
+ // answer the results gathered so far support. Returning '' here is what made
92
+ // a long search end in a dead-end instead of the partial answer it had
93
+ // already collected.
94
+ const content = await finalAnswerWithoutTools({ llm, system, convo, canStream, onTextDelta, onTextReset, signal });
95
+ return { content, iterations, capped: true };
96
+ }
97
+
98
+ async function finalAnswerWithoutTools({
99
+ llm,
100
+ system,
101
+ convo,
102
+ canStream,
103
+ onTextDelta,
104
+ onTextReset,
105
+ signal,
106
+ }) {
107
+ try {
108
+ if (canStream) {
109
+ let text = '';
110
+ const result = await llm.streamWithTools({
111
+ system,
112
+ tools: [],
113
+ messages: convo,
114
+ toolChoice: 'auto',
115
+ onTextDelta: (delta) => { text += delta; onTextDelta(delta); },
116
+ signal,
117
+ });
118
+ // A tool call despite the empty toolset is not an answer: drop whatever
119
+ // it streamed and let the caller fall back to its own message.
120
+ if (result?.tool_calls?.length) { onTextReset?.(); return ''; }
121
+ return String(result?.content ?? text ?? '').trim();
122
+ }
123
+ const result = await llm.completeWithTools({
124
+ system,
125
+ tools: [],
126
+ messages: convo,
127
+ toolChoice: 'auto',
128
+ signal,
129
+ });
130
+ if (result?.tool_calls?.length) return '';
131
+ return String(result?.content ?? result?.message?.content ?? '').trim();
132
+ } catch {
133
+ return '';
134
+ }
81
135
  }
@@ -61,16 +61,47 @@ test('runs concurrent tool calls and replays results in call order', async () =>
61
61
  assert.deepEqual(order, ['a', 'b']); // preserved model call order
62
62
  });
63
63
 
64
- test('reports capped when the model keeps calling tools past the cap', async () => {
64
+ test('stops on a repeated identical tool call instead of burning the cap', async () => {
65
65
  const llm = {
66
- async completeWithTools() {
66
+ async completeWithTools({ tools }) {
67
+ if (tools.length === 0) return { content: 'Synthèse des résultats.', tool_calls: [] };
67
68
  const calls = [toolCall('x', 's__status')];
68
69
  return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
69
70
  },
70
71
  };
71
- const out = await runBoundedToolLoop({ llm, tools: [], executeCall: async () => 'r', maxIterations: 3 });
72
+ const out = await runBoundedToolLoop({
73
+ llm,
74
+ tools: [{ function: { name: 's__status' } }],
75
+ executeCall: async () => 'r',
76
+ maxIterations: 8,
77
+ });
78
+ assert.equal(out.capped, true);
79
+ // The same call twice is a loop: it stopped well before the cap.
80
+ assert.ok(out.iterations < 8, `expected an early stop, got ${out.iterations}`);
81
+ // And the turn still answers from what it gathered instead of a dead-end.
82
+ assert.equal(out.content, 'Synthèse des résultats.');
83
+ });
84
+
85
+ test('answers from the gathered results when the cap is reached', async () => {
86
+ let round = 0;
87
+ const llm = {
88
+ async completeWithTools({ tools }) {
89
+ round += 1;
90
+ if (round <= 2 && tools.length > 0) {
91
+ const calls = [toolCall('x', 's__search', `{"q":"${round}"}`)];
92
+ return { message: { role: 'assistant', content: '', tool_calls: calls }, tool_calls: calls };
93
+ }
94
+ return { content: "Voici ce que j'ai trouvé.", tool_calls: [] };
95
+ },
96
+ };
97
+ const out = await runBoundedToolLoop({
98
+ llm,
99
+ tools: [{ function: { name: 's__search' } }],
100
+ executeCall: async () => 'r',
101
+ maxIterations: 2,
102
+ });
72
103
  assert.equal(out.capped, true);
73
- assert.equal(out.iterations, 3);
104
+ assert.equal(out.content, "Voici ce que j'ai trouvé.");
74
105
  });
75
106
 
76
107
  test('propagates an abort thrown by executeCall', async () => {
@@ -1,5 +1,6 @@
1
1
  import { createAgentEvent, dispatchAgentEvent, dispatchRuntimeLog } from '../core/agentEvents.js';
2
2
  import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
3
+ import { cloneJson } from '../core/json.js';
3
4
  import { assertContract } from '../contracts/schemas.js';
4
5
 
5
6
  const AVAILABLE = 'available';
@@ -320,6 +321,3 @@ function cloneAgent(agent) {
320
321
  };
321
322
  }
322
323
 
323
- function cloneJson(value) {
324
- return value == null ? value : JSON.parse(JSON.stringify(value));
325
- }
@@ -174,6 +174,3 @@ function taskId(task) {
174
174
  return String(task?.id ?? task?.step);
175
175
  }
176
176
 
177
- export function isTerminalTask(task) {
178
- return isTerminal(statusOf(task));
179
- }
@@ -1,4 +1,5 @@
1
1
  import { validateContract } from '../contracts/schemas.js';
2
+ import { cloneJson } from '../core/json.js';
2
3
 
3
4
  const SUPPORTED_CONTRACT_VERSIONS = new Set(['1']);
4
5
  const MUTATING_OPERATIONS = new Set([
@@ -529,6 +530,3 @@ function issue(code, message, details = {}) {
529
530
  return { code, message, details };
530
531
  }
531
532
 
532
- function cloneJson(value) {
533
- return value == null ? value : JSON.parse(JSON.stringify(value));
534
- }
@@ -43,20 +43,6 @@ import { assertContract } from '../../contracts/schemas.js';
43
43
 
44
44
  export const RUNTIME_PROTOCOL_VERSION = '1';
45
45
 
46
- export const RUNTIME_EVENT_TYPES = [
47
- 'run_created',
48
- 'run_started',
49
- 'agent_thinking',
50
- 'tool_started',
51
- 'tool_finished',
52
- 'subagent_started',
53
- 'subagent_finished',
54
- 'message',
55
- 'approval_required',
56
- 'run_completed',
57
- 'run_failed',
58
- 'run_cancelled',
59
- ];
60
46
 
61
47
  export class RuntimeProviderUnavailableError extends Error {
62
48
  constructor(runtime, reason) {
@@ -28,6 +28,14 @@ export const PENDING_STATUSES_LIST = Object.freeze(['pending', 'pending_approval
28
28
  /** En cours : un agent y travaille en ce moment. */
29
29
  export const ACTIVE_STATUSES = Object.freeze(['running', 'in_progress', 'started', 'starting']);
30
30
 
31
+ /**
32
+ * Terminal, réduit à ses quatre formes canoniques (les alias sont normalisés
33
+ * avant comparaison). Les modules qui recopiaient `['done','failed',
34
+ * 'cancelled','skipped']` dans un `Set` importent celui-ci à la place.
35
+ */
36
+ export const TERMINAL_STATUSES = Object.freeze(['done', 'failed', 'cancelled', 'skipped']);
37
+ export const TERMINAL_STATUS_SET = new Set(TERMINAL_STATUSES);
38
+
31
39
  const ALIASES = new Map([
32
40
  ...SUCCESS_STATUSES.map((status) => [status, 'done']),
33
41
  ...FAILURE_STATUSES.map((status) => [status, 'failed']),
@@ -238,19 +238,6 @@ export async function postRuntimeShutdown({
238
238
  return response.json();
239
239
  }
240
240
 
241
- export async function postRuntimeResume({
242
- url = runtimeUrlFromEnv(),
243
- token = runtimeToken(),
244
- workspace = null,
245
- } = {}) {
246
- const response = await fetch(runtimeEndpoint(url, '/resume', workspace), {
247
- method: 'POST',
248
- headers: runtimeHeaders(token),
249
- });
250
- if (!response.ok) throw new Error(`Runtime resume failed: HTTP ${response.status}`);
251
- return response.json();
252
- }
253
-
254
241
  export async function postRuntimeApprove({
255
242
  url = runtimeUrlFromEnv(),
256
243
  token = runtimeToken(),
@@ -331,6 +318,3 @@ export async function* streamRuntimeEvents({
331
318
  }
332
319
  }
333
320
 
334
- export function runtimeFetchOptions(token = runtimeToken()) {
335
- return { headers: runtimeHeaders(token) };
336
- }
@@ -1,10 +1,13 @@
1
1
  /**
2
2
  * @statuses-vocabulary
3
3
  * Control items are queued run requests, not orchestrator tasks. Their
4
- * terminal vocabulary deliberately includes chain-level `skipped` and is
5
- * projected by core/agentEvents.js rather than taskStatuses.js.
4
+ * terminal vocabulary is the same four canonical statuses as a task's
5
+ * (`TERMINAL_STATUS_SET`), including chain-level `skipped`; the projection is
6
+ * still done by core/agentEvents.js, this only shares the vocabulary.
6
7
  */
7
- const TERMINAL = new Set(['done', 'failed', 'cancelled', 'skipped']);
8
+ import { TERMINAL_STATUS_SET } from '../orchestrator/taskStatuses.js';
9
+
10
+ const TERMINAL = TERMINAL_STATUS_SET;
8
11
 
9
12
  export function reconcileControlQueue(context, { startItem, skipItem } = {}) {
10
13
  if (!context?.session || context.running || context.controlDrainActive) return false;
@@ -0,0 +1,53 @@
1
+ /*
2
+ * Coalesces streaming text fragments before they are persisted and pushed.
3
+ *
4
+ * The runtime persisted one SQLite row (plus one SSE write) per streamed token.
5
+ * When a tool pulled a lot of content into the thread, the answer narrating it
6
+ * grew long and those synchronous writes stalled the event loop: both chats
7
+ * (serve and ShellUI) froze while the answer was still being produced. Buffering
8
+ * the fragments and flushing them at a bounded rate turns thousands of writes
9
+ * into a handful without changing what the reader sees.
10
+ *
11
+ * Ordering matters: `flush()` must be called before any non-delta event, or a
12
+ * final message could overtake the fragments that precede it. `reset()` drops
13
+ * buffered text that turned out to be provisional narration (a tool-call
14
+ * iteration), matching `assistant_delta_reset`.
15
+ */
16
+ export function createDeltaCoalescer(flush, { intervalMs = 80 } = {}) {
17
+ if (typeof flush !== 'function') throw new TypeError('createDeltaCoalescer requires a flush callback.');
18
+ const delay = Math.max(1, Math.floor(intervalMs) || 80);
19
+ let buffer = '';
20
+ let timer = null;
21
+ const emit = () => {
22
+ if (timer) {
23
+ clearTimeout(timer);
24
+ timer = null;
25
+ }
26
+ if (!buffer) return;
27
+ const delta = buffer;
28
+ buffer = '';
29
+ flush(delta);
30
+ };
31
+ return {
32
+ push(delta) {
33
+ const text = String(delta ?? '');
34
+ if (!text) return;
35
+ buffer += text;
36
+ if (!timer) timer = setTimeout(emit, delay);
37
+ },
38
+ flush: emit,
39
+ reset() {
40
+ buffer = '';
41
+ if (timer) {
42
+ clearTimeout(timer);
43
+ timer = null;
44
+ }
45
+ },
46
+ dispose() {
47
+ if (timer) {
48
+ clearTimeout(timer);
49
+ timer = null;
50
+ }
51
+ },
52
+ };
53
+ }
@@ -0,0 +1,56 @@
1
+ import assert from 'node:assert/strict';
2
+ import test from 'node:test';
3
+ import { createDeltaCoalescer } from './deltaCoalescer.js';
4
+
5
+ const tick = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
6
+
7
+ test('coalesces fragments pushed within the interval into one flush', async () => {
8
+ const flushed = [];
9
+ const coalescer = createDeltaCoalescer((delta) => flushed.push(delta), { intervalMs: 20 });
10
+ coalescer.push('Bon');
11
+ coalescer.push('jour ');
12
+ coalescer.push('le monde.');
13
+ assert.deepEqual(flushed, [], 'rien ne doit partir avant l\'intervalle');
14
+ await tick(35);
15
+ assert.deepEqual(flushed, ['Bonjour le monde.']);
16
+ });
17
+
18
+ test('flush() emits the buffered fragments immediately, in order', () => {
19
+ const flushed = [];
20
+ const coalescer = createDeltaCoalescer((delta) => flushed.push(delta), { intervalMs: 1000 });
21
+ coalescer.push('a');
22
+ coalescer.push('b');
23
+ coalescer.flush();
24
+ assert.deepEqual(flushed, ['ab']);
25
+ // Nothing left to flush twice.
26
+ coalescer.flush();
27
+ assert.deepEqual(flushed, ['ab']);
28
+ });
29
+
30
+ test('reset() drops buffered provisional narration', () => {
31
+ const flushed = [];
32
+ const coalescer = createDeltaCoalescer((delta) => flushed.push(delta), { intervalMs: 1000 });
33
+ coalescer.push('je vais regarder…');
34
+ coalescer.reset();
35
+ coalescer.flush();
36
+ assert.deepEqual(flushed, []);
37
+ });
38
+
39
+ test('a single flush covers fragments that arrive after a first flush', async () => {
40
+ const flushed = [];
41
+ const coalescer = createDeltaCoalescer((delta) => flushed.push(delta), { intervalMs: 20 });
42
+ coalescer.push('un ');
43
+ await tick(30);
44
+ coalescer.push('deux');
45
+ await tick(30);
46
+ assert.deepEqual(flushed, ['un ', 'deux']);
47
+ });
48
+
49
+ test('dispose() stops the pending timer without emitting', async () => {
50
+ const flushed = [];
51
+ const coalescer = createDeltaCoalescer((delta) => flushed.push(delta), { intervalMs: 20 });
52
+ coalescer.push('perdu');
53
+ coalescer.dispose();
54
+ await tick(40);
55
+ assert.deepEqual(flushed, []);
56
+ });
@@ -127,6 +127,3 @@ export function loginPageHtml({ enrolled = false, secret = null, uri = null, err
127
127
  </html>`;
128
128
  }
129
129
 
130
- export function loginSuccessHtml(sessionExpiresAt) {
131
- return loginPageHtml({ enrolled: true, sessionExpiresAt });
132
- }
@@ -1,4 +1,4 @@
1
- import { chmodSync, existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
1
+ import { chmodSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
2
2
  import { join, resolve } from 'node:path';
3
3
  import { randomBytes } from 'node:crypto';
4
4
  import { defaultRuntimeStateDir } from '../core/env.js';
@@ -221,6 +221,3 @@ export function pruneLoginAttempts() {
221
221
  }
222
222
  }
223
223
 
224
- export function sessionExists() {
225
- return existsSync(sessionPath());
226
- }
@@ -303,12 +303,13 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
303
303
  // instead of the client streaming a per-job line for every task. Uses the
304
304
  // workspace LLM to phrase it, degrading to a plain templated fact line if the
305
305
  // LLM is unavailable or errors — the run must never block on this summary.
306
- async function announceRunOutcome(session, { runId, ok, signal = null } = {}) {
306
+ export async function announceRunOutcome(session, { runId, ok, signal = null } = {}) {
307
307
  const plan = Array.isArray(session.headlessPlan) ? session.headlessPlan : [];
308
308
  if (plan.length === 0) return;
309
309
  let failed = 0;
310
310
  let cancelled = 0;
311
311
  let completed = 0;
312
+ let pending = 0;
312
313
  let firstError = null;
313
314
  for (const step of plan) {
314
315
  const status = String(step?.status ?? '').toLowerCase();
@@ -322,12 +323,24 @@ async function announceRunOutcome(session, { runId, ok, signal = null } = {}) {
322
323
  cancelled += 1;
323
324
  } else if (isSuccessful(status)) {
324
325
  completed += 1;
326
+ } else if (isPending(status) || !isTerminal(status)) {
327
+ // pending_approval, waiting_approval, running, unknown: the work has NOT
328
+ // happened. Counting these as neither success nor failure is what made a
329
+ // run that had only *planned* its mutations announce a success (LLM
330
+ // rephrasing "0/N réussie" into "le livrable a bien été publié") before
331
+ // the approval that would actually run it.
332
+ pending += 1;
325
333
  }
326
334
  }
327
335
  const total = plan.length;
328
- const factLine = ok && failed === 0
336
+ const finished = ok && failed === 0 && cancelled === 0 && pending === 0 && completed === total;
337
+ const factLine = finished
329
338
  ? `Plan terminé avec succès — ${completed}/${total} tâche(s) réussie(s).`
330
- : `Plan terminé en erreur — ${completed}/${total} tâche(s) réussie(s), ${failed} en erreur${cancelled ? `, ${cancelled} annulée(s)` : ''}.${firstError ? ` Première erreur : ${firstError}.` : ''}`;
339
+ : `Plan non terminé — ${completed}/${total} tâche(s) réussie(s)` +
340
+ `${pending ? `, ${pending} en attente (approbation ou exécution)` : ''}` +
341
+ `${failed ? `, ${failed} en erreur` : ''}` +
342
+ `${cancelled ? `, ${cancelled} annulée(s)` : ''}.` +
343
+ `${firstError ? ` Première erreur : ${firstError}.` : ''}`;
331
344
  let content = factLine;
332
345
  const llm = session.llm;
333
346
  if (llm && typeof llm.completeWithTools === 'function') {
@@ -337,6 +350,7 @@ async function announceRunOutcome(session, { runId, ok, signal = null } = {}) {
337
350
  'You are Donna, an orchestration assistant reporting a run result to the user.',
338
351
  'Rephrase the outcome facts in ONE short, natural sentence, in the same language as the facts.',
339
352
  'No lists, no headers, no raw job ids — just a concise human summary.',
353
+ 'If the facts say the plan is NOT finished, say so plainly and name what is still pending or failed: never claim the work was completed, published or successful.',
340
354
  ].join('\n'),
341
355
  tools: [],
342
356
  messages: [{ role: 'user', content: `Run outcome facts:\n${factLine}` }],
@@ -4,7 +4,7 @@ import { createAgentEvent, dispatchAgentEvent, reduceAgentEvents } from '../core
4
4
  import { tasksAwaitingApproval } from '../orchestrator/dependencyResolver.js';
5
5
  import { isTerminal } from '../orchestrator/taskStatuses.js';
6
6
  import { readyPlanTasks } from '../core/planPatch.js';
7
- import { skipImpossibleTasks, structuredPlanEvaluation, ensurePlanProjection, evaluateRuntimeRun, finishRuntimeRun, materializeTaskInputs, replanRuntimeRun, runRuntimeAgenticWorkflow, runRuntimeParallelPlan, shouldUseParallelScheduler } from './runner.js';
7
+ import { skipImpossibleTasks, structuredPlanEvaluation, ensurePlanProjection, evaluateRuntimeRun, finishRuntimeRun, materializeTaskInputs, replanRuntimeRun, runRuntimeAgenticWorkflow, runRuntimeParallelPlan, shouldUseParallelScheduler, announceRunOutcome } from './runner.js';
8
8
 
9
9
  test('ensurePlanProjection re-projects when the chained plan changes shape (step 2)', () => {
10
10
  const session = { agentEvents: [], agentProjection: null };
@@ -1288,3 +1288,34 @@ test('rejouer les événements redonne exactement les mêmes statuts', () => {
1288
1288
  );
1289
1289
  assert.deepEqual(replayed.plan.map((step) => step.status), ['failed', 'skipped']);
1290
1290
  });
1291
+
1292
+ test('announceRunOutcome never calls a plan with pending tasks a success', async () => {
1293
+ // The outcome summary counted only failed/cancelled/successful, so a run
1294
+ // whose single mutating task was still `pending_approval` produced
1295
+ // "Plan terminé avec succès — 0/1 réussie", which the model rephrased into
1296
+ // "le livrable a bien été publié" — before the approval that would run it.
1297
+ const session = {
1298
+ agentEvents: [],
1299
+ agentProjection: null,
1300
+ headlessPlan: [{ id: 'a', description: 'Build TechSections', status: 'pending_approval' }],
1301
+ };
1302
+ await announceRunOutcome(session, { runId: 'run-1', ok: true });
1303
+ const message = session.agentEvents.find((event) => event.type === 'assistant_message');
1304
+ assert.match(message.payload.content, /non terminé/i);
1305
+ assert.match(message.payload.content, /en attente/);
1306
+ assert.doesNotMatch(message.payload.content, /succès/i);
1307
+ });
1308
+
1309
+ test('announceRunOutcome reports success only when every task finished', async () => {
1310
+ const session = {
1311
+ agentEvents: [],
1312
+ agentProjection: null,
1313
+ headlessPlan: [
1314
+ { id: 'a', description: 'Build TechSections', status: 'done' },
1315
+ { id: 'b', description: 'Export', status: 'success' },
1316
+ ],
1317
+ };
1318
+ await announceRunOutcome(session, { runId: 'run-2', ok: true });
1319
+ const message = session.agentEvents.find((event) => event.type === 'assistant_message');
1320
+ assert.match(message.payload.content, /succès/);
1321
+ });