@dotdrelle/wiki-manager 0.15.93 → 0.15.96

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +39 -28
  2. package/docker-compose.yml +1 -1
  3. package/package.json +2 -2
  4. package/src/agent/graph.js +78 -26
  5. package/src/agent/graph.test.js +28 -2
  6. package/src/cli/wiki-manager.js +138 -18
  7. package/src/commands/slash.js +38 -4
  8. package/src/commands/slash.test.js +11 -1
  9. package/src/core/agentEvents.js +24 -3
  10. package/src/core/agentEvents.test.js +37 -0
  11. package/src/core/buildInfo.json +2 -2
  12. package/src/core/dockerCompose.test.js +14 -0
  13. package/src/core/googleGrants.js +0 -3
  14. package/src/core/json.js +9 -0
  15. package/src/core/mcp.js +1 -1
  16. package/src/core/plan.js +0 -4
  17. package/src/core/progressNotes.js +0 -4
  18. package/src/core/skillChainView.js +3 -1
  19. package/src/core/skillCompiler.test.js +21 -1
  20. package/src/core/toolLoop.js +56 -2
  21. package/src/core/toolLoop.test.js +35 -4
  22. package/src/orchestrator/agentRegistry.js +1 -3
  23. package/src/orchestrator/dependencyResolver.js +0 -3
  24. package/src/orchestrator/objectiveResolver.test.js +27 -0
  25. package/src/orchestrator/planValidator.js +1 -3
  26. package/src/orchestrator/providers/runtimeProvider.js +0 -14
  27. package/src/orchestrator/taskStatuses.js +8 -0
  28. package/src/runtime/client.js +0 -16
  29. package/src/runtime/controlDrain.js +6 -3
  30. package/src/runtime/deltaCoalescer.js +53 -0
  31. package/src/runtime/deltaCoalescer.test.js +56 -0
  32. package/src/runtime/loginPage.js +129 -0
  33. package/src/runtime/loginRoutes.test.js +131 -0
  34. package/src/runtime/loginSession.js +223 -0
  35. package/src/runtime/loginSession.test.js +143 -0
  36. package/src/runtime/qrCode.js +15 -0
  37. package/src/runtime/runner.js +29 -3
  38. package/src/runtime/runner.test.js +90 -1
  39. package/src/runtime/server.js +240 -1
  40. package/src/runtime/server.test.js +87 -0
  41. package/src/runtime/skillRun.js +1 -1
  42. package/src/runtime/skillRun.test.js +3 -0
  43. package/src/runtime/totp.js +87 -0
  44. package/src/runtime/totp.test.js +80 -0
  45. package/src/runtime/totpLogin.js +123 -0
  46. package/src/runtime/vendor/qrcode.cjs +2297 -0
  47. package/src/shell/repl.js +7 -3
  48. package/src/orchestrator/.fuse_hidden0000001c00000001 +0 -316
@@ -0,0 +1,143 @@
1
+ import test from 'node:test';
2
+ import assert from 'node:assert/strict';
3
+ import { mkdtempSync, rmSync } from 'node:fs';
4
+ import { tmpdir } from 'node:os';
5
+ import { join } from 'node:path';
6
+ import { totpCode } from './totp.js';
7
+ import {
8
+ enrollment,
9
+ isTotpEnabled,
10
+ issueSessionWithTotp,
11
+ loginAttemptAllowed,
12
+ loginStatus,
13
+ resetTotpEnrollment,
14
+ revokeSession,
15
+ SESSION_TTL_MS,
16
+ verifySessionToken,
17
+ } from './loginSession.js';
18
+
19
+ // Each test gets a fresh in-memory enrollment and a fresh state dir.
20
+ function freshStateDir() {
21
+ const dir = mkdtempSync(join(tmpdir(), 'login-session-'));
22
+ process.env.WIKI_MANAGER_STATE_DIR = dir;
23
+ return dir;
24
+ }
25
+
26
+ test.afterEach(() => {
27
+ const dir = process.env.WIKI_MANAGER_STATE_DIR;
28
+ if (dir && dir.startsWith(tmpdir())) rmSync(dir, { recursive: true, force: true });
29
+ delete process.env.WIKI_MANAGER_STATE_DIR;
30
+ });
31
+
32
+ test('enrollment publishes a secret and URI, then the first code persists it', () => {
33
+ freshStateDir();
34
+ const before = enrollment();
35
+ assert.ok(before.secret.match(/^[A-Z2-7]{32}$/));
36
+ assert.ok(before.uri.startsWith('otpauth://totp/'));
37
+ assert.equal(loginStatus().enrolled, false);
38
+
39
+ // The SAME pending secret is shown again (the page reloads, not re-rolls).
40
+ assert.equal(enrollment().secret, before.secret);
41
+
42
+ const result = issueSessionWithTotp(totpCode(before.secret));
43
+ assert.equal(result.ok, true);
44
+ assert.equal(result.enrolled, true);
45
+ assert.ok(result.token.length >= 32);
46
+ assert.equal(loginStatus().enrolled, true);
47
+ assert.equal(enrollment(), null);
48
+ });
49
+
50
+ test('a wrong code never enrolls and never issues a session', () => {
51
+ freshStateDir();
52
+ const pending = enrollment();
53
+ const wrong = totpCode(pending.secret) === '000000' ? '111111' : '000000';
54
+ const result = issueSessionWithTotp(wrong);
55
+ assert.equal(result.ok, false);
56
+ assert.equal(result.error, 'invalid_code');
57
+ assert.equal(loginStatus().enrolled, false);
58
+ });
59
+
60
+ test('enrollment is refused from a non-loopback address', () => {
61
+ freshStateDir();
62
+ const pending = enrollment();
63
+ const result = issueSessionWithTotp(totpCode(pending.secret), { remoteAddress: '192.168.1.10' });
64
+ assert.equal(result.ok, false);
65
+ assert.equal(result.error, 'enrollment_requires_loopback');
66
+ // From loopback the same code succeeds.
67
+ const ok = issueSessionWithTotp(totpCode(pending.secret), { remoteAddress: '127.0.0.1' });
68
+ assert.equal(ok.ok, true);
69
+ });
70
+
71
+ test('a verified session slides instead of expiring', () => {
72
+ freshStateDir();
73
+ const pending = enrollment();
74
+ const issued = issueSessionWithTotp(totpCode(pending.secret));
75
+ assert.ok(issued.token);
76
+
77
+ // Right after issue: valid.
78
+ assert.equal(verifySessionToken(issued.token).ok, true);
79
+ // Near the end of the TTL: still valid, and the slide pushes expiry forward.
80
+ const late = issued.expiresAt - 1;
81
+ const check = verifySessionToken(issued.token, { timestamp: late });
82
+ assert.equal(check.ok, true);
83
+ assert.ok(check.expiresAt > issued.expiresAt);
84
+ // Past the NEW expiry the token is refused.
85
+ const past = check.expiresAt + 1;
86
+ assert.equal(verifySessionToken(issued.token, { timestamp: past, renew: false }).ok, false);
87
+ });
88
+
89
+ test('an unknown token is refused and revocation invalidates the real one', () => {
90
+ freshStateDir();
91
+ const pending = enrollment();
92
+ const issued = issueSessionWithTotp(totpCode(pending.secret));
93
+
94
+ assert.equal(verifySessionToken('nope').ok, false);
95
+ assert.equal(revokeSession(issued.token), true);
96
+ assert.equal(verifySessionToken(issued.token).ok, false);
97
+ assert.equal(loginStatus().sessionActive, false);
98
+ // Revoking twice is a no-op, not an error.
99
+ assert.equal(revokeSession(issued.token), false);
100
+ });
101
+
102
+ test('resetTotpEnrollment wipes the secret and the session for re-enrollment', () => {
103
+ freshStateDir();
104
+ const pending = enrollment();
105
+ const issued = issueSessionWithTotp(totpCode(pending.secret));
106
+ assert.equal(loginStatus().enrolled, true);
107
+ assert.equal(loginStatus().sessionActive, true);
108
+
109
+ resetTotpEnrollment();
110
+
111
+ const status = loginStatus();
112
+ assert.equal(status.enrolled, false);
113
+ assert.equal(status.sessionActive, false);
114
+ assert.equal(verifySessionToken(issued.token).ok, false);
115
+ // A fresh enrollment secret is generated for the next page.
116
+ const fresh = enrollment();
117
+ assert.ok(fresh.secret);
118
+ assert.notEqual(fresh.secret, pending.secret);
119
+ });
120
+
121
+ test('login attempts are rate-limited per address', () => {
122
+ for (let index = 0; index < 10; index++) {
123
+ const allowed = loginAttemptAllowed('127.0.0.1');
124
+ assert.equal(allowed.ok, true, `attempt ${index}`);
125
+ }
126
+ const refused = loginAttemptAllowed('127.0.0.1');
127
+ assert.equal(refused.ok, false);
128
+ assert.ok(refused.retryAfterSeconds > 0);
129
+ // Another address is unaffected.
130
+ assert.equal(loginAttemptAllowed('10.0.0.2').ok, true);
131
+ });
132
+
133
+ test('TOTP can be disabled from the environment', () => {
134
+ freshStateDir();
135
+ process.env.WIKI_MANAGER_TOTP = 'off';
136
+ assert.equal(isTotpEnabled(), false);
137
+ assert.equal(enrollment(), null);
138
+ delete process.env.WIKI_MANAGER_TOTP;
139
+ });
140
+
141
+ test('the default session TTL is 12 hours', () => {
142
+ assert.equal(SESSION_TTL_MS, 12 * 60 * 60 * 1000);
143
+ });
@@ -0,0 +1,15 @@
1
+ import qrcodeGenerator from './vendor/qrcode.cjs';
2
+
3
+ /*
4
+ Thin wrapper over the vendored MIT qrcode-generator (Kazuhiko Arase, 2009 —
5
+ see vendor/qrcode.cjs header). The login page renders the enrollment QR as a
6
+ responsive SVG; the otpauth:// URI is ASCII-only, so the generator's default
7
+ latin-1 byte conversion is sufficient.
8
+ */
9
+
10
+ export function qrSvg(content, { cellSize = 4, margin = 8 } = {}) {
11
+ const qr = qrcodeGenerator(0, 'M');
12
+ qr.addData(String(content), 'Byte');
13
+ qr.make();
14
+ return qr.createSvgTag({ cellSize, margin, scalable: true, alt: { text: 'QR code' } });
15
+ }
@@ -63,7 +63,12 @@ export function conversationSeed(session, currentInput, { limit = 12, maxChars =
63
63
  const conversation = Array.isArray(session.agentProjection?.conversation)
64
64
  ? session.agentProjection.conversation
65
65
  : [];
66
+ // Everything before the last compact (conversation_reset) is forgotten for
67
+ // grounding, while it stays visible in the displayed thread. The boundary is
68
+ // an index into the same array, so no message is actually removed.
69
+ const seedStart = Math.max(0, Number(session.agentProjection?.conversationSeedStart) || 0);
66
70
  const seed = conversation
71
+ .slice(seedStart)
67
72
  .filter((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim())
68
73
  .slice(-limit)
69
74
  .map((message) => ({ role: message.role, content: String(message.content).slice(0, maxChars) }));
@@ -71,6 +76,13 @@ export function conversationSeed(session, currentInput, { limit = 12, maxChars =
71
76
  // drop it from the seed to avoid sending it twice.
72
77
  const last = seed.at(-1);
73
78
  if (last && last.role === 'user' && last.content === String(currentInput ?? '').slice(0, maxChars)) seed.pop();
79
+ // A compact does not just cut the older turns — it replaces them with a
80
+ // short summary (see /conversation/compact), so what was agreed there is
81
+ // not lost to the grounding window entirely, only condensed.
82
+ const summary = String(session.agentProjection?.conversationSummary ?? '').trim();
83
+ if (summary) {
84
+ seed.unshift({ role: 'user', content: `[Summary of earlier conversation, compacted]\n${summary.slice(0, maxChars)}` });
85
+ }
74
86
  return seed;
75
87
  }
76
88
 
@@ -291,12 +303,13 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
291
303
  // instead of the client streaming a per-job line for every task. Uses the
292
304
  // workspace LLM to phrase it, degrading to a plain templated fact line if the
293
305
  // LLM is unavailable or errors — the run must never block on this summary.
294
- async function announceRunOutcome(session, { runId, ok, signal = null } = {}) {
306
+ export async function announceRunOutcome(session, { runId, ok, signal = null } = {}) {
295
307
  const plan = Array.isArray(session.headlessPlan) ? session.headlessPlan : [];
296
308
  if (plan.length === 0) return;
297
309
  let failed = 0;
298
310
  let cancelled = 0;
299
311
  let completed = 0;
312
+ let pending = 0;
300
313
  let firstError = null;
301
314
  for (const step of plan) {
302
315
  const status = String(step?.status ?? '').toLowerCase();
@@ -310,12 +323,24 @@ async function announceRunOutcome(session, { runId, ok, signal = null } = {}) {
310
323
  cancelled += 1;
311
324
  } else if (isSuccessful(status)) {
312
325
  completed += 1;
326
+ } else if (isPending(status) || !isTerminal(status)) {
327
+ // pending_approval, waiting_approval, running, unknown: the work has NOT
328
+ // happened. Counting these as neither success nor failure is what made a
329
+ // run that had only *planned* its mutations announce a success (LLM
330
+ // rephrasing "0/N réussie" into "le livrable a bien été publié") before
331
+ // the approval that would actually run it.
332
+ pending += 1;
313
333
  }
314
334
  }
315
335
  const total = plan.length;
316
- const factLine = ok && failed === 0
336
+ const finished = ok && failed === 0 && cancelled === 0 && pending === 0 && completed === total;
337
+ const factLine = finished
317
338
  ? `Plan terminé avec succès — ${completed}/${total} tâche(s) réussie(s).`
318
- : `Plan terminé en erreur — ${completed}/${total} tâche(s) réussie(s), ${failed} en erreur${cancelled ? `, ${cancelled} annulée(s)` : ''}.${firstError ? ` Première erreur : ${firstError}.` : ''}`;
339
+ : `Plan non terminé — ${completed}/${total} tâche(s) réussie(s)` +
340
+ `${pending ? `, ${pending} en attente (approbation ou exécution)` : ''}` +
341
+ `${failed ? `, ${failed} en erreur` : ''}` +
342
+ `${cancelled ? `, ${cancelled} annulée(s)` : ''}.` +
343
+ `${firstError ? ` Première erreur : ${firstError}.` : ''}`;
319
344
  let content = factLine;
320
345
  const llm = session.llm;
321
346
  if (llm && typeof llm.completeWithTools === 'function') {
@@ -325,6 +350,7 @@ async function announceRunOutcome(session, { runId, ok, signal = null } = {}) {
325
350
  'You are Donna, an orchestration assistant reporting a run result to the user.',
326
351
  'Rephrase the outcome facts in ONE short, natural sentence, in the same language as the facts.',
327
352
  'No lists, no headers, no raw job ids — just a concise human summary.',
353
+ 'If the facts say the plan is NOT finished, say so plainly and name what is still pending or failed: never claim the work was completed, published or successful.',
328
354
  ].join('\n'),
329
355
  tools: [],
330
356
  messages: [{ role: 'user', content: `Run outcome facts:\n${factLine}` }],
@@ -4,7 +4,7 @@ import { createAgentEvent, dispatchAgentEvent, reduceAgentEvents } from '../core
4
4
  import { tasksAwaitingApproval } from '../orchestrator/dependencyResolver.js';
5
5
  import { isTerminal } from '../orchestrator/taskStatuses.js';
6
6
  import { readyPlanTasks } from '../core/planPatch.js';
7
- import { skipImpossibleTasks, structuredPlanEvaluation, ensurePlanProjection, evaluateRuntimeRun, finishRuntimeRun, materializeTaskInputs, replanRuntimeRun, runRuntimeAgenticWorkflow, runRuntimeParallelPlan, shouldUseParallelScheduler } from './runner.js';
7
+ import { skipImpossibleTasks, structuredPlanEvaluation, ensurePlanProjection, evaluateRuntimeRun, finishRuntimeRun, materializeTaskInputs, replanRuntimeRun, runRuntimeAgenticWorkflow, runRuntimeParallelPlan, shouldUseParallelScheduler, announceRunOutcome } from './runner.js';
8
8
 
9
9
  test('ensurePlanProjection re-projects when the chained plan changes shape (step 2)', () => {
10
10
  const session = { agentEvents: [], agentProjection: null };
@@ -1103,6 +1103,64 @@ test('runtime runs are seeded with the chat that preceded them', async () => {
1103
1103
  assert.equal(clipped[0].content.length, 2000);
1104
1104
  });
1105
1105
 
1106
+ test('a compact boundary hides the prior exchanges from the seed but keeps them in the conversation', async () => {
1107
+ const { conversationSeed } = await import('./runner.js');
1108
+ const session = {
1109
+ agentProjection: {
1110
+ conversation: [
1111
+ { role: 'user', content: 'avant 1' },
1112
+ { role: 'assistant', content: 'avant 2' },
1113
+ { role: 'user', content: 'après 1' },
1114
+ { role: 'assistant', content: 'après 2' },
1115
+ ],
1116
+ conversationSeedStart: 2,
1117
+ },
1118
+ };
1119
+ const seed = conversationSeed(session, 'nouvelle question');
1120
+ assert.deepEqual(seed, [
1121
+ { role: 'user', content: 'après 1' },
1122
+ { role: 'assistant', content: 'après 2' },
1123
+ ]);
1124
+ // The pre-compact exchange is gone from the grounding…
1125
+ assert.equal(seed.some((message) => /avant/.test(message.content)), false);
1126
+ // …but it stays in the displayed conversation.
1127
+ assert.equal(session.agentProjection.conversation.length, 4);
1128
+ });
1129
+
1130
+ test('a stored compact summary is prepended to the seed, ahead of the recent exchanges', async () => {
1131
+ const { conversationSeed } = await import('./runner.js');
1132
+ const session = {
1133
+ agentProjection: {
1134
+ conversation: [
1135
+ { role: 'user', content: 'après 1' },
1136
+ { role: 'assistant', content: 'après 2' },
1137
+ ],
1138
+ conversationSeedStart: 0,
1139
+ conversationSummary: 'Résumé condensé des échanges précédents.',
1140
+ },
1141
+ };
1142
+ const seed = conversationSeed(session, 'nouvelle question');
1143
+ assert.equal(seed.length, 3);
1144
+ assert.equal(seed[0].role, 'user');
1145
+ assert.match(seed[0].content, /Résumé condensé des échanges précédents\./);
1146
+ assert.deepEqual(seed.slice(1), [
1147
+ { role: 'user', content: 'après 1' },
1148
+ { role: 'assistant', content: 'après 2' },
1149
+ ]);
1150
+ });
1151
+
1152
+ test('no stored summary means no synthetic entry is added to the seed', async () => {
1153
+ const { conversationSeed } = await import('./runner.js');
1154
+ const session = {
1155
+ agentProjection: {
1156
+ conversation: [{ role: 'user', content: 'salut' }],
1157
+ conversationSeedStart: 0,
1158
+ },
1159
+ };
1160
+ const seed = conversationSeed(session, 'autre question');
1161
+ assert.deepEqual(seed, [{ role: 'user', content: 'salut' }]);
1162
+ });
1163
+
1106
1164
  test('skipImpossibleTasks propage un échec jusqu’au point fixe', () => {
1107
1165
  /*
1108
1166
  Marquer les seules tâches directement bloquées laissait un résidu : A en
@@ -1230,3 +1288,34 @@ test('rejouer les événements redonne exactement les mêmes statuts', () => {
1230
1288
  );
1231
1289
  assert.deepEqual(replayed.plan.map((step) => step.status), ['failed', 'skipped']);
1232
1290
  });
1291
+
1292
+ test('announceRunOutcome never calls a plan with pending tasks a success', async () => {
1293
+ // The outcome summary counted only failed/cancelled/successful, so a run
1294
+ // whose single mutating task was still `pending_approval` produced
1295
+ // "Plan terminé avec succès — 0/1 réussie", which the model rephrased into
1296
+ // "le livrable a bien été publié" — before the approval that would run it.
1297
+ const session = {
1298
+ agentEvents: [],
1299
+ agentProjection: null,
1300
+ headlessPlan: [{ id: 'a', description: 'Build TechSections', status: 'pending_approval' }],
1301
+ };
1302
+ await announceRunOutcome(session, { runId: 'run-1', ok: true });
1303
+ const message = session.agentEvents.find((event) => event.type === 'assistant_message');
1304
+ assert.match(message.payload.content, /non terminé/i);
1305
+ assert.match(message.payload.content, /en attente/);
1306
+ assert.doesNotMatch(message.payload.content, /succès/i);
1307
+ });
1308
+
1309
+ test('announceRunOutcome reports success only when every task finished', async () => {
1310
+ const session = {
1311
+ agentEvents: [],
1312
+ agentProjection: null,
1313
+ headlessPlan: [
1314
+ { id: 'a', description: 'Build TechSections', status: 'done' },
1315
+ { id: 'b', description: 'Export', status: 'success' },
1316
+ ],
1317
+ };
1318
+ await announceRunOutcome(session, { runId: 'run-2', ok: true });
1319
+ const message = session.agentEvents.find((event) => event.type === 'assistant_message');
1320
+ assert.match(message.payload.content, /succès/);
1321
+ });
@@ -15,6 +15,29 @@ import { cancelControlChain, cancelQueuedControlItem } from './controlCancellati
15
15
  import { generateSkillAcknowledgment, runSkillChain } from './skillRun.js';
16
16
  import { emitRuntimeLog } from './supervisor.js';
17
17
  import { findSkill, listSkills } from '../core/skills.js';
18
+ import {
19
+ enrollment,
20
+ isLoopbackAddress,
21
+ isTotpEnabled,
22
+ issueSessionWithTotp,
23
+ loginAttemptAllowed,
24
+ loginStatus,
25
+ pruneLoginAttempts,
26
+ resetLoginAttempts,
27
+ revokeSession,
28
+ verifySessionToken,
29
+ } from './loginSession.js';
30
+ import { loginPageHtml } from './loginPage.js';
31
+
32
+ function loginErrorText(code) {
33
+ const messages = {
34
+ totp_disabled: 'TOTP login is disabled on this manager.',
35
+ no_enrollment: 'Enrollment expired. Reload this page.',
36
+ enrollment_requires_loopback: 'Enrollment must be done from the machine running the manager.',
37
+ invalid_code: 'Invalid verification code.',
38
+ };
39
+ return messages[code] ?? 'Verification failed.';
40
+ }
18
41
 
19
42
  const PRIVATE_CONTROL_INPUTS = new WeakMap();
20
43
 
@@ -75,12 +98,85 @@ export function startRuntimeServer({
75
98
 
76
99
  const server = createServer(async (request, response) => {
77
100
  try {
101
+ const url = new URL(request.url ?? '/', `http://${request.headers.host ?? 'localhost'}`);
102
+
103
+ // ── TOTP login surface — public by design: it is the door, not a room ──
104
+ if (url.pathname === '/login' && request.method === 'GET') {
105
+ const remoteAddress = request.socket?.remoteAddress ?? null;
106
+ const current = enrollment();
107
+ const status = loginStatus();
108
+ if (!status.enabled) {
109
+ sendHtml(response, 200, loginPageHtml({ error: 'TOTP login is disabled on this manager.' }));
110
+ return;
111
+ }
112
+ if (!status.enrolled && !isLoopbackAddress(remoteAddress)) {
113
+ sendHtml(response, 403, loginPageHtml({ error: 'Enrollment must be done from the machine running the manager.' }));
114
+ return;
115
+ }
116
+ sendHtml(response, 200, loginPageHtml({
117
+ enrolled: status.enrolled,
118
+ secret: current?.secret ?? null,
119
+ uri: current?.uri ?? null,
120
+ }));
121
+ return;
122
+ }
123
+ if (url.pathname === '/login/verify' && request.method === 'POST') {
124
+ const remoteAddress = request.socket?.remoteAddress ?? null;
125
+ const allowed = loginAttemptAllowed(remoteAddress);
126
+ if (!allowed.ok) {
127
+ sendJson(response, 429, { ok: false, error: `Too many attempts. Try again in ${allowed.retryAfterSeconds}s.` });
128
+ return;
129
+ }
130
+ const body = await readJson(request).catch(() => ({}));
131
+ const result = issueSessionWithTotp(body?.code, { remoteAddress });
132
+ if (!result.ok) {
133
+ sendJson(response, 401, { ok: false, error: loginErrorText(result.error) });
134
+ return;
135
+ }
136
+ resetLoginAttempts(remoteAddress);
137
+ // Hand the session to the browser as a cookie, on the runtime's own
138
+ // origin. Cookies ignore the port, so a browser that logged in here
139
+ // (the ShellUI gate) carries the same `wiki_session` to `serve` when
140
+ // it runs on this host — one TOTP login for both surfaces. serve sets
141
+ // its own cookie too when a browser reaches it first.
142
+ setSessionCookie(response, result.token, result.expiresAt, request);
143
+ sendJson(response, 200, {
144
+ ok: true,
145
+ token: result.token,
146
+ expiresAt: result.expiresAt,
147
+ page: loginPageHtml({ enrolled: true, sessionExpiresAt: result.expiresAt }),
148
+ });
149
+ return;
150
+ }
151
+ if (url.pathname === '/login/status' && request.method === 'GET') {
152
+ sendJson(response, 200, { ok: true, ...loginStatus() });
153
+ return;
154
+ }
155
+ if (url.pathname === '/logout' && request.method === 'POST') {
156
+ const body = await readJson(request).catch(() => ({}));
157
+ // revokeSession refuses without the session's own token — this route
158
+ // sits before the bearer gate by design, so the token itself is the
159
+ // only proof that the caller is the session being revoked, not an
160
+ // unauthenticated third party forcing the operator out.
161
+ const revoked = revokeSession(body?.token ?? null);
162
+ sendJson(response, 200, { ok: true, revoked });
163
+ return;
164
+ }
165
+ if (request.method === 'GET' && url.pathname === '/session/verify') {
166
+ // Public on purpose: the token IS the credential being checked — the
167
+ // caller already holds it, so verifying it leaks nothing beyond what
168
+ // possession implies. serve and the shell call this without bearer.
169
+ const sessionToken = String(url.searchParams.get('token') ?? '').trim();
170
+ const result = verifySessionToken(sessionToken);
171
+ sendJson(response, 200, { ok: result.ok, ...result });
172
+ return;
173
+ }
174
+
78
175
  if (!isAuthorized(request, token)) {
79
176
  sendJson(response, 401, { error: 'Unauthorized' });
80
177
  return;
81
178
  }
82
179
 
83
- const url = new URL(request.url ?? '/', `http://${request.headers.host ?? 'localhost'}`);
84
180
  if (request.method === 'GET' && url.pathname === '/health') {
85
181
  const workspace = workspaceFromUrl(url);
86
182
  const context = workspace ? await resolveContext({ workspace }) : null;
@@ -433,6 +529,20 @@ export function startRuntimeServer({
433
529
  }
434
530
  return;
435
531
  }
532
+ // A run/job status question is answered by the runtime itself, whatever
533
+ // the mode and whether or not a run is active. Left to the model it
534
+ // confused the runtime runId with a production job id ("job not
535
+ // found"); in chat mode it had no runtime status tool at all.
536
+ if (asksForRunStatus(input)) {
537
+ const status = controlStatus(context, store);
538
+ sendJson(response, 200, {
539
+ accepted: true,
540
+ kind: 'observe',
541
+ ...status,
542
+ explanation: explainControlState(status),
543
+ });
544
+ return;
545
+ }
436
546
  if (context.running && !readOnlyChat) {
437
547
  // Agent-mode message while a run is active. Classify once: control
438
548
  // verbs and new tasks go to the control lane, plain conversation is
@@ -649,6 +759,51 @@ export function startRuntimeServer({
649
759
  sendJson(response, 200, { truncated: true, index, removedEvents });
650
760
  return;
651
761
  }
762
+ // Compact: a deliberate "forget everything said so far in this
763
+ // workspace" action (served chat's memory gauge). Unlike
764
+ // /conversation/truncate above, nothing is deleted from the event log —
765
+ // the audit trail (GET /audit) stays intact. A single conversation_reset
766
+ // event is enough: the reducer (core/agentEvents.js) moves the
767
+ // conversationSeedStart boundary on it, and since executeInteractiveTurn
768
+ // rebuilds its conversationSeed from a fresh reduceAgentEvents() replay
769
+ // on every turn, future turns stop seeing anything before this point
770
+ // while the displayed conversation (and the ShellUI thread) stays whole.
771
+ if (request.method === 'POST' && url.pathname === '/conversation/compact') {
772
+ const { workspace, context } = await resolveBodyContext(request, url);
773
+ if (context?.running) {
774
+ sendJson(response, 409, { compacted: false, reason: 'run_active' });
775
+ return;
776
+ }
777
+ const resolvedWorkspace = context?.workspace ?? workspace ?? null;
778
+ if (!resolvedWorkspace) {
779
+ sendJson(response, 400, { compacted: false, reason: 'workspace_required' });
780
+ return;
781
+ }
782
+ let summary = null;
783
+ if (context?.session) {
784
+ const conversation = Array.isArray(context.session.agentProjection?.conversation)
785
+ ? context.session.agentProjection.conversation
786
+ : [];
787
+ const seedStart = Math.max(0, Number(context.session.agentProjection?.conversationSeedStart) || 0);
788
+ const previousSummary = context.session.agentProjection?.conversationSummary ?? null;
789
+ // Summarize BEFORE dispatching: the event's payload carries the
790
+ // result so the reducer only ever has to store a plain string, and
791
+ // a run cannot start concurrently (already refused with 409 above)
792
+ // to move conversation.length out from under this read.
793
+ summary = await summarizeCompactedConversation(context.session, {
794
+ previousSummary,
795
+ segment: conversation.slice(seedStart),
796
+ });
797
+ dispatchAgentEvent(context.session, createAgentEvent('conversation_reset', {
798
+ origin: 'user',
799
+ workspace: resolvedWorkspace,
800
+ payload: summary ? { summary } : {},
801
+ }));
802
+ }
803
+ publishState(resolvedWorkspace, context);
804
+ sendJson(response, 200, { compacted: true, summary });
805
+ return;
806
+ }
652
807
  if (request.method === 'POST' && url.pathname === '/resume') {
653
808
  const workspace = workspaceFromUrl(url);
654
809
  const result = await resume?.({ workspace });
@@ -681,6 +836,12 @@ export function startRuntimeServer({
681
836
  }
682
837
  });
683
838
 
839
+ // Housekeeping for the in-memory login-attempt rate limiter: nothing else
840
+ // ever calls pruneLoginAttempts, so without this the `attempts` Map grows
841
+ // by one entry per distinct source address for the life of the process.
842
+ const loginAttemptPruneTimer = setInterval(() => pruneLoginAttempts(), 10 * 60 * 1000);
843
+ loginAttemptPruneTimer.unref?.();
844
+
684
845
  return new Promise((resolve, reject) => {
685
846
  server.once('error', reject);
686
847
  server.listen(port, host, () => {
@@ -692,6 +853,7 @@ export function startRuntimeServer({
692
853
  publish,
693
854
  drainControl: (context) => drainControlQueue(context),
694
855
  close: () => new Promise((closeResolve, closeReject) => {
856
+ clearInterval(loginAttemptPruneTimer);
695
857
  for (const client of clients) client.response.end();
696
858
  clients.clear();
697
859
  server.close((err) => (err ? closeReject(err) : closeResolve()));
@@ -1206,6 +1368,44 @@ async function handleControlMessage(context, store, input, { intent = null, star
1206
1368
  falls back to the deterministic English catalog when no LLM is configured or
1207
1369
  the call fails. The fallback is what keeps the lane deterministic-under-failure.
1208
1370
  */
1371
+ const CONVERSATION_SUMMARY_TIMEOUT_MS = 20_000;
1372
+ const CONVERSATION_SUMMARY_MAX_INPUT_CHARS = 8_000;
1373
+
1374
+ /*
1375
+ A compact does not just cut older turns from conversationSeed — it replaces
1376
+ them with a short rolling summary, so a decision made 20 messages ago is not
1377
+ gone from Donna's grounding entirely, only condensed. Best-effort: no LLM
1378
+ configured, an empty reply, or a call failure all fall back to keeping
1379
+ whatever summary already existed (never worse than before this compact),
1380
+ the same deterministic-under-failure shape as generateControlAcknowledgment.
1381
+ */
1382
+ async function summarizeCompactedConversation(session, { previousSummary, segment }) {
1383
+ const llm = session?.llm;
1384
+ const transcript = (Array.isArray(segment) ? segment : [])
1385
+ .filter((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim())
1386
+ .map((message) => `${message.role === 'user' ? 'User' : 'Assistant'}: ${String(message.content).trim()}`)
1387
+ .join('\n')
1388
+ .slice(0, CONVERSATION_SUMMARY_MAX_INPUT_CHARS);
1389
+ if (!transcript) return previousSummary || null;
1390
+ if (!(llm && typeof llm.complete === 'function')) return previousSummary || null;
1391
+ try {
1392
+ const reply = await llm.complete({
1393
+ system: 'You maintain a compact working memory for Donna, a workspace assistant. You are shown an optional PREVIOUS SUMMARY and a NEW SEGMENT of conversation about to leave the assistant\'s context window. Write ONE updated summary that preserves the facts, decisions, open questions and user preferences that still matter for future turns. Be concise: well under 200 words. Return only the summary text — no preamble, no meta-commentary, no headings.',
1394
+ input: [
1395
+ previousSummary ? `PREVIOUS SUMMARY:\n${previousSummary}` : null,
1396
+ `NEW SEGMENT:\n${transcript}`,
1397
+ ].filter(Boolean).join('\n\n'),
1398
+ signal: AbortSignal.timeout(CONVERSATION_SUMMARY_TIMEOUT_MS),
1399
+ });
1400
+ const text = String(reply ?? '').trim();
1401
+ if (text) return text;
1402
+ emitRuntimeLog(session, 'conversation-compact: LLM returned an empty summary, keeping the previous one');
1403
+ } catch (err) {
1404
+ emitRuntimeLog(session, `conversation-compact: summary LLM call failed, keeping the previous summary — ${err instanceof Error ? err.message : String(err)}`);
1405
+ }
1406
+ return previousSummary || null;
1407
+ }
1408
+
1209
1409
  async function generateControlAcknowledgment(session, { kind, input }) {
1210
1410
  const language = String(session?.language ?? '').trim().toLowerCase() || 'en';
1211
1411
  const llm = session?.llm;
@@ -1446,6 +1646,18 @@ function rejectPlanPatch(context, store, patchId, reason) {
1446
1646
  };
1447
1647
  }
1448
1648
 
1649
+ // A question about the run/job currently executing. Deliberately narrow — a
1650
+ // status word AND a run/job noun — so it never hijacks an ordinary "explain how
1651
+ // X works" question. Such a question must be answered by the runtime itself:
1652
+ // left to the model, a runtime runId was mistaken for a production job id and
1653
+ // reported as "not found", and a read-only chat turn had no runtime status tool.
1654
+ function asksForRunStatus(input) {
1655
+ const text = String(input ?? '');
1656
+ const statusWord = /\b(status|statut|progression|progress|avancement|o[uù] en est|o[uù] en sont)\b/i;
1657
+ const runNoun = /\b(job|run|t[aâ]che|task|build|ingest|pipeline|export|polish|traitement)\b/i;
1658
+ return statusWord.test(text) && runNoun.test(text);
1659
+ }
1660
+
1449
1661
  // Classifier for the control lane's free-text messages. The classification is
1450
1662
  // LLM-backed: the only deterministic matches left are the runtime's own
1451
1663
  // control verbs (cancel, an explicit "later/queue", status and plan-change
@@ -1482,6 +1694,14 @@ async function classifyControlMessage(input, status, { forcedIntent = null, llm
1482
1694
  if (/\b(o[uù] en es[t-]|status|statut|progress|progression|logs?|explique|explain|inspect|show|montre|quoi de neuf)\b/i.test(lower)) {
1483
1695
  return { kind: 'observe', confidence: 0.86, reason: 'status_or_explanation_request' };
1484
1696
  }
1697
+ // A bare "yes" answers the runtime's own last prompt (the launch
1698
+ // acknowledgement used to end on "check progress or cancel?"). While a run is
1699
+ // active, the only thing the runtime can act on is a status check: treating
1700
+ // the word as ordinary conversation made the read-only chat fallback lecture
1701
+ // the user about switching modes instead of answering.
1702
+ if (status.running && /^\s*(oui|yes|yep|ok|okay|vas[- ]?y|d'accord|daccord|entendu)\b/i.test(lower)) {
1703
+ return { kind: 'observe', confidence: 0.7, reason: 'confirmation_of_runtime_prompt' };
1704
+ }
1485
1705
  if (status.running && /\b(ajoute|add|change|modifie|modify|remplace|replace|retire|remove|skip|ignore|apr[eè]s|before|after|chaque|each|plan|step|t[aâ]che)\b/i.test(lower)) {
1486
1706
  return { kind: 'modify_run', confidence: 0.78, reason: 'active_run_change_request' };
1487
1707
  }
@@ -1549,6 +1769,25 @@ function sendJson(response, statusCode, value) {
1549
1769
  response.end(`${JSON.stringify(value)}\n`);
1550
1770
  }
1551
1771
 
1772
+ function sendHtml(response, statusCode, html) {
1773
+ response.writeHead(statusCode, { 'Content-Type': 'text/html; charset=utf-8' });
1774
+ response.end(html);
1775
+ }
1776
+
1777
+ // Mirrors serve's cookie (same name, flags and lifetime) so one runtime-issued
1778
+ // session is also the one serve validates. Secure only when the request
1779
+ // already arrived over TLS — a localhost HTTP install must still get the
1780
+ // cookie, but a proxied HTTPS one must not leak it.
1781
+ function setSessionCookie(response, token, expiresAt, request) {
1782
+ const maxAgeSeconds = Math.max(1, Math.floor((Number(expiresAt) - Date.now()) / 1000));
1783
+ const tls = Boolean(request.socket?.encrypted) || request.headers['x-forwarded-proto'] === 'https';
1784
+ const secure = tls ? '; Secure' : '';
1785
+ response.setHeader(
1786
+ 'Set-Cookie',
1787
+ `wiki_session=${encodeURIComponent(token)}; Path=/; HttpOnly; SameSite=Lax; Max-Age=${maxAgeSeconds}${secure}`,
1788
+ );
1789
+ }
1790
+
1552
1791
  function readRequiredPatchId(body, response) {
1553
1792
  const patchId = String(body.patchId ?? body.id ?? '').trim();
1554
1793
  if (!patchId) {