@dotdrelle/wiki-manager 0.12.11 → 0.12.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,4 @@
1
1
  {
2
- "version": "0.12.11",
3
- "commit": "d8eae0b"
2
+ "version": "0.12.12",
3
+ "commit": "e1e43ae"
4
4
  }
package/src/core/mcp.js CHANGED
@@ -1,7 +1,7 @@
1
1
  import { existsSync, readFileSync } from 'node:fs';
2
2
  import { managerEnvFile, managerMcpEndpointsFile, readEnvFile } from './env.js';
3
3
 
4
- const WIKI_MANAGER_VERSION = '0.12.11';
4
+ const WIKI_MANAGER_VERSION = '0.12.12';
5
5
 
6
6
  function envValue(key) {
7
7
  const filePath = managerEnvFile();
@@ -305,8 +305,7 @@ export function formatMcpToolResult(result) {
305
305
  const DEFAULT_TOOL_RESULT_MAX_CHARS = 16000;
306
306
 
307
307
  function toolResultMaxChars() {
308
- const parsed = Number(process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS);
309
- return Number.isFinite(parsed) && parsed > 0 ? Math.floor(parsed) : DEFAULT_TOOL_RESULT_MAX_CHARS;
308
+ return DEFAULT_TOOL_RESULT_MAX_CHARS;
310
309
  }
311
310
 
312
311
  // Bound what a tool result injects into the LLM context and the conversation
@@ -529,15 +529,3 @@ test('truncateToolResult keeps short results intact and bounds long ones head+ta
529
529
  assert.match(bounded, /caractères tronqués/);
530
530
  });
531
531
 
532
- test('truncateToolResult honours WIKI_MANAGER_TOOL_RESULT_MAX_CHARS', () => {
533
- const previous = process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS;
534
- process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS = '500';
535
- try {
536
- const bounded = truncateToolResult('y'.repeat(5000));
537
- assert.ok(bounded.length < 700);
538
- assert.match(bounded, /caractères tronqués/);
539
- } finally {
540
- if (previous === undefined) delete process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS;
541
- else process.env.WIKI_MANAGER_TOOL_RESULT_MAX_CHARS = previous;
542
- }
543
- });
@@ -9,6 +9,30 @@ export function resolveSchedulerConcurrency(value = process.env.WIKI_MANAGER_SCH
9
9
  : DEFAULT_SCHEDULER_CONCURRENCY;
10
10
  }
11
11
 
12
+ export function resolvePlanConcurrency({ plan = [], agents = [], configured = null } = {}) {
13
+ const capabilities = new Set(plan.map((task) => task?.requiredCapability).filter(Boolean).map(String));
14
+ const assignedAgents = new Set(plan.map((task) => task?.agentInstanceId).filter(Boolean).map(String));
15
+ const relevantAgents = agents.filter((agent) => {
16
+ const id = String(agent?.agentInstanceId ?? agent?.description?.agentInstanceId ?? '');
17
+ if (id && assignedAgents.has(id)) return true;
18
+ return (agent?.description?.capabilities ?? []).some((capability) => capabilities.has(String(capability?.id ?? '')));
19
+ });
20
+ const values = [
21
+ positiveInteger(configured),
22
+ ...plan.flatMap(concurrencyValues),
23
+ ...relevantAgents.flatMap(concurrencyValues),
24
+ ].filter(Boolean);
25
+ return values.length > 0 ? Math.max(1, Math.min(...values)) : DEFAULT_SCHEDULER_CONCURRENCY;
26
+ }
27
+
28
+ export function resolveCapabilityConcurrency(agent = null, ...constraints) {
29
+ const values = [
30
+ ...concurrencyValues(agent),
31
+ ...constraints.map(positiveInteger),
32
+ ].filter(Boolean);
33
+ return values.length > 0 ? Math.max(1, Math.min(...values)) : DEFAULT_SCHEDULER_CONCURRENCY;
34
+ }
35
+
12
36
  export function effectiveConcurrency(group = null, agent = null, donna = null, provider = null) {
13
37
  const limits = [
14
38
  ...concurrencyValues(donna),
@@ -6,7 +6,12 @@ import test from 'node:test';
6
6
  import { createBudgetManager } from './budgetManager.js';
7
7
  import { readyTasks } from './dependencyResolver.js';
8
8
  import { createLockManager } from './lockManager.js';
9
- import { effectiveConcurrency, startReadyTasks } from './scheduler.js';
9
+ import {
10
+ effectiveConcurrency,
11
+ resolveCapabilityConcurrency,
12
+ resolvePlanConcurrency,
13
+ startReadyTasks,
14
+ } from './scheduler.js';
10
15
 
11
16
  test('dependencyResolver holds a barrier task until its dependsOnGroup is done', () => {
12
17
  const plan = [
@@ -76,6 +81,40 @@ test('scheduler.effectiveConcurrency returns the minimum effective concurrency',
76
81
  assert.equal(effectiveConcurrency(group, agent, donna, provider), 2);
77
82
  });
78
83
 
84
+ test('scheduler uses the relevant agent declaration instead of hard-capping plans at three', () => {
85
+ const plan = [{ id: 'task-1', requiredCapability: 'ingest' }];
86
+ const agents = [{
87
+ description: {
88
+ capabilities: [{ id: 'ingest' }],
89
+ limits: { recommendedConcurrency: 10, maxConcurrency: 12 },
90
+ },
91
+ }];
92
+
93
+ assert.equal(resolvePlanConcurrency({ plan, agents }), 10);
94
+ assert.equal(resolvePlanConcurrency({ plan, agents, configured: 3 }), 3);
95
+ assert.equal(resolvePlanConcurrency({ plan, agents, configured: 20 }), 10);
96
+ });
97
+
98
+ test('scheduler ignores unrelated agents and falls back to three without declarations', () => {
99
+ const plan = [{ id: 'task-1', requiredCapability: 'ingest' }];
100
+ const agents = [{
101
+ description: {
102
+ capabilities: [{ id: 'production' }],
103
+ limits: { recommendedConcurrency: 1 },
104
+ },
105
+ }];
106
+
107
+ assert.equal(resolvePlanConcurrency({ plan, agents }), 3);
108
+ });
109
+
110
+ test('capability constraints can lower but never raise an agent declaration', () => {
111
+ const agent = { description: { limits: { recommendedConcurrency: 6, maxConcurrency: 10 } } };
112
+
113
+ assert.equal(resolveCapabilityConcurrency(agent), 6);
114
+ assert.equal(resolveCapabilityConcurrency(agent, 2), 2);
115
+ assert.equal(resolveCapabilityConcurrency(agent, 20), 6);
116
+ });
117
+
79
118
  test('startReadyTasks starts only ready tasks and respects lock starvation', () => {
80
119
  const active = new Map();
81
120
  const lockManager = createLockManager();
@@ -52,6 +52,7 @@ export async function postRuntimeRun(input, {
52
52
  workspace = null,
53
53
  evaluate = undefined,
54
54
  replans = undefined,
55
+ capabilityPlan = undefined,
55
56
  } = {}) {
56
57
  const response = await fetch(runtimeEndpoint(url, '/run', workspace), {
57
58
  method: 'POST',
@@ -59,7 +60,7 @@ export async function postRuntimeRun(input, {
59
60
  ...runtimeHeaders(token),
60
61
  'Content-Type': 'application/json',
61
62
  },
62
- body: JSON.stringify(Object.assign({ input, workspace }, evaluate !== undefined && { evaluate }, replans !== undefined && { replans })),
63
+ body: JSON.stringify(Object.assign({ input, workspace }, evaluate !== undefined && { evaluate }, replans !== undefined && { replans }, capabilityPlan !== undefined && { capabilityPlan })),
63
64
  });
64
65
  if (!response.ok) {
65
66
  const err = new Error(`Runtime run failed: HTTP ${response.status}`);
@@ -109,7 +109,7 @@ export async function runtimeHealthOrNull(url = runtimeUrlFromEnv(), token = run
109
109
  // alive after exit produced zombie runtimes running yesterday's code and
110
110
  // yesterday's endpoints. Nuance preserved: if a run is active anywhere, the
111
111
  // runtime is left alive so the run survives the shell (that promise stays).
112
- export async function shutdownOwnedRuntime(runtime, { log = () => {} } = {}) {
112
+ export async function shutdownOwnedRuntime(runtime, { log = (_message) => {} } = {}) {
113
113
  if (!runtime?.url || !runtime?.started) return { action: 'kept', reason: 'not_owned' };
114
114
  try {
115
115
  const health = await runtimeHealthOrNull(runtime.url, runtime.token);
@@ -9,10 +9,16 @@ import { createBudgetManager, BudgetExceededError } from '../orchestrator/budget
9
9
  import { createDispatcher } from '../orchestrator/dispatcher.js';
10
10
  import { assertValidatedFragment } from '../orchestrator/planValidator.js';
11
11
  import { createResultAggregator } from '../orchestrator/resultAggregator.js';
12
- import { drainActive, resolveSchedulerConcurrency, startReadyTasks } from '../orchestrator/scheduler.js';
12
+ import { drainActive, resolvePlanConcurrency, startReadyTasks } from '../orchestrator/scheduler.js';
13
13
  import { emitRuntimeLog, pollActivitiesOnce } from './supervisor.js';
14
14
 
15
- const DEFAULT_MAX_REPLANS = 2;
15
+ // 0 by default: automatic replans turn evaluator/replanner TEXT into
16
+ // executable pseudo-tasks (no capability, no operation) that stall at 0%
17
+ // and pile up as replan-1/2/3 ghost work — the same disease as the removed
18
+ // text-plan extraction. Failures now end with an honest report; the user
19
+ // (or a stronger model) decides what to do next. Re-enable explicitly with
20
+ // WIKI_MANAGER_REPLANNER_MAX_REPLANS if desired.
21
+ const DEFAULT_MAX_REPLANS = 0;
16
22
 
17
23
  async function waitForRuntimeActivities(session, startedActivities, { timeoutMs, signal, pollBusy }) {
18
24
  const deadline = Date.now() + timeoutMs;
@@ -43,13 +49,35 @@ async function waitForRuntimeActivities(session, startedActivities, { timeoutMs,
43
49
  return { ok: false, timedOut: true, completed: tracked };
44
50
  }
45
51
 
46
- export async function runRuntimeAgenticLoop(agent, session, initialInput, { signal, timeoutMs, maxTurns, runId, pollBusy, parallelHandoff = false }) {
52
+ // Last chat exchanges (user/assistant) that preceded this run, so the run's
53
+ // LLM knows WHAT was agreed before acting. Long messages are clipped: the
54
+ // context is for grounding, not for re-reading novels.
55
+ // Env knobs (documented in .env.example): every tunable introduced by the
56
+ // grounding/orchestration work is overridable — nothing business-critical
57
+ // is frozen in code.
58
+ export function conversationSeed(session, currentInput, { limit = 12, maxChars = 2000 } = {}) {
59
+ const conversation = Array.isArray(session.agentProjection?.conversation)
60
+ ? session.agentProjection.conversation
61
+ : [];
62
+ const seed = conversation
63
+ .filter((message) => ['user', 'assistant'].includes(message?.role) && String(message?.content ?? '').trim())
64
+ .slice(-limit)
65
+ .map((message) => ({ role: message.role, content: String(message.content).slice(0, maxChars) }));
66
+ // The run's own triggering user message is appended by the loop itself —
67
+ // drop it from the seed to avoid sending it twice.
68
+ const last = seed.at(-1);
69
+ if (last && last.role === 'user' && last.content === String(currentInput ?? '').slice(0, maxChars)) seed.pop();
70
+ return seed;
71
+ }
72
+
73
+ export async function runRuntimeAgenticLoop(agent, session, initialInput, { signal, timeoutMs, maxTurns, runId, pollBusy, parallelHandoff = false, initialMessages = [] }) {
47
74
  return runAgenticLoop(agent, session, initialInput, {
48
75
  signal,
49
76
  timeoutMs,
50
77
  maxTurns,
51
78
  runId,
52
79
  parallelHandoff,
80
+ initialMessages,
53
81
  deterministicTerminalSummary: true,
54
82
  abortMessage: 'Runtime run cancelled.',
55
83
  waitForActivities: (turnSession, startedActivities, waitOptions) =>
@@ -107,6 +135,9 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
107
135
  } = {}) {
108
136
  let currentInput = initialInput ?? input;
109
137
  let replansLeft = Math.max(0, Math.floor(Number(maxReplans) || 0));
138
+ // Computed ONCE at run start: the pre-run chat. Re-computing inside the
139
+ // loop would re-ingest this run's own turns and duplicate them.
140
+ const runConversationSeed = conversationSeed(session, currentInput);
110
141
 
111
142
  while (true) {
112
143
  sanitizeSessionPlanForExecution(session, runId);
@@ -127,6 +158,7 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
127
158
  runId,
128
159
  pollBusy,
129
160
  parallelHandoff: true,
161
+ initialMessages: runConversationSeed,
130
162
  });
131
163
  if (result.ok && result.handoff) continue;
132
164
  if (!result.ok) {
@@ -212,6 +244,20 @@ export async function runRuntimeAgenticWorkflow(agent, session, input, {
212
244
  continue;
213
245
  }
214
246
  }
247
+ // Surface the verdict in the CHAT: the work that ran stays done, the
248
+ // user sees why the evaluator was unsatisfied and decides — no
249
+ // self-generated follow-up tasks.
250
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', {
251
+ origin: 'runtime',
252
+ runId,
253
+ payload: {
254
+ content: [
255
+ `Le run est terminé mais l'évaluation le juge incomplet : ${evaluation.reason}`,
256
+ evaluation.suggestedAction ? `Piste suggérée : ${evaluation.suggestedAction}` : null,
257
+ 'Aucune tâche supplémentaire n\'a été créée automatiquement — dis-moi si tu veux poursuivre.',
258
+ ].filter(Boolean).join('\n'),
259
+ },
260
+ }));
215
261
  dispatchAgentEvent(session, createAgentEvent('run_error', {
216
262
  origin: 'runtime',
217
263
  runId,
@@ -240,7 +286,7 @@ export async function runRuntimeParallelPlan(agent, session, input, {
240
286
  maxTurns,
241
287
  runId = null,
242
288
  pollBusy,
243
- concurrency = resolveSchedulerConcurrency(),
289
+ concurrency = null,
244
290
  fragment = null,
245
291
  assignmentManager = null,
246
292
  attemptManager = null,
@@ -252,7 +298,15 @@ export async function runRuntimeParallelPlan(agent, session, input, {
252
298
  dispatcherPollIntervalMs = 250,
253
299
  } = {}) {
254
300
  if (fragment != null) assertValidatedFragment(fragment);
255
- const limit = Math.max(1, Math.floor(Number(concurrency) || 1));
301
+ const agents = session.agentRegistry?.snapshot?.() ?? session.agentRegistrySnapshot ?? [];
302
+ const configuredConcurrency = Number(concurrency) > 0
303
+ ? Number(concurrency)
304
+ : Number(process.env.WIKI_MANAGER_CAPABILITY_CONCURRENCY || process.env.WIKI_MANAGER_SCHEDULER_CONCURRENCY);
305
+ const limit = resolvePlanConcurrency({
306
+ plan: session.headlessPlan ?? [],
307
+ agents,
308
+ configured: configuredConcurrency,
309
+ });
256
310
  const active = new Map();
257
311
  const attempts = attemptManager ?? createAttemptManager();
258
312
  const assigner = assignmentManager ?? createAssignmentManager({ session });
@@ -278,6 +332,15 @@ export async function runRuntimeParallelPlan(agent, session, input, {
278
332
  sanitizeSessionPlanForExecution(session, runId);
279
333
  ensurePlanProjection(session, runId);
280
334
  emitRuntimeLog(session, `scheduler: parallel plan enabled (concurrency ${limit})`);
335
+ let approvalNoticeSent = false;
336
+ // Interactive approvals do NOT expire: the user has /approve, "valide
337
+ // tout", /cancel and /run kill — an arbitrary timer only created mystery
338
+ // failures. A deadline exists only when explicitly configured (headless
339
+ // runs, CI) via the session or the env escape hatch.
340
+ const configuredApprovalWait = Number(session._approvalTimeoutMs) > 0
341
+ ? Number(session._approvalTimeoutMs)
342
+ : (Number(process.env.WIKI_MANAGER_APPROVAL_TIMEOUT_MS) > 0 ? Number(process.env.WIKI_MANAGER_APPROVAL_TIMEOUT_MS) : null);
343
+ const approvalDeadline = configuredApprovalWait ? Date.now() + configuredApprovalWait : Infinity;
281
344
 
282
345
  try {
283
346
  while (true) {
@@ -371,6 +434,44 @@ export async function runRuntimeParallelPlan(agent, session, input, {
371
434
  emitRuntimeLog(session, `scheduler: budget exceeded (${exceeded.reason})`);
372
435
  return { ok: false, budgetExceeded: true, reason: exceeded.reason, budget: exceeded, completed: sessionActivities(session), failures };
373
436
  }
437
+ const needingApproval = pending.filter((step) => step.requiresApproval === true);
438
+ if (needingApproval.length > 0) {
439
+ // The plan is only blocked on a HUMAN decision — wait for it
440
+ // (bounded) instead of declaring the run stalled. Announce once in
441
+ // the chat: users cannot approve what they never saw asked.
442
+ if (!approvalNoticeSent) {
443
+ approvalNoticeSent = true;
444
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', {
445
+ origin: 'runtime',
446
+ runId,
447
+ payload: {
448
+ content: [
449
+ `⏸ Approbation requise avant exécution : ${needingApproval.length} tâche(s) mutante(s) en attente.`,
450
+ ...needingApproval.slice(0, 5).map((step) => ` - ${step.description ?? step.id}`),
451
+ needingApproval.length > 5 ? ` … et ${needingApproval.length - 5} autre(s).` : null,
452
+ 'Réponds « valide tout » (ou tape /approve) pour lancer, « annule » pour abandonner.',
453
+ ].filter(Boolean).join('\n'),
454
+ },
455
+ }));
456
+ emitRuntimeLog(session, `scheduler: waiting for approval (${needingApproval.length} task(s))`);
457
+ }
458
+ if (Date.now() < approvalDeadline) {
459
+ await new Promise((resolveDelay) => setTimeout(resolveDelay, 500));
460
+ continue;
461
+ }
462
+ // Configured timeout (headless/CI) reached: say it PLAINLY in the
463
+ // chat and let run_error clean the plan/activities so nothing
464
+ // lingers in the panels.
465
+ emitRuntimeLog(session, 'scheduler: approval wait timed out');
466
+ dispatchAgentEvent(session, createAgentEvent('assistant_message', {
467
+ origin: 'runtime',
468
+ runId,
469
+ payload: {
470
+ content: `⏱ Approbation non reçue dans le délai imparti — run arrêté, ${needingApproval.length} tâche(s) annulée(s). Relance la demande quand tu veux.`,
471
+ },
472
+ }));
473
+ return { ok: false, stalled: true, reason: 'awaiting_approval', completed: sessionActivities(session), failures };
474
+ }
374
475
  const reason = pending.every((step) => approvalWaitingStatus(step.status)) ? 'awaiting_approval' : 'no_ready_plan_task';
375
476
  emitRuntimeLog(session, `scheduler: stalled (${reason})`);
376
477
  return { ok: false, stalled: true, reason, completed: sessionActivities(session), failures };
@@ -1016,7 +1117,7 @@ function formatRecentConversation(session, n = 12) {
1016
1117
  .join('\n');
1017
1118
  }
1018
1119
 
1019
- function resolveMaxReplans(value = process.env.WIKI_MANAGER_REPLANNER_MAX_REPLANS) {
1120
+ function resolveMaxReplans(value) {
1020
1121
  const parsed = Number(value);
1021
1122
  return Number.isFinite(parsed) ? Math.max(0, Math.floor(parsed)) : DEFAULT_MAX_REPLANS;
1022
1123
  }
@@ -713,7 +713,9 @@ test('runRuntimeParallelPlan fails cleanly when scheduler budget is exceeded', a
713
713
  assert.equal(result.budgetExceeded, true);
714
714
  assert.equal(result.reason, 'max_tasks_exceeded');
715
715
  assert.ok(session.agentEvents.some((event) => event.type === 'run_error' && event.payload?.budget?.reason === 'max_tasks_exceeded'));
716
- assert.equal(session.headlessPlan[0].status, 'pending');
716
+ // run_error now cancels leftover pending steps: a dead run must not leave
717
+ // ghost work in the panels or reappear at the next relaunch.
718
+ assert.equal(session.headlessPlan[0].status, 'cancelled');
717
719
  });
718
720
 
719
721
  test('runRuntimeParallelPlan retries a retryable task on a fallback agent', async () => {
@@ -915,3 +917,36 @@ async function waitFor(predicate, timeoutMs = 500) {
915
917
  }
916
918
  assert.fail('condition was not met before timeout');
917
919
  }
920
+
921
+ test('runtime runs are seeded with the chat that preceded them', async () => {
922
+ // "Donna n'a pas de contexte" was literally true: runs started with an
923
+ // empty history, so the model reinvented the missing context. The seed
924
+ // carries the prior exchanges, minus the run's own triggering message.
925
+ const { conversationSeed } = await import('./runner.js');
926
+ const session = {
927
+ agentProjection: {
928
+ conversation: [
929
+ { role: 'user', content: 'peux-tu configurer le CME ?' },
930
+ { role: 'assistant', content: 'CME configuré sur confluent.meteo.fr. Veux-tu ingérer les documents en attente ?' },
931
+ { role: 'user', content: 'oui lance l\'ingestion' },
932
+ ],
933
+ },
934
+ };
935
+
936
+ const seed = conversationSeed(session, "oui lance l'ingestion");
937
+ assert.deepEqual(seed.map((message) => message.role), ['user', 'assistant']);
938
+ assert.match(seed[1].content, /confluent\.meteo\.fr/);
939
+
940
+ // Long entries are clipped, empty/technical roles dropped.
941
+ const noisy = {
942
+ agentProjection: {
943
+ conversation: [
944
+ { role: 'command', content: 'Runtime is idle.' },
945
+ { role: 'assistant', content: 'x'.repeat(5000) },
946
+ ],
947
+ },
948
+ };
949
+ const clipped = conversationSeed(noisy, 'autre demande');
950
+ assert.equal(clipped.length, 1);
951
+ assert.equal(clipped[0].content.length, 2000);
952
+ });
@@ -1,3 +1,5 @@
1
+ import { openSync, readSync, closeSync, fstatSync } from 'node:fs';
2
+ import { isAbsolute, join, normalize, resolve } from 'node:path';
1
3
  import { createAgentEvent, dispatchAgentEvent } from '../core/agentEvents.js';
2
4
  import { extractActivity, parseJsonText, sessionActivities } from '../core/activity.js';
3
5
  import { callMcpTool, formatMcpToolResult } from '../core/mcp.js';
@@ -144,9 +146,19 @@ export async function pollActivitiesOnce(session, {
144
146
  }
145
147
  session._activityLogCursors[key] = logTail.at(-1);
146
148
  }
149
+ // The job log is terse — the actual narrative (per-document plans,
150
+ // LLM calls with token counts, apply operations) lives in the
151
+ // engine's trace file, whose path the status exposes. Tail it from
152
+ // the host and surface the significant events.
153
+ if (progress.traceFile) {
154
+ for (const line of readNewTraceLines(session, key, progress.traceFile)) {
155
+ emitRuntimeLog(session, `trace: ${line}`);
156
+ }
157
+ }
147
158
  if (polledActivity.terminal) {
148
159
  delete session._activityLogKeys[key];
149
160
  delete session._activityLogCursors?.[key];
161
+ delete session._activityTraceCursors?.[key];
150
162
  await startNextQueuedJob(session, {
151
163
  addLog: (message) => emitRuntimeLog(session, message),
152
164
  });
@@ -210,3 +222,52 @@ export async function cancelActiveActivityJobs(session, { callTool = callMcpTool
210
222
  }
211
223
  return cancelled;
212
224
  }
225
+
226
+ // Trace events worth surfacing in the chat-side log: plan/apply milestones,
227
+ // LLM calls (with token counts), warnings and errors. The raw trace is far
228
+ // chattier — streaming everything would recreate the noise the dedupe killed.
229
+ const TRACE_EVENT_PATTERN = /\b(llm:start|llm:end|llm:json|llm:error|ingest:plan|ingest:operations|ingest:apply|ingest:source|build:template|retrieval:|embedding:|WARN|ERROR)\b/;
230
+ const TRACE_MAX_BYTES_PER_POLL = 64 * 1024;
231
+ const TRACE_MAX_LINES_PER_POLL = 12;
232
+
233
+ export function readNewTraceLines(session, key, traceFile) {
234
+ const workspacePath = session?.workspacePath;
235
+ if (!workspacePath) return [];
236
+ // The trace path comes from an agent payload: never let it escape the
237
+ // workspace directory.
238
+ const resolved = resolve(workspacePath, normalize(String(traceFile)));
239
+ if (isAbsolute(String(traceFile)) || !resolved.startsWith(resolve(workspacePath))) return [];
240
+ let fd;
241
+ try {
242
+ fd = openSync(resolved, 'r');
243
+ const size = fstatSync(fd).size;
244
+ session._activityTraceCursors ??= {};
245
+ let offset = session._activityTraceCursors[key];
246
+ // First sighting (or file rotation): start near the end, not at byte 0 —
247
+ // replaying a long history would flood the panel.
248
+ if (offset == null || offset > size) offset = Math.max(0, size - 4096);
249
+ if (size <= offset) return [];
250
+ const length = Math.min(size - offset, TRACE_MAX_BYTES_PER_POLL);
251
+ const buffer = Buffer.alloc(length);
252
+ readSync(fd, buffer, 0, length, offset);
253
+ const chunk = buffer.toString('utf8');
254
+ // Only advance past COMPLETE lines so a partially-written line is
255
+ // re-read whole on the next poll.
256
+ const lastNewline = chunk.lastIndexOf('\n');
257
+ if (lastNewline === -1) return [];
258
+ session._activityTraceCursors[key] = offset + Buffer.byteLength(chunk.slice(0, lastNewline + 1), 'utf8');
259
+ return chunk
260
+ .slice(0, lastNewline)
261
+ .split('\n')
262
+ .map((line) => line.trim())
263
+ .filter((line) => line && TRACE_EVENT_PATTERN.test(line))
264
+ .slice(0, TRACE_MAX_LINES_PER_POLL)
265
+ // Strip the ISO timestamp + elapsed prefix: the runtime log adds its
266
+ // own clock, and the double timestamp ate half the panel width.
267
+ .map((line) => line.replace(/^\S+\s+\+[\d.]+(?:ms|s|m|h)\s+(INFO|WARN|ERROR)\s+/, (_m, level) => (level === 'INFO' ? '' : `${level} `)));
268
+ } catch {
269
+ return [];
270
+ } finally {
271
+ if (fd !== undefined) closeSync(fd);
272
+ }
273
+ }
@@ -284,3 +284,34 @@ async function waitFor(predicate, timeoutMs = 250) {
284
284
  }
285
285
  assert.fail('Timed out waiting for condition.');
286
286
  }
287
+
288
+ test('readNewTraceLines streams significant trace events with a per-activity cursor', async () => {
289
+ const { readNewTraceLines } = await import('./supervisor.js');
290
+ const { mkdtempSync, writeFileSync, appendFileSync, rmSync } = await import('node:fs');
291
+ const { tmpdir } = await import('node:os');
292
+ const { join } = await import('node:path');
293
+ const workspacePath = mkdtempSync(join(tmpdir(), 'trace-tail-'));
294
+ try {
295
+ const rel = '.wiki/logs/ingest-test.log';
296
+ const abs = join(workspacePath, rel);
297
+ (await import('node:fs')).mkdirSync(join(workspacePath, '.wiki/logs'), { recursive: true });
298
+ writeFileSync(abs, '2026-07-10T13:00:00.000Z +10ms INFO llm:start label=ingest_plan promptChars=42000\n');
299
+ const session = { workspacePath };
300
+
301
+ const first = readNewTraceLines(session, 'k1', rel);
302
+ assert.deepEqual(first, ['llm:start label=ingest_plan promptChars=42000']);
303
+
304
+ // No new content → nothing re-emitted.
305
+ assert.deepEqual(readNewTraceLines(session, 'k1', rel), []);
306
+
307
+ appendFileSync(abs, '2026-07-10T13:01:00.000Z +70s INFO noise: irrelevant heartbeat\n2026-07-10T13:02:00.000Z +130s WARN embedding:neutralized-input status=413\n');
308
+ const second = readNewTraceLines(session, 'k1', rel);
309
+ assert.deepEqual(second, ['WARN embedding:neutralized-input status=413']);
310
+
311
+ // Path traversal attempts are refused.
312
+ assert.deepEqual(readNewTraceLines(session, 'k1', '../../etc/passwd'), []);
313
+ assert.deepEqual(readNewTraceLines(session, 'k1', '/etc/passwd'), []);
314
+ } finally {
315
+ rmSync(workspacePath, { recursive: true, force: true });
316
+ }
317
+ });
@@ -272,9 +272,16 @@ function renderMarkdownLines(lines: Array<{ text: string; isCode: boolean }>, ro
272
272
  for (let index = 0; index < lines.length; index += 1) {
273
273
  const { text, isCode } = lines[index];
274
274
  if (isCode) {
275
- output.push(...wrapLine(text || ' ', columns).map((piece) => ({
276
- segments: [{ text: piece || ' ', color: '#D6DEE8', bg: '#1A2235' }],
275
+ const blockStarts = index === 0 || !lines[index - 1].isCode;
276
+ const blockEnds = index === lines.length - 1 || !lines[index + 1].isCode;
277
+ // Blank line before/after the block + 2-space inner padding: fenced
278
+ // blocks used to render as a dense background slab glued to the text.
279
+ if (blockStarts) output.push({ segments: [{ text: ' ', color: '#D6DEE8' }] });
280
+ const innerWidth = Math.max(8, columns - 4);
281
+ output.push(...wrapLine(text || ' ', innerWidth).map((piece) => ({
282
+ segments: [{ text: ` ${(piece || ' ').padEnd(innerWidth)} `, color: '#D6DEE8', bg: '#1A2235' }],
277
283
  })));
284
+ if (blockEnds) output.push({ segments: [{ text: ' ', color: '#D6DEE8' }] });
278
285
  continue;
279
286
  }
280
287
 
package/src/shell/repl.js CHANGED
@@ -18,7 +18,15 @@ import { listWorkspaces } from '../core/workspaces.js';
18
18
  import { fetchRuntimeState, postRuntimeApprove, postRuntimeCancel, postRuntimeControl, postRuntimeRun, postRuntimeShutdown, streamRuntimeEvents } from '../runtime/client.js';
19
19
  import { versionWithBuild } from '../core/buildInfo.js';
20
20
 
21
- marked.use(markedTerminal());
21
+ // Code blocks: marked-terminal's default paints a dense background block
22
+ // glued to the surrounding text. Indent the content, keep a plain style and
23
+ // guarantee a blank line before/after so fenced blocks breathe in the chat.
24
+ const CODE_INDENT = ' ';
25
+ const CODE_TINT = '\u001b[38;5;152m'; // soft blue-grey, readable on dark bg
26
+ const CODE_RESET = '\u001b[0m';
27
+ marked.use(markedTerminal({
28
+ code: (code) => `\n${String(code).split('\n').map((line) => `${CODE_INDENT}${CODE_TINT}${line}${CODE_RESET}`).join('\n')}\n`,
29
+ }));
22
30
  // marked-terminal's text renderer extracts token.text (raw string) instead of
23
31
  // calling parseInline(token.tokens), so inline Markdown inside list items is
24
32
  // silently dropped. Patch it to call parseInline when tokens are available.
@@ -75,7 +83,7 @@ const COMMAND_COMPLETION_DESCRIPTIONS = {
75
83
  '/chat': 'Switch free text to direct LLM chat without tools.',
76
84
  '/agent': 'Switch free text to the LangGraph agent with tools.',
77
85
  '/openui': 'Open the workspace web UI in the browser.',
78
- '/run': 'Inspect, cancel, or kill runtime runs.',
86
+ '/run': 'Inspect, cancel, kill runtime runs, or start a capability run.',
79
87
  '/approve': 'Approve a pending runtime run or tool.',
80
88
  };
81
89
 
@@ -141,7 +149,7 @@ export function createSession() {
141
149
  wikircConfig: null,
142
150
  language: null,
143
151
  mcp: null,
144
- commands: ['help', 'version', 'exit', 'workspace', 'new', 'use', 'config', 'status', 'services', 'start', 'stop', 'logs', 'mcp', 'wiki', 'skills', 'upload', 'uploads', 'clear', 'chat', 'agent', 'openui', 'run', 'queue', 'approve'],
152
+ commands: ['help', 'version', 'exit', 'workspace', 'new', 'use', 'config', 'status', 'services', 'start', 'stop', 'logs', 'mcp', 'wiki', 'skills', 'upload', 'uploads', 'clear', 'chat', 'agent', 'openui', 'run', 'cancel', 'queue', 'approve'],
145
153
  chatMode: true,
146
154
  llm: null,
147
155
  activities: {},
@@ -246,7 +254,7 @@ function completionValuesFor(parts, inputBuffer, session) {
246
254
  if (command === '/upload' && parts[1] === 'convert' && tokenIndex === 2) return ['pending'];
247
255
  if (command === '/uploads' && tokenIndex === 1) return ['clean', 'list'];
248
256
  if (command === '/uploads' && previousToken === 'clean') return ['--older-than'];
249
- if (command === '/run' && tokenIndex === 1) return ['status', 'cancel', 'kill'];
257
+ if (command === '/run' && tokenIndex === 1) return ['status', 'cancel', 'kill', 'capability'];
250
258
  if (command === '/queue' && tokenIndex === 1) return ['cancel', 'clear'];
251
259
  if (command === '/queue' && previousToken === 'cancel') {
252
260
  return (session.jobQueue ?? [])
package/src/shell/tui.tsx CHANGED
@@ -116,6 +116,29 @@ function App(props: {
116
116
  // sync at every call site.
117
117
  const [screen, setScreen] = createSignal<'startup' | 'setup' | 'main'>('startup');
118
118
  let ctrlCTimer: ReturnType<typeof setTimeout> | null = null;
119
+ let exiting = false;
120
+ // Single exit path: the owned-runtime shutdown MUST happen here, on the
121
+ // user's actual exit gesture. render() resolves at MOUNT, so code placed
122
+ // after `await runOpenTuiShell(...)` runs while the shell is still on
123
+ // screen — 0.12.9 shipped that and killed the runtime mid-session.
124
+ const exitShell = () => {
125
+ if (exiting) return;
126
+ exiting = true;
127
+ void (async () => {
128
+ const messages: string[] = [];
129
+ try {
130
+ if (props.runtime?.url) {
131
+ const { shutdownOwnedRuntime } = await import('../runtime/lifecycle.js');
132
+ await shutdownOwnedRuntime(props.runtime, { log: (message: string) => { messages.push(message); } });
133
+ }
134
+ } catch {
135
+ // Best effort: never block the exit on runtime cleanup.
136
+ }
137
+ renderer.destroy();
138
+ // Print AFTER destroy so the note survives on the restored terminal.
139
+ for (const message of messages) console.log(`[wiki-manager] ${message}`);
140
+ })();
141
+ };
119
142
  let copyHintTimer: ReturnType<typeof setTimeout> | null = null;
120
143
  let selectionCopyTimer: ReturnType<typeof setTimeout> | null = null;
121
144
  let startupKeyboardEventId = 0;
@@ -141,7 +164,7 @@ function App(props: {
141
164
  return;
142
165
  }
143
166
  void state.submitInput(value).then((result) => {
144
- if (result?.exit) renderer.destroy();
167
+ if (result?.exit) exitShell();
145
168
  });
146
169
  };
147
170
 
@@ -263,7 +286,7 @@ function App(props: {
263
286
  return;
264
287
  }
265
288
  if (exitHint()) {
266
- renderer.destroy();
289
+ exitShell();
267
290
  return;
268
291
  }
269
292
  setExitHint(true);
@@ -378,7 +401,7 @@ function App(props: {
378
401
  height={dimensions().height}
379
402
  keyboardEvent={startupKeyboardEvent()}
380
403
  onSelect={openAction}
381
- onQuit={() => renderer.destroy()}
404
+ onQuit={() => exitShell()}
382
405
  />
383
406
  </Show>
384
407
  );