@namzu/cli 11.0.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/README.md +390 -46
  2. package/dist/cli.d.ts +15 -0
  3. package/dist/cli.d.ts.map +1 -1
  4. package/dist/cli.js +132 -4
  5. package/dist/cli.js.map +1 -1
  6. package/dist/commands/acp.d.ts +29 -0
  7. package/dist/commands/acp.d.ts.map +1 -0
  8. package/dist/commands/acp.js +156 -0
  9. package/dist/commands/acp.js.map +1 -0
  10. package/dist/commands/doctor.d.ts +12 -1
  11. package/dist/commands/doctor.d.ts.map +1 -1
  12. package/dist/commands/doctor.js +18 -2
  13. package/dist/commands/doctor.js.map +1 -1
  14. package/dist/commands/drain.d.ts +2 -2
  15. package/dist/commands/drain.d.ts.map +1 -1
  16. package/dist/commands/drain.js +31 -8
  17. package/dist/commands/drain.js.map +1 -1
  18. package/dist/commands/run-flags.d.ts.map +1 -1
  19. package/dist/commands/run-flags.js +33 -0
  20. package/dist/commands/run-flags.js.map +1 -1
  21. package/dist/commands/run-stream.d.ts +3 -3
  22. package/dist/commands/run-stream.d.ts.map +1 -1
  23. package/dist/commands/run-stream.js +27 -10
  24. package/dist/commands/run-stream.js.map +1 -1
  25. package/dist/commands/run.d.ts.map +1 -1
  26. package/dist/commands/run.js +50 -4
  27. package/dist/commands/run.js.map +1 -1
  28. package/dist/commands/types.d.ts +10 -0
  29. package/dist/commands/types.d.ts.map +1 -1
  30. package/dist/config/load.d.ts +64 -0
  31. package/dist/config/load.d.ts.map +1 -1
  32. package/dist/config/load.js +137 -20
  33. package/dist/config/load.js.map +1 -1
  34. package/dist/config/schema.d.ts +51 -0
  35. package/dist/config/schema.d.ts.map +1 -1
  36. package/dist/config/schema.js.map +1 -1
  37. package/dist/context/capabilities.d.ts +88 -0
  38. package/dist/context/capabilities.d.ts.map +1 -0
  39. package/dist/context/capabilities.js +160 -0
  40. package/dist/context/capabilities.js.map +1 -0
  41. package/dist/context/sandbox.d.ts +13 -0
  42. package/dist/context/sandbox.d.ts.map +1 -1
  43. package/dist/context/sandbox.js +15 -0
  44. package/dist/context/sandbox.js.map +1 -1
  45. package/dist/doctor/checks/index.d.ts +6 -1
  46. package/dist/doctor/checks/index.d.ts.map +1 -1
  47. package/dist/doctor/checks/index.js +45 -2
  48. package/dist/doctor/checks/index.js.map +1 -1
  49. package/dist/doctor/checks/invariants.d.ts +23 -0
  50. package/dist/doctor/checks/invariants.d.ts.map +1 -0
  51. package/dist/doctor/checks/invariants.js +79 -0
  52. package/dist/doctor/checks/invariants.js.map +1 -0
  53. package/dist/doctor/checks/logging.d.ts +10 -0
  54. package/dist/doctor/checks/logging.d.ts.map +1 -0
  55. package/dist/doctor/checks/logging.js +69 -0
  56. package/dist/doctor/checks/logging.js.map +1 -0
  57. package/dist/doctor/checks/session-export.d.ts +21 -0
  58. package/dist/doctor/checks/session-export.d.ts.map +1 -0
  59. package/dist/doctor/checks/session-export.js +62 -0
  60. package/dist/doctor/checks/session-export.js.map +1 -0
  61. package/dist/doctor/checks/telemetry.d.ts +15 -28
  62. package/dist/doctor/checks/telemetry.d.ts.map +1 -1
  63. package/dist/doctor/checks/telemetry.js +34 -54
  64. package/dist/doctor/checks/telemetry.js.map +1 -1
  65. package/dist/doctor/checks/vault.d.ts +4 -20
  66. package/dist/doctor/checks/vault.d.ts.map +1 -1
  67. package/dist/doctor/checks/vault.js +58 -19
  68. package/dist/doctor/checks/vault.js.map +1 -1
  69. package/dist/doctor/registry.d.ts.map +1 -1
  70. package/dist/doctor/registry.js +13 -6
  71. package/dist/doctor/registry.js.map +1 -1
  72. package/dist/index.d.ts +3 -2
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +3 -2
  75. package/dist/index.js.map +1 -1
  76. package/dist/integrations/files/attachment-store.d.ts +44 -0
  77. package/dist/integrations/files/attachment-store.d.ts.map +1 -0
  78. package/dist/integrations/files/attachment-store.js +86 -0
  79. package/dist/integrations/files/attachment-store.js.map +1 -0
  80. package/dist/integrations/providers/credential-provider.d.ts +39 -0
  81. package/dist/integrations/providers/credential-provider.d.ts.map +1 -0
  82. package/dist/integrations/providers/credential-provider.js +74 -0
  83. package/dist/integrations/providers/credential-provider.js.map +1 -0
  84. package/dist/integrations/providers/discover.d.ts.map +1 -1
  85. package/dist/integrations/providers/discover.js +15 -3
  86. package/dist/integrations/providers/discover.js.map +1 -1
  87. package/dist/integrations/providers/oauth.d.ts.map +1 -1
  88. package/dist/integrations/providers/oauth.js +24 -1
  89. package/dist/integrations/providers/oauth.js.map +1 -1
  90. package/dist/integrations/sessions/store.d.ts +2 -2
  91. package/dist/integrations/sessions/store.d.ts.map +1 -1
  92. package/dist/integrations/sessions/store.js +21 -9
  93. package/dist/integrations/sessions/store.js.map +1 -1
  94. package/dist/integrations/subagents/runtime.d.ts +3 -3
  95. package/dist/integrations/subagents/runtime.d.ts.map +1 -1
  96. package/dist/integrations/subagents/runtime.js +11 -11
  97. package/dist/integrations/subagents/runtime.js.map +1 -1
  98. package/dist/integrations/telemetry/session-export.d.ts +99 -0
  99. package/dist/integrations/telemetry/session-export.d.ts.map +1 -0
  100. package/dist/integrations/telemetry/session-export.js +119 -0
  101. package/dist/integrations/telemetry/session-export.js.map +1 -0
  102. package/dist/logging.d.ts +79 -0
  103. package/dist/logging.d.ts.map +1 -0
  104. package/dist/logging.js +67 -0
  105. package/dist/logging.js.map +1 -0
  106. package/dist/permissions/rules.d.ts +3 -3
  107. package/dist/permissions/rules.d.ts.map +1 -1
  108. package/dist/permissions/rules.js +1 -1
  109. package/dist/permissions/rules.js.map +1 -1
  110. package/dist/skills/store.d.ts +1 -1
  111. package/dist/skills/store.js +1 -1
  112. package/dist/tui/App.d.ts.map +1 -1
  113. package/dist/tui/App.js +92 -6
  114. package/dist/tui/App.js.map +1 -1
  115. package/dist/tui/StatusBar.d.ts +1 -1
  116. package/dist/tui/StatusBar.d.ts.map +1 -1
  117. package/dist/tui/agent.d.ts +54 -24
  118. package/dist/tui/agent.d.ts.map +1 -1
  119. package/dist/tui/agent.js +327 -124
  120. package/dist/tui/agent.js.map +1 -1
  121. package/dist/tui/index.d.ts.map +1 -1
  122. package/dist/tui/index.js +17 -6
  123. package/dist/tui/index.js.map +1 -1
  124. package/dist/tui/log-pane.d.ts +48 -0
  125. package/dist/tui/log-pane.d.ts.map +1 -0
  126. package/dist/tui/log-pane.js +106 -0
  127. package/dist/tui/log-pane.js.map +1 -0
  128. package/dist/tui/slashCommands.d.ts +134 -11
  129. package/dist/tui/slashCommands.d.ts.map +1 -1
  130. package/dist/tui/slashCommands.js +169 -15
  131. package/dist/tui/slashCommands.js.map +1 -1
  132. package/dist/tui/types.d.ts +11 -2
  133. package/dist/tui/types.d.ts.map +1 -1
  134. package/package.json +8 -6
package/dist/tui/agent.js CHANGED
@@ -20,11 +20,12 @@
20
20
  * `emptySession()` whose `send()` yields a single error event so the UI
21
21
  * renders an actionable hint rather than crashing.
22
22
  */
23
- import { DiskMemoryStore, DiskTaskStore, ProviderRegistry, SearchToolsTool, ToolRegistry, buildMemoryTools, createMemoryPromoter, getBuiltinTools, getRootLogger, isTrustedReadOnly, query, resumeRun, } from '@namzu/sdk';
23
+ import { BOOT_EVENT_NAMES, DiskMemoryStore, DiskTaskStore, EVENT_NAME_ATTRIBUTE, ProviderRegistry, SearchToolsTool, ToolRegistry, asProjectId, asRunId, asSessionId, asTenantId, asTopicId, buildMemoryTools, createMemoryPromoter, createToolPresenter, genericLabel, getBuiltinTools, getRootLogger, isTrustedReadOnly, query, resumeRun, } from '@namzu/sdk';
24
24
  import { join } from 'node:path';
25
+ import { probeCapabilities } from '../context/capabilities.js';
25
26
  import { composeEnvironmentPrompt, readEnvironmentFacts } from '../context/environment.js';
26
27
  import { loadProjectInstructions } from '../context/project.js';
27
- import { resolveSandbox } from '../context/sandbox.js';
28
+ import { resolveSandbox, sandboxResolvedSeverity, } from '../context/sandbox.js';
28
29
  import { connectMcpServers, } from '../integrations/mcp/servers.js';
29
30
  import { PROVIDER_REGISTRY, chainCapabilityDisagreements, chainPositionName, describeAcceptedMismatch, describeCapabilityRefusal, discoverProviders, ensureFreshAnthropicToken, ensureRegistered, findDetected, isAnthropicOAuthToken, isRegistered, missingCredentialMessage, primaryProvider, readPreferences, readSubscriptionCredential, resolveChainCapabilities, unresolvedMembers, unsupportedProviderMessage, } from '../integrations/providers/index.js';
30
31
  import { createSubagentRuntime } from '../integrations/subagents/runtime.js';
@@ -186,6 +187,21 @@ export async function createAgentSession(prefs, detected, options = {}) {
186
187
  // credential rejection for a provider they never configured.
187
188
  const fallbackPlan = planFallbacks(prefs.providers, detected);
188
189
  const model = primary.model ?? entry.defaultModel;
190
+ // One line naming the head and how many declared fallbacks are usable —
191
+ // `fallbackPlan.notices` already carries WHY each skipped member did (no
192
+ // credential, unknown id, not registered); this promotes that same
193
+ // information from a UI notice string to a boot record rather than
194
+ // computing it a second time.
195
+ getRootLogger().info('provider chain resolved', {
196
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.PROVIDER_RESOLVED,
197
+ 'gen_ai.request.model': model,
198
+ 'namzu.provider.id': primary.id,
199
+ 'namzu.provider.chain_length': prefs.providers.length,
200
+ 'namzu.provider.skipped_count': fallbackPlan.notices.length,
201
+ });
202
+ for (const notice of fallbackPlan.notices) {
203
+ getRootLogger().warn(notice, { [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.PROVIDER_RESOLVED });
204
+ }
189
205
  let provider;
190
206
  try {
191
207
  provider = constructProvider(primary.id, det, model);
@@ -225,8 +241,13 @@ export async function createAgentSession(prefs, detected, options = {}) {
225
241
  try {
226
242
  provider = constructProvider('anthropic', { ...det, apiKey: fresh }, model);
227
243
  }
228
- catch {
244
+ catch (err) {
229
245
  // Keep the previous client; the turn may still 401 but won't crash.
246
+ // Silent until now — a client rebuild failing after a token refresh
247
+ // had no trace anywhere, so the first sign of it was a live 401 an
248
+ // operator had no way to connect back to "the refresh happened, the
249
+ // rebuild didn't."
250
+ getRootLogger().warn('provider client rebuild after token refresh failed', exceptionAttributes(err));
230
251
  }
231
252
  };
232
253
  // Read ONCE, here, rather than per turn the way memory is.
@@ -242,8 +263,43 @@ export async function createAgentSession(prefs, detected, options = {}) {
242
263
  // Before the registry, because a `requireIsolation` this machine cannot
243
264
  // meet throws here — and failing before the session is built is the
244
265
  // difference between "namzu refused to start" and a half-constructed
245
- // session reporting a tool error on the first command.
246
- const sandbox = resolveSandbox(getRootLogger(), options.sandbox);
266
+ // session reporting a tool error on the first command. Ordering is load
267
+ // bearing on BOTH sides of this block: `resolveSandbox` stays BEFORE
268
+ // `buildToolRegistry` below (unchanged), and the emit two statements down
269
+ // stays strictly AFTER `resolveSandbox` returns — logging "attempting to
270
+ // resolve the sandbox" ahead of the call would say nothing `resolveSandbox`
271
+ // itself doesn't already say better, for a narrative that is supposed to
272
+ // report facts, not attempts.
273
+ let sandbox;
274
+ try {
275
+ sandbox = resolveSandbox(getRootLogger(), options.sandbox);
276
+ }
277
+ catch (err) {
278
+ // The one refusal in this function that does not go through
279
+ // `emptySession(...)`: `resolveSandbox` THROWS rather than degrading
280
+ // when `sandbox.requireIsolation` names a control this host cannot
281
+ // meet (see that function's own doc comment), and a caller half-built
282
+ // at that point has nothing to return a session FROM. Logged here,
283
+ // then re-thrown unchanged — `runCli`'s own top-level catch (already
284
+ // in place, untouched by this change) is what turns the throw into a
285
+ // non-zero exit; this is only responsible for the record existing
286
+ // before that happens.
287
+ getRootLogger().error(err instanceof Error ? err.message : String(err), {
288
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.BOOT_REFUSED,
289
+ 'namzu.refusal.kind': 'environment',
290
+ });
291
+ throw err;
292
+ }
293
+ // AFTER resolveSandbox returns — the honest report of what THIS run got,
294
+ // never what was attempted. `unconfined` decides the severity: per the
295
+ // design, this is "the single highest-value line in the whole design,
296
+ // today computed and thrown away" — an operator reading default `info`
297
+ // output must see it specifically when nothing is enforced, not only
298
+ // under `--verbose`.
299
+ getRootLogger()[sandboxResolvedSeverity(sandbox)](sandbox.notice, {
300
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.SANDBOX_RESOLVED,
301
+ 'namzu.sandbox.unconfined': sandbox.unconfined,
302
+ });
247
303
  const { registry, memoryStore } = buildToolRegistry(cwd);
248
304
  // External tool servers, before the roster is counted, so `toolNames` and
249
305
  // the `/tools` list a user reads include what they configured. Connecting
@@ -251,6 +307,39 @@ export async function createAgentSession(prefs, detected, options = {}) {
251
307
  const mcp = await connectMcpServers(options.mcpServers, { cwd });
252
308
  if (mcp.tools.length > 0)
253
309
  registry.register([...mcp.tools]);
310
+ // Connectors are the one discovery source THIS function performs —
311
+ // plugins and skills are loaded elsewhere (`run-flags.ts`'s
312
+ // `loadSkillsContext`, per turn) and neither is wired to the boot path
313
+ // yet, so a fabricated "plugins 0 · skills 0" here would claim a
314
+ // measurement that was never taken. This reports only what was.
315
+ getRootLogger().info('discovery complete', {
316
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.DISCOVERY_COMPLETED,
317
+ 'namzu.discovery.kind': 'connector',
318
+ 'namzu.discovery.count': mcp.connected.length,
319
+ 'namzu.discovery.tool_count': mcp.tools.length,
320
+ 'namzu.discovery.failed_count': mcp.failed.length,
321
+ });
322
+ for (const server of mcp.connected) {
323
+ getRootLogger().debug('connector discovered', {
324
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.DISCOVERY_COMPLETED,
325
+ 'namzu.discovery.kind': 'connector',
326
+ 'namzu.connector.name': server.name,
327
+ 'namzu.connector.tool_count': server.toolCount,
328
+ });
329
+ }
330
+ for (const server of mcp.failed) {
331
+ getRootLogger().debug('connector failed to connect', {
332
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.DISCOVERY_COMPLETED,
333
+ 'namzu.discovery.kind': 'connector',
334
+ 'namzu.connector.name': server.name,
335
+ });
336
+ }
337
+ // Detected once per session, not per capability check an operator might
338
+ // separately run via `namzu doctor` — same probe, same three-state
339
+ // answer, so the boot narrative and the doctor report can never disagree
340
+ // about whether @namzu/sandbox loaded.
341
+ const capabilities = await probeCapabilities();
342
+ logCapabilities(capabilities);
254
343
  // This session passes a `taskStore` to query() below, which registers the
255
344
  // task tools deferred — so `search_tools` has something to find here.
256
345
  registry.register([SearchToolsTool]);
@@ -311,7 +400,7 @@ export async function createAgentSession(prefs, detected, options = {}) {
311
400
  verificationGate: gateFor(options.rules),
312
401
  onEvent: (e) => {
313
402
  if (e.type === 'tool_executing') {
314
- childSteps.push(`${e.toolName}(${summarizeToolInput(e.input)})`);
403
+ childSteps.push(`${e.toolName}(${genericLabel(e.input)})`);
315
404
  }
316
405
  },
317
406
  });
@@ -319,8 +408,12 @@ export async function createAgentSession(prefs, detected, options = {}) {
319
408
  subagentGateway = sub.gateway;
320
409
  allowedAgentIds = sub.allowedAgentIds;
321
410
  }
322
- catch {
323
- // Sub-agents unavailable this session — non-fatal.
411
+ catch (err) {
412
+ // Sub-agents unavailable this session — non-fatal: `allowedAgentIds`
413
+ // stays empty and the chat still works. Silent until now, which was
414
+ // the wrong kind of non-fatal — an operator who expected delegation
415
+ // and got none had nothing on stderr to say why.
416
+ getRootLogger().warn('sub-agent runtime unavailable this session', exceptionAttributes(err));
324
417
  }
325
418
  // Task store → query registers task_create / task_update / task_list as
326
419
  // DEFERRED tools and emits task_created/task_updated, so the agent can track
@@ -336,7 +429,7 @@ export async function createAgentSession(prefs, detected, options = {}) {
336
429
  // changes is that asking again later gets a later answer.
337
430
  const taskStore = new DiskTaskStore({
338
431
  baseDir: join(cwd, '.namzu'),
339
- defaultRunId: 'run_namzu-cli',
432
+ defaultRunId: asRunId('run_namzu-cli'),
340
433
  tenantId: scope.tenantId,
341
434
  });
342
435
  // Persists across turns: once the user picks "approve all", later tool
@@ -353,6 +446,16 @@ export async function createAgentSession(prefs, detected, options = {}) {
353
446
  // settle — is what this repository has been doing all along by accident.
354
447
  // A run that learned nothing still writes nothing.
355
448
  const promoteMemory = createMemoryPromoter({ store: memoryStore });
449
+ // The one terminal POSITIVE event on this path, emitted exactly once —
450
+ // every early return above goes through `emptySession`, which emits
451
+ // `namzu.boot.refused` instead, and the `resolveSandbox` throw path above
452
+ // emits its own `boot.refused` and never reaches this line at all. No
453
+ // boolean readiness field anywhere in the record: systemd's own `READY=1`
454
+ // has no `READY=0` counterpart, for the same reason — a field that CAN
455
+ // say "not ready" is a field some unaudited path can wrongly set true.
456
+ getRootLogger().info('agent session ready', {
457
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.BOOT_READY,
458
+ });
356
459
  return {
357
460
  hasProvider: true,
358
461
  providerSummary: entry.label,
@@ -432,6 +535,7 @@ export async function createAgentSession(prefs, detected, options = {}) {
432
535
  messages,
433
536
  opts,
434
537
  taskGateway: subagentGateway,
538
+ onRunEvent: options.onRunEvent,
435
539
  childSteps,
436
540
  ...(sandbox.provider ? { sandboxProvider: sandbox.provider } : {}),
437
541
  });
@@ -475,6 +579,10 @@ export async function createAgentSession(prefs, detected, options = {}) {
475
579
  // No `onPermission`: there is nobody at a drainer's terminal, so a
476
580
  // prompt would block the pass forever on a run nobody is watching.
477
581
  // The gate's deny rules still apply.
582
+ // One presenter for the whole stream, built from the registry this
583
+ // scope already holds. It was the absence of the registry HERE that
584
+ // forced presentation to be name matching: `toAgentEvent` was pure
585
+ // over a `RunEvent` and could not ask a tool anything.
478
586
  resumeHandler: makeResumeHandler(approval, undefined, options.permissionMode, (n, i) => isPromptExempt(registry, n, i)),
479
587
  ...(signal ? { signal } : {}),
480
588
  // Attribution comes from the ENTRY, not from this session: the run
@@ -483,11 +591,11 @@ export async function createAgentSession(prefs, detected, options = {}) {
483
591
  tenantId: entry.tenantId,
484
592
  projectId: entry.projectId,
485
593
  sessionId: entry.sessionId,
486
- // …except the thread, which no checkpoint records — see
594
+ // …except the topic, which no checkpoint records — see
487
595
  // `RunStateScope`. This one is the drainer's, and honestly so:
488
596
  // supplied here rather than pretended to have been recovered.
489
- threadId: scope.threadId,
490
- scope: { ...entry, threadId: scope.threadId },
597
+ topicId: scope.topicId,
598
+ scope: { ...entry, topicId: scope.topicId },
491
599
  checkpointStore,
492
600
  ...(claimFence !== undefined ? { claimFence } : {}),
493
601
  });
@@ -693,11 +801,15 @@ export async function listProviderModels(id, det) {
693
801
  /** One scope per launched TUI session; runId is minted fresh per turn by the SDK. */
694
802
  function mintScope() {
695
803
  const suffix = `tui-${Date.now().toString(36)}`;
804
+ // Through the constructors rather than as four bare template literals.
805
+ // One suffix shared by four ids is exactly the shape a typo hides in —
806
+ // `top_` and `tnt_` differ by two characters, and the types accept either
807
+ // spelling for either field while they are still structural.
696
808
  return {
697
- sessionId: `ses_${suffix}`,
698
- threadId: `thd_${suffix}`,
699
- projectId: `prj_${suffix}`,
700
- tenantId: `tnt_${suffix}`,
809
+ sessionId: asSessionId(`ses_${suffix}`),
810
+ topicId: asTopicId(`top_${suffix}`),
811
+ projectId: asProjectId(`prj_${suffix}`),
812
+ tenantId: asTenantId(`tnt_${suffix}`),
701
813
  };
702
814
  }
703
815
  // Pre-execution safety gate: hard-deny catastrophic shell patterns
@@ -737,6 +849,11 @@ function gateFor(rules) {
737
849
  // a number here would fix one window across every model the CLI can talk to.
738
850
  const COMPACTION_CONFIG = {
739
851
  strategy: 'structured',
852
+ // On, and this is the CLI making a choice rather than taking a default.
853
+ // A local session's transcript is the only record of what was compacted
854
+ // away, and `<cwd>/.namzu` is the operator's own disk — the size trade
855
+ // this costs is theirs to see and theirs to turn off.
856
+ recordShedHistory: true,
740
857
  triggerThreshold: 0.7,
741
858
  resetThreshold: 0.4,
742
859
  keepRecentMessages: 6,
@@ -760,8 +877,13 @@ const COMPACTION_CONFIG = {
760
877
  maxCharsPerRequirement: 300,
761
878
  maxCharsPerTask: 400,
762
879
  };
763
- async function* runTurn({ provider, fallbackProviders, model, tools, scope, workingDirectory, rules, permissionMode, reviewAnswer, maxAnswerReviews, promoteMemory, approval, taskStore, systemPrompt, messages, opts, taskGateway, childSteps, sandboxProvider, }) {
880
+ async function* runTurn({ provider, fallbackProviders, model, tools, scope, workingDirectory, rules, permissionMode, reviewAnswer, maxAnswerReviews, promoteMemory, approval, taskStore, systemPrompt, messages, opts, taskGateway, childSteps, sandboxProvider, onRunEvent, }) {
764
881
  const signal = opts?.signal;
882
+ // One presenter for the whole stream, built from the registry this scope
883
+ // already holds. Its absence HERE is what forced presentation to be name
884
+ // matching in the first place: `toAgentEvent` is pure over a `RunEvent`
885
+ // and could not ask a tool anything, so the host guessed from the name.
886
+ const presenter = createToolPresenter(tools);
765
887
  try {
766
888
  const events = query({
767
889
  provider,
@@ -809,16 +931,22 @@ async function* runTurn({ provider, fallbackProviders, model, tools, scope, work
809
931
  // The exemption reads `tools` at decision time, so it sees the task
810
932
  // tools `query()` registers deferred below and any tool server that
811
933
  // connected after this session was built.
812
- resumeHandler: makeResumeHandler(approval, opts?.onPermission, permissionMode, (name, input) => isPromptExempt(tools, name, input)),
934
+ resumeHandler: makeResumeHandler(approval, opts?.onPermission, permissionMode, (name, input) => isPromptExempt(tools, name, input), presenter),
813
935
  signal,
814
936
  ...scope,
815
937
  });
816
938
  for await (const event of events) {
939
+ // Before the abort check and before `toAgentEvent`: a session
940
+ // cancelled mid-turn still produced the events up to that point, and
941
+ // they are the interesting ones. Every event, not just the ones the
942
+ // TUI renders — an export that only saw what the screen showed would
943
+ // be a recording of the interface rather than of the session.
944
+ onRunEvent?.(event);
817
945
  if (signal?.aborted) {
818
946
  yield { kind: 'error', message: 'aborted' };
819
947
  return;
820
948
  }
821
- const mapped = toAgentEvent(event);
949
+ const mapped = toAgentEvent(event, presenter);
822
950
  if (!mapped)
823
951
  continue;
824
952
  // On an `Agent` delegation finishing, attach the sub-agent's tool
@@ -857,7 +985,14 @@ export function makeResumeHandler(approval, onPermission, mode = onPermission ?
857
985
  * handler stays testable without a registry — and so the answer comes from
858
986
  * the live roster at the moment of the call.
859
987
  */
860
- exempt = () => false) {
988
+ exempt = () => false,
989
+ /**
990
+ * How a prompted call is described. Injected for the same reason
991
+ * `exempt` is — this handler is unit-tested without a registry — and
992
+ * defaulted to the generic view so a caller that has no registry still
993
+ * gets the label the tool's arguments imply, rather than nothing.
994
+ */
995
+ presenter = GENERIC_PRESENTER) {
861
996
  return async (request) => {
862
997
  if (request.type !== 'tool_review') {
863
998
  return request.type === 'plan_approval' ? { action: 'approve_plan' } : { action: 'continue' };
@@ -883,9 +1018,11 @@ exempt = () => false) {
883
1018
  toolCalls: request.toolCalls.map((tc) => ({
884
1019
  id: tc.id,
885
1020
  name: tc.name,
886
- summary: summarizeToolInput(tc.input),
1021
+ ...(() => {
1022
+ const view = presenter.presentCall(tc.name, tc.input);
1023
+ return { summary: viewToSummary(view), preview: viewToPreview(view) };
1024
+ })(),
887
1025
  isDestructive: tc.isDestructive,
888
- preview: previewToolInput(tc.name, tc.input),
889
1026
  })),
890
1027
  });
891
1028
  switch (decision.kind) {
@@ -983,17 +1120,24 @@ export function batchNeedsPrompt(toolCalls, exempt) {
983
1120
  * `null` for events the chat surface doesn't render (iteration markers,
984
1121
  * token usage, checkpoints, plan/task lifecycle, …). Pure — unit-tested.
985
1122
  */
986
- export function toAgentEvent(event) {
1123
+ export function toAgentEvent(event, presenter) {
987
1124
  switch (event.type) {
988
1125
  case 'text_delta':
989
- return { kind: 'delta', text: event.text };
1126
+ return {
1127
+ kind: 'delta',
1128
+ text: event.text,
1129
+ ...(event.messageId ? { messageId: event.messageId } : {}),
1130
+ ...(event.runId ? { runId: event.runId } : {}),
1131
+ };
990
1132
  case 'tool_executing':
991
1133
  return {
992
1134
  kind: 'tool-start',
993
1135
  toolUseId: event.toolUseId,
994
1136
  toolName: event.toolName,
995
- summary: summarizeToolInput(event.input),
996
- detail: toolStartDetail(event.toolName, event.input),
1137
+ ...(() => {
1138
+ const view = presenter.presentCall(event.toolName, event.input);
1139
+ return { summary: viewToSummary(view), detail: viewToLines(view) };
1140
+ })(),
997
1141
  };
998
1142
  case 'tool_completed':
999
1143
  return {
@@ -1002,7 +1146,16 @@ export function toAgentEvent(event) {
1002
1146
  toolName: event.toolName,
1003
1147
  isError: event.isError,
1004
1148
  summary: firstLine(event.result),
1005
- detail: toolEndDetail(event.toolName, event.result),
1149
+ // `tool_completed` carries no input, so the presenter gets an
1150
+ // empty one. A tool whose result rendering depends on its
1151
+ // arguments would need the executing event's input threaded
1152
+ // through; none does yet, and inventing the plumbing for a
1153
+ // caller that does not exist is the declaration this repo
1154
+ // keeps deleting.
1155
+ detail: viewToLines(presenter.presentResult(event.toolName, {}, {
1156
+ success: !event.isError,
1157
+ output: event.result,
1158
+ })),
1006
1159
  };
1007
1160
  case 'token_usage_updated':
1008
1161
  // The context figures are forwarded, not recomputed. They were
@@ -1053,6 +1206,18 @@ export function toAgentEvent(event) {
1053
1206
  };
1054
1207
  case 'compaction_completed':
1055
1208
  return { kind: 'context', text: describeCompaction(event), shed: true };
1209
+ case 'compaction_tool_results_cleared':
1210
+ // `shed: true` on both branches: the tool-result bodies are gone
1211
+ // either way. `reliefWasEnough: false` additionally means a
1212
+ // summarization followed, and the reader will see its own line —
1213
+ // so this one says what IT cost rather than claiming the total.
1214
+ return {
1215
+ kind: 'context',
1216
+ text: `cleared ${event.clearedCount} oversized tool result${event.clearedCount === 1 ? '' : 's'}` +
1217
+ ` (~${event.reclaimedTokens.toLocaleString()} tokens)` +
1218
+ (event.reliefWasEnough ? '' : ' — not enough, compacting'),
1219
+ shed: true,
1220
+ };
1056
1221
  case 'compaction_failed':
1057
1222
  return { kind: 'context', text: describeCompactionFailure(event), shed: false };
1058
1223
  default:
@@ -1148,65 +1313,97 @@ function asTree(steps) {
1148
1313
  return steps.map((s, i) => `${i === steps.length - 1 ? '└─' : '├─'} ${s}`);
1149
1314
  }
1150
1315
  /** Short, human-readable one-liner for a tool call (e.g. `ls -la`, path). */
1151
- function summarizeToolInput(input) {
1152
- if (input && typeof input === 'object') {
1153
- const obj = input;
1154
- const pick = (k) => (typeof obj[k] === 'string' ? obj[k] : undefined);
1155
- const primary = pick('command') ??
1156
- pick('path') ??
1157
- pick('file_path') ??
1158
- pick('pattern') ??
1159
- pick('query') ??
1160
- // Last, so it only speaks for a tool none of the above describe.
1161
- // Those tools were falling through to a truncated `JSON.stringify`
1162
- // — which is how `Agent` came to show a blob of its own arguments
1163
- // while requiring the model to write a label nothing then read.
1164
- // Every input named `description` in this tree is a short
1165
- // human-facing label, so it is a summary by construction.
1166
- pick('description');
1167
- if (primary)
1168
- return truncate(primary, 120);
1316
+ /**
1317
+ * The presenter a caller with no registry gets.
1318
+ *
1319
+ * `makeResumeHandler` is unit-tested without one, and a handler that
1320
+ * described every prompted call as an empty string would make those tests
1321
+ * pass while telling a real user nothing. This is the same fallback the
1322
+ * registry-backed presenter uses when a tool has no opinion, which is what
1323
+ * the four deleted functions did for every tool.
1324
+ */
1325
+ const GENERIC_PRESENTER = {
1326
+ presentCall: (_name, input) => ({ kind: 'generic', label: genericLabel(input) }),
1327
+ presentResult: (_name, _input, result) => ({ kind: 'terminal', output: result.output ?? '' }),
1328
+ };
1329
+ /**
1330
+ * The one presentation function this host keeps.
1331
+ *
1332
+ * There used to be four, and each switched on a lowercased tool NAME:
1333
+ * `name === 'write'` and `name === 'edit'` got a diff, everything else got
1334
+ * a truncated string. So a tool this host had never heard of — an MCP
1335
+ * server's, a plugin's — could not get a diff no matter what it did.
1336
+ *
1337
+ * The tool now says which of three shapes it wants, and this decides what
1338
+ * that looks like in a terminal. Clamping and the `STDOUT:`/`STDERR:`
1339
+ * cleanup stay here on purpose: how many rows fit and how a shell labels
1340
+ * its streams are properties of this surface, not of the tool.
1341
+ */
1342
+ export function viewToLines(view) {
1343
+ switch (view.kind) {
1344
+ case 'generic':
1345
+ // The label IS the summary row. Repeating it underneath adds a
1346
+ // line that says what the line above it already said.
1347
+ return undefined;
1348
+ case 'diff': {
1349
+ // An empty `before` is a whole-file write, not a patch: there is
1350
+ // nothing to contrast against, so the content reads plainly. `edit`
1351
+ // never produces this — it returns no view at all for an insert,
1352
+ // rather than claim the file was empty.
1353
+ if (view.before === '') {
1354
+ const lines = clampLines(view.after);
1355
+ return lines.length > 0 ? lines : undefined;
1356
+ }
1357
+ const lines = [];
1358
+ for (const line of clampLines(view.before))
1359
+ lines.push(`- ${line}`);
1360
+ for (const line of clampLines(view.after))
1361
+ lines.push(`+ ${line}`);
1362
+ return lines.length > 0 ? lines : undefined;
1363
+ }
1364
+ case 'terminal': {
1365
+ if (view.output.trim().length === 0)
1366
+ return undefined;
1367
+ const lines = resultToLines(view.output);
1368
+ // A single short line is already the summary — no need to repeat it.
1369
+ return lines.length <= 1 ? undefined : lines;
1370
+ }
1169
1371
  }
1170
- if (typeof input === 'string')
1171
- return truncate(input, 120);
1172
- return truncate(JSON.stringify(input ?? {}), 120);
1173
1372
  }
1174
- function truncate(value, max) {
1175
- const oneLine = value.replace(/\s+/g, ' ');
1176
- return oneLine.length > max ? `${oneLine.slice(0, max - 1)}…` : oneLine;
1373
+ /** The `⏺` row: one line naming what the call is about. */
1374
+ export function viewToSummary(view) {
1375
+ switch (view.kind) {
1376
+ case 'generic':
1377
+ return truncate(view.label, 120);
1378
+ case 'diff':
1379
+ return truncate(view.path ?? view.after.split('\n')[0] ?? '', 120);
1380
+ case 'terminal':
1381
+ return truncate(view.command ?? view.output.split('\n')[0] ?? '', 120);
1382
+ }
1177
1383
  }
1178
1384
  /**
1179
- * Multi-line preview of a mutating tool's effect, shown in the permission
1180
- * overlay so the user approves with sight of what changes. `write` shows
1181
- * the leading content lines; `edit` shows a minimal -old / +new diff;
1182
- * everything else has no preview (the one-line summary suffices). Pure —
1183
- * unit-tested.
1385
+ * The permission overlay's preview: the same shapes, cut shorter.
1386
+ *
1387
+ * A user approving a call needs enough to recognise it, not the whole
1388
+ * file — the transcript shows that once it has run.
1184
1389
  */
1185
- export function previewToolInput(toolName, input) {
1186
- if (!input || typeof input !== 'object')
1390
+ export function viewToPreview(view) {
1391
+ if (view.kind !== 'diff')
1187
1392
  return undefined;
1188
- const obj = input;
1189
- const str = (k) => (typeof obj[k] === 'string' ? obj[k] : undefined);
1190
- const name = toolName.toLowerCase();
1191
- if (name === 'write') {
1192
- const content = str('content');
1193
- if (content !== undefined)
1194
- return previewLines(content, 8);
1195
- }
1196
- if (name === 'edit') {
1197
- const oldString = str('old_string');
1198
- const newString = str('new_string');
1199
- const lines = [];
1200
- if (oldString)
1201
- for (const line of previewLines(oldString, 4))
1202
- lines.push(`- ${line}`);
1203
- if (newString)
1204
- for (const line of previewLines(newString, 4))
1205
- lines.push(`+ ${line}`);
1206
- if (lines.length > 0)
1207
- return lines;
1393
+ if (view.before === '') {
1394
+ const lines = previewLines(view.after, 8);
1395
+ return lines.length > 0 ? lines : undefined;
1208
1396
  }
1209
- return undefined;
1397
+ const lines = [];
1398
+ for (const line of previewLines(view.before, 4))
1399
+ lines.push(`- ${line}`);
1400
+ for (const line of previewLines(view.after, 4))
1401
+ lines.push(`+ ${line}`);
1402
+ return lines.length > 0 ? lines : undefined;
1403
+ }
1404
+ function truncate(value, max) {
1405
+ const oneLine = value.replace(/\s+/g, ' ');
1406
+ return oneLine.length > max ? `${oneLine.slice(0, max - 1)}…` : oneLine;
1210
1407
  }
1211
1408
  function previewLines(value, max) {
1212
1409
  const lines = value.split('\n');
@@ -1220,35 +1417,6 @@ function clampLines(value) {
1220
1417
  const lines = value.replace(/\s+$/, '').split('\n');
1221
1418
  return lines.length > MAX_DETAIL_LINES ? lines.slice(0, MAX_DETAIL_LINES) : lines;
1222
1419
  }
1223
- /**
1224
- * Diff / content shown under a tool CALL (`⏺`): an `edit` renders a
1225
- * `- old` / `+ new` diff, a `write` renders the content. Other tools show
1226
- * nothing at call time (their output appears under the result instead).
1227
- */
1228
- export function toolStartDetail(toolName, input) {
1229
- if (!input || typeof input !== 'object')
1230
- return undefined;
1231
- const obj = input;
1232
- const str = (k) => (typeof obj[k] === 'string' ? obj[k] : undefined);
1233
- const name = toolName.toLowerCase();
1234
- if (name === 'write') {
1235
- const content = str('content');
1236
- return content !== undefined ? clampLines(content) : undefined;
1237
- }
1238
- if (name === 'edit') {
1239
- const oldString = str('old_string');
1240
- const newString = str('new_string');
1241
- const lines = [];
1242
- if (oldString)
1243
- for (const line of clampLines(oldString))
1244
- lines.push(`- ${line}`);
1245
- if (newString)
1246
- for (const line of clampLines(newString))
1247
- lines.push(`+ ${line}`);
1248
- return lines.length > 0 ? lines : undefined;
1249
- }
1250
- return undefined;
1251
- }
1252
1420
  /** Parse a string as a JSON object, or null. Connector tools return JSON. */
1253
1421
  function parseJsonObject(s) {
1254
1422
  const t = s.trim();
@@ -1295,22 +1463,6 @@ function resultToLines(result) {
1295
1463
  return clampLines(JSON.stringify(obj, null, 2));
1296
1464
  return clampLines(cleanToolText(result.trim()));
1297
1465
  }
1298
- /**
1299
- * Output shown under a tool RESULT (`⎿`). For `edit`/`write` the diff was
1300
- * already shown at call time, so the result stays a one-line confirmation;
1301
- * every other tool (read/bash/grep/…) shows its captured output here — JSON
1302
- * results are pretty-printed / unwrapped so they don't read as a raw blob.
1303
- */
1304
- export function toolEndDetail(toolName, result) {
1305
- const name = toolName.toLowerCase();
1306
- if (name === 'edit' || name === 'write')
1307
- return undefined;
1308
- if (result.trim().length === 0)
1309
- return undefined;
1310
- const lines = resultToLines(result);
1311
- // A single short line is already the summary — no need to repeat it.
1312
- return lines.length <= 1 ? undefined : lines;
1313
- }
1314
1466
  /** Concise one-line summary of a tool result for the `⎿` line. */
1315
1467
  function firstLine(result) {
1316
1468
  const payload = payloadString(result);
@@ -1329,7 +1481,58 @@ function firstLine(result) {
1329
1481
  const cleaned = cleanToolText(result.trim());
1330
1482
  return truncate(cleaned.split('\n').find((l) => l.trim().length > 0) ?? '', 120);
1331
1483
  }
1484
+ function exceptionAttributes(err) {
1485
+ const error = err instanceof Error ? err : new Error(String(err));
1486
+ return {
1487
+ 'exception.type': error.constructor?.name ?? 'Error',
1488
+ 'exception.message': error.message,
1489
+ };
1490
+ }
1491
+ /**
1492
+ * `namzu.capability.detected` per package at `debug`, one aggregate summary
1493
+ * at `info` (the design's §6.3 `capability sandbox yes · files yes · …`
1494
+ * line), and `namzu.capability.broken` at `error` for any package that
1495
+ * resolved and failed to load. Never refuses the boot: nothing in
1496
+ * `NamzuCliConfig` marks a capability required yet, so `broken` here is
1497
+ * always the "not required by config" case §6.5 describes — an optional
1498
+ * capability's failure degrades what this line SAYS, never whether
1499
+ * `namzu.boot.ready` fires.
1500
+ */
1501
+ function logCapabilities(probes) {
1502
+ const log = getRootLogger();
1503
+ const summary = probes
1504
+ .map((p) => `${p.specifier.split('/').pop()} ${p.state === 'present' ? 'yes' : 'no'}`)
1505
+ .join(' · ');
1506
+ log.info(summary, { [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.CAPABILITY_DETECTED });
1507
+ for (const probe of probes) {
1508
+ if (probe.state === 'broken') {
1509
+ log.error('Capability probe failed to load', {
1510
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.CAPABILITY_BROKEN,
1511
+ 'namzu.capability.name': probe.specifier,
1512
+ ...exceptionAttributes(probe.error),
1513
+ });
1514
+ continue;
1515
+ }
1516
+ log.debug('Capability probe completed', {
1517
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.CAPABILITY_DETECTED,
1518
+ 'namzu.capability.name': probe.specifier,
1519
+ 'namzu.capability.state': probe.state,
1520
+ 'namzu.capability.present': probe.state === 'present',
1521
+ ...(probe.state === 'present' ? { 'namzu.capability.version': probe.version } : {}),
1522
+ });
1523
+ }
1524
+ }
1332
1525
  function emptySession(errorHint, errorKind = 'environment') {
1526
+ // Every path into this function is a boot refusal — `createAgentSession`
1527
+ // is the whole extent of the session-construction half of the boot
1528
+ // narrative, and every one of its early returns comes through here. One
1529
+ // emission point instead of five call-site ones is what keeps that true
1530
+ // instead of "true until the sixth `emptySession(...)` someone adds
1531
+ // forgets it."
1532
+ getRootLogger().error(errorHint, {
1533
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.BOOT_REFUSED,
1534
+ 'namzu.refusal.kind': errorKind,
1535
+ });
1333
1536
  return {
1334
1537
  hasProvider: false,
1335
1538
  errorKind,