@namzu/cli 10.0.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/README.md +390 -46
  2. package/dist/cli.d.ts +15 -0
  3. package/dist/cli.d.ts.map +1 -1
  4. package/dist/cli.js +133 -4
  5. package/dist/cli.js.map +1 -1
  6. package/dist/commands/acp.d.ts +29 -0
  7. package/dist/commands/acp.d.ts.map +1 -0
  8. package/dist/commands/acp.js +156 -0
  9. package/dist/commands/acp.js.map +1 -0
  10. package/dist/commands/doctor.d.ts +12 -1
  11. package/dist/commands/doctor.d.ts.map +1 -1
  12. package/dist/commands/doctor.js +18 -2
  13. package/dist/commands/doctor.js.map +1 -1
  14. package/dist/commands/drain.d.ts +2 -2
  15. package/dist/commands/drain.d.ts.map +1 -1
  16. package/dist/commands/drain.js +31 -8
  17. package/dist/commands/drain.js.map +1 -1
  18. package/dist/commands/run-flags.d.ts.map +1 -1
  19. package/dist/commands/run-flags.js +33 -0
  20. package/dist/commands/run-flags.js.map +1 -1
  21. package/dist/commands/run-stream.d.ts +3 -3
  22. package/dist/commands/run-stream.d.ts.map +1 -1
  23. package/dist/commands/run-stream.js +27 -10
  24. package/dist/commands/run-stream.js.map +1 -1
  25. package/dist/commands/run.d.ts.map +1 -1
  26. package/dist/commands/run.js +50 -4
  27. package/dist/commands/run.js.map +1 -1
  28. package/dist/commands/types.d.ts +10 -0
  29. package/dist/commands/types.d.ts.map +1 -1
  30. package/dist/config/load.d.ts +64 -0
  31. package/dist/config/load.d.ts.map +1 -1
  32. package/dist/config/load.js +156 -20
  33. package/dist/config/load.js.map +1 -1
  34. package/dist/config/schema.d.ts +88 -0
  35. package/dist/config/schema.d.ts.map +1 -1
  36. package/dist/config/schema.js.map +1 -1
  37. package/dist/context/capabilities.d.ts +88 -0
  38. package/dist/context/capabilities.d.ts.map +1 -0
  39. package/dist/context/capabilities.js +160 -0
  40. package/dist/context/capabilities.js.map +1 -0
  41. package/dist/context/sandbox.d.ts +54 -0
  42. package/dist/context/sandbox.d.ts.map +1 -0
  43. package/dist/context/sandbox.js +61 -0
  44. package/dist/context/sandbox.js.map +1 -0
  45. package/dist/doctor/checks/index.d.ts +6 -1
  46. package/dist/doctor/checks/index.d.ts.map +1 -1
  47. package/dist/doctor/checks/index.js +45 -2
  48. package/dist/doctor/checks/index.js.map +1 -1
  49. package/dist/doctor/checks/invariants.d.ts +23 -0
  50. package/dist/doctor/checks/invariants.d.ts.map +1 -0
  51. package/dist/doctor/checks/invariants.js +79 -0
  52. package/dist/doctor/checks/invariants.js.map +1 -0
  53. package/dist/doctor/checks/logging.d.ts +10 -0
  54. package/dist/doctor/checks/logging.d.ts.map +1 -0
  55. package/dist/doctor/checks/logging.js +69 -0
  56. package/dist/doctor/checks/logging.js.map +1 -0
  57. package/dist/doctor/checks/session-export.d.ts +21 -0
  58. package/dist/doctor/checks/session-export.d.ts.map +1 -0
  59. package/dist/doctor/checks/session-export.js +62 -0
  60. package/dist/doctor/checks/session-export.js.map +1 -0
  61. package/dist/doctor/checks/telemetry.d.ts +15 -28
  62. package/dist/doctor/checks/telemetry.d.ts.map +1 -1
  63. package/dist/doctor/checks/telemetry.js +34 -54
  64. package/dist/doctor/checks/telemetry.js.map +1 -1
  65. package/dist/doctor/checks/vault.d.ts +4 -20
  66. package/dist/doctor/checks/vault.d.ts.map +1 -1
  67. package/dist/doctor/checks/vault.js +58 -19
  68. package/dist/doctor/checks/vault.js.map +1 -1
  69. package/dist/doctor/registry.d.ts.map +1 -1
  70. package/dist/doctor/registry.js +13 -6
  71. package/dist/doctor/registry.js.map +1 -1
  72. package/dist/index.d.ts +3 -2
  73. package/dist/index.d.ts.map +1 -1
  74. package/dist/index.js +3 -2
  75. package/dist/index.js.map +1 -1
  76. package/dist/integrations/files/attachment-store.d.ts +44 -0
  77. package/dist/integrations/files/attachment-store.d.ts.map +1 -0
  78. package/dist/integrations/files/attachment-store.js +86 -0
  79. package/dist/integrations/files/attachment-store.js.map +1 -0
  80. package/dist/integrations/providers/credential-provider.d.ts +39 -0
  81. package/dist/integrations/providers/credential-provider.d.ts.map +1 -0
  82. package/dist/integrations/providers/credential-provider.js +74 -0
  83. package/dist/integrations/providers/credential-provider.js.map +1 -0
  84. package/dist/integrations/providers/discover.d.ts.map +1 -1
  85. package/dist/integrations/providers/discover.js +15 -3
  86. package/dist/integrations/providers/discover.js.map +1 -1
  87. package/dist/integrations/providers/oauth.d.ts.map +1 -1
  88. package/dist/integrations/providers/oauth.js +24 -1
  89. package/dist/integrations/providers/oauth.js.map +1 -1
  90. package/dist/integrations/sessions/store.d.ts +2 -2
  91. package/dist/integrations/sessions/store.d.ts.map +1 -1
  92. package/dist/integrations/sessions/store.js +21 -9
  93. package/dist/integrations/sessions/store.js.map +1 -1
  94. package/dist/integrations/subagents/runtime.d.ts +3 -3
  95. package/dist/integrations/subagents/runtime.d.ts.map +1 -1
  96. package/dist/integrations/subagents/runtime.js +11 -11
  97. package/dist/integrations/subagents/runtime.js.map +1 -1
  98. package/dist/integrations/telemetry/session-export.d.ts +99 -0
  99. package/dist/integrations/telemetry/session-export.d.ts.map +1 -0
  100. package/dist/integrations/telemetry/session-export.js +119 -0
  101. package/dist/integrations/telemetry/session-export.js.map +1 -0
  102. package/dist/logging.d.ts +79 -0
  103. package/dist/logging.d.ts.map +1 -0
  104. package/dist/logging.js +67 -0
  105. package/dist/logging.js.map +1 -0
  106. package/dist/permissions/rules.d.ts +3 -3
  107. package/dist/permissions/rules.d.ts.map +1 -1
  108. package/dist/permissions/rules.js +1 -1
  109. package/dist/permissions/rules.js.map +1 -1
  110. package/dist/skills/store.d.ts +1 -1
  111. package/dist/skills/store.js +1 -1
  112. package/dist/tui/App.d.ts.map +1 -1
  113. package/dist/tui/App.js +93 -6
  114. package/dist/tui/App.js.map +1 -1
  115. package/dist/tui/StatusBar.d.ts +1 -1
  116. package/dist/tui/StatusBar.d.ts.map +1 -1
  117. package/dist/tui/agent.d.ts +65 -24
  118. package/dist/tui/agent.d.ts.map +1 -1
  119. package/dist/tui/agent.js +337 -122
  120. package/dist/tui/agent.js.map +1 -1
  121. package/dist/tui/index.d.ts.map +1 -1
  122. package/dist/tui/index.js +17 -6
  123. package/dist/tui/index.js.map +1 -1
  124. package/dist/tui/log-pane.d.ts +48 -0
  125. package/dist/tui/log-pane.d.ts.map +1 -0
  126. package/dist/tui/log-pane.js +106 -0
  127. package/dist/tui/log-pane.js.map +1 -0
  128. package/dist/tui/slashCommands.d.ts +134 -11
  129. package/dist/tui/slashCommands.d.ts.map +1 -1
  130. package/dist/tui/slashCommands.js +169 -15
  131. package/dist/tui/slashCommands.js.map +1 -1
  132. package/dist/tui/types.d.ts +18 -2
  133. package/dist/tui/types.d.ts.map +1 -1
  134. package/package.json +8 -6
package/dist/tui/agent.js CHANGED
@@ -20,10 +20,12 @@
20
20
  * `emptySession()` whose `send()` yields a single error event so the UI
21
21
  * renders an actionable hint rather than crashing.
22
22
  */
23
- import { DiskMemoryStore, DiskTaskStore, ProviderRegistry, SearchToolsTool, ToolRegistry, buildMemoryTools, createMemoryPromoter, getBuiltinTools, query, resumeRun, } from '@namzu/sdk';
23
+ import { BOOT_EVENT_NAMES, DiskMemoryStore, DiskTaskStore, EVENT_NAME_ATTRIBUTE, ProviderRegistry, SearchToolsTool, ToolRegistry, asProjectId, asRunId, asSessionId, asTenantId, asTopicId, buildMemoryTools, createMemoryPromoter, createToolPresenter, genericLabel, getBuiltinTools, getRootLogger, isTrustedReadOnly, query, resumeRun, } from '@namzu/sdk';
24
24
  import { join } from 'node:path';
25
+ import { probeCapabilities } from '../context/capabilities.js';
25
26
  import { composeEnvironmentPrompt, readEnvironmentFacts } from '../context/environment.js';
26
27
  import { loadProjectInstructions } from '../context/project.js';
28
+ import { resolveSandbox, sandboxResolvedSeverity, } from '../context/sandbox.js';
27
29
  import { connectMcpServers, } from '../integrations/mcp/servers.js';
28
30
  import { PROVIDER_REGISTRY, chainCapabilityDisagreements, chainPositionName, describeAcceptedMismatch, describeCapabilityRefusal, discoverProviders, ensureFreshAnthropicToken, ensureRegistered, findDetected, isAnthropicOAuthToken, isRegistered, missingCredentialMessage, primaryProvider, readPreferences, readSubscriptionCredential, resolveChainCapabilities, unresolvedMembers, unsupportedProviderMessage, } from '../integrations/providers/index.js';
29
31
  import { createSubagentRuntime } from '../integrations/subagents/runtime.js';
@@ -185,6 +187,21 @@ export async function createAgentSession(prefs, detected, options = {}) {
185
187
  // credential rejection for a provider they never configured.
186
188
  const fallbackPlan = planFallbacks(prefs.providers, detected);
187
189
  const model = primary.model ?? entry.defaultModel;
190
+ // One line naming the head and how many declared fallbacks are usable —
191
+ // `fallbackPlan.notices` already carries WHY each skipped member did (no
192
+ // credential, unknown id, not registered); this promotes that same
193
+ // information from a UI notice string to a boot record rather than
194
+ // computing it a second time.
195
+ getRootLogger().info('provider chain resolved', {
196
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.PROVIDER_RESOLVED,
197
+ 'gen_ai.request.model': model,
198
+ 'namzu.provider.id': primary.id,
199
+ 'namzu.provider.chain_length': prefs.providers.length,
200
+ 'namzu.provider.skipped_count': fallbackPlan.notices.length,
201
+ });
202
+ for (const notice of fallbackPlan.notices) {
203
+ getRootLogger().warn(notice, { [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.PROVIDER_RESOLVED });
204
+ }
188
205
  let provider;
189
206
  try {
190
207
  provider = constructProvider(primary.id, det, model);
@@ -224,8 +241,13 @@ export async function createAgentSession(prefs, detected, options = {}) {
224
241
  try {
225
242
  provider = constructProvider('anthropic', { ...det, apiKey: fresh }, model);
226
243
  }
227
- catch {
244
+ catch (err) {
228
245
  // Keep the previous client; the turn may still 401 but won't crash.
246
+ // Silent until now — a client rebuild failing after a token refresh
247
+ // had no trace anywhere, so the first sign of it was a live 401 an
248
+ // operator had no way to connect back to "the refresh happened, the
249
+ // rebuild didn't."
250
+ getRootLogger().warn('provider client rebuild after token refresh failed', exceptionAttributes(err));
229
251
  }
230
252
  };
231
253
  // Read ONCE, here, rather than per turn the way memory is.
@@ -238,6 +260,46 @@ export async function createAgentSession(prefs, detected, options = {}) {
238
260
  // turn would make the line a surface prints at connect time a claim about
239
261
  // the past. An edited file takes effect on the next session.
240
262
  const projectInstructions = loadProjectInstructions(cwd);
263
+ // Before the registry, because a `requireIsolation` this machine cannot
264
+ // meet throws here — and failing before the session is built is the
265
+ // difference between "namzu refused to start" and a half-constructed
266
+ // session reporting a tool error on the first command. Ordering is load
267
+ // bearing on BOTH sides of this block: `resolveSandbox` stays BEFORE
268
+ // `buildToolRegistry` below (unchanged), and the emit two statements down
269
+ // stays strictly AFTER `resolveSandbox` returns — logging "attempting to
270
+ // resolve the sandbox" ahead of the call would say nothing `resolveSandbox`
271
+ // itself doesn't already say better, for a narrative that is supposed to
272
+ // report facts, not attempts.
273
+ let sandbox;
274
+ try {
275
+ sandbox = resolveSandbox(getRootLogger(), options.sandbox);
276
+ }
277
+ catch (err) {
278
+ // The one refusal in this function that does not go through
279
+ // `emptySession(...)`: `resolveSandbox` THROWS rather than degrading
280
+ // when `sandbox.requireIsolation` names a control this host cannot
281
+ // meet (see that function's own doc comment), and a caller half-built
282
+ // at that point has nothing to return a session FROM. Logged here,
283
+ // then re-thrown unchanged — `runCli`'s own top-level catch (already
284
+ // in place, untouched by this change) is what turns the throw into a
285
+ // non-zero exit; this is only responsible for the record existing
286
+ // before that happens.
287
+ getRootLogger().error(err instanceof Error ? err.message : String(err), {
288
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.BOOT_REFUSED,
289
+ 'namzu.refusal.kind': 'environment',
290
+ });
291
+ throw err;
292
+ }
293
+ // AFTER resolveSandbox returns — the honest report of what THIS run got,
294
+ // never what was attempted. `unconfined` decides the severity: per the
295
+ // design, this is "the single highest-value line in the whole design,
296
+ // today computed and thrown away" — an operator reading default `info`
297
+ // output must see it specifically when nothing is enforced, not only
298
+ // under `--verbose`.
299
+ getRootLogger()[sandboxResolvedSeverity(sandbox)](sandbox.notice, {
300
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.SANDBOX_RESOLVED,
301
+ 'namzu.sandbox.unconfined': sandbox.unconfined,
302
+ });
241
303
  const { registry, memoryStore } = buildToolRegistry(cwd);
242
304
  // External tool servers, before the roster is counted, so `toolNames` and
243
305
  // the `/tools` list a user reads include what they configured. Connecting
@@ -245,6 +307,39 @@ export async function createAgentSession(prefs, detected, options = {}) {
245
307
  const mcp = await connectMcpServers(options.mcpServers, { cwd });
246
308
  if (mcp.tools.length > 0)
247
309
  registry.register([...mcp.tools]);
310
+ // Connectors are the one discovery source THIS function performs —
311
+ // plugins and skills are loaded elsewhere (`run-flags.ts`'s
312
+ // `loadSkillsContext`, per turn) and neither is wired to the boot path
313
+ // yet, so a fabricated "plugins 0 · skills 0" here would claim a
314
+ // measurement that was never taken. This reports only what was.
315
+ getRootLogger().info('discovery complete', {
316
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.DISCOVERY_COMPLETED,
317
+ 'namzu.discovery.kind': 'connector',
318
+ 'namzu.discovery.count': mcp.connected.length,
319
+ 'namzu.discovery.tool_count': mcp.tools.length,
320
+ 'namzu.discovery.failed_count': mcp.failed.length,
321
+ });
322
+ for (const server of mcp.connected) {
323
+ getRootLogger().debug('connector discovered', {
324
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.DISCOVERY_COMPLETED,
325
+ 'namzu.discovery.kind': 'connector',
326
+ 'namzu.connector.name': server.name,
327
+ 'namzu.connector.tool_count': server.toolCount,
328
+ });
329
+ }
330
+ for (const server of mcp.failed) {
331
+ getRootLogger().debug('connector failed to connect', {
332
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.DISCOVERY_COMPLETED,
333
+ 'namzu.discovery.kind': 'connector',
334
+ 'namzu.connector.name': server.name,
335
+ });
336
+ }
337
+ // Detected once per session, not per capability check an operator might
338
+ // separately run via `namzu doctor` — same probe, same three-state
339
+ // answer, so the boot narrative and the doctor report can never disagree
340
+ // about whether @namzu/sandbox loaded.
341
+ const capabilities = await probeCapabilities();
342
+ logCapabilities(capabilities);
248
343
  // This session passes a `taskStore` to query() below, which registers the
249
344
  // task tools deferred — so `search_tools` has something to find here.
250
345
  registry.register([SearchToolsTool]);
@@ -305,7 +400,7 @@ export async function createAgentSession(prefs, detected, options = {}) {
305
400
  verificationGate: gateFor(options.rules),
306
401
  onEvent: (e) => {
307
402
  if (e.type === 'tool_executing') {
308
- childSteps.push(`${e.toolName}(${summarizeToolInput(e.input)})`);
403
+ childSteps.push(`${e.toolName}(${genericLabel(e.input)})`);
309
404
  }
310
405
  },
311
406
  });
@@ -313,8 +408,12 @@ export async function createAgentSession(prefs, detected, options = {}) {
313
408
  subagentGateway = sub.gateway;
314
409
  allowedAgentIds = sub.allowedAgentIds;
315
410
  }
316
- catch {
317
- // Sub-agents unavailable this session — non-fatal.
411
+ catch (err) {
412
+ // Sub-agents unavailable this session — non-fatal: `allowedAgentIds`
413
+ // stays empty and the chat still works. Silent until now, which was
414
+ // the wrong kind of non-fatal — an operator who expected delegation
415
+ // and got none had nothing on stderr to say why.
416
+ getRootLogger().warn('sub-agent runtime unavailable this session', exceptionAttributes(err));
318
417
  }
319
418
  // Task store → query registers task_create / task_update / task_list as
320
419
  // DEFERRED tools and emits task_created/task_updated, so the agent can track
@@ -330,7 +429,7 @@ export async function createAgentSession(prefs, detected, options = {}) {
330
429
  // changes is that asking again later gets a later answer.
331
430
  const taskStore = new DiskTaskStore({
332
431
  baseDir: join(cwd, '.namzu'),
333
- defaultRunId: 'run_namzu-cli',
432
+ defaultRunId: asRunId('run_namzu-cli'),
334
433
  tenantId: scope.tenantId,
335
434
  });
336
435
  // Persists across turns: once the user picks "approve all", later tool
@@ -347,6 +446,16 @@ export async function createAgentSession(prefs, detected, options = {}) {
347
446
  // settle — is what this repository has been doing all along by accident.
348
447
  // A run that learned nothing still writes nothing.
349
448
  const promoteMemory = createMemoryPromoter({ store: memoryStore });
449
+ // The one terminal POSITIVE event on this path, emitted exactly once —
450
+ // every early return above goes through `emptySession`, which emits
451
+ // `namzu.boot.refused` instead, and the `resolveSandbox` throw path above
452
+ // emits its own `boot.refused` and never reaches this line at all. No
453
+ // boolean readiness field anywhere in the record: systemd's own `READY=1`
454
+ // has no `READY=0` counterpart, for the same reason — a field that CAN
455
+ // say "not ready" is a field some unaudited path can wrongly set true.
456
+ getRootLogger().info('agent session ready', {
457
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.BOOT_READY,
458
+ });
350
459
  return {
351
460
  hasProvider: true,
352
461
  providerSummary: entry.label,
@@ -426,7 +535,9 @@ export async function createAgentSession(prefs, detected, options = {}) {
426
535
  messages,
427
536
  opts,
428
537
  taskGateway: subagentGateway,
538
+ onRunEvent: options.onRunEvent,
429
539
  childSteps,
540
+ ...(sandbox.provider ? { sandboxProvider: sandbox.provider } : {}),
430
541
  });
431
542
  },
432
543
  resumeDurable: async ({ entry, checkpointStore, claimFence, signal }) => {
@@ -468,6 +579,10 @@ export async function createAgentSession(prefs, detected, options = {}) {
468
579
  // No `onPermission`: there is nobody at a drainer's terminal, so a
469
580
  // prompt would block the pass forever on a run nobody is watching.
470
581
  // The gate's deny rules still apply.
582
+ // One presenter for the whole stream, built from the registry this
583
+ // scope already holds. It was the absence of the registry HERE that
584
+ // forced presentation to be name matching: `toAgentEvent` was pure
585
+ // over a `RunEvent` and could not ask a tool anything.
471
586
  resumeHandler: makeResumeHandler(approval, undefined, options.permissionMode, (n, i) => isPromptExempt(registry, n, i)),
472
587
  ...(signal ? { signal } : {}),
473
588
  // Attribution comes from the ENTRY, not from this session: the run
@@ -476,11 +591,11 @@ export async function createAgentSession(prefs, detected, options = {}) {
476
591
  tenantId: entry.tenantId,
477
592
  projectId: entry.projectId,
478
593
  sessionId: entry.sessionId,
479
- // …except the thread, which no checkpoint records — see
594
+ // …except the topic, which no checkpoint records — see
480
595
  // `RunStateScope`. This one is the drainer's, and honestly so:
481
596
  // supplied here rather than pretended to have been recovered.
482
- threadId: scope.threadId,
483
- scope: { ...entry, threadId: scope.threadId },
597
+ topicId: scope.topicId,
598
+ scope: { ...entry, topicId: scope.topicId },
484
599
  checkpointStore,
485
600
  ...(claimFence !== undefined ? { claimFence } : {}),
486
601
  });
@@ -686,11 +801,15 @@ export async function listProviderModels(id, det) {
686
801
  /** One scope per launched TUI session; runId is minted fresh per turn by the SDK. */
687
802
  function mintScope() {
688
803
  const suffix = `tui-${Date.now().toString(36)}`;
804
+ // Through the constructors rather than as four bare template literals.
805
+ // One suffix shared by four ids is exactly the shape a typo hides in —
806
+ // `top_` and `tnt_` differ by two characters, and the types accept either
807
+ // spelling for either field while they are still structural.
689
808
  return {
690
- sessionId: `ses_${suffix}`,
691
- threadId: `thd_${suffix}`,
692
- projectId: `prj_${suffix}`,
693
- tenantId: `tnt_${suffix}`,
809
+ sessionId: asSessionId(`ses_${suffix}`),
810
+ topicId: asTopicId(`top_${suffix}`),
811
+ projectId: asProjectId(`prj_${suffix}`),
812
+ tenantId: asTenantId(`tnt_${suffix}`),
694
813
  };
695
814
  }
696
815
  // Pre-execution safety gate: hard-deny catastrophic shell patterns
@@ -730,6 +849,11 @@ function gateFor(rules) {
730
849
  // a number here would fix one window across every model the CLI can talk to.
731
850
  const COMPACTION_CONFIG = {
732
851
  strategy: 'structured',
852
+ // On, and this is the CLI making a choice rather than taking a default.
853
+ // A local session's transcript is the only record of what was compacted
854
+ // away, and `<cwd>/.namzu` is the operator's own disk — the size trade
855
+ // this costs is theirs to see and theirs to turn off.
856
+ recordShedHistory: true,
733
857
  triggerThreshold: 0.7,
734
858
  resetThreshold: 0.4,
735
859
  keepRecentMessages: 6,
@@ -753,8 +877,13 @@ const COMPACTION_CONFIG = {
753
877
  maxCharsPerRequirement: 300,
754
878
  maxCharsPerTask: 400,
755
879
  };
756
- async function* runTurn({ provider, fallbackProviders, model, tools, scope, workingDirectory, rules, permissionMode, reviewAnswer, maxAnswerReviews, promoteMemory, approval, taskStore, systemPrompt, messages, opts, taskGateway, childSteps, }) {
880
+ async function* runTurn({ provider, fallbackProviders, model, tools, scope, workingDirectory, rules, permissionMode, reviewAnswer, maxAnswerReviews, promoteMemory, approval, taskStore, systemPrompt, messages, opts, taskGateway, childSteps, sandboxProvider, onRunEvent, }) {
757
881
  const signal = opts?.signal;
882
+ // One presenter for the whole stream, built from the registry this scope
883
+ // already holds. Its absence HERE is what forced presentation to be name
884
+ // matching in the first place: `toAgentEvent` is pure over a `RunEvent`
885
+ // and could not ask a tool anything, so the host guessed from the name.
886
+ const presenter = createToolPresenter(tools);
758
887
  try {
759
888
  const events = query({
760
889
  provider,
@@ -769,6 +898,7 @@ async function* runTurn({ provider, fallbackProviders, model, tools, scope, work
769
898
  // empty array, so passing it here discarded the operator's rules on the
770
899
  // path that runs every top-level turn. The sub-agent path called
771
900
  // `gateFor` and this one did not.
901
+ ...(sandboxProvider ? { sandboxProvider } : {}),
772
902
  verificationGate: gateFor(rules),
773
903
  compactionConfig: COMPACTION_CONFIG,
774
904
  // The CLI owns its process end to end, so it can safely hand the
@@ -801,16 +931,22 @@ async function* runTurn({ provider, fallbackProviders, model, tools, scope, work
801
931
  // The exemption reads `tools` at decision time, so it sees the task
802
932
  // tools `query()` registers deferred below and any tool server that
803
933
  // connected after this session was built.
804
- resumeHandler: makeResumeHandler(approval, opts?.onPermission, permissionMode, (name, input) => isPromptExempt(tools, name, input)),
934
+ resumeHandler: makeResumeHandler(approval, opts?.onPermission, permissionMode, (name, input) => isPromptExempt(tools, name, input), presenter),
805
935
  signal,
806
936
  ...scope,
807
937
  });
808
938
  for await (const event of events) {
939
+ // Before the abort check and before `toAgentEvent`: a session
940
+ // cancelled mid-turn still produced the events up to that point, and
941
+ // they are the interesting ones. Every event, not just the ones the
942
+ // TUI renders — an export that only saw what the screen showed would
943
+ // be a recording of the interface rather than of the session.
944
+ onRunEvent?.(event);
809
945
  if (signal?.aborted) {
810
946
  yield { kind: 'error', message: 'aborted' };
811
947
  return;
812
948
  }
813
- const mapped = toAgentEvent(event);
949
+ const mapped = toAgentEvent(event, presenter);
814
950
  if (!mapped)
815
951
  continue;
816
952
  // On an `Agent` delegation finishing, attach the sub-agent's tool
@@ -849,7 +985,14 @@ export function makeResumeHandler(approval, onPermission, mode = onPermission ?
849
985
  * handler stays testable without a registry — and so the answer comes from
850
986
  * the live roster at the moment of the call.
851
987
  */
852
- exempt = () => false) {
988
+ exempt = () => false,
989
+ /**
990
+ * How a prompted call is described. Injected for the same reason
991
+ * `exempt` is — this handler is unit-tested without a registry — and
992
+ * defaulted to the generic view so a caller that has no registry still
993
+ * gets the label the tool's arguments imply, rather than nothing.
994
+ */
995
+ presenter = GENERIC_PRESENTER) {
853
996
  return async (request) => {
854
997
  if (request.type !== 'tool_review') {
855
998
  return request.type === 'plan_approval' ? { action: 'approve_plan' } : { action: 'continue' };
@@ -875,9 +1018,11 @@ exempt = () => false) {
875
1018
  toolCalls: request.toolCalls.map((tc) => ({
876
1019
  id: tc.id,
877
1020
  name: tc.name,
878
- summary: summarizeToolInput(tc.input),
1021
+ ...(() => {
1022
+ const view = presenter.presentCall(tc.name, tc.input);
1023
+ return { summary: viewToSummary(view), preview: viewToPreview(view) };
1024
+ })(),
879
1025
  isDestructive: tc.isDestructive,
880
- preview: previewToolInput(tc.name, tc.input),
881
1026
  })),
882
1027
  });
883
1028
  switch (decision.kind) {
@@ -949,7 +1094,11 @@ export function isPromptExempt(registry, name, input) {
949
1094
  if (PROMPT_EXEMPT_WRITES.has(name.toLowerCase()))
950
1095
  return true;
951
1096
  const tool = registry.get(name) ?? registry.get(name.toLowerCase());
952
- return tool?.isReadOnly?.(input) ?? false;
1097
+ // A connected server's own claim about its own tool cannot skip the
1098
+ // prompt. Same predicate the kernel gate and plan mode use -- three
1099
+ // doors, one rule, because fixing two would close the issue and leave
1100
+ // the boundary open.
1101
+ return isTrustedReadOnly(tool, input);
953
1102
  }
954
1103
  /** The exempt roster, sorted, for the surface that has to NAME it. */
955
1104
  export function promptExemptToolNames(registry) {
@@ -971,17 +1120,24 @@ export function batchNeedsPrompt(toolCalls, exempt) {
971
1120
  * `null` for events the chat surface doesn't render (iteration markers,
972
1121
  * token usage, checkpoints, plan/task lifecycle, …). Pure — unit-tested.
973
1122
  */
974
- export function toAgentEvent(event) {
1123
+ export function toAgentEvent(event, presenter) {
975
1124
  switch (event.type) {
976
1125
  case 'text_delta':
977
- return { kind: 'delta', text: event.text };
1126
+ return {
1127
+ kind: 'delta',
1128
+ text: event.text,
1129
+ ...(event.messageId ? { messageId: event.messageId } : {}),
1130
+ ...(event.runId ? { runId: event.runId } : {}),
1131
+ };
978
1132
  case 'tool_executing':
979
1133
  return {
980
1134
  kind: 'tool-start',
981
1135
  toolUseId: event.toolUseId,
982
1136
  toolName: event.toolName,
983
- summary: summarizeToolInput(event.input),
984
- detail: toolStartDetail(event.toolName, event.input),
1137
+ ...(() => {
1138
+ const view = presenter.presentCall(event.toolName, event.input);
1139
+ return { summary: viewToSummary(view), detail: viewToLines(view) };
1140
+ })(),
985
1141
  };
986
1142
  case 'tool_completed':
987
1143
  return {
@@ -990,7 +1146,16 @@ export function toAgentEvent(event) {
990
1146
  toolName: event.toolName,
991
1147
  isError: event.isError,
992
1148
  summary: firstLine(event.result),
993
- detail: toolEndDetail(event.toolName, event.result),
1149
+ // `tool_completed` carries no input, so the presenter gets an
1150
+ // empty one. A tool whose result rendering depends on its
1151
+ // arguments would need the executing event's input threaded
1152
+ // through; none does yet, and inventing the plumbing for a
1153
+ // caller that does not exist is the declaration this repo
1154
+ // keeps deleting.
1155
+ detail: viewToLines(presenter.presentResult(event.toolName, {}, {
1156
+ success: !event.isError,
1157
+ output: event.result,
1158
+ })),
994
1159
  };
995
1160
  case 'token_usage_updated':
996
1161
  // The context figures are forwarded, not recomputed. They were
@@ -1041,6 +1206,18 @@ export function toAgentEvent(event) {
1041
1206
  };
1042
1207
  case 'compaction_completed':
1043
1208
  return { kind: 'context', text: describeCompaction(event), shed: true };
1209
+ case 'compaction_tool_results_cleared':
1210
+ // `shed: true` on both branches: the tool-result bodies are gone
1211
+ // either way. `reliefWasEnough: false` additionally means a
1212
+ // summarization followed, and the reader will see its own line —
1213
+ // so this one says what IT cost rather than claiming the total.
1214
+ return {
1215
+ kind: 'context',
1216
+ text: `cleared ${event.clearedCount} oversized tool result${event.clearedCount === 1 ? '' : 's'}` +
1217
+ ` (~${event.reclaimedTokens.toLocaleString()} tokens)` +
1218
+ (event.reliefWasEnough ? '' : ' — not enough, compacting'),
1219
+ shed: true,
1220
+ };
1044
1221
  case 'compaction_failed':
1045
1222
  return { kind: 'context', text: describeCompactionFailure(event), shed: false };
1046
1223
  default:
@@ -1136,65 +1313,97 @@ function asTree(steps) {
1136
1313
  return steps.map((s, i) => `${i === steps.length - 1 ? '└─' : '├─'} ${s}`);
1137
1314
  }
1138
1315
  /** Short, human-readable one-liner for a tool call (e.g. `ls -la`, path). */
1139
- function summarizeToolInput(input) {
1140
- if (input && typeof input === 'object') {
1141
- const obj = input;
1142
- const pick = (k) => (typeof obj[k] === 'string' ? obj[k] : undefined);
1143
- const primary = pick('command') ??
1144
- pick('path') ??
1145
- pick('file_path') ??
1146
- pick('pattern') ??
1147
- pick('query') ??
1148
- // Last, so it only speaks for a tool none of the above describe.
1149
- // Those tools were falling through to a truncated `JSON.stringify`
1150
- // — which is how `Agent` came to show a blob of its own arguments
1151
- // while requiring the model to write a label nothing then read.
1152
- // Every input named `description` in this tree is a short
1153
- // human-facing label, so it is a summary by construction.
1154
- pick('description');
1155
- if (primary)
1156
- return truncate(primary, 120);
1316
+ /**
1317
+ * The presenter a caller with no registry gets.
1318
+ *
1319
+ * `makeResumeHandler` is unit-tested without one, and a handler that
1320
+ * described every prompted call as an empty string would make those tests
1321
+ * pass while telling a real user nothing. This is the same fallback the
1322
+ * registry-backed presenter uses when a tool has no opinion, which is what
1323
+ * the four deleted functions did for every tool.
1324
+ */
1325
+ const GENERIC_PRESENTER = {
1326
+ presentCall: (_name, input) => ({ kind: 'generic', label: genericLabel(input) }),
1327
+ presentResult: (_name, _input, result) => ({ kind: 'terminal', output: result.output ?? '' }),
1328
+ };
1329
+ /**
1330
+ * The one presentation function this host keeps.
1331
+ *
1332
+ * There used to be four, and each switched on a lowercased tool NAME:
1333
+ * `name === 'write'` and `name === 'edit'` got a diff, everything else got
1334
+ * a truncated string. So a tool this host had never heard of — an MCP
1335
+ * server's, a plugin's — could not get a diff no matter what it did.
1336
+ *
1337
+ * The tool now says which of three shapes it wants, and this decides what
1338
+ * that looks like in a terminal. Clamping and the `STDOUT:`/`STDERR:`
1339
+ * cleanup stay here on purpose: how many rows fit and how a shell labels
1340
+ * its streams are properties of this surface, not of the tool.
1341
+ */
1342
+ export function viewToLines(view) {
1343
+ switch (view.kind) {
1344
+ case 'generic':
1345
+ // The label IS the summary row. Repeating it underneath adds a
1346
+ // line that says what the line above it already said.
1347
+ return undefined;
1348
+ case 'diff': {
1349
+ // An empty `before` is a whole-file write, not a patch: there is
1350
+ // nothing to contrast against, so the content reads plainly. `edit`
1351
+ // never produces this — it returns no view at all for an insert,
1352
+ // rather than claim the file was empty.
1353
+ if (view.before === '') {
1354
+ const lines = clampLines(view.after);
1355
+ return lines.length > 0 ? lines : undefined;
1356
+ }
1357
+ const lines = [];
1358
+ for (const line of clampLines(view.before))
1359
+ lines.push(`- ${line}`);
1360
+ for (const line of clampLines(view.after))
1361
+ lines.push(`+ ${line}`);
1362
+ return lines.length > 0 ? lines : undefined;
1363
+ }
1364
+ case 'terminal': {
1365
+ if (view.output.trim().length === 0)
1366
+ return undefined;
1367
+ const lines = resultToLines(view.output);
1368
+ // A single short line is already the summary — no need to repeat it.
1369
+ return lines.length <= 1 ? undefined : lines;
1370
+ }
1157
1371
  }
1158
- if (typeof input === 'string')
1159
- return truncate(input, 120);
1160
- return truncate(JSON.stringify(input ?? {}), 120);
1161
1372
  }
1162
- function truncate(value, max) {
1163
- const oneLine = value.replace(/\s+/g, ' ');
1164
- return oneLine.length > max ? `${oneLine.slice(0, max - 1)}…` : oneLine;
1373
+ /** The `⏺` row: one line naming what the call is about. */
1374
+ export function viewToSummary(view) {
1375
+ switch (view.kind) {
1376
+ case 'generic':
1377
+ return truncate(view.label, 120);
1378
+ case 'diff':
1379
+ return truncate(view.path ?? view.after.split('\n')[0] ?? '', 120);
1380
+ case 'terminal':
1381
+ return truncate(view.command ?? view.output.split('\n')[0] ?? '', 120);
1382
+ }
1165
1383
  }
1166
1384
  /**
1167
- * Multi-line preview of a mutating tool's effect, shown in the permission
1168
- * overlay so the user approves with sight of what changes. `write` shows
1169
- * the leading content lines; `edit` shows a minimal -old / +new diff;
1170
- * everything else has no preview (the one-line summary suffices). Pure —
1171
- * unit-tested.
1385
+ * The permission overlay's preview: the same shapes, cut shorter.
1386
+ *
1387
+ * A user approving a call needs enough to recognise it, not the whole
1388
+ * file — the transcript shows that once it has run.
1172
1389
  */
1173
- export function previewToolInput(toolName, input) {
1174
- if (!input || typeof input !== 'object')
1390
+ export function viewToPreview(view) {
1391
+ if (view.kind !== 'diff')
1175
1392
  return undefined;
1176
- const obj = input;
1177
- const str = (k) => (typeof obj[k] === 'string' ? obj[k] : undefined);
1178
- const name = toolName.toLowerCase();
1179
- if (name === 'write') {
1180
- const content = str('content');
1181
- if (content !== undefined)
1182
- return previewLines(content, 8);
1183
- }
1184
- if (name === 'edit') {
1185
- const oldString = str('old_string');
1186
- const newString = str('new_string');
1187
- const lines = [];
1188
- if (oldString)
1189
- for (const line of previewLines(oldString, 4))
1190
- lines.push(`- ${line}`);
1191
- if (newString)
1192
- for (const line of previewLines(newString, 4))
1193
- lines.push(`+ ${line}`);
1194
- if (lines.length > 0)
1195
- return lines;
1393
+ if (view.before === '') {
1394
+ const lines = previewLines(view.after, 8);
1395
+ return lines.length > 0 ? lines : undefined;
1196
1396
  }
1197
- return undefined;
1397
+ const lines = [];
1398
+ for (const line of previewLines(view.before, 4))
1399
+ lines.push(`- ${line}`);
1400
+ for (const line of previewLines(view.after, 4))
1401
+ lines.push(`+ ${line}`);
1402
+ return lines.length > 0 ? lines : undefined;
1403
+ }
1404
+ function truncate(value, max) {
1405
+ const oneLine = value.replace(/\s+/g, ' ');
1406
+ return oneLine.length > max ? `${oneLine.slice(0, max - 1)}…` : oneLine;
1198
1407
  }
1199
1408
  function previewLines(value, max) {
1200
1409
  const lines = value.split('\n');
@@ -1208,35 +1417,6 @@ function clampLines(value) {
1208
1417
  const lines = value.replace(/\s+$/, '').split('\n');
1209
1418
  return lines.length > MAX_DETAIL_LINES ? lines.slice(0, MAX_DETAIL_LINES) : lines;
1210
1419
  }
1211
- /**
1212
- * Diff / content shown under a tool CALL (`⏺`): an `edit` renders a
1213
- * `- old` / `+ new` diff, a `write` renders the content. Other tools show
1214
- * nothing at call time (their output appears under the result instead).
1215
- */
1216
- export function toolStartDetail(toolName, input) {
1217
- if (!input || typeof input !== 'object')
1218
- return undefined;
1219
- const obj = input;
1220
- const str = (k) => (typeof obj[k] === 'string' ? obj[k] : undefined);
1221
- const name = toolName.toLowerCase();
1222
- if (name === 'write') {
1223
- const content = str('content');
1224
- return content !== undefined ? clampLines(content) : undefined;
1225
- }
1226
- if (name === 'edit') {
1227
- const oldString = str('old_string');
1228
- const newString = str('new_string');
1229
- const lines = [];
1230
- if (oldString)
1231
- for (const line of clampLines(oldString))
1232
- lines.push(`- ${line}`);
1233
- if (newString)
1234
- for (const line of clampLines(newString))
1235
- lines.push(`+ ${line}`);
1236
- return lines.length > 0 ? lines : undefined;
1237
- }
1238
- return undefined;
1239
- }
1240
1420
  /** Parse a string as a JSON object, or null. Connector tools return JSON. */
1241
1421
  function parseJsonObject(s) {
1242
1422
  const t = s.trim();
@@ -1283,22 +1463,6 @@ function resultToLines(result) {
1283
1463
  return clampLines(JSON.stringify(obj, null, 2));
1284
1464
  return clampLines(cleanToolText(result.trim()));
1285
1465
  }
1286
- /**
1287
- * Output shown under a tool RESULT (`⎿`). For `edit`/`write` the diff was
1288
- * already shown at call time, so the result stays a one-line confirmation;
1289
- * every other tool (read/bash/grep/…) shows its captured output here — JSON
1290
- * results are pretty-printed / unwrapped so they don't read as a raw blob.
1291
- */
1292
- export function toolEndDetail(toolName, result) {
1293
- const name = toolName.toLowerCase();
1294
- if (name === 'edit' || name === 'write')
1295
- return undefined;
1296
- if (result.trim().length === 0)
1297
- return undefined;
1298
- const lines = resultToLines(result);
1299
- // A single short line is already the summary — no need to repeat it.
1300
- return lines.length <= 1 ? undefined : lines;
1301
- }
1302
1466
  /** Concise one-line summary of a tool result for the `⎿` line. */
1303
1467
  function firstLine(result) {
1304
1468
  const payload = payloadString(result);
@@ -1317,7 +1481,58 @@ function firstLine(result) {
1317
1481
  const cleaned = cleanToolText(result.trim());
1318
1482
  return truncate(cleaned.split('\n').find((l) => l.trim().length > 0) ?? '', 120);
1319
1483
  }
1484
+ function exceptionAttributes(err) {
1485
+ const error = err instanceof Error ? err : new Error(String(err));
1486
+ return {
1487
+ 'exception.type': error.constructor?.name ?? 'Error',
1488
+ 'exception.message': error.message,
1489
+ };
1490
+ }
1491
+ /**
1492
+ * `namzu.capability.detected` per package at `debug`, one aggregate summary
1493
+ * at `info` (the design's §6.3 `capability sandbox yes · files yes · …`
1494
+ * line), and `namzu.capability.broken` at `error` for any package that
1495
+ * resolved and failed to load. Never refuses the boot: nothing in
1496
+ * `NamzuCliConfig` marks a capability required yet, so `broken` here is
1497
+ * always the "not required by config" case §6.5 describes — an optional
1498
+ * capability's failure degrades what this line SAYS, never whether
1499
+ * `namzu.boot.ready` fires.
1500
+ */
1501
+ function logCapabilities(probes) {
1502
+ const log = getRootLogger();
1503
+ const summary = probes
1504
+ .map((p) => `${p.specifier.split('/').pop()} ${p.state === 'present' ? 'yes' : 'no'}`)
1505
+ .join(' · ');
1506
+ log.info(summary, { [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.CAPABILITY_DETECTED });
1507
+ for (const probe of probes) {
1508
+ if (probe.state === 'broken') {
1509
+ log.error('Capability probe failed to load', {
1510
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.CAPABILITY_BROKEN,
1511
+ 'namzu.capability.name': probe.specifier,
1512
+ ...exceptionAttributes(probe.error),
1513
+ });
1514
+ continue;
1515
+ }
1516
+ log.debug('Capability probe completed', {
1517
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.CAPABILITY_DETECTED,
1518
+ 'namzu.capability.name': probe.specifier,
1519
+ 'namzu.capability.state': probe.state,
1520
+ 'namzu.capability.present': probe.state === 'present',
1521
+ ...(probe.state === 'present' ? { 'namzu.capability.version': probe.version } : {}),
1522
+ });
1523
+ }
1524
+ }
1320
1525
  function emptySession(errorHint, errorKind = 'environment') {
1526
+ // Every path into this function is a boot refusal — `createAgentSession`
1527
+ // is the whole extent of the session-construction half of the boot
1528
+ // narrative, and every one of its early returns comes through here. One
1529
+ // emission point instead of five call-site ones is what keeps that true
1530
+ // instead of "true until the sixth `emptySession(...)` someone adds
1531
+ // forgets it."
1532
+ getRootLogger().error(errorHint, {
1533
+ [EVENT_NAME_ATTRIBUTE]: BOOT_EVENT_NAMES.BOOT_REFUSED,
1534
+ 'namzu.refusal.kind': errorKind,
1535
+ });
1321
1536
  return {
1322
1537
  hasProvider: false,
1323
1538
  errorKind,