mixdog 0.9.38 → 0.9.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/package.json +9 -4
  2. package/scripts/abort-recovery-test.mjs +43 -2
  3. package/scripts/agent-tag-reuse-smoke.mjs +146 -6
  4. package/scripts/agent-terminal-reap-test.mjs +127 -0
  5. package/scripts/agent-trace-io-test.mjs +69 -0
  6. package/scripts/code-graph-disk-hit-test.mjs +224 -0
  7. package/scripts/execution-completion-dedup-test.mjs +157 -0
  8. package/scripts/execution-pending-resume-kick-test.mjs +57 -2
  9. package/scripts/execution-resume-esc-integration-test.mjs +174 -0
  10. package/scripts/explore-bench.mjs +38 -2
  11. package/scripts/explore-prompt-policy-test.mjs +152 -11
  12. package/scripts/find-fuzzy-hidden-test.mjs +122 -0
  13. package/scripts/internal-comms-bench-test.mjs +226 -0
  14. package/scripts/internal-comms-bench.mjs +185 -58
  15. package/scripts/internal-comms-smoke.mjs +171 -23
  16. package/scripts/live-worker-smoke.mjs +38 -2
  17. package/scripts/memory-cycle-routing-test.mjs +111 -0
  18. package/scripts/memory-rule-contract-test.mjs +93 -0
  19. package/scripts/output-style-smoke.mjs +2 -2
  20. package/scripts/rg-runner-test.mjs +240 -0
  21. package/scripts/routing-corpus-test.mjs +349 -0
  22. package/scripts/routing-corpus.mjs +211 -32
  23. package/scripts/session-orphan-sweep-test.mjs +83 -0
  24. package/scripts/steering-drain-buckets-test.mjs +179 -0
  25. package/scripts/tool-smoke.mjs +21 -13
  26. package/scripts/tool-tui-presentation-test.mjs +202 -0
  27. package/src/agents/heavy-worker/AGENT.md +10 -7
  28. package/src/agents/reviewer/AGENT.md +6 -4
  29. package/src/agents/worker/AGENT.md +7 -5
  30. package/src/rules/agent/00-common.md +4 -4
  31. package/src/rules/agent/00-core.md +11 -14
  32. package/src/rules/agent/20-skip-protocol.md +3 -3
  33. package/src/rules/agent/30-explorer.md +50 -60
  34. package/src/rules/agent/40-cycle1-agent.md +15 -24
  35. package/src/rules/agent/41-cycle2-agent.md +33 -57
  36. package/src/rules/agent/42-cycle3-agent.md +28 -42
  37. package/src/rules/lead/01-general.md +7 -10
  38. package/src/rules/lead/lead-brief.md +11 -14
  39. package/src/rules/lead/lead-tool.md +6 -5
  40. package/src/rules/shared/01-tool.md +44 -45
  41. package/src/runtime/agent/orchestrator/agent-trace-format.mjs +37 -1
  42. package/src/runtime/agent/orchestrator/agent-trace-io.mjs +20 -5
  43. package/src/runtime/agent/orchestrator/providers/codex-client-meta.mjs +21 -1
  44. package/src/runtime/agent/orchestrator/providers/openai-oauth.mjs +17 -38
  45. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +17 -8
  46. package/src/runtime/agent/orchestrator/session/context-utils.mjs +270 -78
  47. package/src/runtime/agent/orchestrator/session/loop/tool-helpers.mjs +2 -1
  48. package/src/runtime/agent/orchestrator/session/manager/delivered-completions.mjs +123 -0
  49. package/src/runtime/agent/orchestrator/session/manager/idle-cleanup.mjs +15 -2
  50. package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +41 -2
  51. package/src/runtime/agent/orchestrator/session/manager/session-close.mjs +1 -1
  52. package/src/runtime/agent/orchestrator/session/store.mjs +223 -42
  53. package/src/runtime/agent/orchestrator/tools/builtin/arg-guard.mjs +12 -1
  54. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +25 -20
  55. package/src/runtime/agent/orchestrator/tools/builtin/fs-reachability.mjs +6 -2
  56. package/src/runtime/agent/orchestrator/tools/builtin/fuzzy-match.mjs +118 -53
  57. package/src/runtime/agent/orchestrator/tools/builtin/list-tool.mjs +106 -29
  58. package/src/runtime/agent/orchestrator/tools/builtin/path-utils.mjs +8 -1
  59. package/src/runtime/agent/orchestrator/tools/builtin/read-batch.mjs +4 -2
  60. package/src/runtime/agent/orchestrator/tools/builtin/read-range-index.mjs +3 -1
  61. package/src/runtime/agent/orchestrator/tools/builtin/read-snapshot-runtime.mjs +4 -1
  62. package/src/runtime/agent/orchestrator/tools/builtin/read-tool.mjs +33 -2
  63. package/src/runtime/agent/orchestrator/tools/builtin/rg-runner.mjs +151 -64
  64. package/src/runtime/agent/orchestrator/tools/builtin/search-path-diagnostics.mjs +37 -13
  65. package/src/runtime/agent/orchestrator/tools/builtin/search-tool.mjs +84 -49
  66. package/src/runtime/agent/orchestrator/tools/code-graph/build.mjs +172 -23
  67. package/src/runtime/agent/orchestrator/tools/code-graph/constants.mjs +3 -0
  68. package/src/runtime/agent/orchestrator/tools/code-graph/disk-cache.mjs +68 -5
  69. package/src/runtime/agent/orchestrator/tools/code-graph/dispatch.mjs +17 -3
  70. package/src/runtime/agent/orchestrator/tools/code-graph/graph-binary.mjs +24 -10
  71. package/src/runtime/agent/orchestrator/tools/code-graph-prewarm-worker.mjs +6 -3
  72. package/src/runtime/agent/orchestrator/tools/code-graph-tool-defs.mjs +4 -7
  73. package/src/runtime/agent/orchestrator/tools/code-graph.mjs +1 -0
  74. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +1 -1
  75. package/src/runtime/memory/lib/memory-cycle2-gate.mjs +4 -4
  76. package/src/runtime/memory/lib/memory-cycle2-mutations.mjs +4 -3
  77. package/src/runtime/memory/lib/memory-cycle3.mjs +12 -2
  78. package/src/runtime/shared/tool-primitives.mjs +4 -1
  79. package/src/runtime/shared/tool-status.mjs +27 -0
  80. package/src/runtime/shared/tool-surface.mjs +6 -3
  81. package/src/session-runtime/config-helpers.mjs +14 -0
  82. package/src/session-runtime/context-status.mjs +1 -0
  83. package/src/session-runtime/effort.mjs +6 -2
  84. package/src/session-runtime/model-recency.mjs +5 -2
  85. package/src/session-runtime/runtime-core.mjs +35 -2
  86. package/src/session-runtime/tool-catalog.mjs +34 -0
  87. package/src/standalone/agent-tool/notify.mjs +13 -0
  88. package/src/standalone/agent-tool.mjs +45 -69
  89. package/src/standalone/explore-tool.mjs +6 -7
  90. package/src/tui/App.jsx +31 -0
  91. package/src/tui/app/model-options.mjs +5 -3
  92. package/src/tui/app/model-picker.mjs +12 -24
  93. package/src/tui/app/transcript-window.mjs +1 -0
  94. package/src/tui/components/ToolExecution.jsx +11 -6
  95. package/src/tui/components/TranscriptItem.jsx +1 -1
  96. package/src/tui/components/tool-execution/surface-detail.mjs +24 -10
  97. package/src/tui/components/tool-execution/text-format.mjs +10 -19
  98. package/src/tui/dist/index.mjs +517 -142
  99. package/src/tui/engine/agent-job-feed.mjs +144 -17
  100. package/src/tui/engine/agent-response-tail.mjs +68 -0
  101. package/src/tui/engine/notification-plan.mjs +16 -0
  102. package/src/tui/engine/session-api.mjs +8 -2
  103. package/src/tui/engine/session-flow.mjs +19 -1
  104. package/src/tui/engine/tool-card-results.mjs +54 -32
  105. package/src/tui/engine/tool-result-status.mjs +75 -21
  106. package/src/tui/engine/turn.mjs +77 -42
  107. package/src/tui/engine.mjs +63 -2
  108. package/src/workflows/bench/WORKFLOW.md +25 -35
  109. package/src/workflows/default/WORKFLOW.md +38 -32
  110. package/src/workflows/solo/WORKFLOW.md +19 -22
  111. package/scripts/_jitter-fuzz.mjs +0 -44410
  112. package/scripts/_jitter-fuzz2.mjs +0 -44400
  113. package/scripts/_jitter-probe.mjs +0 -44397
  114. package/scripts/_jp2.mjs +0 -45614
@@ -27,6 +27,35 @@ export const MEASURED_TOOL_USAGE = Object.freeze({
27
27
  web_fetch: 2,
28
28
  });
29
29
  const MEASURED_TOOL_ORDER = Object.freeze(Object.keys(MEASURED_TOOL_USAGE));
30
+ const toolSchemaBreakdownMemo = new WeakMap();
31
+
32
+ function sameToolSchemaEntries(cached, tools) {
33
+ if (!cached || cached.entries.length !== tools.length) return false;
34
+ for (let index = 0; index < tools.length; index += 1) {
35
+ const entry = cached.entries[index];
36
+ const tool = tools[index];
37
+ if (entry.tool !== tool
38
+ || entry.name !== tool?.name
39
+ || entry.description !== tool?.description
40
+ || entry.inputSchema !== tool?.inputSchema
41
+ || entry.input_schema !== tool?.input_schema
42
+ || entry.parameters !== tool?.parameters
43
+ || entry.schema !== tool?.schema) return false;
44
+ }
45
+ return true;
46
+ }
47
+
48
+ function toolSchemaEntry(tool) {
49
+ return {
50
+ tool,
51
+ name: tool?.name,
52
+ description: tool?.description,
53
+ inputSchema: tool?.inputSchema,
54
+ input_schema: tool?.input_schema,
55
+ parameters: tool?.parameters,
56
+ schema: tool?.schema,
57
+ };
58
+ }
30
59
  export const DEFERRED_DEFAULT_FULL_TOOLS = Object.freeze([
31
60
  'read',
32
61
  'code_graph',
@@ -125,6 +154,10 @@ export function toolSchemaBucket(tool) {
125
154
  }
126
155
 
127
156
  export function estimateToolSchemaBreakdown(tools) {
157
+ if (Array.isArray(tools)) {
158
+ const cached = toolSchemaBreakdownMemo.get(tools);
159
+ if (sameToolSchemaEntries(cached, tools)) return cached.value;
160
+ }
128
161
  const out = {};
129
162
  for (const tool of Array.isArray(tools) ? tools : []) {
130
163
  const bucket = toolSchemaBucket(tool);
@@ -133,6 +166,7 @@ export function estimateToolSchemaBreakdown(tools) {
133
166
  row.tokens += estimateToolSchemaTokens([tool]);
134
167
  out[bucket] = row;
135
168
  }
169
+ if (Array.isArray(tools)) toolSchemaBreakdownMemo.set(tools, { entries: tools.map(toolSchemaEntry), value: out });
136
170
  return out;
137
171
  }
138
172
 
@@ -4,6 +4,7 @@
4
4
  import { modelVisibleToolCompletionMessage } from '../../runtime/shared/tool-execution-contract.mjs';
5
5
  import { renderBackgroundTask, sanitizeTaskMeta, setBackgroundTaskEnqueueFallback } from '../../runtime/shared/background-tasks.mjs';
6
6
  import { markCompletionEntry } from '../../runtime/agent/orchestrator/session/manager/pending-messages.mjs';
7
+ import { isDeliveredCompletion, logDuplicateSkip } from '../../runtime/agent/orchestrator/session/manager/delivered-completions.mjs';
7
8
  import { clean } from './helpers.mjs';
8
9
 
9
10
  export function createNotify(mgr) {
@@ -13,6 +14,18 @@ export function createNotify(mgr) {
13
14
  try {
14
15
  const visible = modelVisibleToolCompletionMessage(text, meta);
15
16
  if (!visible) return false;
17
+ // Skip-if-delivered: the TUI already injected + ACKed this completion
18
+ // body into the active loop, so this racing enqueue (fallback/reconcile
19
+ // or the async reject/false-resolve rescue) would double-inject it.
20
+ // Report DELIVERED (truthy), not false — a false return propagates through
21
+ // tryEnqueueFallback→onSettled(false), which un-marks notified/
22
+ // notifiedWithBody (background-tasks.mjs) and makes reconcile refire
23
+ // forever, eventually enqueuing a post-eviction duplicate. Suppressed here
24
+ // == already delivered, so the caller must mark it notified and stop.
25
+ if (isDeliveredCompletion({ executionId: meta?.execution_id, text: visible })) {
26
+ logDuplicateSkip('notify-enqueue', { executionId: meta?.execution_id, text: visible });
27
+ return true;
28
+ }
16
29
  // Mark this as a deferred completion/task notification so a later session
17
30
  // resume drops it rather than replaying it out-of-order (owner decision).
18
31
  return Boolean(mgr.enqueuePendingMessage(target, markCompletionEntry(visible)) > 0);
@@ -60,6 +60,7 @@ import {
60
60
  import { abnormalEmptyFinishError, renderResult } from './agent-tool/render.mjs';
61
61
  import { createProviderInit } from './agent-tool/provider-init.mjs';
62
62
  import { createNotify } from './agent-tool/notify.mjs';
63
+ import { resolveAgentTerminalReapMs } from '../session-runtime/config-helpers.mjs';
63
64
  // Re-export the static tool descriptor so importers of this facade keep the
64
65
  // identical public surface (`import { AGENT_TOOL } from './agent-tool.mjs'`).
65
66
  export { AGENT_TOOL };
@@ -68,30 +69,6 @@ ensureProcessListenerHeadroom(64);
68
69
 
69
70
  const STANDALONE_SOURCE_ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..');
70
71
 
71
- // Grace window during which a terminated/idle worker row is kept around so the
72
- // same terminal can re-use (or cleanly re-spawn) the same tag. Cached as a
73
- // constant like the other timeouts; override with MIXDOG_AGENT_TERMINAL_REAP_MS.
74
- // 5m (was 1h): aligned with the BP4 message-tail cache TTL — past 5m the
75
- // session's prompt cache is cold anyway, so same-tag reuse pays a full prefix
76
- // rewrite; a fresh respawn (send dead-tag fallback) starts smaller and cheaper.
77
- // Trace replay (2026-07): agent tail cost 1h 1.77M -> 5m 1.11M (-37%).
78
- const TERMINAL_REAP_MS = envTimeoutMs('MIXDOG_AGENT_TERMINAL_REAP_MS', 5 * 60_000);
79
- // Provider-aware reap windows: the 5m default matches Anthropic's 5m agent
80
- // tail TTL, but providers with long-lived automatic prefix caches (OpenAI 24h
81
- // retention, DeepSeek hours-days, Groq 2h) keep a warm cache well past 5m, so
82
- // same-tag reuse stays cheaper than a respawn. Capped at 1h — holding closed
83
- // worker sessions longer hoards manager memory for a reuse pattern that
84
- // rarely spans hours. Explicit MIXDOG_AGENT_TERMINAL_REAP_MS overrides all.
85
- const PROVIDER_REAP_MS = {
86
- 'openai': 60 * 60_000,
87
- 'deepseek': 60 * 60_000,
88
- 'groq': 60 * 60_000,
89
- 'openai-oauth': 10 * 60_000,
90
- };
91
- function reapMsForProvider(provider) {
92
- if (clean(process.env.MIXDOG_AGENT_TERMINAL_REAP_MS)) return TERMINAL_REAP_MS;
93
- return PROVIDER_REAP_MS[clean(provider)] || TERMINAL_REAP_MS;
94
- }
95
72
  // Independent hard cap for the spawn *prep* phase (ensureProvider /
96
73
  // prepareAgentSession / catalog+rules load). Kept separate from the
97
74
  // first-response watchdog so prep cannot hang a whole fanout before the model
@@ -161,7 +138,8 @@ export function createStandaloneAgent({ cfgMod, reg, mgr, dataDir, cwd: defaultC
161
138
  const t = workerRowTime(row);
162
139
  if (!t) return true;
163
140
  if (!isTerminalWorkerStatus(row.status || row.stage)) return true;
164
- return Date.now() - t < reapMsForProvider(row.provider);
141
+ const reapMs = resolveAgentTerminalReapMs(cfgMod.loadConfig(), row.provider);
142
+ return reapMs == null || Date.now() - t < reapMs;
165
143
  }
166
144
 
167
145
  function normalizeWorkerRows(value) {
@@ -562,6 +540,17 @@ export function createStandaloneAgent({ cfgMod, reg, mgr, dataDir, cwd: defaultC
562
540
  removeWorkerRow({ tag, sessionId });
563
541
  }
564
542
 
543
+ function forgetTerminalSession(tag, sessionId) {
544
+ const value = clean(tag);
545
+ const id = clean(sessionId);
546
+ if (value && id && tags.get(value) === id) {
547
+ tags.delete(value);
548
+ tagAgents.delete(value);
549
+ tagCwds.delete(value);
550
+ }
551
+ if (id) removeWorkerRow({ sessionId: id });
552
+ }
553
+
565
554
  function cancelReap(sessionId) {
566
555
  const handle = reapTimers.get(sessionId);
567
556
  if (!handle) return false;
@@ -574,15 +563,16 @@ export function createStandaloneAgent({ cfgMod, reg, mgr, dataDir, cwd: defaultC
574
563
  if (!sessionId) return;
575
564
  cancelReap(sessionId);
576
565
  const reapProvider = provider || getLiveSession(sessionId)?.provider || null;
566
+ const reapMs = resolveAgentTerminalReapMs(cfgMod.loadConfig(), reapProvider);
567
+ if (reapMs == null) return;
577
568
  const handle = setTimeout(() => {
578
569
  reapTimers.delete(sessionId);
579
570
  try { mgr.hideSessionFromList?.(sessionId); } catch {}
580
571
  const tag = tagForSession(sessionId);
581
- if (tag) forgetTag(tag);
582
- removeWorkerRow({ tag, sessionId });
572
+ forgetTerminalSession(tag, sessionId);
583
573
  clearAgentStatuslineRoute(sessionId);
584
574
  try { mgr.closeSession(sessionId, 'terminal-reap'); } catch {}
585
- }, reapMsForProvider(reapProvider));
575
+ }, reapMs);
586
576
  handle.unref?.();
587
577
  reapTimers.set(sessionId, handle);
588
578
  }
@@ -872,8 +862,7 @@ export function createStandaloneAgent({ cfgMod, reg, mgr, dataDir, cwd: defaultC
872
862
  if (!prepared?.session?.id) return;
873
863
  try { mgr.closeSession(prepared.session.id, reason); } catch {}
874
864
  try { clearAgentStatuslineRoute(prepared.session.id); } catch {}
875
- if (prepared.tag) forgetTag(prepared.tag);
876
- removeWorkerRow({ tag: prepared.tag, sessionId: prepared.session.id });
865
+ forgetTerminalSession(prepared.tag, prepared.session.id);
877
866
  }
878
867
 
879
868
  // Owner/worker completion notification lives in ./agent-tool/notify.mjs; the
@@ -1464,7 +1453,12 @@ export function createStandaloneAgent({ cfgMod, reg, mgr, dataDir, cwd: defaultC
1464
1453
  const sessionId = resolveTag(target, scopedContext, { scanSessions: wantsSessionScan(args) });
1465
1454
  if (!sessionId) {
1466
1455
  if (!target.startsWith('sess_') && tagAgents.has(target)) {
1467
- forgetTag(target);
1456
+ // This is only stale local metadata: resolveTag found no session in
1457
+ // this terminal/scope, so there is no sessionId-safe worker row to
1458
+ // delete. Never turn it into a tag-wide persisted-row removal.
1459
+ tags.delete(target);
1460
+ tagAgents.delete(target);
1461
+ tagCwds.delete(target);
1468
1462
  if (task?.taskId) cancelBackgroundTask(task.taskId, 'cancelled by agent close');
1469
1463
  return { closed: true, forgotten: true, tag: target, sessionId: null, task_id: task?.taskId || null };
1470
1464
  }
@@ -1472,8 +1466,7 @@ export function createStandaloneAgent({ cfgMod, reg, mgr, dataDir, cwd: defaultC
1472
1466
  }
1473
1467
  cancelReap(sessionId);
1474
1468
  const tag = tagForSession(sessionId);
1475
- if (tag) forgetTag(tag);
1476
- removeWorkerRow({ tag, sessionId });
1469
+ forgetTerminalSession(tag, sessionId);
1477
1470
  clearAgentStatuslineRoute(sessionId);
1478
1471
  // Cancel any running background task bound to this session BEFORE closing
1479
1472
  // the session. Otherwise closeSession rejects the in-flight runSpawn with
@@ -1549,16 +1542,12 @@ export function createStandaloneAgent({ cfgMod, reg, mgr, dataDir, cwd: defaultC
1549
1542
  function reapTerminalTraceForTag(tag, context = {}) {
1550
1543
  const value = clean(tag);
1551
1544
  if (!value || value.startsWith('sess_')) return false;
1552
- if (!hasTerminalTrace(value, context)) return false;
1545
+ const row = terminalWorkerRowForTag(value, context);
1546
+ if (!row) return false;
1553
1547
  refreshTagsFromSessions({ context });
1554
- let sessionId = tags.get(value) || '';
1555
- if (!sessionId) {
1556
- const row = readWorkerRows(context).find((item) => clean(item.tag) === value);
1557
- sessionId = clean(row?.sessionId) || '';
1558
- }
1548
+ const sessionId = clean(row.sessionId);
1559
1549
  if (sessionId) cancelReap(sessionId);
1560
- forgetTag(value);
1561
- removeWorkerRow({ tag: value, sessionId });
1550
+ forgetTerminalSession(value, sessionId);
1562
1551
  return true;
1563
1552
  }
1564
1553
 
@@ -1593,38 +1582,25 @@ export function createStandaloneAgent({ cfgMod, reg, mgr, dataDir, cwd: defaultC
1593
1582
  if (!fallbackTag || fallbackTag.startsWith('sess_') || !isDeadTarget) throw err;
1594
1583
  const prompt = clean(args.message || args.prompt);
1595
1584
  if (!prompt) throw err;
1596
- // Inherit the dead session's agent/cwd so the auto-respawn doesn't
1597
- // die on "agent spawn: agent is required" when Lead's send (correctly)
1598
- // omitted them. Read BEFORE forgetTag wipes the registry entries.
1599
- //
1600
- // Root cause of the surfaced "target not found" errors: when a tag is
1601
- // FULLY reaped (session gone past the 5m grace window), the in-memory
1602
- // tagAgents/tagCwds maps are cleared, so inheritedAgent came back empty
1603
- // and this path re-threw instead of respawning. Recover the agent/cwd
1604
- // from the persisted worker row, then fall back to the default `worker`
1605
- // agent so an unknown-but-plausible tag still cold-respawns. The guards
1606
- // above (empty tag / raw sess_ id / non-dead error) keep plainly invalid
1607
- // targets erroring.
1608
- // readWorkerRows is already scoped to THIS terminal (clientHostPid,
1609
- // via rowMatchesContext), so inheritedRow — and every mutation below
1610
- // keyed on its sessionId — never touches another terminal's same-tag
1611
- // worker row.
1585
+ // A retained row is proof that this terminal owned the tag. Once the
1586
+ // reaper removes that row, an explicit agent plus usable cwd is enough
1587
+ // to cold-spawn a replacement without retaining closed sessions.
1612
1588
  let inheritedRow = null;
1613
1589
  try {
1614
1590
  inheritedRow = readWorkerRows(scopedContext).find((row) => clean(row.tag) === fallbackTag) || null;
1615
1591
  } catch { inheritedRow = null; }
1616
- // Gate the respawn on evidence the tag previously existed IN THIS
1617
- // terminal: a persisted, clientHostPid-scoped worker row. The
1618
- // in-memory tags/tagAgents/tagCwds maps are NOT reliable evidence —
1619
- // closeAll() calls refreshTagsFromSessions() with NO context
1620
- // (see ~L1501), which populates them from EVERY terminal's rows
1621
- // (readWorkerRows({}) — rowMatchesContext returns true with no pid),
1622
- // so they can hold a peer terminal's tag. A typo'd / never-existed
1623
- // tag has no scoped row and must keep the original "not found" error.
1624
- if (!inheritedRow) throw err;
1625
- const inheritedSessionId = clean(inheritedRow.sessionId);
1626
- const inheritedAgent = clean(args.agent) || clean(inheritedRow.agent) || 'worker';
1627
- const inheritedCwd = clean(args.cwd) || clean(inheritedRow.cwd) || clean(callerCwd);
1592
+ // Never use explicit cold-spawn identity to take over a tag retained
1593
+ // by another terminal. In-memory maps can contain peer entries, so
1594
+ // inspect the persisted index directly.
1595
+ const peerRow = readAllWorkerRows().find((row) => (
1596
+ clean(row.tag) === fallbackTag && !rowMatchesContext(row, scopedContext)
1597
+ ));
1598
+ const explicitAgent = clean(args.agent);
1599
+ const explicitCwd = clean(args.cwd) || clean(callerCwd);
1600
+ if (peerRow || (!inheritedRow && (!explicitAgent || !explicitCwd))) throw err;
1601
+ const inheritedSessionId = clean(inheritedRow?.sessionId);
1602
+ const inheritedAgent = explicitAgent || clean(inheritedRow?.agent) || 'worker';
1603
+ const inheritedCwd = clean(args.cwd) || clean(inheritedRow?.cwd) || clean(callerCwd);
1628
1604
  // Drop this terminal's in-memory trace and remove ONLY the persisted
1629
1605
  // row matching inheritedRow.sessionId. Do NOT call forgetTag here: it
1630
1606
  // does a tag-wide removeWorkerRow({tag,sessionId}) (L556) whose OR
@@ -9,6 +9,8 @@ import { ensureProcessListenerHeadroom } from '../runtime/shared/process-listene
9
9
 
10
10
  ensureProcessListenerHeadroom(64);
11
11
 
12
+ const EXPLORE_NO_RELOOKUP_CONTRACT = 'Every returned requested path:line freezes the LOCATION only; read/code_graph detail inspection is valid when content was not returned; never re-locate it, and search only unresolved facets.';
13
+
12
14
  // Ported from the original mixdog tool-defs.mjs `explore` entry.
13
15
  // `aiWrapped` is dropped: in the standalone build there is no aiWrapped
14
16
  // dispatch hub — execution is wired directly in the runtime executor below
@@ -18,7 +20,7 @@ export const EXPLORE_TOOL = {
18
20
  name: 'explore',
19
21
  title: 'Explore',
20
22
  annotations: { title: 'Explore', readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: false },
21
- description: 'Locator for broad/uncertain targets with no known path repo anchors AND out-of-repo/machine-wide file locations (e.g. "where does X store its logs/config"). Searches dot-directories via the hardened find internally. Array = independent targets: when starting a task, decompose what you need to know into facets (implementation site / config-load path / tests / error origin, ...) and fan them out as one query[] call (max 8, parallel) never send rephrasings of the same target.',
23
+ description: `Broad/uncertain locator with no known path; use after input-state classification for repo anchors or out-of-repo/machine-wide locations. Hardened find searches dot-directories. Array = independent targets; array-first, facets in one query[] (max 8, parallel), never rephrasings. ${EXPLORE_NO_RELOOKUP_CONTRACT}`,
22
24
  inputSchema: {
23
25
  type: 'object',
24
26
  properties: {
@@ -87,14 +89,11 @@ function escapeXml(str) {
87
89
  return String(str).replace(/&/g, '&amp;').replace(/</g, '&lt;').replace(/>/g, '&gt;');
88
90
  }
89
91
 
90
- // Mirrors buildExplorerPrompt from the original ai-wrapped-dispatch: each
91
- // explorer receives ONLY its own query, with a trailing descriptive-only
92
- // reminder. The full no-verdict contract lives at system level
93
- // (rules/agent/30-explorer.md).
92
+ // Each explorer receives only its own query. The full routing and behavioral
93
+ // contract lives at system level (rules/agent/30-explorer.md).
94
94
  export function buildExplorerPrompt(query) {
95
- return `<query>${escapeXml(query)}</query>\nReminder: BUDGET = TWO MESSAGES NORMALLY: message 1 = your multi-tool batch, message 2 = your answer text. A 3rd message (ANY extra tool call) is a DEFECT unless message 1 returned ZERO specific-token lines — extra code_graph/grep calls to confirm or upgrade an anchor you already hold are the single biggest source of overspend. TURN 1 IS ONE BATCH, NON-NEGOTIABLE: your FIRST tool message fires, in that single message together, grep (pattern[] packing 3-6 token variants) AND code_graph symbol_search (identifiers) AND find/glob (path/name fragments) — never emit grep alone and wait for its result before adding code_graph/find. Serial one-tool-per-turn is the #1 budget defect and forfeits the turn-1 -> answer path; a first turn that is not this multi-tool batch is malformed. After that batch returns, if ANY line carries a specific query token you MUST emit the ANSWER as your very next message: a further tool call to refine, confirm, or upgrade is the primary overspend defect — a second tool turn is legal ONLY when the batch returned ZERO specific-token lines. Then RULE ZERO is a binary gate after every tool result: >=1 path:line matching a SPECIFIC query token (product/library/domain name) -> STOP and answer NOW with those exact coordinates, this IS your final turn; zero -> one more batch if budget remains. Turns 2-3 exist SOLELY as zero-hit recovery (the previous turn matched ZERO specific tokens); spending a turn to confirm, refine, or upgrade an anchor you already hold is a defect. A generic-word-only match (schema, handler, config, resolver, index, error...) while the query's specific tokens match nowhere counts as ZERO, not a hit. There is no third branch: "hits exist but I want better ones" IS branch one — answer now. A code_graph symbol hit (find_symbol/symbol_search returning path:line) IS an anchor — emit it directly, never re-locate it with grep. Credibility is mechanical (specific-token match), never judged: "is this the real implementation / final handler / just a wrapper" are caller questions, FORBIDDEN here; mark weak anchors ? and answer anyway. Flow/how/trigger queries: first matching definition/entry anchors ARE the complete answer — trace no chains, one anchor per concept; the SOLE exception: a query that EXPLICITLY asks flow/default-resolution whose turn 1 yielded only an entry anchor (not the resolved value) may take ONE turn-2 hop to the resolving site, then stop. You locate WHERE, never WHY. Scope is ALWAYS the session working directory: omit path (project cwd default) or pass only a path seen in an earlier result; an unverified path/name fragment rides the SAME turn-1 find query[] batch (a find-only turn is a defect) and scopes grep/glob only through the exact path find returns; inventing a directory (/workspace/..., another repo's layout) is a defect — on zero hits change TOKENS, never guess paths. HARD max 3 tool turns; counter line (turn 1/3, turn 2/3, turn 3/3) on every tool message; expected shape is turn 1 -> answer, turns 2-3 are miss-recovery only (previous turn had ZERO matching lines), with changed tokens; after turn 3/3 you MUST answer best-so-far. Each turn = ONE maximal batch: a single grep whose pattern[] packs ALL token variants PLUS code_graph/find/glob in the SAME message; a single-tool turn is a wasted turn. Before emitting EXPLORATION_FAILED re-scan ALL earlier results: any line matching a specific query token -> answer with that anchor (? if weak) — a weak anchor beats a false miss; EXPLORATION_FAILED only when all 3 turns found zero token-matching lines. Output only anchor lines formatted as path:line — symbol/name — short reason, max 3 lines, or EXPLORATION_FAILED. For a CODE-location answer every line MUST carry :line (an explicit line number) — a bare filename with no :line is a DEFECT; with no line-anchored code evidence, return EXPLORATION_FAILED rather than a vague file-only or prose answer. EXCEPTION — file/dir-LOCATION queries (where does X store its config/logs/data on disk, which directory or file holds Y): an exact VERIFIED path (the file or directory itself, no :line) IS the valid answer — do not force a line number and do not FAIL. No preamble, bullets, numbering, headings, summary, code quotes, analysis, verdicts, ratings, or recommendations.`;
95
+ return `<query>${escapeXml(query)}</query>`;
96
96
  }
97
-
98
97
  export function normalizeExploreQueries(rawQuery) {
99
98
  let raw = rawQuery;
100
99
  // Some clients JSON-stringify arrays when the schema field is loosely typed.
package/src/tui/App.jsx CHANGED
@@ -592,6 +592,37 @@ export function App({ store, initialStatusLine = '', forceOnboarding = false })
592
592
  searchModelsCacheRef.current = { models: null, at: 0 };
593
593
  }
594
594
  }, []);
595
+ // Boot-time catalog prefetch: warm the /model & /agents provider catalog and
596
+ // the /search catalog once at startup so those pickers open instantly from
597
+ // cache (openModelPicker still TTL-refreshes stale rows in the background).
598
+ // Provider models load first so the search catalog derives from the full
599
+ // runtime cache instead of the sparse quick rows. Guarded by the same
600
+ // generation seq as the onboarding prefetch so an auth-triggered
601
+ // clearModelCaches() can't be clobbered by a stale in-flight result.
602
+ useEffect(() => {
603
+ let alive = true;
604
+ const timer = setTimeout(async () => {
605
+ const seq = onboardingPrefetchSeqRef.current;
606
+ try {
607
+ const models = await Promise.resolve(store.listProviderModels?.() || []);
608
+ if (alive && seq === onboardingPrefetchSeqRef.current
609
+ && Array.isArray(models) && models.length > 0
610
+ && !Array.isArray(providerModelsCacheRef.current.models)) {
611
+ providerModelsCacheRef.current = { models, at: Date.now() };
612
+ }
613
+ } catch { /* prefetch is advisory; pickers fall back to their own load */ }
614
+ if (!alive) return;
615
+ try {
616
+ const searchModels = await Promise.resolve(store.listSearchModels?.() || []);
617
+ if (alive && Array.isArray(searchModels) && searchModels.length > 0
618
+ && !Array.isArray(searchModelsCacheRef.current.models)) {
619
+ searchModelsCacheRef.current = { models: searchModels, at: Date.now() };
620
+ }
621
+ } catch { /* prefetch is advisory; /search falls back to its own load */ }
622
+ }, 1500);
623
+ timer.unref?.();
624
+ return () => { alive = false; clearTimeout(timer); };
625
+ }, [store]);
595
626
  // Onboarding wizard + channel setup picker factories. Instantiated here —
596
627
  // after the onboarding refs above (const-TDZ) — with later-defined openers
597
628
  // (openProviderSetupPicker/openOutputStylePicker) threaded as lazy getters
@@ -70,11 +70,13 @@ export const compareModelRecency = (a, b) => {
70
70
 
71
71
  const ta = releaseTime(a);
72
72
  const tb = releaseTime(b);
73
- if (ta !== tb) return tb - ta;
74
-
75
- if (!!a?.latest !== !!b?.latest) return a?.latest ? -1 : 1;
76
73
  const versionDelta = compareModelVersion(a, b);
74
+ // Release dates win only when both sides have one; sparse OAuth catalogs
75
+ // must not sink undated (often newest) models below dated/latest ones.
76
+ if (ta > 0 && tb > 0 && ta !== tb) return tb - ta;
77
77
  if (versionDelta) return versionDelta;
78
+ if (!!a?.latest !== !!b?.latest) return a?.latest ? -1 : 1;
79
+ if (ta !== tb) return tb - ta;
78
80
  return String(a?.display || a?.id || '').localeCompare(String(b?.display || b?.id || ''));
79
81
  };
80
82
 
@@ -48,11 +48,9 @@ export function createModelPicker({
48
48
  setChannelPrompt(null);
49
49
  setHookPrompt(null);
50
50
  setSettingsPrompt(null);
51
- const modelPickerRequest = ++modelPickerRequestRef.current;
51
+ modelPickerRequestRef.current += 1;
52
52
  let modelPickerClosed = false;
53
- let activeModelProvider = null;
54
53
  let providerListHighlightProvider = null;
55
- const isActiveModelPicker = () => !modelPickerClosed && modelPickerRequestRef.current === modelPickerRequest;
56
54
  const returnTo = typeof options.returnTo === 'function' ? options.returnTo : null;
57
55
  const returnLabel = String(options.returnLabel || 'Agents');
58
56
  const returnOnNestedCancel = options.returnOnNestedCancel === true;
@@ -68,7 +66,6 @@ export function createModelPicker({
68
66
  : [];
69
67
  let refreshModelsPromise = null;
70
68
  let renderedQuickModels = false;
71
- let renderActiveProviderModels = null;
72
69
  if (!providerModels.length || options.refreshModels === true) {
73
70
  setPicker({
74
71
  title: options.title || 'Model',
@@ -130,16 +127,12 @@ export function createModelPicker({
130
127
  providerListHighlightProvider = renderOptions.highlightProvider;
131
128
  }
132
129
  const highlightProvider = renderOptions.highlightProvider || providerListHighlightProvider || null;
133
- activeModelProvider = null;
134
- renderActiveProviderModels = null;
135
130
  const openProviderModelsPicker = (provider) => {
136
131
  if (!provider) return;
137
- activeModelProvider = provider;
138
- renderActiveProviderModels = () => openProviderModelsPicker(provider);
139
132
  const providerModels = models.filter((model) => model.provider === provider);
140
133
  const preferredEffort = (values = []) => {
141
134
  const allowed = values.filter(Boolean);
142
- for (const value of ['high', 'medium', 'low', 'none', 'xhigh', 'max']) {
135
+ for (const value of ['high', 'medium', 'low', 'none', 'xhigh', 'max', 'ultra']) {
143
136
  if (allowed.includes(value)) return value;
144
137
  }
145
138
  return allowed[0] || null;
@@ -213,6 +206,7 @@ export function createModelPicker({
213
206
  if (value === 'medium') return '◑';
214
207
  if (value === 'high') return '◕';
215
208
  if (value === 'max') return '◆';
209
+ if (value === 'ultra') return '✦';
216
210
  return '●';
217
211
  };
218
212
  const effortColor = (value) => {
@@ -221,6 +215,7 @@ export function createModelPicker({
221
215
  if (value === 'medium') return theme.claude;
222
216
  if (value === 'high') return theme.error;
223
217
  if (value === 'max') return theme.permission;
218
+ if (value === 'ultra') return theme.permission;
224
219
  return theme.error;
225
220
  };
226
221
  const modelFooter = (model = null) => {
@@ -366,34 +361,27 @@ export function createModelPicker({
366
361
  };
367
362
 
368
363
  renderModelPicker();
369
- const applyFreshModels = (freshModels) => {
370
- if (!isActiveModelPicker()) return;
364
+ // Freshness policy: an open picker keeps the catalog it first rendered.
365
+ // Background refreshes only update the cache, so fresh rows apply on the
366
+ // NEXT open (re-entry) instead of re-sorting the list mid-selection.
367
+ const adoptFreshModels = (freshModels) => {
371
368
  if (!Array.isArray(freshModels) || freshModels.length === 0) return;
372
- providerModels = freshModels;
373
- models = normalizeModelOptions(providerModels);
374
- cacheRef.current = { models: providerModels, at: Date.now() };
375
- if (activeModelProvider === null) {
376
- renderModelPicker();
377
- } else if (typeof renderActiveProviderModels === 'function') {
378
- renderActiveProviderModels();
379
- }
369
+ cacheRef.current = { models: freshModels, at: Date.now() };
380
370
  };
381
371
  if (renderedQuickModels && refreshModelsPromise) {
382
- void refreshModelsPromise.then(applyFreshModels).catch(() => {});
372
+ void refreshModelsPromise.then(adoptFreshModels).catch(() => {});
383
373
  } else if (cacheIsStale) {
384
374
  if (!providerModelsTtlRefreshPromise) {
385
375
  providerModelsTtlRefreshPromise = Promise.resolve(loadModels({ force: true }))
386
376
  .then((freshModels) => {
387
- if (Array.isArray(freshModels) && freshModels.length > 0) {
388
- cacheRef.current = { models: freshModels, at: Date.now() };
389
- }
377
+ adoptFreshModels(freshModels);
390
378
  return freshModels;
391
379
  })
392
380
  .finally(() => {
393
381
  providerModelsTtlRefreshPromise = null;
394
382
  });
395
383
  }
396
- void providerModelsTtlRefreshPromise.then(applyFreshModels).catch(() => {});
384
+ void providerModelsTtlRefreshPromise.catch(() => {});
397
385
  }
398
386
  };
399
387
 
@@ -245,6 +245,7 @@ function toolHasDisplayResultForRows(item) {
245
245
 
246
246
  function toolExpandedRawTextForRows(item, rawRt) {
247
247
  if (item?.aggregate) return rawRt;
248
+ if (item?.agentResponseAggregate) return rawRt;
248
249
  const hasDisplayResult = toolHasDisplayResultForRows(item);
249
250
  if (hasDisplayResult) return toolDisplayedResultTextForRows(item);
250
251
  return stripLeadingStatusMarkerFromTextForRows(rawRt || '');
@@ -86,7 +86,7 @@ function statusCopy(name, label, count, doneCount, pending, isError, args = {})
86
86
  // dropping the pad just normalizes the spacing.
87
87
  return formatToolActionHeader(name, args, { pending, count });
88
88
  }
89
- export function ToolExecution({ name, args, result, rawResult, isError, errorCount, callErrorCount, expanded, columns = 80, attached = false, count = 1, completedCount = 0, startedAt = 0, completedAt = 0, aggregate = false, categories = {}, doneCategories = null, headerFinalized = true, deferredDisplayReady = false }) {
89
+ export function ToolExecution({ name, args, result, rawResult, isError, errorCount, callErrorCount, exitErrorCount, expanded, columns = 80, attached = false, count = 1, completedCount = 0, startedAt = 0, completedAt = 0, aggregate = false, categories = {}, doneCategories = null, headerFinalized = true, deferredDisplayReady = false, agentResponseAggregate = false }) {
90
90
  const rowWidth = Math.max(1, Number(columns || 80));
91
91
  const groupCount = Math.max(1, Number(count || 1));
92
92
  const doneCount = Math.max(0, Math.min(groupCount, Number(completedCount || (result == null ? 0 : groupCount))));
@@ -138,6 +138,9 @@ export function ToolExecution({ name, args, result, rawResult, isError, errorCou
138
138
  // paint the dot red. Fall back to 0 (never `isError`) when the engine did not
139
139
  // supply a call-error count so a result failure can't leak into the dot.
140
140
  const callFailedCount = clampFailureCount(callErrorCount, groupCount, false);
141
+ // Shell command-exits (ran, non-zero exit). Counted separately so the dot
142
+ // paints the neutral warning "Exit" color instead of red or green success.
143
+ const exitFailedCount = clampFailureCount(exitErrorCount, groupCount, false);
141
144
  const displayGroupCount = groupCount;
142
145
  const displayCategories = normalizeCountMap(categories || {});
143
146
  // In the DONE state, count only successful calls: error-terminated calls are
@@ -212,7 +215,7 @@ export function ToolExecution({ name, args, result, rawResult, isError, errorCou
212
215
  const aggregateTerminalStatus = pending
213
216
  ? 'running'
214
217
  : (resultTerminalStatus(rt) || (isError || failedCount > 0 ? 'failed' : 'completed'));
215
- const dotColor = toolStatusColor({ pending, groupCount, callFailedCount, terminalStatus: aggregateTerminalStatus });
218
+ const dotColor = toolStatusColor({ pending, groupCount, callFailedCount, exitFailedCount, terminalStatus: aggregateTerminalStatus });
216
219
  const dotText = pending && !blinkOn ? ' ' : TURN_MARKER;
217
220
  const gutter = 2;
218
221
  const showHeaderExpandHint = hasRawResult;
@@ -310,7 +313,7 @@ export function ToolExecution({ name, args, result, rawResult, isError, errorCou
310
313
  const firstResultLine = hasDisplayResult ? String(lines[0] ?? '') : '';
311
314
  const firstResultLineClipped = hasDisplayBody && stringWidth(firstResultLine) > maxResultChars;
312
315
  const hasHiddenDetail = !pending && hasDisplayBody && (totalLines > 1 || firstResultLineClipped || Boolean(resultSummary));
313
- const shellStatus = isShellSurface ? shellDisplayStatus({ pending, failedCount, isError, result: displayedResultText }) : '';
316
+ const shellStatus = isShellSurface ? shellDisplayStatus({ pending, failedCount, exitFailedCount, isError, result: displayedResultText }) : '';
314
317
  const shellElapsed = isShellSurface ? (shellResultElapsed(displayedResultText) || elapsed) : '';
315
318
  const backgroundElapsed = backgroundMeta
316
319
  ? backgroundTaskElapsed(backgroundMeta, elapsed)
@@ -384,7 +387,9 @@ export function ToolExecution({ name, args, result, rawResult, isError, errorCou
384
387
  const showRawResult = expanded && (hasDisplayBody || hasRawResult)
385
388
  && (!isBackgroundMetadataResult || hasRawResult);
386
389
  const detailLines = showRawResult
387
- ? (hasDisplayBody ? lines : (rawRt ? stripLeadingStatusMarkerLines(rawRt.split('\n')) : []))
390
+ ? (agentResponseAggregate && hasRawResult
391
+ ? stripLeadingStatusMarkerLines(rawRt.split('\n'))
392
+ : (hasDisplayBody ? lines : (rawRt ? stripLeadingStatusMarkerLines(rawRt.split('\n')) : [])))
388
393
  : (collapsedDetail ? [collapsedDetail] : []);
389
394
  const isPendingPlaceholderDetail = !showRawResult && Boolean(pendingDetailPlaceholder);
390
395
  const detailColor = isPendingPlaceholderDetail ? theme.subtle : theme.text;
@@ -426,7 +431,7 @@ export function ToolExecution({ name, args, result, rawResult, isError, errorCou
426
431
  const keepAgentDetail = (isError || agentFailureText) && !(agentHeaderFailure && !agentSurfaceBrief);
427
432
  visibleDetailLines = keepAgentDetail ? [agentDetailLine] : [];
428
433
  }
429
- const finalStatusColor = toolStatusColor({ pending, groupCount, callFailedCount, terminalStatus });
434
+ const finalStatusColor = toolStatusColor({ pending, groupCount, callFailedCount, exitFailedCount, terminalStatus });
430
435
  const dotColor = finalStatusColor;
431
436
  // Agent surface cards use directional markers: `←` for requests going OUT
432
437
  // (spawn/send/etc.) and `→` for the response coming back IN. Background
@@ -446,7 +451,7 @@ export function ToolExecution({ name, args, result, rawResult, isError, errorCou
446
451
  const markerText = isDirectionalMarker ? `${markerGlyph} ` : markerGlyph;
447
452
  const dotText = pending && !blinkOn ? ' ' : markerText;
448
453
  let labelText;
449
- if (isAgentResponse) labelText = agentResponseTitle(parsedArgs);
454
+ if (isAgentResponse) labelText = agentResponseTitle(parsedArgs, displayGroupCount);
450
455
  else if (isBackgroundResponse) labelText = backgroundTaskResultTitle(normalizedName, backgroundMeta || parsedArgs);
451
456
  else if (isBackgroundMetadataResult) labelText = backgroundTaskActionTitle(normalizedName, backgroundMeta);
452
457
  else if (isShellSurface) labelText = shellHeader(shellStatus, displayGroupCount);
@@ -77,7 +77,7 @@ export const Item = React.memo(function Item({ item, prevKind, columns, toolOutp
77
77
  node = <ToolHookDenialCard item={item} columns={columns} />;
78
78
  break;
79
79
  }
80
- node = <ToolExecution name={item.name} args={item.args} result={item.result} rawResult={item.rawResult} isError={item.isError} errorCount={item.errorCount} callErrorCount={item.callErrorCount} expanded={toolOutputExpanded || item.expanded} columns={columns} attached={false} count={item.count} completedCount={item.completedCount} startedAt={item.startedAt} completedAt={item.completedAt} aggregate={item.aggregate} categories={item.categories} doneCategories={item.doneCategories} headerFinalized={item.headerFinalized} deferredDisplayReady={item.deferredDisplayReady} />;
80
+ node = <ToolExecution name={item.name} args={item.args} result={item.result} rawResult={item.rawResult} isError={item.isError} errorCount={item.errorCount} callErrorCount={item.callErrorCount} exitErrorCount={item.exitErrorCount} expanded={toolOutputExpanded || item.expanded} columns={columns} attached={false} count={item.count} completedCount={item.completedCount} startedAt={item.startedAt} completedAt={item.completedAt} aggregate={item.aggregate} categories={item.categories} doneCategories={item.doneCategories} headerFinalized={item.headerFinalized} deferredDisplayReady={item.deferredDisplayReady} agentResponseAggregate={item.agentResponseAggregate} />;
81
81
  break;
82
82
  }
83
83
  case 'notice':
@@ -21,11 +21,18 @@ export function isShellTool(normalizedName, label = '') {
21
21
  return n === 'shell' || n === 'bash' || n === 'bash_session' || n === 'shell_command' || n === 'job_wait' || l === 'run';
22
22
  }
23
23
 
24
- export function shellDisplayStatus({ pending = false, failedCount = 0, isError = false, result = '' } = {}) {
24
+ export function shellDisplayStatus({ pending = false, failedCount = 0, exitFailedCount = 0, isError = false, result = '' } = {}) {
25
25
  const status = shellResultStatus(result);
26
26
  if (pending || /^(running|pending|queued)$/.test(status)) return 'running';
27
27
  if (/^cancel/.test(status)) return 'cancelled';
28
- if (/^(failed|error|killed|timeout)$/.test(status) || isError || failedCount > 0) return 'failed';
28
+ if (/^(failed|error|killed|timeout)$/.test(status)) return 'failed';
29
+ // A command that RAN but exited non-zero is a command-exit, not a real
30
+ // failure: render the neutral "Exit" state unless there is ALSO a real
31
+ // tool-call/result failure in the group.
32
+ const realFailed = Math.max(0, Number(failedCount) - Number(exitFailedCount));
33
+ if (realFailed > 0) return 'failed';
34
+ if (Number(exitFailedCount) > 0) return 'exit';
35
+ if (isError || failedCount > 0) return 'failed';
29
36
  return 'completed';
30
37
  }
31
38
 
@@ -98,7 +105,9 @@ export function joinActionAgent(action, agent) {
98
105
  return agent ? `${action} ${agent}` : action;
99
106
  }
100
107
 
101
- export function agentResponseTitle(args) {
108
+ export function agentResponseTitle(args, count = 1) {
109
+ const total = Math.max(1, Number(count) || 1);
110
+ if (total > 1) return `Responses ${total} agents`;
102
111
  const name = titleizeAgentName(args?.agent || args?.subagent_type || args?.name || '');
103
112
  // The agent + model identify the responder; the response summary itself
104
113
  // is hidden in the collapsed card (ctrl+o expand still shows the full body).
@@ -392,14 +401,19 @@ export function clampFailureCount(errorCount, groupCount, isError) {
392
401
  // cancelled -> theme.warning
393
402
  // The RED/orange failure color is driven ONLY by real tool-call errors
394
403
  // (`callFailedCount` — backend isError / error toolKind), NOT by command/result
395
- // failures like a shell non-zero exit or a `[status: failed]` result. Those
396
- // keep the card's L2 detail showing "Failed" but leave the dot on the success
397
- // color. `terminalStatus` is still consulted so a cancelled card stays warning.
398
- export function toolStatusColor({ pending, groupCount, callFailedCount = 0, terminalStatus = '' }) {
404
+ // failures like a `[status: failed]` result. A shell command-exit
405
+ // (`exitFailedCount`) is its own distinct neutral state: warning color, never
406
+ // red. `terminalStatus` is still consulted so a cancelled card stays warning.
407
+ export function toolStatusColor({ pending, groupCount, callFailedCount = 0, exitFailedCount = 0, terminalStatus = '' }) {
399
408
  if (pending) return theme.text;
400
409
  const status = normalizeTerminalStatus(terminalStatus);
401
410
  if (status === 'cancelled') return theme.warning;
402
- if (callFailedCount <= 0) return theme.success;
403
- if (groupCount > 1 && callFailedCount < groupCount) return theme.mixdogOrange || theme.warning;
404
- return theme.error;
411
+ if (status === 'denied') return theme.warning;
412
+ if (callFailedCount > 0) {
413
+ if (groupCount > 1 && callFailedCount < groupCount) return theme.mixdogOrange || theme.warning;
414
+ return theme.error;
415
+ }
416
+ // Command-exit(s) with no real tool-call failure: distinct warning state.
417
+ if (exitFailedCount > 0) return theme.warning;
418
+ return theme.success;
405
419
  }