@phnx-labs/agents-cli 1.22.57 → 1.22.59

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (152) hide show
  1. package/CHANGELOG.md +294 -0
  2. package/README.md +29 -0
  3. package/dist/bootstrap.js +39 -1
  4. package/dist/commands/accounts.js +7 -3
  5. package/dist/commands/apply.js +10 -2
  6. package/dist/commands/fork.d.ts +23 -10
  7. package/dist/commands/fork.js +115 -58
  8. package/dist/commands/monitors.js +198 -23
  9. package/dist/commands/prune.js +5 -3
  10. package/dist/commands/routines.d.ts +8 -0
  11. package/dist/commands/routines.js +57 -3
  12. package/dist/commands/routines.test-fixture.js +5 -0
  13. package/dist/commands/send.d.ts +2 -1
  14. package/dist/commands/send.js +7 -5
  15. package/dist/commands/sessions-picker.d.ts +11 -0
  16. package/dist/commands/sessions-picker.js +16 -0
  17. package/dist/commands/sessions-stats.js +37 -5
  18. package/dist/commands/sessions.js +40 -5
  19. package/dist/commands/share.d.ts +14 -0
  20. package/dist/commands/share.js +43 -2
  21. package/dist/commands/ssh.js +12 -1
  22. package/dist/commands/status.js +1 -1
  23. package/dist/commands/sync.js +83 -7
  24. package/dist/commands/traces.js +7 -0
  25. package/dist/commands/versions.js +12 -4
  26. package/dist/commands/view.js +7 -2
  27. package/dist/index.d.ts +1 -1
  28. package/dist/index.js +6 -1
  29. package/dist/lib/account-registry.d.ts +5 -1
  30. package/dist/lib/account-registry.js +47 -14
  31. package/dist/lib/accounting/capacity.d.ts +18 -7
  32. package/dist/lib/accounting/capacity.js +19 -8
  33. package/dist/lib/accounting/usage-sync.d.ts +29 -1
  34. package/dist/lib/accounting/usage-sync.js +76 -2
  35. package/dist/lib/accounting/usage.js +7 -1
  36. package/dist/lib/auth-mint.d.ts +11 -1
  37. package/dist/lib/auth-mint.js +21 -6
  38. package/dist/lib/auto-pull-worker.js +7 -2
  39. package/dist/lib/browser/ipc.d.ts +8 -0
  40. package/dist/lib/browser/ipc.js +87 -0
  41. package/dist/lib/browser/service.d.ts +19 -0
  42. package/dist/lib/browser/service.js +96 -11
  43. package/dist/lib/browser/sessions-list.js +10 -1
  44. package/dist/lib/cloud/rush.d.ts +7 -0
  45. package/dist/lib/cloud/rush.js +29 -1
  46. package/dist/lib/daemon/daemon.d.ts +22 -0
  47. package/dist/lib/daemon/daemon.js +39 -0
  48. package/dist/lib/daemon/runner.d.ts +3 -0
  49. package/dist/lib/daemon/runner.js +86 -45
  50. package/dist/lib/daemon/session-index-service.js +9 -1
  51. package/dist/lib/daemon/usage-sync-service.d.ts +3 -3
  52. package/dist/lib/daemon/usage-sync-service.js +14 -8
  53. package/dist/lib/daemon-services.js +1 -1
  54. package/dist/lib/daemon-ticks.d.ts +15 -0
  55. package/dist/lib/daemon-ticks.js +26 -0
  56. package/dist/lib/device-config.d.ts +5 -1
  57. package/dist/lib/device-config.js +2 -2
  58. package/dist/lib/devices/connect.d.ts +17 -8
  59. package/dist/lib/devices/connect.js +31 -14
  60. package/dist/lib/devices/health.js +5 -1
  61. package/dist/lib/devices/pool.d.ts +25 -2
  62. package/dist/lib/devices/pool.js +32 -2
  63. package/dist/lib/devices/stats-cache.d.ts +0 -6
  64. package/dist/lib/devices/stats-cache.js +2 -9
  65. package/dist/lib/doctor-diff.d.ts +14 -0
  66. package/dist/lib/doctor-diff.js +120 -9
  67. package/dist/lib/fleet/manifest.d.ts +17 -0
  68. package/dist/lib/fleet/manifest.js +26 -0
  69. package/dist/lib/git.d.ts +38 -0
  70. package/dist/lib/git.js +58 -0
  71. package/dist/lib/hooks/install.d.ts +27 -11
  72. package/dist/lib/hooks/install.js +42 -17
  73. package/dist/lib/hosts/ready.d.ts +8 -0
  74. package/dist/lib/hosts/ready.js +13 -2
  75. package/dist/lib/hosts/reconnect.d.ts +52 -203
  76. package/dist/lib/hosts/reconnect.js +64 -284
  77. package/dist/lib/installations/migrate.d.ts +6 -120
  78. package/dist/lib/installations/migrate.js +27 -259
  79. package/dist/lib/installations/shims.d.ts +13 -95
  80. package/dist/lib/installations/shims.js +22 -139
  81. package/dist/lib/installations/store.js +1 -1
  82. package/dist/lib/installations/versions.d.ts +43 -133
  83. package/dist/lib/installations/versions.js +94 -206
  84. package/dist/lib/monitors/config.d.ts +71 -3
  85. package/dist/lib/monitors/config.js +100 -12
  86. package/dist/lib/monitors/pid-watch.d.ts +35 -0
  87. package/dist/lib/monitors/pid-watch.js +45 -0
  88. package/dist/lib/monitors/remote.d.ts +18 -0
  89. package/dist/lib/monitors/remote.js +11 -0
  90. package/dist/lib/permissions.js +7 -2
  91. package/dist/lib/plugins/plugins.d.ts +17 -3
  92. package/dist/lib/plugins/plugins.js +84 -9
  93. package/dist/lib/plugins/skills.d.ts +8 -1
  94. package/dist/lib/plugins/skills.js +18 -2
  95. package/dist/lib/pty-server.d.ts +14 -0
  96. package/dist/lib/pty-server.js +49 -5
  97. package/dist/lib/refresh.d.ts +9 -0
  98. package/dist/lib/refresh.js +3 -1
  99. package/dist/lib/routine-readiness.d.ts +15 -1
  100. package/dist/lib/routine-readiness.js +41 -0
  101. package/dist/lib/sandbox.d.ts +4 -1
  102. package/dist/lib/sandbox.js +30 -1
  103. package/dist/lib/secrets/agent.d.ts +80 -225
  104. package/dist/lib/secrets/agent.js +139 -401
  105. package/dist/lib/secrets/bundles.d.ts +73 -222
  106. package/dist/lib/secrets/bundles.js +168 -467
  107. package/dist/lib/secrets/drivers/rush.js +5 -0
  108. package/dist/lib/secrets/reaper.d.ts +28 -70
  109. package/dist/lib/secrets/reaper.js +30 -85
  110. package/dist/lib/secrets/remote.d.ts +42 -129
  111. package/dist/lib/secrets/remote.js +55 -173
  112. package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
  113. package/dist/lib/self-heal/checks/install-staging.js +96 -0
  114. package/dist/lib/self-heal/registry.js +2 -0
  115. package/dist/lib/self-heal/types.d.ts +1 -1
  116. package/dist/lib/self-update.d.ts +65 -0
  117. package/dist/lib/self-update.js +138 -0
  118. package/dist/lib/session/active.d.ts +13 -1
  119. package/dist/lib/session/active.js +2 -0
  120. package/dist/lib/session/cloud.js +5 -0
  121. package/dist/lib/session/db.d.ts +51 -6
  122. package/dist/lib/session/db.js +266 -20
  123. package/dist/lib/session/fork.d.ts +45 -26
  124. package/dist/lib/session/fork.js +32 -95
  125. package/dist/lib/session/tool-calls.d.ts +43 -1
  126. package/dist/lib/session/tool-calls.js +74 -44
  127. package/dist/lib/session/tool-store.d.ts +33 -2
  128. package/dist/lib/session/tool-store.js +56 -3
  129. package/dist/lib/smart-launch.d.ts +6 -0
  130. package/dist/lib/smart-launch.js +5 -2
  131. package/dist/lib/staleness/writers/plugins.js +5 -2
  132. package/dist/lib/staleness/writers/sources.d.ts +5 -0
  133. package/dist/lib/staleness/writers/sources.js +2 -1
  134. package/dist/lib/staleness/writers/subagents.js +13 -3
  135. package/dist/lib/state.d.ts +7 -4
  136. package/dist/lib/state.js +7 -4
  137. package/dist/lib/subagents.js +8 -2
  138. package/dist/lib/sync-status.d.ts +22 -0
  139. package/dist/lib/sync-status.js +27 -0
  140. package/dist/lib/sync-umbrella.d.ts +9 -0
  141. package/dist/lib/sync-umbrella.js +21 -2
  142. package/dist/lib/teams/scheduler.d.ts +10 -0
  143. package/dist/lib/teams/scheduler.js +8 -0
  144. package/dist/lib/traces/insights.d.ts +47 -14
  145. package/dist/lib/traces/insights.js +92 -21
  146. package/dist/lib/traces/phenotype.d.ts +23 -3
  147. package/dist/lib/traces/phenotype.js +72 -24
  148. package/dist/lib/traces/sync.d.ts +128 -6
  149. package/dist/lib/traces/sync.js +294 -35
  150. package/dist/lib/traces/worker-template.js +154 -1
  151. package/dist/lib/view-types.d.ts +12 -0
  152. package/package.json +2 -2
@@ -49,6 +49,14 @@ export interface SyncResult {
49
49
  parseFailed: number;
50
50
  /** Parsed fine but the upload PUT failed (network/5xx) — retried on the next sync. */
51
51
  uploadFailed: number;
52
+ /**
53
+ * The index shard build or upload failed. Undefined on success. The per-session
54
+ * data still uploaded (that loop runs first), but the aggregated console shard —
55
+ * stats, needs-attention, failure clusters, latency — was NOT refreshed, so the
56
+ * console keeps serving the last good index. Surfaced (not swallowed) so a stale
57
+ * console is diagnosable instead of looking like a clean sync. See PHNX-3401.
58
+ */
59
+ indexError?: string;
52
60
  }
53
61
  /** Push derived, redacted trajectories for this device to the traces store. */
54
62
  export declare function syncTraces(opts?: SyncOpts): Promise<SyncResult>;
@@ -59,18 +67,46 @@ export interface TracesIndexShard {
59
67
  owner: string;
60
68
  stats: {
61
69
  sessionsImported: number;
70
+ /**
71
+ * Median ACTIVE duration (span − idle gaps > 120s), ms — the meaningful figure
72
+ * (PHNX-3457). Same key/shape as before this change, so the fleet-aggregate
73
+ * worker (`worker-template.ts`) keeps weighted-averaging it unchanged; only its
74
+ * VALUE moved from raw span to active time. The raw span stays available per
75
+ * session on `SessionDetail.meta.spanMs`.
76
+ */
62
77
  medianMs: number;
78
+ /** p90 ACTIVE duration, ms. */
63
79
  p90Ms: number;
80
+ /**
81
+ * SEGMENTED active-time stats (PHNX-3472). The blended `medianMs`/`p90Ms`
82
+ * above conflate one-shot interactive queries (63% of the corpus, ~15s
83
+ * median) with substantial agent runs (~15min median), so they headline
84
+ * neither. A session is an AGENT run when it made any tool call OR has more
85
+ * than 8 messages; otherwise INTERACTIVE. These segment the same active-time
86
+ * figure so the console can headline agent runs on their own axis. Each is
87
+ * computed only over sessions with a non-null duration.
88
+ */
89
+ agentMedianMs: number;
90
+ /** p90 ACTIVE duration over AGENT sessions, ms. */
91
+ agentP90Ms: number;
92
+ /** Median ACTIVE duration over INTERACTIVE sessions, ms. */
93
+ interactiveMedianMs: number;
94
+ /** (sessions with a non-null duration) / (total sessions), 0..1 — coverage of the duration stats. */
95
+ measuredFraction: number;
64
96
  needAttention: number;
65
97
  toolErrorRate: number;
66
98
  };
99
+ /**
100
+ * Sessions excluded from the eval corpus as internal utility plumbing (PHNX-3474):
101
+ * single-shot machine calls (no tool call AND ≤2 messages) or a known
102
+ * internal-prompt signature (title generation, watchdog, commit-message, factory
103
+ * worker). Every `stats` figure above, `topics` counts, and `needsAttention` are
104
+ * computed over the AGENT set ONLY — `sessionsImported` is the real agent count,
105
+ * not the raw row count. This is the number that was dropped.
106
+ */
107
+ utilityCount: number;
67
108
  needsAttention: IndexedSession[];
68
- topics: Array<{
69
- key: string;
70
- label: string;
71
- count: number;
72
- group: TraceTopicGroup;
73
- }>;
109
+ topics: TopicItem[];
74
110
  failures: {
75
111
  byToolError: Array<{
76
112
  tool: string;
@@ -96,16 +132,68 @@ export interface IndexedSession {
96
132
  title: string;
97
133
  repo: string;
98
134
  device: string;
135
+ /** The harness that produced the session (claude/codex/rush/grok/…). Same as `harness`. */
99
136
  agent: string;
100
137
  model: string;
138
+ /** Corpus classification (PHNX-3474). Always `'agent'` here — utility rows are excluded. */
139
+ kind: SessionKind;
101
140
  severity: number;
102
141
  flags: string[];
103
142
  }
143
+ /**
144
+ * One example session under a topic tile — the shape the console drill-down consumes.
145
+ * Carries `kind` + `harness` (PHNX-3474) so the console can filter a tile's session
146
+ * list by corpus class and by harness. Refs on a topic tile are always `'agent'`
147
+ * (utility rows never reach a bucket), but the field is explicit for the consumer.
148
+ */
149
+ export interface TopicSessionRef {
150
+ id: string;
151
+ title: string;
152
+ kind: SessionKind;
153
+ harness: string;
154
+ }
155
+ /**
156
+ * Corpus class of a session (PHNX-3474). `utility` is internal machine plumbing —
157
+ * a single-shot call with no tool use and ≤2 messages, or one whose topic/label
158
+ * matches a known internal-prompt signature (title generation, watchdog,
159
+ * commit-message writer, factory worker). Everything else is `agent`: real agent
160
+ * work the Evals console counts and scores. Utility rows are tagged, never deleted,
161
+ * and excluded from every index statistic.
162
+ */
163
+ export type SessionKind = 'utility' | 'agent';
164
+ /**
165
+ * Classify a session as internal `utility` plumbing vs real `agent` work (PHNX-3474).
166
+ * `toolCallCount` is the AUTHORITATIVE per-session tool-call count from the loaded
167
+ * `tool_calls` rows (the row's own `tool_call_count` column is a fallback for a row
168
+ * whose calls weren't loaded). A session is `utility` when a known internal-prompt
169
+ * signature matches its topic/label, OR it made no tool call AND has ≤2 messages —
170
+ * the single-shot machine-call shape. Otherwise it is `agent`.
171
+ */
172
+ export declare function classifySessionKind(row: Pick<SyncRow, 'topic' | 'label' | 'message_count' | 'tool_call_count'>, toolCallCount: number): SessionKind;
173
+ /**
174
+ * One topic bucket in the treemap. `sessions` carries up to {@link TOPIC_SESSION_CAP}
175
+ * example refs so the console can drill from the tile into its session list — a tile
176
+ * with no refs renders display-only (PHNX-3408). `count` stays the true total.
177
+ */
178
+ export interface TopicItem {
179
+ key: string;
180
+ label: string;
181
+ count: number;
182
+ group: TraceTopicGroup;
183
+ sessions: TopicSessionRef[];
184
+ }
104
185
  /** A row from `tool_calls`. `ordinal`/`timestamp` order calls within a session for computeInsights(). */
105
186
  export interface ToolCallRow {
106
187
  session_id: string;
107
188
  ordinal: number;
108
189
  timestamp: string;
190
+ /**
191
+ * When the call's result arrived — its own end time (PHNX-3437). Optional
192
+ * because rows an older extractor stored, and calls that never produced a
193
+ * result, carry NULL; `computeInsights` falls back to the bounded inter-call
194
+ * gap when it is absent.
195
+ */
196
+ end_timestamp?: string | null;
109
197
  tool: string;
110
198
  outcome: string;
111
199
  exit_code: number | null;
@@ -114,6 +202,31 @@ export interface ToolCallRow {
114
202
  error: string | null;
115
203
  parse_error: string | null;
116
204
  }
205
+ /**
206
+ * Active time for a session in the index shard: its recorded span minus every idle
207
+ * gap > 120s (PHNX-3457). The index build already holds the ordered `tool_calls`
208
+ * rows for the whole corpus, so idle is derived from them here — no transcript
209
+ * re-parse. A cursor sweeps the span from `sessionStartMs`: each stretch where the
210
+ * cursor sits idle for more than the threshold before the next call starts is
211
+ * subtracted, and idle is measured from a call's END (its own `end_timestamp` when
212
+ * known, else its start) so a call's own blocking duration is never mistaken for
213
+ * idle. Crucially the sweep also books the gaps at the two BOUNDARIES — before the
214
+ * first call and after the last call to the session end — so a session with a lone
215
+ * tool call that was then abandoned and resumed hours later (the case a
216
+ * between-calls-only measure missed entirely, leaving the whole 345h span counted
217
+ * as active) has that trailing idle stripped. A session end is `sessionStartMs +
218
+ * spanMs`, so the two agree by construction.
219
+ *
220
+ * Bounded to `[0, spanMs]`. A session with NO tool calls returns the full span
221
+ * unchanged rather than a fabricated zero: there is no tool-call evidence of idle
222
+ * either way, and treating a chat-only turn as 100% idle would be a worse error
223
+ * than leaving its span uncorrected. Where the full event stream IS available (a
224
+ * per-session `SessionDetail`), {@link activeMsFromTrajectory} is used instead —
225
+ * it sees message events this call-only approximation cannot, so the two are close
226
+ * but not identical by design (the corpus-scale index build cannot afford the
227
+ * per-session parse the detail view does).
228
+ */
229
+ export declare function sessionActiveMs(spanMs: number, sessionCalls: ToolCallRow[], sessionStartMs: number): number;
117
230
  /** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
118
231
  export declare function failureDescription(call: ToolCallRow, cause: TraceFailureCause): string;
119
232
  /** Build the redacted rich console shard from indexed metadata and derived caches. */
@@ -123,7 +236,16 @@ export interface SessionDetail {
123
236
  schema: 1;
124
237
  id: string;
125
238
  meta: {
239
+ /** Raw wall-clock span (last event − first event), idle time included. */
126
240
  spanMs: number;
241
+ /**
242
+ * Active time: `spanMs` minus every idle gap > 120s (PHNX-3457). A session
243
+ * resumed hours later, or left idle mid-turn, inflates `spanMs` with wall-clock
244
+ * the agent did no work in — active time strips those gaps so a duration reads
245
+ * as effort, not calendar span. This is what the console's duration median/p90
246
+ * should trust; `spanMs` stays available as the raw figure.
247
+ */
248
+ activeMs: number;
127
249
  turns: number;
128
250
  tools: number;
129
251
  errorCount: number;
@@ -22,7 +22,7 @@
22
22
  import fs from 'node:fs';
23
23
  import os from 'node:os';
24
24
  import path from 'node:path';
25
- import { getDB, readSessionInsights, readSessionTopics, writeSessionInsights, writeSessionTopics, } from '../session/db.js';
25
+ import { getDB, readSessionInsights, readSessionPhenotypes, readSessionTopics, writeSessionInsights, writeSessionPhenotypes, writeSessionTopics, } from '../session/db.js';
26
26
  import { parseSession } from '../session/parse.js';
27
27
  import { buildTrajectory } from '../session/trajectory.js';
28
28
  import { computeInsightFacets } from '../session/insights.js';
@@ -31,6 +31,7 @@ import { getRuntimeStateDir } from '../state.js';
31
31
  import { resolveTracesBackend } from './backend.js';
32
32
  import { classifyCause, classifyTopic, computeDriftSignal, } from './classify.js';
33
33
  import { computeInsights } from './insights.js';
34
+ import { classifyPhenotype, recoveredAfterErrors } from './phenotype.js';
34
35
  /** Push derived, redacted trajectories for this device to the traces store. */
35
36
  export async function syncTraces(opts = {}) {
36
37
  const dryRun = opts.dryRun === true;
@@ -147,6 +148,7 @@ export async function syncTraces(opts = {}) {
147
148
  recordFailure(row, 'upload-failed', err);
148
149
  }
149
150
  }
151
+ let indexError;
150
152
  if (!opts.skipIndex) {
151
153
  try {
152
154
  const allRows = db
@@ -172,8 +174,13 @@ export async function syncTraces(opts = {}) {
172
174
  await putIndexShard(backend, device, owner, shard);
173
175
  }
174
176
  }
175
- catch {
176
- // index PUT/write failure is not fatal — the per-session data is already written
177
+ catch (err) {
178
+ // Not fatal to the per-session upload (that loop already ran), but it DOES
179
+ // mean the console shard is now stale. Record it so the caller can surface a
180
+ // warning instead of reporting a clean, green sync — a silent swallow here is
181
+ // exactly what let a 59h-stale, insight-less index hide in plain sight
182
+ // (PHNX-3401). Do NOT re-throw: the session data is durable and worth keeping.
183
+ indexError = err instanceof Error ? err.message : String(err);
177
184
  }
178
185
  }
179
186
  // A dry-run never advances the incremental watermark: it is a read-only export.
@@ -203,14 +210,112 @@ export async function syncTraces(opts = {}) {
203
210
  transcriptUnavailable,
204
211
  parseFailed,
205
212
  uploadFailed,
213
+ indexError,
206
214
  };
207
215
  }
216
+ /**
217
+ * Topic/label substrings that identify an internal-prompt session regardless of its
218
+ * message/tool shape. These are the harness-spawned utility prompts the Rush app
219
+ * fires (they run under the `claude` harness): title generation writes the 3–4 word
220
+ * session title, the watchdog polls for stalled agents, the commit-message writer
221
+ * drafts a conventional commit, and factory workers are dispatched sub-agents. The
222
+ * title-generation prompt lives in the `topic` column, the rest can land in either
223
+ * `topic` or `label`, so both are matched.
224
+ */
225
+ const UTILITY_PROMPT_SIGNATURES = [
226
+ /generate a 3-4 word title/i, // title generation
227
+ /you are a watchdog|watchdog monitoring/i, // watchdog tick
228
+ /conventional[- ]commit/i, // commit-message writer
229
+ /factory worker/i, // dispatched factory worker
230
+ ];
231
+ /**
232
+ * Classify a session as internal `utility` plumbing vs real `agent` work (PHNX-3474).
233
+ * `toolCallCount` is the AUTHORITATIVE per-session tool-call count from the loaded
234
+ * `tool_calls` rows (the row's own `tool_call_count` column is a fallback for a row
235
+ * whose calls weren't loaded). A session is `utility` when a known internal-prompt
236
+ * signature matches its topic/label, OR it made no tool call AND has ≤2 messages —
237
+ * the single-shot machine-call shape. Otherwise it is `agent`.
238
+ */
239
+ export function classifySessionKind(row, toolCallCount) {
240
+ const haystack = `${row.topic ?? ''}\n${row.label ?? ''}`;
241
+ if (UTILITY_PROMPT_SIGNATURES.some((re) => re.test(haystack)))
242
+ return 'utility';
243
+ const hasToolCalls = toolCallCount > 0 || (row.tool_call_count ?? 0) > 0;
244
+ const messages = row.message_count ?? 0;
245
+ if (!hasToolCalls && messages <= 2)
246
+ return 'utility';
247
+ return 'agent';
248
+ }
208
249
  function percentile(values, ratio) {
209
250
  if (values.length === 0)
210
251
  return 0;
211
252
  const sorted = [...values].sort((a, b) => a - b);
212
253
  return sorted[Math.min(sorted.length - 1, Math.ceil(sorted.length * ratio) - 1)];
213
254
  }
255
+ /**
256
+ * A delta between consecutive events longer than this reads as an idle stall, not
257
+ * work — the same threshold the trajectory uses for its gap detection
258
+ * (`DEFAULT_IDLE_THRESHOLD_MS`, trajectory.ts). Kept in lockstep so active time
259
+ * here and the gaps drawn in a session's detail view agree on what "idle" means.
260
+ */
261
+ const IDLE_GAP_THRESHOLD_MS = 120_000;
262
+ /** Max example session refs carried per topic tile so a tile is drillable (PHNX-3408). */
263
+ const TOPIC_SESSION_CAP = 30;
264
+ /**
265
+ * Active time for a session in the index shard: its recorded span minus every idle
266
+ * gap > 120s (PHNX-3457). The index build already holds the ordered `tool_calls`
267
+ * rows for the whole corpus, so idle is derived from them here — no transcript
268
+ * re-parse. A cursor sweeps the span from `sessionStartMs`: each stretch where the
269
+ * cursor sits idle for more than the threshold before the next call starts is
270
+ * subtracted, and idle is measured from a call's END (its own `end_timestamp` when
271
+ * known, else its start) so a call's own blocking duration is never mistaken for
272
+ * idle. Crucially the sweep also books the gaps at the two BOUNDARIES — before the
273
+ * first call and after the last call to the session end — so a session with a lone
274
+ * tool call that was then abandoned and resumed hours later (the case a
275
+ * between-calls-only measure missed entirely, leaving the whole 345h span counted
276
+ * as active) has that trailing idle stripped. A session end is `sessionStartMs +
277
+ * spanMs`, so the two agree by construction.
278
+ *
279
+ * Bounded to `[0, spanMs]`. A session with NO tool calls returns the full span
280
+ * unchanged rather than a fabricated zero: there is no tool-call evidence of idle
281
+ * either way, and treating a chat-only turn as 100% idle would be a worse error
282
+ * than leaving its span uncorrected. Where the full event stream IS available (a
283
+ * per-session `SessionDetail`), {@link activeMsFromTrajectory} is used instead —
284
+ * it sees message events this call-only approximation cannot, so the two are close
285
+ * but not identical by design (the corpus-scale index build cannot afford the
286
+ * per-session parse the detail view does).
287
+ */
288
+ export function sessionActiveMs(spanMs, sessionCalls, sessionStartMs) {
289
+ if (spanMs <= 0)
290
+ return Math.max(0, spanMs);
291
+ if (!Number.isFinite(sessionStartMs))
292
+ return spanMs; // can't place calls on the span
293
+ const spanEndMs = sessionStartMs + spanMs;
294
+ const ordered = sessionCalls
295
+ .map((c) => ({ startMs: Date.parse(c.timestamp), endMs: Date.parse(c.end_timestamp ?? c.timestamp) }))
296
+ .filter((c) => Number.isFinite(c.startMs))
297
+ .sort((a, b) => a.startMs - b.startMs);
298
+ if (ordered.length === 0)
299
+ return spanMs; // no tool-call evidence of idle
300
+ let idleMs = 0;
301
+ let cursor = sessionStartMs;
302
+ for (const call of ordered) {
303
+ if (call.startMs > cursor + IDLE_GAP_THRESHOLD_MS)
304
+ idleMs += call.startMs - cursor;
305
+ const endMs = Number.isFinite(call.endMs) ? Math.max(call.endMs, call.startMs) : call.startMs;
306
+ if (endMs > cursor)
307
+ cursor = endMs;
308
+ }
309
+ // Trailing idle: the stretch from the last call's end to the session's end.
310
+ if (spanEndMs > cursor + IDLE_GAP_THRESHOLD_MS)
311
+ idleMs += spanEndMs - cursor;
312
+ return Math.max(0, spanMs - Math.min(idleMs, spanMs));
313
+ }
314
+ /** Active time from an already-built trajectory: span minus its idle gaps (all > threshold). */
315
+ function activeMsFromTrajectory(traj) {
316
+ const idleMs = traj.gaps.reduce((sum, gap) => sum + gap.durationMs, 0);
317
+ return Math.max(0, traj.spanMs - Math.min(idleMs, traj.spanMs));
318
+ }
214
319
  /** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
215
320
  export function failureDescription(call, cause) {
216
321
  if (cause === 'guard')
@@ -247,6 +352,32 @@ function attentionFlags(errorCount, facets) {
247
352
  flags.push(`${correctionCount} correction${correctionCount === 1 ? '' : 's'}`);
248
353
  return flags;
249
354
  }
355
+ /**
356
+ * Persist a derived-cache warm-up (topics / insights) without letting a
357
+ * contended DB take down the whole index build.
358
+ *
359
+ * These write-backs only speed up the NEXT sync — the shard about to be built
360
+ * reads from the in-memory `topics` / `insights` maps that were already
361
+ * populated above, never from what this write persists. So the write is
362
+ * genuinely optional to the shard's correctness.
363
+ *
364
+ * Yet it was the single point that broke the console: on an active machine the
365
+ * Rush app holds `sessions.db`, this `BEGIN IMMEDIATE` waits out the 30s
366
+ * `busy_timeout` and throws `SQLITE_BUSY`, the throw escaped `buildIndexShard`,
367
+ * and `syncTraces` swallowed it — so the index (with wasted-time / failure
368
+ * clusters / latency) never re-uploaded and the dashboard sat 59h stale
369
+ * (PHNX-3401). Isolating the failure here keeps the index building; the warning
370
+ * makes the degraded cache visible instead of silent.
371
+ */
372
+ function persistDerivedCache(label, write) {
373
+ try {
374
+ write();
375
+ }
376
+ catch (err) {
377
+ const msg = err instanceof Error ? err.message : String(err);
378
+ console.warn(`traces: ${label} cache warm-up skipped (${msg}) — index still built`);
379
+ }
380
+ }
250
381
  /** Build the redacted rich console shard from indexed metadata and derived caches. */
251
382
  export function buildIndexShard(rows, device, owner, prevShard) {
252
383
  const db = getDB();
@@ -256,13 +387,14 @@ export function buildIndexShard(rows, device, owner, prevShard) {
256
387
  for (let i = 0; i < ids.length; i += 400) {
257
388
  const chunk = ids.slice(i, i + 400);
258
389
  calls.push(...db.prepare(`
259
- SELECT session_id, ordinal, timestamp, tool, outcome, exit_code, status_code, error_code, error, parse_error
390
+ SELECT session_id, ordinal, timestamp, end_timestamp, tool, outcome, exit_code, status_code, error_code, error, parse_error
260
391
  FROM tool_calls
261
392
  WHERE session_id IN (${chunk.map(() => '?').join(',')})
262
393
  `).all(...chunk));
263
394
  }
264
395
  const toolMix = new Map();
265
396
  const errorCounts = new Map();
397
+ const callsBySession = new Map();
266
398
  for (const call of calls) {
267
399
  const mix = toolMix.get(call.session_id) ?? {};
268
400
  mix[call.tool] = (mix[call.tool] ?? 0) + 1;
@@ -270,9 +402,27 @@ export function buildIndexShard(rows, device, owner, prevShard) {
270
402
  if (call.outcome === 'error') {
271
403
  errorCounts.set(call.session_id, (errorCounts.get(call.session_id) ?? 0) + 1);
272
404
  }
405
+ const list = callsBySession.get(call.session_id);
406
+ if (list)
407
+ list.push(call);
408
+ else
409
+ callsBySession.set(call.session_id, [call]);
273
410
  }
274
- const topics = readSessionTopics(ids);
275
- const missingTopics = rows.filter((row) => !topics.has(row.id)).map((row) => {
411
+ // Classify every row as real `agent` work or internal `utility` plumbing, then run
412
+ // the ENTIRE rest of the shard build over the agent set ONLY (PHNX-3474). Utility
413
+ // rows — title-gen / watchdog / commit-message / factory-worker calls, ~68% of the
414
+ // corpus — are tagged and excluded here, never deleted from sessions.db, so the
415
+ // console counts and scores real agent work: `sessionsImported`, the medians,
416
+ // needs-attention, tool-error-rate, and the topic buckets all measure `agentRows`.
417
+ const kindOf = new Map(rows.map((row) => [row.id, classifySessionKind(row, callsBySession.get(row.id)?.length ?? 0)]));
418
+ const agentRows = rows.filter((row) => kindOf.get(row.id) === 'agent');
419
+ const agentIds = agentRows.map((row) => row.id);
420
+ const utilityCount = rows.length - agentRows.length;
421
+ // Tool calls belonging to utility rows never contribute to the failure/latency
422
+ // stats or the tool-error rate — filter them out at the source alongside the rows.
423
+ const agentCalls = calls.filter((call) => kindOf.get(call.session_id) === 'agent');
424
+ const topics = readSessionTopics(agentIds);
425
+ const missingTopics = agentRows.filter((row) => !topics.has(row.id)).map((row) => {
276
426
  const topic = classifyTopic({
277
427
  cwd: row.cwd,
278
428
  gitBranch: row.git_branch,
@@ -283,27 +433,53 @@ export function buildIndexShard(rows, device, owner, prevShard) {
283
433
  topics.set(row.id, topic);
284
434
  return { id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, topic };
285
435
  });
286
- writeSessionTopics(missingTopics);
287
- const insights = readSessionInsights(ids);
436
+ persistDerivedCache('session-topics', () => writeSessionTopics(missingTopics));
437
+ // Insights (frictionSignals) and phenotype (false-termination / …) both need
438
+ // the parsed transcript, which flat tool_calls rows don't carry — so both are
439
+ // lazily derived per-session and cached by transcript mtime+size, then read for
440
+ // the WHOLE corpus (`rows` = allRows) every sync. That full-corpus read is what
441
+ // keeps the phenotype grouping dimension consistent: a session synced weeks ago
442
+ // still contributes its real phenotype from cache, so it can never fragment away
443
+ // from an identically-signatured session synced this run purely by *when* each
444
+ // was first seen (PHNX-3327). A cache-miss row is parsed at most once here even
445
+ // when both derivations are missing.
446
+ const insights = readSessionInsights(agentIds);
447
+ const phenotypes = readSessionPhenotypes(agentIds);
288
448
  const missingInsights = [];
289
- for (const row of rows.filter((candidate) => !insights.has(candidate.id))) {
449
+ const missingPhenotypes = [];
450
+ for (const row of agentRows) {
451
+ const needInsights = !insights.has(row.id);
452
+ const needPhenotype = !phenotypes.has(row.id);
453
+ if (!needInsights && !needPhenotype)
454
+ continue;
455
+ let events;
290
456
  try {
291
- const events = parseSession(row.file_path, row.agent);
292
- const facets = computeInsightFacets(events);
293
- insights.set(row.id, facets);
294
- missingInsights.push({
295
- id: row.id,
296
- fileMtimeMs: row.file_mtime_ms,
297
- fileSize: row.file_size,
298
- facets,
299
- });
457
+ events = parseSession(row.file_path, row.agent);
300
458
  }
301
459
  catch {
302
- continue;
460
+ continue; // gone/unreadable transcript — leave both uncached, same as before
461
+ }
462
+ if (needInsights) {
463
+ try {
464
+ const facets = computeInsightFacets(events);
465
+ insights.set(row.id, facets);
466
+ missingInsights.push({ id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, facets });
467
+ }
468
+ catch { /* leave this session's facets uncached; recompute next sync */ }
469
+ }
470
+ if (needPhenotype) {
471
+ try {
472
+ const traj = buildTrajectory(events, rowToMeta(row), { redact: true, knownSecrets });
473
+ const phenotype = classifyPhenotype(buildSessionDetail(traj));
474
+ phenotypes.set(row.id, phenotype);
475
+ missingPhenotypes.push({ id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, phenotype });
476
+ }
477
+ catch { /* leave this session's phenotype uncached; recompute next sync */ }
303
478
  }
304
479
  }
305
- writeSessionInsights(missingInsights);
306
- const needsAttention = rows.flatMap((row) => {
480
+ persistDerivedCache('session-insights', () => writeSessionInsights(missingInsights));
481
+ persistDerivedCache('session-phenotypes', () => writeSessionPhenotypes(missingPhenotypes));
482
+ const needsAttention = agentRows.flatMap((row) => {
307
483
  const facets = insights.get(row.id);
308
484
  const errorCount = errorCounts.get(row.id) ?? 0;
309
485
  const flags = attentionFlags(errorCount, facets);
@@ -318,17 +494,28 @@ export function buildIndexShard(rows, device, owner, prevShard) {
318
494
  device,
319
495
  agent: row.agent,
320
496
  model: row.model ?? 'unknown',
497
+ kind: 'agent',
321
498
  severity: errorCount * 2 + friction * 3 + corrections * 2,
322
499
  flags,
323
500
  }];
324
501
  }).sort((a, b) => b.severity - a.severity || a.id.localeCompare(b.id));
325
502
  const topicCounts = new Map();
326
- for (const topic of topics.values()) {
327
- const current = topicCounts.get(topic.key) ?? { ...topic, count: 0 };
328
- current.count++;
329
- topicCounts.set(topic.key, current);
503
+ for (const row of agentRows) {
504
+ const topic = topics.get(row.id);
505
+ if (!topic)
506
+ continue;
507
+ const bucket = topicCounts.get(topic.key)
508
+ ?? { key: topic.key, label: topic.label, group: topic.group, count: 0, refs: [] };
509
+ bucket.count++;
510
+ bucket.refs.push({
511
+ id: row.id,
512
+ title: redactSecrets(row.label ?? row.topic ?? topic.label ?? 'Untitled session', knownSecrets),
513
+ harness: row.agent,
514
+ recencyMs: Date.parse(row.last_activity ?? row.timestamp) || 0,
515
+ });
516
+ topicCounts.set(topic.key, bucket);
330
517
  }
331
- const failedCalls = calls.filter((call) => call.outcome === 'error');
518
+ const failedCalls = agentCalls.filter((call) => call.outcome === 'error');
332
519
  const byCause = { real: 0, guard: 0, hook: 0 };
333
520
  const failureCounts = new Map();
334
521
  for (const call of failedCalls) {
@@ -340,14 +527,38 @@ export function buildIndexShard(rows, device, owner, prevShard) {
340
527
  current.count++;
341
528
  failureCounts.set(key, current);
342
529
  }
343
- const durations = rows.flatMap((row) => row.duration_ms == null ? [] : [row.duration_ms]);
530
+ // Duration stats run over ACTIVE time, not raw span (PHNX-3457): span minus idle
531
+ // gaps > 120s, derived per session from the tool_calls already loaded above. A
532
+ // session resumed after hours, or left idle mid-turn, otherwise inflates the
533
+ // median/p90 with wall-clock the agent did no work in (real corpus max span:
534
+ // 345h). The raw span stays available per session on `SessionDetail.meta.spanMs`.
535
+ const activeDurations = agentRows.flatMap((row) => row.duration_ms == null
536
+ ? []
537
+ : [sessionActiveMs(row.duration_ms, callsBySession.get(row.id) ?? [], Date.parse(row.timestamp))]);
538
+ // Segment the same active-time figure into AGENT vs INTERACTIVE runs (PHNX-3472).
539
+ // A session is an AGENT run when it made any tool call OR has more than 8
540
+ // messages; otherwise INTERACTIVE (a one-shot query). Only sessions with a
541
+ // non-null duration contribute to the medians; `measuredFraction` reports how
542
+ // much of the corpus that covers.
543
+ const agentActive = [];
544
+ const interactiveActive = [];
545
+ let measured = 0;
546
+ for (const row of agentRows) {
547
+ if (row.duration_ms == null)
548
+ continue;
549
+ measured++;
550
+ const active = sessionActiveMs(row.duration_ms, callsBySession.get(row.id) ?? [], Date.parse(row.timestamp));
551
+ const isAgent = (callsBySession.get(row.id)?.length ?? 0) > 0 || (row.message_count ?? 0) > 8;
552
+ (isAgent ? agentActive : interactiveActive).push(active);
553
+ }
554
+ const measuredFraction = agentRows.length === 0 ? 0 : measured / agentRows.length;
344
555
  // Build today's per-bucket stats for the rolling drift window.
345
556
  const todayDate = new Date().toISOString().slice(0, 10);
346
557
  const todayStats = [...topicCounts.values()].map(({ key }) => {
347
558
  const sessionsInBucket = [...topics.entries()]
348
559
  .filter(([, t]) => t.key === key)
349
560
  .map(([id]) => id);
350
- const bucketCalls = calls.filter((c) => sessionsInBucket.includes(c.session_id));
561
+ const bucketCalls = agentCalls.filter((c) => sessionsInBucket.includes(c.session_id));
351
562
  const bucketErrors = bucketCalls.filter((c) => c.outcome === 'error').length;
352
563
  const errorRate = bucketCalls.length === 0 ? 0 : bucketErrors / bucketCalls.length;
353
564
  const stallCount = sessionsInBucket.filter((id) => {
@@ -361,21 +572,43 @@ export function buildIndexShard(rows, device, owner, prevShard) {
361
572
  const prevHistory = prevShard?.bucketHistory ?? [];
362
573
  const bucketHistory = [...prevHistory, todayStats].slice(-14);
363
574
  const driftSignals = computeDriftSignal(prevHistory, todayStats);
364
- const patternInsights = computeInsights(rows, calls, prevShard);
575
+ const patternInsights = computeInsights(agentRows, agentCalls, prevShard, phenotypes);
365
576
  return {
366
577
  schema: 1,
367
578
  device,
368
579
  syncedAt: Date.now(),
369
580
  owner,
370
581
  stats: {
371
- sessionsImported: rows.length,
372
- medianMs: percentile(durations, 0.5),
373
- p90Ms: percentile(durations, 0.9),
582
+ sessionsImported: agentRows.length,
583
+ medianMs: percentile(activeDurations, 0.5),
584
+ p90Ms: percentile(activeDurations, 0.9),
585
+ agentMedianMs: percentile(agentActive, 0.5),
586
+ agentP90Ms: percentile(agentActive, 0.9),
587
+ interactiveMedianMs: percentile(interactiveActive, 0.5),
588
+ measuredFraction,
374
589
  needAttention: needsAttention.length,
375
- toolErrorRate: calls.length === 0 ? 0 : failedCalls.length / calls.length,
590
+ toolErrorRate: agentCalls.length === 0 ? 0 : failedCalls.length / agentCalls.length,
376
591
  },
592
+ utilityCount,
377
593
  needsAttention,
378
- topics: [...topicCounts.values()].sort((a, b) => b.count - a.count || a.key.localeCompare(b.key)),
594
+ topics: [...topicCounts.values()]
595
+ .sort((a, b) => b.count - a.count || a.key.localeCompare(b.key))
596
+ .map((bucket) => ({
597
+ key: bucket.key,
598
+ label: bucket.label,
599
+ count: bucket.count,
600
+ group: bucket.group,
601
+ // Up to TOPIC_SESSION_CAP most-recent example sessions, so the console can
602
+ // drill from a tile into its session list. Capped to keep the shard small;
603
+ // the tile's `count` remains the true total. These are correct on this
604
+ // per-device shard; the fleet-aggregate `/all` view (worker-template.ts,
605
+ // a must-not-touch R2 worker here) carries only the first device's refs
606
+ // per topic until PHNX-3464 merges them across devices.
607
+ sessions: bucket.refs
608
+ .sort((a, b) => b.recencyMs - a.recencyMs)
609
+ .slice(0, TOPIC_SESSION_CAP)
610
+ .map(({ id, title, harness }) => ({ id, title, kind: 'agent', harness })),
611
+ })),
379
612
  failures: {
380
613
  byToolError: [...failureCounts.values()].sort((a, b) => b.count - a.count || a.tool.localeCompare(b.tool)),
381
614
  byCause,
@@ -404,6 +637,31 @@ function buildWhereItWentWrong(traj) {
404
637
  return null;
405
638
  return `This run hit ${parts.join('; ')}.`;
406
639
  }
640
+ /**
641
+ * Truthful run-level outcome (PHNX-3387).
642
+ *
643
+ * A run with zero tool errors `completed`. A run that hit tool errors is
644
+ * `completed` ONLY when it *causally recovered* — a substantive, non-human-facing
645
+ * tool step succeeded strictly after the last error AND resolved the failed work
646
+ * (its work signature matches an errored step's), the exact predicate the
647
+ * false-termination phenotype uses ({@link recoveredAfterErrors}). A run whose
648
+ * last substantive step is the error, whose only post-error steps are human-facing
649
+ * (a punt to `AskUserQuestion` — the case the broken "last tool call ok" heuristic
650
+ * mislabeled `completed`), or whose only post-error success is unrelated work (a
651
+ * failed `bun test` followed by an incidental `ls`) stays `errored`.
652
+ *
653
+ * This is what makes `surfacedToolFailures` on a `completed` run honest: those
654
+ * are failures the run recovered from, not a green status hiding an unresolved
655
+ * failure. It never flips a run that ended unresolved to `completed` (no
656
+ * regression vs the old `errorCount > 0 ? errored : completed`), and it does not
657
+ * flip a run whose failed work was never resolved just because some later,
658
+ * unrelated call happened to succeed.
659
+ */
660
+ function deriveRunOutcome(traj) {
661
+ if (traj.errorCount === 0)
662
+ return 'completed';
663
+ return recoveredAfterErrors({ steps: traj.steps }) ? 'completed' : 'errored';
664
+ }
407
665
  /**
408
666
  * Map the derived trajectory to the console's SessionDetail shape, stripping
409
667
  * local-machine PII (full cwd, account) that would expose filesystem paths if
@@ -418,12 +676,13 @@ export function buildSessionDetail(traj) {
418
676
  id: s.id,
419
677
  meta: {
420
678
  spanMs: traj.spanMs,
679
+ activeMs: activeMsFromTrajectory(traj),
421
680
  turns: (stats.userTurns ?? 0) + (stats.assistantTurns ?? 0),
422
681
  tools: stats.toolCount ?? 0,
423
682
  errorCount: traj.errorCount,
424
683
  tokens: stats.outputTokens ?? 0,
425
684
  costUsd: s.costUsd ?? 0,
426
- outcome: traj.errorCount > 0 ? 'errored' : 'completed',
685
+ outcome: deriveRunOutcome(traj),
427
686
  repo,
428
687
  agent: s.agent,
429
688
  model: s.model ?? 'unknown',