@phnx-labs/agents-cli 1.22.58 → 1.22.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/CHANGELOG.md +260 -0
  2. package/README.md +29 -0
  3. package/dist/bootstrap.js +32 -1
  4. package/dist/commands/monitors.js +187 -23
  5. package/dist/commands/perf.js +10 -0
  6. package/dist/commands/routines.test-fixture.js +5 -0
  7. package/dist/commands/send.d.ts +2 -1
  8. package/dist/commands/send.js +7 -5
  9. package/dist/commands/sessions-picker.d.ts +13 -0
  10. package/dist/commands/sessions-picker.js +17 -8
  11. package/dist/commands/sessions-stats.js +37 -5
  12. package/dist/commands/sessions.js +52 -16
  13. package/dist/commands/ssh.js +12 -1
  14. package/dist/commands/teams-picker.js +20 -6
  15. package/dist/commands/teams.d.ts +2 -2
  16. package/dist/commands/teams.js +77 -21
  17. package/dist/commands/versions.js +12 -4
  18. package/dist/commands/view.js +7 -2
  19. package/dist/lib/accounting/rotate.js +12 -4
  20. package/dist/lib/accounting/usage-sync.d.ts +12 -2
  21. package/dist/lib/accounting/usage-sync.js +34 -6
  22. package/dist/lib/auto-pull-worker.js +7 -2
  23. package/dist/lib/cloud/rush.d.ts +7 -0
  24. package/dist/lib/cloud/rush.js +29 -1
  25. package/dist/lib/daemon/daemon.d.ts +22 -0
  26. package/dist/lib/daemon/daemon.js +39 -0
  27. package/dist/lib/daemon/session-index-service.js +9 -1
  28. package/dist/lib/daemon-ticks.d.ts +15 -0
  29. package/dist/lib/daemon-ticks.js +26 -0
  30. package/dist/lib/device-config.d.ts +5 -1
  31. package/dist/lib/device-config.js +2 -2
  32. package/dist/lib/devices/health.js +5 -1
  33. package/dist/lib/devices/pool.d.ts +25 -2
  34. package/dist/lib/devices/pool.js +32 -2
  35. package/dist/lib/devices/stats-cache.d.ts +0 -6
  36. package/dist/lib/devices/stats-cache.js +2 -9
  37. package/dist/lib/doctor-diff.d.ts +14 -0
  38. package/dist/lib/doctor-diff.js +43 -2
  39. package/dist/lib/feed/events.js +4 -0
  40. package/dist/lib/git.d.ts +38 -0
  41. package/dist/lib/git.js +58 -0
  42. package/dist/lib/hosts/ready.d.ts +8 -0
  43. package/dist/lib/hosts/ready.js +13 -2
  44. package/dist/lib/installations/versions.d.ts +17 -0
  45. package/dist/lib/installations/versions.js +53 -2
  46. package/dist/lib/monitors/config.d.ts +71 -3
  47. package/dist/lib/monitors/config.js +100 -12
  48. package/dist/lib/monitors/pid-watch.d.ts +35 -0
  49. package/dist/lib/monitors/pid-watch.js +45 -0
  50. package/dist/lib/monitors/remote.d.ts +18 -0
  51. package/dist/lib/monitors/remote.js +11 -0
  52. package/dist/lib/perf/db.d.ts +1 -1
  53. package/dist/lib/perf/db.js +53 -2
  54. package/dist/lib/perf/types.d.ts +14 -0
  55. package/dist/lib/permissions.js +7 -2
  56. package/dist/lib/plugins/plugins.d.ts +17 -3
  57. package/dist/lib/plugins/plugins.js +84 -9
  58. package/dist/lib/pty-server.d.ts +14 -0
  59. package/dist/lib/pty-server.js +49 -5
  60. package/dist/lib/secrets/drivers/rush.js +5 -0
  61. package/dist/lib/self-update.d.ts +42 -0
  62. package/dist/lib/self-update.js +88 -0
  63. package/dist/lib/session/cloud.js +5 -0
  64. package/dist/lib/session/db.d.ts +32 -6
  65. package/dist/lib/session/db.js +128 -12
  66. package/dist/lib/session/live-metadata.js +3 -3
  67. package/dist/lib/smart-launch.d.ts +6 -0
  68. package/dist/lib/smart-launch.js +5 -2
  69. package/dist/lib/staleness/writers/plugins.js +5 -2
  70. package/dist/lib/staleness/writers/subagents.js +13 -3
  71. package/dist/lib/state.d.ts +7 -4
  72. package/dist/lib/state.js +7 -4
  73. package/dist/lib/subagents.js +8 -2
  74. package/dist/lib/teams/api.d.ts +8 -0
  75. package/dist/lib/teams/api.js +50 -6
  76. package/dist/lib/teams/delivery.d.ts +14 -4
  77. package/dist/lib/teams/delivery.js +15 -5
  78. package/dist/lib/teams/scheduler.d.ts +10 -0
  79. package/dist/lib/teams/scheduler.js +8 -0
  80. package/dist/lib/traces/sync.d.ts +113 -6
  81. package/dist/lib/traces/sync.js +193 -19
  82. package/dist/lib/view-types.d.ts +12 -0
  83. package/package.json +2 -2
@@ -17,6 +17,10 @@ import { AgentStatus } from './agents.js';
17
17
  * When the process completed with a `prUrl` and merge is unknown or false,
18
18
  * delivery is `pr_open` — pessimistic: assume open until proven merged so an
19
19
  * orchestrator never mistakes "agent stopped" for "work on main".
20
+ *
21
+ * When the process completed with no PR and uncommitted changes remain in the
22
+ * worktree, delivery is `stranded` — the work exists only locally and will be
23
+ * lost if the worktree is cleaned up (PHNX-2951).
20
24
  */
21
25
  export function resolveTeammateDelivery(opts) {
22
26
  const status = String(opts.status);
@@ -30,8 +34,9 @@ export function resolveTeammateDelivery(opts) {
30
34
  return 'stopped';
31
35
  if (status === AgentStatus.COMPLETED || status === 'completed') {
32
36
  const prUrl = opts.prUrl?.trim();
33
- if (!prUrl)
34
- return 'no_pr';
37
+ if (!prUrl) {
38
+ return opts.hasUncommittedChanges ? 'stranded' : 'no_pr';
39
+ }
35
40
  if (opts.prMerged === true)
36
41
  return 'pr_merged';
37
42
  return 'pr_open';
@@ -40,21 +45,26 @@ export function resolveTeammateDelivery(opts) {
40
45
  }
41
46
  /**
42
47
  * Human label for `teams status` rows. Replaces bare COMPLETED with PR OPEN
43
- * when delivery is still pending merge.
48
+ * when delivery is still pending merge, and with STRANDED when uncommitted
49
+ * work is stranded in the worktree.
44
50
  */
45
51
  export function deliveryDisplayLabel(delivery, processStatus) {
46
52
  if (delivery === 'pr_open')
47
53
  return 'PR OPEN';
48
54
  if (delivery === 'pr_merged')
49
55
  return 'COMPLETED';
56
+ if (delivery === 'stranded')
57
+ return 'STRANDED';
50
58
  return String(processStatus).toUpperCase();
51
59
  }
52
60
  /**
53
- * Color key for statusColor-style switches. `pr_open` is its own key so the
54
- * row is visually distinct from green COMPLETED.
61
+ * Color key for statusColor-style switches. `pr_open` and `stranded` get their
62
+ * own keys so the rows are visually distinct from green COMPLETED.
55
63
  */
56
64
  export function deliveryColorKey(delivery, processStatus) {
57
65
  if (delivery === 'pr_open')
58
66
  return 'pr_open';
67
+ if (delivery === 'stranded')
68
+ return 'stranded';
59
69
  return String(processStatus);
60
70
  }
@@ -48,6 +48,16 @@ export interface PlacementOptions {
48
48
  /** Human label of the requested agent (e.g. `claude@2.1.112`) for the
49
49
  * fail-loud message. */
50
50
  agentLabel?: string;
51
+ /**
52
+ * Normalized hosts boosted with `auto-launch.preferred` (set by
53
+ * `agents devices prefer <name>`). A preferred device ranks ahead of a
54
+ * non-preferred one among the eligible survivors — after the signed-in tier
55
+ * (a preferred box that can't run the agent is still no use) and before load,
56
+ * so an operator boost overrides load-based ordering without overriding hard
57
+ * health. Empty/undefined leaves the ranking unchanged. See
58
+ * {@link autoLaunchPreferredSet}.
59
+ */
60
+ preferred?: ReadonlySet<string>;
51
61
  }
52
62
  /** Why a device was excluded from the viable set, for the fail-loud message. */
53
63
  export type ExclusionReason = 'unreachable' | 'overloaded' | 'capped' | 'not-installed';
@@ -266,6 +266,14 @@ export function pickBestDevice(devices, roster, opts) {
266
266
  const signedIn = (sa?.signedIn === true ? 0 : 1) - (sb?.signedIn === true ? 0 : 1);
267
267
  if (signedIn !== 0)
268
268
  return signedIn;
269
+ // (a2) operator-preferred device next — `agents devices prefer <name>`
270
+ // boosts a box above its load-equal peers, overriding load-based order.
271
+ const preferred = opts?.preferred;
272
+ if (preferred && preferred.size > 0) {
273
+ const pref = (preferred.has(a) ? 0 : 1) - (preferred.has(b) ? 0 : 1);
274
+ if (pref !== 0)
275
+ return pref;
276
+ }
269
277
  // (b) lower load — coarse headroom tier, then raw load cost.
270
278
  const tier = headroomTier(sa?.headroom) - headroomTier(sb?.headroom);
271
279
  if (tier !== 0)
@@ -67,18 +67,46 @@ export interface TracesIndexShard {
67
67
  owner: string;
68
68
  stats: {
69
69
  sessionsImported: number;
70
+ /**
71
+ * Median ACTIVE duration (span − idle gaps > 120s), ms — the meaningful figure
72
+ * (PHNX-3457). Same key/shape as before this change, so the fleet-aggregate
73
+ * worker (`worker-template.ts`) keeps weighted-averaging it unchanged; only its
74
+ * VALUE moved from raw span to active time. The raw span stays available per
75
+ * session on `SessionDetail.meta.spanMs`.
76
+ */
70
77
  medianMs: number;
78
+ /** p90 ACTIVE duration, ms. */
71
79
  p90Ms: number;
80
+ /**
81
+ * SEGMENTED active-time stats (PHNX-3472). The blended `medianMs`/`p90Ms`
82
+ * above conflate one-shot interactive queries (63% of the corpus, ~15s
83
+ * median) with substantial agent runs (~15min median), so they headline
84
+ * neither. A session is an AGENT run when it made any tool call OR has more
85
+ * than 8 messages; otherwise INTERACTIVE. These segment the same active-time
86
+ * figure so the console can headline agent runs on their own axis. Each is
87
+ * computed only over sessions with a non-null duration.
88
+ */
89
+ agentMedianMs: number;
90
+ /** p90 ACTIVE duration over AGENT sessions, ms. */
91
+ agentP90Ms: number;
92
+ /** Median ACTIVE duration over INTERACTIVE sessions, ms. */
93
+ interactiveMedianMs: number;
94
+ /** (sessions with a non-null duration) / (total sessions), 0..1 — coverage of the duration stats. */
95
+ measuredFraction: number;
72
96
  needAttention: number;
73
97
  toolErrorRate: number;
74
98
  };
99
+ /**
100
+ * Sessions excluded from the eval corpus as internal utility plumbing (PHNX-3474):
101
+ * single-shot machine calls (no tool call AND ≤2 messages) or a known
102
+ * internal-prompt signature (title generation, watchdog, commit-message, factory
103
+ * worker). Every `stats` figure above, `topics` counts, and `needsAttention` are
104
+ * computed over the AGENT set ONLY — `sessionsImported` is the real agent count,
105
+ * not the raw row count. This is the number that was dropped.
106
+ */
107
+ utilityCount: number;
75
108
  needsAttention: IndexedSession[];
76
- topics: Array<{
77
- key: string;
78
- label: string;
79
- count: number;
80
- group: TraceTopicGroup;
81
- }>;
109
+ topics: TopicItem[];
82
110
  failures: {
83
111
  byToolError: Array<{
84
112
  tool: string;
@@ -104,11 +132,56 @@ export interface IndexedSession {
104
132
  title: string;
105
133
  repo: string;
106
134
  device: string;
135
+ /** The harness that produced the session (claude/codex/rush/grok/…). Same as `harness`. */
107
136
  agent: string;
108
137
  model: string;
138
+ /** Corpus classification (PHNX-3474). Always `'agent'` here — utility rows are excluded. */
139
+ kind: SessionKind;
109
140
  severity: number;
110
141
  flags: string[];
111
142
  }
143
+ /**
144
+ * One example session under a topic tile — the shape the console drill-down consumes.
145
+ * Carries `kind` + `harness` (PHNX-3474) so the console can filter a tile's session
146
+ * list by corpus class and by harness. Refs on a topic tile are always `'agent'`
147
+ * (utility rows never reach a bucket), but the field is explicit for the consumer.
148
+ */
149
+ export interface TopicSessionRef {
150
+ id: string;
151
+ title: string;
152
+ kind: SessionKind;
153
+ harness: string;
154
+ }
155
+ /**
156
+ * Corpus class of a session (PHNX-3474). `utility` is internal machine plumbing —
157
+ * a single-shot call with no tool use and ≤2 messages, or one whose topic/label
158
+ * matches a known internal-prompt signature (title generation, watchdog,
159
+ * commit-message writer, factory worker). Everything else is `agent`: real agent
160
+ * work the Evals console counts and scores. Utility rows are tagged, never deleted,
161
+ * and excluded from every index statistic.
162
+ */
163
+ export type SessionKind = 'utility' | 'agent';
164
+ /**
165
+ * Classify a session as internal `utility` plumbing vs real `agent` work (PHNX-3474).
166
+ * `toolCallCount` is the AUTHORITATIVE per-session tool-call count from the loaded
167
+ * `tool_calls` rows (the row's own `tool_call_count` column is a fallback for a row
168
+ * whose calls weren't loaded). A session is `utility` when a known internal-prompt
169
+ * signature matches its topic/label, OR it made no tool call AND has ≤2 messages —
170
+ * the single-shot machine-call shape. Otherwise it is `agent`.
171
+ */
172
+ export declare function classifySessionKind(row: Pick<SyncRow, 'topic' | 'label' | 'message_count' | 'tool_call_count'>, toolCallCount: number): SessionKind;
173
+ /**
174
+ * One topic bucket in the treemap. `sessions` carries up to {@link TOPIC_SESSION_CAP}
175
+ * example refs so the console can drill from the tile into its session list — a tile
176
+ * with no refs renders display-only (PHNX-3408). `count` stays the true total.
177
+ */
178
+ export interface TopicItem {
179
+ key: string;
180
+ label: string;
181
+ count: number;
182
+ group: TraceTopicGroup;
183
+ sessions: TopicSessionRef[];
184
+ }
112
185
  /** A row from `tool_calls`. `ordinal`/`timestamp` order calls within a session for computeInsights(). */
113
186
  export interface ToolCallRow {
114
187
  session_id: string;
@@ -129,6 +202,31 @@ export interface ToolCallRow {
129
202
  error: string | null;
130
203
  parse_error: string | null;
131
204
  }
205
+ /**
206
+ * Active time for a session in the index shard: its recorded span minus every idle
207
+ * gap > 120s (PHNX-3457). The index build already holds the ordered `tool_calls`
208
+ * rows for the whole corpus, so idle is derived from them here — no transcript
209
+ * re-parse. A cursor sweeps the span from `sessionStartMs`: each stretch where the
210
+ * cursor sits idle for more than the threshold before the next call starts is
211
+ * subtracted, and idle is measured from a call's END (its own `end_timestamp` when
212
+ * known, else its start) so a call's own blocking duration is never mistaken for
213
+ * idle. Crucially the sweep also books the gaps at the two BOUNDARIES — before the
214
+ * first call and after the last call to the session end — so a session with a lone
215
+ * tool call that was then abandoned and resumed hours later (the case a
216
+ * between-calls-only measure missed entirely, leaving the whole 345h span counted
217
+ * as active) has that trailing idle stripped. A session end is `sessionStartMs +
218
+ * spanMs`, so the two agree by construction.
219
+ *
220
+ * Bounded to `[0, spanMs]`. A session with NO tool calls returns the full span
221
+ * unchanged rather than a fabricated zero: there is no tool-call evidence of idle
222
+ * either way, and treating a chat-only turn as 100% idle would be a worse error
223
+ * than leaving its span uncorrected. Where the full event stream IS available (a
224
+ * per-session `SessionDetail`), {@link activeMsFromTrajectory} is used instead —
225
+ * it sees message events this call-only approximation cannot, so the two are close
226
+ * but not identical by design (the corpus-scale index build cannot afford the
227
+ * per-session parse the detail view does).
228
+ */
229
+ export declare function sessionActiveMs(spanMs: number, sessionCalls: ToolCallRow[], sessionStartMs: number): number;
132
230
  /** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
133
231
  export declare function failureDescription(call: ToolCallRow, cause: TraceFailureCause): string;
134
232
  /** Build the redacted rich console shard from indexed metadata and derived caches. */
@@ -138,7 +236,16 @@ export interface SessionDetail {
138
236
  schema: 1;
139
237
  id: string;
140
238
  meta: {
239
+ /** Raw wall-clock span (last event − first event), idle time included. */
141
240
  spanMs: number;
241
+ /**
242
+ * Active time: `spanMs` minus every idle gap > 120s (PHNX-3457). A session
243
+ * resumed hours later, or left idle mid-turn, inflates `spanMs` with wall-clock
244
+ * the agent did no work in — active time strips those gaps so a duration reads
245
+ * as effort, not calendar span. This is what the console's duration median/p90
246
+ * should trust; `spanMs` stays available as the raw figure.
247
+ */
248
+ activeMs: number;
142
249
  turns: number;
143
250
  tools: number;
144
251
  errorCount: number;
@@ -213,12 +213,109 @@ export async function syncTraces(opts = {}) {
213
213
  indexError,
214
214
  };
215
215
  }
216
+ /**
217
+ * Topic/label substrings that identify an internal-prompt session regardless of its
218
+ * message/tool shape. These are the harness-spawned utility prompts the Rush app
219
+ * fires (they run under the `claude` harness): title generation writes the 3–4 word
220
+ * session title, the watchdog polls for stalled agents, the commit-message writer
221
+ * drafts a conventional commit, and factory workers are dispatched sub-agents. The
222
+ * title-generation prompt lives in the `topic` column, the rest can land in either
223
+ * `topic` or `label`, so both are matched.
224
+ */
225
+ const UTILITY_PROMPT_SIGNATURES = [
226
+ /generate a 3-4 word title/i, // title generation
227
+ /you are a watchdog|watchdog monitoring/i, // watchdog tick
228
+ /conventional[- ]commit/i, // commit-message writer
229
+ /factory worker/i, // dispatched factory worker
230
+ ];
231
+ /**
232
+ * Classify a session as internal `utility` plumbing vs real `agent` work (PHNX-3474).
233
+ * `toolCallCount` is the AUTHORITATIVE per-session tool-call count from the loaded
234
+ * `tool_calls` rows (the row's own `tool_call_count` column is a fallback for a row
235
+ * whose calls weren't loaded). A session is `utility` when a known internal-prompt
236
+ * signature matches its topic/label, OR it made no tool call AND has ≤2 messages —
237
+ * the single-shot machine-call shape. Otherwise it is `agent`.
238
+ */
239
+ export function classifySessionKind(row, toolCallCount) {
240
+ const haystack = `${row.topic ?? ''}\n${row.label ?? ''}`;
241
+ if (UTILITY_PROMPT_SIGNATURES.some((re) => re.test(haystack)))
242
+ return 'utility';
243
+ const hasToolCalls = toolCallCount > 0 || (row.tool_call_count ?? 0) > 0;
244
+ const messages = row.message_count ?? 0;
245
+ if (!hasToolCalls && messages <= 2)
246
+ return 'utility';
247
+ return 'agent';
248
+ }
216
249
  function percentile(values, ratio) {
217
250
  if (values.length === 0)
218
251
  return 0;
219
252
  const sorted = [...values].sort((a, b) => a - b);
220
253
  return sorted[Math.min(sorted.length - 1, Math.ceil(sorted.length * ratio) - 1)];
221
254
  }
255
+ /**
256
+ * A delta between consecutive events longer than this reads as an idle stall, not
257
+ * work — the same threshold the trajectory uses for its gap detection
258
+ * (`DEFAULT_IDLE_THRESHOLD_MS`, trajectory.ts). Kept in lockstep so active time
259
+ * here and the gaps drawn in a session's detail view agree on what "idle" means.
260
+ */
261
+ const IDLE_GAP_THRESHOLD_MS = 120_000;
262
+ /** Max example session refs carried per topic tile so a tile is drillable (PHNX-3408). */
263
+ const TOPIC_SESSION_CAP = 30;
264
+ /**
265
+ * Active time for a session in the index shard: its recorded span minus every idle
266
+ * gap > 120s (PHNX-3457). The index build already holds the ordered `tool_calls`
267
+ * rows for the whole corpus, so idle is derived from them here — no transcript
268
+ * re-parse. A cursor sweeps the span from `sessionStartMs`: each stretch where the
269
+ * cursor sits idle for more than the threshold before the next call starts is
270
+ * subtracted, and idle is measured from a call's END (its own `end_timestamp` when
271
+ * known, else its start) so a call's own blocking duration is never mistaken for
272
+ * idle. Crucially the sweep also books the gaps at the two BOUNDARIES — before the
273
+ * first call and after the last call to the session end — so a session with a lone
274
+ * tool call that was then abandoned and resumed hours later (the case a
275
+ * between-calls-only measure missed entirely, leaving the whole 345h span counted
276
+ * as active) has that trailing idle stripped. A session end is `sessionStartMs +
277
+ * spanMs`, so the two agree by construction.
278
+ *
279
+ * Bounded to `[0, spanMs]`. A session with NO tool calls returns the full span
280
+ * unchanged rather than a fabricated zero: there is no tool-call evidence of idle
281
+ * either way, and treating a chat-only turn as 100% idle would be a worse error
282
+ * than leaving its span uncorrected. Where the full event stream IS available (a
283
+ * per-session `SessionDetail`), {@link activeMsFromTrajectory} is used instead —
284
+ * it sees message events this call-only approximation cannot, so the two are close
285
+ * but not identical by design (the corpus-scale index build cannot afford the
286
+ * per-session parse the detail view does).
287
+ */
288
+ export function sessionActiveMs(spanMs, sessionCalls, sessionStartMs) {
289
+ if (spanMs <= 0)
290
+ return Math.max(0, spanMs);
291
+ if (!Number.isFinite(sessionStartMs))
292
+ return spanMs; // can't place calls on the span
293
+ const spanEndMs = sessionStartMs + spanMs;
294
+ const ordered = sessionCalls
295
+ .map((c) => ({ startMs: Date.parse(c.timestamp), endMs: Date.parse(c.end_timestamp ?? c.timestamp) }))
296
+ .filter((c) => Number.isFinite(c.startMs))
297
+ .sort((a, b) => a.startMs - b.startMs);
298
+ if (ordered.length === 0)
299
+ return spanMs; // no tool-call evidence of idle
300
+ let idleMs = 0;
301
+ let cursor = sessionStartMs;
302
+ for (const call of ordered) {
303
+ if (call.startMs > cursor + IDLE_GAP_THRESHOLD_MS)
304
+ idleMs += call.startMs - cursor;
305
+ const endMs = Number.isFinite(call.endMs) ? Math.max(call.endMs, call.startMs) : call.startMs;
306
+ if (endMs > cursor)
307
+ cursor = endMs;
308
+ }
309
+ // Trailing idle: the stretch from the last call's end to the session's end.
310
+ if (spanEndMs > cursor + IDLE_GAP_THRESHOLD_MS)
311
+ idleMs += spanEndMs - cursor;
312
+ return Math.max(0, spanMs - Math.min(idleMs, spanMs));
313
+ }
314
+ /** Active time from an already-built trajectory: span minus its idle gaps (all > threshold). */
315
+ function activeMsFromTrajectory(traj) {
316
+ const idleMs = traj.gaps.reduce((sum, gap) => sum + gap.durationMs, 0);
317
+ return Math.max(0, traj.spanMs - Math.min(idleMs, traj.spanMs));
318
+ }
222
319
  /** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
223
320
  export function failureDescription(call, cause) {
224
321
  if (cause === 'guard')
@@ -297,6 +394,7 @@ export function buildIndexShard(rows, device, owner, prevShard) {
297
394
  }
298
395
  const toolMix = new Map();
299
396
  const errorCounts = new Map();
397
+ const callsBySession = new Map();
300
398
  for (const call of calls) {
301
399
  const mix = toolMix.get(call.session_id) ?? {};
302
400
  mix[call.tool] = (mix[call.tool] ?? 0) + 1;
@@ -304,9 +402,27 @@ export function buildIndexShard(rows, device, owner, prevShard) {
304
402
  if (call.outcome === 'error') {
305
403
  errorCounts.set(call.session_id, (errorCounts.get(call.session_id) ?? 0) + 1);
306
404
  }
405
+ const list = callsBySession.get(call.session_id);
406
+ if (list)
407
+ list.push(call);
408
+ else
409
+ callsBySession.set(call.session_id, [call]);
307
410
  }
308
- const topics = readSessionTopics(ids);
309
- const missingTopics = rows.filter((row) => !topics.has(row.id)).map((row) => {
411
+ // Classify every row as real `agent` work or internal `utility` plumbing, then run
412
+ // the ENTIRE rest of the shard build over the agent set ONLY (PHNX-3474). Utility
413
+ // rows — title-gen / watchdog / commit-message / factory-worker calls, ~68% of the
414
+ // corpus — are tagged and excluded here, never deleted from sessions.db, so the
415
+ // console counts and scores real agent work: `sessionsImported`, the medians,
416
+ // needs-attention, tool-error-rate, and the topic buckets all measure `agentRows`.
417
+ const kindOf = new Map(rows.map((row) => [row.id, classifySessionKind(row, callsBySession.get(row.id)?.length ?? 0)]));
418
+ const agentRows = rows.filter((row) => kindOf.get(row.id) === 'agent');
419
+ const agentIds = agentRows.map((row) => row.id);
420
+ const utilityCount = rows.length - agentRows.length;
421
+ // Tool calls belonging to utility rows never contribute to the failure/latency
422
+ // stats or the tool-error rate — filter them out at the source alongside the rows.
423
+ const agentCalls = calls.filter((call) => kindOf.get(call.session_id) === 'agent');
424
+ const topics = readSessionTopics(agentIds);
425
+ const missingTopics = agentRows.filter((row) => !topics.has(row.id)).map((row) => {
310
426
  const topic = classifyTopic({
311
427
  cwd: row.cwd,
312
428
  gitBranch: row.git_branch,
@@ -327,11 +443,11 @@ export function buildIndexShard(rows, device, owner, prevShard) {
327
443
  // from an identically-signatured session synced this run purely by *when* each
328
444
  // was first seen (PHNX-3327). A cache-miss row is parsed at most once here even
329
445
  // when both derivations are missing.
330
- const insights = readSessionInsights(ids);
331
- const phenotypes = readSessionPhenotypes(ids);
446
+ const insights = readSessionInsights(agentIds);
447
+ const phenotypes = readSessionPhenotypes(agentIds);
332
448
  const missingInsights = [];
333
449
  const missingPhenotypes = [];
334
- for (const row of rows) {
450
+ for (const row of agentRows) {
335
451
  const needInsights = !insights.has(row.id);
336
452
  const needPhenotype = !phenotypes.has(row.id);
337
453
  if (!needInsights && !needPhenotype)
@@ -363,7 +479,7 @@ export function buildIndexShard(rows, device, owner, prevShard) {
363
479
  }
364
480
  persistDerivedCache('session-insights', () => writeSessionInsights(missingInsights));
365
481
  persistDerivedCache('session-phenotypes', () => writeSessionPhenotypes(missingPhenotypes));
366
- const needsAttention = rows.flatMap((row) => {
482
+ const needsAttention = agentRows.flatMap((row) => {
367
483
  const facets = insights.get(row.id);
368
484
  const errorCount = errorCounts.get(row.id) ?? 0;
369
485
  const flags = attentionFlags(errorCount, facets);
@@ -378,17 +494,28 @@ export function buildIndexShard(rows, device, owner, prevShard) {
378
494
  device,
379
495
  agent: row.agent,
380
496
  model: row.model ?? 'unknown',
497
+ kind: 'agent',
381
498
  severity: errorCount * 2 + friction * 3 + corrections * 2,
382
499
  flags,
383
500
  }];
384
501
  }).sort((a, b) => b.severity - a.severity || a.id.localeCompare(b.id));
385
502
  const topicCounts = new Map();
386
- for (const topic of topics.values()) {
387
- const current = topicCounts.get(topic.key) ?? { ...topic, count: 0 };
388
- current.count++;
389
- topicCounts.set(topic.key, current);
503
+ for (const row of agentRows) {
504
+ const topic = topics.get(row.id);
505
+ if (!topic)
506
+ continue;
507
+ const bucket = topicCounts.get(topic.key)
508
+ ?? { key: topic.key, label: topic.label, group: topic.group, count: 0, refs: [] };
509
+ bucket.count++;
510
+ bucket.refs.push({
511
+ id: row.id,
512
+ title: redactSecrets(row.label ?? row.topic ?? topic.label ?? 'Untitled session', knownSecrets),
513
+ harness: row.agent,
514
+ recencyMs: Date.parse(row.last_activity ?? row.timestamp) || 0,
515
+ });
516
+ topicCounts.set(topic.key, bucket);
390
517
  }
391
- const failedCalls = calls.filter((call) => call.outcome === 'error');
518
+ const failedCalls = agentCalls.filter((call) => call.outcome === 'error');
392
519
  const byCause = { real: 0, guard: 0, hook: 0 };
393
520
  const failureCounts = new Map();
394
521
  for (const call of failedCalls) {
@@ -400,14 +527,38 @@ export function buildIndexShard(rows, device, owner, prevShard) {
400
527
  current.count++;
401
528
  failureCounts.set(key, current);
402
529
  }
403
- const durations = rows.flatMap((row) => row.duration_ms == null ? [] : [row.duration_ms]);
530
+ // Duration stats run over ACTIVE time, not raw span (PHNX-3457): span minus idle
531
+ // gaps > 120s, derived per session from the tool_calls already loaded above. A
532
+ // session resumed after hours, or left idle mid-turn, otherwise inflates the
533
+ // median/p90 with wall-clock the agent did no work in (real corpus max span:
534
+ // 345h). The raw span stays available per session on `SessionDetail.meta.spanMs`.
535
+ const activeDurations = agentRows.flatMap((row) => row.duration_ms == null
536
+ ? []
537
+ : [sessionActiveMs(row.duration_ms, callsBySession.get(row.id) ?? [], Date.parse(row.timestamp))]);
538
+ // Segment the same active-time figure into AGENT vs INTERACTIVE runs (PHNX-3472).
539
+ // A session is an AGENT run when it made any tool call OR has more than 8
540
+ // messages; otherwise INTERACTIVE (a one-shot query). Only sessions with a
541
+ // non-null duration contribute to the medians; `measuredFraction` reports how
542
+ // much of the corpus that covers.
543
+ const agentActive = [];
544
+ const interactiveActive = [];
545
+ let measured = 0;
546
+ for (const row of agentRows) {
547
+ if (row.duration_ms == null)
548
+ continue;
549
+ measured++;
550
+ const active = sessionActiveMs(row.duration_ms, callsBySession.get(row.id) ?? [], Date.parse(row.timestamp));
551
+ const isAgent = (callsBySession.get(row.id)?.length ?? 0) > 0 || (row.message_count ?? 0) > 8;
552
+ (isAgent ? agentActive : interactiveActive).push(active);
553
+ }
554
+ const measuredFraction = agentRows.length === 0 ? 0 : measured / agentRows.length;
404
555
  // Build today's per-bucket stats for the rolling drift window.
405
556
  const todayDate = new Date().toISOString().slice(0, 10);
406
557
  const todayStats = [...topicCounts.values()].map(({ key }) => {
407
558
  const sessionsInBucket = [...topics.entries()]
408
559
  .filter(([, t]) => t.key === key)
409
560
  .map(([id]) => id);
410
- const bucketCalls = calls.filter((c) => sessionsInBucket.includes(c.session_id));
561
+ const bucketCalls = agentCalls.filter((c) => sessionsInBucket.includes(c.session_id));
411
562
  const bucketErrors = bucketCalls.filter((c) => c.outcome === 'error').length;
412
563
  const errorRate = bucketCalls.length === 0 ? 0 : bucketErrors / bucketCalls.length;
413
564
  const stallCount = sessionsInBucket.filter((id) => {
@@ -421,21 +572,43 @@ export function buildIndexShard(rows, device, owner, prevShard) {
421
572
  const prevHistory = prevShard?.bucketHistory ?? [];
422
573
  const bucketHistory = [...prevHistory, todayStats].slice(-14);
423
574
  const driftSignals = computeDriftSignal(prevHistory, todayStats);
424
- const patternInsights = computeInsights(rows, calls, prevShard, phenotypes);
575
+ const patternInsights = computeInsights(agentRows, agentCalls, prevShard, phenotypes);
425
576
  return {
426
577
  schema: 1,
427
578
  device,
428
579
  syncedAt: Date.now(),
429
580
  owner,
430
581
  stats: {
431
- sessionsImported: rows.length,
432
- medianMs: percentile(durations, 0.5),
433
- p90Ms: percentile(durations, 0.9),
582
+ sessionsImported: agentRows.length,
583
+ medianMs: percentile(activeDurations, 0.5),
584
+ p90Ms: percentile(activeDurations, 0.9),
585
+ agentMedianMs: percentile(agentActive, 0.5),
586
+ agentP90Ms: percentile(agentActive, 0.9),
587
+ interactiveMedianMs: percentile(interactiveActive, 0.5),
588
+ measuredFraction,
434
589
  needAttention: needsAttention.length,
435
- toolErrorRate: calls.length === 0 ? 0 : failedCalls.length / calls.length,
590
+ toolErrorRate: agentCalls.length === 0 ? 0 : failedCalls.length / agentCalls.length,
436
591
  },
592
+ utilityCount,
437
593
  needsAttention,
438
- topics: [...topicCounts.values()].sort((a, b) => b.count - a.count || a.key.localeCompare(b.key)),
594
+ topics: [...topicCounts.values()]
595
+ .sort((a, b) => b.count - a.count || a.key.localeCompare(b.key))
596
+ .map((bucket) => ({
597
+ key: bucket.key,
598
+ label: bucket.label,
599
+ count: bucket.count,
600
+ group: bucket.group,
601
+ // Up to TOPIC_SESSION_CAP most-recent example sessions, so the console can
602
+ // drill from a tile into its session list. Capped to keep the shard small;
603
+ // the tile's `count` remains the true total. These are correct on this
604
+ // per-device shard; the fleet-aggregate `/all` view (worker-template.ts,
605
+ // a must-not-touch R2 worker here) carries only the first device's refs
606
+ // per topic until PHNX-3464 merges them across devices.
607
+ sessions: bucket.refs
608
+ .sort((a, b) => b.recencyMs - a.recencyMs)
609
+ .slice(0, TOPIC_SESSION_CAP)
610
+ .map(({ id, title, harness }) => ({ id, title, kind: 'agent', harness })),
611
+ })),
439
612
  failures: {
440
613
  byToolError: [...failureCounts.values()].sort((a, b) => b.count - a.count || a.tool.localeCompare(b.tool)),
441
614
  byCause,
@@ -503,6 +676,7 @@ export function buildSessionDetail(traj) {
503
676
  id: s.id,
504
677
  meta: {
505
678
  spanMs: traj.spanMs,
679
+ activeMs: activeMsFromTrajectory(traj),
506
680
  turns: (stats.userTurns ?? 0) + (stats.assistantTurns ?? 0),
507
681
  tools: stats.toolCount ?? 0,
508
682
  errorCount: traj.errorCount,
@@ -9,6 +9,18 @@ export interface ViewJsonVersion {
9
9
  isolated: boolean;
10
10
  isIsolatedDefault: boolean;
11
11
  signedIn: boolean;
12
+ /**
13
+ * Whether THIS version home can actually spawn a signed-in agent — the strict
14
+ * per-version launch truth (`isLaunchableSignedIn`), not the display `signedIn`
15
+ * above. `signedIn` is true when the version *inherits* the active/global HOME
16
+ * login even with no per-version credential of its own; such a home shows "who
17
+ * is logged in" but dies at spawn once launch isolates HOME to it. Automatic
18
+ * `--device auto` placement gates on THIS field so a remote box is judged by
19
+ * the same launchability the local candidate uses (`collectRunCandidates` →
20
+ * `isLaunchableSignedIn`), closing the local/remote asymmetry (PHNX-3466).
21
+ * Absent on an older remote CLI, whose consumers fall back to `signedIn`.
22
+ */
23
+ launchable: boolean;
12
24
  /** Live cached authentication verdict for this installed version. */
13
25
  authVerdict: AuthVerdict | null;
14
26
  email: string | null;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@phnx-labs/agents-cli",
3
- "version": "1.22.58",
3
+ "version": "1.22.60",
4
4
  "description": "One CLI for all your AI coding agents - versions, config, cloud dispatch, sessions, and teams (now with first-class Grok Build CLI support)",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
@@ -88,7 +88,7 @@
88
88
  "dependencies": {
89
89
  "@fontsource/inter": "^5.3.0",
90
90
  "@fontsource/jetbrains-mono": "^5.3.0",
91
- "@homebridge/node-pty-prebuilt-multiarch": "0.13.1",
91
+ "@homebridge/node-pty-prebuilt-multiarch": "0.14.1",
92
92
  "@inquirer/prompts": "8.5.2",
93
93
  "@resvg/resvg-wasm": "^2.6.2",
94
94
  "@types/proper-lockfile": "4.1.4",