@phnx-labs/agents-cli 1.22.58 → 1.22.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/CHANGELOG.md +260 -0
  2. package/README.md +29 -0
  3. package/dist/bootstrap.js +32 -1
  4. package/dist/commands/monitors.js +187 -23
  5. package/dist/commands/perf.js +10 -0
  6. package/dist/commands/routines.test-fixture.js +5 -0
  7. package/dist/commands/send.d.ts +2 -1
  8. package/dist/commands/send.js +7 -5
  9. package/dist/commands/sessions-picker.d.ts +13 -0
  10. package/dist/commands/sessions-picker.js +17 -8
  11. package/dist/commands/sessions-stats.js +37 -5
  12. package/dist/commands/sessions.js +52 -16
  13. package/dist/commands/ssh.js +12 -1
  14. package/dist/commands/teams-picker.js +20 -6
  15. package/dist/commands/teams.d.ts +2 -2
  16. package/dist/commands/teams.js +77 -21
  17. package/dist/commands/versions.js +12 -4
  18. package/dist/commands/view.js +7 -2
  19. package/dist/lib/accounting/rotate.js +12 -4
  20. package/dist/lib/accounting/usage-sync.d.ts +12 -2
  21. package/dist/lib/accounting/usage-sync.js +34 -6
  22. package/dist/lib/auto-pull-worker.js +7 -2
  23. package/dist/lib/cloud/rush.d.ts +7 -0
  24. package/dist/lib/cloud/rush.js +29 -1
  25. package/dist/lib/daemon/daemon.d.ts +22 -0
  26. package/dist/lib/daemon/daemon.js +39 -0
  27. package/dist/lib/daemon/session-index-service.js +9 -1
  28. package/dist/lib/daemon-ticks.d.ts +15 -0
  29. package/dist/lib/daemon-ticks.js +26 -0
  30. package/dist/lib/device-config.d.ts +5 -1
  31. package/dist/lib/device-config.js +2 -2
  32. package/dist/lib/devices/health.js +5 -1
  33. package/dist/lib/devices/pool.d.ts +25 -2
  34. package/dist/lib/devices/pool.js +32 -2
  35. package/dist/lib/devices/stats-cache.d.ts +0 -6
  36. package/dist/lib/devices/stats-cache.js +2 -9
  37. package/dist/lib/doctor-diff.d.ts +14 -0
  38. package/dist/lib/doctor-diff.js +43 -2
  39. package/dist/lib/feed/events.js +4 -0
  40. package/dist/lib/git.d.ts +38 -0
  41. package/dist/lib/git.js +58 -0
  42. package/dist/lib/hosts/ready.d.ts +8 -0
  43. package/dist/lib/hosts/ready.js +13 -2
  44. package/dist/lib/installations/versions.d.ts +17 -0
  45. package/dist/lib/installations/versions.js +53 -2
  46. package/dist/lib/monitors/config.d.ts +71 -3
  47. package/dist/lib/monitors/config.js +100 -12
  48. package/dist/lib/monitors/pid-watch.d.ts +35 -0
  49. package/dist/lib/monitors/pid-watch.js +45 -0
  50. package/dist/lib/monitors/remote.d.ts +18 -0
  51. package/dist/lib/monitors/remote.js +11 -0
  52. package/dist/lib/perf/db.d.ts +1 -1
  53. package/dist/lib/perf/db.js +53 -2
  54. package/dist/lib/perf/types.d.ts +14 -0
  55. package/dist/lib/permissions.js +7 -2
  56. package/dist/lib/plugins/plugins.d.ts +17 -3
  57. package/dist/lib/plugins/plugins.js +84 -9
  58. package/dist/lib/pty-server.d.ts +14 -0
  59. package/dist/lib/pty-server.js +49 -5
  60. package/dist/lib/secrets/drivers/rush.js +5 -0
  61. package/dist/lib/self-update.d.ts +42 -0
  62. package/dist/lib/self-update.js +88 -0
  63. package/dist/lib/session/cloud.js +5 -0
  64. package/dist/lib/session/db.d.ts +32 -6
  65. package/dist/lib/session/db.js +128 -12
  66. package/dist/lib/session/live-metadata.js +3 -3
  67. package/dist/lib/smart-launch.d.ts +6 -0
  68. package/dist/lib/smart-launch.js +5 -2
  69. package/dist/lib/staleness/writers/plugins.js +5 -2
  70. package/dist/lib/staleness/writers/subagents.js +13 -3
  71. package/dist/lib/state.d.ts +7 -4
  72. package/dist/lib/state.js +7 -4
  73. package/dist/lib/subagents.js +8 -2
  74. package/dist/lib/teams/api.d.ts +8 -0
  75. package/dist/lib/teams/api.js +50 -6
  76. package/dist/lib/teams/delivery.d.ts +14 -4
  77. package/dist/lib/teams/delivery.js +15 -5
  78. package/dist/lib/teams/scheduler.d.ts +10 -0
  79. package/dist/lib/teams/scheduler.js +8 -0
  80. package/dist/lib/traces/sync.d.ts +113 -6
  81. package/dist/lib/traces/sync.js +193 -19
  82. package/dist/lib/view-types.d.ts +12 -0
  83. package/package.json +2 -2
@@ -13,7 +13,7 @@ import { type ToolScanResumePoint } from './tool-store.js';
13
13
  /** Current schema version; bumped when migrations are added. Exported so tests
14
14
  * assert against the constant instead of hardcoding a number that every bump
15
15
  * then has to chase (docs/sessions.md calls the constant the source of truth). */
16
- export declare const SCHEMA_VERSION = 43;
16
+ export declare const SCHEMA_VERSION = 44;
17
17
  /**
18
18
  * Bump to force the content extractor (assistant-answer text, alongside the
19
19
  * user-prompt text every harness already accumulates) to re-derive on every
@@ -317,6 +317,16 @@ export declare function getSessionExistenceCacheStats(): {
317
317
  };
318
318
  /** Query sessions from the database, applying filters and ordering by last-activity descending (default). */
319
319
  export declare function querySessions(options?: QueryOptions): SessionMeta[];
320
+ /**
321
+ * Cheap query for the daemon's deferred tool-index pass (PHNX-3411).
322
+ *
323
+ * Returns the most-recently-active sessions whose parseSession reads a large
324
+ * flat transcript (kimi: wire.jsonl, grok: chat_history.jsonl). Their scanners
325
+ * produce no events, so upsertSessionsBatch skips them in the warm tick to
326
+ * avoid wedging the event loop. ensureToolIndex uses tool_scan_ledger stamps
327
+ * to skip already-current rows and applies byte/file budget caps.
328
+ */
329
+ export declare function querySessionsForDeferredToolIndex(limit: number): SessionMeta[];
320
330
  /** Count sessions matching the given filter options. */
321
331
  export declare function countSessions(options?: QueryOptions): number;
322
332
  /** One grouped row in a cost/duration rollup. */
@@ -519,14 +529,30 @@ export declare function queryResourceUsageStats(options: QueryOptions & {
519
529
  limit?: number;
520
530
  }): ResourceStatRow[];
521
531
  /**
522
- * Coverage of the resource-usage signal: how many distinct sessions carry any
523
- * row in session_resource_usage vs. the total indexed. A low ratio means the
524
- * historical backfill (`agents sessions backfill resources`) hasn't run — the
525
- * stats surface uses this to tell the user their zero-counts may just be
526
- * un-scanned history, not genuine non-use.
532
+ * Coverage of the resource-usage signal, as three honest facts:
533
+ *
534
+ * - `scanned` — sessions the resource extractor has actually processed, i.e.
535
+ * those carrying a `resource_scan_ledger` row at the current
536
+ * `RESOURCE_INDEX_VERSION`. This is the true "has the historical backfill run"
537
+ * signal: the ledger is stamped for EVERY scanned session, including ones that
538
+ * invoked nothing (`resource_count = 0`), so `scanned/total` rises to ~1 after
539
+ * `agents sessions backfill resources` regardless of how sparse explicit
540
+ * invocations are.
541
+ * - `covered` — distinct sessions that carry AT LEAST ONE row in
542
+ * `session_resource_usage`, i.e. that actually recorded an explicit invocation.
543
+ * This is an ABSOLUTE signal count, not a coverage ratio: it stays small even
544
+ * at full scan coverage because most sessions invoke no skill/command, and a
545
+ * non-recording harness contributes none by construction.
546
+ * - `total` — sessions indexed.
547
+ *
548
+ * The two were previously conflated: `covered/total` was framed as coverage and
549
+ * read ~1.2% even after a full backfill (most sessions genuinely invoke nothing),
550
+ * so the "run the backfill" hint never cleared. Keying the hint on `scanned/total`
551
+ * fixes that — see `commands/sessions-stats.ts` (PHNX-2301).
527
552
  */
528
553
  export declare function resourceUsageCoverage(): {
529
554
  covered: number;
555
+ scanned: number;
530
556
  total: number;
531
557
  };
532
558
  /** Outcome of a resource-usage backfill run. */
@@ -27,7 +27,7 @@ const DB_PATH = getSessionsDbPath();
27
27
  /** Current schema version; bumped when migrations are added. Exported so tests
28
28
  * assert against the constant instead of hardcoding a number that every bump
29
29
  * then has to chase (docs/sessions.md calls the constant the source of truth). */
30
- export const SCHEMA_VERSION = 43;
30
+ export const SCHEMA_VERSION = 44;
31
31
  /**
32
32
  * Bump to force the content extractor (assistant-answer text, alongside the
33
33
  * user-prompt text every harness already accumulates) to re-derive on every
@@ -1205,6 +1205,32 @@ function migrateSchema(db, fromVersion) {
1205
1205
  if (!cols.has('end_timestamp'))
1206
1206
  db.exec(`ALTER TABLE tool_calls ADD COLUMN end_timestamp TEXT`);
1207
1207
  }
1208
+ if (fromVersion < 44) {
1209
+ // v43 -> v44: backfill duration_ms for sessions whose harness scan extractor
1210
+ // never derived it (PHNX-3457). rush/grok/kimi/cursor/muse/antigravity/hermes/
1211
+ // openclaw left duration_ms NULL — 52% of the corpus, 100% of rush — so the
1212
+ // console median was computed over only the ~48% that carried it, skewing it
1213
+ // short. Going forward resolveDurationMs() populates it at every upsert; this
1214
+ // repairs already-indexed rows in place from the timestamps they already store
1215
+ // (last_activity, itself resolved from the last-message time else file mtime,
1216
+ // minus the creation timestamp), so no transcript is re-parsed. julianday is
1217
+ // avoided because it does not accept a trailing 'Z'; the arithmetic is done in
1218
+ // JS with the exact Date.parse resolveDurationMs uses, keeping the backfill and
1219
+ // the live path consistent. Only rows with a positive span are touched; a NULL
1220
+ // that cannot be resolved stays NULL rather than becoming a fabricated 0.
1221
+ const nullDurationRows = db.prepare(`SELECT id, timestamp, last_activity FROM sessions
1222
+ WHERE duration_ms IS NULL AND last_activity IS NOT NULL`).all();
1223
+ // Runs inside migrateSchema's own transaction (db.ts:1468), so no nested
1224
+ // db.transaction() here — that would raise "transaction within a transaction".
1225
+ const update = db.prepare(`UPDATE sessions SET duration_ms = ? WHERE id = ?`);
1226
+ for (const row of nullDurationRows) {
1227
+ const startMs = Date.parse(row.timestamp);
1228
+ const lastMs = Date.parse(row.last_activity);
1229
+ if (Number.isFinite(startMs) && Number.isFinite(lastMs) && lastMs > startMs) {
1230
+ update.run(lastMs - startMs, row.id);
1231
+ }
1232
+ }
1233
+ }
1208
1234
  }
1209
1235
  /**
1210
1236
  * Stamp `account_key` / `account_org` / `account` on every Claude row from its
@@ -2075,7 +2101,7 @@ export function upsertSession(meta, content, scan, assistantContent = '') {
2075
2101
  cache_write_tokens: meta.cacheWriteTokens ?? null,
2076
2102
  cost_usd: meta.costUsd ?? null,
2077
2103
  cost_usd_nocache: meta.costUsdNoCache ?? null,
2078
- duration_ms: meta.durationMs ?? null,
2104
+ duration_ms: resolveDurationMs(meta, scan),
2079
2105
  model: meta.model ?? null,
2080
2106
  tool_call_count: meta.toolCallCount ?? null,
2081
2107
  file_path: meta.filePath,
@@ -2163,6 +2189,23 @@ export function upsertSessionsBatch(entries) {
2163
2189
  ? { ...entry, meta: { ...entry.meta, ...fanOutCounts(entry.events, entry.meta.agent) } }
2164
2190
  : entry;
2165
2191
  }
2192
+ // Harnesses whose scanners produce no events AND whose parseSession reads a
2193
+ // potentially large flat transcript file (not a compact SQLite DB). Calling
2194
+ // parseSession on the warm tick for an active large session wedges the Node
2195
+ // event loop for seconds, making browser IPC miss its connection window
2196
+ // (PHNX-3411). Defer their tool-call indexing to runDeferredToolIndex, which
2197
+ // uses ensureToolIndex with tool_scan_ledger stamps and byte/file budget caps.
2198
+ // NOT opencode — parseOpenCode issues a targeted SQLite query, so it is fast
2199
+ // even for large sessions and its results populate recentDirectoriesTouched.
2200
+ const LARGE_TRANSCRIPT_AGENTS = new Set(['kimi', 'grok']);
2201
+ if (!entry.events && LARGE_TRANSCRIPT_AGENTS.has(entry.meta.agent)) {
2202
+ return entry;
2203
+ }
2204
+ // Enrich the entry. When the scanner already provided events, use them
2205
+ // directly (no transcript re-read). When it didn't — e.g. OpenCode whose
2206
+ // scanner produces only metadata but whose parseOpenCode is a fast SQLite
2207
+ // query — fall back to parseSession. Agents that would call an expensive
2208
+ // flat-file parse already returned above.
2166
2209
  try {
2167
2210
  const toolSourcePath = toolEvidenceSourcePath(entry.meta.filePath, entry.meta.agent);
2168
2211
  const toolScan = toolSourcePath === entry.meta.filePath
@@ -2171,9 +2214,6 @@ export function upsertSessionsBatch(entries) {
2171
2214
  const stat = fs.statSync(toolSourcePath);
2172
2215
  return { fileMtimeMs: stat.mtimeMs, fileSize: stat.size };
2173
2216
  })();
2174
- // Some non-resumable scanners already normalized the transcript while
2175
- // deriving metadata. Reuse those events; scanners that only read summary
2176
- // metadata fall back to exactly one normalized parse here.
2177
2217
  const events = entry.events ?? parseSession(entry.meta.filePath, entry.meta.agent);
2178
2218
  writeResourceUsage(entry.meta.id, events, entry.meta.cwd);
2179
2219
  // Resume the tool index from the last scan of this append-only stream when
@@ -2302,7 +2342,7 @@ export function upsertSessionsBatch(entries) {
2302
2342
  cache_write_tokens: meta.cacheWriteTokens ?? null,
2303
2343
  cost_usd: meta.costUsd ?? null,
2304
2344
  cost_usd_nocache: meta.costUsdNoCache ?? null,
2305
- duration_ms: meta.durationMs ?? null,
2345
+ duration_ms: resolveDurationMs(meta, scan),
2306
2346
  model: meta.model ?? null,
2307
2347
  tool_call_count: meta.toolCallCount ?? null,
2308
2348
  file_path: meta.filePath,
@@ -2599,6 +2639,36 @@ function resolveLastActivity(meta, scan) {
2599
2639
  return new Date(scan.fileMtimeMs).toISOString();
2600
2640
  return meta.timestamp;
2601
2641
  }
2642
+ /**
2643
+ * The persisted wall-clock span for a session (PHNX-3457).
2644
+ *
2645
+ * `durationMs` is canonically `lastTs − firstTs`. Some harness scan extractors
2646
+ * derive it themselves from per-event timestamps (claude/codex/droid/gemini/
2647
+ * opencode set `meta.durationMs`); the rest (rush/grok/kimi/cursor/muse/
2648
+ * antigravity/hermes/openclaw) never did, so `duration_ms` landed NULL for them —
2649
+ * 52% of the corpus, including 100% of the dominant `rush` usage — and the
2650
+ * console median was computed over only the ~48% that happened to carry it,
2651
+ * skewing it short and misleading.
2652
+ *
2653
+ * This closes the gap at the single write boundary every harness funnels
2654
+ * through, so parity is automatic rather than per-extractor: when the extractor
2655
+ * already computed a precise span, keep it; otherwise derive it from the same
2656
+ * `timestamp` (creation) and `resolveLastActivity` (last event, itself resolved
2657
+ * from the harness's last-message time, else file mtime) that the row already
2658
+ * stores. Returns null only when no positive span can be established (a single
2659
+ * timestamped event, or a clock that runs backwards), which reads as NULL rather
2660
+ * than a fabricated 0.
2661
+ */
2662
+ function resolveDurationMs(meta, scan) {
2663
+ if (meta.durationMs != null)
2664
+ return meta.durationMs;
2665
+ const startMs = Date.parse(meta.timestamp);
2666
+ const lastMs = Date.parse(resolveLastActivity(meta, scan));
2667
+ if (Number.isFinite(startMs) && Number.isFinite(lastMs) && lastMs > startMs) {
2668
+ return lastMs - startMs;
2669
+ }
2670
+ return null;
2671
+ }
2602
2672
  export function isSessionActivityFresh(row, maxAgeMs, nowMs) {
2603
2673
  const parsedActivityMs = Date.parse(row.last_activity ?? row.timestamp);
2604
2674
  const activityMs = Number.isFinite(parsedActivityMs) ? parsedActivityMs : row.file_mtime_ms ?? undefined;
@@ -2892,6 +2962,26 @@ export function querySessions(options = {}) {
2892
2962
  const trimmed = options.limit ? live.slice(0, options.limit) : live;
2893
2963
  return trimmed.map(rowToMeta);
2894
2964
  }
2965
+ /**
2966
+ * Cheap query for the daemon's deferred tool-index pass (PHNX-3411).
2967
+ *
2968
+ * Returns the most-recently-active sessions whose parseSession reads a large
2969
+ * flat transcript (kimi: wire.jsonl, grok: chat_history.jsonl). Their scanners
2970
+ * produce no events, so upsertSessionsBatch skips them in the warm tick to
2971
+ * avoid wedging the event loop. ensureToolIndex uses tool_scan_ledger stamps
2972
+ * to skip already-current rows and applies byte/file budget caps.
2973
+ */
2974
+ export function querySessionsForDeferredToolIndex(limit) {
2975
+ const db = getDB();
2976
+ const rows = db.prepare(`
2977
+ SELECT * FROM sessions
2978
+ WHERE file_path IS NOT NULL
2979
+ AND agent IN ('kimi', 'grok')
2980
+ ORDER BY last_activity DESC, timestamp DESC
2981
+ LIMIT ?
2982
+ `).all(limit);
2983
+ return rows.map(rowToMeta);
2984
+ }
2895
2985
  /** Count sessions matching the given filter options. */
2896
2986
  export function countSessions(options = {}) {
2897
2987
  const db = getDB();
@@ -3333,17 +3423,43 @@ export function queryResourceUsageStats(options) {
3333
3423
  return db.prepare(sql).all(...allParams);
3334
3424
  }
3335
3425
  /**
3336
- * Coverage of the resource-usage signal: how many distinct sessions carry any
3337
- * row in session_resource_usage vs. the total indexed. A low ratio means the
3338
- * historical backfill (`agents sessions backfill resources`) hasn't run — the
3339
- * stats surface uses this to tell the user their zero-counts may just be
3340
- * un-scanned history, not genuine non-use.
3426
+ * Coverage of the resource-usage signal, as three honest facts:
3427
+ *
3428
+ * - `scanned` — sessions the resource extractor has actually processed, i.e.
3429
+ * those carrying a `resource_scan_ledger` row at the current
3430
+ * `RESOURCE_INDEX_VERSION`. This is the true "has the historical backfill run"
3431
+ * signal: the ledger is stamped for EVERY scanned session, including ones that
3432
+ * invoked nothing (`resource_count = 0`), so `scanned/total` rises to ~1 after
3433
+ * `agents sessions backfill resources` regardless of how sparse explicit
3434
+ * invocations are.
3435
+ * - `covered` — distinct sessions that carry AT LEAST ONE row in
3436
+ * `session_resource_usage`, i.e. that actually recorded an explicit invocation.
3437
+ * This is an ABSOLUTE signal count, not a coverage ratio: it stays small even
3438
+ * at full scan coverage because most sessions invoke no skill/command, and a
3439
+ * non-recording harness contributes none by construction.
3440
+ * - `total` — sessions indexed.
3441
+ *
3442
+ * The two were previously conflated: `covered/total` was framed as coverage and
3443
+ * read ~1.2% even after a full backfill (most sessions genuinely invoke nothing),
3444
+ * so the "run the backfill" hint never cleared. Keying the hint on `scanned/total`
3445
+ * fixes that — see `commands/sessions-stats.ts` (PHNX-2301).
3341
3446
  */
3342
3447
  export function resourceUsageCoverage() {
3343
3448
  const db = getDB();
3344
3449
  const covered = db.prepare(`SELECT COUNT(DISTINCT session_id) AS n FROM session_resource_usage`).get().n;
3450
+ // JOIN sessions so a ledger row for a since-vanished transcript (dropped from
3451
+ // `sessions` but not yet from the ledger) can't inflate scan coverage past the
3452
+ // indexed set. Only rows at the current extractor version count as scanned —
3453
+ // a stale-version row is re-derived on the next backfill, so it is not yet
3454
+ // "covered" for this extractor.
3455
+ const scanned = db.prepare(`
3456
+ SELECT COUNT(*) AS n
3457
+ FROM resource_scan_ledger l
3458
+ JOIN sessions s ON s.id = l.session_id
3459
+ WHERE l.extractor_version = ?
3460
+ `).get(RESOURCE_INDEX_VERSION).n;
3345
3461
  const total = db.prepare(`SELECT COUNT(*) AS n FROM sessions`).get().n;
3346
- return { covered, total };
3462
+ return { covered, scanned, total };
3347
3463
  }
3348
3464
  /** Has this session's resource usage been derived at the current extractor version for this exact file? */
3349
3465
  function needsResourceIndex(db, sessionId, stamp) {
@@ -15,7 +15,8 @@
15
15
  // exist. It parses no transcript and renders nothing — it only reshapes process
16
16
  // state the live registry already computed. When the transcript IS on disk, the
17
17
  // synthesized row carries its path, so the downstream `buildPreview` renders the
18
- // real digest; otherwise it renders the header + a "not indexed here" note.
18
+ // real digest. Without a path, a peer-owned row fetches its digest from that peer;
19
+ // a genuinely local row renders the header + a "not indexed here" note.
19
20
  import { deriveShortId } from '../text/short-id.js';
20
21
  import { isSessionTrackedAgent } from './types.js';
21
22
  /**
@@ -50,9 +51,8 @@ export function activeSessionToSessionMeta(active, self, nowMs) {
50
51
  topic: active.topic,
51
52
  version: active.version,
52
53
  messageCount: undefined,
53
- // The row runs HERE — the local live registry is machine-local, so a match
54
- // resolves without an SSH hop back to a peer.
55
54
  machine: active.machine ?? self,
55
+ _remote: (active.machine ?? self) !== self,
56
56
  ticketId: active.ticket?.id,
57
57
  prUrl: active.pr?.url,
58
58
  prNumber: active.pr?.number,
@@ -95,6 +95,12 @@ export declare function resolveDeviceAuto(agent?: string, opts?: {
95
95
  /** Route only to devices with a row the interactive account picker can launch. */
96
96
  accountPicker?: boolean;
97
97
  probe?: (pool: string[], agent?: AgentType) => Promise<Map<string, DevicePlacementSignal>>;
98
+ /**
99
+ * Preferred hosts (`auto-launch.preferred`) that get the ranking boost.
100
+ * Defaults to the stored fleet block resolved over the candidate pool;
101
+ * injectable so a test pins it without touching disk.
102
+ */
103
+ preferred?: ReadonlySet<string>;
98
104
  }): Promise<DeviceAutoPlan>;
99
105
  /**
100
106
  * Resolve host for `--device auto`. Does NOT pick harness or accounts.
@@ -8,7 +8,7 @@
8
8
  import { queryAffinityRollup } from './session/db.js';
9
9
  import { localMachineId } from './session/origin-machine.js';
10
10
  import { loadDevicesSync } from './devices/registry.js';
11
- import { describeAutoPool, filterAutoPool, isAutoPoolMember } from './devices/pool.js';
11
+ import { autoLaunchPreferredSet, describeAutoPool, filterAutoPool, isAutoPoolMember } from './devices/pool.js';
12
12
  import { normalizeHost } from './machine-id.js';
13
13
  import { probePoolSignals } from './teams/placement-probe.js';
14
14
  import { pickBestDevice } from './teams/scheduler.js';
@@ -153,7 +153,10 @@ export async function resolveDeviceAuto(agent, opts = {}) {
153
153
  if (eligiblePool.length === 0) {
154
154
  throw new Error(formatNoHealthyDeviceError(pool, signals, agent));
155
155
  }
156
- const picked = pickBestDevice(eligiblePool, [], { signals, agentLabel: agent });
156
+ // `agents devices prefer <name>` boosts a device in the ranking — resolved
157
+ // over the same pool so a fleet default reaches a doc-less box.
158
+ const preferred = opts.preferred ?? autoLaunchPreferredSet(pool, { roster: pool });
159
+ const picked = pickBestDevice(eligiblePool, [], { signals, agentLabel: agent, preferred });
157
160
  return {
158
161
  host: picked === local ? null : picked,
159
162
  pickedDeviceKey: picked,
@@ -8,8 +8,11 @@ function buildPluginsWriter(agent) {
8
8
  write({ version, versionHome, selection }) {
9
9
  const all = discoverPlugins();
10
10
  const map = new Map(all.map(p => [p.name, p]));
11
- // Clean orphan plugin-skills from plugins that no longer exist.
12
- cleanOrphanedPluginSkills(agent, versionHome, new Set(all.map(p => p.name)));
11
+ // Clean orphan plugin-skills from plugins that no longer exist. Pass the
12
+ // discovered plugins (not just names) so a stale install under one
13
+ // marketplace is trashed even when another marketplace still ships that
14
+ // name — the PHNX-2618 shadow `code` plugin.
15
+ cleanOrphanedPluginSkills(agent, versionHome, all);
13
16
  const synced = [];
14
17
  for (const name of selection) {
15
18
  const plugin = map.get(name);
@@ -25,10 +25,16 @@ function buildSubagentsWriter(agent) {
25
25
  const dir = target.dir(versionHome);
26
26
  const synced = [];
27
27
  const paths = [];
28
+ const errors = [];
28
29
  for (const name of selection) {
29
30
  const sub = map.get(name);
30
- if (!sub)
31
+ if (!sub) {
32
+ // Requested but not discoverable as an installed central subagent —
33
+ // e.g. its AGENT.md failed to parse. Say so instead of silently
34
+ // dropping it, which read as an unactionable doctor "hold" (PHNX-3187).
35
+ errors.push(`subagent '${name}': no parseable AGENT.md in ~/.agents/subagents`);
31
36
  continue;
37
+ }
32
38
  try {
33
39
  target.write(dir, sub);
34
40
  synced.push(sub.name);
@@ -39,9 +45,13 @@ function buildSubagentsWriter(agent) {
39
45
  paths.push(entry.path);
40
46
  }
41
47
  }
42
- catch { /* per-item sync failure: skip */ }
48
+ catch (e) {
49
+ // A genuine fs/transform failure must surface with its reason, not
50
+ // vanish behind a bare `catch` (RUSH-2677 / PHNX-3187).
51
+ errors.push(`subagent '${sub.name}': ${e.message}`);
52
+ }
43
53
  }
44
- return { synced, paths };
54
+ return errors.length > 0 ? { synced, paths, errors } : { synced, paths };
45
55
  },
46
56
  };
47
57
  }
@@ -198,10 +198,13 @@ export declare function getMonitorsDir(): string;
198
198
  * Path to built-in monitor definitions shipped in the system repo
199
199
  * (`~/.agents/.system/monitors/`). Unioned under the user monitors dir by
200
200
  * listMonitors()/readMonitor(): a monitor shipped here is available on every
201
- * install, and a user monitor of the same name overrides it. A built-in with no
202
- * `enabled:` field is opt-in — it stays disabled until the user enables it,
203
- * which materializes a user copy (writes never touch this pull-only mirror). The
204
- * directory need not exist.
201
+ * install, and a user monitor of the same name overrides it. A built-in is
202
+ * enabled by default like every other system-layer resource — it runs on every
203
+ * install unless the user shadows it with `enabled: false` (via `agents monitors
204
+ * pause`, which materializes a user copy; writes never touch this pull-only
205
+ * mirror). A shared-input built-in still carries its own `device:` owner pin in
206
+ * the shipped YAML so exactly one box fires it (SING-9). The directory need not
207
+ * exist.
205
208
  */
206
209
  export declare function getSystemMonitorsDir(): string;
207
210
  /** Path to the durable per-monitor state-diff store + fire history
package/dist/lib/state.js CHANGED
@@ -491,10 +491,13 @@ export function getMonitorsDir() { return process.env.AGENTS_MONITORS_DIR ?? MON
491
491
  * Path to built-in monitor definitions shipped in the system repo
492
492
  * (`~/.agents/.system/monitors/`). Unioned under the user monitors dir by
493
493
  * listMonitors()/readMonitor(): a monitor shipped here is available on every
494
- * install, and a user monitor of the same name overrides it. A built-in with no
495
- * `enabled:` field is opt-in — it stays disabled until the user enables it,
496
- * which materializes a user copy (writes never touch this pull-only mirror). The
497
- * directory need not exist.
494
+ * install, and a user monitor of the same name overrides it. A built-in is
495
+ * enabled by default like every other system-layer resource — it runs on every
496
+ * install unless the user shadows it with `enabled: false` (via `agents monitors
497
+ * pause`, which materializes a user copy; writes never touch this pull-only
498
+ * mirror). A shared-input built-in still carries its own `device:` owner pin in
499
+ * the shipped YAML so exactly one box fires it (SING-9). The directory need not
500
+ * exist.
498
501
  */
499
502
  export function getSystemMonitorsDir() { return process.env.AGENTS_SYSTEM_MONITORS_DIR ?? SYSTEM_MONITORS_DIR; }
500
503
  /** Path to the durable per-monitor state-diff store + fire history
@@ -23,7 +23,12 @@ export function parseSubagentFrontmatter(filePath) {
23
23
  }
24
24
  try {
25
25
  const content = fs.readFileSync(filePath, 'utf-8');
26
- const lines = content.split('\n');
26
+ // Split on CRLF or LF: git checks text files out with CRLF on Windows
27
+ // (core.autocrlf), so a plain split('\n') leaves a trailing '\r' on every
28
+ // line and the frontmatter fence `'---\r' !== '---'` never matches —
29
+ // silently dropping the subagent from discovery so it can never be
30
+ // installed or reconciled (PHNX-3187).
31
+ const lines = content.split(/\r?\n/);
27
32
  // Check for YAML frontmatter
28
33
  if (lines[0] === '---') {
29
34
  const endIndex = lines.slice(1).findIndex((l) => l === '---');
@@ -52,7 +57,8 @@ export function getSubagentBody(filePath) {
52
57
  return '';
53
58
  }
54
59
  const content = fs.readFileSync(filePath, 'utf-8');
55
- const lines = content.split('\n');
60
+ // CRLF-robust for the same reason as parseSubagentFrontmatter (PHNX-3187).
61
+ const lines = content.split(/\r?\n/);
56
62
  // Skip YAML frontmatter
57
63
  if (lines[0] === '---') {
58
64
  const endIndex = lines.slice(1).findIndex((l) => l === '---');
@@ -62,6 +62,8 @@ export interface AgentStatusDetail {
62
62
  task_type?: TaskType | null;
63
63
  /** Device name the teammate runs on for a distributed (--on) teammate; null for local. */
64
64
  host?: string | null;
65
+ /** Absolute path to the teammate's worktree, when known. */
66
+ workspace_dir?: string | null;
65
67
  /** Sanitized evidence observed at the lifecycle boundary that failed. */
66
68
  failure?: TeammateFailure | null;
67
69
  }
@@ -73,6 +75,7 @@ export interface TaskStatusResult {
73
75
  pending: number;
74
76
  running: number;
75
77
  completed: number;
78
+ stranded: number;
76
79
  failed: number;
77
80
  stopped: number;
78
81
  };
@@ -126,6 +129,8 @@ export interface AgentStatusSummary {
126
129
  /** Device name for a distributed (--on) teammate; null for local. */
127
130
  host: string | null;
128
131
  failure: TeammateFailure | null;
132
+ /** Absolute path to the teammate's worktree, when known. */
133
+ workspace_dir?: string | null;
129
134
  /** ISO timestamp — feed back via --since for delta polling. */
130
135
  cursor: string;
131
136
  }
@@ -137,6 +142,7 @@ export interface TaskStatusSummaryResult {
137
142
  pending: number;
138
143
  running: number;
139
144
  completed: number;
145
+ stranded: number;
140
146
  failed: number;
141
147
  stopped: number;
142
148
  };
@@ -164,6 +170,8 @@ export interface TaskInfo {
164
170
  pending: number;
165
171
  running: number;
166
172
  completed: number;
173
+ /** Completed teammates with uncommitted work and no PR (PHNX-2951). */
174
+ stranded: number;
167
175
  failed: number;
168
176
  stopped: number;
169
177
  workspace_dir: string | null;
@@ -3,6 +3,7 @@ import { getDelta } from './summarizer.js';
3
3
  import { debug } from './debug.js';
4
4
  import { buildClaudeLabelMap } from '../session/discover.js';
5
5
  import { resolveTeammateDelivery } from './delivery.js';
6
+ import { hasUncommittedChanges } from './worktree.js';
6
7
  /**
7
8
  * Truncate a bash command for status output.
8
9
  * Handles heredocs specially - shows the redirect target instead of contents.
@@ -126,6 +127,7 @@ export function toAgentStatusSummary(detail) {
126
127
  .map((m) => trimMessage(m)),
127
128
  host: detail.host ?? null,
128
129
  failure: detail.failure ?? null,
130
+ workspace_dir: detail.workspace_dir ?? null,
129
131
  cursor: detail.cursor,
130
132
  };
131
133
  }
@@ -207,7 +209,7 @@ parentSessionId) {
207
209
  ? allAgents
208
210
  : allAgents.filter((a) => a.status === effectiveFilter);
209
211
  const agentStatuses = [];
210
- const counts = { pending: 0, running: 0, completed: 0, failed: 0, stopped: 0 };
212
+ const counts = { pending: 0, running: 0, completed: 0, stranded: 0, failed: 0, stopped: 0 };
211
213
  // Count ALL agents for summary (not just filtered)
212
214
  for (const agent of allAgents) {
213
215
  if (agent.status === AgentStatus.PENDING)
@@ -221,6 +223,30 @@ parentSessionId) {
221
223
  else if (agent.status === AgentStatus.STOPPED)
222
224
  counts.stopped++;
223
225
  }
226
+ // Stranded count is computed over ALL agents, not the filter-narrowed view,
227
+ // so a `--filter running` status still reports teammates that completed dirty.
228
+ // Cache the probe result so we can reuse it when building details for the
229
+ // filtered subset without probing the worktree twice.
230
+ const uncommittedCache = new Map();
231
+ for (const agent of allAgents) {
232
+ if (agent.status !== AgentStatus.COMPLETED)
233
+ continue;
234
+ const shouldProbeWorktree = !agent.prUrl?.trim() &&
235
+ !agent.hostName &&
236
+ Boolean(agent.workspaceDir);
237
+ const hasUncommitted = shouldProbeWorktree
238
+ ? await hasUncommittedChanges(agent.workspaceDir)
239
+ : false;
240
+ uncommittedCache.set(agent.agentId, hasUncommitted);
241
+ const delivery = resolveTeammateDelivery({
242
+ status: agent.status,
243
+ prUrl: agent.prUrl,
244
+ hasUncommittedChanges: hasUncommitted,
245
+ });
246
+ if (delivery === 'stranded') {
247
+ counts.stranded++;
248
+ }
249
+ }
224
250
  // Build details only for filtered agents
225
251
  let maxTimestamp = since || new Date(0).toISOString(); // Track max timestamp for cursor
226
252
  const claudeLabels = allAgents.some((agent) => agent.agentType === 'claude')
@@ -237,6 +263,12 @@ parentSessionId) {
237
263
  if (agentTimestamp > maxTimestamp) {
238
264
  maxTimestamp = agentTimestamp;
239
265
  }
266
+ const hasUncommitted = uncommittedCache.get(agent.agentId) ?? false;
267
+ const delivery = resolveTeammateDelivery({
268
+ status: agent.status,
269
+ prUrl: agent.prUrl,
270
+ hasUncommittedChanges: hasUncommitted,
271
+ });
240
272
  const detail = {
241
273
  agent_id: agent.agentId,
242
274
  agent_type: agent.agentType,
@@ -254,15 +286,13 @@ parentSessionId) {
254
286
  after: agent.after,
255
287
  task_type: agent.taskType,
256
288
  host: agent.hostName,
289
+ workspace_dir: agent.workspaceDir,
257
290
  failure: agent.failure,
258
291
  mode: agent.mode,
259
292
  cloud_session_id: agent.cloudSessionId,
260
293
  cloud_provider: agent.cloudProvider,
261
294
  pr_url: agent.prUrl,
262
- delivery: resolveTeammateDelivery({
263
- status: agent.status,
264
- prUrl: agent.prUrl,
265
- }),
295
+ delivery,
266
296
  files_created: delta.new_files_created,
267
297
  files_modified: delta.new_files_modified,
268
298
  files_read: delta.new_files_read,
@@ -297,7 +327,7 @@ export async function handleTasks(manager, limit = 10) {
297
327
  }
298
328
  const tasks = [];
299
329
  for (const [taskName, agents] of taskMap) {
300
- let pending = 0, running = 0, completed = 0, failed = 0, stopped = 0;
330
+ let pending = 0, running = 0, completed = 0, stranded = 0, failed = 0, stopped = 0;
301
331
  let earliestStart = null;
302
332
  let latestActivity = null;
303
333
  let workspaceDir = null;
@@ -313,6 +343,19 @@ export async function handleTasks(manager, limit = 10) {
313
343
  failed++;
314
344
  else if (agent.status === AgentStatus.STOPPED)
315
345
  stopped++;
346
+ // Stranded = completed, no PR, local worktree still dirty. Probes the real
347
+ // worktree so `teams tasks` and `teams list --status` don't classify lost
348
+ // work as done (PHNX-2951).
349
+ if (agent.status === AgentStatus.COMPLETED && !agent.prUrl?.trim() && !agent.hostName && agent.workspaceDir) {
350
+ const delivery = resolveTeammateDelivery({
351
+ status: agent.status,
352
+ prUrl: agent.prUrl,
353
+ hasUncommittedChanges: await hasUncommittedChanges(agent.workspaceDir),
354
+ });
355
+ if (delivery === 'stranded') {
356
+ stranded++;
357
+ }
358
+ }
316
359
  // Track earliest start (created_at)
317
360
  if (!earliestStart || agent.startedAt < earliestStart) {
318
361
  earliestStart = agent.startedAt;
@@ -336,6 +379,7 @@ export async function handleTasks(manager, limit = 10) {
336
379
  pending,
337
380
  running,
338
381
  completed,
382
+ stranded,
339
383
  failed,
340
384
  stopped,
341
385
  workspace_dir: workspaceDir,
@@ -12,13 +12,17 @@
12
12
  */
13
13
  import { AgentStatus } from './agents.js';
14
14
  /** Postcondition-facing delivery of a teammate's work. */
15
- export type TeammateDelivery = 'pending' | 'in_progress' | 'pr_open' | 'pr_merged' | 'no_pr' | 'failed' | 'stopped';
15
+ export type TeammateDelivery = 'pending' | 'in_progress' | 'pr_open' | 'pr_merged' | 'no_pr' | 'stranded' | 'failed' | 'stopped';
16
16
  /**
17
17
  * Derive delivery from process status and PR state.
18
18
  *
19
19
  * When the process completed with a `prUrl` and merge is unknown or false,
20
20
  * delivery is `pr_open` — pessimistic: assume open until proven merged so an
21
21
  * orchestrator never mistakes "agent stopped" for "work on main".
22
+ *
23
+ * When the process completed with no PR and uncommitted changes remain in the
24
+ * worktree, delivery is `stranded` — the work exists only locally and will be
25
+ * lost if the worktree is cleaned up (PHNX-2951).
22
26
  */
23
27
  export declare function resolveTeammateDelivery(opts: {
24
28
  status: AgentStatus | string;
@@ -28,14 +32,20 @@ export declare function resolveTeammateDelivery(opts: {
28
32
  * `null`/`undefined` = unknown; with a `prUrl` that means `pr_open`.
29
33
  */
30
34
  prMerged?: boolean | null;
35
+ /**
36
+ * Whether the teammate's worktree has uncommitted changes. Only consulted
37
+ * when the process completed without a PR URL.
38
+ */
39
+ hasUncommittedChanges?: boolean | null;
31
40
  }): TeammateDelivery;
32
41
  /**
33
42
  * Human label for `teams status` rows. Replaces bare COMPLETED with PR OPEN
34
- * when delivery is still pending merge.
43
+ * when delivery is still pending merge, and with STRANDED when uncommitted
44
+ * work is stranded in the worktree.
35
45
  */
36
46
  export declare function deliveryDisplayLabel(delivery: TeammateDelivery, processStatus: AgentStatus | string): string;
37
47
  /**
38
- * Color key for statusColor-style switches. `pr_open` is its own key so the
39
- * row is visually distinct from green COMPLETED.
48
+ * Color key for statusColor-style switches. `pr_open` and `stranded` get their
49
+ * own keys so the rows are visually distinct from green COMPLETED.
40
50
  */
41
51
  export declare function deliveryColorKey(delivery: TeammateDelivery, processStatus: string): string;