@phnx-labs/agents-cli 1.22.58 → 1.22.59

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/CHANGELOG.md +238 -0
  2. package/README.md +29 -0
  3. package/dist/bootstrap.js +32 -1
  4. package/dist/commands/monitors.js +187 -23
  5. package/dist/commands/routines.test-fixture.js +5 -0
  6. package/dist/commands/send.d.ts +2 -1
  7. package/dist/commands/send.js +7 -5
  8. package/dist/commands/sessions-stats.js +37 -5
  9. package/dist/commands/sessions.js +39 -5
  10. package/dist/commands/ssh.js +12 -1
  11. package/dist/commands/versions.js +12 -4
  12. package/dist/commands/view.js +7 -2
  13. package/dist/lib/auto-pull-worker.js +7 -2
  14. package/dist/lib/cloud/rush.d.ts +7 -0
  15. package/dist/lib/cloud/rush.js +29 -1
  16. package/dist/lib/daemon/daemon.d.ts +22 -0
  17. package/dist/lib/daemon/daemon.js +39 -0
  18. package/dist/lib/daemon/session-index-service.js +9 -1
  19. package/dist/lib/daemon-ticks.d.ts +15 -0
  20. package/dist/lib/daemon-ticks.js +26 -0
  21. package/dist/lib/device-config.d.ts +5 -1
  22. package/dist/lib/device-config.js +2 -2
  23. package/dist/lib/devices/health.js +5 -1
  24. package/dist/lib/devices/pool.d.ts +25 -2
  25. package/dist/lib/devices/pool.js +32 -2
  26. package/dist/lib/devices/stats-cache.d.ts +0 -6
  27. package/dist/lib/devices/stats-cache.js +2 -9
  28. package/dist/lib/doctor-diff.d.ts +14 -0
  29. package/dist/lib/doctor-diff.js +43 -2
  30. package/dist/lib/git.d.ts +38 -0
  31. package/dist/lib/git.js +58 -0
  32. package/dist/lib/hosts/ready.d.ts +8 -0
  33. package/dist/lib/hosts/ready.js +13 -2
  34. package/dist/lib/installations/versions.d.ts +17 -0
  35. package/dist/lib/installations/versions.js +53 -2
  36. package/dist/lib/monitors/config.d.ts +71 -3
  37. package/dist/lib/monitors/config.js +100 -12
  38. package/dist/lib/monitors/pid-watch.d.ts +35 -0
  39. package/dist/lib/monitors/pid-watch.js +45 -0
  40. package/dist/lib/monitors/remote.d.ts +18 -0
  41. package/dist/lib/monitors/remote.js +11 -0
  42. package/dist/lib/permissions.js +7 -2
  43. package/dist/lib/plugins/plugins.d.ts +17 -3
  44. package/dist/lib/plugins/plugins.js +84 -9
  45. package/dist/lib/pty-server.d.ts +14 -0
  46. package/dist/lib/pty-server.js +49 -5
  47. package/dist/lib/secrets/drivers/rush.js +5 -0
  48. package/dist/lib/self-update.d.ts +42 -0
  49. package/dist/lib/self-update.js +88 -0
  50. package/dist/lib/session/cloud.js +5 -0
  51. package/dist/lib/session/db.d.ts +32 -6
  52. package/dist/lib/session/db.js +128 -12
  53. package/dist/lib/smart-launch.d.ts +6 -0
  54. package/dist/lib/smart-launch.js +5 -2
  55. package/dist/lib/staleness/writers/plugins.js +5 -2
  56. package/dist/lib/staleness/writers/subagents.js +13 -3
  57. package/dist/lib/state.d.ts +7 -4
  58. package/dist/lib/state.js +7 -4
  59. package/dist/lib/subagents.js +8 -2
  60. package/dist/lib/teams/scheduler.d.ts +10 -0
  61. package/dist/lib/teams/scheduler.js +8 -0
  62. package/dist/lib/traces/sync.d.ts +113 -6
  63. package/dist/lib/traces/sync.js +193 -19
  64. package/dist/lib/view-types.d.ts +12 -0
  65. package/package.json +2 -2
@@ -27,7 +27,7 @@ const DB_PATH = getSessionsDbPath();
27
27
  /** Current schema version; bumped when migrations are added. Exported so tests
28
28
  * assert against the constant instead of hardcoding a number that every bump
29
29
  * then has to chase (docs/sessions.md calls the constant the source of truth). */
30
- export const SCHEMA_VERSION = 43;
30
+ export const SCHEMA_VERSION = 44;
31
31
  /**
32
32
  * Bump to force the content extractor (assistant-answer text, alongside the
33
33
  * user-prompt text every harness already accumulates) to re-derive on every
@@ -1205,6 +1205,32 @@ function migrateSchema(db, fromVersion) {
1205
1205
  if (!cols.has('end_timestamp'))
1206
1206
  db.exec(`ALTER TABLE tool_calls ADD COLUMN end_timestamp TEXT`);
1207
1207
  }
1208
+ if (fromVersion < 44) {
1209
+ // v43 -> v44: backfill duration_ms for sessions whose harness scan extractor
1210
+ // never derived it (PHNX-3457). rush/grok/kimi/cursor/muse/antigravity/hermes/
1211
+ // openclaw left duration_ms NULL — 52% of the corpus, 100% of rush — so the
1212
+ // console median was computed over only the ~48% that carried it, skewing it
1213
+ // short. Going forward resolveDurationMs() populates it at every upsert; this
1214
+ // repairs already-indexed rows in place from the timestamps they already store
1215
+ // (last_activity, itself resolved from the last-message time else file mtime,
1216
+ // minus the creation timestamp), so no transcript is re-parsed. julianday is
1217
+ // avoided because it does not accept a trailing 'Z'; the arithmetic is done in
1218
+ // JS with the exact Date.parse resolveDurationMs uses, keeping the backfill and
1219
+ // the live path consistent. Only rows with a positive span are touched; a NULL
1220
+ // that cannot be resolved stays NULL rather than becoming a fabricated 0.
1221
+ const nullDurationRows = db.prepare(`SELECT id, timestamp, last_activity FROM sessions
1222
+ WHERE duration_ms IS NULL AND last_activity IS NOT NULL`).all();
1223
+ // Runs inside migrateSchema's own transaction (db.ts:1468), so no nested
1224
+ // db.transaction() here — that would raise "transaction within a transaction".
1225
+ const update = db.prepare(`UPDATE sessions SET duration_ms = ? WHERE id = ?`);
1226
+ for (const row of nullDurationRows) {
1227
+ const startMs = Date.parse(row.timestamp);
1228
+ const lastMs = Date.parse(row.last_activity);
1229
+ if (Number.isFinite(startMs) && Number.isFinite(lastMs) && lastMs > startMs) {
1230
+ update.run(lastMs - startMs, row.id);
1231
+ }
1232
+ }
1233
+ }
1208
1234
  }
1209
1235
  /**
1210
1236
  * Stamp `account_key` / `account_org` / `account` on every Claude row from its
@@ -2075,7 +2101,7 @@ export function upsertSession(meta, content, scan, assistantContent = '') {
2075
2101
  cache_write_tokens: meta.cacheWriteTokens ?? null,
2076
2102
  cost_usd: meta.costUsd ?? null,
2077
2103
  cost_usd_nocache: meta.costUsdNoCache ?? null,
2078
- duration_ms: meta.durationMs ?? null,
2104
+ duration_ms: resolveDurationMs(meta, scan),
2079
2105
  model: meta.model ?? null,
2080
2106
  tool_call_count: meta.toolCallCount ?? null,
2081
2107
  file_path: meta.filePath,
@@ -2163,6 +2189,23 @@ export function upsertSessionsBatch(entries) {
2163
2189
  ? { ...entry, meta: { ...entry.meta, ...fanOutCounts(entry.events, entry.meta.agent) } }
2164
2190
  : entry;
2165
2191
  }
2192
+ // Harnesses whose scanners produce no events AND whose parseSession reads a
2193
+ // potentially large flat transcript file (not a compact SQLite DB). Calling
2194
+ // parseSession on the warm tick for an active large session wedges the Node
2195
+ // event loop for seconds, making browser IPC miss its connection window
2196
+ // (PHNX-3411). Defer their tool-call indexing to runDeferredToolIndex, which
2197
+ // uses ensureToolIndex with tool_scan_ledger stamps and byte/file budget caps.
2198
+ // NOT opencode — parseOpenCode issues a targeted SQLite query, so it is fast
2199
+ // even for large sessions and its results populate recentDirectoriesTouched.
2200
+ const LARGE_TRANSCRIPT_AGENTS = new Set(['kimi', 'grok']);
2201
+ if (!entry.events && LARGE_TRANSCRIPT_AGENTS.has(entry.meta.agent)) {
2202
+ return entry;
2203
+ }
2204
+ // Enrich the entry. When the scanner already provided events, use them
2205
+ // directly (no transcript re-read). When it didn't — e.g. OpenCode whose
2206
+ // scanner produces only metadata but whose parseOpenCode is a fast SQLite
2207
+ // query — fall back to parseSession. Agents that would call an expensive
2208
+ // flat-file parse already returned above.
2166
2209
  try {
2167
2210
  const toolSourcePath = toolEvidenceSourcePath(entry.meta.filePath, entry.meta.agent);
2168
2211
  const toolScan = toolSourcePath === entry.meta.filePath
@@ -2171,9 +2214,6 @@ export function upsertSessionsBatch(entries) {
2171
2214
  const stat = fs.statSync(toolSourcePath);
2172
2215
  return { fileMtimeMs: stat.mtimeMs, fileSize: stat.size };
2173
2216
  })();
2174
- // Some non-resumable scanners already normalized the transcript while
2175
- // deriving metadata. Reuse those events; scanners that only read summary
2176
- // metadata fall back to exactly one normalized parse here.
2177
2217
  const events = entry.events ?? parseSession(entry.meta.filePath, entry.meta.agent);
2178
2218
  writeResourceUsage(entry.meta.id, events, entry.meta.cwd);
2179
2219
  // Resume the tool index from the last scan of this append-only stream when
@@ -2302,7 +2342,7 @@ export function upsertSessionsBatch(entries) {
2302
2342
  cache_write_tokens: meta.cacheWriteTokens ?? null,
2303
2343
  cost_usd: meta.costUsd ?? null,
2304
2344
  cost_usd_nocache: meta.costUsdNoCache ?? null,
2305
- duration_ms: meta.durationMs ?? null,
2345
+ duration_ms: resolveDurationMs(meta, scan),
2306
2346
  model: meta.model ?? null,
2307
2347
  tool_call_count: meta.toolCallCount ?? null,
2308
2348
  file_path: meta.filePath,
@@ -2599,6 +2639,36 @@ function resolveLastActivity(meta, scan) {
2599
2639
  return new Date(scan.fileMtimeMs).toISOString();
2600
2640
  return meta.timestamp;
2601
2641
  }
2642
+ /**
2643
+ * The persisted wall-clock span for a session (PHNX-3457).
2644
+ *
2645
+ * `durationMs` is canonically `lastTs − firstTs`. Some harness scan extractors
2646
+ * derive it themselves from per-event timestamps (claude/codex/droid/gemini/
2647
+ * opencode set `meta.durationMs`); the rest (rush/grok/kimi/cursor/muse/
2648
+ * antigravity/hermes/openclaw) never did, so `duration_ms` landed NULL for them —
2649
+ * 52% of the corpus, including 100% of the dominant `rush` usage — and the
2650
+ * console median was computed over only the ~48% that happened to carry it,
2651
+ * skewing it short and misleading.
2652
+ *
2653
+ * This closes the gap at the single write boundary every harness funnels
2654
+ * through, so parity is automatic rather than per-extractor: when the extractor
2655
+ * already computed a precise span, keep it; otherwise derive it from the same
2656
+ * `timestamp` (creation) and `resolveLastActivity` (last event, itself resolved
2657
+ * from the harness's last-message time, else file mtime) that the row already
2658
+ * stores. Returns null only when no positive span can be established (a single
2659
+ * timestamped event, or a clock that runs backwards), which reads as NULL rather
2660
+ * than a fabricated 0.
2661
+ */
2662
+ function resolveDurationMs(meta, scan) {
2663
+ if (meta.durationMs != null)
2664
+ return meta.durationMs;
2665
+ const startMs = Date.parse(meta.timestamp);
2666
+ const lastMs = Date.parse(resolveLastActivity(meta, scan));
2667
+ if (Number.isFinite(startMs) && Number.isFinite(lastMs) && lastMs > startMs) {
2668
+ return lastMs - startMs;
2669
+ }
2670
+ return null;
2671
+ }
2602
2672
  export function isSessionActivityFresh(row, maxAgeMs, nowMs) {
2603
2673
  const parsedActivityMs = Date.parse(row.last_activity ?? row.timestamp);
2604
2674
  const activityMs = Number.isFinite(parsedActivityMs) ? parsedActivityMs : row.file_mtime_ms ?? undefined;
@@ -2892,6 +2962,26 @@ export function querySessions(options = {}) {
2892
2962
  const trimmed = options.limit ? live.slice(0, options.limit) : live;
2893
2963
  return trimmed.map(rowToMeta);
2894
2964
  }
2965
+ /**
2966
+ * Cheap query for the daemon's deferred tool-index pass (PHNX-3411).
2967
+ *
2968
+ * Returns the most-recently-active sessions whose parseSession reads a large
2969
+ * flat transcript (kimi: wire.jsonl, grok: chat_history.jsonl). Their scanners
2970
+ * produce no events, so upsertSessionsBatch skips them in the warm tick to
2971
+ * avoid wedging the event loop. ensureToolIndex uses tool_scan_ledger stamps
2972
+ * to skip already-current rows and applies byte/file budget caps.
2973
+ */
2974
+ export function querySessionsForDeferredToolIndex(limit) {
2975
+ const db = getDB();
2976
+ const rows = db.prepare(`
2977
+ SELECT * FROM sessions
2978
+ WHERE file_path IS NOT NULL
2979
+ AND agent IN ('kimi', 'grok')
2980
+ ORDER BY last_activity DESC, timestamp DESC
2981
+ LIMIT ?
2982
+ `).all(limit);
2983
+ return rows.map(rowToMeta);
2984
+ }
2895
2985
  /** Count sessions matching the given filter options. */
2896
2986
  export function countSessions(options = {}) {
2897
2987
  const db = getDB();
@@ -3333,17 +3423,43 @@ export function queryResourceUsageStats(options) {
3333
3423
  return db.prepare(sql).all(...allParams);
3334
3424
  }
3335
3425
  /**
3336
- * Coverage of the resource-usage signal: how many distinct sessions carry any
3337
- * row in session_resource_usage vs. the total indexed. A low ratio means the
3338
- * historical backfill (`agents sessions backfill resources`) hasn't run — the
3339
- * stats surface uses this to tell the user their zero-counts may just be
3340
- * un-scanned history, not genuine non-use.
3426
+ * Coverage of the resource-usage signal, as three honest facts:
3427
+ *
3428
+ * - `scanned` — sessions the resource extractor has actually processed, i.e.
3429
+ * those carrying a `resource_scan_ledger` row at the current
3430
+ * `RESOURCE_INDEX_VERSION`. This is the true "has the historical backfill run"
3431
+ * signal: the ledger is stamped for EVERY scanned session, including ones that
3432
+ * invoked nothing (`resource_count = 0`), so `scanned/total` rises to ~1 after
3433
+ * `agents sessions backfill resources` regardless of how sparse explicit
3434
+ * invocations are.
3435
+ * - `covered` — distinct sessions that carry AT LEAST ONE row in
3436
+ * `session_resource_usage`, i.e. that actually recorded an explicit invocation.
3437
+ * This is an ABSOLUTE signal count, not a coverage ratio: it stays small even
3438
+ * at full scan coverage because most sessions invoke no skill/command, and a
3439
+ * non-recording harness contributes none by construction.
3440
+ * - `total` — sessions indexed.
3441
+ *
3442
+ * The two were previously conflated: `covered/total` was framed as coverage and
3443
+ * read ~1.2% even after a full backfill (most sessions genuinely invoke nothing),
3444
+ * so the "run the backfill" hint never cleared. Keying the hint on `scanned/total`
3445
+ * fixes that — see `commands/sessions-stats.ts` (PHNX-2301).
3341
3446
  */
3342
3447
  export function resourceUsageCoverage() {
3343
3448
  const db = getDB();
3344
3449
  const covered = db.prepare(`SELECT COUNT(DISTINCT session_id) AS n FROM session_resource_usage`).get().n;
3450
+ // JOIN sessions so a ledger row for a since-vanished transcript (dropped from
3451
+ // `sessions` but not yet from the ledger) can't inflate scan coverage past the
3452
+ // indexed set. Only rows at the current extractor version count as scanned —
3453
+ // a stale-version row is re-derived on the next backfill, so it is not yet
3454
+ // "covered" for this extractor.
3455
+ const scanned = db.prepare(`
3456
+ SELECT COUNT(*) AS n
3457
+ FROM resource_scan_ledger l
3458
+ JOIN sessions s ON s.id = l.session_id
3459
+ WHERE l.extractor_version = ?
3460
+ `).get(RESOURCE_INDEX_VERSION).n;
3345
3461
  const total = db.prepare(`SELECT COUNT(*) AS n FROM sessions`).get().n;
3346
- return { covered, total };
3462
+ return { covered, scanned, total };
3347
3463
  }
3348
3464
  /** Has this session's resource usage been derived at the current extractor version for this exact file? */
3349
3465
  function needsResourceIndex(db, sessionId, stamp) {
@@ -95,6 +95,12 @@ export declare function resolveDeviceAuto(agent?: string, opts?: {
95
95
  /** Route only to devices with a row the interactive account picker can launch. */
96
96
  accountPicker?: boolean;
97
97
  probe?: (pool: string[], agent?: AgentType) => Promise<Map<string, DevicePlacementSignal>>;
98
+ /**
99
+ * Preferred hosts (`auto-launch.preferred`) that get the ranking boost.
100
+ * Defaults to the stored fleet block resolved over the candidate pool;
101
+ * injectable so a test pins it without touching disk.
102
+ */
103
+ preferred?: ReadonlySet<string>;
98
104
  }): Promise<DeviceAutoPlan>;
99
105
  /**
100
106
  * Resolve host for `--device auto`. Does NOT pick harness or accounts.
@@ -8,7 +8,7 @@
8
8
  import { queryAffinityRollup } from './session/db.js';
9
9
  import { localMachineId } from './session/origin-machine.js';
10
10
  import { loadDevicesSync } from './devices/registry.js';
11
- import { describeAutoPool, filterAutoPool, isAutoPoolMember } from './devices/pool.js';
11
+ import { autoLaunchPreferredSet, describeAutoPool, filterAutoPool, isAutoPoolMember } from './devices/pool.js';
12
12
  import { normalizeHost } from './machine-id.js';
13
13
  import { probePoolSignals } from './teams/placement-probe.js';
14
14
  import { pickBestDevice } from './teams/scheduler.js';
@@ -153,7 +153,10 @@ export async function resolveDeviceAuto(agent, opts = {}) {
153
153
  if (eligiblePool.length === 0) {
154
154
  throw new Error(formatNoHealthyDeviceError(pool, signals, agent));
155
155
  }
156
- const picked = pickBestDevice(eligiblePool, [], { signals, agentLabel: agent });
156
+ // `agents devices prefer <name>` boosts a device in the ranking — resolved
157
+ // over the same pool so a fleet default reaches a doc-less box.
158
+ const preferred = opts.preferred ?? autoLaunchPreferredSet(pool, { roster: pool });
159
+ const picked = pickBestDevice(eligiblePool, [], { signals, agentLabel: agent, preferred });
157
160
  return {
158
161
  host: picked === local ? null : picked,
159
162
  pickedDeviceKey: picked,
@@ -8,8 +8,11 @@ function buildPluginsWriter(agent) {
8
8
  write({ version, versionHome, selection }) {
9
9
  const all = discoverPlugins();
10
10
  const map = new Map(all.map(p => [p.name, p]));
11
- // Clean orphan plugin-skills from plugins that no longer exist.
12
- cleanOrphanedPluginSkills(agent, versionHome, new Set(all.map(p => p.name)));
11
+ // Clean orphan plugin-skills from plugins that no longer exist. Pass the
12
+ // discovered plugins (not just names) so a stale install under one
13
+ // marketplace is trashed even when another marketplace still ships that
14
+ // name — the PHNX-2618 shadow `code` plugin.
15
+ cleanOrphanedPluginSkills(agent, versionHome, all);
13
16
  const synced = [];
14
17
  for (const name of selection) {
15
18
  const plugin = map.get(name);
@@ -25,10 +25,16 @@ function buildSubagentsWriter(agent) {
25
25
  const dir = target.dir(versionHome);
26
26
  const synced = [];
27
27
  const paths = [];
28
+ const errors = [];
28
29
  for (const name of selection) {
29
30
  const sub = map.get(name);
30
- if (!sub)
31
+ if (!sub) {
32
+ // Requested but not discoverable as an installed central subagent —
33
+ // e.g. its AGENT.md failed to parse. Say so instead of silently
34
+ // dropping it, which read as an unactionable doctor "hold" (PHNX-3187).
35
+ errors.push(`subagent '${name}': no parseable AGENT.md in ~/.agents/subagents`);
31
36
  continue;
37
+ }
32
38
  try {
33
39
  target.write(dir, sub);
34
40
  synced.push(sub.name);
@@ -39,9 +45,13 @@ function buildSubagentsWriter(agent) {
39
45
  paths.push(entry.path);
40
46
  }
41
47
  }
42
- catch { /* per-item sync failure: skip */ }
48
+ catch (e) {
49
+ // A genuine fs/transform failure must surface with its reason, not
50
+ // vanish behind a bare `catch` (RUSH-2677 / PHNX-3187).
51
+ errors.push(`subagent '${sub.name}': ${e.message}`);
52
+ }
43
53
  }
44
- return { synced, paths };
54
+ return errors.length > 0 ? { synced, paths, errors } : { synced, paths };
45
55
  },
46
56
  };
47
57
  }
@@ -198,10 +198,13 @@ export declare function getMonitorsDir(): string;
198
198
  * Path to built-in monitor definitions shipped in the system repo
199
199
  * (`~/.agents/.system/monitors/`). Unioned under the user monitors dir by
200
200
  * listMonitors()/readMonitor(): a monitor shipped here is available on every
201
- * install, and a user monitor of the same name overrides it. A built-in with no
202
- * `enabled:` field is opt-in — it stays disabled until the user enables it,
203
- * which materializes a user copy (writes never touch this pull-only mirror). The
204
- * directory need not exist.
201
+ * install, and a user monitor of the same name overrides it. A built-in is
202
+ * enabled by default like every other system-layer resource — it runs on every
203
+ * install unless the user shadows it with `enabled: false` (via `agents monitors
204
+ * pause`, which materializes a user copy; writes never touch this pull-only
205
+ * mirror). A shared-input built-in still carries its own `device:` owner pin in
206
+ * the shipped YAML so exactly one box fires it (SING-9). The directory need not
207
+ * exist.
205
208
  */
206
209
  export declare function getSystemMonitorsDir(): string;
207
210
  /** Path to the durable per-monitor state-diff store + fire history
package/dist/lib/state.js CHANGED
@@ -491,10 +491,13 @@ export function getMonitorsDir() { return process.env.AGENTS_MONITORS_DIR ?? MON
491
491
  * Path to built-in monitor definitions shipped in the system repo
492
492
  * (`~/.agents/.system/monitors/`). Unioned under the user monitors dir by
493
493
  * listMonitors()/readMonitor(): a monitor shipped here is available on every
494
- * install, and a user monitor of the same name overrides it. A built-in with no
495
- * `enabled:` field is opt-in — it stays disabled until the user enables it,
496
- * which materializes a user copy (writes never touch this pull-only mirror). The
497
- * directory need not exist.
494
+ * install, and a user monitor of the same name overrides it. A built-in is
495
+ * enabled by default like every other system-layer resource — it runs on every
496
+ * install unless the user shadows it with `enabled: false` (via `agents monitors
497
+ * pause`, which materializes a user copy; writes never touch this pull-only
498
+ * mirror). A shared-input built-in still carries its own `device:` owner pin in
499
+ * the shipped YAML so exactly one box fires it (SING-9). The directory need not
500
+ * exist.
498
501
  */
499
502
  export function getSystemMonitorsDir() { return process.env.AGENTS_SYSTEM_MONITORS_DIR ?? SYSTEM_MONITORS_DIR; }
500
503
  /** Path to the durable per-monitor state-diff store + fire history
@@ -23,7 +23,12 @@ export function parseSubagentFrontmatter(filePath) {
23
23
  }
24
24
  try {
25
25
  const content = fs.readFileSync(filePath, 'utf-8');
26
- const lines = content.split('\n');
26
+ // Split on CRLF or LF: git checks text files out with CRLF on Windows
27
+ // (core.autocrlf), so a plain split('\n') leaves a trailing '\r' on every
28
+ // line and the frontmatter fence `'---\r' !== '---'` never matches —
29
+ // silently dropping the subagent from discovery so it can never be
30
+ // installed or reconciled (PHNX-3187).
31
+ const lines = content.split(/\r?\n/);
27
32
  // Check for YAML frontmatter
28
33
  if (lines[0] === '---') {
29
34
  const endIndex = lines.slice(1).findIndex((l) => l === '---');
@@ -52,7 +57,8 @@ export function getSubagentBody(filePath) {
52
57
  return '';
53
58
  }
54
59
  const content = fs.readFileSync(filePath, 'utf-8');
55
- const lines = content.split('\n');
60
+ // CRLF-robust for the same reason as parseSubagentFrontmatter (PHNX-3187).
61
+ const lines = content.split(/\r?\n/);
56
62
  // Skip YAML frontmatter
57
63
  if (lines[0] === '---') {
58
64
  const endIndex = lines.slice(1).findIndex((l) => l === '---');
@@ -48,6 +48,16 @@ export interface PlacementOptions {
48
48
  /** Human label of the requested agent (e.g. `claude@2.1.112`) for the
49
49
  * fail-loud message. */
50
50
  agentLabel?: string;
51
+ /**
52
+ * Normalized hosts boosted with `auto-launch.preferred` (set by
53
+ * `agents devices prefer <name>`). A preferred device ranks ahead of a
54
+ * non-preferred one among the eligible survivors — after the signed-in tier
55
+ * (a preferred box that can't run the agent is still no use) and before load,
56
+ * so an operator boost overrides load-based ordering without overriding hard
57
+ * health. Empty/undefined leaves the ranking unchanged. See
58
+ * {@link autoLaunchPreferredSet}.
59
+ */
60
+ preferred?: ReadonlySet<string>;
51
61
  }
52
62
  /** Why a device was excluded from the viable set, for the fail-loud message. */
53
63
  export type ExclusionReason = 'unreachable' | 'overloaded' | 'capped' | 'not-installed';
@@ -266,6 +266,14 @@ export function pickBestDevice(devices, roster, opts) {
266
266
  const signedIn = (sa?.signedIn === true ? 0 : 1) - (sb?.signedIn === true ? 0 : 1);
267
267
  if (signedIn !== 0)
268
268
  return signedIn;
269
+ // (a2) operator-preferred device next — `agents devices prefer <name>`
270
+ // boosts a box above its load-equal peers, overriding load-based order.
271
+ const preferred = opts?.preferred;
272
+ if (preferred && preferred.size > 0) {
273
+ const pref = (preferred.has(a) ? 0 : 1) - (preferred.has(b) ? 0 : 1);
274
+ if (pref !== 0)
275
+ return pref;
276
+ }
269
277
  // (b) lower load — coarse headroom tier, then raw load cost.
270
278
  const tier = headroomTier(sa?.headroom) - headroomTier(sb?.headroom);
271
279
  if (tier !== 0)
@@ -67,18 +67,46 @@ export interface TracesIndexShard {
67
67
  owner: string;
68
68
  stats: {
69
69
  sessionsImported: number;
70
+ /**
71
+ * Median ACTIVE duration (span − idle gaps > 120s), ms — the meaningful figure
72
+ * (PHNX-3457). Same key/shape as before this change, so the fleet-aggregate
73
+ * worker (`worker-template.ts`) keeps weighted-averaging it unchanged; only its
74
+ * VALUE moved from raw span to active time. The raw span stays available per
75
+ * session on `SessionDetail.meta.spanMs`.
76
+ */
70
77
  medianMs: number;
78
+ /** p90 ACTIVE duration, ms. */
71
79
  p90Ms: number;
80
+ /**
81
+ * SEGMENTED active-time stats (PHNX-3472). The blended `medianMs`/`p90Ms`
82
+ * above conflate one-shot interactive queries (63% of the corpus, ~15s
83
+ * median) with substantial agent runs (~15min median), so they headline
84
+ * neither. A session is an AGENT run when it made any tool call OR has more
85
+ * than 8 messages; otherwise INTERACTIVE. These segment the same active-time
86
+ * figure so the console can headline agent runs on their own axis. Each is
87
+ * computed only over sessions with a non-null duration.
88
+ */
89
+ agentMedianMs: number;
90
+ /** p90 ACTIVE duration over AGENT sessions, ms. */
91
+ agentP90Ms: number;
92
+ /** Median ACTIVE duration over INTERACTIVE sessions, ms. */
93
+ interactiveMedianMs: number;
94
+ /** (sessions with a non-null duration) / (total sessions), 0..1 — coverage of the duration stats. */
95
+ measuredFraction: number;
72
96
  needAttention: number;
73
97
  toolErrorRate: number;
74
98
  };
99
+ /**
100
+ * Sessions excluded from the eval corpus as internal utility plumbing (PHNX-3474):
101
+ * single-shot machine calls (no tool call AND ≤2 messages) or a known
102
+ * internal-prompt signature (title generation, watchdog, commit-message, factory
103
+ * worker). Every `stats` figure above, `topics` counts, and `needsAttention` are
104
+ * computed over the AGENT set ONLY — `sessionsImported` is the real agent count,
105
+ * not the raw row count. This is the number that was dropped.
106
+ */
107
+ utilityCount: number;
75
108
  needsAttention: IndexedSession[];
76
- topics: Array<{
77
- key: string;
78
- label: string;
79
- count: number;
80
- group: TraceTopicGroup;
81
- }>;
109
+ topics: TopicItem[];
82
110
  failures: {
83
111
  byToolError: Array<{
84
112
  tool: string;
@@ -104,11 +132,56 @@ export interface IndexedSession {
104
132
  title: string;
105
133
  repo: string;
106
134
  device: string;
135
+ /** The harness that produced the session (claude/codex/rush/grok/…). Same as `harness`. */
107
136
  agent: string;
108
137
  model: string;
138
+ /** Corpus classification (PHNX-3474). Always `'agent'` here — utility rows are excluded. */
139
+ kind: SessionKind;
109
140
  severity: number;
110
141
  flags: string[];
111
142
  }
143
+ /**
144
+ * One example session under a topic tile — the shape the console drill-down consumes.
145
+ * Carries `kind` + `harness` (PHNX-3474) so the console can filter a tile's session
146
+ * list by corpus class and by harness. Refs on a topic tile are always `'agent'`
147
+ * (utility rows never reach a bucket), but the field is explicit for the consumer.
148
+ */
149
+ export interface TopicSessionRef {
150
+ id: string;
151
+ title: string;
152
+ kind: SessionKind;
153
+ harness: string;
154
+ }
155
+ /**
156
+ * Corpus class of a session (PHNX-3474). `utility` is internal machine plumbing —
157
+ * a single-shot call with no tool use and ≤2 messages, or one whose topic/label
158
+ * matches a known internal-prompt signature (title generation, watchdog,
159
+ * commit-message writer, factory worker). Everything else is `agent`: real agent
160
+ * work the Evals console counts and scores. Utility rows are tagged, never deleted,
161
+ * and excluded from every index statistic.
162
+ */
163
+ export type SessionKind = 'utility' | 'agent';
164
+ /**
165
+ * Classify a session as internal `utility` plumbing vs real `agent` work (PHNX-3474).
166
+ * `toolCallCount` is the AUTHORITATIVE per-session tool-call count from the loaded
167
+ * `tool_calls` rows (the row's own `tool_call_count` column is a fallback for a row
168
+ * whose calls weren't loaded). A session is `utility` when a known internal-prompt
169
+ * signature matches its topic/label, OR it made no tool call AND has ≤2 messages —
170
+ * the single-shot machine-call shape. Otherwise it is `agent`.
171
+ */
172
+ export declare function classifySessionKind(row: Pick<SyncRow, 'topic' | 'label' | 'message_count' | 'tool_call_count'>, toolCallCount: number): SessionKind;
173
+ /**
174
+ * One topic bucket in the treemap. `sessions` carries up to {@link TOPIC_SESSION_CAP}
175
+ * example refs so the console can drill from the tile into its session list — a tile
176
+ * with no refs renders display-only (PHNX-3408). `count` stays the true total.
177
+ */
178
+ export interface TopicItem {
179
+ key: string;
180
+ label: string;
181
+ count: number;
182
+ group: TraceTopicGroup;
183
+ sessions: TopicSessionRef[];
184
+ }
112
185
  /** A row from `tool_calls`. `ordinal`/`timestamp` order calls within a session for computeInsights(). */
113
186
  export interface ToolCallRow {
114
187
  session_id: string;
@@ -129,6 +202,31 @@ export interface ToolCallRow {
129
202
  error: string | null;
130
203
  parse_error: string | null;
131
204
  }
205
+ /**
206
+ * Active time for a session in the index shard: its recorded span minus every idle
207
+ * gap > 120s (PHNX-3457). The index build already holds the ordered `tool_calls`
208
+ * rows for the whole corpus, so idle is derived from them here — no transcript
209
+ * re-parse. A cursor sweeps the span from `sessionStartMs`: each stretch where the
210
+ * cursor sits idle for more than the threshold before the next call starts is
211
+ * subtracted, and idle is measured from a call's END (its own `end_timestamp` when
212
+ * known, else its start) so a call's own blocking duration is never mistaken for
213
+ * idle. Crucially the sweep also books the gaps at the two BOUNDARIES — before the
214
+ * first call and after the last call to the session end — so a session with a lone
215
+ * tool call that was then abandoned and resumed hours later (the case a
216
+ * between-calls-only measure missed entirely, leaving the whole 345h span counted
217
+ * as active) has that trailing idle stripped. A session end is `sessionStartMs +
218
+ * spanMs`, so the two agree by construction.
219
+ *
220
+ * Bounded to `[0, spanMs]`. A session with NO tool calls returns the full span
221
+ * unchanged rather than a fabricated zero: there is no tool-call evidence of idle
222
+ * either way, and treating a chat-only turn as 100% idle would be a worse error
223
+ * than leaving its span uncorrected. Where the full event stream IS available (a
224
+ * per-session `SessionDetail`), {@link activeMsFromTrajectory} is used instead —
225
+ * it sees message events this call-only approximation cannot, so the two are close
226
+ * but not identical by design (the corpus-scale index build cannot afford the
227
+ * per-session parse the detail view does).
228
+ */
229
+ export declare function sessionActiveMs(spanMs: number, sessionCalls: ToolCallRow[], sessionStartMs: number): number;
132
230
  /** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
133
231
  export declare function failureDescription(call: ToolCallRow, cause: TraceFailureCause): string;
134
232
  /** Build the redacted rich console shard from indexed metadata and derived caches. */
@@ -138,7 +236,16 @@ export interface SessionDetail {
138
236
  schema: 1;
139
237
  id: string;
140
238
  meta: {
239
+ /** Raw wall-clock span (last event − first event), idle time included. */
141
240
  spanMs: number;
241
+ /**
242
+ * Active time: `spanMs` minus every idle gap > 120s (PHNX-3457). A session
243
+ * resumed hours later, or left idle mid-turn, inflates `spanMs` with wall-clock
244
+ * the agent did no work in — active time strips those gaps so a duration reads
245
+ * as effort, not calendar span. This is what the console's duration median/p90
246
+ * should trust; `spanMs` stays available as the raw figure.
247
+ */
248
+ activeMs: number;
142
249
  turns: number;
143
250
  tools: number;
144
251
  errorCount: number;