@phnx-labs/agents-cli 1.22.70 → 1.22.72

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/README.md +31 -1
  3. package/dist/bootstrap.js +4 -4
  4. package/dist/commands/repo.js +2 -2
  5. package/dist/commands/sessions-export.d.ts +5 -1
  6. package/dist/commands/sessions-export.js +100 -24
  7. package/dist/commands/sessions-import.d.ts +2 -1
  8. package/dist/commands/sessions-import.js +85 -21
  9. package/dist/lib/accounting/usage-sync.d.ts +1 -1
  10. package/dist/lib/accounting/usage-sync.js +3 -3
  11. package/dist/lib/browser/ipc.d.ts +34 -0
  12. package/dist/lib/browser/ipc.js +140 -19
  13. package/dist/lib/browser/types.d.ts +3 -1
  14. package/dist/lib/daemon/auth-sync-service.js +1 -1
  15. package/dist/lib/daemon/browser-task-reap-service.js +1 -1
  16. package/dist/lib/daemon/daemon.js +13 -3
  17. package/dist/lib/daemon/heartbeat-service.js +3 -3
  18. package/dist/lib/daemon/keychain-reap-service.js +1 -1
  19. package/dist/lib/daemon/runner.d.ts +18 -1
  20. package/dist/lib/daemon/runner.js +231 -78
  21. package/dist/lib/daemon/self-heal-service.js +13 -3
  22. package/dist/lib/daemon/self-update-service.d.ts +174 -0
  23. package/dist/lib/daemon/self-update-service.js +353 -0
  24. package/dist/lib/daemon/state-dir-check-service.js +3 -3
  25. package/dist/lib/daemon/usage-sync-service.js +1 -1
  26. package/dist/lib/daemon/watchdog-service.js +4 -4
  27. package/dist/lib/daemon-services.d.ts +1 -1
  28. package/dist/lib/daemon-services.js +5 -0
  29. package/dist/lib/device-config.d.ts +12 -1
  30. package/dist/lib/device-config.js +63 -13
  31. package/dist/lib/exec-bounded.d.ts +52 -0
  32. package/dist/lib/exec-bounded.js +113 -0
  33. package/dist/lib/feed/events.d.ts +22 -14
  34. package/dist/lib/feed/events.js +84 -44
  35. package/dist/lib/fleet-shared-state.d.ts +12 -5
  36. package/dist/lib/fleet-shared-state.js +50 -20
  37. package/dist/lib/fs-atomic.d.ts +11 -0
  38. package/dist/lib/fs-atomic.js +60 -0
  39. package/dist/lib/hosts/reconcile.d.ts +11 -4
  40. package/dist/lib/hosts/reconcile.js +31 -5
  41. package/dist/lib/project-resources.d.ts +12 -0
  42. package/dist/lib/project-resources.js +138 -0
  43. package/dist/lib/routine-process-cleanup.d.ts +2 -2
  44. package/dist/lib/routine-process-cleanup.js +45 -34
  45. package/dist/lib/secrets/reaper.d.ts +2 -2
  46. package/dist/lib/secrets/reaper.js +13 -10
  47. package/dist/lib/secrets/reserved-sync.d.ts +1 -1
  48. package/dist/lib/secrets/reserved-sync.js +4 -4
  49. package/dist/lib/self-update.d.ts +21 -8
  50. package/dist/lib/self-update.js +54 -31
  51. package/dist/lib/session/sync/backend.d.ts +61 -0
  52. package/dist/lib/session/sync/backend.js +89 -0
  53. package/dist/lib/session/sync/managed-config.d.ts +29 -0
  54. package/dist/lib/session/sync/managed-config.js +23 -0
  55. package/dist/lib/session/sync/managed-key.d.ts +45 -0
  56. package/dist/lib/session/sync/managed-key.js +128 -0
  57. package/dist/lib/session/sync/net-client.d.ts +65 -0
  58. package/dist/lib/session/sync/net-client.js +117 -0
  59. package/dist/lib/session/sync/provision.d.ts +19 -0
  60. package/dist/lib/session/sync/provision.js +38 -0
  61. package/dist/lib/session/sync/r2.d.ts +5 -2
  62. package/dist/lib/session/sync/r2.js +5 -2
  63. package/dist/lib/session/sync/worker-template.d.ts +6 -0
  64. package/dist/lib/session/sync/worker-template.js +847 -0
  65. package/dist/lib/tmux/orphan-reap.js +6 -4
  66. package/dist/lib/tmux/session.js +4 -1
  67. package/dist/lib/traces/classify.d.ts +8 -1
  68. package/dist/lib/traces/insights.d.ts +13 -1
  69. package/dist/lib/traces/insights.js +78 -3
  70. package/dist/lib/traces/sync.js +8 -3
  71. package/dist/lib/traces/worker-template.js +9 -5
  72. package/package.json +1 -1
@@ -99,7 +99,7 @@
99
99
  * helper); it is not a parsing bug {@link parseTmuxSessionMarker} can fix.
100
100
  */
101
101
  import { execFile } from 'child_process';
102
- import * as fs from 'fs';
102
+ import * as fsp from 'fs/promises';
103
103
  import { promisify } from 'util';
104
104
  const execFileAsync = promisify(execFile);
105
105
  /** Ceiling on the `ps` snapshot so a wedged `ps` can never stall the daemon tick. */
@@ -407,7 +407,9 @@ export async function readAgentProcesses(opts = {}) {
407
407
  if (process.platform === 'win32')
408
408
  return [];
409
409
  const scope = opts.pids && opts.pids.length > 0 ? ['-p', opts.pids.join(',')] : ['-A'];
410
- const hasProc = fs.existsSync('/proc/self/environ');
410
+ // Async /proc reads — this runs on the daemon's tmux-reap tick, so a sync scan
411
+ // of the whole process table's environ files would freeze the loop (PHNX-3695).
412
+ const hasProc = await fsp.access('/proc/self/environ').then(() => true, () => false);
411
413
  const args = hasProc ? [...scope, '-o', 'pid=,ppid=,args='] : [...scope, '-E', '-o', 'pid=,ppid=,args='];
412
414
  let stdout;
413
415
  try {
@@ -426,7 +428,7 @@ export async function readAgentProcesses(opts = {}) {
426
428
  for (const row of rows) {
427
429
  let blob;
428
430
  try {
429
- blob = fs.readFileSync(`/proc/${row.pid}/environ`, 'utf8');
431
+ blob = await fsp.readFile(`/proc/${row.pid}/environ`, 'utf8');
430
432
  }
431
433
  catch {
432
434
  continue; // exited between ps and the read, or not ours to inspect
@@ -450,7 +452,7 @@ export async function readPaneOwners(socket) {
450
452
  // No socket file at all means no server was ever started here — for the
451
453
  // shared agents socket this is reliable (tmux unlinks its own socket on
452
454
  // exit), so this is a confident, reliable EMPTY read, not an unknown one.
453
- if (!fs.existsSync(socket))
455
+ if (!(await fsp.access(socket).then(() => true, () => false)))
454
456
  return { ok: true, owners: new Map(), panePids: new Set() };
455
457
  const { runTmux } = await import('./binary.js');
456
458
  try {
@@ -10,6 +10,7 @@
10
10
  * `tmux list-sessions` and prunes stale entries on the fly.
11
11
  */
12
12
  import * as fs from 'fs';
13
+ import * as fsp from 'fs/promises';
13
14
  import * as os from 'os';
14
15
  import * as path from 'path';
15
16
  import { runTmux, TmuxCommandError } from './binary.js';
@@ -412,7 +413,9 @@ export async function reapDeadTmuxPanes(socket, opts = {}) {
412
413
  result.warnings = orphans.warnings;
413
414
  result.processes = opts.dryRun ? orphans.candidates.length : orphans.killed;
414
415
  result.processDetails = orphans.details;
415
- if (!fs.existsSync(sock))
416
+ // Async existence check — this runs on the daemon's tmux-reap tick, so a sync
417
+ // `existsSync` would block the shared event loop (PHNX-3695).
418
+ if (!(await fsp.access(sock).then(() => true, () => false)))
416
419
  return result;
417
420
  const res = await runTmux({
418
421
  socket: sock,
@@ -1,5 +1,12 @@
1
1
  export type TraceTopicGroup = 'code' | 'research' | 'review' | 'content' | 'ops';
2
- export type TraceFailureCause = 'real' | 'guard' | 'hook';
2
+ /**
3
+ * `real`/`guard`/`hook` are the cause buckets of a FAILED tool call (`classifyCause`).
4
+ * `behavioral` is different in kind: a silent failure with no error code at all — the
5
+ * agent went idle after its last event and a human had to nudge it. It is derived from
6
+ * per-session friction facets (`computeBehavioralPatterns` in `insights.ts`), never from
7
+ * `classifyCause`, so a failed-tool-call classifier never returns it.
8
+ */
9
+ export type TraceFailureCause = 'real' | 'guard' | 'hook' | 'behavioral';
3
10
  /** Per-bucket aggregate stats for one day, stored in the rolling bucketHistory. */
4
11
  export interface BucketStats {
5
12
  key: string;
@@ -29,6 +29,7 @@ import { type TraceFailureCause } from './classify.js';
29
29
  import type { FailurePhenotype } from './phenotype.js';
30
30
  import { type LatencyInsight } from './segments.js';
31
31
  import { type SyncRow, type ToolCallRow, type TracesIndexShard } from './sync.js';
32
+ import type { InsightFacets } from '../session/insights.js';
32
33
  export interface FailureSignature {
33
34
  tool: string;
34
35
  cause: TraceFailureCause;
@@ -97,4 +98,15 @@ export declare function normalizeErrorKey(desc: string, raw: string | null): str
97
98
  * estimate, not ground truth; it is not inflated by folding in ordinary
98
99
  * processing time between unrelated calls.
99
100
  */
100
- export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null, phenotypes?: ReadonlyMap<string, FailurePhenotype | null>): ComputedInsights;
101
+ export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null, phenotypes?: ReadonlyMap<string, FailurePhenotype | null>, behavioralPatterns?: readonly FailurePattern[]): ComputedInsights;
102
+ /**
103
+ * Promote the per-session silent-stall friction signals — already computed by
104
+ * `computeInsightFacets` in `session/insights.ts` and keyed `silent stall: <bucket>`
105
+ * — into cross-session behavioral `FailurePattern`s. These are silent failures
106
+ * with NO error code: the agent went idle after its last event and a human had to
107
+ * nudge it. They never surface through `computeInsights` (which only clusters
108
+ * `outcome === 'error'` tool calls) and were previously only a `needsAttention`
109
+ * friction counter, never a ranked issue. Cause is `behavioral`; the signature
110
+ * `tool` is the synthetic `silent-stall` and `key` is the duration bucket.
111
+ */
112
+ export declare function computeBehavioralPatterns(facetsBySession: ReadonlyMap<string, Pick<InsightFacets, 'frictionSignals'>>, prevShard?: TracesIndexShard | null): FailurePattern[];
@@ -132,7 +132,7 @@ function labelFor(tool, cause, key) {
132
132
  * estimate, not ground truth; it is not inflated by folding in ordinary
133
133
  * processing time between unrelated calls.
134
134
  */
135
- export function computeInsights(rows, calls, prevShard, phenotypes) {
135
+ export function computeInsights(rows, calls, prevShard, phenotypes, behavioralPatterns = []) {
136
136
  const bySession = new Map();
137
137
  for (const call of calls) {
138
138
  const list = bySession.get(call.session_id);
@@ -226,13 +226,88 @@ export function computeInsights(rows, calls, prevShard, phenotypes) {
226
226
  drift,
227
227
  };
228
228
  });
229
- const wastedMsTotal = allPatterns.reduce((sum, p) => sum + p.wastedMs, 0);
230
- const failurePatterns = [...allPatterns]
229
+ // Behavioral patterns (silent stalls — no failed tool call, so absent from the
230
+ // error-anchored clustering above) join the SAME ranking and total, so the top-K
231
+ // is by impact across both kinds and `wastedMsTotal` counts the idle time too.
232
+ const combined = [...allPatterns, ...behavioralPatterns];
233
+ const wastedMsTotal = combined.reduce((sum, p) => sum + p.wastedMs, 0);
234
+ const failurePatterns = [...combined]
231
235
  .sort((a, b) => b.wastedMs - a.wastedMs || b.occurrences - a.occurrences || a.id.localeCompare(b.id))
232
236
  .slice(0, TOP_K_PATTERNS);
233
237
  const latency = computeLatency(firstToolSegments(rows, bySession));
234
238
  return { failurePatterns, wastedMsTotal, latency };
235
239
  }
240
+ // ---------------------------------------------------------------------------
241
+ // Behavioral patterns — silent failures with no error code
242
+ // ---------------------------------------------------------------------------
243
+ /**
244
+ * Estimated agent-owned idle ms per silent-stall bucket. A bucket's midpoint,
245
+ * capped at MAX_GAP_ATTRIBUTION_MS exactly like the tool-error gap attribution
246
+ * above so one long overnight stall can't book hours of "waste" — the same
247
+ * honesty bound the error path uses. The open-ended buckets are the cap.
248
+ */
249
+ const SILENT_STALL_WASTED_MS = {
250
+ '5-15m': 10 * 60_000,
251
+ '15-60m': MAX_GAP_ATTRIBUTION_MS,
252
+ '1h+': MAX_GAP_ATTRIBUTION_MS,
253
+ };
254
+ /**
255
+ * Promote the per-session silent-stall friction signals — already computed by
256
+ * `computeInsightFacets` in `session/insights.ts` and keyed `silent stall: <bucket>`
257
+ * — into cross-session behavioral `FailurePattern`s. These are silent failures
258
+ * with NO error code: the agent went idle after its last event and a human had to
259
+ * nudge it. They never surface through `computeInsights` (which only clusters
260
+ * `outcome === 'error'` tool calls) and were previously only a `needsAttention`
261
+ * friction counter, never a ranked issue. Cause is `behavioral`; the signature
262
+ * `tool` is the synthetic `silent-stall` and `key` is the duration bucket.
263
+ */
264
+ export function computeBehavioralPatterns(facetsBySession, prevShard) {
265
+ const groups = new Map();
266
+ for (const [sessionId, facets] of facetsBySession) {
267
+ for (const [signal, count] of Object.entries(facets.frictionSignals ?? {})) {
268
+ if (count <= 0)
269
+ continue;
270
+ const match = /^silent stall: (.+)$/.exec(signal);
271
+ if (!match)
272
+ continue;
273
+ const bucket = match[1];
274
+ let group = groups.get(bucket);
275
+ if (!group) {
276
+ group = { bucket, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
277
+ groups.set(bucket, group);
278
+ }
279
+ group.occurrences += count;
280
+ group.sessions.add(sessionId);
281
+ group.wastedMs += (SILENT_STALL_WASTED_MS[bucket] ?? MAX_GAP_ATTRIBUTION_MS) * count;
282
+ if (group.examples.length < MAX_EXAMPLE_SESSIONS && !group.examples.includes(sessionId)) {
283
+ group.examples.push(sessionId);
284
+ }
285
+ }
286
+ }
287
+ const prevById = new Map((prevShard?.failurePatterns ?? []).map((p) => [p.id, p]));
288
+ return [...groups.values()].map((group) => {
289
+ const id = hashSignature('silent-stall', 'behavioral', group.bucket, null);
290
+ const prev = prevById.get(id);
291
+ const drift = !prev
292
+ ? 'up'
293
+ : group.occurrences > prev.occurrences
294
+ ? 'up'
295
+ : group.occurrences < prev.occurrences
296
+ ? 'down'
297
+ : 'flat';
298
+ return {
299
+ id,
300
+ label: `Agent silent stall (${group.bucket}) — idle until nudged`,
301
+ signature: { tool: 'silent-stall', cause: 'behavioral', key: group.bucket },
302
+ phenotype: null,
303
+ sessions: group.sessions.size,
304
+ occurrences: group.occurrences,
305
+ wastedMs: group.wastedMs,
306
+ exampleSessionIds: group.examples,
307
+ drift,
308
+ };
309
+ });
310
+ }
236
311
  /** Synthesize one-step SegmentSessions carrying only the time-to-first-tool offset, for computeLatency() reuse. */
237
312
  function firstToolSegments(rows, bySession) {
238
313
  return rows.flatMap((row) => {
@@ -30,7 +30,7 @@ import { knownSecretValuesFromEnv, redactSecrets } from '../redact.js';
30
30
  import { getRuntimeStateDir } from '../state.js';
31
31
  import { resolveTracesBackend } from './backend.js';
32
32
  import { classifyCause, classifyTopic, computeDriftSignal, } from './classify.js';
33
- import { computeInsights } from './insights.js';
33
+ import { computeBehavioralPatterns, computeInsights } from './insights.js';
34
34
  import { classifyPhenotype, recoveredAfterErrors } from './phenotype.js';
35
35
  import { buildSessionDetailV2 } from './schema2-build.js';
36
36
  /**
@@ -529,7 +529,9 @@ export function buildIndexShard(rows, device, owner, prevShard) {
529
529
  topicCounts.set(topic.key, bucket);
530
530
  }
531
531
  const failedCalls = agentCalls.filter((call) => call.outcome === 'error');
532
- const byCause = { real: 0, guard: 0, hook: 0 };
532
+ // `behavioral` is not a failed-tool-call cause (classifyCause never returns it),
533
+ // so it stays 0 in this tool-error split; it surfaces as its own failurePatterns.
534
+ const byCause = { real: 0, guard: 0, hook: 0, behavioral: 0 };
533
535
  const failureCounts = new Map();
534
536
  for (const call of failedCalls) {
535
537
  const cause = classifyCause(call);
@@ -585,7 +587,10 @@ export function buildIndexShard(rows, device, owner, prevShard) {
585
587
  const prevHistory = prevShard?.bucketHistory ?? [];
586
588
  const bucketHistory = [...prevHistory, todayStats].slice(-14);
587
589
  const driftSignals = computeDriftSignal(prevHistory, todayStats);
588
- const patternInsights = computeInsights(agentRows, agentCalls, prevShard, phenotypes);
590
+ // Silent-stall friction (per-session, no failed tool call) becomes cross-session
591
+ // behavioral FailurePatterns, ranked into the same top-K by wasted idle time.
592
+ const behavioralPatterns = computeBehavioralPatterns(insights, prevShard);
593
+ const patternInsights = computeInsights(agentRows, agentCalls, prevShard, phenotypes, behavioralPatterns);
589
594
  // Per-session roster (PHNX-3483): one flat scalar row per agent session, the raw
590
595
  // material the Rush console filters and re-aggregates client-side. `durationMs`
591
596
  // reuses `sessionActiveMs` (the value behind `stats.medianMs`; 0 for a null-duration
@@ -170,7 +170,11 @@ function mergeIndexShards(shards, owner) {
170
170
  topicCounts.set(t.key, cur);
171
171
  }
172
172
  const byToolError = new Map();
173
- let real = 0, guard = 0, hook = 0;
173
+ // Accumulate byCause over WHATEVER cause keys each shard carries (real / guard /
174
+ // hook / behavioral / any future TraceFailureCause) rather than a hand-enumerated
175
+ // set — a hardcoded {real,guard,hook} silently dropped a new member from the /all
176
+ // view and made the cause sum NaN (RUSH-2988).
177
+ const byCause = {};
174
178
  for (const s of sorted) {
175
179
  const f = s.failures || {};
176
180
  for (const e of f.byToolError || []) {
@@ -179,9 +183,9 @@ function mergeIndexShards(shards, owner) {
179
183
  cur.count += e.count || 0;
180
184
  byToolError.set(key, cur);
181
185
  }
182
- real += (f.byCause && f.byCause.real) || 0;
183
- guard += (f.byCause && f.byCause.guard) || 0;
184
- hook += (f.byCause && f.byCause.hook) || 0;
186
+ for (const [cause, n] of Object.entries((f.byCause) || {})) {
187
+ byCause[cause] = (byCause[cause] || 0) + (n || 0);
188
+ }
185
189
  }
186
190
  // Fold failure patterns by their stable id, the same way topics/byToolError fold
187
191
  // by key above — a signature that fired on two devices (a rate limit on the laptop
@@ -220,7 +224,7 @@ function mergeIndexShards(shards, owner) {
220
224
  topics: Array.from(topicCounts.values()).sort((a, b) => b.count - a.count),
221
225
  failures: {
222
226
  byToolError: Array.from(byToolError.values()).sort((a, b) => b.count - a.count).slice(0, 50),
223
- byCause: { real, guard, hook },
227
+ byCause,
224
228
  },
225
229
  failurePatterns: Array.from(patternById.values())
226
230
  .sort((a, b) => (b.wastedMs || 0) - (a.wastedMs || 0)).slice(0, 25),
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@phnx-labs/agents-cli",
3
- "version": "1.22.70",
3
+ "version": "1.22.72",
4
4
  "description": "One CLI for all your AI coding agents - versions, config, cloud dispatch, sessions, and teams (now with first-class Grok Build CLI support)",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",