@phnx-labs/agents-cli 1.22.56 → 1.22.58

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (146) hide show
  1. package/CHANGELOG.md +70 -0
  2. package/README.md +4 -4
  3. package/dist/bootstrap.js +11 -2
  4. package/dist/cli/command-registry.d.ts +0 -1
  5. package/dist/cli/command-registry.js +0 -3
  6. package/dist/commands/accounts.js +7 -3
  7. package/dist/commands/apply.js +10 -2
  8. package/dist/commands/exec.js +1 -1
  9. package/dist/commands/fork.d.ts +23 -10
  10. package/dist/commands/fork.js +115 -58
  11. package/dist/commands/hooks.js +4 -4
  12. package/dist/commands/insights.d.ts +7 -5
  13. package/dist/commands/insights.js +16 -9
  14. package/dist/commands/monitors.js +11 -0
  15. package/dist/commands/perf.d.ts +16 -7
  16. package/dist/commands/perf.js +29 -20
  17. package/dist/commands/prune.js +5 -3
  18. package/dist/commands/routines.d.ts +8 -0
  19. package/dist/commands/routines.js +57 -3
  20. package/dist/commands/rules.js +1 -1
  21. package/dist/commands/sessions-picker.d.ts +11 -0
  22. package/dist/commands/sessions-picker.js +16 -0
  23. package/dist/commands/sessions.js +1 -0
  24. package/dist/commands/share.d.ts +14 -0
  25. package/dist/commands/share.js +43 -2
  26. package/dist/commands/ssh.js +24 -14
  27. package/dist/commands/status.js +1 -1
  28. package/dist/commands/sync.js +83 -7
  29. package/dist/commands/traces.js +7 -0
  30. package/dist/commands/trash.d.ts +2 -2
  31. package/dist/commands/trash.js +2 -6
  32. package/dist/commands/versions.d.ts +2 -2
  33. package/dist/commands/versions.js +1 -10
  34. package/dist/commands/view.d.ts +2 -2
  35. package/dist/commands/view.js +7 -6
  36. package/dist/index.d.ts +1 -0
  37. package/dist/index.js +14 -0
  38. package/dist/lib/account-registry.d.ts +5 -1
  39. package/dist/lib/account-registry.js +47 -14
  40. package/dist/lib/accounting/capacity.d.ts +18 -7
  41. package/dist/lib/accounting/capacity.js +19 -8
  42. package/dist/lib/accounting/usage-ingest.d.ts +1 -0
  43. package/dist/lib/accounting/usage-ingest.js +75 -0
  44. package/dist/lib/accounting/usage-sync.d.ts +97 -0
  45. package/dist/lib/accounting/usage-sync.js +203 -0
  46. package/dist/lib/accounting/usage.d.ts +48 -2
  47. package/dist/lib/accounting/usage.js +79 -2
  48. package/dist/lib/agent-spec/agents.js +1 -1
  49. package/dist/lib/analytics/mix-commands.d.ts +8 -7
  50. package/dist/lib/analytics/mix-commands.js +50 -73
  51. package/dist/lib/auth-mint.d.ts +11 -1
  52. package/dist/lib/auth-mint.js +21 -6
  53. package/dist/lib/browser/ipc.d.ts +8 -0
  54. package/dist/lib/browser/ipc.js +87 -0
  55. package/dist/lib/browser/service.d.ts +19 -0
  56. package/dist/lib/browser/service.js +96 -11
  57. package/dist/lib/browser/sessions-list.js +10 -1
  58. package/dist/lib/daemon/daemon.js +5 -0
  59. package/dist/lib/daemon/runner.d.ts +3 -0
  60. package/dist/lib/daemon/runner.js +95 -53
  61. package/dist/lib/daemon/usage-sync-service.d.ts +21 -0
  62. package/dist/lib/daemon/usage-sync-service.js +42 -0
  63. package/dist/lib/daemon-services.d.ts +1 -1
  64. package/dist/lib/daemon-services.js +5 -0
  65. package/dist/lib/device-config.d.ts +17 -6
  66. package/dist/lib/device-config.js +25 -11
  67. package/dist/lib/devices/connect.d.ts +17 -8
  68. package/dist/lib/devices/connect.js +31 -14
  69. package/dist/lib/devices/pool.d.ts +4 -3
  70. package/dist/lib/devices/pool.js +13 -5
  71. package/dist/lib/doctor-diff.js +77 -7
  72. package/dist/lib/exec.d.ts +6 -41
  73. package/dist/lib/exec.js +6 -41
  74. package/dist/lib/fleet/manifest.d.ts +17 -0
  75. package/dist/lib/fleet/manifest.js +26 -0
  76. package/dist/lib/git.d.ts +13 -1
  77. package/dist/lib/git.js +36 -7
  78. package/dist/lib/harness/adapter.d.ts +7 -7
  79. package/dist/lib/harness/adapters/claude.js +3 -2
  80. package/dist/lib/hooks/install.d.ts +27 -11
  81. package/dist/lib/hooks/install.js +42 -17
  82. package/dist/lib/hosts/reconnect.d.ts +52 -203
  83. package/dist/lib/hosts/reconnect.js +64 -284
  84. package/dist/lib/hosts/remote-cmd.d.ts +9 -0
  85. package/dist/lib/hosts/remote-cmd.js +22 -0
  86. package/dist/lib/installations/migrate.d.ts +6 -120
  87. package/dist/lib/installations/migrate.js +27 -259
  88. package/dist/lib/installations/shims.d.ts +13 -95
  89. package/dist/lib/installations/shims.js +22 -139
  90. package/dist/lib/installations/store.js +1 -1
  91. package/dist/lib/installations/versions.d.ts +26 -133
  92. package/dist/lib/installations/versions.js +41 -204
  93. package/dist/lib/perf/db.d.ts +1 -1
  94. package/dist/lib/perf/db.js +1 -1
  95. package/dist/lib/plugins/skills.d.ts +8 -1
  96. package/dist/lib/plugins/skills.js +18 -2
  97. package/dist/lib/refresh.d.ts +9 -0
  98. package/dist/lib/refresh.js +3 -1
  99. package/dist/lib/routine-readiness.d.ts +15 -1
  100. package/dist/lib/routine-readiness.js +41 -0
  101. package/dist/lib/sandbox.d.ts +4 -1
  102. package/dist/lib/sandbox.js +30 -1
  103. package/dist/lib/secrets/agent.d.ts +80 -225
  104. package/dist/lib/secrets/agent.js +139 -401
  105. package/dist/lib/secrets/bundles.d.ts +73 -222
  106. package/dist/lib/secrets/bundles.js +168 -467
  107. package/dist/lib/secrets/reaper.d.ts +28 -70
  108. package/dist/lib/secrets/reaper.js +30 -85
  109. package/dist/lib/secrets/remote.d.ts +42 -129
  110. package/dist/lib/secrets/remote.js +55 -173
  111. package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
  112. package/dist/lib/self-heal/checks/install-staging.js +96 -0
  113. package/dist/lib/self-heal/registry.js +2 -0
  114. package/dist/lib/self-heal/types.d.ts +1 -1
  115. package/dist/lib/self-update.d.ts +23 -0
  116. package/dist/lib/self-update.js +50 -0
  117. package/dist/lib/session/active.d.ts +16 -32
  118. package/dist/lib/session/active.js +10 -68
  119. package/dist/lib/session/db.d.ts +24 -36
  120. package/dist/lib/session/db.js +143 -44
  121. package/dist/lib/session/discover.d.ts +6 -58
  122. package/dist/lib/session/discover.js +5 -43
  123. package/dist/lib/session/fork.d.ts +45 -26
  124. package/dist/lib/session/fork.js +32 -95
  125. package/dist/lib/session/parse.d.ts +1 -19
  126. package/dist/lib/session/parse.js +2 -15
  127. package/dist/lib/session/tool-calls.d.ts +43 -1
  128. package/dist/lib/session/tool-calls.js +74 -44
  129. package/dist/lib/session/tool-store.d.ts +33 -2
  130. package/dist/lib/session/tool-store.js +56 -3
  131. package/dist/lib/staleness/writers/sources.d.ts +5 -0
  132. package/dist/lib/staleness/writers/sources.js +2 -1
  133. package/dist/lib/startup/command-registry.d.ts +8 -2
  134. package/dist/lib/startup/command-registry.js +12 -4
  135. package/dist/lib/sync-status.d.ts +22 -0
  136. package/dist/lib/sync-status.js +27 -0
  137. package/dist/lib/sync-umbrella.d.ts +9 -0
  138. package/dist/lib/sync-umbrella.js +21 -2
  139. package/dist/lib/traces/insights.d.ts +47 -14
  140. package/dist/lib/traces/insights.js +92 -21
  141. package/dist/lib/traces/phenotype.d.ts +23 -3
  142. package/dist/lib/traces/phenotype.js +72 -24
  143. package/dist/lib/traces/sync.d.ts +15 -0
  144. package/dist/lib/traces/sync.js +104 -19
  145. package/dist/lib/traces/worker-template.js +154 -1
  146. package/package.json +1 -1
@@ -1,5 +1,5 @@
1
1
  import type Database from '../sqlite.js';
2
- import { type IndexedToolCall } from './tool-calls.js';
2
+ import { type IndexedToolCall, type EventToolScanResumePoint } from './tool-calls.js';
3
3
  import type { SessionMeta } from './types.js';
4
4
  export declare function canonicalToolLedgerPath(filePath: string): string;
5
5
  /** Resolve the transcript that actually carries tool events for split-file harnesses. */
@@ -12,7 +12,14 @@ export declare function purgeMissingToolCallsInDirectory(db: Database.Database,
12
12
  export interface ToolScanResumePoint {
13
13
  /** Serialized ToolCallCollector snapshot at `parsedOffset`. */
14
14
  parserState: string;
15
- /** Byte offset just past the last complete record consumed. */
15
+ /**
16
+ * Where a later scan may resume. For the streaming (claude/codex) path this is
17
+ * a BYTE offset just past the last complete record consumed. For the full-file
18
+ * harness path (`planEventToolResume`) it is instead the COUNT of normalized
19
+ * events already folded — the two never share a session (a session is one
20
+ * agent, and the streaming path is claude/codex only), so the column carries
21
+ * whichever meaning that agent's resumer wrote.
22
+ */
16
23
  parsedOffset: number;
17
24
  }
18
25
  export interface PersistToolCallsOptions {
@@ -37,3 +44,27 @@ export declare function persistToolCalls(db: Database.Database, session: Session
37
44
  fileMtimeMs: number;
38
45
  fileSize: number;
39
46
  }, options?: PersistToolCallsOptions): void;
47
+ /**
48
+ * Decide whether a changed full-file-harness session can RESUME its tool index
49
+ * from the last scan instead of re-deriving every call (PHNX-3411).
50
+ *
51
+ * The warm-tick indexer re-derives tool calls for a changed session on every
52
+ * tick, and for an ACTIVE large session that means re-sanitizing tens of
53
+ * thousands of calls each time — the synchronous work that wedged the daemon
54
+ * event loop. Streaming harnesses (claude/codex) already resume from a byte
55
+ * offset; every other harness re-parses the whole file, so this brings them the
56
+ * same benefit at the EVENT level: the ledger stores how many events were folded
57
+ * (`parsed_offset`) plus the collector snapshot (`parser_state`), and a later
58
+ * scan folds only the newly appended events.
59
+ *
60
+ * Returns the prior resume point when it is safe to append, or `null` (full
61
+ * re-scan) whenever the stored prefix may no longer describe the current file:
62
+ * a different extractor, no recorded resume point, a source the ledger row does
63
+ * not describe, a transcript that shrank below what was already parsed (a
64
+ * truncation/rewrite, not an append), more events already folded than the file
65
+ * now yields, or a snapshot that does not read back.
66
+ */
67
+ export declare function planEventToolResume(db: Database.Database, sessionId: string, sourcePath: string, stamp: {
68
+ fileMtimeMs: number;
69
+ fileSize: number;
70
+ }, currentEventCount: number): EventToolScanResumePoint | null;
@@ -75,13 +75,14 @@ export function persistToolCalls(db, session, calls, sourceStamp, options = {})
75
75
  const sourcePath = toolEvidenceSourcePath(session.filePath, session.agent);
76
76
  const insertCall = db.prepare(`
77
77
  INSERT INTO tool_calls (
78
- call_key, session_id, ordinal, source_call_id, timestamp, tool, input,
78
+ call_key, session_id, ordinal, source_call_id, timestamp, end_timestamp, tool, input,
79
79
  outcome, exit_code, status_code, error_code, output, error, parse_error
80
80
  , evidence_bytes
81
- ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
81
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
82
82
  ON CONFLICT(session_id, ordinal) DO UPDATE SET
83
83
  source_call_id = excluded.source_call_id,
84
84
  timestamp = excluded.timestamp,
85
+ end_timestamp = excluded.end_timestamp,
85
86
  tool = excluded.tool,
86
87
  input = excluded.input,
87
88
  outcome = excluded.outcome,
@@ -197,7 +198,7 @@ export function persistToolCalls(db, session, calls, sourceStamp, options = {})
197
198
  }
198
199
  for (const call of accepted) {
199
200
  const key = toolCallKey(session.id, call.ordinal);
200
- insertCall.run(key, session.id, call.ordinal, call.sourceCallId ?? null, call.timestamp, call.tool, call.input, call.outcome, call.exitCode ?? null, call.statusCode ?? null, call.errorCode ?? null, call.output ?? null, call.error ?? null, call.parseError ?? null, toolCallEvidenceBytes(call));
201
+ insertCall.run(key, session.id, call.ordinal, call.sourceCallId ?? null, call.timestamp, call.endTimestamp ?? null, call.tool, call.input, call.outcome, call.exitCode ?? null, call.statusCode ?? null, call.errorCode ?? null, call.output ?? null, call.error ?? null, call.parseError ?? null, toolCallEvidenceBytes(call));
201
202
  // The upsert above preserves the rowid of a call it updated, so this is
202
203
  // the same rowid the existing text row (if any) was written under.
203
204
  const { rowid } = callRowid.get(key);
@@ -215,3 +216,55 @@ export function persistToolCalls(db, session, calls, sourceStamp, options = {})
215
216
  });
216
217
  txn();
217
218
  }
219
+ /**
220
+ * Decide whether a changed full-file-harness session can RESUME its tool index
221
+ * from the last scan instead of re-deriving every call (PHNX-3411).
222
+ *
223
+ * The warm-tick indexer re-derives tool calls for a changed session on every
224
+ * tick, and for an ACTIVE large session that means re-sanitizing tens of
225
+ * thousands of calls each time — the synchronous work that wedged the daemon
226
+ * event loop. Streaming harnesses (claude/codex) already resume from a byte
227
+ * offset; every other harness re-parses the whole file, so this brings them the
228
+ * same benefit at the EVENT level: the ledger stores how many events were folded
229
+ * (`parsed_offset`) plus the collector snapshot (`parser_state`), and a later
230
+ * scan folds only the newly appended events.
231
+ *
232
+ * Returns the prior resume point when it is safe to append, or `null` (full
233
+ * re-scan) whenever the stored prefix may no longer describe the current file:
234
+ * a different extractor, no recorded resume point, a source the ledger row does
235
+ * not describe, a transcript that shrank below what was already parsed (a
236
+ * truncation/rewrite, not an append), more events already folded than the file
237
+ * now yields, or a snapshot that does not read back.
238
+ */
239
+ export function planEventToolResume(db, sessionId, sourcePath, stamp, currentEventCount) {
240
+ const row = db.prepare(`
241
+ SELECT file_path, file_size, extractor_version, parsed_offset, parser_state
242
+ FROM tool_scan_ledger WHERE session_id = ?
243
+ `).get(sessionId);
244
+ if (!row)
245
+ return null;
246
+ if (row.extractor_version !== TOOL_INDEX_VERSION)
247
+ return null;
248
+ if (row.parsed_offset === null || !Number.isSafeInteger(row.parsed_offset) || row.parsed_offset < 0)
249
+ return null;
250
+ if (row.file_path !== canonicalToolLedgerPath(sourcePath))
251
+ return null;
252
+ if (stamp.fileSize < row.file_size)
253
+ return null;
254
+ // The file grew but reports fewer events than were already folded — the prefix
255
+ // was rewritten, not appended to. Re-scan from scratch.
256
+ if (row.parsed_offset > currentEventCount)
257
+ return null;
258
+ if (row.parser_state === null)
259
+ return null;
260
+ let snapshot;
261
+ try {
262
+ snapshot = JSON.parse(row.parser_state);
263
+ }
264
+ catch {
265
+ return null;
266
+ }
267
+ if (snapshot?.v !== 1 || !Number.isSafeInteger(snapshot.nextOrdinal))
268
+ return null;
269
+ return { snapshot, eventCount: row.parsed_offset };
270
+ }
@@ -7,6 +7,11 @@ export type EnabledExtra = {
7
7
  export declare function trustedSourceBases(): {
8
8
  dir: string;
9
9
  }[];
10
+ /** Every `plugins/<plugin>/skills` dir across trusted source bases, filtered to an optional plugin/agent scope. */
11
+ export declare function pluginSkillDirs(options?: {
12
+ agent?: AgentId;
13
+ plugins?: Set<string>;
14
+ }): string[];
10
15
  /** Find the trusted source for a command markdown by name. */
11
16
  export declare function resolveCommandSource(name: string): string | null;
12
17
  /** Find the trusted source directory for a skill by name. */
@@ -54,7 +54,8 @@ function pluginSupportsAgent(manifest, agent) {
54
54
  return true;
55
55
  return !manifest.agents || manifest.agents.length === 0 || manifest.agents.includes(agent);
56
56
  }
57
- function pluginSkillDirs(options = {}) {
57
+ /** Every `plugins/<plugin>/skills` dir across trusted source bases, filtered to an optional plugin/agent scope. */
58
+ export function pluginSkillDirs(options = {}) {
58
59
  const dirs = [];
59
60
  for (const base of trustedSourceBases()) {
60
61
  const pluginsDir = path.join(base.dir, 'plugins');
@@ -26,7 +26,8 @@ export declare const KNOWN_TOP_LEVEL_COMMANDS: ReadonlySet<string>;
26
26
  * (linear-cli) (RUSH-2932). `alias` moved under `agents setup alias` (RUSH-2965).
27
27
  * `inbox` was a pure alias of `agents feed` (RUSH-2984). `unshare` nested under
28
28
  * `agents artifacts unshare` (RUSH-2989). `audit` nested under `agents events audit`.
29
- * `trends` nested under `agents insights mix` / `insights trends`. `serve` (the
29
+ * `trends` was removed with the insights recipe collapse — the one counter
30
+ * surface is `agents insights mix` (PHNX-3391). `serve` (the
30
31
  * read-only local web companion + `--control` anchor) was removed with the
31
32
  * unshipped iOS Fleet Cockpit it existed for (RUSH-3001). `apply` nested under
32
33
  * `agents fleet apply` / `agents devices apply`. `beta` nested under
@@ -34,7 +35,12 @@ export declare const KNOWN_TOP_LEVEL_COMMANDS: ReadonlySet<string>;
34
35
  * removed; `agents auth` returned against Phoenix ID with `auth space` as the
35
36
  * team surface (RUSH-2581). `usage` was removed as a duplicate surface of
36
37
  * `agents view`, which renders per-account usage with account, version, and
37
- * auth state beside it (RUSH-3079).
38
+ * auth state beside it (RUSH-3079). `perf` nested under `agents insights perf`
39
+ * — performance metrics are an insight, not a top-level noun (PHNX-3391).
40
+ * `list` was removed — it was a long-deprecated full duplicate of `agents view`
41
+ * (it already printed "agents list is now agents view"); `agents view` is the
42
+ * one version-listing surface (PHNX-3391). The `agents trash restore` subcommand
43
+ * was likewise removed as an exact duplicate of top-level `agents restore`.
38
44
  */
39
45
  export declare const RETIRED_TOP_LEVEL_COMMANDS: ReadonlySet<string>;
40
46
  export declare function isKnownTopLevelCommand(name: string): boolean;
@@ -1,11 +1,11 @@
1
1
  const LOADED_COMMAND_NAMES = [
2
2
  'accounts', 'auth', 'view', 'inspect', 'feedback', 'commands', 'hooks', 'skills', 'rules', 'memory',
3
- 'permissions', 'mcp', 'clis', 'subagents', 'plugins', 'workflows', 'add', 'use', 'list',
3
+ 'permissions', 'mcp', 'clis', 'subagents', 'plugins', 'workflows', 'add', 'use',
4
4
  'remove', 'rm', 'purge', 'update', 'prune', 'import', 'registry', 'search', 'install',
5
5
  'routines', 'monitors', 'projects', 'run', 'open', 'reconnect', 'fork', 'config',
6
6
  'models', 'modes', 'trash', 'restore', 'doctor',
7
7
  'route', 'harness', 'harnesses', 'secrets', 'menubar', 'sync',
8
- 'refresh-rules', 'factory', 'insights', 'trace', 'perf',
8
+ 'refresh-rules', 'factory', 'insights', 'trace',
9
9
  'pty', 'tmux', 'watchdog', 'browser', 'computer', 'logs', 'events',
10
10
  'ssh', 'devices', 'fleet', 'repos', 'repo', 'setup', 'uninstall', 'upgrade', 'sessions',
11
11
  'teams', 'cloud', 'message', 'send', 'notify', 'feed',
@@ -46,7 +46,8 @@ export const KNOWN_TOP_LEVEL_COMMANDS = new Set([
46
46
  * (linear-cli) (RUSH-2932). `alias` moved under `agents setup alias` (RUSH-2965).
47
47
  * `inbox` was a pure alias of `agents feed` (RUSH-2984). `unshare` nested under
48
48
  * `agents artifacts unshare` (RUSH-2989). `audit` nested under `agents events audit`.
49
- * `trends` nested under `agents insights mix` / `insights trends`. `serve` (the
49
+ * `trends` was removed with the insights recipe collapse — the one counter
50
+ * surface is `agents insights mix` (PHNX-3391). `serve` (the
50
51
  * read-only local web companion + `--control` anchor) was removed with the
51
52
  * unshipped iOS Fleet Cockpit it existed for (RUSH-3001). `apply` nested under
52
53
  * `agents fleet apply` / `agents devices apply`. `beta` nested under
@@ -54,7 +55,12 @@ export const KNOWN_TOP_LEVEL_COMMANDS = new Set([
54
55
  * removed; `agents auth` returned against Phoenix ID with `auth space` as the
55
56
  * team surface (RUSH-2581). `usage` was removed as a duplicate surface of
56
57
  * `agents view`, which renders per-account usage with account, version, and
57
- * auth state beside it (RUSH-3079).
58
+ * auth state beside it (RUSH-3079). `perf` nested under `agents insights perf`
59
+ * — performance metrics are an insight, not a top-level noun (PHNX-3391).
60
+ * `list` was removed — it was a long-deprecated full duplicate of `agents view`
61
+ * (it already printed "agents list is now agents view"); `agents view` is the
62
+ * one version-listing surface (PHNX-3391). The `agents trash restore` subcommand
63
+ * was likewise removed as an exact duplicate of top-level `agents restore`.
58
64
  */
59
65
  export const RETIRED_TOP_LEVEL_COMMANDS = new Set([
60
66
  'webhook',
@@ -85,6 +91,8 @@ export const RETIRED_TOP_LEVEL_COMMANDS = new Set([
85
91
  'trends',
86
92
  'apply',
87
93
  'beta',
94
+ 'perf',
95
+ 'list',
88
96
  ]);
89
97
  export function isKnownTopLevelCommand(name) {
90
98
  return KNOWN_TOP_LEVEL_COMMANDS.has(name);
@@ -93,6 +93,28 @@ export interface UnifiedSyncStatus {
93
93
  agentsNeedingSync: number;
94
94
  };
95
95
  }
96
+ /**
97
+ * Residual drift for a single (agent, version) after a reconcile — the drifted
98
+ * and missing rows that a sync claimed to fix but did not. `orphan` rows are
99
+ * deliberately excluded: sync never removes them (that is `agents prune`'s job),
100
+ * so they are not "unfinished sync". Empty `drifted`+`missing` ⇒ converged.
101
+ */
102
+ export interface ResidualDrift {
103
+ agent: AgentId;
104
+ version: string;
105
+ rows: ResourceStatusRow[];
106
+ }
107
+ /**
108
+ * Re-check that a version's home now matches its resolved sources after a
109
+ * reconcile. This is the post-write verification the `agents sync` success line
110
+ * depends on (PHNX-3186): the exit line must not read "reconciled" while drift
111
+ * it was asked to fix stays put. Resolves against non-project layers only
112
+ * (`excludeProject: true`), mirroring what the sync writer targets. Returns null
113
+ * when the version converged (no drifted/missing), else the residual rows.
114
+ */
115
+ export declare function verifyVersionConverged(agent: AgentId, version: string, cwd?: string): ResidualDrift | null;
116
+ /** One-line-per-resource description of residual drift, for the sync exit line. */
117
+ export declare function formatResidualDrift(residual: ResidualDrift[]): string[];
96
118
  export interface SyncStatusOptions {
97
119
  cwd?: string;
98
120
  /** Restrict to specific agent ids; undefined = every supported agent. */
@@ -27,6 +27,33 @@ import { getSystemAgentsDir, getUserAgentsDir } from './state.js';
27
27
  import * as fs from 'fs';
28
28
  import { isGitRepo, readOriginUrl } from './git.js';
29
29
  import { detectConfigDrift } from './config-drift.js';
30
+ /**
31
+ * Re-check that a version's home now matches its resolved sources after a
32
+ * reconcile. This is the post-write verification the `agents sync` success line
33
+ * depends on (PHNX-3186): the exit line must not read "reconciled" while drift
34
+ * it was asked to fix stays put. Resolves against non-project layers only
35
+ * (`excludeProject: true`), mirroring what the sync writer targets. Returns null
36
+ * when the version converged (no drifted/missing), else the residual rows.
37
+ */
38
+ export function verifyVersionConverged(agent, version, cwd = process.cwd()) {
39
+ const report = diffVersionResources(agent, version, { cwd, excludeProject: true });
40
+ const rows = rowsFromReport(agent, version, report)
41
+ .filter((r) => r.status === 'drifted' || r.status === 'missing');
42
+ if (rows.length === 0)
43
+ return null;
44
+ return { agent, version, rows };
45
+ }
46
+ /** One-line-per-resource description of residual drift, for the sync exit line. */
47
+ export function formatResidualDrift(residual) {
48
+ const lines = [];
49
+ for (const r of residual) {
50
+ for (const row of r.rows) {
51
+ const detail = row.detail ? ` (${row.detail})` : '';
52
+ lines.push(`${r.agent}@${r.version}: ${row.status} ${row.kind} '${row.name}'${detail}`);
53
+ }
54
+ }
55
+ return lines;
56
+ }
30
57
  const STATUS_MAP = {
31
58
  ok: 'synced',
32
59
  diff: 'drifted',
@@ -64,6 +64,15 @@ export interface UmbrellaResult {
64
64
  * Empty when nothing was declined — never conflated with "nothing to do".
65
65
  */
66
66
  declined: string[];
67
+ /**
68
+ * The (agent, version) pairs the reconcile stage actually wrote into — the set
69
+ * the caller re-verifies for residual drift so the `✓ sync: reconciled` line is
70
+ * never printed while drift it was asked to fix stays put (PHNX-3186).
71
+ */
72
+ reconciledVersions: Array<{
73
+ agent: string;
74
+ version: string;
75
+ }>;
67
76
  }
68
77
  export interface RunUmbrellaArgs {
69
78
  flags: UmbrellaFlags;
@@ -15,7 +15,7 @@
15
15
  * arrives with `agents secrets vault unlock`, #366/#367)
16
16
  * reconcile-> refresh({ skipPrompts }) — re-materialize resources into homes
17
17
  */
18
- import { pullRepo } from './git.js';
18
+ import { pullRepo, adoptUserRepoIfNeeded } from './git.js';
19
19
  import { getUserAgentsDir, getEnabledExtraRepos } from './state.js';
20
20
  import { listRemoteBundles, pullBundle } from './secrets/sync.js';
21
21
  import { SYNC_PASSPHRASE_ENV } from './secrets/sync-passphrase.js';
@@ -55,7 +55,7 @@ export function planUmbrellaStages(f) {
55
55
  export async function runUmbrellaSync(args) {
56
56
  const { flags, log, yes, passphrase, quiet = false } = args;
57
57
  const plan = planUmbrellaStages(flags);
58
- const result = { plan, reconciled: false, declined: [] };
58
+ const result = { plan, reconciled: false, declined: [], reconciledVersions: [] };
59
59
  if (plan.fetchRepos) {
60
60
  const dirs = [
61
61
  { alias: 'user', dir: getUserAgentsDir() },
@@ -64,6 +64,24 @@ export async function runUmbrellaSync(args) {
64
64
  let pulled = 0;
65
65
  const errors = [];
66
66
  for (const { alias, dir } of dirs) {
67
+ // A box where `~/.agents` is present but not a git repo (or lost its
68
+ // `.git`) is a partial install: `pullRepo` there silently fails while the
69
+ // umbrella still reported `✓ reconciled`, so fleet dotfiles/resources never
70
+ // propagated and nothing said so (PHNX-3239, m0). Adopt it in place first —
71
+ // the same self-heal `agents sync user` runs (PHNX-3301) — so the pull has a
72
+ // real repo to fast-forward. A box that cannot be adopted (no recorded
73
+ // remote) fails LOUD into `errors` with the reason, never a silent no-op.
74
+ if (alias === 'user') {
75
+ const adopted = await adoptUserRepoIfNeeded(dir);
76
+ if (adopted && !adopted.success) {
77
+ const hint = adopted.needsUrl ? ' — git-back it: agents repo pull user <git-url>' : '';
78
+ errors.push(`${alias}: ${adopted.error}${hint}`);
79
+ continue;
80
+ }
81
+ if (adopted?.success) {
82
+ log(`repos: ${alias} adopted in place → ${adopted.commit} (${adopted.materialized} file(s) materialized)`);
83
+ }
84
+ }
67
85
  const r = await pullRepo(dir);
68
86
  if (r.success) {
69
87
  pulled++;
@@ -116,6 +134,7 @@ export async function runUmbrellaSync(args) {
116
134
  const refreshed = await refresh({ skipPrompts: yes, quiet });
117
135
  result.reconciled = true;
118
136
  result.declined = refreshed.declined;
137
+ result.reconciledVersions = refreshed.reconciled;
119
138
  // Keep already-registered devices' reachability current, and surface newly
120
139
  // appeared tailnet nodes as "pending" for the menu-bar Register/Ignore gate
121
140
  // rather than silently adding them (refresh mode). Soft: a machine without
@@ -12,14 +12,21 @@
12
12
  * is enough to reconstruct per-session call order and inter-call gaps without a
13
13
  * full `SessionTrajectory` — that is what makes this incremental at scale.
14
14
  *
15
- * Scope note: `FailureSignature` does not yet carry a `phenotype`
16
- * (false-termination / out-of-order / …, `phenotype.ts`) — classifying that
17
- * needs the full derived trajectory (turns, ordered steps, gaps), which is
18
- * only ever materialized per-session during upload, not cached the way
19
- * `InsightFacets` is. Folding it in is a real, scoped follow-up (see
20
- * `cli/AGENTS.md`), not a silent omission.
15
+ * Failure phenotype (false-termination / out-of-order / premature-completion /
16
+ * failure-to-act, `phenotype.ts`) is folded in as a fourth grouping dimension
17
+ * (PHNX-3327). Classifying it needs the full derived trajectory, so it is NOT
18
+ * computed here — the caller passes a per-session `phenotypes` map that
19
+ * `buildIndexShard` fills from the persisted, mtime+size-keyed
20
+ * `session_phenotypes` cache for the WHOLE corpus. Keying the group on the
21
+ * cached-per-session phenotype (never on this run's incremental batch) is what
22
+ * keeps two identically-signatured sessions in one cluster regardless of when
23
+ * each was synced. The `signature` OUTPUT is unchanged (`{ tool, cause, key }`);
24
+ * phenotype is an added dimension carried alongside it, so callers that never
25
+ * pass a map (the unit tests, a pre-phenotype caller) see the exact prior
26
+ * grouping.
21
27
  */
22
28
  import { type TraceFailureCause } from './classify.js';
29
+ import type { FailurePhenotype } from './phenotype.js';
23
30
  import { type LatencyInsight } from './segments.js';
24
31
  import { type SyncRow, type ToolCallRow, type TracesIndexShard } from './sync.js';
25
32
  export interface FailureSignature {
@@ -29,10 +36,17 @@ export interface FailureSignature {
29
36
  key: string;
30
37
  }
31
38
  export interface FailurePattern {
32
- /** Stable hash of the signature — deep-linkable, unaffected by row order. */
39
+ /** Stable hash of the signature (incl. phenotype) — deep-linkable, unaffected by row order. */
33
40
  id: string;
34
41
  label: string;
35
42
  signature: FailureSignature;
43
+ /**
44
+ * Dominant failure phenotype of the sessions in this cluster, or `null` when
45
+ * none was classifiable. A fourth grouping dimension (PHNX-3327): two failures
46
+ * with the same `(tool, cause, key)` but different phenotypes are distinct
47
+ * patterns.
48
+ */
49
+ phenotype: FailurePhenotype | null;
36
50
  /** Distinct sessions this pattern occurred in. */
37
51
  sessions: number;
38
52
  /** Total failing calls matching this signature. */
@@ -57,11 +71,30 @@ export declare function normalizeErrorKey(desc: string, raw: string | null): str
57
71
  * Cluster failed tool calls into ranked patterns and estimate the wasted time
58
72
  * behind each, plus device-wide time-to-first-tool latency.
59
73
  *
60
- * wastedMs attribution: the gap between a failed call and the NEXT call in the
61
- * same session counts as wasted when either (a) the next call repeats the same
62
- * signature (a retry loop) or (b) the gap itself is a stall (≥60s) — an idle
63
- * gap unrelated to a nearby failure is never counted. This is an estimate, not
64
- * ground truth (a stall could be legitimate user think-time); it is not
65
- * inflated by folding in ordinary processing time between unrelated calls.
74
+ * wastedMs attribution has two parts that sum:
75
+ *
76
+ * (1) The failed call's OWN blocking duration — `end_timestamp - timestamp`
77
+ * (PHNX-3437). A call that hung for minutes and then failed wasted that whole
78
+ * time even if it was the last call in its session or was followed quickly by an
79
+ * unrelated call — the case the gap heuristic alone booked as ~0. This is what
80
+ * makes a fail-fast fix measurable: a channel that stops hanging 5.5min on stdin
81
+ * and instead fails in <1s (PHNX-3407) drops from ~5.5min of attributed waste to
82
+ * ~0. Bounded by MAX_GAP_ATTRIBUTION_MS so a corrupt end timestamp can't dominate.
83
+ *
84
+ * (2) The idle gap AFTER the call, before the NEXT call in the same session,
85
+ * counted when either (a) the next call repeats the same signature (a retry
86
+ * loop) or (b) the gap is a stall (≥60s) before an unrelated next call — each
87
+ * bounded by MAX_GAP_ATTRIBUTION_MS so one failure can't absorb hours of
88
+ * human-away idle in an async channel session, whether as a lone stall or a
89
+ * same-signature re-ask hours later; a genuine active retry loop is many short
90
+ * gaps that each clear the cap and still sum large. When the end timestamp is
91
+ * known, this gap is measured from the call's END, so the blocking time counted
92
+ * in (1) is never double-counted; for a NULL end (rows an older extractor stored,
93
+ * or a call still pending at scan end) it falls back to the original
94
+ * gap-from-START heuristic unchanged — no crash, no NaN, no regression.
95
+ *
96
+ * An idle gap unrelated to a nearby failure is never counted. This is an
97
+ * estimate, not ground truth; it is not inflated by folding in ordinary
98
+ * processing time between unrelated calls.
66
99
  */
67
- export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null): ComputedInsights;
100
+ export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null, phenotypes?: ReadonlyMap<string, FailurePhenotype | null>): ComputedInsights;
@@ -12,12 +12,18 @@
12
12
  * is enough to reconstruct per-session call order and inter-call gaps without a
13
13
  * full `SessionTrajectory` — that is what makes this incremental at scale.
14
14
  *
15
- * Scope note: `FailureSignature` does not yet carry a `phenotype`
16
- * (false-termination / out-of-order / …, `phenotype.ts`) — classifying that
17
- * needs the full derived trajectory (turns, ordered steps, gaps), which is
18
- * only ever materialized per-session during upload, not cached the way
19
- * `InsightFacets` is. Folding it in is a real, scoped follow-up (see
20
- * `cli/AGENTS.md`), not a silent omission.
15
+ * Failure phenotype (false-termination / out-of-order / premature-completion /
16
+ * failure-to-act, `phenotype.ts`) is folded in as a fourth grouping dimension
17
+ * (PHNX-3327). Classifying it needs the full derived trajectory, so it is NOT
18
+ * computed here — the caller passes a per-session `phenotypes` map that
19
+ * `buildIndexShard` fills from the persisted, mtime+size-keyed
20
+ * `session_phenotypes` cache for the WHOLE corpus. Keying the group on the
21
+ * cached-per-session phenotype (never on this run's incremental batch) is what
22
+ * keeps two identically-signatured sessions in one cluster regardless of when
23
+ * each was synced. The `signature` OUTPUT is unchanged (`{ tool, cause, key }`);
24
+ * phenotype is an added dimension carried alongside it, so callers that never
25
+ * pass a map (the unit tests, a pre-phenotype caller) see the exact prior
26
+ * grouping.
21
27
  */
22
28
  import { classifyCause } from './classify.js';
23
29
  import { computeLatency } from './segments.js';
@@ -30,6 +36,24 @@ const TOP_K_PATTERNS = 25;
30
36
  const MAX_EXAMPLE_SESSIONS = 5;
31
37
  /** A gap this long right after a failure reads as an idle stall, not think-time (matches sync.ts's own "stalled Xm" threshold). */
32
38
  const STALL_MS = 60_000;
39
+ /**
40
+ * Upper bound on how much of a SINGLE inter-call gap is attributable to a
41
+ * failure. Beyond a recovery window a gap is not failure-loop waste — it is a
42
+ * human away from a chat thread, an abandoned session, or an outage. This bounds
43
+ * BOTH branches, and that matters: `nextIsSameFailure` only compares
44
+ * `(tool, cause, normalized-key)`, with no temporal check, so a deterministic
45
+ * failure that recurs identically hours apart (a permanently-denied capability,
46
+ * a missing credential — e.g. a Slack user re-asking "where am I" at 2pm and
47
+ * 6pm) would otherwise look like an "active retry loop" and absorb the whole
48
+ * multi-hour gap — the very artifact this fix targets. A genuine active loop has
49
+ * MANY short gaps that each stay under this cap and still sum to a large total,
50
+ * so bounding a single gap doesn't hide it. Real single-tool stalls (a hung
51
+ * typecheck, a slow build) are minutes and stay fully counted.
52
+ *
53
+ * The same cap bounds the call's OWN blocking duration (PHNX-3437) so a corrupt
54
+ * or backwards end timestamp can't book a single call as hours of waste either.
55
+ */
56
+ const MAX_GAP_ATTRIBUTION_MS = 30 * 60_000;
33
57
  // ---------------------------------------------------------------------------
34
58
  // Signature normalization — fold volatile per-instance text together
35
59
  // ---------------------------------------------------------------------------
@@ -48,8 +72,8 @@ export function normalizeErrorKey(desc, raw) {
48
72
  }
49
73
  return text.replace(/\s+/g, ' ').trim().slice(0, 160);
50
74
  }
51
- function hashSignature(tool, cause, key) {
52
- const input = `${tool} ${cause} ${key}`;
75
+ function hashSignature(tool, cause, key, phenotype) {
76
+ const input = `${tool} ${cause} ${key} ${phenotype ?? ''}`;
53
77
  let hash = 5381;
54
78
  for (let i = 0; i < input.length; i++) {
55
79
  hash = ((hash << 5) + hash + input.charCodeAt(i)) >>> 0;
@@ -82,14 +106,33 @@ function labelFor(tool, cause, key) {
82
106
  * Cluster failed tool calls into ranked patterns and estimate the wasted time
83
107
  * behind each, plus device-wide time-to-first-tool latency.
84
108
  *
85
- * wastedMs attribution: the gap between a failed call and the NEXT call in the
86
- * same session counts as wasted when either (a) the next call repeats the same
87
- * signature (a retry loop) or (b) the gap itself is a stall (≥60s) — an idle
88
- * gap unrelated to a nearby failure is never counted. This is an estimate, not
89
- * ground truth (a stall could be legitimate user think-time); it is not
90
- * inflated by folding in ordinary processing time between unrelated calls.
109
+ * wastedMs attribution has two parts that sum:
110
+ *
111
+ * (1) The failed call's OWN blocking duration — `end_timestamp - timestamp`
112
+ * (PHNX-3437). A call that hung for minutes and then failed wasted that whole
113
+ * time even if it was the last call in its session or was followed quickly by an
114
+ * unrelated call — the case the gap heuristic alone booked as ~0. This is what
115
+ * makes a fail-fast fix measurable: a channel that stops hanging 5.5min on stdin
116
+ * and instead fails in <1s (PHNX-3407) drops from ~5.5min of attributed waste to
117
+ * ~0. Bounded by MAX_GAP_ATTRIBUTION_MS so a corrupt end timestamp can't dominate.
118
+ *
119
+ * (2) The idle gap AFTER the call, before the NEXT call in the same session,
120
+ * counted when either (a) the next call repeats the same signature (a retry
121
+ * loop) or (b) the gap is a stall (≥60s) before an unrelated next call — each
122
+ * bounded by MAX_GAP_ATTRIBUTION_MS so one failure can't absorb hours of
123
+ * human-away idle in an async channel session, whether as a lone stall or a
124
+ * same-signature re-ask hours later; a genuine active retry loop is many short
125
+ * gaps that each clear the cap and still sum large. When the end timestamp is
126
+ * known, this gap is measured from the call's END, so the blocking time counted
127
+ * in (1) is never double-counted; for a NULL end (rows an older extractor stored,
128
+ * or a call still pending at scan end) it falls back to the original
129
+ * gap-from-START heuristic unchanged — no crash, no NaN, no regression.
130
+ *
131
+ * An idle gap unrelated to a nearby failure is never counted. This is an
132
+ * estimate, not ground truth; it is not inflated by folding in ordinary
133
+ * processing time between unrelated calls.
91
134
  */
92
- export function computeInsights(rows, calls, prevShard) {
135
+ export function computeInsights(rows, calls, prevShard, phenotypes) {
93
136
  const bySession = new Map();
94
137
  for (const call of calls) {
95
138
  const list = bySession.get(call.session_id);
@@ -100,6 +143,11 @@ export function computeInsights(rows, calls, prevShard) {
100
143
  }
101
144
  const groups = new Map();
102
145
  for (const [sessionId, sessionCalls] of bySession) {
146
+ // Phenotype is a per-session property (one classification per session), so
147
+ // every failing call in this session shares it. A caller that passes no map
148
+ // (unit tests, a pre-phenotype caller) collapses the dimension to `null`,
149
+ // yielding the exact prior grouping.
150
+ const phenotype = phenotypes?.get(sessionId) ?? null;
103
151
  const ordered = [...sessionCalls].sort((a, b) => a.ordinal - b.ordinal);
104
152
  for (let i = 0; i < ordered.length; i++) {
105
153
  const call = ordered[i];
@@ -107,10 +155,10 @@ export function computeInsights(rows, calls, prevShard) {
107
155
  continue;
108
156
  const cause = classifyCause(call);
109
157
  const key = normalizeErrorKey(failureDescription(call, cause), call.error);
110
- const groupKey = `${call.tool} ${cause} ${key}`;
158
+ const groupKey = `${call.tool} ${cause} ${key} ${phenotype ?? ''}`;
111
159
  let group = groups.get(groupKey);
112
160
  if (!group) {
113
- group = { tool: call.tool, cause, key, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
161
+ group = { tool: call.tool, cause, key, phenotype, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
114
162
  groups.set(groupKey, group);
115
163
  }
116
164
  group.occurrences++;
@@ -118,24 +166,46 @@ export function computeInsights(rows, calls, prevShard) {
118
166
  if (group.examples.length < MAX_EXAMPLE_SESSIONS && !group.examples.includes(sessionId)) {
119
167
  group.examples.push(sessionId);
120
168
  }
169
+ // (1) The call's OWN blocking duration (end minus start) is the primary
170
+ // signal — see the computeInsights docblock. Attributed whenever the end
171
+ // timestamp is present, independent of whether a next call follows.
172
+ const startMs = Date.parse(call.timestamp);
173
+ const endMs = call.end_timestamp ? Date.parse(call.end_timestamp) : NaN;
174
+ const hasEnd = Number.isFinite(endMs) && Number.isFinite(startMs);
175
+ if (hasEnd) {
176
+ const ownMs = endMs - startMs;
177
+ if (ownMs > 0)
178
+ group.wastedMs += Math.min(ownMs, MAX_GAP_ATTRIBUTION_MS);
179
+ }
180
+ // (2) The idle gap after the call, before the next one. Measured from the
181
+ // call's END when known (so the blocking time in (1) isn't double-counted),
182
+ // else from its START — the original heuristic, unchanged for NULL ends.
121
183
  const next = ordered[i + 1];
122
184
  if (!next)
123
185
  continue;
124
- const gapMs = Date.parse(next.timestamp) - Date.parse(call.timestamp);
186
+ const gapFromMs = hasEnd ? endMs : startMs;
187
+ const gapMs = Date.parse(next.timestamp) - gapFromMs;
125
188
  if (!Number.isFinite(gapMs) || gapMs <= 0)
126
189
  continue;
127
190
  const nextIsSameFailure = next.outcome === 'error' &&
128
191
  next.tool === call.tool &&
129
192
  classifyCause(next) === cause &&
130
193
  normalizeErrorKey(failureDescription(next, cause), next.error) === key;
131
- if (nextIsSameFailure || gapMs >= STALL_MS) {
132
- group.wastedMs += gapMs;
194
+ if (nextIsSameFailure) {
195
+ // Retry loop: a real active loop is many short gaps, each under the cap,
196
+ // summing to a large total. Bound a single gap so one huge same-signature
197
+ // gap (a re-ask hours later, not active retrying) can't absorb it all.
198
+ group.wastedMs += Math.min(gapMs, MAX_GAP_ATTRIBUTION_MS);
199
+ }
200
+ else if (gapMs >= STALL_MS) {
201
+ // Lone stall before an unrelated next call: same bounded recovery window.
202
+ group.wastedMs += Math.min(gapMs, MAX_GAP_ATTRIBUTION_MS);
133
203
  }
134
204
  }
135
205
  }
136
206
  const prevById = new Map((prevShard?.failurePatterns ?? []).map((p) => [p.id, p]));
137
207
  const allPatterns = [...groups.values()].map((group) => {
138
- const id = hashSignature(group.tool, group.cause, group.key);
208
+ const id = hashSignature(group.tool, group.cause, group.key, group.phenotype);
139
209
  const prev = prevById.get(id);
140
210
  const drift = !prev
141
211
  ? 'up'
@@ -148,6 +218,7 @@ export function computeInsights(rows, calls, prevShard) {
148
218
  id,
149
219
  label: labelFor(group.tool, group.cause, group.key),
150
220
  signature: { tool: group.tool, cause: group.cause, key: group.key },
221
+ phenotype: group.phenotype,
151
222
  sessions: group.sessions.size,
152
223
  occurrences: group.occurrences,
153
224
  wastedMs: group.wastedMs,