@phnx-labs/agents-cli 1.22.51 → 1.22.53

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/CHANGELOG.md +238 -0
  2. package/README.md +1 -1
  3. package/dist/commands/accounts.js +1 -1
  4. package/dist/commands/attach.js +7 -0
  5. package/dist/commands/browser.js +118 -56
  6. package/dist/commands/daemon.d.ts +2 -0
  7. package/dist/commands/daemon.js +8 -4
  8. package/dist/commands/detach.js +1 -1
  9. package/dist/commands/exec.js +16 -9
  10. package/dist/commands/fleet-capture.js +7 -0
  11. package/dist/commands/focus.d.ts +1 -10
  12. package/dist/commands/focus.js +16 -79
  13. package/dist/commands/go.d.ts +26 -0
  14. package/dist/commands/go.js +65 -6
  15. package/dist/commands/monitors.js +1 -1
  16. package/dist/commands/repo.js +31 -3
  17. package/dist/commands/sessions-inject.js +8 -3
  18. package/dist/commands/sessions-picker.js +2 -1
  19. package/dist/commands/sessions-resume.d.ts +1 -0
  20. package/dist/commands/sessions-resume.js +13 -2
  21. package/dist/commands/sessions-stop.js +1 -1
  22. package/dist/commands/sessions.d.ts +23 -13
  23. package/dist/commands/sessions.js +69 -39
  24. package/dist/commands/setup-browser.d.ts +5 -2
  25. package/dist/commands/setup-browser.js +14 -29
  26. package/dist/commands/setup-preferences.d.ts +22 -3
  27. package/dist/commands/setup-preferences.js +25 -8
  28. package/dist/commands/share.js +12 -8
  29. package/dist/commands/ssh.js +35 -12
  30. package/dist/commands/status.js +5 -0
  31. package/dist/commands/sync.js +102 -2
  32. package/dist/commands/tmux.d.ts +8 -1
  33. package/dist/commands/tmux.js +167 -17
  34. package/dist/lib/account-registry.d.ts +15 -5
  35. package/dist/lib/account-registry.js +150 -50
  36. package/dist/lib/answer-router.js +2 -1
  37. package/dist/lib/browser/ipc.d.ts +44 -0
  38. package/dist/lib/browser/ipc.js +120 -8
  39. package/dist/lib/browser/profiles.d.ts +57 -17
  40. package/dist/lib/browser/profiles.js +77 -53
  41. package/dist/lib/browser/registry.d.ts +44 -14
  42. package/dist/lib/browser/registry.js +141 -45
  43. package/dist/lib/browser/runtime-state.d.ts +4 -2
  44. package/dist/lib/browser/runtime-state.js +4 -2
  45. package/dist/lib/browser/service.js +4 -3
  46. package/dist/lib/channels/owner-forward.d.ts +88 -0
  47. package/dist/lib/channels/owner-forward.js +116 -0
  48. package/dist/lib/channels/owner-sink.js +7 -0
  49. package/dist/lib/daemon/runner.js +10 -2
  50. package/dist/lib/device-config.js +3 -2
  51. package/dist/lib/devices/config-migration.js +147 -1
  52. package/dist/lib/devices/device-docs.d.ts +35 -0
  53. package/dist/lib/devices/device-docs.js +163 -0
  54. package/dist/lib/devices/discovery-policy.d.ts +14 -2
  55. package/dist/lib/devices/discovery-policy.js +31 -21
  56. package/dist/lib/devices/registry.d.ts +11 -5
  57. package/dist/lib/devices/registry.js +46 -18
  58. package/dist/lib/exec.d.ts +66 -28
  59. package/dist/lib/exec.js +71 -26
  60. package/dist/lib/feed/feed.d.ts +10 -2
  61. package/dist/lib/feed/feed.js +12 -1
  62. package/dist/lib/feed-broadcast.js +15 -1
  63. package/dist/lib/git.d.ts +93 -0
  64. package/dist/lib/git.js +232 -0
  65. package/dist/lib/hosts/dispatch.d.ts +4 -3
  66. package/dist/lib/hosts/dispatch.js +12 -8
  67. package/dist/lib/hosts/providers/local.d.ts +9 -3
  68. package/dist/lib/hosts/providers/local.js +23 -12
  69. package/dist/lib/hosts/reconnect.d.ts +7 -4
  70. package/dist/lib/hosts/reconnect.js +29 -25
  71. package/dist/lib/hosts/registry.js +4 -1
  72. package/dist/lib/hosts/remote-os.js +3 -1
  73. package/dist/lib/monitors/remote.d.ts +18 -1
  74. package/dist/lib/monitors/remote.js +15 -2
  75. package/dist/lib/notify.d.ts +7 -0
  76. package/dist/lib/notify.js +15 -1
  77. package/dist/lib/session/active.d.ts +10 -1
  78. package/dist/lib/session/active.js +7 -1
  79. package/dist/lib/session/actor-sidecar.d.ts +7 -0
  80. package/dist/lib/session/actor-sidecar.js +2 -0
  81. package/dist/lib/session/db.d.ts +1 -1
  82. package/dist/lib/session/db.js +39 -3
  83. package/dist/lib/session/discover.js +7 -12
  84. package/dist/lib/session/live-metadata.js +1 -0
  85. package/dist/lib/session/local-tmux-attach.d.ts +69 -0
  86. package/dist/lib/session/local-tmux-attach.js +164 -0
  87. package/dist/lib/session/pid-registry.d.ts +7 -0
  88. package/dist/lib/session/prompt.d.ts +15 -0
  89. package/dist/lib/session/prompt.js +21 -0
  90. package/dist/lib/session/remote-active.d.ts +8 -0
  91. package/dist/lib/session/remote-active.js +1 -0
  92. package/dist/lib/session/types.d.ts +17 -0
  93. package/dist/lib/session/types.js +10 -0
  94. package/dist/lib/share/publish.d.ts +8 -11
  95. package/dist/lib/share/publish.js +16 -20
  96. package/dist/lib/share/worker-template.js +104 -12
  97. package/dist/lib/state.d.ts +8 -0
  98. package/dist/lib/state.js +143 -11
  99. package/dist/lib/sync-status.d.ts +17 -0
  100. package/dist/lib/sync-status.js +21 -2
  101. package/dist/lib/terminal/resolve.d.ts +7 -0
  102. package/dist/lib/terminal/resolve.js +41 -2
  103. package/dist/lib/tmux/index.d.ts +1 -1
  104. package/dist/lib/tmux/index.js +1 -1
  105. package/dist/lib/tmux/session.d.ts +10 -0
  106. package/dist/lib/tmux/session.js +29 -0
  107. package/dist/lib/traces/insights.d.ts +67 -0
  108. package/dist/lib/traces/insights.js +178 -0
  109. package/dist/lib/traces/phenotype.d.ts +67 -0
  110. package/dist/lib/traces/phenotype.js +437 -0
  111. package/dist/lib/traces/segments.d.ts +133 -0
  112. package/dist/lib/traces/segments.js +301 -0
  113. package/dist/lib/traces/sync.d.ts +33 -0
  114. package/dist/lib/traces/sync.js +11 -2
  115. package/dist/lib/types.d.ts +47 -1
  116. package/dist/lib/watchdog/runner.js +18 -4
  117. package/package.json +1 -1
@@ -23,8 +23,9 @@ import { ALL_AGENT_IDS } from './agents.js';
23
23
  import { diffVersionResources, } from './doctor-diff.js';
24
24
  import { listInstalledVersions, getGlobalDefault } from './installations/versions.js';
25
25
  import { loadManifest } from './staleness/index.js';
26
- import { getSystemAgentsDir } from './state.js';
27
- import { isGitRepo } from './git.js';
26
+ import { getSystemAgentsDir, getUserAgentsDir } from './state.js';
27
+ import * as fs from 'fs';
28
+ import { isGitRepo, readOriginUrl } from './git.js';
28
29
  const STATUS_MAP = {
29
30
  ok: 'synced',
30
31
  diff: 'drifted',
@@ -73,6 +74,22 @@ export async function getSystemRepoStatus() {
73
74
  return base;
74
75
  }
75
76
  }
77
+ /**
78
+ * Detect whether `~/.agents` (the user config layer) is git-backed. A partial
79
+ * install — runtime state present but no `.git` (or no `origin`) — is a distinct
80
+ * drift state that `agents repo sync user` heals by adopting in place (PHNX-3301),
81
+ * surfaced separately from per-version resource gaps. Purely local; no network.
82
+ */
83
+ export async function getUserRepoStatus() {
84
+ const dir = getUserAgentsDir();
85
+ if (!fs.existsSync(dir))
86
+ return { dir, notGitRepo: false };
87
+ if (!isGitRepo(dir))
88
+ return { dir, notGitRepo: true };
89
+ // A repo with no `origin` is just as partial for adopt's purposes — reuse the
90
+ // single origin-URL reader rather than a second remote check.
91
+ return { dir, notGitRepo: readOriginUrl(dir) === null };
92
+ }
76
93
  /**
77
94
  * Compute unified sync status across the fleet. Resolves against non-project
78
95
  * layers only (`excludeProject: true`) — the GLOBAL version home is never
@@ -107,6 +124,7 @@ export async function computeSyncStatus(options = {}) {
107
124
  }
108
125
  }
109
126
  const system = await getSystemRepoStatus();
127
+ const user = await getUserRepoStatus();
110
128
  const agentsNeedingSync = new Set();
111
129
  let drifted = 0, missing = 0, orphan = 0, versionsNeedingSync = 0, versionsNeverSynced = 0;
112
130
  for (const v of agents) {
@@ -122,6 +140,7 @@ export async function computeSyncStatus(options = {}) {
122
140
  }
123
141
  return {
124
142
  system,
143
+ user,
125
144
  agents,
126
145
  totals: {
127
146
  drifted,
@@ -34,6 +34,7 @@ export type InjectResolution = {
34
34
  } | {
35
35
  addressable: false;
36
36
  reason: string;
37
+ hint?: string;
37
38
  };
38
39
  export interface ResolveOptions {
39
40
  /**
@@ -49,6 +50,12 @@ export interface ResolveOptions {
49
50
  */
50
51
  ptyId?: string;
51
52
  }
53
+ /**
54
+ * Human-facing recovery hint for a session the resolver judged un-addressable.
55
+ * Tells the user both how to continue THIS session and how to make FUTURE runs
56
+ * addressable, so the failure is not silent and the fix is actionable.
57
+ */
58
+ export declare function addressabilityRecoveryHint(session: ActiveSession, fallbackId?: string): string;
52
59
  /**
53
60
  * Resolve a target from an already-fetched ActiveSession. Pure — no I/O — so the
54
61
  * precedence logic is unit-testable without the process table. This is where the
@@ -1,10 +1,40 @@
1
1
  import { getActiveSessions } from '../session/active.js';
2
+ import { machineId } from '../machine-id.js';
2
3
  /** The editor CLIs that speak the swarm-ext URI protocol, keyed by the host detectHost() reports. */
3
4
  const IDE_INJECT_VARIANTS = {
4
5
  codium: { cli: 'codium', scheme: 'vscodium' },
5
6
  cursor: { cli: 'cursor', scheme: 'cursor' },
6
7
  code: { cli: 'code', scheme: 'vscode' },
7
8
  };
9
+ /**
10
+ * Human-facing recovery hint for a session the resolver judged un-addressable.
11
+ * Tells the user both how to continue THIS session and how to make FUTURE runs
12
+ * addressable, so the failure is not silent and the fix is actionable.
13
+ */
14
+ export function addressabilityRecoveryHint(session, fallbackId) {
15
+ const sid = session.sessionId;
16
+ // Branch on the live sessionId (an IDE terminal that has not registered one
17
+ // yet is a distinct message), but render the resume command with the real id
18
+ // when the caller can supply it — e.g. `focus` has meta.id even though the
19
+ // live row's sessionId is falsy, so without this the hint printed a useless
20
+ // `agents sessions resume <id>` placeholder in exactly that case (PHNX-3070).
21
+ const resumeId = sid ?? fallbackId;
22
+ const shortId = resumeId ? resumeId.slice(0, 8) : '<id>';
23
+ const device = session.machine ?? machineId();
24
+ const resumeCmd = resumeId ? `agents sessions resume ${shortId}` : 'agents sessions resume <id>';
25
+ const tmuxCmd = `agents config set devices.${device}.tmux on`;
26
+ const interactive = session.context === 'terminal' || !!session.tty;
27
+ if (session.host === 'ghostty') {
28
+ return `Ghostty has no per-split addressing. ${interactive ? `Enable tmux wrapping with \`${tmuxCmd}\` and re-launch, or ` : ''}use \`${resumeCmd}\` to continue this session.`;
29
+ }
30
+ if (session.host && session.host in IDE_INJECT_VARIANTS && !sid) {
31
+ return `This IDE terminal has not registered a session id yet. Wait a moment and retry, or use \`${resumeCmd}\` to continue.`;
32
+ }
33
+ if (session.host) {
34
+ return `Host '${session.host}' has no addressable rail here. ${interactive ? `Enable tmux wrapping with \`${tmuxCmd}\` and re-launch, or ` : ''}use \`${resumeCmd}\` to continue this session.`;
35
+ }
36
+ return `This session has no addressable terminal rail (not tmux, iTerm, an IDE terminal, or a pty sidecar). ${interactive ? `Enable tmux wrapping with \`${tmuxCmd}\` and re-launch, or ` : ''}use \`${resumeCmd}\` to continue this session.`;
37
+ }
8
38
  /**
9
39
  * Resolve a target from an already-fetched ActiveSession. Pure — no I/O — so the
10
40
  * precedence logic is unit-testable without the process table. This is where the
@@ -35,7 +65,11 @@ export function resolveInjectTargetForSession(session, opts = {}) {
35
65
  const variant = session.host ? IDE_INJECT_VARIANTS[session.host] : undefined;
36
66
  if (variant) {
37
67
  if (!session.sessionId) {
38
- return { addressable: false, reason: `IDE terminal (${session.host}) has no session id to address` };
68
+ return {
69
+ addressable: false,
70
+ reason: `IDE terminal (${session.host}) has no session id to address`,
71
+ hint: addressabilityRecoveryHint(session),
72
+ };
39
73
  }
40
74
  return {
41
75
  addressable: true,
@@ -58,13 +92,18 @@ export function resolveInjectTargetForSession(session, opts = {}) {
58
92
  note: 'coarse Ghostty window path (opt-in): raises a window and types into the FOCUSED split — not split-precise',
59
93
  };
60
94
  }
61
- return { addressable: false, reason: 'un-addressable (ghostty, no tmux): no per-split addressing; watchdog skips' };
95
+ return {
96
+ addressable: false,
97
+ reason: 'un-addressable (ghostty, no tmux): no per-split addressing; watchdog skips',
98
+ hint: addressabilityRecoveryHint(session),
99
+ };
62
100
  }
63
101
  return {
64
102
  addressable: false,
65
103
  reason: session.host
66
104
  ? `no precise inject rail for host '${session.host}' (no tmux/iterm/IDE terminal detected)`
67
105
  : 'no inject rail: session is not inside tmux, iTerm, or an IDE terminal',
106
+ hint: addressabilityRecoveryHint(session),
68
107
  };
69
108
  }
70
109
  /**
@@ -5,4 +5,4 @@
5
5
  */
6
6
  export { findTmuxBinary, isTmuxInstalled, getTmuxVersion, isTmuxVersionSupported, MIN_TMUX_VERSION, assertTmuxAvailable, TmuxUnavailableError, TmuxCommandError, runTmux, attachTmux, } from './binary.js';
7
7
  export { getDefaultSocketPath, getSessionMetaPath, ensureTmuxDir, } from './paths.js';
8
- export { assertValidSessionName, slugifyName, hasSession, createSession, killSession, killAll, listSessions, splitPane, sendKeys, capturePane, readSessionMeta, TmuxSessionError, reconcileSessionHooks, ensureSessionHookRepaired, type SessionMeta, type CreateSessionOptions, type ListedSession, type SplitOptions, type SendOptions, type CaptureOptions, } from './session.js';
8
+ export { assertValidSessionName, slugifyName, hasSession, createSession, killSession, teardownIfAgentExited, killAll, listSessions, splitPane, sendKeys, capturePane, readSessionMeta, TmuxSessionError, reconcileSessionHooks, ensureSessionHookRepaired, type SessionMeta, type CreateSessionOptions, type ListedSession, type SplitOptions, type SendOptions, type CaptureOptions, } from './session.js';
@@ -5,4 +5,4 @@
5
5
  */
6
6
  export { findTmuxBinary, isTmuxInstalled, getTmuxVersion, isTmuxVersionSupported, MIN_TMUX_VERSION, assertTmuxAvailable, TmuxUnavailableError, TmuxCommandError, runTmux, attachTmux, } from './binary.js';
7
7
  export { getDefaultSocketPath, getSessionMetaPath, ensureTmuxDir, } from './paths.js';
8
- export { assertValidSessionName, slugifyName, hasSession, createSession, killSession, killAll, listSessions, splitPane, sendKeys, capturePane, readSessionMeta, TmuxSessionError, reconcileSessionHooks, ensureSessionHookRepaired, } from './session.js';
8
+ export { assertValidSessionName, slugifyName, hasSession, createSession, killSession, teardownIfAgentExited, killAll, listSessions, splitPane, sendKeys, capturePane, readSessionMeta, TmuxSessionError, reconcileSessionHooks, ensureSessionHookRepaired, } from './session.js';
@@ -109,6 +109,16 @@ export declare function createSession(opts: CreateSessionOptions): Promise<Sessi
109
109
  export declare function killSession(name: string, socket?: string, opts?: {
110
110
  reapOrphans?: boolean;
111
111
  }): Promise<boolean>;
112
+ /**
113
+ * After an attach client returns: destroy the session if every pane is dead
114
+ * (the agent exited); leave it if any pane is still alive (Ctrl-b d).
115
+ *
116
+ * `runInTmux` already does this via `resolveAfterAttach`. The attach verbs
117
+ * (`agents tmux attach`, `sessions focus`/`resume --attach-only`, `jumpTo`)
118
+ * used to `process.exit` the tmux client status and leave a `remain-on-exit`
119
+ * husk — the session "came back" in `tmux ls` after the user exited the agent.
120
+ */
121
+ export declare function teardownIfAgentExited(name: string, socket?: string): Promise<'killed' | 'kept' | 'absent'>;
112
122
  /**
113
123
  * Kill every session on the shared server AND the server itself, then prune
114
124
  * meta files. Wipes the socket so the next `new` starts from a clean slate.
@@ -315,6 +315,35 @@ export async function killSession(name, socket, opts = {}) {
315
315
  removeSessionMeta(name);
316
316
  return true;
317
317
  }
318
+ /**
319
+ * After an attach client returns: destroy the session if every pane is dead
320
+ * (the agent exited); leave it if any pane is still alive (Ctrl-b d).
321
+ *
322
+ * `runInTmux` already does this via `resolveAfterAttach`. The attach verbs
323
+ * (`agents tmux attach`, `sessions focus`/`resume --attach-only`, `jumpTo`)
324
+ * used to `process.exit` the tmux client status and leave a `remain-on-exit`
325
+ * husk — the session "came back" in `tmux ls` after the user exited the agent.
326
+ */
327
+ export async function teardownIfAgentExited(name, socket) {
328
+ assertValidSessionName(name);
329
+ const sock = socket ?? getDefaultSocketPath();
330
+ if (!(await hasSession(name, sock)))
331
+ return 'absent';
332
+ const res = await runTmux({
333
+ socket: sock,
334
+ args: ['list-panes', '-t', `=${name}`, '-F', '#{pane_dead}'],
335
+ throwOnError: false,
336
+ }).catch(() => null);
337
+ if (!res || res.code !== 0) {
338
+ await killSession(name, sock).catch(() => { });
339
+ return 'killed';
340
+ }
341
+ const flags = res.stdout.split('\n').map((s) => s.trim()).filter(Boolean);
342
+ if (flags.some((d) => d === '0'))
343
+ return 'kept';
344
+ await killSession(name, sock).catch(() => { });
345
+ return 'killed';
346
+ }
318
347
  /**
319
348
  * Kill every session on the shared server AND the server itself, then prune
320
349
  * meta files. Wipes the socket so the next `new` starts from a clean slate.
@@ -0,0 +1,67 @@
1
+ /**
2
+ * Cross-session failure clustering + time-wasted attribution for the traces
3
+ * insight engine (PHNX-3141) — the piece that turns per-tool error counts
4
+ * into "here is your #1 systemic problem and what it cost."
5
+ *
6
+ * Pure and SQL-shaped: it consumes the same `SyncRow[]` + `tool_calls` rows
7
+ * `buildIndexShard` already loads (no re-parsing of transcripts), so cost
8
+ * stays proportional to this sync's row count, never the full corpus.
9
+ *
10
+ * `tool_calls` rows carry `ordinal`/`timestamp` per call within a session
11
+ * (`db.ts`'s `idx_tool_calls_session ON tool_calls(session_id, ordinal)`), which
12
+ * is enough to reconstruct per-session call order and inter-call gaps without a
13
+ * full `SessionTrajectory` — that is what makes this incremental at scale.
14
+ *
15
+ * Scope note: `FailureSignature` does not yet carry a `phenotype`
16
+ * (false-termination / out-of-order / …, `phenotype.ts`) — classifying that
17
+ * needs the full derived trajectory (turns, ordered steps, gaps), which is
18
+ * only ever materialized per-session during upload, not cached the way
19
+ * `InsightFacets` is. Folding it in is a real, scoped follow-up (see
20
+ * `cli/AGENTS.md`), not a silent omission.
21
+ */
22
+ import { type TraceFailureCause } from './classify.js';
23
+ import { type LatencyInsight } from './segments.js';
24
+ import { type SyncRow, type ToolCallRow, type TracesIndexShard } from './sync.js';
25
+ export interface FailureSignature {
26
+ tool: string;
27
+ cause: TraceFailureCause;
28
+ /** Normalized error text — volatile tokens (ids, counts, countdowns) stripped so instances fold together. */
29
+ key: string;
30
+ }
31
+ export interface FailurePattern {
32
+ /** Stable hash of the signature — deep-linkable, unaffected by row order. */
33
+ id: string;
34
+ label: string;
35
+ signature: FailureSignature;
36
+ /** Distinct sessions this pattern occurred in. */
37
+ sessions: number;
38
+ /** Total failing calls matching this signature. */
39
+ occurrences: number;
40
+ /** Estimated ms of retry/stall time attributable to this pattern (see attribution rule below). */
41
+ wastedMs: number;
42
+ /** Bounded example session ids for drill-down. */
43
+ exampleSessionIds: string[];
44
+ /** Movement vs the same pattern id in the previous shard. */
45
+ drift: 'up' | 'flat' | 'down';
46
+ }
47
+ export interface ComputedInsights {
48
+ /** Top-K patterns ranked by wastedMs (impact) — a rare 1-occurrence/8h loop still surfaces. */
49
+ failurePatterns: FailurePattern[];
50
+ /** Sum of wastedMs across every cluster found this sync, not just the top-K rows above. */
51
+ wastedMsTotal: number;
52
+ latency: LatencyInsight;
53
+ }
54
+ /** Strip volatile tokens from a failure's evidence text so repeat instances hash identically. */
55
+ export declare function normalizeErrorKey(desc: string, raw: string | null): string;
56
+ /**
57
+ * Cluster failed tool calls into ranked patterns and estimate the wasted time
58
+ * behind each, plus device-wide time-to-first-tool latency.
59
+ *
60
+ * wastedMs attribution: the gap between a failed call and the NEXT call in the
61
+ * same session counts as wasted when either (a) the next call repeats the same
62
+ * signature (a retry loop) or (b) the gap itself is a stall (≥60s) — an idle
63
+ * gap unrelated to a nearby failure is never counted. This is an estimate, not
64
+ * ground truth (a stall could be legitimate user think-time); it is not
65
+ * inflated by folding in ordinary processing time between unrelated calls.
66
+ */
67
+ export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null): ComputedInsights;
@@ -0,0 +1,178 @@
1
+ /**
2
+ * Cross-session failure clustering + time-wasted attribution for the traces
3
+ * insight engine (PHNX-3141) — the piece that turns per-tool error counts
4
+ * into "here is your #1 systemic problem and what it cost."
5
+ *
6
+ * Pure and SQL-shaped: it consumes the same `SyncRow[]` + `tool_calls` rows
7
+ * `buildIndexShard` already loads (no re-parsing of transcripts), so cost
8
+ * stays proportional to this sync's row count, never the full corpus.
9
+ *
10
+ * `tool_calls` rows carry `ordinal`/`timestamp` per call within a session
11
+ * (`db.ts`'s `idx_tool_calls_session ON tool_calls(session_id, ordinal)`), which
12
+ * is enough to reconstruct per-session call order and inter-call gaps without a
13
+ * full `SessionTrajectory` — that is what makes this incremental at scale.
14
+ *
15
+ * Scope note: `FailureSignature` does not yet carry a `phenotype`
16
+ * (false-termination / out-of-order / …, `phenotype.ts`) — classifying that
17
+ * needs the full derived trajectory (turns, ordered steps, gaps), which is
18
+ * only ever materialized per-session during upload, not cached the way
19
+ * `InsightFacets` is. Folding it in is a real, scoped follow-up (see
20
+ * `cli/AGENTS.md`), not a silent omission.
21
+ */
22
+ import { classifyCause } from './classify.js';
23
+ import { computeLatency } from './segments.js';
24
+ import { failureDescription } from './sync.js';
25
+ // ---------------------------------------------------------------------------
26
+ // Tunables
27
+ // ---------------------------------------------------------------------------
28
+ /** Bounded shard size — patterns, not sessions, so 738 or 738k render identically. */
29
+ const TOP_K_PATTERNS = 25;
30
+ const MAX_EXAMPLE_SESSIONS = 5;
31
+ /** A gap this long right after a failure reads as an idle stall, not think-time (matches sync.ts's own "stalled Xm" threshold). */
32
+ const STALL_MS = 60_000;
33
+ // ---------------------------------------------------------------------------
34
+ // Signature normalization — fold volatile per-instance text together
35
+ // ---------------------------------------------------------------------------
36
+ const VOLATILE_TOKEN_PATTERNS = [
37
+ { pattern: /\bfor user [\w.-]+/gi, replacement: 'for user _' },
38
+ { pattern: /\btry again in [\w.]+s?\b/gi, replacement: 'try again in _s' },
39
+ { pattern: /\b[0-9a-f]{7,40}\b/gi, replacement: '_sha_' },
40
+ { pattern: /\b[\w.-]+@[\w.-]+\.\w+\b/gi, replacement: '_email_' },
41
+ { pattern: /\b\d+\b/g, replacement: '_n_' },
42
+ ];
43
+ /** Strip volatile tokens from a failure's evidence text so repeat instances hash identically. */
44
+ export function normalizeErrorKey(desc, raw) {
45
+ let text = (raw && raw.trim().length > 0 ? raw : desc).toLowerCase();
46
+ for (const { pattern, replacement } of VOLATILE_TOKEN_PATTERNS) {
47
+ text = text.replace(pattern, replacement);
48
+ }
49
+ return text.replace(/\s+/g, ' ').trim().slice(0, 160);
50
+ }
51
+ function hashSignature(tool, cause, key) {
52
+ const input = `${tool} ${cause} ${key}`;
53
+ let hash = 5381;
54
+ for (let i = 0; i < input.length; i++) {
55
+ hash = ((hash << 5) + hash + input.charCodeAt(i)) >>> 0;
56
+ }
57
+ return hash.toString(16).padStart(8, '0');
58
+ }
59
+ // ---------------------------------------------------------------------------
60
+ // Label rules — a table, not an if/else-by-name chain (matches segments.ts's TASK_TYPE_RULES)
61
+ // ---------------------------------------------------------------------------
62
+ const LABEL_RULES = [
63
+ { pattern: /rate limit/i, label: 'rate limit back-off loop' },
64
+ { pattern: /permission denied/i, label: 'permission denied' },
65
+ { pattern: /not found|no such file/i, label: 'missing resource' },
66
+ { pattern: /timed? ?out/i, label: 'timeout' },
67
+ { pattern: /econnrefused|connection refused|network/i, label: 'network error' },
68
+ { pattern: /conflict|diverged/i, label: 'git conflict' },
69
+ ];
70
+ function labelFor(tool, cause, key) {
71
+ if (cause === 'guard')
72
+ return `${tool}: git guard denial`;
73
+ if (cause === 'hook')
74
+ return `${tool}: hook denial`;
75
+ const rule = LABEL_RULES.find((row) => row.pattern.test(key));
76
+ return `${tool}: ${rule ? rule.label : key.slice(0, 48)}`;
77
+ }
78
+ // ---------------------------------------------------------------------------
79
+ // Public API
80
+ // ---------------------------------------------------------------------------
81
+ /**
82
+ * Cluster failed tool calls into ranked patterns and estimate the wasted time
83
+ * behind each, plus device-wide time-to-first-tool latency.
84
+ *
85
+ * wastedMs attribution: the gap between a failed call and the NEXT call in the
86
+ * same session counts as wasted when either (a) the next call repeats the same
87
+ * signature (a retry loop) or (b) the gap itself is a stall (≥60s) — an idle
88
+ * gap unrelated to a nearby failure is never counted. This is an estimate, not
89
+ * ground truth (a stall could be legitimate user think-time); it is not
90
+ * inflated by folding in ordinary processing time between unrelated calls.
91
+ */
92
+ export function computeInsights(rows, calls, prevShard) {
93
+ const bySession = new Map();
94
+ for (const call of calls) {
95
+ const list = bySession.get(call.session_id);
96
+ if (list)
97
+ list.push(call);
98
+ else
99
+ bySession.set(call.session_id, [call]);
100
+ }
101
+ const groups = new Map();
102
+ for (const [sessionId, sessionCalls] of bySession) {
103
+ const ordered = [...sessionCalls].sort((a, b) => a.ordinal - b.ordinal);
104
+ for (let i = 0; i < ordered.length; i++) {
105
+ const call = ordered[i];
106
+ if (call.outcome !== 'error')
107
+ continue;
108
+ const cause = classifyCause(call);
109
+ const key = normalizeErrorKey(failureDescription(call, cause), call.error);
110
+ const groupKey = `${call.tool} ${cause} ${key}`;
111
+ let group = groups.get(groupKey);
112
+ if (!group) {
113
+ group = { tool: call.tool, cause, key, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
114
+ groups.set(groupKey, group);
115
+ }
116
+ group.occurrences++;
117
+ group.sessions.add(sessionId);
118
+ if (group.examples.length < MAX_EXAMPLE_SESSIONS && !group.examples.includes(sessionId)) {
119
+ group.examples.push(sessionId);
120
+ }
121
+ const next = ordered[i + 1];
122
+ if (!next)
123
+ continue;
124
+ const gapMs = Date.parse(next.timestamp) - Date.parse(call.timestamp);
125
+ if (!Number.isFinite(gapMs) || gapMs <= 0)
126
+ continue;
127
+ const nextIsSameFailure = next.outcome === 'error' &&
128
+ next.tool === call.tool &&
129
+ classifyCause(next) === cause &&
130
+ normalizeErrorKey(failureDescription(next, cause), next.error) === key;
131
+ if (nextIsSameFailure || gapMs >= STALL_MS) {
132
+ group.wastedMs += gapMs;
133
+ }
134
+ }
135
+ }
136
+ const prevById = new Map((prevShard?.failurePatterns ?? []).map((p) => [p.id, p]));
137
+ const allPatterns = [...groups.values()].map((group) => {
138
+ const id = hashSignature(group.tool, group.cause, group.key);
139
+ const prev = prevById.get(id);
140
+ const drift = !prev
141
+ ? 'up'
142
+ : group.occurrences > prev.occurrences
143
+ ? 'up'
144
+ : group.occurrences < prev.occurrences
145
+ ? 'down'
146
+ : 'flat';
147
+ return {
148
+ id,
149
+ label: labelFor(group.tool, group.cause, group.key),
150
+ signature: { tool: group.tool, cause: group.cause, key: group.key },
151
+ sessions: group.sessions.size,
152
+ occurrences: group.occurrences,
153
+ wastedMs: group.wastedMs,
154
+ exampleSessionIds: group.examples,
155
+ drift,
156
+ };
157
+ });
158
+ const wastedMsTotal = allPatterns.reduce((sum, p) => sum + p.wastedMs, 0);
159
+ const failurePatterns = [...allPatterns]
160
+ .sort((a, b) => b.wastedMs - a.wastedMs || b.occurrences - a.occurrences || a.id.localeCompare(b.id))
161
+ .slice(0, TOP_K_PATTERNS);
162
+ const latency = computeLatency(firstToolSegments(rows, bySession));
163
+ return { failurePatterns, wastedMsTotal, latency };
164
+ }
165
+ /** Synthesize one-step SegmentSessions carrying only the time-to-first-tool offset, for computeLatency() reuse. */
166
+ function firstToolSegments(rows, bySession) {
167
+ return rows.flatMap((row) => {
168
+ const sessionCalls = bySession.get(row.id);
169
+ if (!sessionCalls || sessionCalls.length === 0)
170
+ return [];
171
+ const first = sessionCalls.reduce((min, call) => (call.ordinal < min.ordinal ? call : min));
172
+ const sessionStartMs = Date.parse(row.timestamp);
173
+ const firstCallMs = Date.parse(first.timestamp);
174
+ if (!Number.isFinite(sessionStartMs) || !Number.isFinite(firstCallMs))
175
+ return [];
176
+ return [{ steps: [{ startMs: Math.max(0, firstCallMs - sessionStartMs) }] }];
177
+ });
178
+ }
@@ -0,0 +1,67 @@
1
+ /**
2
+ * Failure phenotype classifier + outcome taxonomy for the traces insight engine.
3
+ *
4
+ * Both functions are pure: they take a redacted {@link SessionDetail} (the same
5
+ * shape the traces sync writes to `sessions/<id>.json`) and return a decision
6
+ * derived only from the already-derived step/gap/meta signal. They never read
7
+ * raw transcript text and never fabricate a signal that is not in the data.
8
+ *
9
+ * The rubrics are expressed as data-driven tables of conditions, not as
10
+ * if/else-by-name chains. Each table row is a named phenotype/outcome with a
11
+ * declarative predicate; the classifier walks the table in priority order and
12
+ * returns the first match, or the honest lower-confidence default when no
13
+ * high-confidence signal is present.
14
+ */
15
+ import type { SessionDetail } from './sync.js';
16
+ /** A failure mode detectable from the derived trajectory of a session. */
17
+ export type FailurePhenotype = 'false-termination' | 'premature-completion' | 'out-of-order' | 'failure-to-act';
18
+ /**
19
+ * The coarse outcome of a session's work.
20
+ *
21
+ * - `merged` : explicit PR/branch merge signal in the steps (high confidence).
22
+ * - `tests-green` : explicit test command returned ok with no later failure (high confidence).
23
+ * - `partial` : progress was made but no landing/test signal is present (low confidence default).
24
+ * - `abandoned` : errored, stalled, and unresolved.
25
+ * - `human-takeover`: the final substantive action was a human-facing ask/wait.
26
+ * - `invalid-env` : environment/setup failures dominated the session.
27
+ */
28
+ export type TraceOutcome = 'merged' | 'tests-green' | 'partial' | 'abandoned' | 'human-takeover' | 'invalid-env';
29
+ export interface PhenotypeResult {
30
+ phenotype: FailurePhenotype | null;
31
+ reason: string;
32
+ }
33
+ export interface OutcomeResult {
34
+ outcome: TraceOutcome;
35
+ confidence: 'high' | 'medium' | 'low';
36
+ reason: string;
37
+ }
38
+ /**
39
+ * Classify the failure phenotype of a session from its derived trajectory.
40
+ *
41
+ * Definitions (from agent-failure research):
42
+ * - `false-termination` — stopped with an unresolved error.
43
+ * - `premature-completion` — declared done while tests were failing or no
44
+ * verification step ran for the engineering work.
45
+ * - `out-of-order` — a write/edit step occurred before any read/plan of the
46
+ * target.
47
+ * - `failure-to-act` — stalled or produced no meaningful tool use.
48
+ *
49
+ * Returns `null` when none of the failure phenotypes apply.
50
+ */
51
+ export declare function classifyPhenotype(session: SessionDetail): FailurePhenotype | null;
52
+ /** Detailed phenotype result with a human-readable reason. */
53
+ export declare function classifyPhenotypeDetailed(session: SessionDetail): PhenotypeResult;
54
+ /**
55
+ * Derive the coarse outcome of a session from its end state + tool signals.
56
+ *
57
+ * `merged` and `tests-green` are high-confidence only when an explicit signal
58
+ * is present in the derived steps. When that signal is genuinely not in the
59
+ * data, the function returns the honest lower-confidence value (`partial` for
60
+ * completed work without a landing signal, `abandoned` for errored/unresolved
61
+ * work, `invalid-env` for setup-dominant failures, `human-takeover` when the
62
+ * session ends on a human-facing ask).
63
+ */
64
+ export declare function deriveOutcome(session: SessionDetail): TraceOutcome;
65
+ /** Detailed outcome result with confidence and a human-readable reason. */
66
+ export declare function deriveOutcomeDetailed(session: SessionDetail): OutcomeResult;
67
+ export type { SessionDetail } from './sync.js';