@phnx-labs/agents-cli 1.22.57 → 1.22.58

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. package/CHANGELOG.md +56 -0
  2. package/dist/bootstrap.js +8 -1
  3. package/dist/commands/accounts.js +7 -3
  4. package/dist/commands/apply.js +10 -2
  5. package/dist/commands/fork.d.ts +23 -10
  6. package/dist/commands/fork.js +115 -58
  7. package/dist/commands/monitors.js +11 -0
  8. package/dist/commands/prune.js +5 -3
  9. package/dist/commands/routines.d.ts +8 -0
  10. package/dist/commands/routines.js +57 -3
  11. package/dist/commands/sessions-picker.d.ts +11 -0
  12. package/dist/commands/sessions-picker.js +16 -0
  13. package/dist/commands/sessions.js +1 -0
  14. package/dist/commands/share.d.ts +14 -0
  15. package/dist/commands/share.js +43 -2
  16. package/dist/commands/status.js +1 -1
  17. package/dist/commands/sync.js +83 -7
  18. package/dist/commands/traces.js +7 -0
  19. package/dist/index.d.ts +1 -1
  20. package/dist/index.js +6 -1
  21. package/dist/lib/account-registry.d.ts +5 -1
  22. package/dist/lib/account-registry.js +47 -14
  23. package/dist/lib/accounting/capacity.d.ts +18 -7
  24. package/dist/lib/accounting/capacity.js +19 -8
  25. package/dist/lib/accounting/usage-sync.d.ts +29 -1
  26. package/dist/lib/accounting/usage-sync.js +76 -2
  27. package/dist/lib/accounting/usage.js +7 -1
  28. package/dist/lib/auth-mint.d.ts +11 -1
  29. package/dist/lib/auth-mint.js +21 -6
  30. package/dist/lib/browser/ipc.d.ts +8 -0
  31. package/dist/lib/browser/ipc.js +87 -0
  32. package/dist/lib/browser/service.d.ts +19 -0
  33. package/dist/lib/browser/service.js +96 -11
  34. package/dist/lib/browser/sessions-list.js +10 -1
  35. package/dist/lib/daemon/runner.d.ts +3 -0
  36. package/dist/lib/daemon/runner.js +86 -45
  37. package/dist/lib/daemon/usage-sync-service.d.ts +3 -3
  38. package/dist/lib/daemon/usage-sync-service.js +14 -8
  39. package/dist/lib/daemon-services.js +1 -1
  40. package/dist/lib/devices/connect.d.ts +17 -8
  41. package/dist/lib/devices/connect.js +31 -14
  42. package/dist/lib/doctor-diff.js +77 -7
  43. package/dist/lib/fleet/manifest.d.ts +17 -0
  44. package/dist/lib/fleet/manifest.js +26 -0
  45. package/dist/lib/hooks/install.d.ts +27 -11
  46. package/dist/lib/hooks/install.js +42 -17
  47. package/dist/lib/hosts/reconnect.d.ts +52 -203
  48. package/dist/lib/hosts/reconnect.js +64 -284
  49. package/dist/lib/installations/migrate.d.ts +6 -120
  50. package/dist/lib/installations/migrate.js +27 -259
  51. package/dist/lib/installations/shims.d.ts +13 -95
  52. package/dist/lib/installations/shims.js +22 -139
  53. package/dist/lib/installations/store.js +1 -1
  54. package/dist/lib/installations/versions.d.ts +26 -133
  55. package/dist/lib/installations/versions.js +41 -204
  56. package/dist/lib/plugins/skills.d.ts +8 -1
  57. package/dist/lib/plugins/skills.js +18 -2
  58. package/dist/lib/refresh.d.ts +9 -0
  59. package/dist/lib/refresh.js +3 -1
  60. package/dist/lib/routine-readiness.d.ts +15 -1
  61. package/dist/lib/routine-readiness.js +41 -0
  62. package/dist/lib/sandbox.d.ts +4 -1
  63. package/dist/lib/sandbox.js +30 -1
  64. package/dist/lib/secrets/agent.d.ts +80 -225
  65. package/dist/lib/secrets/agent.js +139 -401
  66. package/dist/lib/secrets/bundles.d.ts +73 -222
  67. package/dist/lib/secrets/bundles.js +168 -467
  68. package/dist/lib/secrets/reaper.d.ts +28 -70
  69. package/dist/lib/secrets/reaper.js +30 -85
  70. package/dist/lib/secrets/remote.d.ts +42 -129
  71. package/dist/lib/secrets/remote.js +55 -173
  72. package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
  73. package/dist/lib/self-heal/checks/install-staging.js +96 -0
  74. package/dist/lib/self-heal/registry.js +2 -0
  75. package/dist/lib/self-heal/types.d.ts +1 -1
  76. package/dist/lib/self-update.d.ts +23 -0
  77. package/dist/lib/self-update.js +50 -0
  78. package/dist/lib/session/active.d.ts +13 -1
  79. package/dist/lib/session/active.js +2 -0
  80. package/dist/lib/session/db.d.ts +20 -1
  81. package/dist/lib/session/db.js +139 -9
  82. package/dist/lib/session/fork.d.ts +45 -26
  83. package/dist/lib/session/fork.js +32 -95
  84. package/dist/lib/session/tool-calls.d.ts +43 -1
  85. package/dist/lib/session/tool-calls.js +74 -44
  86. package/dist/lib/session/tool-store.d.ts +33 -2
  87. package/dist/lib/session/tool-store.js +56 -3
  88. package/dist/lib/staleness/writers/sources.d.ts +5 -0
  89. package/dist/lib/staleness/writers/sources.js +2 -1
  90. package/dist/lib/sync-status.d.ts +22 -0
  91. package/dist/lib/sync-status.js +27 -0
  92. package/dist/lib/sync-umbrella.d.ts +9 -0
  93. package/dist/lib/sync-umbrella.js +21 -2
  94. package/dist/lib/traces/insights.d.ts +47 -14
  95. package/dist/lib/traces/insights.js +92 -21
  96. package/dist/lib/traces/phenotype.d.ts +23 -3
  97. package/dist/lib/traces/phenotype.js +72 -24
  98. package/dist/lib/traces/sync.d.ts +15 -0
  99. package/dist/lib/traces/sync.js +104 -19
  100. package/dist/lib/traces/worker-template.js +154 -1
  101. package/package.json +1 -1
@@ -12,12 +12,18 @@
12
12
  * is enough to reconstruct per-session call order and inter-call gaps without a
13
13
  * full `SessionTrajectory` — that is what makes this incremental at scale.
14
14
  *
15
- * Scope note: `FailureSignature` does not yet carry a `phenotype`
16
- * (false-termination / out-of-order / …, `phenotype.ts`) — classifying that
17
- * needs the full derived trajectory (turns, ordered steps, gaps), which is
18
- * only ever materialized per-session during upload, not cached the way
19
- * `InsightFacets` is. Folding it in is a real, scoped follow-up (see
20
- * `cli/AGENTS.md`), not a silent omission.
15
+ * Failure phenotype (false-termination / out-of-order / premature-completion /
16
+ * failure-to-act, `phenotype.ts`) is folded in as a fourth grouping dimension
17
+ * (PHNX-3327). Classifying it needs the full derived trajectory, so it is NOT
18
+ * computed here — the caller passes a per-session `phenotypes` map that
19
+ * `buildIndexShard` fills from the persisted, mtime+size-keyed
20
+ * `session_phenotypes` cache for the WHOLE corpus. Keying the group on the
21
+ * cached-per-session phenotype (never on this run's incremental batch) is what
22
+ * keeps two identically-signatured sessions in one cluster regardless of when
23
+ * each was synced. The `signature` OUTPUT is unchanged (`{ tool, cause, key }`);
24
+ * phenotype is an added dimension carried alongside it, so callers that never
25
+ * pass a map (the unit tests, a pre-phenotype caller) see the exact prior
26
+ * grouping.
21
27
  */
22
28
  import { classifyCause } from './classify.js';
23
29
  import { computeLatency } from './segments.js';
@@ -30,6 +36,24 @@ const TOP_K_PATTERNS = 25;
30
36
  const MAX_EXAMPLE_SESSIONS = 5;
31
37
  /** A gap this long right after a failure reads as an idle stall, not think-time (matches sync.ts's own "stalled Xm" threshold). */
32
38
  const STALL_MS = 60_000;
39
+ /**
40
+ * Upper bound on how much of a SINGLE inter-call gap is attributable to a
41
+ * failure. Beyond a recovery window a gap is not failure-loop waste — it is a
42
+ * human away from a chat thread, an abandoned session, or an outage. This bounds
43
+ * BOTH branches, and that matters: `nextIsSameFailure` only compares
44
+ * `(tool, cause, normalized-key)`, with no temporal check, so a deterministic
45
+ * failure that recurs identically hours apart (a permanently-denied capability,
46
+ * a missing credential — e.g. a Slack user re-asking "where am I" at 2pm and
47
+ * 6pm) would otherwise look like an "active retry loop" and absorb the whole
48
+ * multi-hour gap — the very artifact this fix targets. A genuine active loop has
49
+ * MANY short gaps that each stay under this cap and still sum to a large total,
50
+ * so bounding a single gap doesn't hide it. Real single-tool stalls (a hung
51
+ * typecheck, a slow build) are minutes and stay fully counted.
52
+ *
53
+ * The same cap bounds the call's OWN blocking duration (PHNX-3437) so a corrupt
54
+ * or backwards end timestamp can't book a single call as hours of waste either.
55
+ */
56
+ const MAX_GAP_ATTRIBUTION_MS = 30 * 60_000;
33
57
  // ---------------------------------------------------------------------------
34
58
  // Signature normalization — fold volatile per-instance text together
35
59
  // ---------------------------------------------------------------------------
@@ -48,8 +72,8 @@ export function normalizeErrorKey(desc, raw) {
48
72
  }
49
73
  return text.replace(/\s+/g, ' ').trim().slice(0, 160);
50
74
  }
51
- function hashSignature(tool, cause, key) {
52
- const input = `${tool} ${cause} ${key}`;
75
+ function hashSignature(tool, cause, key, phenotype) {
76
+ const input = `${tool} ${cause} ${key} ${phenotype ?? ''}`;
53
77
  let hash = 5381;
54
78
  for (let i = 0; i < input.length; i++) {
55
79
  hash = ((hash << 5) + hash + input.charCodeAt(i)) >>> 0;
@@ -82,14 +106,33 @@ function labelFor(tool, cause, key) {
82
106
  * Cluster failed tool calls into ranked patterns and estimate the wasted time
83
107
  * behind each, plus device-wide time-to-first-tool latency.
84
108
  *
85
- * wastedMs attribution: the gap between a failed call and the NEXT call in the
86
- * same session counts as wasted when either (a) the next call repeats the same
87
- * signature (a retry loop) or (b) the gap itself is a stall (≥60s) — an idle
88
- * gap unrelated to a nearby failure is never counted. This is an estimate, not
89
- * ground truth (a stall could be legitimate user think-time); it is not
90
- * inflated by folding in ordinary processing time between unrelated calls.
109
+ * wastedMs attribution has two parts that sum:
110
+ *
111
+ * (1) The failed call's OWN blocking duration — `end_timestamp - timestamp`
112
+ * (PHNX-3437). A call that hung for minutes and then failed wasted that whole
113
+ * time even if it was the last call in its session or was followed quickly by an
114
+ * unrelated call — the case the gap heuristic alone booked as ~0. This is what
115
+ * makes a fail-fast fix measurable: a channel that stops hanging 5.5min on stdin
116
+ * and instead fails in <1s (PHNX-3407) drops from ~5.5min of attributed waste to
117
+ * ~0. Bounded by MAX_GAP_ATTRIBUTION_MS so a corrupt end timestamp can't dominate.
118
+ *
119
+ * (2) The idle gap AFTER the call, before the NEXT call in the same session,
120
+ * counted when either (a) the next call repeats the same signature (a retry
121
+ * loop) or (b) the gap is a stall (≥60s) before an unrelated next call — each
122
+ * bounded by MAX_GAP_ATTRIBUTION_MS so one failure can't absorb hours of
123
+ * human-away idle in an async channel session, whether as a lone stall or a
124
+ * same-signature re-ask hours later; a genuine active retry loop is many short
125
+ * gaps that each clear the cap and still sum large. When the end timestamp is
126
+ * known, this gap is measured from the call's END, so the blocking time counted
127
+ * in (1) is never double-counted; for a NULL end (rows an older extractor stored,
128
+ * or a call still pending at scan end) it falls back to the original
129
+ * gap-from-START heuristic unchanged — no crash, no NaN, no regression.
130
+ *
131
+ * An idle gap unrelated to a nearby failure is never counted. This is an
132
+ * estimate, not ground truth; it is not inflated by folding in ordinary
133
+ * processing time between unrelated calls.
91
134
  */
92
- export function computeInsights(rows, calls, prevShard) {
135
+ export function computeInsights(rows, calls, prevShard, phenotypes) {
93
136
  const bySession = new Map();
94
137
  for (const call of calls) {
95
138
  const list = bySession.get(call.session_id);
@@ -100,6 +143,11 @@ export function computeInsights(rows, calls, prevShard) {
100
143
  }
101
144
  const groups = new Map();
102
145
  for (const [sessionId, sessionCalls] of bySession) {
146
+ // Phenotype is a per-session property (one classification per session), so
147
+ // every failing call in this session shares it. A caller that passes no map
148
+ // (unit tests, a pre-phenotype caller) collapses the dimension to `null`,
149
+ // yielding the exact prior grouping.
150
+ const phenotype = phenotypes?.get(sessionId) ?? null;
103
151
  const ordered = [...sessionCalls].sort((a, b) => a.ordinal - b.ordinal);
104
152
  for (let i = 0; i < ordered.length; i++) {
105
153
  const call = ordered[i];
@@ -107,10 +155,10 @@ export function computeInsights(rows, calls, prevShard) {
107
155
  continue;
108
156
  const cause = classifyCause(call);
109
157
  const key = normalizeErrorKey(failureDescription(call, cause), call.error);
110
- const groupKey = `${call.tool} ${cause} ${key}`;
158
+ const groupKey = `${call.tool} ${cause} ${key} ${phenotype ?? ''}`;
111
159
  let group = groups.get(groupKey);
112
160
  if (!group) {
113
- group = { tool: call.tool, cause, key, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
161
+ group = { tool: call.tool, cause, key, phenotype, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
114
162
  groups.set(groupKey, group);
115
163
  }
116
164
  group.occurrences++;
@@ -118,24 +166,46 @@ export function computeInsights(rows, calls, prevShard) {
118
166
  if (group.examples.length < MAX_EXAMPLE_SESSIONS && !group.examples.includes(sessionId)) {
119
167
  group.examples.push(sessionId);
120
168
  }
169
+ // (1) The call's OWN blocking duration (end minus start) is the primary
170
+ // signal — see the computeInsights docblock. Attributed whenever the end
171
+ // timestamp is present, independent of whether a next call follows.
172
+ const startMs = Date.parse(call.timestamp);
173
+ const endMs = call.end_timestamp ? Date.parse(call.end_timestamp) : NaN;
174
+ const hasEnd = Number.isFinite(endMs) && Number.isFinite(startMs);
175
+ if (hasEnd) {
176
+ const ownMs = endMs - startMs;
177
+ if (ownMs > 0)
178
+ group.wastedMs += Math.min(ownMs, MAX_GAP_ATTRIBUTION_MS);
179
+ }
180
+ // (2) The idle gap after the call, before the next one. Measured from the
181
+ // call's END when known (so the blocking time in (1) isn't double-counted),
182
+ // else from its START — the original heuristic, unchanged for NULL ends.
121
183
  const next = ordered[i + 1];
122
184
  if (!next)
123
185
  continue;
124
- const gapMs = Date.parse(next.timestamp) - Date.parse(call.timestamp);
186
+ const gapFromMs = hasEnd ? endMs : startMs;
187
+ const gapMs = Date.parse(next.timestamp) - gapFromMs;
125
188
  if (!Number.isFinite(gapMs) || gapMs <= 0)
126
189
  continue;
127
190
  const nextIsSameFailure = next.outcome === 'error' &&
128
191
  next.tool === call.tool &&
129
192
  classifyCause(next) === cause &&
130
193
  normalizeErrorKey(failureDescription(next, cause), next.error) === key;
131
- if (nextIsSameFailure || gapMs >= STALL_MS) {
132
- group.wastedMs += gapMs;
194
+ if (nextIsSameFailure) {
195
+ // Retry loop: a real active loop is many short gaps, each under the cap,
196
+ // summing to a large total. Bound a single gap so one huge same-signature
197
+ // gap (a re-ask hours later, not active retrying) can't absorb it all.
198
+ group.wastedMs += Math.min(gapMs, MAX_GAP_ATTRIBUTION_MS);
199
+ }
200
+ else if (gapMs >= STALL_MS) {
201
+ // Lone stall before an unrelated next call: same bounded recovery window.
202
+ group.wastedMs += Math.min(gapMs, MAX_GAP_ATTRIBUTION_MS);
133
203
  }
134
204
  }
135
205
  }
136
206
  const prevById = new Map((prevShard?.failurePatterns ?? []).map((p) => [p.id, p]));
137
207
  const allPatterns = [...groups.values()].map((group) => {
138
- const id = hashSignature(group.tool, group.cause, group.key);
208
+ const id = hashSignature(group.tool, group.cause, group.key, group.phenotype);
139
209
  const prev = prevById.get(id);
140
210
  const drift = !prev
141
211
  ? 'up'
@@ -148,6 +218,7 @@ export function computeInsights(rows, calls, prevShard) {
148
218
  id,
149
219
  label: labelFor(group.tool, group.cause, group.key),
150
220
  signature: { tool: group.tool, cause: group.cause, key: group.key },
221
+ phenotype: group.phenotype,
151
222
  sessions: group.sessions.size,
152
223
  occurrences: group.occurrences,
153
224
  wastedMs: group.wastedMs,
@@ -35,13 +35,33 @@ export interface OutcomeResult {
35
35
  confidence: 'high' | 'medium' | 'low';
36
36
  reason: string;
37
37
  }
38
+ /**
39
+ * Did a run that hit tool errors nonetheless *recover and finish*?
40
+ *
41
+ * The causal recovery test: a **substantive** tool step (not a human-facing
42
+ * `AskUserQuestion` / `SendMessage` / `wait`) SUCCEEDED strictly AFTER the last
43
+ * error's ordinal, AND that success resolves the failed work — its
44
+ * {@link workSignature} matches an errored step's. A later success of *unrelated*
45
+ * work does NOT count: a `bun test` failure followed by an incidental `ls` leaves
46
+ * the failed test unresolved, so the run stays errored, even though the `ls`
47
+ * succeeded after it. A punt to a human is excluded twice over — human-facing
48
+ * tools are outside the substantive set, and they never match a failed signature.
49
+ *
50
+ * This is the single source of truth for "did this finish", shared by the
51
+ * false-termination phenotype below and `sync.ts`'s `deriveRunOutcome` — so a
52
+ * run's console outcome and its failure phenotype can never disagree about
53
+ * whether it recovered. It reads only the derived steps, never `meta.outcome`,
54
+ * precisely so `deriveRunOutcome` can call it while it is still *computing*
55
+ * `meta.outcome`.
56
+ */
57
+ export declare function recoveredAfterErrors(session: Pick<SessionDetail, 'steps'>): boolean;
38
58
  /**
39
59
  * Classify the failure phenotype of a session from its derived trajectory.
40
60
  *
41
61
  * Definitions (from agent-failure research):
42
- * - `false-termination` — stopped with an unresolved error.
43
- * - `premature-completion` — declared done while tests were failing or no
44
- * verification step ran for the engineering work.
62
+ * - `false-termination` — stopped with an unresolved error (did not causally recover).
63
+ * - `premature-completion` — declared done with no verification step for the
64
+ * engineering work.
45
65
  * - `out-of-order` — a write/edit step occurred before any read/plan of the
46
66
  * target.
47
67
  * - `failure-to-act` — stalled or produced no meaningful tool use.
@@ -156,30 +156,75 @@ function reasonOutOfOrder(session) {
156
156
  const tool = session.steps.find((s) => WRITE_EDIT_TOOLS.has(s.tool ?? s.lane));
157
157
  return `write/edit step ${tool?.tool ?? tool?.lane ?? ''} preceded any read/plan`;
158
158
  }
159
+ /**
160
+ * The **work signature** of a step — what work it represents, so a later success
161
+ * can be matched back to the specific failure it resolves. For a shell step this
162
+ * is the effective program (`bun`, `git`, `gh`, …, from `TrajectoryStep.program`),
163
+ * so a failed `bun test` is resolved by a later `bun test` but NOT by an incidental
164
+ * `ls`. For any other tool it is the tool identity, so a failed `Edit` is resolved
165
+ * by a later successful `Edit`, not by an unrelated `Read`. A shell step whose
166
+ * command did not parse (no `program`) degrades to the bare tool name, matching
167
+ * the pre-`program` behavior only for that unparseable minority.
168
+ */
169
+ function workSignature(step) {
170
+ const tool = step.tool ?? step.lane;
171
+ if (SHELL_TOOLS.has(tool) && step.program)
172
+ return `${tool}:${step.program}`;
173
+ return tool;
174
+ }
175
+ /**
176
+ * Did a run that hit tool errors nonetheless *recover and finish*?
177
+ *
178
+ * The causal recovery test: a **substantive** tool step (not a human-facing
179
+ * `AskUserQuestion` / `SendMessage` / `wait`) SUCCEEDED strictly AFTER the last
180
+ * error's ordinal, AND that success resolves the failed work — its
181
+ * {@link workSignature} matches an errored step's. A later success of *unrelated*
182
+ * work does NOT count: a `bun test` failure followed by an incidental `ls` leaves
183
+ * the failed test unresolved, so the run stays errored, even though the `ls`
184
+ * succeeded after it. A punt to a human is excluded twice over — human-facing
185
+ * tools are outside the substantive set, and they never match a failed signature.
186
+ *
187
+ * This is the single source of truth for "did this finish", shared by the
188
+ * false-termination phenotype below and `sync.ts`'s `deriveRunOutcome` — so a
189
+ * run's console outcome and its failure phenotype can never disagree about
190
+ * whether it recovered. It reads only the derived steps, never `meta.outcome`,
191
+ * precisely so `deriveRunOutcome` can call it while it is still *computing*
192
+ * `meta.outcome`.
193
+ */
194
+ export function recoveredAfterErrors(session) {
195
+ const substantive = substantiveSteps(session);
196
+ if (substantive.length === 0)
197
+ return false;
198
+ const last = substantive[substantive.length - 1];
199
+ if (last.outcome === 'error')
200
+ return false;
201
+ const lastErrorOrdinal = lastStepOrdinalOf(session, (s) => s.outcome === 'error');
202
+ if (lastErrorOrdinal === undefined)
203
+ return true;
204
+ const failedSignatures = new Set();
205
+ for (const s of session.steps) {
206
+ if (s.outcome === 'error')
207
+ failedSignatures.add(workSignature(s));
208
+ }
209
+ return substantive.some((s) => s.ordinal > lastErrorOrdinal &&
210
+ s.outcome === 'ok' &&
211
+ !HUMAN_FACING_TOOLS.has(s.tool ?? s.lane) &&
212
+ failedSignatures.has(workSignature(s)));
213
+ }
159
214
  /**
160
215
  * False-termination: the session stopped with an unresolved error.
161
216
  *
162
217
  * Rubric:
163
218
  * - meta outcome is `errored`, and
164
219
  * - at least one step outcome is `error`, and
165
- * - the last non-thinking step is an error or is followed by no successful recovery.
220
+ * - the run did not causally recover ({@link recoveredAfterErrors}).
166
221
  */
167
222
  function isFalseTermination(session) {
168
223
  if (session.meta.outcome !== 'errored')
169
224
  return false;
170
225
  if (session.meta.errorCount === 0)
171
226
  return false;
172
- const substantive = substantiveSteps(session);
173
- if (substantive.length === 0)
174
- return true;
175
- const last = substantive[substantive.length - 1];
176
- if (last.outcome === 'error')
177
- return true;
178
- const lastErrorOrdinal = lastStepOrdinalOf(session, (s) => s.outcome === 'error');
179
- if (lastErrorOrdinal === undefined)
180
- return false;
181
- const recoveryAfter = substantive.some((s) => s.ordinal > lastErrorOrdinal && s.outcome === 'ok' && !HUMAN_FACING_TOOLS.has(s.tool ?? s.lane));
182
- return !recoveryAfter;
227
+ return !recoveredAfterErrors(session);
183
228
  }
184
229
  function reasonFalseTermination(session) {
185
230
  const substantive = substantiveSteps(session);
@@ -192,13 +237,21 @@ function reasonFalseTermination(session) {
192
237
  return 'errored with no successful recovery after the last error';
193
238
  }
194
239
  /**
195
- * Premature-completion: declared done while tests were failing or no verification
196
- * step ran for an engineering task.
240
+ * Premature-completion: declared done without a verification step for an
241
+ * engineering task.
197
242
  *
198
243
  * Rubric:
199
244
  * - meta outcome is `completed`, and
200
245
  * - the session performed write/edit work, and
201
- * - either errors were present or no test/build/lint verification step ran.
246
+ * - no test/build/lint verification step ran.
247
+ *
248
+ * `errorCount` is deliberately NOT a signal here. Under truthful outcomes
249
+ * (PHNX-3387) a `completed` run CAN carry `errorCount > 0` — that is precisely a
250
+ * recover-then-succeed run, where the errors were *resolved*, not left hanging.
251
+ * Keying prematurity off `errorCount > 0` would mislabel every such recovery as
252
+ * premature. (Under the pre-PHNX-3387 `errorCount > 0 ? errored : completed`
253
+ * derivation this branch was unreachable — a `completed` run always had
254
+ * `errorCount === 0` — so dropping it changes nothing for a clean-completed run.)
202
255
  */
203
256
  function isPrematureCompletion(session) {
204
257
  if (session.meta.outcome !== 'completed')
@@ -206,8 +259,6 @@ function isPrematureCompletion(session) {
206
259
  const didWriteEdit = session.steps.some((s) => WRITE_EDIT_TOOLS.has(s.tool ?? s.lane));
207
260
  if (!didWriteEdit)
208
261
  return false;
209
- if (session.meta.errorCount > 0)
210
- return true;
211
262
  const verified = session.steps.some((s) => {
212
263
  if (!SHELL_TOOLS.has(s.tool ?? s.lane))
213
264
  return false;
@@ -216,10 +267,7 @@ function isPrematureCompletion(session) {
216
267
  });
217
268
  return !verified;
218
269
  }
219
- function reasonPrematureCompletion(session) {
220
- if (session.meta.errorCount > 0) {
221
- return `declared completed with ${session.meta.errorCount} unresolved error(s)`;
222
- }
270
+ function reasonPrematureCompletion(_session) {
223
271
  return 'engineering work completed without a test/build/lint verification step';
224
272
  }
225
273
  /** Ordered rubric: the first matching phenotype wins. */
@@ -391,9 +439,9 @@ const OUTCOME_RULES = [
391
439
  * Classify the failure phenotype of a session from its derived trajectory.
392
440
  *
393
441
  * Definitions (from agent-failure research):
394
- * - `false-termination` — stopped with an unresolved error.
395
- * - `premature-completion` — declared done while tests were failing or no
396
- * verification step ran for the engineering work.
442
+ * - `false-termination` — stopped with an unresolved error (did not causally recover).
443
+ * - `premature-completion` — declared done with no verification step for the
444
+ * engineering work.
397
445
  * - `out-of-order` — a write/edit step occurred before any read/plan of the
398
446
  * target.
399
447
  * - `failure-to-act` — stalled or produced no meaningful tool use.
@@ -49,6 +49,14 @@ export interface SyncResult {
49
49
  parseFailed: number;
50
50
  /** Parsed fine but the upload PUT failed (network/5xx) — retried on the next sync. */
51
51
  uploadFailed: number;
52
+ /**
53
+ * The index shard build or upload failed. Undefined on success. The per-session
54
+ * data still uploaded (that loop runs first), but the aggregated console shard —
55
+ * stats, needs-attention, failure clusters, latency — was NOT refreshed, so the
56
+ * console keeps serving the last good index. Surfaced (not swallowed) so a stale
57
+ * console is diagnosable instead of looking like a clean sync. See PHNX-3401.
58
+ */
59
+ indexError?: string;
52
60
  }
53
61
  /** Push derived, redacted trajectories for this device to the traces store. */
54
62
  export declare function syncTraces(opts?: SyncOpts): Promise<SyncResult>;
@@ -106,6 +114,13 @@ export interface ToolCallRow {
106
114
  session_id: string;
107
115
  ordinal: number;
108
116
  timestamp: string;
117
+ /**
118
+ * When the call's result arrived — its own end time (PHNX-3437). Optional
119
+ * because rows an older extractor stored, and calls that never produced a
120
+ * result, carry NULL; `computeInsights` falls back to the bounded inter-call
121
+ * gap when it is absent.
122
+ */
123
+ end_timestamp?: string | null;
109
124
  tool: string;
110
125
  outcome: string;
111
126
  exit_code: number | null;
@@ -22,7 +22,7 @@
22
22
  import fs from 'node:fs';
23
23
  import os from 'node:os';
24
24
  import path from 'node:path';
25
- import { getDB, readSessionInsights, readSessionTopics, writeSessionInsights, writeSessionTopics, } from '../session/db.js';
25
+ import { getDB, readSessionInsights, readSessionPhenotypes, readSessionTopics, writeSessionInsights, writeSessionPhenotypes, writeSessionTopics, } from '../session/db.js';
26
26
  import { parseSession } from '../session/parse.js';
27
27
  import { buildTrajectory } from '../session/trajectory.js';
28
28
  import { computeInsightFacets } from '../session/insights.js';
@@ -31,6 +31,7 @@ import { getRuntimeStateDir } from '../state.js';
31
31
  import { resolveTracesBackend } from './backend.js';
32
32
  import { classifyCause, classifyTopic, computeDriftSignal, } from './classify.js';
33
33
  import { computeInsights } from './insights.js';
34
+ import { classifyPhenotype, recoveredAfterErrors } from './phenotype.js';
34
35
  /** Push derived, redacted trajectories for this device to the traces store. */
35
36
  export async function syncTraces(opts = {}) {
36
37
  const dryRun = opts.dryRun === true;
@@ -147,6 +148,7 @@ export async function syncTraces(opts = {}) {
147
148
  recordFailure(row, 'upload-failed', err);
148
149
  }
149
150
  }
151
+ let indexError;
150
152
  if (!opts.skipIndex) {
151
153
  try {
152
154
  const allRows = db
@@ -172,8 +174,13 @@ export async function syncTraces(opts = {}) {
172
174
  await putIndexShard(backend, device, owner, shard);
173
175
  }
174
176
  }
175
- catch {
176
- // index PUT/write failure is not fatal — the per-session data is already written
177
+ catch (err) {
178
+ // Not fatal to the per-session upload (that loop already ran), but it DOES
179
+ // mean the console shard is now stale. Record it so the caller can surface a
180
+ // warning instead of reporting a clean, green sync — a silent swallow here is
181
+ // exactly what let a 59h-stale, insight-less index hide in plain sight
182
+ // (PHNX-3401). Do NOT re-throw: the session data is durable and worth keeping.
183
+ indexError = err instanceof Error ? err.message : String(err);
177
184
  }
178
185
  }
179
186
  // A dry-run never advances the incremental watermark: it is a read-only export.
@@ -203,6 +210,7 @@ export async function syncTraces(opts = {}) {
203
210
  transcriptUnavailable,
204
211
  parseFailed,
205
212
  uploadFailed,
213
+ indexError,
206
214
  };
207
215
  }
208
216
  function percentile(values, ratio) {
@@ -247,6 +255,32 @@ function attentionFlags(errorCount, facets) {
247
255
  flags.push(`${correctionCount} correction${correctionCount === 1 ? '' : 's'}`);
248
256
  return flags;
249
257
  }
258
+ /**
259
+ * Persist a derived-cache warm-up (topics / insights) without letting a
260
+ * contended DB take down the whole index build.
261
+ *
262
+ * These write-backs only speed up the NEXT sync — the shard about to be built
263
+ * reads from the in-memory `topics` / `insights` maps that were already
264
+ * populated above, never from what this write persists. So the write is
265
+ * genuinely optional to the shard's correctness.
266
+ *
267
+ * Yet it was the single point that broke the console: on an active machine the
268
+ * Rush app holds `sessions.db`, this `BEGIN IMMEDIATE` waits out the 30s
269
+ * `busy_timeout` and throws `SQLITE_BUSY`, the throw escaped `buildIndexShard`,
270
+ * and `syncTraces` swallowed it — so the index (with wasted-time / failure
271
+ * clusters / latency) never re-uploaded and the dashboard sat 59h stale
272
+ * (PHNX-3401). Isolating the failure here keeps the index building; the warning
273
+ * makes the degraded cache visible instead of silent.
274
+ */
275
+ function persistDerivedCache(label, write) {
276
+ try {
277
+ write();
278
+ }
279
+ catch (err) {
280
+ const msg = err instanceof Error ? err.message : String(err);
281
+ console.warn(`traces: ${label} cache warm-up skipped (${msg}) — index still built`);
282
+ }
283
+ }
250
284
  /** Build the redacted rich console shard from indexed metadata and derived caches. */
251
285
  export function buildIndexShard(rows, device, owner, prevShard) {
252
286
  const db = getDB();
@@ -256,7 +290,7 @@ export function buildIndexShard(rows, device, owner, prevShard) {
256
290
  for (let i = 0; i < ids.length; i += 400) {
257
291
  const chunk = ids.slice(i, i + 400);
258
292
  calls.push(...db.prepare(`
259
- SELECT session_id, ordinal, timestamp, tool, outcome, exit_code, status_code, error_code, error, parse_error
293
+ SELECT session_id, ordinal, timestamp, end_timestamp, tool, outcome, exit_code, status_code, error_code, error, parse_error
260
294
  FROM tool_calls
261
295
  WHERE session_id IN (${chunk.map(() => '?').join(',')})
262
296
  `).all(...chunk));
@@ -283,26 +317,52 @@ export function buildIndexShard(rows, device, owner, prevShard) {
283
317
  topics.set(row.id, topic);
284
318
  return { id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, topic };
285
319
  });
286
- writeSessionTopics(missingTopics);
320
+ persistDerivedCache('session-topics', () => writeSessionTopics(missingTopics));
321
+ // Insights (frictionSignals) and phenotype (false-termination / …) both need
322
+ // the parsed transcript, which flat tool_calls rows don't carry — so both are
323
+ // lazily derived per-session and cached by transcript mtime+size, then read for
324
+ // the WHOLE corpus (`rows` = allRows) every sync. That full-corpus read is what
325
+ // keeps the phenotype grouping dimension consistent: a session synced weeks ago
326
+ // still contributes its real phenotype from cache, so it can never fragment away
327
+ // from an identically-signatured session synced this run purely by *when* each
328
+ // was first seen (PHNX-3327). A cache-miss row is parsed at most once here even
329
+ // when both derivations are missing.
287
330
  const insights = readSessionInsights(ids);
331
+ const phenotypes = readSessionPhenotypes(ids);
288
332
  const missingInsights = [];
289
- for (const row of rows.filter((candidate) => !insights.has(candidate.id))) {
333
+ const missingPhenotypes = [];
334
+ for (const row of rows) {
335
+ const needInsights = !insights.has(row.id);
336
+ const needPhenotype = !phenotypes.has(row.id);
337
+ if (!needInsights && !needPhenotype)
338
+ continue;
339
+ let events;
290
340
  try {
291
- const events = parseSession(row.file_path, row.agent);
292
- const facets = computeInsightFacets(events);
293
- insights.set(row.id, facets);
294
- missingInsights.push({
295
- id: row.id,
296
- fileMtimeMs: row.file_mtime_ms,
297
- fileSize: row.file_size,
298
- facets,
299
- });
341
+ events = parseSession(row.file_path, row.agent);
300
342
  }
301
343
  catch {
302
- continue;
344
+ continue; // gone/unreadable transcript — leave both uncached, same as before
345
+ }
346
+ if (needInsights) {
347
+ try {
348
+ const facets = computeInsightFacets(events);
349
+ insights.set(row.id, facets);
350
+ missingInsights.push({ id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, facets });
351
+ }
352
+ catch { /* leave this session's facets uncached; recompute next sync */ }
353
+ }
354
+ if (needPhenotype) {
355
+ try {
356
+ const traj = buildTrajectory(events, rowToMeta(row), { redact: true, knownSecrets });
357
+ const phenotype = classifyPhenotype(buildSessionDetail(traj));
358
+ phenotypes.set(row.id, phenotype);
359
+ missingPhenotypes.push({ id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, phenotype });
360
+ }
361
+ catch { /* leave this session's phenotype uncached; recompute next sync */ }
303
362
  }
304
363
  }
305
- writeSessionInsights(missingInsights);
364
+ persistDerivedCache('session-insights', () => writeSessionInsights(missingInsights));
365
+ persistDerivedCache('session-phenotypes', () => writeSessionPhenotypes(missingPhenotypes));
306
366
  const needsAttention = rows.flatMap((row) => {
307
367
  const facets = insights.get(row.id);
308
368
  const errorCount = errorCounts.get(row.id) ?? 0;
@@ -361,7 +421,7 @@ export function buildIndexShard(rows, device, owner, prevShard) {
361
421
  const prevHistory = prevShard?.bucketHistory ?? [];
362
422
  const bucketHistory = [...prevHistory, todayStats].slice(-14);
363
423
  const driftSignals = computeDriftSignal(prevHistory, todayStats);
364
- const patternInsights = computeInsights(rows, calls, prevShard);
424
+ const patternInsights = computeInsights(rows, calls, prevShard, phenotypes);
365
425
  return {
366
426
  schema: 1,
367
427
  device,
@@ -404,6 +464,31 @@ function buildWhereItWentWrong(traj) {
404
464
  return null;
405
465
  return `This run hit ${parts.join('; ')}.`;
406
466
  }
467
+ /**
468
+ * Truthful run-level outcome (PHNX-3387).
469
+ *
470
+ * A run with zero tool errors `completed`. A run that hit tool errors is
471
+ * `completed` ONLY when it *causally recovered* — a substantive, non-human-facing
472
+ * tool step succeeded strictly after the last error AND resolved the failed work
473
+ * (its work signature matches an errored step's), the exact predicate the
474
+ * false-termination phenotype uses ({@link recoveredAfterErrors}). A run whose
475
+ * last substantive step is the error, whose only post-error steps are human-facing
476
+ * (a punt to `AskUserQuestion` — the case the broken "last tool call ok" heuristic
477
+ * mislabeled `completed`), or whose only post-error success is unrelated work (a
478
+ * failed `bun test` followed by an incidental `ls`) stays `errored`.
479
+ *
480
+ * This is what makes `surfacedToolFailures` on a `completed` run honest: those
481
+ * are failures the run recovered from, not a green status hiding an unresolved
482
+ * failure. It never flips a run that ended unresolved to `completed` (no
483
+ * regression vs the old `errorCount > 0 ? errored : completed`), and it does not
484
+ * flip a run whose failed work was never resolved just because some later,
485
+ * unrelated call happened to succeed.
486
+ */
487
+ function deriveRunOutcome(traj) {
488
+ if (traj.errorCount === 0)
489
+ return 'completed';
490
+ return recoveredAfterErrors({ steps: traj.steps }) ? 'completed' : 'errored';
491
+ }
407
492
  /**
408
493
  * Map the derived trajectory to the console's SessionDetail shape, stripping
409
494
  * local-machine PII (full cwd, account) that would expose filesystem paths if
@@ -423,7 +508,7 @@ export function buildSessionDetail(traj) {
423
508
  errorCount: traj.errorCount,
424
509
  tokens: stats.outputTokens ?? 0,
425
510
  costUsd: s.costUsd ?? 0,
426
- outcome: traj.errorCount > 0 ? 'errored' : 'completed',
511
+ outcome: deriveRunOutcome(traj),
427
512
  repo,
428
513
  agent: s.agent,
429
514
  model: s.model ?? 'unknown',