@phnx-labs/agents-cli 1.22.57 → 1.22.59
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +294 -0
- package/README.md +29 -0
- package/dist/bootstrap.js +39 -1
- package/dist/commands/accounts.js +7 -3
- package/dist/commands/apply.js +10 -2
- package/dist/commands/fork.d.ts +23 -10
- package/dist/commands/fork.js +115 -58
- package/dist/commands/monitors.js +198 -23
- package/dist/commands/prune.js +5 -3
- package/dist/commands/routines.d.ts +8 -0
- package/dist/commands/routines.js +57 -3
- package/dist/commands/routines.test-fixture.js +5 -0
- package/dist/commands/send.d.ts +2 -1
- package/dist/commands/send.js +7 -5
- package/dist/commands/sessions-picker.d.ts +11 -0
- package/dist/commands/sessions-picker.js +16 -0
- package/dist/commands/sessions-stats.js +37 -5
- package/dist/commands/sessions.js +40 -5
- package/dist/commands/share.d.ts +14 -0
- package/dist/commands/share.js +43 -2
- package/dist/commands/ssh.js +12 -1
- package/dist/commands/status.js +1 -1
- package/dist/commands/sync.js +83 -7
- package/dist/commands/traces.js +7 -0
- package/dist/commands/versions.js +12 -4
- package/dist/commands/view.js +7 -2
- package/dist/index.d.ts +1 -1
- package/dist/index.js +6 -1
- package/dist/lib/account-registry.d.ts +5 -1
- package/dist/lib/account-registry.js +47 -14
- package/dist/lib/accounting/capacity.d.ts +18 -7
- package/dist/lib/accounting/capacity.js +19 -8
- package/dist/lib/accounting/usage-sync.d.ts +29 -1
- package/dist/lib/accounting/usage-sync.js +76 -2
- package/dist/lib/accounting/usage.js +7 -1
- package/dist/lib/auth-mint.d.ts +11 -1
- package/dist/lib/auth-mint.js +21 -6
- package/dist/lib/auto-pull-worker.js +7 -2
- package/dist/lib/browser/ipc.d.ts +8 -0
- package/dist/lib/browser/ipc.js +87 -0
- package/dist/lib/browser/service.d.ts +19 -0
- package/dist/lib/browser/service.js +96 -11
- package/dist/lib/browser/sessions-list.js +10 -1
- package/dist/lib/cloud/rush.d.ts +7 -0
- package/dist/lib/cloud/rush.js +29 -1
- package/dist/lib/daemon/daemon.d.ts +22 -0
- package/dist/lib/daemon/daemon.js +39 -0
- package/dist/lib/daemon/runner.d.ts +3 -0
- package/dist/lib/daemon/runner.js +86 -45
- package/dist/lib/daemon/session-index-service.js +9 -1
- package/dist/lib/daemon/usage-sync-service.d.ts +3 -3
- package/dist/lib/daemon/usage-sync-service.js +14 -8
- package/dist/lib/daemon-services.js +1 -1
- package/dist/lib/daemon-ticks.d.ts +15 -0
- package/dist/lib/daemon-ticks.js +26 -0
- package/dist/lib/device-config.d.ts +5 -1
- package/dist/lib/device-config.js +2 -2
- package/dist/lib/devices/connect.d.ts +17 -8
- package/dist/lib/devices/connect.js +31 -14
- package/dist/lib/devices/health.js +5 -1
- package/dist/lib/devices/pool.d.ts +25 -2
- package/dist/lib/devices/pool.js +32 -2
- package/dist/lib/devices/stats-cache.d.ts +0 -6
- package/dist/lib/devices/stats-cache.js +2 -9
- package/dist/lib/doctor-diff.d.ts +14 -0
- package/dist/lib/doctor-diff.js +120 -9
- package/dist/lib/fleet/manifest.d.ts +17 -0
- package/dist/lib/fleet/manifest.js +26 -0
- package/dist/lib/git.d.ts +38 -0
- package/dist/lib/git.js +58 -0
- package/dist/lib/hooks/install.d.ts +27 -11
- package/dist/lib/hooks/install.js +42 -17
- package/dist/lib/hosts/ready.d.ts +8 -0
- package/dist/lib/hosts/ready.js +13 -2
- package/dist/lib/hosts/reconnect.d.ts +52 -203
- package/dist/lib/hosts/reconnect.js +64 -284
- package/dist/lib/installations/migrate.d.ts +6 -120
- package/dist/lib/installations/migrate.js +27 -259
- package/dist/lib/installations/shims.d.ts +13 -95
- package/dist/lib/installations/shims.js +22 -139
- package/dist/lib/installations/store.js +1 -1
- package/dist/lib/installations/versions.d.ts +43 -133
- package/dist/lib/installations/versions.js +94 -206
- package/dist/lib/monitors/config.d.ts +71 -3
- package/dist/lib/monitors/config.js +100 -12
- package/dist/lib/monitors/pid-watch.d.ts +35 -0
- package/dist/lib/monitors/pid-watch.js +45 -0
- package/dist/lib/monitors/remote.d.ts +18 -0
- package/dist/lib/monitors/remote.js +11 -0
- package/dist/lib/permissions.js +7 -2
- package/dist/lib/plugins/plugins.d.ts +17 -3
- package/dist/lib/plugins/plugins.js +84 -9
- package/dist/lib/plugins/skills.d.ts +8 -1
- package/dist/lib/plugins/skills.js +18 -2
- package/dist/lib/pty-server.d.ts +14 -0
- package/dist/lib/pty-server.js +49 -5
- package/dist/lib/refresh.d.ts +9 -0
- package/dist/lib/refresh.js +3 -1
- package/dist/lib/routine-readiness.d.ts +15 -1
- package/dist/lib/routine-readiness.js +41 -0
- package/dist/lib/sandbox.d.ts +4 -1
- package/dist/lib/sandbox.js +30 -1
- package/dist/lib/secrets/agent.d.ts +80 -225
- package/dist/lib/secrets/agent.js +139 -401
- package/dist/lib/secrets/bundles.d.ts +73 -222
- package/dist/lib/secrets/bundles.js +168 -467
- package/dist/lib/secrets/drivers/rush.js +5 -0
- package/dist/lib/secrets/reaper.d.ts +28 -70
- package/dist/lib/secrets/reaper.js +30 -85
- package/dist/lib/secrets/remote.d.ts +42 -129
- package/dist/lib/secrets/remote.js +55 -173
- package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
- package/dist/lib/self-heal/checks/install-staging.js +96 -0
- package/dist/lib/self-heal/registry.js +2 -0
- package/dist/lib/self-heal/types.d.ts +1 -1
- package/dist/lib/self-update.d.ts +65 -0
- package/dist/lib/self-update.js +138 -0
- package/dist/lib/session/active.d.ts +13 -1
- package/dist/lib/session/active.js +2 -0
- package/dist/lib/session/cloud.js +5 -0
- package/dist/lib/session/db.d.ts +51 -6
- package/dist/lib/session/db.js +266 -20
- package/dist/lib/session/fork.d.ts +45 -26
- package/dist/lib/session/fork.js +32 -95
- package/dist/lib/session/tool-calls.d.ts +43 -1
- package/dist/lib/session/tool-calls.js +74 -44
- package/dist/lib/session/tool-store.d.ts +33 -2
- package/dist/lib/session/tool-store.js +56 -3
- package/dist/lib/smart-launch.d.ts +6 -0
- package/dist/lib/smart-launch.js +5 -2
- package/dist/lib/staleness/writers/plugins.js +5 -2
- package/dist/lib/staleness/writers/sources.d.ts +5 -0
- package/dist/lib/staleness/writers/sources.js +2 -1
- package/dist/lib/staleness/writers/subagents.js +13 -3
- package/dist/lib/state.d.ts +7 -4
- package/dist/lib/state.js +7 -4
- package/dist/lib/subagents.js +8 -2
- package/dist/lib/sync-status.d.ts +22 -0
- package/dist/lib/sync-status.js +27 -0
- package/dist/lib/sync-umbrella.d.ts +9 -0
- package/dist/lib/sync-umbrella.js +21 -2
- package/dist/lib/teams/scheduler.d.ts +10 -0
- package/dist/lib/teams/scheduler.js +8 -0
- package/dist/lib/traces/insights.d.ts +47 -14
- package/dist/lib/traces/insights.js +92 -21
- package/dist/lib/traces/phenotype.d.ts +23 -3
- package/dist/lib/traces/phenotype.js +72 -24
- package/dist/lib/traces/sync.d.ts +128 -6
- package/dist/lib/traces/sync.js +294 -35
- package/dist/lib/traces/worker-template.js +154 -1
- package/dist/lib/view-types.d.ts +12 -0
- package/package.json +2 -2
|
@@ -48,6 +48,16 @@ export interface PlacementOptions {
|
|
|
48
48
|
/** Human label of the requested agent (e.g. `claude@2.1.112`) for the
|
|
49
49
|
* fail-loud message. */
|
|
50
50
|
agentLabel?: string;
|
|
51
|
+
/**
|
|
52
|
+
* Normalized hosts boosted with `auto-launch.preferred` (set by
|
|
53
|
+
* `agents devices prefer <name>`). A preferred device ranks ahead of a
|
|
54
|
+
* non-preferred one among the eligible survivors — after the signed-in tier
|
|
55
|
+
* (a preferred box that can't run the agent is still no use) and before load,
|
|
56
|
+
* so an operator boost overrides load-based ordering without overriding hard
|
|
57
|
+
* health. Empty/undefined leaves the ranking unchanged. See
|
|
58
|
+
* {@link autoLaunchPreferredSet}.
|
|
59
|
+
*/
|
|
60
|
+
preferred?: ReadonlySet<string>;
|
|
51
61
|
}
|
|
52
62
|
/** Why a device was excluded from the viable set, for the fail-loud message. */
|
|
53
63
|
export type ExclusionReason = 'unreachable' | 'overloaded' | 'capped' | 'not-installed';
|
|
@@ -266,6 +266,14 @@ export function pickBestDevice(devices, roster, opts) {
|
|
|
266
266
|
const signedIn = (sa?.signedIn === true ? 0 : 1) - (sb?.signedIn === true ? 0 : 1);
|
|
267
267
|
if (signedIn !== 0)
|
|
268
268
|
return signedIn;
|
|
269
|
+
// (a2) operator-preferred device next — `agents devices prefer <name>`
|
|
270
|
+
// boosts a box above its load-equal peers, overriding load-based order.
|
|
271
|
+
const preferred = opts?.preferred;
|
|
272
|
+
if (preferred && preferred.size > 0) {
|
|
273
|
+
const pref = (preferred.has(a) ? 0 : 1) - (preferred.has(b) ? 0 : 1);
|
|
274
|
+
if (pref !== 0)
|
|
275
|
+
return pref;
|
|
276
|
+
}
|
|
269
277
|
// (b) lower load — coarse headroom tier, then raw load cost.
|
|
270
278
|
const tier = headroomTier(sa?.headroom) - headroomTier(sb?.headroom);
|
|
271
279
|
if (tier !== 0)
|
|
@@ -12,14 +12,21 @@
|
|
|
12
12
|
* is enough to reconstruct per-session call order and inter-call gaps without a
|
|
13
13
|
* full `SessionTrajectory` — that is what makes this incremental at scale.
|
|
14
14
|
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
* needs the full derived trajectory
|
|
18
|
-
*
|
|
19
|
-
* `
|
|
20
|
-
* `
|
|
15
|
+
* Failure phenotype (false-termination / out-of-order / premature-completion /
|
|
16
|
+
* failure-to-act, `phenotype.ts`) is folded in as a fourth grouping dimension
|
|
17
|
+
* (PHNX-3327). Classifying it needs the full derived trajectory, so it is NOT
|
|
18
|
+
* computed here — the caller passes a per-session `phenotypes` map that
|
|
19
|
+
* `buildIndexShard` fills from the persisted, mtime+size-keyed
|
|
20
|
+
* `session_phenotypes` cache for the WHOLE corpus. Keying the group on the
|
|
21
|
+
* cached-per-session phenotype (never on this run's incremental batch) is what
|
|
22
|
+
* keeps two identically-signatured sessions in one cluster regardless of when
|
|
23
|
+
* each was synced. The `signature` OUTPUT is unchanged (`{ tool, cause, key }`);
|
|
24
|
+
* phenotype is an added dimension carried alongside it, so callers that never
|
|
25
|
+
* pass a map (the unit tests, a pre-phenotype caller) see the exact prior
|
|
26
|
+
* grouping.
|
|
21
27
|
*/
|
|
22
28
|
import { type TraceFailureCause } from './classify.js';
|
|
29
|
+
import type { FailurePhenotype } from './phenotype.js';
|
|
23
30
|
import { type LatencyInsight } from './segments.js';
|
|
24
31
|
import { type SyncRow, type ToolCallRow, type TracesIndexShard } from './sync.js';
|
|
25
32
|
export interface FailureSignature {
|
|
@@ -29,10 +36,17 @@ export interface FailureSignature {
|
|
|
29
36
|
key: string;
|
|
30
37
|
}
|
|
31
38
|
export interface FailurePattern {
|
|
32
|
-
/** Stable hash of the signature — deep-linkable, unaffected by row order. */
|
|
39
|
+
/** Stable hash of the signature (incl. phenotype) — deep-linkable, unaffected by row order. */
|
|
33
40
|
id: string;
|
|
34
41
|
label: string;
|
|
35
42
|
signature: FailureSignature;
|
|
43
|
+
/**
|
|
44
|
+
* Dominant failure phenotype of the sessions in this cluster, or `null` when
|
|
45
|
+
* none was classifiable. A fourth grouping dimension (PHNX-3327): two failures
|
|
46
|
+
* with the same `(tool, cause, key)` but different phenotypes are distinct
|
|
47
|
+
* patterns.
|
|
48
|
+
*/
|
|
49
|
+
phenotype: FailurePhenotype | null;
|
|
36
50
|
/** Distinct sessions this pattern occurred in. */
|
|
37
51
|
sessions: number;
|
|
38
52
|
/** Total failing calls matching this signature. */
|
|
@@ -57,11 +71,30 @@ export declare function normalizeErrorKey(desc: string, raw: string | null): str
|
|
|
57
71
|
* Cluster failed tool calls into ranked patterns and estimate the wasted time
|
|
58
72
|
* behind each, plus device-wide time-to-first-tool latency.
|
|
59
73
|
*
|
|
60
|
-
* wastedMs attribution
|
|
61
|
-
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
74
|
+
* wastedMs attribution has two parts that sum:
|
|
75
|
+
*
|
|
76
|
+
* (1) The failed call's OWN blocking duration — `end_timestamp - timestamp`
|
|
77
|
+
* (PHNX-3437). A call that hung for minutes and then failed wasted that whole
|
|
78
|
+
* time even if it was the last call in its session or was followed quickly by an
|
|
79
|
+
* unrelated call — the case the gap heuristic alone booked as ~0. This is what
|
|
80
|
+
* makes a fail-fast fix measurable: a channel that stops hanging 5.5min on stdin
|
|
81
|
+
* and instead fails in <1s (PHNX-3407) drops from ~5.5min of attributed waste to
|
|
82
|
+
* ~0. Bounded by MAX_GAP_ATTRIBUTION_MS so a corrupt end timestamp can't dominate.
|
|
83
|
+
*
|
|
84
|
+
* (2) The idle gap AFTER the call, before the NEXT call in the same session,
|
|
85
|
+
* counted when either (a) the next call repeats the same signature (a retry
|
|
86
|
+
* loop) or (b) the gap is a stall (≥60s) before an unrelated next call — each
|
|
87
|
+
* bounded by MAX_GAP_ATTRIBUTION_MS so one failure can't absorb hours of
|
|
88
|
+
* human-away idle in an async channel session, whether as a lone stall or a
|
|
89
|
+
* same-signature re-ask hours later; a genuine active retry loop is many short
|
|
90
|
+
* gaps that each clear the cap and still sum large. When the end timestamp is
|
|
91
|
+
* known, this gap is measured from the call's END, so the blocking time counted
|
|
92
|
+
* in (1) is never double-counted; for a NULL end (rows an older extractor stored,
|
|
93
|
+
* or a call still pending at scan end) it falls back to the original
|
|
94
|
+
* gap-from-START heuristic unchanged — no crash, no NaN, no regression.
|
|
95
|
+
*
|
|
96
|
+
* An idle gap unrelated to a nearby failure is never counted. This is an
|
|
97
|
+
* estimate, not ground truth; it is not inflated by folding in ordinary
|
|
98
|
+
* processing time between unrelated calls.
|
|
66
99
|
*/
|
|
67
|
-
export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null): ComputedInsights;
|
|
100
|
+
export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null, phenotypes?: ReadonlyMap<string, FailurePhenotype | null>): ComputedInsights;
|
|
@@ -12,12 +12,18 @@
|
|
|
12
12
|
* is enough to reconstruct per-session call order and inter-call gaps without a
|
|
13
13
|
* full `SessionTrajectory` — that is what makes this incremental at scale.
|
|
14
14
|
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
* needs the full derived trajectory
|
|
18
|
-
*
|
|
19
|
-
* `
|
|
20
|
-
* `
|
|
15
|
+
* Failure phenotype (false-termination / out-of-order / premature-completion /
|
|
16
|
+
* failure-to-act, `phenotype.ts`) is folded in as a fourth grouping dimension
|
|
17
|
+
* (PHNX-3327). Classifying it needs the full derived trajectory, so it is NOT
|
|
18
|
+
* computed here — the caller passes a per-session `phenotypes` map that
|
|
19
|
+
* `buildIndexShard` fills from the persisted, mtime+size-keyed
|
|
20
|
+
* `session_phenotypes` cache for the WHOLE corpus. Keying the group on the
|
|
21
|
+
* cached-per-session phenotype (never on this run's incremental batch) is what
|
|
22
|
+
* keeps two identically-signatured sessions in one cluster regardless of when
|
|
23
|
+
* each was synced. The `signature` OUTPUT is unchanged (`{ tool, cause, key }`);
|
|
24
|
+
* phenotype is an added dimension carried alongside it, so callers that never
|
|
25
|
+
* pass a map (the unit tests, a pre-phenotype caller) see the exact prior
|
|
26
|
+
* grouping.
|
|
21
27
|
*/
|
|
22
28
|
import { classifyCause } from './classify.js';
|
|
23
29
|
import { computeLatency } from './segments.js';
|
|
@@ -30,6 +36,24 @@ const TOP_K_PATTERNS = 25;
|
|
|
30
36
|
const MAX_EXAMPLE_SESSIONS = 5;
|
|
31
37
|
/** A gap this long right after a failure reads as an idle stall, not think-time (matches sync.ts's own "stalled Xm" threshold). */
|
|
32
38
|
const STALL_MS = 60_000;
|
|
39
|
+
/**
|
|
40
|
+
* Upper bound on how much of a SINGLE inter-call gap is attributable to a
|
|
41
|
+
* failure. Beyond a recovery window a gap is not failure-loop waste — it is a
|
|
42
|
+
* human away from a chat thread, an abandoned session, or an outage. This bounds
|
|
43
|
+
* BOTH branches, and that matters: `nextIsSameFailure` only compares
|
|
44
|
+
* `(tool, cause, normalized-key)`, with no temporal check, so a deterministic
|
|
45
|
+
* failure that recurs identically hours apart (a permanently-denied capability,
|
|
46
|
+
* a missing credential — e.g. a Slack user re-asking "where am I" at 2pm and
|
|
47
|
+
* 6pm) would otherwise look like an "active retry loop" and absorb the whole
|
|
48
|
+
* multi-hour gap — the very artifact this fix targets. A genuine active loop has
|
|
49
|
+
* MANY short gaps that each stay under this cap and still sum to a large total,
|
|
50
|
+
* so bounding a single gap doesn't hide it. Real single-tool stalls (a hung
|
|
51
|
+
* typecheck, a slow build) are minutes and stay fully counted.
|
|
52
|
+
*
|
|
53
|
+
* The same cap bounds the call's OWN blocking duration (PHNX-3437) so a corrupt
|
|
54
|
+
* or backwards end timestamp can't book a single call as hours of waste either.
|
|
55
|
+
*/
|
|
56
|
+
const MAX_GAP_ATTRIBUTION_MS = 30 * 60_000;
|
|
33
57
|
// ---------------------------------------------------------------------------
|
|
34
58
|
// Signature normalization — fold volatile per-instance text together
|
|
35
59
|
// ---------------------------------------------------------------------------
|
|
@@ -48,8 +72,8 @@ export function normalizeErrorKey(desc, raw) {
|
|
|
48
72
|
}
|
|
49
73
|
return text.replace(/\s+/g, ' ').trim().slice(0, 160);
|
|
50
74
|
}
|
|
51
|
-
function hashSignature(tool, cause, key) {
|
|
52
|
-
const input = `${tool} ${cause} ${key}`;
|
|
75
|
+
function hashSignature(tool, cause, key, phenotype) {
|
|
76
|
+
const input = `${tool} ${cause} ${key} ${phenotype ?? ''}`;
|
|
53
77
|
let hash = 5381;
|
|
54
78
|
for (let i = 0; i < input.length; i++) {
|
|
55
79
|
hash = ((hash << 5) + hash + input.charCodeAt(i)) >>> 0;
|
|
@@ -82,14 +106,33 @@ function labelFor(tool, cause, key) {
|
|
|
82
106
|
* Cluster failed tool calls into ranked patterns and estimate the wasted time
|
|
83
107
|
* behind each, plus device-wide time-to-first-tool latency.
|
|
84
108
|
*
|
|
85
|
-
* wastedMs attribution
|
|
86
|
-
*
|
|
87
|
-
*
|
|
88
|
-
*
|
|
89
|
-
*
|
|
90
|
-
*
|
|
109
|
+
* wastedMs attribution has two parts that sum:
|
|
110
|
+
*
|
|
111
|
+
* (1) The failed call's OWN blocking duration — `end_timestamp - timestamp`
|
|
112
|
+
* (PHNX-3437). A call that hung for minutes and then failed wasted that whole
|
|
113
|
+
* time even if it was the last call in its session or was followed quickly by an
|
|
114
|
+
* unrelated call — the case the gap heuristic alone booked as ~0. This is what
|
|
115
|
+
* makes a fail-fast fix measurable: a channel that stops hanging 5.5min on stdin
|
|
116
|
+
* and instead fails in <1s (PHNX-3407) drops from ~5.5min of attributed waste to
|
|
117
|
+
* ~0. Bounded by MAX_GAP_ATTRIBUTION_MS so a corrupt end timestamp can't dominate.
|
|
118
|
+
*
|
|
119
|
+
* (2) The idle gap AFTER the call, before the NEXT call in the same session,
|
|
120
|
+
* counted when either (a) the next call repeats the same signature (a retry
|
|
121
|
+
* loop) or (b) the gap is a stall (≥60s) before an unrelated next call — each
|
|
122
|
+
* bounded by MAX_GAP_ATTRIBUTION_MS so one failure can't absorb hours of
|
|
123
|
+
* human-away idle in an async channel session, whether as a lone stall or a
|
|
124
|
+
* same-signature re-ask hours later; a genuine active retry loop is many short
|
|
125
|
+
* gaps that each clear the cap and still sum large. When the end timestamp is
|
|
126
|
+
* known, this gap is measured from the call's END, so the blocking time counted
|
|
127
|
+
* in (1) is never double-counted; for a NULL end (rows an older extractor stored,
|
|
128
|
+
* or a call still pending at scan end) it falls back to the original
|
|
129
|
+
* gap-from-START heuristic unchanged — no crash, no NaN, no regression.
|
|
130
|
+
*
|
|
131
|
+
* An idle gap unrelated to a nearby failure is never counted. This is an
|
|
132
|
+
* estimate, not ground truth; it is not inflated by folding in ordinary
|
|
133
|
+
* processing time between unrelated calls.
|
|
91
134
|
*/
|
|
92
|
-
export function computeInsights(rows, calls, prevShard) {
|
|
135
|
+
export function computeInsights(rows, calls, prevShard, phenotypes) {
|
|
93
136
|
const bySession = new Map();
|
|
94
137
|
for (const call of calls) {
|
|
95
138
|
const list = bySession.get(call.session_id);
|
|
@@ -100,6 +143,11 @@ export function computeInsights(rows, calls, prevShard) {
|
|
|
100
143
|
}
|
|
101
144
|
const groups = new Map();
|
|
102
145
|
for (const [sessionId, sessionCalls] of bySession) {
|
|
146
|
+
// Phenotype is a per-session property (one classification per session), so
|
|
147
|
+
// every failing call in this session shares it. A caller that passes no map
|
|
148
|
+
// (unit tests, a pre-phenotype caller) collapses the dimension to `null`,
|
|
149
|
+
// yielding the exact prior grouping.
|
|
150
|
+
const phenotype = phenotypes?.get(sessionId) ?? null;
|
|
103
151
|
const ordered = [...sessionCalls].sort((a, b) => a.ordinal - b.ordinal);
|
|
104
152
|
for (let i = 0; i < ordered.length; i++) {
|
|
105
153
|
const call = ordered[i];
|
|
@@ -107,10 +155,10 @@ export function computeInsights(rows, calls, prevShard) {
|
|
|
107
155
|
continue;
|
|
108
156
|
const cause = classifyCause(call);
|
|
109
157
|
const key = normalizeErrorKey(failureDescription(call, cause), call.error);
|
|
110
|
-
const groupKey = `${call.tool} ${cause} ${key}`;
|
|
158
|
+
const groupKey = `${call.tool} ${cause} ${key} ${phenotype ?? ''}`;
|
|
111
159
|
let group = groups.get(groupKey);
|
|
112
160
|
if (!group) {
|
|
113
|
-
group = { tool: call.tool, cause, key, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
|
|
161
|
+
group = { tool: call.tool, cause, key, phenotype, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
|
|
114
162
|
groups.set(groupKey, group);
|
|
115
163
|
}
|
|
116
164
|
group.occurrences++;
|
|
@@ -118,24 +166,46 @@ export function computeInsights(rows, calls, prevShard) {
|
|
|
118
166
|
if (group.examples.length < MAX_EXAMPLE_SESSIONS && !group.examples.includes(sessionId)) {
|
|
119
167
|
group.examples.push(sessionId);
|
|
120
168
|
}
|
|
169
|
+
// (1) The call's OWN blocking duration (end minus start) is the primary
|
|
170
|
+
// signal — see the computeInsights docblock. Attributed whenever the end
|
|
171
|
+
// timestamp is present, independent of whether a next call follows.
|
|
172
|
+
const startMs = Date.parse(call.timestamp);
|
|
173
|
+
const endMs = call.end_timestamp ? Date.parse(call.end_timestamp) : NaN;
|
|
174
|
+
const hasEnd = Number.isFinite(endMs) && Number.isFinite(startMs);
|
|
175
|
+
if (hasEnd) {
|
|
176
|
+
const ownMs = endMs - startMs;
|
|
177
|
+
if (ownMs > 0)
|
|
178
|
+
group.wastedMs += Math.min(ownMs, MAX_GAP_ATTRIBUTION_MS);
|
|
179
|
+
}
|
|
180
|
+
// (2) The idle gap after the call, before the next one. Measured from the
|
|
181
|
+
// call's END when known (so the blocking time in (1) isn't double-counted),
|
|
182
|
+
// else from its START — the original heuristic, unchanged for NULL ends.
|
|
121
183
|
const next = ordered[i + 1];
|
|
122
184
|
if (!next)
|
|
123
185
|
continue;
|
|
124
|
-
const
|
|
186
|
+
const gapFromMs = hasEnd ? endMs : startMs;
|
|
187
|
+
const gapMs = Date.parse(next.timestamp) - gapFromMs;
|
|
125
188
|
if (!Number.isFinite(gapMs) || gapMs <= 0)
|
|
126
189
|
continue;
|
|
127
190
|
const nextIsSameFailure = next.outcome === 'error' &&
|
|
128
191
|
next.tool === call.tool &&
|
|
129
192
|
classifyCause(next) === cause &&
|
|
130
193
|
normalizeErrorKey(failureDescription(next, cause), next.error) === key;
|
|
131
|
-
if (nextIsSameFailure
|
|
132
|
-
|
|
194
|
+
if (nextIsSameFailure) {
|
|
195
|
+
// Retry loop: a real active loop is many short gaps, each under the cap,
|
|
196
|
+
// summing to a large total. Bound a single gap so one huge same-signature
|
|
197
|
+
// gap (a re-ask hours later, not active retrying) can't absorb it all.
|
|
198
|
+
group.wastedMs += Math.min(gapMs, MAX_GAP_ATTRIBUTION_MS);
|
|
199
|
+
}
|
|
200
|
+
else if (gapMs >= STALL_MS) {
|
|
201
|
+
// Lone stall before an unrelated next call: same bounded recovery window.
|
|
202
|
+
group.wastedMs += Math.min(gapMs, MAX_GAP_ATTRIBUTION_MS);
|
|
133
203
|
}
|
|
134
204
|
}
|
|
135
205
|
}
|
|
136
206
|
const prevById = new Map((prevShard?.failurePatterns ?? []).map((p) => [p.id, p]));
|
|
137
207
|
const allPatterns = [...groups.values()].map((group) => {
|
|
138
|
-
const id = hashSignature(group.tool, group.cause, group.key);
|
|
208
|
+
const id = hashSignature(group.tool, group.cause, group.key, group.phenotype);
|
|
139
209
|
const prev = prevById.get(id);
|
|
140
210
|
const drift = !prev
|
|
141
211
|
? 'up'
|
|
@@ -148,6 +218,7 @@ export function computeInsights(rows, calls, prevShard) {
|
|
|
148
218
|
id,
|
|
149
219
|
label: labelFor(group.tool, group.cause, group.key),
|
|
150
220
|
signature: { tool: group.tool, cause: group.cause, key: group.key },
|
|
221
|
+
phenotype: group.phenotype,
|
|
151
222
|
sessions: group.sessions.size,
|
|
152
223
|
occurrences: group.occurrences,
|
|
153
224
|
wastedMs: group.wastedMs,
|
|
@@ -35,13 +35,33 @@ export interface OutcomeResult {
|
|
|
35
35
|
confidence: 'high' | 'medium' | 'low';
|
|
36
36
|
reason: string;
|
|
37
37
|
}
|
|
38
|
+
/**
|
|
39
|
+
* Did a run that hit tool errors nonetheless *recover and finish*?
|
|
40
|
+
*
|
|
41
|
+
* The causal recovery test: a **substantive** tool step (not a human-facing
|
|
42
|
+
* `AskUserQuestion` / `SendMessage` / `wait`) SUCCEEDED strictly AFTER the last
|
|
43
|
+
* error's ordinal, AND that success resolves the failed work — its
|
|
44
|
+
* {@link workSignature} matches an errored step's. A later success of *unrelated*
|
|
45
|
+
* work does NOT count: a `bun test` failure followed by an incidental `ls` leaves
|
|
46
|
+
* the failed test unresolved, so the run stays errored, even though the `ls`
|
|
47
|
+
* succeeded after it. A punt to a human is excluded twice over — human-facing
|
|
48
|
+
* tools are outside the substantive set, and they never match a failed signature.
|
|
49
|
+
*
|
|
50
|
+
* This is the single source of truth for "did this finish", shared by the
|
|
51
|
+
* false-termination phenotype below and `sync.ts`'s `deriveRunOutcome` — so a
|
|
52
|
+
* run's console outcome and its failure phenotype can never disagree about
|
|
53
|
+
* whether it recovered. It reads only the derived steps, never `meta.outcome`,
|
|
54
|
+
* precisely so `deriveRunOutcome` can call it while it is still *computing*
|
|
55
|
+
* `meta.outcome`.
|
|
56
|
+
*/
|
|
57
|
+
export declare function recoveredAfterErrors(session: Pick<SessionDetail, 'steps'>): boolean;
|
|
38
58
|
/**
|
|
39
59
|
* Classify the failure phenotype of a session from its derived trajectory.
|
|
40
60
|
*
|
|
41
61
|
* Definitions (from agent-failure research):
|
|
42
|
-
* - `false-termination` — stopped with an unresolved error.
|
|
43
|
-
* - `premature-completion` — declared done
|
|
44
|
-
*
|
|
62
|
+
* - `false-termination` — stopped with an unresolved error (did not causally recover).
|
|
63
|
+
* - `premature-completion` — declared done with no verification step for the
|
|
64
|
+
* engineering work.
|
|
45
65
|
* - `out-of-order` — a write/edit step occurred before any read/plan of the
|
|
46
66
|
* target.
|
|
47
67
|
* - `failure-to-act` — stalled or produced no meaningful tool use.
|
|
@@ -156,30 +156,75 @@ function reasonOutOfOrder(session) {
|
|
|
156
156
|
const tool = session.steps.find((s) => WRITE_EDIT_TOOLS.has(s.tool ?? s.lane));
|
|
157
157
|
return `write/edit step ${tool?.tool ?? tool?.lane ?? ''} preceded any read/plan`;
|
|
158
158
|
}
|
|
159
|
+
/**
|
|
160
|
+
* The **work signature** of a step — what work it represents, so a later success
|
|
161
|
+
* can be matched back to the specific failure it resolves. For a shell step this
|
|
162
|
+
* is the effective program (`bun`, `git`, `gh`, …, from `TrajectoryStep.program`),
|
|
163
|
+
* so a failed `bun test` is resolved by a later `bun test` but NOT by an incidental
|
|
164
|
+
* `ls`. For any other tool it is the tool identity, so a failed `Edit` is resolved
|
|
165
|
+
* by a later successful `Edit`, not by an unrelated `Read`. A shell step whose
|
|
166
|
+
* command did not parse (no `program`) degrades to the bare tool name, matching
|
|
167
|
+
* the pre-`program` behavior only for that unparseable minority.
|
|
168
|
+
*/
|
|
169
|
+
function workSignature(step) {
|
|
170
|
+
const tool = step.tool ?? step.lane;
|
|
171
|
+
if (SHELL_TOOLS.has(tool) && step.program)
|
|
172
|
+
return `${tool}:${step.program}`;
|
|
173
|
+
return tool;
|
|
174
|
+
}
|
|
175
|
+
/**
|
|
176
|
+
* Did a run that hit tool errors nonetheless *recover and finish*?
|
|
177
|
+
*
|
|
178
|
+
* The causal recovery test: a **substantive** tool step (not a human-facing
|
|
179
|
+
* `AskUserQuestion` / `SendMessage` / `wait`) SUCCEEDED strictly AFTER the last
|
|
180
|
+
* error's ordinal, AND that success resolves the failed work — its
|
|
181
|
+
* {@link workSignature} matches an errored step's. A later success of *unrelated*
|
|
182
|
+
* work does NOT count: a `bun test` failure followed by an incidental `ls` leaves
|
|
183
|
+
* the failed test unresolved, so the run stays errored, even though the `ls`
|
|
184
|
+
* succeeded after it. A punt to a human is excluded twice over — human-facing
|
|
185
|
+
* tools are outside the substantive set, and they never match a failed signature.
|
|
186
|
+
*
|
|
187
|
+
* This is the single source of truth for "did this finish", shared by the
|
|
188
|
+
* false-termination phenotype below and `sync.ts`'s `deriveRunOutcome` — so a
|
|
189
|
+
* run's console outcome and its failure phenotype can never disagree about
|
|
190
|
+
* whether it recovered. It reads only the derived steps, never `meta.outcome`,
|
|
191
|
+
* precisely so `deriveRunOutcome` can call it while it is still *computing*
|
|
192
|
+
* `meta.outcome`.
|
|
193
|
+
*/
|
|
194
|
+
export function recoveredAfterErrors(session) {
|
|
195
|
+
const substantive = substantiveSteps(session);
|
|
196
|
+
if (substantive.length === 0)
|
|
197
|
+
return false;
|
|
198
|
+
const last = substantive[substantive.length - 1];
|
|
199
|
+
if (last.outcome === 'error')
|
|
200
|
+
return false;
|
|
201
|
+
const lastErrorOrdinal = lastStepOrdinalOf(session, (s) => s.outcome === 'error');
|
|
202
|
+
if (lastErrorOrdinal === undefined)
|
|
203
|
+
return true;
|
|
204
|
+
const failedSignatures = new Set();
|
|
205
|
+
for (const s of session.steps) {
|
|
206
|
+
if (s.outcome === 'error')
|
|
207
|
+
failedSignatures.add(workSignature(s));
|
|
208
|
+
}
|
|
209
|
+
return substantive.some((s) => s.ordinal > lastErrorOrdinal &&
|
|
210
|
+
s.outcome === 'ok' &&
|
|
211
|
+
!HUMAN_FACING_TOOLS.has(s.tool ?? s.lane) &&
|
|
212
|
+
failedSignatures.has(workSignature(s)));
|
|
213
|
+
}
|
|
159
214
|
/**
|
|
160
215
|
* False-termination: the session stopped with an unresolved error.
|
|
161
216
|
*
|
|
162
217
|
* Rubric:
|
|
163
218
|
* - meta outcome is `errored`, and
|
|
164
219
|
* - at least one step outcome is `error`, and
|
|
165
|
-
* - the
|
|
220
|
+
* - the run did not causally recover ({@link recoveredAfterErrors}).
|
|
166
221
|
*/
|
|
167
222
|
function isFalseTermination(session) {
|
|
168
223
|
if (session.meta.outcome !== 'errored')
|
|
169
224
|
return false;
|
|
170
225
|
if (session.meta.errorCount === 0)
|
|
171
226
|
return false;
|
|
172
|
-
|
|
173
|
-
if (substantive.length === 0)
|
|
174
|
-
return true;
|
|
175
|
-
const last = substantive[substantive.length - 1];
|
|
176
|
-
if (last.outcome === 'error')
|
|
177
|
-
return true;
|
|
178
|
-
const lastErrorOrdinal = lastStepOrdinalOf(session, (s) => s.outcome === 'error');
|
|
179
|
-
if (lastErrorOrdinal === undefined)
|
|
180
|
-
return false;
|
|
181
|
-
const recoveryAfter = substantive.some((s) => s.ordinal > lastErrorOrdinal && s.outcome === 'ok' && !HUMAN_FACING_TOOLS.has(s.tool ?? s.lane));
|
|
182
|
-
return !recoveryAfter;
|
|
227
|
+
return !recoveredAfterErrors(session);
|
|
183
228
|
}
|
|
184
229
|
function reasonFalseTermination(session) {
|
|
185
230
|
const substantive = substantiveSteps(session);
|
|
@@ -192,13 +237,21 @@ function reasonFalseTermination(session) {
|
|
|
192
237
|
return 'errored with no successful recovery after the last error';
|
|
193
238
|
}
|
|
194
239
|
/**
|
|
195
|
-
* Premature-completion: declared done
|
|
196
|
-
*
|
|
240
|
+
* Premature-completion: declared done without a verification step for an
|
|
241
|
+
* engineering task.
|
|
197
242
|
*
|
|
198
243
|
* Rubric:
|
|
199
244
|
* - meta outcome is `completed`, and
|
|
200
245
|
* - the session performed write/edit work, and
|
|
201
|
-
* -
|
|
246
|
+
* - no test/build/lint verification step ran.
|
|
247
|
+
*
|
|
248
|
+
* `errorCount` is deliberately NOT a signal here. Under truthful outcomes
|
|
249
|
+
* (PHNX-3387) a `completed` run CAN carry `errorCount > 0` — that is precisely a
|
|
250
|
+
* recover-then-succeed run, where the errors were *resolved*, not left hanging.
|
|
251
|
+
* Keying prematurity off `errorCount > 0` would mislabel every such recovery as
|
|
252
|
+
* premature. (Under the pre-PHNX-3387 `errorCount > 0 ? errored : completed`
|
|
253
|
+
* derivation this branch was unreachable — a `completed` run always had
|
|
254
|
+
* `errorCount === 0` — so dropping it changes nothing for a clean-completed run.)
|
|
202
255
|
*/
|
|
203
256
|
function isPrematureCompletion(session) {
|
|
204
257
|
if (session.meta.outcome !== 'completed')
|
|
@@ -206,8 +259,6 @@ function isPrematureCompletion(session) {
|
|
|
206
259
|
const didWriteEdit = session.steps.some((s) => WRITE_EDIT_TOOLS.has(s.tool ?? s.lane));
|
|
207
260
|
if (!didWriteEdit)
|
|
208
261
|
return false;
|
|
209
|
-
if (session.meta.errorCount > 0)
|
|
210
|
-
return true;
|
|
211
262
|
const verified = session.steps.some((s) => {
|
|
212
263
|
if (!SHELL_TOOLS.has(s.tool ?? s.lane))
|
|
213
264
|
return false;
|
|
@@ -216,10 +267,7 @@ function isPrematureCompletion(session) {
|
|
|
216
267
|
});
|
|
217
268
|
return !verified;
|
|
218
269
|
}
|
|
219
|
-
function reasonPrematureCompletion(
|
|
220
|
-
if (session.meta.errorCount > 0) {
|
|
221
|
-
return `declared completed with ${session.meta.errorCount} unresolved error(s)`;
|
|
222
|
-
}
|
|
270
|
+
function reasonPrematureCompletion(_session) {
|
|
223
271
|
return 'engineering work completed without a test/build/lint verification step';
|
|
224
272
|
}
|
|
225
273
|
/** Ordered rubric: the first matching phenotype wins. */
|
|
@@ -391,9 +439,9 @@ const OUTCOME_RULES = [
|
|
|
391
439
|
* Classify the failure phenotype of a session from its derived trajectory.
|
|
392
440
|
*
|
|
393
441
|
* Definitions (from agent-failure research):
|
|
394
|
-
* - `false-termination` — stopped with an unresolved error.
|
|
395
|
-
* - `premature-completion` — declared done
|
|
396
|
-
*
|
|
442
|
+
* - `false-termination` — stopped with an unresolved error (did not causally recover).
|
|
443
|
+
* - `premature-completion` — declared done with no verification step for the
|
|
444
|
+
* engineering work.
|
|
397
445
|
* - `out-of-order` — a write/edit step occurred before any read/plan of the
|
|
398
446
|
* target.
|
|
399
447
|
* - `failure-to-act` — stalled or produced no meaningful tool use.
|