@phnx-labs/agents-cli 1.22.51 → 1.22.53
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +238 -0
- package/README.md +1 -1
- package/dist/commands/accounts.js +1 -1
- package/dist/commands/attach.js +7 -0
- package/dist/commands/browser.js +118 -56
- package/dist/commands/daemon.d.ts +2 -0
- package/dist/commands/daemon.js +8 -4
- package/dist/commands/detach.js +1 -1
- package/dist/commands/exec.js +16 -9
- package/dist/commands/fleet-capture.js +7 -0
- package/dist/commands/focus.d.ts +1 -10
- package/dist/commands/focus.js +16 -79
- package/dist/commands/go.d.ts +26 -0
- package/dist/commands/go.js +65 -6
- package/dist/commands/monitors.js +1 -1
- package/dist/commands/repo.js +31 -3
- package/dist/commands/sessions-inject.js +8 -3
- package/dist/commands/sessions-picker.js +2 -1
- package/dist/commands/sessions-resume.d.ts +1 -0
- package/dist/commands/sessions-resume.js +13 -2
- package/dist/commands/sessions-stop.js +1 -1
- package/dist/commands/sessions.d.ts +23 -13
- package/dist/commands/sessions.js +69 -39
- package/dist/commands/setup-browser.d.ts +5 -2
- package/dist/commands/setup-browser.js +14 -29
- package/dist/commands/setup-preferences.d.ts +22 -3
- package/dist/commands/setup-preferences.js +25 -8
- package/dist/commands/share.js +12 -8
- package/dist/commands/ssh.js +35 -12
- package/dist/commands/status.js +5 -0
- package/dist/commands/sync.js +102 -2
- package/dist/commands/tmux.d.ts +8 -1
- package/dist/commands/tmux.js +167 -17
- package/dist/lib/account-registry.d.ts +15 -5
- package/dist/lib/account-registry.js +150 -50
- package/dist/lib/answer-router.js +2 -1
- package/dist/lib/browser/ipc.d.ts +44 -0
- package/dist/lib/browser/ipc.js +120 -8
- package/dist/lib/browser/profiles.d.ts +57 -17
- package/dist/lib/browser/profiles.js +77 -53
- package/dist/lib/browser/registry.d.ts +44 -14
- package/dist/lib/browser/registry.js +141 -45
- package/dist/lib/browser/runtime-state.d.ts +4 -2
- package/dist/lib/browser/runtime-state.js +4 -2
- package/dist/lib/browser/service.js +4 -3
- package/dist/lib/channels/owner-forward.d.ts +88 -0
- package/dist/lib/channels/owner-forward.js +116 -0
- package/dist/lib/channels/owner-sink.js +7 -0
- package/dist/lib/daemon/runner.js +10 -2
- package/dist/lib/device-config.js +3 -2
- package/dist/lib/devices/config-migration.js +147 -1
- package/dist/lib/devices/device-docs.d.ts +35 -0
- package/dist/lib/devices/device-docs.js +163 -0
- package/dist/lib/devices/discovery-policy.d.ts +14 -2
- package/dist/lib/devices/discovery-policy.js +31 -21
- package/dist/lib/devices/registry.d.ts +11 -5
- package/dist/lib/devices/registry.js +46 -18
- package/dist/lib/exec.d.ts +66 -28
- package/dist/lib/exec.js +71 -26
- package/dist/lib/feed/feed.d.ts +10 -2
- package/dist/lib/feed/feed.js +12 -1
- package/dist/lib/feed-broadcast.js +15 -1
- package/dist/lib/git.d.ts +93 -0
- package/dist/lib/git.js +232 -0
- package/dist/lib/hosts/dispatch.d.ts +4 -3
- package/dist/lib/hosts/dispatch.js +12 -8
- package/dist/lib/hosts/providers/local.d.ts +9 -3
- package/dist/lib/hosts/providers/local.js +23 -12
- package/dist/lib/hosts/reconnect.d.ts +7 -4
- package/dist/lib/hosts/reconnect.js +29 -25
- package/dist/lib/hosts/registry.js +4 -1
- package/dist/lib/hosts/remote-os.js +3 -1
- package/dist/lib/monitors/remote.d.ts +18 -1
- package/dist/lib/monitors/remote.js +15 -2
- package/dist/lib/notify.d.ts +7 -0
- package/dist/lib/notify.js +15 -1
- package/dist/lib/session/active.d.ts +10 -1
- package/dist/lib/session/active.js +7 -1
- package/dist/lib/session/actor-sidecar.d.ts +7 -0
- package/dist/lib/session/actor-sidecar.js +2 -0
- package/dist/lib/session/db.d.ts +1 -1
- package/dist/lib/session/db.js +39 -3
- package/dist/lib/session/discover.js +7 -12
- package/dist/lib/session/live-metadata.js +1 -0
- package/dist/lib/session/local-tmux-attach.d.ts +69 -0
- package/dist/lib/session/local-tmux-attach.js +164 -0
- package/dist/lib/session/pid-registry.d.ts +7 -0
- package/dist/lib/session/prompt.d.ts +15 -0
- package/dist/lib/session/prompt.js +21 -0
- package/dist/lib/session/remote-active.d.ts +8 -0
- package/dist/lib/session/remote-active.js +1 -0
- package/dist/lib/session/types.d.ts +17 -0
- package/dist/lib/session/types.js +10 -0
- package/dist/lib/share/publish.d.ts +8 -11
- package/dist/lib/share/publish.js +16 -20
- package/dist/lib/share/worker-template.js +104 -12
- package/dist/lib/state.d.ts +8 -0
- package/dist/lib/state.js +143 -11
- package/dist/lib/sync-status.d.ts +17 -0
- package/dist/lib/sync-status.js +21 -2
- package/dist/lib/terminal/resolve.d.ts +7 -0
- package/dist/lib/terminal/resolve.js +41 -2
- package/dist/lib/tmux/index.d.ts +1 -1
- package/dist/lib/tmux/index.js +1 -1
- package/dist/lib/tmux/session.d.ts +10 -0
- package/dist/lib/tmux/session.js +29 -0
- package/dist/lib/traces/insights.d.ts +67 -0
- package/dist/lib/traces/insights.js +178 -0
- package/dist/lib/traces/phenotype.d.ts +67 -0
- package/dist/lib/traces/phenotype.js +437 -0
- package/dist/lib/traces/segments.d.ts +133 -0
- package/dist/lib/traces/segments.js +301 -0
- package/dist/lib/traces/sync.d.ts +33 -0
- package/dist/lib/traces/sync.js +11 -2
- package/dist/lib/types.d.ts +47 -1
- package/dist/lib/watchdog/runner.js +18 -4
- package/package.json +1 -1
package/dist/lib/sync-status.js
CHANGED
|
@@ -23,8 +23,9 @@ import { ALL_AGENT_IDS } from './agents.js';
|
|
|
23
23
|
import { diffVersionResources, } from './doctor-diff.js';
|
|
24
24
|
import { listInstalledVersions, getGlobalDefault } from './installations/versions.js';
|
|
25
25
|
import { loadManifest } from './staleness/index.js';
|
|
26
|
-
import { getSystemAgentsDir } from './state.js';
|
|
27
|
-
import
|
|
26
|
+
import { getSystemAgentsDir, getUserAgentsDir } from './state.js';
|
|
27
|
+
import * as fs from 'fs';
|
|
28
|
+
import { isGitRepo, readOriginUrl } from './git.js';
|
|
28
29
|
const STATUS_MAP = {
|
|
29
30
|
ok: 'synced',
|
|
30
31
|
diff: 'drifted',
|
|
@@ -73,6 +74,22 @@ export async function getSystemRepoStatus() {
|
|
|
73
74
|
return base;
|
|
74
75
|
}
|
|
75
76
|
}
|
|
77
|
+
/**
|
|
78
|
+
* Detect whether `~/.agents` (the user config layer) is git-backed. A partial
|
|
79
|
+
* install — runtime state present but no `.git` (or no `origin`) — is a distinct
|
|
80
|
+
* drift state that `agents repo sync user` heals by adopting in place (PHNX-3301),
|
|
81
|
+
* surfaced separately from per-version resource gaps. Purely local; no network.
|
|
82
|
+
*/
|
|
83
|
+
export async function getUserRepoStatus() {
|
|
84
|
+
const dir = getUserAgentsDir();
|
|
85
|
+
if (!fs.existsSync(dir))
|
|
86
|
+
return { dir, notGitRepo: false };
|
|
87
|
+
if (!isGitRepo(dir))
|
|
88
|
+
return { dir, notGitRepo: true };
|
|
89
|
+
// A repo with no `origin` is just as partial for adopt's purposes — reuse the
|
|
90
|
+
// single origin-URL reader rather than a second remote check.
|
|
91
|
+
return { dir, notGitRepo: readOriginUrl(dir) === null };
|
|
92
|
+
}
|
|
76
93
|
/**
|
|
77
94
|
* Compute unified sync status across the fleet. Resolves against non-project
|
|
78
95
|
* layers only (`excludeProject: true`) — the GLOBAL version home is never
|
|
@@ -107,6 +124,7 @@ export async function computeSyncStatus(options = {}) {
|
|
|
107
124
|
}
|
|
108
125
|
}
|
|
109
126
|
const system = await getSystemRepoStatus();
|
|
127
|
+
const user = await getUserRepoStatus();
|
|
110
128
|
const agentsNeedingSync = new Set();
|
|
111
129
|
let drifted = 0, missing = 0, orphan = 0, versionsNeedingSync = 0, versionsNeverSynced = 0;
|
|
112
130
|
for (const v of agents) {
|
|
@@ -122,6 +140,7 @@ export async function computeSyncStatus(options = {}) {
|
|
|
122
140
|
}
|
|
123
141
|
return {
|
|
124
142
|
system,
|
|
143
|
+
user,
|
|
125
144
|
agents,
|
|
126
145
|
totals: {
|
|
127
146
|
drifted,
|
|
@@ -34,6 +34,7 @@ export type InjectResolution = {
|
|
|
34
34
|
} | {
|
|
35
35
|
addressable: false;
|
|
36
36
|
reason: string;
|
|
37
|
+
hint?: string;
|
|
37
38
|
};
|
|
38
39
|
export interface ResolveOptions {
|
|
39
40
|
/**
|
|
@@ -49,6 +50,12 @@ export interface ResolveOptions {
|
|
|
49
50
|
*/
|
|
50
51
|
ptyId?: string;
|
|
51
52
|
}
|
|
53
|
+
/**
|
|
54
|
+
* Human-facing recovery hint for a session the resolver judged un-addressable.
|
|
55
|
+
* Tells the user both how to continue THIS session and how to make FUTURE runs
|
|
56
|
+
* addressable, so the failure is not silent and the fix is actionable.
|
|
57
|
+
*/
|
|
58
|
+
export declare function addressabilityRecoveryHint(session: ActiveSession, fallbackId?: string): string;
|
|
52
59
|
/**
|
|
53
60
|
* Resolve a target from an already-fetched ActiveSession. Pure — no I/O — so the
|
|
54
61
|
* precedence logic is unit-testable without the process table. This is where the
|
|
@@ -1,10 +1,40 @@
|
|
|
1
1
|
import { getActiveSessions } from '../session/active.js';
|
|
2
|
+
import { machineId } from '../machine-id.js';
|
|
2
3
|
/** The editor CLIs that speak the swarm-ext URI protocol, keyed by the host detectHost() reports. */
|
|
3
4
|
const IDE_INJECT_VARIANTS = {
|
|
4
5
|
codium: { cli: 'codium', scheme: 'vscodium' },
|
|
5
6
|
cursor: { cli: 'cursor', scheme: 'cursor' },
|
|
6
7
|
code: { cli: 'code', scheme: 'vscode' },
|
|
7
8
|
};
|
|
9
|
+
/**
|
|
10
|
+
* Human-facing recovery hint for a session the resolver judged un-addressable.
|
|
11
|
+
* Tells the user both how to continue THIS session and how to make FUTURE runs
|
|
12
|
+
* addressable, so the failure is not silent and the fix is actionable.
|
|
13
|
+
*/
|
|
14
|
+
export function addressabilityRecoveryHint(session, fallbackId) {
|
|
15
|
+
const sid = session.sessionId;
|
|
16
|
+
// Branch on the live sessionId (an IDE terminal that has not registered one
|
|
17
|
+
// yet is a distinct message), but render the resume command with the real id
|
|
18
|
+
// when the caller can supply it — e.g. `focus` has meta.id even though the
|
|
19
|
+
// live row's sessionId is falsy, so without this the hint printed a useless
|
|
20
|
+
// `agents sessions resume <id>` placeholder in exactly that case (PHNX-3070).
|
|
21
|
+
const resumeId = sid ?? fallbackId;
|
|
22
|
+
const shortId = resumeId ? resumeId.slice(0, 8) : '<id>';
|
|
23
|
+
const device = session.machine ?? machineId();
|
|
24
|
+
const resumeCmd = resumeId ? `agents sessions resume ${shortId}` : 'agents sessions resume <id>';
|
|
25
|
+
const tmuxCmd = `agents config set devices.${device}.tmux on`;
|
|
26
|
+
const interactive = session.context === 'terminal' || !!session.tty;
|
|
27
|
+
if (session.host === 'ghostty') {
|
|
28
|
+
return `Ghostty has no per-split addressing. ${interactive ? `Enable tmux wrapping with \`${tmuxCmd}\` and re-launch, or ` : ''}use \`${resumeCmd}\` to continue this session.`;
|
|
29
|
+
}
|
|
30
|
+
if (session.host && session.host in IDE_INJECT_VARIANTS && !sid) {
|
|
31
|
+
return `This IDE terminal has not registered a session id yet. Wait a moment and retry, or use \`${resumeCmd}\` to continue.`;
|
|
32
|
+
}
|
|
33
|
+
if (session.host) {
|
|
34
|
+
return `Host '${session.host}' has no addressable rail here. ${interactive ? `Enable tmux wrapping with \`${tmuxCmd}\` and re-launch, or ` : ''}use \`${resumeCmd}\` to continue this session.`;
|
|
35
|
+
}
|
|
36
|
+
return `This session has no addressable terminal rail (not tmux, iTerm, an IDE terminal, or a pty sidecar). ${interactive ? `Enable tmux wrapping with \`${tmuxCmd}\` and re-launch, or ` : ''}use \`${resumeCmd}\` to continue this session.`;
|
|
37
|
+
}
|
|
8
38
|
/**
|
|
9
39
|
* Resolve a target from an already-fetched ActiveSession. Pure — no I/O — so the
|
|
10
40
|
* precedence logic is unit-testable without the process table. This is where the
|
|
@@ -35,7 +65,11 @@ export function resolveInjectTargetForSession(session, opts = {}) {
|
|
|
35
65
|
const variant = session.host ? IDE_INJECT_VARIANTS[session.host] : undefined;
|
|
36
66
|
if (variant) {
|
|
37
67
|
if (!session.sessionId) {
|
|
38
|
-
return {
|
|
68
|
+
return {
|
|
69
|
+
addressable: false,
|
|
70
|
+
reason: `IDE terminal (${session.host}) has no session id to address`,
|
|
71
|
+
hint: addressabilityRecoveryHint(session),
|
|
72
|
+
};
|
|
39
73
|
}
|
|
40
74
|
return {
|
|
41
75
|
addressable: true,
|
|
@@ -58,13 +92,18 @@ export function resolveInjectTargetForSession(session, opts = {}) {
|
|
|
58
92
|
note: 'coarse Ghostty window path (opt-in): raises a window and types into the FOCUSED split — not split-precise',
|
|
59
93
|
};
|
|
60
94
|
}
|
|
61
|
-
return {
|
|
95
|
+
return {
|
|
96
|
+
addressable: false,
|
|
97
|
+
reason: 'un-addressable (ghostty, no tmux): no per-split addressing; watchdog skips',
|
|
98
|
+
hint: addressabilityRecoveryHint(session),
|
|
99
|
+
};
|
|
62
100
|
}
|
|
63
101
|
return {
|
|
64
102
|
addressable: false,
|
|
65
103
|
reason: session.host
|
|
66
104
|
? `no precise inject rail for host '${session.host}' (no tmux/iterm/IDE terminal detected)`
|
|
67
105
|
: 'no inject rail: session is not inside tmux, iTerm, or an IDE terminal',
|
|
106
|
+
hint: addressabilityRecoveryHint(session),
|
|
68
107
|
};
|
|
69
108
|
}
|
|
70
109
|
/**
|
package/dist/lib/tmux/index.d.ts
CHANGED
|
@@ -5,4 +5,4 @@
|
|
|
5
5
|
*/
|
|
6
6
|
export { findTmuxBinary, isTmuxInstalled, getTmuxVersion, isTmuxVersionSupported, MIN_TMUX_VERSION, assertTmuxAvailable, TmuxUnavailableError, TmuxCommandError, runTmux, attachTmux, } from './binary.js';
|
|
7
7
|
export { getDefaultSocketPath, getSessionMetaPath, ensureTmuxDir, } from './paths.js';
|
|
8
|
-
export { assertValidSessionName, slugifyName, hasSession, createSession, killSession, killAll, listSessions, splitPane, sendKeys, capturePane, readSessionMeta, TmuxSessionError, reconcileSessionHooks, ensureSessionHookRepaired, type SessionMeta, type CreateSessionOptions, type ListedSession, type SplitOptions, type SendOptions, type CaptureOptions, } from './session.js';
|
|
8
|
+
export { assertValidSessionName, slugifyName, hasSession, createSession, killSession, teardownIfAgentExited, killAll, listSessions, splitPane, sendKeys, capturePane, readSessionMeta, TmuxSessionError, reconcileSessionHooks, ensureSessionHookRepaired, type SessionMeta, type CreateSessionOptions, type ListedSession, type SplitOptions, type SendOptions, type CaptureOptions, } from './session.js';
|
package/dist/lib/tmux/index.js
CHANGED
|
@@ -5,4 +5,4 @@
|
|
|
5
5
|
*/
|
|
6
6
|
export { findTmuxBinary, isTmuxInstalled, getTmuxVersion, isTmuxVersionSupported, MIN_TMUX_VERSION, assertTmuxAvailable, TmuxUnavailableError, TmuxCommandError, runTmux, attachTmux, } from './binary.js';
|
|
7
7
|
export { getDefaultSocketPath, getSessionMetaPath, ensureTmuxDir, } from './paths.js';
|
|
8
|
-
export { assertValidSessionName, slugifyName, hasSession, createSession, killSession, killAll, listSessions, splitPane, sendKeys, capturePane, readSessionMeta, TmuxSessionError, reconcileSessionHooks, ensureSessionHookRepaired, } from './session.js';
|
|
8
|
+
export { assertValidSessionName, slugifyName, hasSession, createSession, killSession, teardownIfAgentExited, killAll, listSessions, splitPane, sendKeys, capturePane, readSessionMeta, TmuxSessionError, reconcileSessionHooks, ensureSessionHookRepaired, } from './session.js';
|
|
@@ -109,6 +109,16 @@ export declare function createSession(opts: CreateSessionOptions): Promise<Sessi
|
|
|
109
109
|
export declare function killSession(name: string, socket?: string, opts?: {
|
|
110
110
|
reapOrphans?: boolean;
|
|
111
111
|
}): Promise<boolean>;
|
|
112
|
+
/**
|
|
113
|
+
* After an attach client returns: destroy the session if every pane is dead
|
|
114
|
+
* (the agent exited); leave it if any pane is still alive (Ctrl-b d).
|
|
115
|
+
*
|
|
116
|
+
* `runInTmux` already does this via `resolveAfterAttach`. The attach verbs
|
|
117
|
+
* (`agents tmux attach`, `sessions focus`/`resume --attach-only`, `jumpTo`)
|
|
118
|
+
* used to `process.exit` the tmux client status and leave a `remain-on-exit`
|
|
119
|
+
* husk — the session "came back" in `tmux ls` after the user exited the agent.
|
|
120
|
+
*/
|
|
121
|
+
export declare function teardownIfAgentExited(name: string, socket?: string): Promise<'killed' | 'kept' | 'absent'>;
|
|
112
122
|
/**
|
|
113
123
|
* Kill every session on the shared server AND the server itself, then prune
|
|
114
124
|
* meta files. Wipes the socket so the next `new` starts from a clean slate.
|
package/dist/lib/tmux/session.js
CHANGED
|
@@ -315,6 +315,35 @@ export async function killSession(name, socket, opts = {}) {
|
|
|
315
315
|
removeSessionMeta(name);
|
|
316
316
|
return true;
|
|
317
317
|
}
|
|
318
|
+
/**
|
|
319
|
+
* After an attach client returns: destroy the session if every pane is dead
|
|
320
|
+
* (the agent exited); leave it if any pane is still alive (Ctrl-b d).
|
|
321
|
+
*
|
|
322
|
+
* `runInTmux` already does this via `resolveAfterAttach`. The attach verbs
|
|
323
|
+
* (`agents tmux attach`, `sessions focus`/`resume --attach-only`, `jumpTo`)
|
|
324
|
+
* used to `process.exit` the tmux client status and leave a `remain-on-exit`
|
|
325
|
+
* husk — the session "came back" in `tmux ls` after the user exited the agent.
|
|
326
|
+
*/
|
|
327
|
+
export async function teardownIfAgentExited(name, socket) {
|
|
328
|
+
assertValidSessionName(name);
|
|
329
|
+
const sock = socket ?? getDefaultSocketPath();
|
|
330
|
+
if (!(await hasSession(name, sock)))
|
|
331
|
+
return 'absent';
|
|
332
|
+
const res = await runTmux({
|
|
333
|
+
socket: sock,
|
|
334
|
+
args: ['list-panes', '-t', `=${name}`, '-F', '#{pane_dead}'],
|
|
335
|
+
throwOnError: false,
|
|
336
|
+
}).catch(() => null);
|
|
337
|
+
if (!res || res.code !== 0) {
|
|
338
|
+
await killSession(name, sock).catch(() => { });
|
|
339
|
+
return 'killed';
|
|
340
|
+
}
|
|
341
|
+
const flags = res.stdout.split('\n').map((s) => s.trim()).filter(Boolean);
|
|
342
|
+
if (flags.some((d) => d === '0'))
|
|
343
|
+
return 'kept';
|
|
344
|
+
await killSession(name, sock).catch(() => { });
|
|
345
|
+
return 'killed';
|
|
346
|
+
}
|
|
318
347
|
/**
|
|
319
348
|
* Kill every session on the shared server AND the server itself, then prune
|
|
320
349
|
* meta files. Wipes the socket so the next `new` starts from a clean slate.
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cross-session failure clustering + time-wasted attribution for the traces
|
|
3
|
+
* insight engine (PHNX-3141) — the piece that turns per-tool error counts
|
|
4
|
+
* into "here is your #1 systemic problem and what it cost."
|
|
5
|
+
*
|
|
6
|
+
* Pure and SQL-shaped: it consumes the same `SyncRow[]` + `tool_calls` rows
|
|
7
|
+
* `buildIndexShard` already loads (no re-parsing of transcripts), so cost
|
|
8
|
+
* stays proportional to this sync's row count, never the full corpus.
|
|
9
|
+
*
|
|
10
|
+
* `tool_calls` rows carry `ordinal`/`timestamp` per call within a session
|
|
11
|
+
* (`db.ts`'s `idx_tool_calls_session ON tool_calls(session_id, ordinal)`), which
|
|
12
|
+
* is enough to reconstruct per-session call order and inter-call gaps without a
|
|
13
|
+
* full `SessionTrajectory` — that is what makes this incremental at scale.
|
|
14
|
+
*
|
|
15
|
+
* Scope note: `FailureSignature` does not yet carry a `phenotype`
|
|
16
|
+
* (false-termination / out-of-order / …, `phenotype.ts`) — classifying that
|
|
17
|
+
* needs the full derived trajectory (turns, ordered steps, gaps), which is
|
|
18
|
+
* only ever materialized per-session during upload, not cached the way
|
|
19
|
+
* `InsightFacets` is. Folding it in is a real, scoped follow-up (see
|
|
20
|
+
* `cli/AGENTS.md`), not a silent omission.
|
|
21
|
+
*/
|
|
22
|
+
import { type TraceFailureCause } from './classify.js';
|
|
23
|
+
import { type LatencyInsight } from './segments.js';
|
|
24
|
+
import { type SyncRow, type ToolCallRow, type TracesIndexShard } from './sync.js';
|
|
25
|
+
export interface FailureSignature {
|
|
26
|
+
tool: string;
|
|
27
|
+
cause: TraceFailureCause;
|
|
28
|
+
/** Normalized error text — volatile tokens (ids, counts, countdowns) stripped so instances fold together. */
|
|
29
|
+
key: string;
|
|
30
|
+
}
|
|
31
|
+
export interface FailurePattern {
|
|
32
|
+
/** Stable hash of the signature — deep-linkable, unaffected by row order. */
|
|
33
|
+
id: string;
|
|
34
|
+
label: string;
|
|
35
|
+
signature: FailureSignature;
|
|
36
|
+
/** Distinct sessions this pattern occurred in. */
|
|
37
|
+
sessions: number;
|
|
38
|
+
/** Total failing calls matching this signature. */
|
|
39
|
+
occurrences: number;
|
|
40
|
+
/** Estimated ms of retry/stall time attributable to this pattern (see attribution rule below). */
|
|
41
|
+
wastedMs: number;
|
|
42
|
+
/** Bounded example session ids for drill-down. */
|
|
43
|
+
exampleSessionIds: string[];
|
|
44
|
+
/** Movement vs the same pattern id in the previous shard. */
|
|
45
|
+
drift: 'up' | 'flat' | 'down';
|
|
46
|
+
}
|
|
47
|
+
export interface ComputedInsights {
|
|
48
|
+
/** Top-K patterns ranked by wastedMs (impact) — a rare 1-occurrence/8h loop still surfaces. */
|
|
49
|
+
failurePatterns: FailurePattern[];
|
|
50
|
+
/** Sum of wastedMs across every cluster found this sync, not just the top-K rows above. */
|
|
51
|
+
wastedMsTotal: number;
|
|
52
|
+
latency: LatencyInsight;
|
|
53
|
+
}
|
|
54
|
+
/** Strip volatile tokens from a failure's evidence text so repeat instances hash identically. */
|
|
55
|
+
export declare function normalizeErrorKey(desc: string, raw: string | null): string;
|
|
56
|
+
/**
|
|
57
|
+
* Cluster failed tool calls into ranked patterns and estimate the wasted time
|
|
58
|
+
* behind each, plus device-wide time-to-first-tool latency.
|
|
59
|
+
*
|
|
60
|
+
* wastedMs attribution: the gap between a failed call and the NEXT call in the
|
|
61
|
+
* same session counts as wasted when either (a) the next call repeats the same
|
|
62
|
+
* signature (a retry loop) or (b) the gap itself is a stall (≥60s) — an idle
|
|
63
|
+
* gap unrelated to a nearby failure is never counted. This is an estimate, not
|
|
64
|
+
* ground truth (a stall could be legitimate user think-time); it is not
|
|
65
|
+
* inflated by folding in ordinary processing time between unrelated calls.
|
|
66
|
+
*/
|
|
67
|
+
export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null): ComputedInsights;
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cross-session failure clustering + time-wasted attribution for the traces
|
|
3
|
+
* insight engine (PHNX-3141) — the piece that turns per-tool error counts
|
|
4
|
+
* into "here is your #1 systemic problem and what it cost."
|
|
5
|
+
*
|
|
6
|
+
* Pure and SQL-shaped: it consumes the same `SyncRow[]` + `tool_calls` rows
|
|
7
|
+
* `buildIndexShard` already loads (no re-parsing of transcripts), so cost
|
|
8
|
+
* stays proportional to this sync's row count, never the full corpus.
|
|
9
|
+
*
|
|
10
|
+
* `tool_calls` rows carry `ordinal`/`timestamp` per call within a session
|
|
11
|
+
* (`db.ts`'s `idx_tool_calls_session ON tool_calls(session_id, ordinal)`), which
|
|
12
|
+
* is enough to reconstruct per-session call order and inter-call gaps without a
|
|
13
|
+
* full `SessionTrajectory` — that is what makes this incremental at scale.
|
|
14
|
+
*
|
|
15
|
+
* Scope note: `FailureSignature` does not yet carry a `phenotype`
|
|
16
|
+
* (false-termination / out-of-order / …, `phenotype.ts`) — classifying that
|
|
17
|
+
* needs the full derived trajectory (turns, ordered steps, gaps), which is
|
|
18
|
+
* only ever materialized per-session during upload, not cached the way
|
|
19
|
+
* `InsightFacets` is. Folding it in is a real, scoped follow-up (see
|
|
20
|
+
* `cli/AGENTS.md`), not a silent omission.
|
|
21
|
+
*/
|
|
22
|
+
import { classifyCause } from './classify.js';
|
|
23
|
+
import { computeLatency } from './segments.js';
|
|
24
|
+
import { failureDescription } from './sync.js';
|
|
25
|
+
// ---------------------------------------------------------------------------
|
|
26
|
+
// Tunables
|
|
27
|
+
// ---------------------------------------------------------------------------
|
|
28
|
+
/** Bounded shard size — patterns, not sessions, so 738 or 738k render identically. */
|
|
29
|
+
const TOP_K_PATTERNS = 25;
|
|
30
|
+
const MAX_EXAMPLE_SESSIONS = 5;
|
|
31
|
+
/** A gap this long right after a failure reads as an idle stall, not think-time (matches sync.ts's own "stalled Xm" threshold). */
|
|
32
|
+
const STALL_MS = 60_000;
|
|
33
|
+
// ---------------------------------------------------------------------------
|
|
34
|
+
// Signature normalization — fold volatile per-instance text together
|
|
35
|
+
// ---------------------------------------------------------------------------
|
|
36
|
+
const VOLATILE_TOKEN_PATTERNS = [
|
|
37
|
+
{ pattern: /\bfor user [\w.-]+/gi, replacement: 'for user _' },
|
|
38
|
+
{ pattern: /\btry again in [\w.]+s?\b/gi, replacement: 'try again in _s' },
|
|
39
|
+
{ pattern: /\b[0-9a-f]{7,40}\b/gi, replacement: '_sha_' },
|
|
40
|
+
{ pattern: /\b[\w.-]+@[\w.-]+\.\w+\b/gi, replacement: '_email_' },
|
|
41
|
+
{ pattern: /\b\d+\b/g, replacement: '_n_' },
|
|
42
|
+
];
|
|
43
|
+
/** Strip volatile tokens from a failure's evidence text so repeat instances hash identically. */
|
|
44
|
+
export function normalizeErrorKey(desc, raw) {
|
|
45
|
+
let text = (raw && raw.trim().length > 0 ? raw : desc).toLowerCase();
|
|
46
|
+
for (const { pattern, replacement } of VOLATILE_TOKEN_PATTERNS) {
|
|
47
|
+
text = text.replace(pattern, replacement);
|
|
48
|
+
}
|
|
49
|
+
return text.replace(/\s+/g, ' ').trim().slice(0, 160);
|
|
50
|
+
}
|
|
51
|
+
function hashSignature(tool, cause, key) {
|
|
52
|
+
const input = `${tool} ${cause} ${key}`;
|
|
53
|
+
let hash = 5381;
|
|
54
|
+
for (let i = 0; i < input.length; i++) {
|
|
55
|
+
hash = ((hash << 5) + hash + input.charCodeAt(i)) >>> 0;
|
|
56
|
+
}
|
|
57
|
+
return hash.toString(16).padStart(8, '0');
|
|
58
|
+
}
|
|
59
|
+
// ---------------------------------------------------------------------------
|
|
60
|
+
// Label rules — a table, not an if/else-by-name chain (matches segments.ts's TASK_TYPE_RULES)
|
|
61
|
+
// ---------------------------------------------------------------------------
|
|
62
|
+
const LABEL_RULES = [
|
|
63
|
+
{ pattern: /rate limit/i, label: 'rate limit back-off loop' },
|
|
64
|
+
{ pattern: /permission denied/i, label: 'permission denied' },
|
|
65
|
+
{ pattern: /not found|no such file/i, label: 'missing resource' },
|
|
66
|
+
{ pattern: /timed? ?out/i, label: 'timeout' },
|
|
67
|
+
{ pattern: /econnrefused|connection refused|network/i, label: 'network error' },
|
|
68
|
+
{ pattern: /conflict|diverged/i, label: 'git conflict' },
|
|
69
|
+
];
|
|
70
|
+
function labelFor(tool, cause, key) {
|
|
71
|
+
if (cause === 'guard')
|
|
72
|
+
return `${tool}: git guard denial`;
|
|
73
|
+
if (cause === 'hook')
|
|
74
|
+
return `${tool}: hook denial`;
|
|
75
|
+
const rule = LABEL_RULES.find((row) => row.pattern.test(key));
|
|
76
|
+
return `${tool}: ${rule ? rule.label : key.slice(0, 48)}`;
|
|
77
|
+
}
|
|
78
|
+
// ---------------------------------------------------------------------------
|
|
79
|
+
// Public API
|
|
80
|
+
// ---------------------------------------------------------------------------
|
|
81
|
+
/**
|
|
82
|
+
* Cluster failed tool calls into ranked patterns and estimate the wasted time
|
|
83
|
+
* behind each, plus device-wide time-to-first-tool latency.
|
|
84
|
+
*
|
|
85
|
+
* wastedMs attribution: the gap between a failed call and the NEXT call in the
|
|
86
|
+
* same session counts as wasted when either (a) the next call repeats the same
|
|
87
|
+
* signature (a retry loop) or (b) the gap itself is a stall (≥60s) — an idle
|
|
88
|
+
* gap unrelated to a nearby failure is never counted. This is an estimate, not
|
|
89
|
+
* ground truth (a stall could be legitimate user think-time); it is not
|
|
90
|
+
* inflated by folding in ordinary processing time between unrelated calls.
|
|
91
|
+
*/
|
|
92
|
+
export function computeInsights(rows, calls, prevShard) {
|
|
93
|
+
const bySession = new Map();
|
|
94
|
+
for (const call of calls) {
|
|
95
|
+
const list = bySession.get(call.session_id);
|
|
96
|
+
if (list)
|
|
97
|
+
list.push(call);
|
|
98
|
+
else
|
|
99
|
+
bySession.set(call.session_id, [call]);
|
|
100
|
+
}
|
|
101
|
+
const groups = new Map();
|
|
102
|
+
for (const [sessionId, sessionCalls] of bySession) {
|
|
103
|
+
const ordered = [...sessionCalls].sort((a, b) => a.ordinal - b.ordinal);
|
|
104
|
+
for (let i = 0; i < ordered.length; i++) {
|
|
105
|
+
const call = ordered[i];
|
|
106
|
+
if (call.outcome !== 'error')
|
|
107
|
+
continue;
|
|
108
|
+
const cause = classifyCause(call);
|
|
109
|
+
const key = normalizeErrorKey(failureDescription(call, cause), call.error);
|
|
110
|
+
const groupKey = `${call.tool} ${cause} ${key}`;
|
|
111
|
+
let group = groups.get(groupKey);
|
|
112
|
+
if (!group) {
|
|
113
|
+
group = { tool: call.tool, cause, key, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
|
|
114
|
+
groups.set(groupKey, group);
|
|
115
|
+
}
|
|
116
|
+
group.occurrences++;
|
|
117
|
+
group.sessions.add(sessionId);
|
|
118
|
+
if (group.examples.length < MAX_EXAMPLE_SESSIONS && !group.examples.includes(sessionId)) {
|
|
119
|
+
group.examples.push(sessionId);
|
|
120
|
+
}
|
|
121
|
+
const next = ordered[i + 1];
|
|
122
|
+
if (!next)
|
|
123
|
+
continue;
|
|
124
|
+
const gapMs = Date.parse(next.timestamp) - Date.parse(call.timestamp);
|
|
125
|
+
if (!Number.isFinite(gapMs) || gapMs <= 0)
|
|
126
|
+
continue;
|
|
127
|
+
const nextIsSameFailure = next.outcome === 'error' &&
|
|
128
|
+
next.tool === call.tool &&
|
|
129
|
+
classifyCause(next) === cause &&
|
|
130
|
+
normalizeErrorKey(failureDescription(next, cause), next.error) === key;
|
|
131
|
+
if (nextIsSameFailure || gapMs >= STALL_MS) {
|
|
132
|
+
group.wastedMs += gapMs;
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
const prevById = new Map((prevShard?.failurePatterns ?? []).map((p) => [p.id, p]));
|
|
137
|
+
const allPatterns = [...groups.values()].map((group) => {
|
|
138
|
+
const id = hashSignature(group.tool, group.cause, group.key);
|
|
139
|
+
const prev = prevById.get(id);
|
|
140
|
+
const drift = !prev
|
|
141
|
+
? 'up'
|
|
142
|
+
: group.occurrences > prev.occurrences
|
|
143
|
+
? 'up'
|
|
144
|
+
: group.occurrences < prev.occurrences
|
|
145
|
+
? 'down'
|
|
146
|
+
: 'flat';
|
|
147
|
+
return {
|
|
148
|
+
id,
|
|
149
|
+
label: labelFor(group.tool, group.cause, group.key),
|
|
150
|
+
signature: { tool: group.tool, cause: group.cause, key: group.key },
|
|
151
|
+
sessions: group.sessions.size,
|
|
152
|
+
occurrences: group.occurrences,
|
|
153
|
+
wastedMs: group.wastedMs,
|
|
154
|
+
exampleSessionIds: group.examples,
|
|
155
|
+
drift,
|
|
156
|
+
};
|
|
157
|
+
});
|
|
158
|
+
const wastedMsTotal = allPatterns.reduce((sum, p) => sum + p.wastedMs, 0);
|
|
159
|
+
const failurePatterns = [...allPatterns]
|
|
160
|
+
.sort((a, b) => b.wastedMs - a.wastedMs || b.occurrences - a.occurrences || a.id.localeCompare(b.id))
|
|
161
|
+
.slice(0, TOP_K_PATTERNS);
|
|
162
|
+
const latency = computeLatency(firstToolSegments(rows, bySession));
|
|
163
|
+
return { failurePatterns, wastedMsTotal, latency };
|
|
164
|
+
}
|
|
165
|
+
/** Synthesize one-step SegmentSessions carrying only the time-to-first-tool offset, for computeLatency() reuse. */
|
|
166
|
+
function firstToolSegments(rows, bySession) {
|
|
167
|
+
return rows.flatMap((row) => {
|
|
168
|
+
const sessionCalls = bySession.get(row.id);
|
|
169
|
+
if (!sessionCalls || sessionCalls.length === 0)
|
|
170
|
+
return [];
|
|
171
|
+
const first = sessionCalls.reduce((min, call) => (call.ordinal < min.ordinal ? call : min));
|
|
172
|
+
const sessionStartMs = Date.parse(row.timestamp);
|
|
173
|
+
const firstCallMs = Date.parse(first.timestamp);
|
|
174
|
+
if (!Number.isFinite(sessionStartMs) || !Number.isFinite(firstCallMs))
|
|
175
|
+
return [];
|
|
176
|
+
return [{ steps: [{ startMs: Math.max(0, firstCallMs - sessionStartMs) }] }];
|
|
177
|
+
});
|
|
178
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Failure phenotype classifier + outcome taxonomy for the traces insight engine.
|
|
3
|
+
*
|
|
4
|
+
* Both functions are pure: they take a redacted {@link SessionDetail} (the same
|
|
5
|
+
* shape the traces sync writes to `sessions/<id>.json`) and return a decision
|
|
6
|
+
* derived only from the already-derived step/gap/meta signal. They never read
|
|
7
|
+
* raw transcript text and never fabricate a signal that is not in the data.
|
|
8
|
+
*
|
|
9
|
+
* The rubrics are expressed as data-driven tables of conditions, not as
|
|
10
|
+
* if/else-by-name chains. Each table row is a named phenotype/outcome with a
|
|
11
|
+
* declarative predicate; the classifier walks the table in priority order and
|
|
12
|
+
* returns the first match, or the honest lower-confidence default when no
|
|
13
|
+
* high-confidence signal is present.
|
|
14
|
+
*/
|
|
15
|
+
import type { SessionDetail } from './sync.js';
|
|
16
|
+
/** A failure mode detectable from the derived trajectory of a session. */
|
|
17
|
+
export type FailurePhenotype = 'false-termination' | 'premature-completion' | 'out-of-order' | 'failure-to-act';
|
|
18
|
+
/**
|
|
19
|
+
* The coarse outcome of a session's work.
|
|
20
|
+
*
|
|
21
|
+
* - `merged` : explicit PR/branch merge signal in the steps (high confidence).
|
|
22
|
+
* - `tests-green` : explicit test command returned ok with no later failure (high confidence).
|
|
23
|
+
* - `partial` : progress was made but no landing/test signal is present (low confidence default).
|
|
24
|
+
* - `abandoned` : errored, stalled, and unresolved.
|
|
25
|
+
* - `human-takeover`: the final substantive action was a human-facing ask/wait.
|
|
26
|
+
* - `invalid-env` : environment/setup failures dominated the session.
|
|
27
|
+
*/
|
|
28
|
+
export type TraceOutcome = 'merged' | 'tests-green' | 'partial' | 'abandoned' | 'human-takeover' | 'invalid-env';
|
|
29
|
+
export interface PhenotypeResult {
|
|
30
|
+
phenotype: FailurePhenotype | null;
|
|
31
|
+
reason: string;
|
|
32
|
+
}
|
|
33
|
+
export interface OutcomeResult {
|
|
34
|
+
outcome: TraceOutcome;
|
|
35
|
+
confidence: 'high' | 'medium' | 'low';
|
|
36
|
+
reason: string;
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Classify the failure phenotype of a session from its derived trajectory.
|
|
40
|
+
*
|
|
41
|
+
* Definitions (from agent-failure research):
|
|
42
|
+
* - `false-termination` — stopped with an unresolved error.
|
|
43
|
+
* - `premature-completion` — declared done while tests were failing or no
|
|
44
|
+
* verification step ran for the engineering work.
|
|
45
|
+
* - `out-of-order` — a write/edit step occurred before any read/plan of the
|
|
46
|
+
* target.
|
|
47
|
+
* - `failure-to-act` — stalled or produced no meaningful tool use.
|
|
48
|
+
*
|
|
49
|
+
* Returns `null` when none of the failure phenotypes apply.
|
|
50
|
+
*/
|
|
51
|
+
export declare function classifyPhenotype(session: SessionDetail): FailurePhenotype | null;
|
|
52
|
+
/** Detailed phenotype result with a human-readable reason. */
|
|
53
|
+
export declare function classifyPhenotypeDetailed(session: SessionDetail): PhenotypeResult;
|
|
54
|
+
/**
|
|
55
|
+
* Derive the coarse outcome of a session from its end state + tool signals.
|
|
56
|
+
*
|
|
57
|
+
* `merged` and `tests-green` are high-confidence only when an explicit signal
|
|
58
|
+
* is present in the derived steps. When that signal is genuinely not in the
|
|
59
|
+
* data, the function returns the honest lower-confidence value (`partial` for
|
|
60
|
+
* completed work without a landing signal, `abandoned` for errored/unresolved
|
|
61
|
+
* work, `invalid-env` for setup-dominant failures, `human-takeover` when the
|
|
62
|
+
* session ends on a human-facing ask).
|
|
63
|
+
*/
|
|
64
|
+
export declare function deriveOutcome(session: SessionDetail): TraceOutcome;
|
|
65
|
+
/** Detailed outcome result with confidence and a human-readable reason. */
|
|
66
|
+
export declare function deriveOutcomeDetailed(session: SessionDetail): OutcomeResult;
|
|
67
|
+
export type { SessionDetail } from './sync.js';
|