@phnx-labs/agents-cli 1.22.59 → 1.22.61

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/CHANGELOG.md +71 -0
  2. package/dist/cli/command-registry.d.ts +1 -0
  3. package/dist/cli/command-registry.js +2 -0
  4. package/dist/commands/browser.js +9 -4
  5. package/dist/commands/doctor.js +1 -1
  6. package/dist/commands/exec.js +35 -1
  7. package/dist/commands/harness-hooks.d.ts +55 -0
  8. package/dist/commands/harness-hooks.js +104 -0
  9. package/dist/commands/harness-wizard.d.ts +33 -14
  10. package/dist/commands/harness-wizard.js +53 -23
  11. package/dist/commands/harness.d.ts +14 -0
  12. package/dist/commands/harness.js +86 -5
  13. package/dist/commands/perf.js +10 -0
  14. package/dist/commands/reminders.d.ts +9 -0
  15. package/dist/commands/reminders.js +49 -0
  16. package/dist/commands/run-account-picker.d.ts +14 -0
  17. package/dist/commands/run-account-picker.js +13 -0
  18. package/dist/commands/sessions-picker.d.ts +13 -0
  19. package/dist/commands/sessions-picker.js +17 -8
  20. package/dist/commands/sessions.js +13 -11
  21. package/dist/commands/teams-picker.js +20 -6
  22. package/dist/commands/teams.d.ts +3 -3
  23. package/dist/commands/teams.js +86 -24
  24. package/dist/index.js +9 -0
  25. package/dist/lib/accounting/rotate.d.ts +63 -0
  26. package/dist/lib/accounting/rotate.js +240 -16
  27. package/dist/lib/accounting/usage-sync.d.ts +12 -2
  28. package/dist/lib/accounting/usage-sync.js +34 -6
  29. package/dist/lib/browser/drivers/local.d.ts +11 -0
  30. package/dist/lib/browser/drivers/local.js +26 -0
  31. package/dist/lib/browser/profiles.js +8 -6
  32. package/dist/lib/browser/service.d.ts +12 -8
  33. package/dist/lib/browser/service.js +38 -10
  34. package/dist/lib/claude-statusline.d.ts +14 -1
  35. package/dist/lib/claude-statusline.js +27 -2
  36. package/dist/lib/daemon/runner.js +17 -2
  37. package/dist/lib/devices/doctor-findings.d.ts +1 -1
  38. package/dist/lib/devices/doctor-findings.js +22 -4
  39. package/dist/lib/doctor-diff.d.ts +21 -5
  40. package/dist/lib/doctor-diff.js +242 -76
  41. package/dist/lib/feed/events.d.ts +1 -1
  42. package/dist/lib/feed/events.js +28 -15
  43. package/dist/lib/github/gh-overload.d.ts +58 -0
  44. package/dist/lib/github/gh-overload.js +246 -0
  45. package/dist/lib/github/rest.d.ts +64 -0
  46. package/dist/lib/github/rest.js +111 -0
  47. package/dist/lib/harness-connection-test.d.ts +57 -0
  48. package/dist/lib/harness-connection-test.js +80 -0
  49. package/dist/lib/heal.js +8 -3
  50. package/dist/lib/installations/shims.d.ts +22 -0
  51. package/dist/lib/installations/shims.js +104 -0
  52. package/dist/lib/linear-project-counts.js +8 -0
  53. package/dist/lib/linear-rate-limit.d.ts +26 -0
  54. package/dist/lib/linear-rate-limit.js +163 -0
  55. package/dist/lib/mcp.d.ts +9 -0
  56. package/dist/lib/mcp.js +37 -1
  57. package/dist/lib/open-url.js +5 -3
  58. package/dist/lib/perf/db.d.ts +1 -1
  59. package/dist/lib/perf/db.js +53 -2
  60. package/dist/lib/perf/types.d.ts +14 -0
  61. package/dist/lib/permissions.d.ts +28 -0
  62. package/dist/lib/permissions.js +156 -1
  63. package/dist/lib/refresh.js +9 -1
  64. package/dist/lib/reminders.d.ts +29 -0
  65. package/dist/lib/reminders.js +88 -0
  66. package/dist/lib/resource-content-diff.d.ts +33 -0
  67. package/dist/lib/resource-content-diff.js +103 -0
  68. package/dist/lib/rules/compile.d.ts +7 -0
  69. package/dist/lib/rules/compile.js +7 -1
  70. package/dist/lib/session/active.d.ts +41 -4
  71. package/dist/lib/session/active.js +58 -7
  72. package/dist/lib/session/host-link.d.ts +22 -0
  73. package/dist/lib/session/host-link.js +40 -4
  74. package/dist/lib/session/live-metadata.js +3 -3
  75. package/dist/lib/session/trajectory.d.ts +42 -0
  76. package/dist/lib/session/trajectory.js +46 -27
  77. package/dist/lib/ssh-exec.d.ts +30 -0
  78. package/dist/lib/ssh-exec.js +37 -5
  79. package/dist/lib/startup/command-registry.js +1 -1
  80. package/dist/lib/subagents-registry.d.ts +18 -0
  81. package/dist/lib/subagents-registry.js +79 -0
  82. package/dist/lib/teams/agents.d.ts +12 -0
  83. package/dist/lib/teams/agents.js +51 -0
  84. package/dist/lib/teams/api.d.ts +8 -0
  85. package/dist/lib/teams/api.js +50 -6
  86. package/dist/lib/teams/delivery.d.ts +14 -4
  87. package/dist/lib/teams/delivery.js +15 -5
  88. package/dist/lib/traces/schema2-build.d.ts +85 -0
  89. package/dist/lib/traces/schema2-build.js +637 -0
  90. package/dist/lib/traces/schema2-danger.d.ts +36 -0
  91. package/dist/lib/traces/schema2-danger.js +185 -0
  92. package/dist/lib/traces/schema2.d.ts +149 -0
  93. package/dist/lib/traces/schema2.js +20 -0
  94. package/dist/lib/traces/sync.d.ts +93 -0
  95. package/dist/lib/traces/sync.js +75 -22
  96. package/dist/lib/traces/worker-template.js +5 -0
  97. package/dist/lib/uninstall.js +10 -1
  98. package/dist/lib/workflows.d.ts +11 -0
  99. package/dist/lib/workflows.js +67 -8
  100. package/package.json +1 -1
@@ -0,0 +1,185 @@
1
+ /**
2
+ * Danger classifier for a single shell action's argv (PHNX-3442, producer side).
3
+ *
4
+ * Safety-sensitive: the schema-2 `BashAction.danger` drives the console's
5
+ * destructive-operation surfacing and risk scoring. So this is CONSERVATIVE by
6
+ * construction — it defaults to `normal` and only escalates on CLEAR structural
7
+ * evidence in the tokenized argv, never on a substring of raw command text. The
8
+ * argv it reads is one tokenizeBash segment (see `tokenizeBash` in
9
+ * `session/bash-command.ts`): the executable at argv[0] and its already-split
10
+ * arguments, so a flag like `-rf` is a whole token, not a substring hunt.
11
+ *
12
+ * The three levels mirror the shipped consumer union (`BashDanger`):
13
+ * - DESTRUCTIVE — irrecoverable data loss / history rewrite / force
14
+ * overwrite of an important path. Requires a WHERE-less
15
+ * DELETE, a recursive/force delete, a hard reset, etc.
16
+ * - potentially-destructive — plain `rm`, soft/mixed `git reset`, `mv` over a
17
+ * path, plain `kill` — recoverable-ish but worth a flag.
18
+ * - normal — everything else.
19
+ *
20
+ * `destructiveOperation` is a short stable label (never raw text) naming WHY the
21
+ * action was flagged, so the console can group by operation without re-parsing.
22
+ */
23
+ const NORMAL = { danger: 'normal' };
24
+ /** basename of an executable token so `/bin/rm` and `rm` classify alike. */
25
+ function baseName(token) {
26
+ const noArgs = token.replace(/^.*\//, '');
27
+ return noArgs.toLowerCase();
28
+ }
29
+ /** True when any argv token is exactly one of `names`. */
30
+ function hasToken(argv, names) {
31
+ return argv.some((t) => names.has(t));
32
+ }
33
+ /** A single-dash cluster flag that CONTAINS every letter in `letters` (e.g. `-rf` ⊇ r,f). */
34
+ function hasClusterFlag(argv, letters) {
35
+ return argv.some((t) => {
36
+ if (!/^-[a-zA-Z]+$/.test(t))
37
+ return false; // single-dash short cluster only
38
+ const body = t.slice(1);
39
+ return letters.every((l) => body.includes(l));
40
+ });
41
+ }
42
+ /** A GNU long flag present as its own token (e.g. `--force`, `--hard`). */
43
+ function hasLongFlag(argv, flag) {
44
+ return argv.includes(flag);
45
+ }
46
+ const RM_TOKEN = new Set(['rm']);
47
+ const KILL_TOKEN = new Set(['kill', 'pkill', 'killall']);
48
+ /** Paths that are catastrophic to force-overwrite via a redirect target. */
49
+ const IMPORTANT_REDIRECT_TARGET = /^\/dev\/(?:sd|nvme|disk|hd|mmcblk|vd)/i;
50
+ /**
51
+ * SQL-ish argv reconstruction: for a `psql -c "DROP TABLE x"` the SQL lives in a
52
+ * single quoted token, so danger scanning of SQL joins the argv back into one
53
+ * lower-cased string and matches structural SQL, not shell tokens. Bounded to the
54
+ * argv we already hold — no new parse.
55
+ */
56
+ function joinedSql(argv) {
57
+ return argv.join(' ').toLowerCase();
58
+ }
59
+ /**
60
+ * Classify one tokenized shell action. `argvComplete` is false when a dynamic node
61
+ * (command substitution / glob / var expansion) kept the argv incomplete; when a
62
+ * DESTRUCTIVE signal depends on a token that could have been mangled by expansion
63
+ * we DO still flag it (a `rm -rf $DIR` is destructive regardless of what `$DIR`
64
+ * expands to), because the operation itself is the danger, not its target.
65
+ */
66
+ export function classifyActionDanger(argv, _argvComplete = true) {
67
+ if (argv.length === 0)
68
+ return NORMAL;
69
+ const exe = baseName(argv[0]);
70
+ // Everything after the executable — the flags/args the danger tests read.
71
+ const rest = argv.slice(1);
72
+ // ── rm ──────────────────────────────────────────────────────────────────
73
+ if (exe === 'rm' || hasToken(argv, RM_TOKEN)) {
74
+ // Only treat a real `rm` invocation (argv[0]) — a stray `rm` argument to some
75
+ // other tool is not an rm call.
76
+ if (exe === 'rm') {
77
+ const recursive = hasClusterFlag(rest, ['r']) || hasLongFlag(rest, '--recursive');
78
+ const force = hasClusterFlag(rest, ['f']) || hasLongFlag(rest, '--force');
79
+ if (recursive && force) {
80
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'recursive-force-delete' };
81
+ }
82
+ if (recursive) {
83
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'recursive-delete' };
84
+ }
85
+ // Plain `rm file` (or `rm -f file` without recursion) — recoverable-ish.
86
+ return { danger: 'potentially-destructive', destructiveOperation: 'delete' };
87
+ }
88
+ }
89
+ // ── git ─────────────────────────────────────────────────────────────────
90
+ if (exe === 'git') {
91
+ const sub = rest.find((t) => !t.startsWith('-'));
92
+ if (sub === 'reset') {
93
+ if (hasLongFlag(rest, '--hard')) {
94
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'git-reset-hard' };
95
+ }
96
+ // soft / mixed reset — moves HEAD but keeps the working tree.
97
+ return { danger: 'potentially-destructive', destructiveOperation: 'git-reset' };
98
+ }
99
+ if (sub === 'clean') {
100
+ // `git clean -fd` / `-fdx` — deletes untracked files irrecoverably.
101
+ if (hasClusterFlag(rest, ['f'])) {
102
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'git-clean-force' };
103
+ }
104
+ return { danger: 'potentially-destructive', destructiveOperation: 'git-clean' };
105
+ }
106
+ if (sub === 'push') {
107
+ // A leading-`+` refspec (`git push origin +main`, `+refs/heads/main:main`,
108
+ // `+HEAD:main`) forces the push just like `--force`, with no flag to catch.
109
+ // Match a `+` followed by a ref char — not a lone `+` or a `+-`-style flag.
110
+ const forceRefspec = rest.some((t) => /^\+[^-\s]/.test(t));
111
+ if (hasLongFlag(rest, '--force') ||
112
+ hasClusterFlag(rest, ['f']) ||
113
+ rest.includes('--force-with-lease') ||
114
+ forceRefspec) {
115
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'git-push-force' };
116
+ }
117
+ }
118
+ if (sub === 'checkout') {
119
+ // `git checkout -- .` / `git checkout -- <path>` throws away working changes.
120
+ if (rest.includes('--')) {
121
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'git-checkout-discard' };
122
+ }
123
+ }
124
+ if (sub === 'stash') {
125
+ const after = rest.slice(rest.indexOf('stash') + 1);
126
+ if (after.includes('drop') || after.includes('clear')) {
127
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'git-stash-drop' };
128
+ }
129
+ }
130
+ return NORMAL;
131
+ }
132
+ // ── kill ────────────────────────────────────────────────────────────────
133
+ if (exe === 'kill' || exe === 'pkill' || exe === 'killall') {
134
+ if (hasToken(rest, new Set(['-9', '-SIGKILL', '-KILL']))) {
135
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'kill-9' };
136
+ }
137
+ if (hasToken(argv, KILL_TOKEN)) {
138
+ return { danger: 'potentially-destructive', destructiveOperation: 'kill' };
139
+ }
140
+ }
141
+ // ── mv (over an existing path — we cannot know if the target exists, so this is
142
+ // the recoverable-ish tier, never DESTRUCTIVE) ────────────────────────────
143
+ if (exe === 'mv') {
144
+ return { danger: 'potentially-destructive', destructiveOperation: 'move-overwrite' };
145
+ }
146
+ // ── dd / mkfs (disk writers) ──────────────────────────────────────────────
147
+ if (exe === 'dd') {
148
+ if (rest.some((t) => /^of=/.test(t))) {
149
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'dd-write' };
150
+ }
151
+ }
152
+ if (/^mkfs(\.|$)/.test(exe)) {
153
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'mkfs' };
154
+ }
155
+ // ── redirect to a raw device / important path ─────────────────────────────
156
+ // A `> /dev/sda`-style redirect target appears as a token in the argv (the
157
+ // tokenizer keeps `>` and its target). Flag only clearly catastrophic targets.
158
+ for (let i = 0; i < argv.length; i++) {
159
+ const t = argv[i];
160
+ if (t === '>' || t === '>>') {
161
+ const target = argv[i + 1];
162
+ if (target && IMPORTANT_REDIRECT_TARGET.test(target)) {
163
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'overwrite-device' };
164
+ }
165
+ }
166
+ // Fused form `>/dev/sda`.
167
+ const fused = t.match(/^>>?(\/\S+)$/);
168
+ if (fused && IMPORTANT_REDIRECT_TARGET.test(fused[1])) {
169
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'overwrite-device' };
170
+ }
171
+ }
172
+ // ── SQL (psql/mysql/sqlite3 -c "…", or a bare SQL statement) ───────────────
173
+ const sql = joinedSql(argv);
174
+ if (/\bdrop\s+table\b/.test(sql)) {
175
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'sql-drop-table' };
176
+ }
177
+ if (/\btruncate\b/.test(sql)) {
178
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'sql-truncate' };
179
+ }
180
+ // DELETE FROM without a WHERE clause. A DELETE with WHERE is scoped → normal.
181
+ if (/\bdelete\s+from\b/.test(sql) && !/\bwhere\b/.test(sql)) {
182
+ return { danger: 'DESTRUCTIVE', destructiveOperation: 'sql-delete-no-where' };
183
+ }
184
+ return NORMAL;
185
+ }
@@ -0,0 +1,149 @@
1
+ export type StepOutcome = 'ok' | 'error' | 'running' | 'unknown';
2
+ export interface SessionStepBase {
3
+ ordinal: number;
4
+ startMs: number;
5
+ durationMs: number;
6
+ /**
7
+ * Whether `durationMs` was measured from a paired result (`false`) or inferred
8
+ * from the next event (`true`). Mandatory — the console must visually
9
+ * distinguish measured from inferred duration.
10
+ */
11
+ durationEstimated: boolean;
12
+ outcome: StepOutcome;
13
+ label: string;
14
+ }
15
+ export interface ThinkingStep extends SessionStepBase {
16
+ kind: 'thinking';
17
+ lane: 'think';
18
+ outcome: 'ok' | 'running' | 'unknown';
19
+ }
20
+ export interface ExecutionBase extends SessionStepBase {
21
+ kind: 'execution';
22
+ lane: string;
23
+ callId?: string;
24
+ /** The tool call this execution targets (hook/permission → the guarded call). */
25
+ targetCallId?: string;
26
+ }
27
+ export type ToolExecutionType = 'bash' | 'edit' | 'write' | 'read' | 'grep' | 'generic';
28
+ export interface ToolExecutionBase extends ExecutionBase {
29
+ executionType: ToolExecutionType;
30
+ tool: string;
31
+ result: ExecutionResult;
32
+ }
33
+ export interface ExecutionResult {
34
+ exitCode?: number;
35
+ statusCode?: number;
36
+ errorCode?: string;
37
+ stdout?: TextPreview;
38
+ stderr?: TextPreview;
39
+ combined?: TextPreview;
40
+ }
41
+ export interface TextPreview {
42
+ text: string;
43
+ truncated: boolean;
44
+ originalBytes: number;
45
+ }
46
+ export type BashCategory = 'build' | 'test' | 'git' | 'network' | 'other';
47
+ export type BashDanger = 'normal' | 'potentially-destructive' | 'DESTRUCTIVE';
48
+ export interface BashAction {
49
+ ordinal: number;
50
+ source: string;
51
+ argv: string[];
52
+ /** false when a dynamic node (substitution/expansion) kept argv incomplete. */
53
+ argvComplete: boolean;
54
+ program?: string;
55
+ categories: BashCategory[];
56
+ danger: BashDanger;
57
+ destructiveOperation?: string;
58
+ }
59
+ export interface BashExecution extends ToolExecutionBase {
60
+ executionType: 'bash';
61
+ /** Redacted outer command. */
62
+ command: string;
63
+ /** Redacted shell payload after unwrapping `/bin/zsh -lc "…"`. */
64
+ unwrappedCommand: string;
65
+ parseStatus: 'parsed' | 'partial' | 'unparseable';
66
+ parseDiagnostics: string[];
67
+ actions: BashAction[];
68
+ }
69
+ export type FileOperation = 'create' | 'update' | 'overwrite' | 'delete' | 'rename' | 'unknown';
70
+ export interface FileHunk {
71
+ id: string;
72
+ oldStart?: number;
73
+ oldLines?: number;
74
+ newStart?: number;
75
+ newLines?: number;
76
+ addedLines: number;
77
+ removedLines: number;
78
+ beforeHash?: string;
79
+ afterHash?: string;
80
+ before?: TextPreview;
81
+ after?: TextPreview;
82
+ revertedByStep?: number;
83
+ }
84
+ export interface FileMutation {
85
+ path: string;
86
+ operation: FileOperation;
87
+ hunks: FileHunk[];
88
+ revertedByStep?: number;
89
+ }
90
+ export interface RevertLink {
91
+ revertedStep: number;
92
+ path: string;
93
+ revertedHunkIds: string[];
94
+ }
95
+ export interface EditExecution extends ToolExecutionBase {
96
+ executionType: 'edit';
97
+ files: FileMutation[];
98
+ reverts: RevertLink[];
99
+ }
100
+ export interface WriteExecution extends ToolExecutionBase {
101
+ executionType: 'write';
102
+ files: FileMutation[];
103
+ reverts: RevertLink[];
104
+ }
105
+ export interface CountResult {
106
+ value: number;
107
+ relation: 'exact' | 'at-least';
108
+ }
109
+ export interface ReadExecution extends ToolExecutionBase {
110
+ executionType: 'read';
111
+ file: string;
112
+ offset?: number;
113
+ limit?: number;
114
+ returnedLines?: CountResult;
115
+ returnedBytes?: CountResult;
116
+ }
117
+ export interface GrepExecution extends ToolExecutionBase {
118
+ executionType: 'grep';
119
+ query: string;
120
+ path?: string;
121
+ glob?: string;
122
+ outputMode?: 'content' | 'files' | 'count' | 'unknown';
123
+ hits?: CountResult;
124
+ }
125
+ export interface GenericToolExecution extends ToolExecutionBase {
126
+ executionType: 'generic';
127
+ input?: TextPreview;
128
+ }
129
+ /** A permission request/decision — a first-class sibling, NOT a fake tool. */
130
+ export interface PermissionExecution extends ExecutionBase {
131
+ executionType: 'permission';
132
+ lane: 'permission';
133
+ requestId?: string;
134
+ permissionKind: 'command' | 'filesystem' | 'network' | 'mcp' | 'other';
135
+ decision: 'approved' | 'denied' | 'cancelled' | 'pending';
136
+ scope?: string;
137
+ reason?: TextPreview;
138
+ }
139
+ /** A hook firing — a first-class sibling, NOT a fake tool. */
140
+ export interface HookExecution extends ExecutionBase {
141
+ executionType: 'hook';
142
+ lane: 'hook';
143
+ hookName?: string;
144
+ hookEvent?: string;
145
+ phase: 'pre' | 'post' | 'session' | 'other';
146
+ decision: 'allowed' | 'blocked' | 'error' | 'unknown';
147
+ result: ExecutionResult;
148
+ }
149
+ export type SessionStepV2 = ThinkingStep | BashExecution | EditExecution | WriteExecution | ReadExecution | GrepExecution | GenericToolExecution | PermissionExecution | HookExecution;
@@ -0,0 +1,20 @@
1
+ /* ────────────────────────────────────────────────────────────────────────
2
+ * Schema 2 — canonical ToolExecution session detail (PHNX-3442, producer side).
3
+ *
4
+ * This is the PRODUCER's authoring contract for the `schema: 2` per-session
5
+ * detail shard. It mirrors, field for field, the CONSUMER contract already
6
+ * shipped and live in `prix/web/lib/traces/types.ts` (SessionDetailV2). The
7
+ * split is deliberate: the command / patch / output PARSING lives here in
8
+ * agents-cli — the worker stores the shard opaquely and the console reads the
9
+ * union directly and NEVER reparses (spec §3, §5).
10
+ *
11
+ * Step 1 (the consumer decoder that reads BOTH schema 1 and schema 2) is merged
12
+ * and live in prod, so emitting schema 2 from here is safe: an old console
13
+ * would still read it, a new one reads it richly.
14
+ *
15
+ * This module is TYPES ONLY. The per-tool mappers that populate these shapes
16
+ * (bash argv + classification, edit/write hunks + the cross-step revert ledger,
17
+ * read/grep counts, first-class permission/hook steps) land in follow-up
18
+ * increments alongside `buildSessionDetailV2`.
19
+ * ──────────────────────────────────────────────────────────────────────── */
20
+ export {};
@@ -23,6 +23,17 @@ import { type SessionTrajectory } from '../session/trajectory.js';
23
23
  import { type BucketStats, type DriftSignal, type TraceFailureCause, type TraceTopicGroup } from './classify.js';
24
24
  import { type FailurePattern } from './insights.js';
25
25
  import type { LatencyInsight } from './segments.js';
26
+ import { buildSessionDetailV2 } from './schema2-build.js';
27
+ import type { SessionEvent } from '../session/types.js';
28
+ /**
29
+ * The per-session shard body: the schema-2 rich `ToolExecution` detail. The
30
+ * console decoder (prix/web) reads BOTH schema 1 and schema 2 and is the only
31
+ * shard consumer, so emitting schema 2 is backward-compatible by construction —
32
+ * no rollout flag is needed and none should exist (a producer that runs across
33
+ * the fleet must not depend on an operator setting an env var). Both the upload
34
+ * and dry-run paths call this one builder.
35
+ */
36
+ export declare function buildSessionShard(traj: SessionTrajectory, events: SessionEvent[], knownSecrets: readonly string[] | undefined): ReturnType<typeof buildSessionDetailV2>;
26
37
  export interface SyncOpts {
27
38
  /** Limit to N sessions (for testing / --dry-run); no limit when undefined. */
28
39
  limit?: number;
@@ -126,6 +137,57 @@ export interface TracesIndexShard {
126
137
  wastedMsTotal: number;
127
138
  /** Time-to-first-tool percentiles across this device's sessions. */
128
139
  latency: LatencyInsight;
140
+ /**
141
+ * Per-session roster — one flat scalar row per AGENT session (PHNX-3483), the
142
+ * raw material the Rush console filters and re-aggregates live. Every `stats`
143
+ * figure above is a pre-rolled scalar over the whole agent corpus; the console
144
+ * cannot re-derive a filtered headline (e.g. "median for `claude` only") from a
145
+ * scalar, so it needs the underlying rows. `durationMs` is the ACTIVE duration
146
+ * (`sessionActiveMs`, the same value backing `stats.medianMs`; 0 when the span is
147
+ * unmeasured), and `mode` encodes the AGENT-vs-INTERACTIVE segmentation
148
+ * (`headless` = an agent run, `interactive` = a one-shot query) — the SAME
149
+ * partition behind `stats.agentMedianMs` / `stats.interactiveMedianMs`, so a
150
+ * mode-split median over the MEASURED rows reproduces them. The segmented stats
151
+ * skip unmeasured (null-duration) sessions, which the roster still carries at
152
+ * `durationMs: 0`, so a consumer reproducing the medians must exclude those the
153
+ * same way (`measuredFraction` reports the covered share). Utility rows are
154
+ * excluded, exactly like every other index statistic, so the roster length equals
155
+ * the agent count (`stats.sessionsImported`). Optional here for schema
156
+ * compatibility with a shard produced before this field.
157
+ */
158
+ sessions?: SessionRosterRow[];
159
+ }
160
+ /**
161
+ * One per-session row in the index roster (PHNX-3483) — flat scalars only, so the
162
+ * Rush console can filter the session set and re-aggregate the headline metrics
163
+ * client-side without re-parsing transcripts. Built over AGENT rows only.
164
+ */
165
+ export interface SessionRosterRow {
166
+ id: string;
167
+ /** `label ?? topic ?? classified-topic label`, secret-redacted. */
168
+ title: string;
169
+ /** The producing harness (`row.agent`). */
170
+ harness: string;
171
+ model: string;
172
+ /** Short repo name from `project` / `cwd` basename / `git_branch`. */
173
+ repo: string;
174
+ /**
175
+ * `headless` = an agent run (any tool call OR more than 8 messages),
176
+ * `interactive` = a one-shot query — the SAME split behind
177
+ * `stats.agentMedianMs` / `stats.interactiveMedianMs`.
178
+ */
179
+ mode: 'interactive' | 'headless';
180
+ /** Corpus topic group of the session; `code` when unclassified. */
181
+ projectType: TraceTopicGroup;
182
+ /** Session start, epoch ms. */
183
+ startedAt: number;
184
+ /** ACTIVE duration in ms (`sessionActiveMs`); 0 when the span is unmeasured. */
185
+ durationMs: number;
186
+ toolCount: number;
187
+ errorCount: number;
188
+ needsAttention: boolean;
189
+ /** Best-effort — omitted when the source figure is unavailable. */
190
+ costUsd?: number;
129
191
  }
130
192
  export interface IndexedSession {
131
193
  id: string;
@@ -227,6 +289,8 @@ export interface ToolCallRow {
227
289
  * per-session parse the detail view does).
228
290
  */
229
291
  export declare function sessionActiveMs(spanMs: number, sessionCalls: ToolCallRow[], sessionStartMs: number): number;
292
+ /** Active time from an already-built trajectory: span minus its idle gaps (all > threshold). */
293
+ export declare function activeMsFromTrajectory(traj: SessionTrajectory): number;
230
294
  /** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
231
295
  export declare function failureDescription(call: ToolCallRow, cause: TraceFailureCause): string;
232
296
  /** Build the redacted rich console shard from indexed metadata and derived caches. */
@@ -272,11 +336,40 @@ export interface SessionDetail {
272
336
  detail?: string;
273
337
  }>;
274
338
  }
339
+ /** Plain-language summary of the friction in a run, or null when it ran clean. */
340
+ export declare function buildWhereItWentWrong(traj: SessionTrajectory): string | null;
341
+ /**
342
+ * Truthful run-level outcome (PHNX-3387).
343
+ *
344
+ * A run with zero tool errors `completed`. A run that hit tool errors is
345
+ * `completed` ONLY when it *causally recovered* — a substantive, non-human-facing
346
+ * tool step succeeded strictly after the last error AND resolved the failed work
347
+ * (its work signature matches an errored step's), the exact predicate the
348
+ * false-termination phenotype uses ({@link recoveredAfterErrors}). A run whose
349
+ * last substantive step is the error, whose only post-error steps are human-facing
350
+ * (a punt to `AskUserQuestion` — the case the broken "last tool call ok" heuristic
351
+ * mislabeled `completed`), or whose only post-error success is unrelated work (a
352
+ * failed `bun test` followed by an incidental `ls`) stays `errored`.
353
+ *
354
+ * This is what makes `surfacedToolFailures` on a `completed` run honest: those
355
+ * are failures the run recovered from, not a green status hiding an unresolved
356
+ * failure. It never flips a run that ended unresolved to `completed` (no
357
+ * regression vs the old `errorCount > 0 ? errored : completed`), and it does not
358
+ * flip a run whose failed work was never resolved just because some later,
359
+ * unrelated call happened to succeed.
360
+ */
361
+ export declare function deriveRunOutcome(traj: SessionTrajectory): 'completed' | 'errored';
275
362
  /**
276
363
  * Map the derived trajectory to the console's SessionDetail shape, stripping
277
364
  * local-machine PII (full cwd, account) that would expose filesystem paths if
278
365
  * written to R2. `repo` is the cwd basename only.
279
366
  */
367
+ /**
368
+ * The `meta` block shared by the schema-1 {@link SessionDetail} and the schema-2
369
+ * `SessionDetailV2`. Strips local-machine PII (full cwd, account): `repo` is the
370
+ * cwd basename only. Factored so both producers stamp identical meta.
371
+ */
372
+ export declare function buildDetailMeta(traj: SessionTrajectory): SessionDetail['meta'];
280
373
  export declare function buildSessionDetail(traj: SessionTrajectory): SessionDetail;
281
374
  /** How a session failed to sync. Only parse/upload failures are re-queried every run. */
282
375
  export type SyncFailureKind = 'transcript-unavailable' | 'parse-failed' | 'upload-failed';
@@ -32,6 +32,18 @@ import { resolveTracesBackend } from './backend.js';
32
32
  import { classifyCause, classifyTopic, computeDriftSignal, } from './classify.js';
33
33
  import { computeInsights } from './insights.js';
34
34
  import { classifyPhenotype, recoveredAfterErrors } from './phenotype.js';
35
+ import { buildSessionDetailV2 } from './schema2-build.js';
36
+ /**
37
+ * The per-session shard body: the schema-2 rich `ToolExecution` detail. The
38
+ * console decoder (prix/web) reads BOTH schema 1 and schema 2 and is the only
39
+ * shard consumer, so emitting schema 2 is backward-compatible by construction —
40
+ * no rollout flag is needed and none should exist (a producer that runs across
41
+ * the fleet must not depend on an operator setting an env var). Both the upload
42
+ * and dry-run paths call this one builder.
43
+ */
44
+ export function buildSessionShard(traj, events, knownSecrets) {
45
+ return buildSessionDetailV2(traj, events, { redact: true, knownSecrets });
46
+ }
35
47
  /** Push derived, redacted trajectories for this device to the traces store. */
36
48
  export async function syncTraces(opts = {}) {
37
49
  const dryRun = opts.dryRun === true;
@@ -114,9 +126,10 @@ export async function syncTraces(opts = {}) {
114
126
  continue;
115
127
  }
116
128
  let traj;
129
+ let events = [];
117
130
  try {
118
131
  const session = rowToMeta(row);
119
- const events = parseSession(row.file_path, row.agent);
132
+ events = parseSession(row.file_path, row.agent);
120
133
  traj = buildTrajectory(events, session, { redact: true, knownSecrets });
121
134
  }
122
135
  catch (err) {
@@ -134,10 +147,10 @@ export async function syncTraces(opts = {}) {
134
147
  }
135
148
  try {
136
149
  if (dryRun && outDir) {
137
- fs.writeFileSync(path.join(outDir, 'sessions', `${row.id}.json`), JSON.stringify(buildSessionDetail(traj)));
150
+ fs.writeFileSync(path.join(outDir, 'sessions', `${row.id}.json`), JSON.stringify(buildSessionShard(traj, events, knownSecrets)));
138
151
  }
139
152
  else {
140
- await putSessionTrace(backend, device, row.id, traj);
153
+ await putSessionTrace(backend, device, row.id, traj, events, knownSecrets);
141
154
  }
142
155
  uploaded++;
143
156
  maxSuccessMtime = Math.max(maxSuccessMtime, row.file_mtime_ms ?? 0);
@@ -312,7 +325,7 @@ export function sessionActiveMs(spanMs, sessionCalls, sessionStartMs) {
312
325
  return Math.max(0, spanMs - Math.min(idleMs, spanMs));
313
326
  }
314
327
  /** Active time from an already-built trajectory: span minus its idle gaps (all > threshold). */
315
- function activeMsFromTrajectory(traj) {
328
+ export function activeMsFromTrajectory(traj) {
316
329
  const idleMs = traj.gaps.reduce((sum, gap) => sum + gap.durationMs, 0);
317
330
  return Math.max(0, traj.spanMs - Math.min(idleMs, traj.spanMs));
318
331
  }
@@ -573,6 +586,36 @@ export function buildIndexShard(rows, device, owner, prevShard) {
573
586
  const bucketHistory = [...prevHistory, todayStats].slice(-14);
574
587
  const driftSignals = computeDriftSignal(prevHistory, todayStats);
575
588
  const patternInsights = computeInsights(agentRows, agentCalls, prevShard, phenotypes);
589
+ // Per-session roster (PHNX-3483): one flat scalar row per agent session, the raw
590
+ // material the Rush console filters and re-aggregates client-side. `durationMs`
591
+ // reuses `sessionActiveMs` (the value behind `stats.medianMs`; 0 for a null-duration
592
+ // row, which the segmented stats above skip entirely), and `mode` reuses the
593
+ // AGENT-vs-INTERACTIVE predicate from the segmentation above so a mode-split median
594
+ // over the MEASURED rows reproduces `stats.agentMedianMs` / `interactiveMedianMs`.
595
+ const needsAttentionIds = new Set(needsAttention.map((s) => s.id));
596
+ const sessions = agentRows.map((row) => {
597
+ const isAgent = (callsBySession.get(row.id)?.length ?? 0) > 0 || (row.message_count ?? 0) > 8;
598
+ const durationMs = row.duration_ms == null
599
+ ? 0
600
+ : sessionActiveMs(row.duration_ms, callsBySession.get(row.id) ?? [], Date.parse(row.timestamp));
601
+ const rosterRow = {
602
+ id: row.id,
603
+ title: redactSecrets(row.label ?? row.topic ?? topics.get(row.id)?.label ?? 'Untitled session', knownSecrets),
604
+ harness: row.agent,
605
+ model: row.model ?? 'unknown',
606
+ repo: row.project ?? (row.cwd ? path.basename(row.cwd) : (row.git_branch ?? 'unknown')),
607
+ mode: isAgent ? 'headless' : 'interactive',
608
+ projectType: topics.get(row.id)?.group ?? 'code',
609
+ startedAt: Date.parse(row.timestamp) || 0,
610
+ durationMs,
611
+ toolCount: row.tool_call_count ?? 0,
612
+ errorCount: errorCounts.get(row.id) ?? 0,
613
+ needsAttention: needsAttentionIds.has(row.id),
614
+ };
615
+ if (row.cost_usd != null)
616
+ rosterRow.costUsd = row.cost_usd;
617
+ return rosterRow;
618
+ });
576
619
  return {
577
620
  schema: 1,
578
621
  device,
@@ -618,10 +661,11 @@ export function buildIndexShard(rows, device, owner, prevShard) {
618
661
  failurePatterns: patternInsights.failurePatterns,
619
662
  wastedMsTotal: patternInsights.wastedMsTotal,
620
663
  latency: patternInsights.latency,
664
+ sessions,
621
665
  };
622
666
  }
623
667
  /** Plain-language summary of the friction in a run, or null when it ran clean. */
624
- function buildWhereItWentWrong(traj) {
668
+ export function buildWhereItWentWrong(traj) {
625
669
  const errorSteps = traj.steps.filter((s) => s.outcome === 'error');
626
670
  const biggestGap = traj.gaps.reduce((max, g) => (!max || g.durationMs > max.durationMs ? g : max), null);
627
671
  const parts = [];
@@ -657,7 +701,7 @@ function buildWhereItWentWrong(traj) {
657
701
  * flip a run whose failed work was never resolved just because some later,
658
702
  * unrelated call happened to succeed.
659
703
  */
660
- function deriveRunOutcome(traj) {
704
+ export function deriveRunOutcome(traj) {
661
705
  if (traj.errorCount === 0)
662
706
  return 'completed';
663
707
  return recoveredAfterErrors({ steps: traj.steps }) ? 'completed' : 'errored';
@@ -667,26 +711,35 @@ function deriveRunOutcome(traj) {
667
711
  * local-machine PII (full cwd, account) that would expose filesystem paths if
668
712
  * written to R2. `repo` is the cwd basename only.
669
713
  */
670
- export function buildSessionDetail(traj) {
714
+ /**
715
+ * The `meta` block shared by the schema-1 {@link SessionDetail} and the schema-2
716
+ * `SessionDetailV2`. Strips local-machine PII (full cwd, account): `repo` is the
717
+ * cwd basename only. Factored so both producers stamp identical meta.
718
+ */
719
+ export function buildDetailMeta(traj) {
671
720
  const s = traj.session;
672
721
  const stats = traj.stats;
673
722
  const repo = s.project ?? (s.cwd ? path.basename(s.cwd) : 'unknown');
723
+ return {
724
+ spanMs: traj.spanMs,
725
+ activeMs: activeMsFromTrajectory(traj),
726
+ turns: (stats.userTurns ?? 0) + (stats.assistantTurns ?? 0),
727
+ tools: stats.toolCount ?? 0,
728
+ errorCount: traj.errorCount,
729
+ tokens: stats.outputTokens ?? 0,
730
+ costUsd: s.costUsd ?? 0,
731
+ outcome: deriveRunOutcome(traj),
732
+ repo,
733
+ agent: s.agent,
734
+ model: s.model ?? 'unknown',
735
+ };
736
+ }
737
+ export function buildSessionDetail(traj) {
738
+ const s = traj.session;
674
739
  return {
675
740
  schema: 1,
676
741
  id: s.id,
677
- meta: {
678
- spanMs: traj.spanMs,
679
- activeMs: activeMsFromTrajectory(traj),
680
- turns: (stats.userTurns ?? 0) + (stats.assistantTurns ?? 0),
681
- tools: stats.toolCount ?? 0,
682
- errorCount: traj.errorCount,
683
- tokens: stats.outputTokens ?? 0,
684
- costUsd: s.costUsd ?? 0,
685
- outcome: deriveRunOutcome(traj),
686
- repo,
687
- agent: s.agent,
688
- model: s.model ?? 'unknown',
689
- },
742
+ meta: buildDetailMeta(traj),
690
743
  steps: traj.steps,
691
744
  gaps: traj.gaps,
692
745
  truncatedSteps: traj.truncatedSteps,
@@ -710,7 +763,7 @@ async function getIndexShard(backend, device) {
710
763
  return null;
711
764
  }
712
765
  }
713
- async function putSessionTrace(backend, device, sessionId, traj) {
766
+ async function putSessionTrace(backend, device, sessionId, traj, events, knownSecrets) {
714
767
  const url = `${backend.baseUrl}/${backend.userId}/${device}/sessions/${sessionId}.json`;
715
768
  const res = await fetch(url, {
716
769
  method: 'PUT',
@@ -718,7 +771,7 @@ async function putSessionTrace(backend, device, sessionId, traj) {
718
771
  authorization: `Bearer ${backend.token}`,
719
772
  'content-type': 'application/json; charset=utf-8',
720
773
  },
721
- body: JSON.stringify(buildSessionDetail(traj)),
774
+ body: JSON.stringify(buildSessionShard(traj, events, knownSecrets)),
722
775
  });
723
776
  if (!res.ok) {
724
777
  throw new Error(`PUT ${url} → ${res.status}`);
@@ -225,6 +225,11 @@ function mergeIndexShards(shards, owner) {
225
225
  failurePatterns: Array.from(patternById.values())
226
226
  .sort((a, b) => (b.wastedMs || 0) - (a.wastedMs || 0)).slice(0, 25),
227
227
  wastedMsTotal: sorted.reduce((n, s) => n + (s.wastedMsTotal || 0), 0),
228
+ // Per-session roster (PHNX-3483): concat every device's rows so the "all"
229
+ // view can filter + re-aggregate client-side exactly like a single device.
230
+ // Devices still on a pre-roster CLI contribute none (|| []) and degrade to
231
+ // the pre-rolled stats, so coverage grows as devices update — never crashes.
232
+ sessions: sorted.flatMap((s) => s.sessions || []),
228
233
  latency: latencies.length ? {
229
234
  firstToolMs: {
230
235
  p50: Math.round(wsum((s) => s.latency && s.latency.firstToolMs && s.latency.firstToolMs.p50)),