@phnx-labs/agents-cli 1.22.57 → 1.22.59
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +294 -0
- package/README.md +29 -0
- package/dist/bootstrap.js +39 -1
- package/dist/commands/accounts.js +7 -3
- package/dist/commands/apply.js +10 -2
- package/dist/commands/fork.d.ts +23 -10
- package/dist/commands/fork.js +115 -58
- package/dist/commands/monitors.js +198 -23
- package/dist/commands/prune.js +5 -3
- package/dist/commands/routines.d.ts +8 -0
- package/dist/commands/routines.js +57 -3
- package/dist/commands/routines.test-fixture.js +5 -0
- package/dist/commands/send.d.ts +2 -1
- package/dist/commands/send.js +7 -5
- package/dist/commands/sessions-picker.d.ts +11 -0
- package/dist/commands/sessions-picker.js +16 -0
- package/dist/commands/sessions-stats.js +37 -5
- package/dist/commands/sessions.js +40 -5
- package/dist/commands/share.d.ts +14 -0
- package/dist/commands/share.js +43 -2
- package/dist/commands/ssh.js +12 -1
- package/dist/commands/status.js +1 -1
- package/dist/commands/sync.js +83 -7
- package/dist/commands/traces.js +7 -0
- package/dist/commands/versions.js +12 -4
- package/dist/commands/view.js +7 -2
- package/dist/index.d.ts +1 -1
- package/dist/index.js +6 -1
- package/dist/lib/account-registry.d.ts +5 -1
- package/dist/lib/account-registry.js +47 -14
- package/dist/lib/accounting/capacity.d.ts +18 -7
- package/dist/lib/accounting/capacity.js +19 -8
- package/dist/lib/accounting/usage-sync.d.ts +29 -1
- package/dist/lib/accounting/usage-sync.js +76 -2
- package/dist/lib/accounting/usage.js +7 -1
- package/dist/lib/auth-mint.d.ts +11 -1
- package/dist/lib/auth-mint.js +21 -6
- package/dist/lib/auto-pull-worker.js +7 -2
- package/dist/lib/browser/ipc.d.ts +8 -0
- package/dist/lib/browser/ipc.js +87 -0
- package/dist/lib/browser/service.d.ts +19 -0
- package/dist/lib/browser/service.js +96 -11
- package/dist/lib/browser/sessions-list.js +10 -1
- package/dist/lib/cloud/rush.d.ts +7 -0
- package/dist/lib/cloud/rush.js +29 -1
- package/dist/lib/daemon/daemon.d.ts +22 -0
- package/dist/lib/daemon/daemon.js +39 -0
- package/dist/lib/daemon/runner.d.ts +3 -0
- package/dist/lib/daemon/runner.js +86 -45
- package/dist/lib/daemon/session-index-service.js +9 -1
- package/dist/lib/daemon/usage-sync-service.d.ts +3 -3
- package/dist/lib/daemon/usage-sync-service.js +14 -8
- package/dist/lib/daemon-services.js +1 -1
- package/dist/lib/daemon-ticks.d.ts +15 -0
- package/dist/lib/daemon-ticks.js +26 -0
- package/dist/lib/device-config.d.ts +5 -1
- package/dist/lib/device-config.js +2 -2
- package/dist/lib/devices/connect.d.ts +17 -8
- package/dist/lib/devices/connect.js +31 -14
- package/dist/lib/devices/health.js +5 -1
- package/dist/lib/devices/pool.d.ts +25 -2
- package/dist/lib/devices/pool.js +32 -2
- package/dist/lib/devices/stats-cache.d.ts +0 -6
- package/dist/lib/devices/stats-cache.js +2 -9
- package/dist/lib/doctor-diff.d.ts +14 -0
- package/dist/lib/doctor-diff.js +120 -9
- package/dist/lib/fleet/manifest.d.ts +17 -0
- package/dist/lib/fleet/manifest.js +26 -0
- package/dist/lib/git.d.ts +38 -0
- package/dist/lib/git.js +58 -0
- package/dist/lib/hooks/install.d.ts +27 -11
- package/dist/lib/hooks/install.js +42 -17
- package/dist/lib/hosts/ready.d.ts +8 -0
- package/dist/lib/hosts/ready.js +13 -2
- package/dist/lib/hosts/reconnect.d.ts +52 -203
- package/dist/lib/hosts/reconnect.js +64 -284
- package/dist/lib/installations/migrate.d.ts +6 -120
- package/dist/lib/installations/migrate.js +27 -259
- package/dist/lib/installations/shims.d.ts +13 -95
- package/dist/lib/installations/shims.js +22 -139
- package/dist/lib/installations/store.js +1 -1
- package/dist/lib/installations/versions.d.ts +43 -133
- package/dist/lib/installations/versions.js +94 -206
- package/dist/lib/monitors/config.d.ts +71 -3
- package/dist/lib/monitors/config.js +100 -12
- package/dist/lib/monitors/pid-watch.d.ts +35 -0
- package/dist/lib/monitors/pid-watch.js +45 -0
- package/dist/lib/monitors/remote.d.ts +18 -0
- package/dist/lib/monitors/remote.js +11 -0
- package/dist/lib/permissions.js +7 -2
- package/dist/lib/plugins/plugins.d.ts +17 -3
- package/dist/lib/plugins/plugins.js +84 -9
- package/dist/lib/plugins/skills.d.ts +8 -1
- package/dist/lib/plugins/skills.js +18 -2
- package/dist/lib/pty-server.d.ts +14 -0
- package/dist/lib/pty-server.js +49 -5
- package/dist/lib/refresh.d.ts +9 -0
- package/dist/lib/refresh.js +3 -1
- package/dist/lib/routine-readiness.d.ts +15 -1
- package/dist/lib/routine-readiness.js +41 -0
- package/dist/lib/sandbox.d.ts +4 -1
- package/dist/lib/sandbox.js +30 -1
- package/dist/lib/secrets/agent.d.ts +80 -225
- package/dist/lib/secrets/agent.js +139 -401
- package/dist/lib/secrets/bundles.d.ts +73 -222
- package/dist/lib/secrets/bundles.js +168 -467
- package/dist/lib/secrets/drivers/rush.js +5 -0
- package/dist/lib/secrets/reaper.d.ts +28 -70
- package/dist/lib/secrets/reaper.js +30 -85
- package/dist/lib/secrets/remote.d.ts +42 -129
- package/dist/lib/secrets/remote.js +55 -173
- package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
- package/dist/lib/self-heal/checks/install-staging.js +96 -0
- package/dist/lib/self-heal/registry.js +2 -0
- package/dist/lib/self-heal/types.d.ts +1 -1
- package/dist/lib/self-update.d.ts +65 -0
- package/dist/lib/self-update.js +138 -0
- package/dist/lib/session/active.d.ts +13 -1
- package/dist/lib/session/active.js +2 -0
- package/dist/lib/session/cloud.js +5 -0
- package/dist/lib/session/db.d.ts +51 -6
- package/dist/lib/session/db.js +266 -20
- package/dist/lib/session/fork.d.ts +45 -26
- package/dist/lib/session/fork.js +32 -95
- package/dist/lib/session/tool-calls.d.ts +43 -1
- package/dist/lib/session/tool-calls.js +74 -44
- package/dist/lib/session/tool-store.d.ts +33 -2
- package/dist/lib/session/tool-store.js +56 -3
- package/dist/lib/smart-launch.d.ts +6 -0
- package/dist/lib/smart-launch.js +5 -2
- package/dist/lib/staleness/writers/plugins.js +5 -2
- package/dist/lib/staleness/writers/sources.d.ts +5 -0
- package/dist/lib/staleness/writers/sources.js +2 -1
- package/dist/lib/staleness/writers/subagents.js +13 -3
- package/dist/lib/state.d.ts +7 -4
- package/dist/lib/state.js +7 -4
- package/dist/lib/subagents.js +8 -2
- package/dist/lib/sync-status.d.ts +22 -0
- package/dist/lib/sync-status.js +27 -0
- package/dist/lib/sync-umbrella.d.ts +9 -0
- package/dist/lib/sync-umbrella.js +21 -2
- package/dist/lib/teams/scheduler.d.ts +10 -0
- package/dist/lib/teams/scheduler.js +8 -0
- package/dist/lib/traces/insights.d.ts +47 -14
- package/dist/lib/traces/insights.js +92 -21
- package/dist/lib/traces/phenotype.d.ts +23 -3
- package/dist/lib/traces/phenotype.js +72 -24
- package/dist/lib/traces/sync.d.ts +128 -6
- package/dist/lib/traces/sync.js +294 -35
- package/dist/lib/traces/worker-template.js +154 -1
- package/dist/lib/view-types.d.ts +12 -0
- package/package.json +2 -2
|
@@ -49,6 +49,14 @@ export interface SyncResult {
|
|
|
49
49
|
parseFailed: number;
|
|
50
50
|
/** Parsed fine but the upload PUT failed (network/5xx) — retried on the next sync. */
|
|
51
51
|
uploadFailed: number;
|
|
52
|
+
/**
|
|
53
|
+
* The index shard build or upload failed. Undefined on success. The per-session
|
|
54
|
+
* data still uploaded (that loop runs first), but the aggregated console shard —
|
|
55
|
+
* stats, needs-attention, failure clusters, latency — was NOT refreshed, so the
|
|
56
|
+
* console keeps serving the last good index. Surfaced (not swallowed) so a stale
|
|
57
|
+
* console is diagnosable instead of looking like a clean sync. See PHNX-3401.
|
|
58
|
+
*/
|
|
59
|
+
indexError?: string;
|
|
52
60
|
}
|
|
53
61
|
/** Push derived, redacted trajectories for this device to the traces store. */
|
|
54
62
|
export declare function syncTraces(opts?: SyncOpts): Promise<SyncResult>;
|
|
@@ -59,18 +67,46 @@ export interface TracesIndexShard {
|
|
|
59
67
|
owner: string;
|
|
60
68
|
stats: {
|
|
61
69
|
sessionsImported: number;
|
|
70
|
+
/**
|
|
71
|
+
* Median ACTIVE duration (span − idle gaps > 120s), ms — the meaningful figure
|
|
72
|
+
* (PHNX-3457). Same key/shape as before this change, so the fleet-aggregate
|
|
73
|
+
* worker (`worker-template.ts`) keeps weighted-averaging it unchanged; only its
|
|
74
|
+
* VALUE moved from raw span to active time. The raw span stays available per
|
|
75
|
+
* session on `SessionDetail.meta.spanMs`.
|
|
76
|
+
*/
|
|
62
77
|
medianMs: number;
|
|
78
|
+
/** p90 ACTIVE duration, ms. */
|
|
63
79
|
p90Ms: number;
|
|
80
|
+
/**
|
|
81
|
+
* SEGMENTED active-time stats (PHNX-3472). The blended `medianMs`/`p90Ms`
|
|
82
|
+
* above conflate one-shot interactive queries (63% of the corpus, ~15s
|
|
83
|
+
* median) with substantial agent runs (~15min median), so they headline
|
|
84
|
+
* neither. A session is an AGENT run when it made any tool call OR has more
|
|
85
|
+
* than 8 messages; otherwise INTERACTIVE. These segment the same active-time
|
|
86
|
+
* figure so the console can headline agent runs on their own axis. Each is
|
|
87
|
+
* computed only over sessions with a non-null duration.
|
|
88
|
+
*/
|
|
89
|
+
agentMedianMs: number;
|
|
90
|
+
/** p90 ACTIVE duration over AGENT sessions, ms. */
|
|
91
|
+
agentP90Ms: number;
|
|
92
|
+
/** Median ACTIVE duration over INTERACTIVE sessions, ms. */
|
|
93
|
+
interactiveMedianMs: number;
|
|
94
|
+
/** (sessions with a non-null duration) / (total sessions), 0..1 — coverage of the duration stats. */
|
|
95
|
+
measuredFraction: number;
|
|
64
96
|
needAttention: number;
|
|
65
97
|
toolErrorRate: number;
|
|
66
98
|
};
|
|
99
|
+
/**
|
|
100
|
+
* Sessions excluded from the eval corpus as internal utility plumbing (PHNX-3474):
|
|
101
|
+
* single-shot machine calls (no tool call AND ≤2 messages) or a known
|
|
102
|
+
* internal-prompt signature (title generation, watchdog, commit-message, factory
|
|
103
|
+
* worker). Every `stats` figure above, `topics` counts, and `needsAttention` are
|
|
104
|
+
* computed over the AGENT set ONLY — `sessionsImported` is the real agent count,
|
|
105
|
+
* not the raw row count. This is the number that was dropped.
|
|
106
|
+
*/
|
|
107
|
+
utilityCount: number;
|
|
67
108
|
needsAttention: IndexedSession[];
|
|
68
|
-
topics:
|
|
69
|
-
key: string;
|
|
70
|
-
label: string;
|
|
71
|
-
count: number;
|
|
72
|
-
group: TraceTopicGroup;
|
|
73
|
-
}>;
|
|
109
|
+
topics: TopicItem[];
|
|
74
110
|
failures: {
|
|
75
111
|
byToolError: Array<{
|
|
76
112
|
tool: string;
|
|
@@ -96,16 +132,68 @@ export interface IndexedSession {
|
|
|
96
132
|
title: string;
|
|
97
133
|
repo: string;
|
|
98
134
|
device: string;
|
|
135
|
+
/** The harness that produced the session (claude/codex/rush/grok/…). Same as `harness`. */
|
|
99
136
|
agent: string;
|
|
100
137
|
model: string;
|
|
138
|
+
/** Corpus classification (PHNX-3474). Always `'agent'` here — utility rows are excluded. */
|
|
139
|
+
kind: SessionKind;
|
|
101
140
|
severity: number;
|
|
102
141
|
flags: string[];
|
|
103
142
|
}
|
|
143
|
+
/**
|
|
144
|
+
* One example session under a topic tile — the shape the console drill-down consumes.
|
|
145
|
+
* Carries `kind` + `harness` (PHNX-3474) so the console can filter a tile's session
|
|
146
|
+
* list by corpus class and by harness. Refs on a topic tile are always `'agent'`
|
|
147
|
+
* (utility rows never reach a bucket), but the field is explicit for the consumer.
|
|
148
|
+
*/
|
|
149
|
+
export interface TopicSessionRef {
|
|
150
|
+
id: string;
|
|
151
|
+
title: string;
|
|
152
|
+
kind: SessionKind;
|
|
153
|
+
harness: string;
|
|
154
|
+
}
|
|
155
|
+
/**
|
|
156
|
+
* Corpus class of a session (PHNX-3474). `utility` is internal machine plumbing —
|
|
157
|
+
* a single-shot call with no tool use and ≤2 messages, or one whose topic/label
|
|
158
|
+
* matches a known internal-prompt signature (title generation, watchdog,
|
|
159
|
+
* commit-message writer, factory worker). Everything else is `agent`: real agent
|
|
160
|
+
* work the Evals console counts and scores. Utility rows are tagged, never deleted,
|
|
161
|
+
* and excluded from every index statistic.
|
|
162
|
+
*/
|
|
163
|
+
export type SessionKind = 'utility' | 'agent';
|
|
164
|
+
/**
|
|
165
|
+
* Classify a session as internal `utility` plumbing vs real `agent` work (PHNX-3474).
|
|
166
|
+
* `toolCallCount` is the AUTHORITATIVE per-session tool-call count from the loaded
|
|
167
|
+
* `tool_calls` rows (the row's own `tool_call_count` column is a fallback for a row
|
|
168
|
+
* whose calls weren't loaded). A session is `utility` when a known internal-prompt
|
|
169
|
+
* signature matches its topic/label, OR it made no tool call AND has ≤2 messages —
|
|
170
|
+
* the single-shot machine-call shape. Otherwise it is `agent`.
|
|
171
|
+
*/
|
|
172
|
+
export declare function classifySessionKind(row: Pick<SyncRow, 'topic' | 'label' | 'message_count' | 'tool_call_count'>, toolCallCount: number): SessionKind;
|
|
173
|
+
/**
|
|
174
|
+
* One topic bucket in the treemap. `sessions` carries up to {@link TOPIC_SESSION_CAP}
|
|
175
|
+
* example refs so the console can drill from the tile into its session list — a tile
|
|
176
|
+
* with no refs renders display-only (PHNX-3408). `count` stays the true total.
|
|
177
|
+
*/
|
|
178
|
+
export interface TopicItem {
|
|
179
|
+
key: string;
|
|
180
|
+
label: string;
|
|
181
|
+
count: number;
|
|
182
|
+
group: TraceTopicGroup;
|
|
183
|
+
sessions: TopicSessionRef[];
|
|
184
|
+
}
|
|
104
185
|
/** A row from `tool_calls`. `ordinal`/`timestamp` order calls within a session for computeInsights(). */
|
|
105
186
|
export interface ToolCallRow {
|
|
106
187
|
session_id: string;
|
|
107
188
|
ordinal: number;
|
|
108
189
|
timestamp: string;
|
|
190
|
+
/**
|
|
191
|
+
* When the call's result arrived — its own end time (PHNX-3437). Optional
|
|
192
|
+
* because rows an older extractor stored, and calls that never produced a
|
|
193
|
+
* result, carry NULL; `computeInsights` falls back to the bounded inter-call
|
|
194
|
+
* gap when it is absent.
|
|
195
|
+
*/
|
|
196
|
+
end_timestamp?: string | null;
|
|
109
197
|
tool: string;
|
|
110
198
|
outcome: string;
|
|
111
199
|
exit_code: number | null;
|
|
@@ -114,6 +202,31 @@ export interface ToolCallRow {
|
|
|
114
202
|
error: string | null;
|
|
115
203
|
parse_error: string | null;
|
|
116
204
|
}
|
|
205
|
+
/**
|
|
206
|
+
* Active time for a session in the index shard: its recorded span minus every idle
|
|
207
|
+
* gap > 120s (PHNX-3457). The index build already holds the ordered `tool_calls`
|
|
208
|
+
* rows for the whole corpus, so idle is derived from them here — no transcript
|
|
209
|
+
* re-parse. A cursor sweeps the span from `sessionStartMs`: each stretch where the
|
|
210
|
+
* cursor sits idle for more than the threshold before the next call starts is
|
|
211
|
+
* subtracted, and idle is measured from a call's END (its own `end_timestamp` when
|
|
212
|
+
* known, else its start) so a call's own blocking duration is never mistaken for
|
|
213
|
+
* idle. Crucially the sweep also books the gaps at the two BOUNDARIES — before the
|
|
214
|
+
* first call and after the last call to the session end — so a session with a lone
|
|
215
|
+
* tool call that was then abandoned and resumed hours later (the case a
|
|
216
|
+
* between-calls-only measure missed entirely, leaving the whole 345h span counted
|
|
217
|
+
* as active) has that trailing idle stripped. A session end is `sessionStartMs +
|
|
218
|
+
* spanMs`, so the two agree by construction.
|
|
219
|
+
*
|
|
220
|
+
* Bounded to `[0, spanMs]`. A session with NO tool calls returns the full span
|
|
221
|
+
* unchanged rather than a fabricated zero: there is no tool-call evidence of idle
|
|
222
|
+
* either way, and treating a chat-only turn as 100% idle would be a worse error
|
|
223
|
+
* than leaving its span uncorrected. Where the full event stream IS available (a
|
|
224
|
+
* per-session `SessionDetail`), {@link activeMsFromTrajectory} is used instead —
|
|
225
|
+
* it sees message events this call-only approximation cannot, so the two are close
|
|
226
|
+
* but not identical by design (the corpus-scale index build cannot afford the
|
|
227
|
+
* per-session parse the detail view does).
|
|
228
|
+
*/
|
|
229
|
+
export declare function sessionActiveMs(spanMs: number, sessionCalls: ToolCallRow[], sessionStartMs: number): number;
|
|
117
230
|
/** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
|
|
118
231
|
export declare function failureDescription(call: ToolCallRow, cause: TraceFailureCause): string;
|
|
119
232
|
/** Build the redacted rich console shard from indexed metadata and derived caches. */
|
|
@@ -123,7 +236,16 @@ export interface SessionDetail {
|
|
|
123
236
|
schema: 1;
|
|
124
237
|
id: string;
|
|
125
238
|
meta: {
|
|
239
|
+
/** Raw wall-clock span (last event − first event), idle time included. */
|
|
126
240
|
spanMs: number;
|
|
241
|
+
/**
|
|
242
|
+
* Active time: `spanMs` minus every idle gap > 120s (PHNX-3457). A session
|
|
243
|
+
* resumed hours later, or left idle mid-turn, inflates `spanMs` with wall-clock
|
|
244
|
+
* the agent did no work in — active time strips those gaps so a duration reads
|
|
245
|
+
* as effort, not calendar span. This is what the console's duration median/p90
|
|
246
|
+
* should trust; `spanMs` stays available as the raw figure.
|
|
247
|
+
*/
|
|
248
|
+
activeMs: number;
|
|
127
249
|
turns: number;
|
|
128
250
|
tools: number;
|
|
129
251
|
errorCount: number;
|
package/dist/lib/traces/sync.js
CHANGED
|
@@ -22,7 +22,7 @@
|
|
|
22
22
|
import fs from 'node:fs';
|
|
23
23
|
import os from 'node:os';
|
|
24
24
|
import path from 'node:path';
|
|
25
|
-
import { getDB, readSessionInsights, readSessionTopics, writeSessionInsights, writeSessionTopics, } from '../session/db.js';
|
|
25
|
+
import { getDB, readSessionInsights, readSessionPhenotypes, readSessionTopics, writeSessionInsights, writeSessionPhenotypes, writeSessionTopics, } from '../session/db.js';
|
|
26
26
|
import { parseSession } from '../session/parse.js';
|
|
27
27
|
import { buildTrajectory } from '../session/trajectory.js';
|
|
28
28
|
import { computeInsightFacets } from '../session/insights.js';
|
|
@@ -31,6 +31,7 @@ import { getRuntimeStateDir } from '../state.js';
|
|
|
31
31
|
import { resolveTracesBackend } from './backend.js';
|
|
32
32
|
import { classifyCause, classifyTopic, computeDriftSignal, } from './classify.js';
|
|
33
33
|
import { computeInsights } from './insights.js';
|
|
34
|
+
import { classifyPhenotype, recoveredAfterErrors } from './phenotype.js';
|
|
34
35
|
/** Push derived, redacted trajectories for this device to the traces store. */
|
|
35
36
|
export async function syncTraces(opts = {}) {
|
|
36
37
|
const dryRun = opts.dryRun === true;
|
|
@@ -147,6 +148,7 @@ export async function syncTraces(opts = {}) {
|
|
|
147
148
|
recordFailure(row, 'upload-failed', err);
|
|
148
149
|
}
|
|
149
150
|
}
|
|
151
|
+
let indexError;
|
|
150
152
|
if (!opts.skipIndex) {
|
|
151
153
|
try {
|
|
152
154
|
const allRows = db
|
|
@@ -172,8 +174,13 @@ export async function syncTraces(opts = {}) {
|
|
|
172
174
|
await putIndexShard(backend, device, owner, shard);
|
|
173
175
|
}
|
|
174
176
|
}
|
|
175
|
-
catch {
|
|
176
|
-
//
|
|
177
|
+
catch (err) {
|
|
178
|
+
// Not fatal to the per-session upload (that loop already ran), but it DOES
|
|
179
|
+
// mean the console shard is now stale. Record it so the caller can surface a
|
|
180
|
+
// warning instead of reporting a clean, green sync — a silent swallow here is
|
|
181
|
+
// exactly what let a 59h-stale, insight-less index hide in plain sight
|
|
182
|
+
// (PHNX-3401). Do NOT re-throw: the session data is durable and worth keeping.
|
|
183
|
+
indexError = err instanceof Error ? err.message : String(err);
|
|
177
184
|
}
|
|
178
185
|
}
|
|
179
186
|
// A dry-run never advances the incremental watermark: it is a read-only export.
|
|
@@ -203,14 +210,112 @@ export async function syncTraces(opts = {}) {
|
|
|
203
210
|
transcriptUnavailable,
|
|
204
211
|
parseFailed,
|
|
205
212
|
uploadFailed,
|
|
213
|
+
indexError,
|
|
206
214
|
};
|
|
207
215
|
}
|
|
216
|
+
/**
|
|
217
|
+
* Topic/label substrings that identify an internal-prompt session regardless of its
|
|
218
|
+
* message/tool shape. These are the harness-spawned utility prompts the Rush app
|
|
219
|
+
* fires (they run under the `claude` harness): title generation writes the 3–4 word
|
|
220
|
+
* session title, the watchdog polls for stalled agents, the commit-message writer
|
|
221
|
+
* drafts a conventional commit, and factory workers are dispatched sub-agents. The
|
|
222
|
+
* title-generation prompt lives in the `topic` column, the rest can land in either
|
|
223
|
+
* `topic` or `label`, so both are matched.
|
|
224
|
+
*/
|
|
225
|
+
const UTILITY_PROMPT_SIGNATURES = [
|
|
226
|
+
/generate a 3-4 word title/i, // title generation
|
|
227
|
+
/you are a watchdog|watchdog monitoring/i, // watchdog tick
|
|
228
|
+
/conventional[- ]commit/i, // commit-message writer
|
|
229
|
+
/factory worker/i, // dispatched factory worker
|
|
230
|
+
];
|
|
231
|
+
/**
|
|
232
|
+
* Classify a session as internal `utility` plumbing vs real `agent` work (PHNX-3474).
|
|
233
|
+
* `toolCallCount` is the AUTHORITATIVE per-session tool-call count from the loaded
|
|
234
|
+
* `tool_calls` rows (the row's own `tool_call_count` column is a fallback for a row
|
|
235
|
+
* whose calls weren't loaded). A session is `utility` when a known internal-prompt
|
|
236
|
+
* signature matches its topic/label, OR it made no tool call AND has ≤2 messages —
|
|
237
|
+
* the single-shot machine-call shape. Otherwise it is `agent`.
|
|
238
|
+
*/
|
|
239
|
+
export function classifySessionKind(row, toolCallCount) {
|
|
240
|
+
const haystack = `${row.topic ?? ''}\n${row.label ?? ''}`;
|
|
241
|
+
if (UTILITY_PROMPT_SIGNATURES.some((re) => re.test(haystack)))
|
|
242
|
+
return 'utility';
|
|
243
|
+
const hasToolCalls = toolCallCount > 0 || (row.tool_call_count ?? 0) > 0;
|
|
244
|
+
const messages = row.message_count ?? 0;
|
|
245
|
+
if (!hasToolCalls && messages <= 2)
|
|
246
|
+
return 'utility';
|
|
247
|
+
return 'agent';
|
|
248
|
+
}
|
|
208
249
|
function percentile(values, ratio) {
|
|
209
250
|
if (values.length === 0)
|
|
210
251
|
return 0;
|
|
211
252
|
const sorted = [...values].sort((a, b) => a - b);
|
|
212
253
|
return sorted[Math.min(sorted.length - 1, Math.ceil(sorted.length * ratio) - 1)];
|
|
213
254
|
}
|
|
255
|
+
/**
|
|
256
|
+
* A delta between consecutive events longer than this reads as an idle stall, not
|
|
257
|
+
* work — the same threshold the trajectory uses for its gap detection
|
|
258
|
+
* (`DEFAULT_IDLE_THRESHOLD_MS`, trajectory.ts). Kept in lockstep so active time
|
|
259
|
+
* here and the gaps drawn in a session's detail view agree on what "idle" means.
|
|
260
|
+
*/
|
|
261
|
+
const IDLE_GAP_THRESHOLD_MS = 120_000;
|
|
262
|
+
/** Max example session refs carried per topic tile so a tile is drillable (PHNX-3408). */
|
|
263
|
+
const TOPIC_SESSION_CAP = 30;
|
|
264
|
+
/**
|
|
265
|
+
* Active time for a session in the index shard: its recorded span minus every idle
|
|
266
|
+
* gap > 120s (PHNX-3457). The index build already holds the ordered `tool_calls`
|
|
267
|
+
* rows for the whole corpus, so idle is derived from them here — no transcript
|
|
268
|
+
* re-parse. A cursor sweeps the span from `sessionStartMs`: each stretch where the
|
|
269
|
+
* cursor sits idle for more than the threshold before the next call starts is
|
|
270
|
+
* subtracted, and idle is measured from a call's END (its own `end_timestamp` when
|
|
271
|
+
* known, else its start) so a call's own blocking duration is never mistaken for
|
|
272
|
+
* idle. Crucially the sweep also books the gaps at the two BOUNDARIES — before the
|
|
273
|
+
* first call and after the last call to the session end — so a session with a lone
|
|
274
|
+
* tool call that was then abandoned and resumed hours later (the case a
|
|
275
|
+
* between-calls-only measure missed entirely, leaving the whole 345h span counted
|
|
276
|
+
* as active) has that trailing idle stripped. A session end is `sessionStartMs +
|
|
277
|
+
* spanMs`, so the two agree by construction.
|
|
278
|
+
*
|
|
279
|
+
* Bounded to `[0, spanMs]`. A session with NO tool calls returns the full span
|
|
280
|
+
* unchanged rather than a fabricated zero: there is no tool-call evidence of idle
|
|
281
|
+
* either way, and treating a chat-only turn as 100% idle would be a worse error
|
|
282
|
+
* than leaving its span uncorrected. Where the full event stream IS available (a
|
|
283
|
+
* per-session `SessionDetail`), {@link activeMsFromTrajectory} is used instead —
|
|
284
|
+
* it sees message events this call-only approximation cannot, so the two are close
|
|
285
|
+
* but not identical by design (the corpus-scale index build cannot afford the
|
|
286
|
+
* per-session parse the detail view does).
|
|
287
|
+
*/
|
|
288
|
+
export function sessionActiveMs(spanMs, sessionCalls, sessionStartMs) {
|
|
289
|
+
if (spanMs <= 0)
|
|
290
|
+
return Math.max(0, spanMs);
|
|
291
|
+
if (!Number.isFinite(sessionStartMs))
|
|
292
|
+
return spanMs; // can't place calls on the span
|
|
293
|
+
const spanEndMs = sessionStartMs + spanMs;
|
|
294
|
+
const ordered = sessionCalls
|
|
295
|
+
.map((c) => ({ startMs: Date.parse(c.timestamp), endMs: Date.parse(c.end_timestamp ?? c.timestamp) }))
|
|
296
|
+
.filter((c) => Number.isFinite(c.startMs))
|
|
297
|
+
.sort((a, b) => a.startMs - b.startMs);
|
|
298
|
+
if (ordered.length === 0)
|
|
299
|
+
return spanMs; // no tool-call evidence of idle
|
|
300
|
+
let idleMs = 0;
|
|
301
|
+
let cursor = sessionStartMs;
|
|
302
|
+
for (const call of ordered) {
|
|
303
|
+
if (call.startMs > cursor + IDLE_GAP_THRESHOLD_MS)
|
|
304
|
+
idleMs += call.startMs - cursor;
|
|
305
|
+
const endMs = Number.isFinite(call.endMs) ? Math.max(call.endMs, call.startMs) : call.startMs;
|
|
306
|
+
if (endMs > cursor)
|
|
307
|
+
cursor = endMs;
|
|
308
|
+
}
|
|
309
|
+
// Trailing idle: the stretch from the last call's end to the session's end.
|
|
310
|
+
if (spanEndMs > cursor + IDLE_GAP_THRESHOLD_MS)
|
|
311
|
+
idleMs += spanEndMs - cursor;
|
|
312
|
+
return Math.max(0, spanMs - Math.min(idleMs, spanMs));
|
|
313
|
+
}
|
|
314
|
+
/** Active time from an already-built trajectory: span minus its idle gaps (all > threshold). */
|
|
315
|
+
function activeMsFromTrajectory(traj) {
|
|
316
|
+
const idleMs = traj.gaps.reduce((sum, gap) => sum + gap.durationMs, 0);
|
|
317
|
+
return Math.max(0, traj.spanMs - Math.min(idleMs, traj.spanMs));
|
|
318
|
+
}
|
|
214
319
|
/** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
|
|
215
320
|
export function failureDescription(call, cause) {
|
|
216
321
|
if (cause === 'guard')
|
|
@@ -247,6 +352,32 @@ function attentionFlags(errorCount, facets) {
|
|
|
247
352
|
flags.push(`${correctionCount} correction${correctionCount === 1 ? '' : 's'}`);
|
|
248
353
|
return flags;
|
|
249
354
|
}
|
|
355
|
+
/**
|
|
356
|
+
* Persist a derived-cache warm-up (topics / insights) without letting a
|
|
357
|
+
* contended DB take down the whole index build.
|
|
358
|
+
*
|
|
359
|
+
* These write-backs only speed up the NEXT sync — the shard about to be built
|
|
360
|
+
* reads from the in-memory `topics` / `insights` maps that were already
|
|
361
|
+
* populated above, never from what this write persists. So the write is
|
|
362
|
+
* genuinely optional to the shard's correctness.
|
|
363
|
+
*
|
|
364
|
+
* Yet it was the single point that broke the console: on an active machine the
|
|
365
|
+
* Rush app holds `sessions.db`, this `BEGIN IMMEDIATE` waits out the 30s
|
|
366
|
+
* `busy_timeout` and throws `SQLITE_BUSY`, the throw escaped `buildIndexShard`,
|
|
367
|
+
* and `syncTraces` swallowed it — so the index (with wasted-time / failure
|
|
368
|
+
* clusters / latency) never re-uploaded and the dashboard sat 59h stale
|
|
369
|
+
* (PHNX-3401). Isolating the failure here keeps the index building; the warning
|
|
370
|
+
* makes the degraded cache visible instead of silent.
|
|
371
|
+
*/
|
|
372
|
+
function persistDerivedCache(label, write) {
|
|
373
|
+
try {
|
|
374
|
+
write();
|
|
375
|
+
}
|
|
376
|
+
catch (err) {
|
|
377
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
378
|
+
console.warn(`traces: ${label} cache warm-up skipped (${msg}) — index still built`);
|
|
379
|
+
}
|
|
380
|
+
}
|
|
250
381
|
/** Build the redacted rich console shard from indexed metadata and derived caches. */
|
|
251
382
|
export function buildIndexShard(rows, device, owner, prevShard) {
|
|
252
383
|
const db = getDB();
|
|
@@ -256,13 +387,14 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
256
387
|
for (let i = 0; i < ids.length; i += 400) {
|
|
257
388
|
const chunk = ids.slice(i, i + 400);
|
|
258
389
|
calls.push(...db.prepare(`
|
|
259
|
-
SELECT session_id, ordinal, timestamp, tool, outcome, exit_code, status_code, error_code, error, parse_error
|
|
390
|
+
SELECT session_id, ordinal, timestamp, end_timestamp, tool, outcome, exit_code, status_code, error_code, error, parse_error
|
|
260
391
|
FROM tool_calls
|
|
261
392
|
WHERE session_id IN (${chunk.map(() => '?').join(',')})
|
|
262
393
|
`).all(...chunk));
|
|
263
394
|
}
|
|
264
395
|
const toolMix = new Map();
|
|
265
396
|
const errorCounts = new Map();
|
|
397
|
+
const callsBySession = new Map();
|
|
266
398
|
for (const call of calls) {
|
|
267
399
|
const mix = toolMix.get(call.session_id) ?? {};
|
|
268
400
|
mix[call.tool] = (mix[call.tool] ?? 0) + 1;
|
|
@@ -270,9 +402,27 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
270
402
|
if (call.outcome === 'error') {
|
|
271
403
|
errorCounts.set(call.session_id, (errorCounts.get(call.session_id) ?? 0) + 1);
|
|
272
404
|
}
|
|
405
|
+
const list = callsBySession.get(call.session_id);
|
|
406
|
+
if (list)
|
|
407
|
+
list.push(call);
|
|
408
|
+
else
|
|
409
|
+
callsBySession.set(call.session_id, [call]);
|
|
273
410
|
}
|
|
274
|
-
|
|
275
|
-
|
|
411
|
+
// Classify every row as real `agent` work or internal `utility` plumbing, then run
|
|
412
|
+
// the ENTIRE rest of the shard build over the agent set ONLY (PHNX-3474). Utility
|
|
413
|
+
// rows — title-gen / watchdog / commit-message / factory-worker calls, ~68% of the
|
|
414
|
+
// corpus — are tagged and excluded here, never deleted from sessions.db, so the
|
|
415
|
+
// console counts and scores real agent work: `sessionsImported`, the medians,
|
|
416
|
+
// needs-attention, tool-error-rate, and the topic buckets all measure `agentRows`.
|
|
417
|
+
const kindOf = new Map(rows.map((row) => [row.id, classifySessionKind(row, callsBySession.get(row.id)?.length ?? 0)]));
|
|
418
|
+
const agentRows = rows.filter((row) => kindOf.get(row.id) === 'agent');
|
|
419
|
+
const agentIds = agentRows.map((row) => row.id);
|
|
420
|
+
const utilityCount = rows.length - agentRows.length;
|
|
421
|
+
// Tool calls belonging to utility rows never contribute to the failure/latency
|
|
422
|
+
// stats or the tool-error rate — filter them out at the source alongside the rows.
|
|
423
|
+
const agentCalls = calls.filter((call) => kindOf.get(call.session_id) === 'agent');
|
|
424
|
+
const topics = readSessionTopics(agentIds);
|
|
425
|
+
const missingTopics = agentRows.filter((row) => !topics.has(row.id)).map((row) => {
|
|
276
426
|
const topic = classifyTopic({
|
|
277
427
|
cwd: row.cwd,
|
|
278
428
|
gitBranch: row.git_branch,
|
|
@@ -283,27 +433,53 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
283
433
|
topics.set(row.id, topic);
|
|
284
434
|
return { id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, topic };
|
|
285
435
|
});
|
|
286
|
-
writeSessionTopics(missingTopics);
|
|
287
|
-
|
|
436
|
+
persistDerivedCache('session-topics', () => writeSessionTopics(missingTopics));
|
|
437
|
+
// Insights (frictionSignals) and phenotype (false-termination / …) both need
|
|
438
|
+
// the parsed transcript, which flat tool_calls rows don't carry — so both are
|
|
439
|
+
// lazily derived per-session and cached by transcript mtime+size, then read for
|
|
440
|
+
// the WHOLE corpus (`rows` = allRows) every sync. That full-corpus read is what
|
|
441
|
+
// keeps the phenotype grouping dimension consistent: a session synced weeks ago
|
|
442
|
+
// still contributes its real phenotype from cache, so it can never fragment away
|
|
443
|
+
// from an identically-signatured session synced this run purely by *when* each
|
|
444
|
+
// was first seen (PHNX-3327). A cache-miss row is parsed at most once here even
|
|
445
|
+
// when both derivations are missing.
|
|
446
|
+
const insights = readSessionInsights(agentIds);
|
|
447
|
+
const phenotypes = readSessionPhenotypes(agentIds);
|
|
288
448
|
const missingInsights = [];
|
|
289
|
-
|
|
449
|
+
const missingPhenotypes = [];
|
|
450
|
+
for (const row of agentRows) {
|
|
451
|
+
const needInsights = !insights.has(row.id);
|
|
452
|
+
const needPhenotype = !phenotypes.has(row.id);
|
|
453
|
+
if (!needInsights && !needPhenotype)
|
|
454
|
+
continue;
|
|
455
|
+
let events;
|
|
290
456
|
try {
|
|
291
|
-
|
|
292
|
-
const facets = computeInsightFacets(events);
|
|
293
|
-
insights.set(row.id, facets);
|
|
294
|
-
missingInsights.push({
|
|
295
|
-
id: row.id,
|
|
296
|
-
fileMtimeMs: row.file_mtime_ms,
|
|
297
|
-
fileSize: row.file_size,
|
|
298
|
-
facets,
|
|
299
|
-
});
|
|
457
|
+
events = parseSession(row.file_path, row.agent);
|
|
300
458
|
}
|
|
301
459
|
catch {
|
|
302
|
-
continue;
|
|
460
|
+
continue; // gone/unreadable transcript — leave both uncached, same as before
|
|
461
|
+
}
|
|
462
|
+
if (needInsights) {
|
|
463
|
+
try {
|
|
464
|
+
const facets = computeInsightFacets(events);
|
|
465
|
+
insights.set(row.id, facets);
|
|
466
|
+
missingInsights.push({ id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, facets });
|
|
467
|
+
}
|
|
468
|
+
catch { /* leave this session's facets uncached; recompute next sync */ }
|
|
469
|
+
}
|
|
470
|
+
if (needPhenotype) {
|
|
471
|
+
try {
|
|
472
|
+
const traj = buildTrajectory(events, rowToMeta(row), { redact: true, knownSecrets });
|
|
473
|
+
const phenotype = classifyPhenotype(buildSessionDetail(traj));
|
|
474
|
+
phenotypes.set(row.id, phenotype);
|
|
475
|
+
missingPhenotypes.push({ id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, phenotype });
|
|
476
|
+
}
|
|
477
|
+
catch { /* leave this session's phenotype uncached; recompute next sync */ }
|
|
303
478
|
}
|
|
304
479
|
}
|
|
305
|
-
writeSessionInsights(missingInsights);
|
|
306
|
-
|
|
480
|
+
persistDerivedCache('session-insights', () => writeSessionInsights(missingInsights));
|
|
481
|
+
persistDerivedCache('session-phenotypes', () => writeSessionPhenotypes(missingPhenotypes));
|
|
482
|
+
const needsAttention = agentRows.flatMap((row) => {
|
|
307
483
|
const facets = insights.get(row.id);
|
|
308
484
|
const errorCount = errorCounts.get(row.id) ?? 0;
|
|
309
485
|
const flags = attentionFlags(errorCount, facets);
|
|
@@ -318,17 +494,28 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
318
494
|
device,
|
|
319
495
|
agent: row.agent,
|
|
320
496
|
model: row.model ?? 'unknown',
|
|
497
|
+
kind: 'agent',
|
|
321
498
|
severity: errorCount * 2 + friction * 3 + corrections * 2,
|
|
322
499
|
flags,
|
|
323
500
|
}];
|
|
324
501
|
}).sort((a, b) => b.severity - a.severity || a.id.localeCompare(b.id));
|
|
325
502
|
const topicCounts = new Map();
|
|
326
|
-
for (const
|
|
327
|
-
const
|
|
328
|
-
|
|
329
|
-
|
|
503
|
+
for (const row of agentRows) {
|
|
504
|
+
const topic = topics.get(row.id);
|
|
505
|
+
if (!topic)
|
|
506
|
+
continue;
|
|
507
|
+
const bucket = topicCounts.get(topic.key)
|
|
508
|
+
?? { key: topic.key, label: topic.label, group: topic.group, count: 0, refs: [] };
|
|
509
|
+
bucket.count++;
|
|
510
|
+
bucket.refs.push({
|
|
511
|
+
id: row.id,
|
|
512
|
+
title: redactSecrets(row.label ?? row.topic ?? topic.label ?? 'Untitled session', knownSecrets),
|
|
513
|
+
harness: row.agent,
|
|
514
|
+
recencyMs: Date.parse(row.last_activity ?? row.timestamp) || 0,
|
|
515
|
+
});
|
|
516
|
+
topicCounts.set(topic.key, bucket);
|
|
330
517
|
}
|
|
331
|
-
const failedCalls =
|
|
518
|
+
const failedCalls = agentCalls.filter((call) => call.outcome === 'error');
|
|
332
519
|
const byCause = { real: 0, guard: 0, hook: 0 };
|
|
333
520
|
const failureCounts = new Map();
|
|
334
521
|
for (const call of failedCalls) {
|
|
@@ -340,14 +527,38 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
340
527
|
current.count++;
|
|
341
528
|
failureCounts.set(key, current);
|
|
342
529
|
}
|
|
343
|
-
|
|
530
|
+
// Duration stats run over ACTIVE time, not raw span (PHNX-3457): span minus idle
|
|
531
|
+
// gaps > 120s, derived per session from the tool_calls already loaded above. A
|
|
532
|
+
// session resumed after hours, or left idle mid-turn, otherwise inflates the
|
|
533
|
+
// median/p90 with wall-clock the agent did no work in (real corpus max span:
|
|
534
|
+
// 345h). The raw span stays available per session on `SessionDetail.meta.spanMs`.
|
|
535
|
+
const activeDurations = agentRows.flatMap((row) => row.duration_ms == null
|
|
536
|
+
? []
|
|
537
|
+
: [sessionActiveMs(row.duration_ms, callsBySession.get(row.id) ?? [], Date.parse(row.timestamp))]);
|
|
538
|
+
// Segment the same active-time figure into AGENT vs INTERACTIVE runs (PHNX-3472).
|
|
539
|
+
// A session is an AGENT run when it made any tool call OR has more than 8
|
|
540
|
+
// messages; otherwise INTERACTIVE (a one-shot query). Only sessions with a
|
|
541
|
+
// non-null duration contribute to the medians; `measuredFraction` reports how
|
|
542
|
+
// much of the corpus that covers.
|
|
543
|
+
const agentActive = [];
|
|
544
|
+
const interactiveActive = [];
|
|
545
|
+
let measured = 0;
|
|
546
|
+
for (const row of agentRows) {
|
|
547
|
+
if (row.duration_ms == null)
|
|
548
|
+
continue;
|
|
549
|
+
measured++;
|
|
550
|
+
const active = sessionActiveMs(row.duration_ms, callsBySession.get(row.id) ?? [], Date.parse(row.timestamp));
|
|
551
|
+
const isAgent = (callsBySession.get(row.id)?.length ?? 0) > 0 || (row.message_count ?? 0) > 8;
|
|
552
|
+
(isAgent ? agentActive : interactiveActive).push(active);
|
|
553
|
+
}
|
|
554
|
+
const measuredFraction = agentRows.length === 0 ? 0 : measured / agentRows.length;
|
|
344
555
|
// Build today's per-bucket stats for the rolling drift window.
|
|
345
556
|
const todayDate = new Date().toISOString().slice(0, 10);
|
|
346
557
|
const todayStats = [...topicCounts.values()].map(({ key }) => {
|
|
347
558
|
const sessionsInBucket = [...topics.entries()]
|
|
348
559
|
.filter(([, t]) => t.key === key)
|
|
349
560
|
.map(([id]) => id);
|
|
350
|
-
const bucketCalls =
|
|
561
|
+
const bucketCalls = agentCalls.filter((c) => sessionsInBucket.includes(c.session_id));
|
|
351
562
|
const bucketErrors = bucketCalls.filter((c) => c.outcome === 'error').length;
|
|
352
563
|
const errorRate = bucketCalls.length === 0 ? 0 : bucketErrors / bucketCalls.length;
|
|
353
564
|
const stallCount = sessionsInBucket.filter((id) => {
|
|
@@ -361,21 +572,43 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
361
572
|
const prevHistory = prevShard?.bucketHistory ?? [];
|
|
362
573
|
const bucketHistory = [...prevHistory, todayStats].slice(-14);
|
|
363
574
|
const driftSignals = computeDriftSignal(prevHistory, todayStats);
|
|
364
|
-
const patternInsights = computeInsights(
|
|
575
|
+
const patternInsights = computeInsights(agentRows, agentCalls, prevShard, phenotypes);
|
|
365
576
|
return {
|
|
366
577
|
schema: 1,
|
|
367
578
|
device,
|
|
368
579
|
syncedAt: Date.now(),
|
|
369
580
|
owner,
|
|
370
581
|
stats: {
|
|
371
|
-
sessionsImported:
|
|
372
|
-
medianMs: percentile(
|
|
373
|
-
p90Ms: percentile(
|
|
582
|
+
sessionsImported: agentRows.length,
|
|
583
|
+
medianMs: percentile(activeDurations, 0.5),
|
|
584
|
+
p90Ms: percentile(activeDurations, 0.9),
|
|
585
|
+
agentMedianMs: percentile(agentActive, 0.5),
|
|
586
|
+
agentP90Ms: percentile(agentActive, 0.9),
|
|
587
|
+
interactiveMedianMs: percentile(interactiveActive, 0.5),
|
|
588
|
+
measuredFraction,
|
|
374
589
|
needAttention: needsAttention.length,
|
|
375
|
-
toolErrorRate:
|
|
590
|
+
toolErrorRate: agentCalls.length === 0 ? 0 : failedCalls.length / agentCalls.length,
|
|
376
591
|
},
|
|
592
|
+
utilityCount,
|
|
377
593
|
needsAttention,
|
|
378
|
-
topics: [...topicCounts.values()]
|
|
594
|
+
topics: [...topicCounts.values()]
|
|
595
|
+
.sort((a, b) => b.count - a.count || a.key.localeCompare(b.key))
|
|
596
|
+
.map((bucket) => ({
|
|
597
|
+
key: bucket.key,
|
|
598
|
+
label: bucket.label,
|
|
599
|
+
count: bucket.count,
|
|
600
|
+
group: bucket.group,
|
|
601
|
+
// Up to TOPIC_SESSION_CAP most-recent example sessions, so the console can
|
|
602
|
+
// drill from a tile into its session list. Capped to keep the shard small;
|
|
603
|
+
// the tile's `count` remains the true total. These are correct on this
|
|
604
|
+
// per-device shard; the fleet-aggregate `/all` view (worker-template.ts,
|
|
605
|
+
// a must-not-touch R2 worker here) carries only the first device's refs
|
|
606
|
+
// per topic until PHNX-3464 merges them across devices.
|
|
607
|
+
sessions: bucket.refs
|
|
608
|
+
.sort((a, b) => b.recencyMs - a.recencyMs)
|
|
609
|
+
.slice(0, TOPIC_SESSION_CAP)
|
|
610
|
+
.map(({ id, title, harness }) => ({ id, title, kind: 'agent', harness })),
|
|
611
|
+
})),
|
|
379
612
|
failures: {
|
|
380
613
|
byToolError: [...failureCounts.values()].sort((a, b) => b.count - a.count || a.tool.localeCompare(b.tool)),
|
|
381
614
|
byCause,
|
|
@@ -404,6 +637,31 @@ function buildWhereItWentWrong(traj) {
|
|
|
404
637
|
return null;
|
|
405
638
|
return `This run hit ${parts.join('; ')}.`;
|
|
406
639
|
}
|
|
640
|
+
/**
|
|
641
|
+
* Truthful run-level outcome (PHNX-3387).
|
|
642
|
+
*
|
|
643
|
+
* A run with zero tool errors `completed`. A run that hit tool errors is
|
|
644
|
+
* `completed` ONLY when it *causally recovered* — a substantive, non-human-facing
|
|
645
|
+
* tool step succeeded strictly after the last error AND resolved the failed work
|
|
646
|
+
* (its work signature matches an errored step's), the exact predicate the
|
|
647
|
+
* false-termination phenotype uses ({@link recoveredAfterErrors}). A run whose
|
|
648
|
+
* last substantive step is the error, whose only post-error steps are human-facing
|
|
649
|
+
* (a punt to `AskUserQuestion` — the case the broken "last tool call ok" heuristic
|
|
650
|
+
* mislabeled `completed`), or whose only post-error success is unrelated work (a
|
|
651
|
+
* failed `bun test` followed by an incidental `ls`) stays `errored`.
|
|
652
|
+
*
|
|
653
|
+
* This is what makes `surfacedToolFailures` on a `completed` run honest: those
|
|
654
|
+
* are failures the run recovered from, not a green status hiding an unresolved
|
|
655
|
+
* failure. It never flips a run that ended unresolved to `completed` (no
|
|
656
|
+
* regression vs the old `errorCount > 0 ? errored : completed`), and it does not
|
|
657
|
+
* flip a run whose failed work was never resolved just because some later,
|
|
658
|
+
* unrelated call happened to succeed.
|
|
659
|
+
*/
|
|
660
|
+
function deriveRunOutcome(traj) {
|
|
661
|
+
if (traj.errorCount === 0)
|
|
662
|
+
return 'completed';
|
|
663
|
+
return recoveredAfterErrors({ steps: traj.steps }) ? 'completed' : 'errored';
|
|
664
|
+
}
|
|
407
665
|
/**
|
|
408
666
|
* Map the derived trajectory to the console's SessionDetail shape, stripping
|
|
409
667
|
* local-machine PII (full cwd, account) that would expose filesystem paths if
|
|
@@ -418,12 +676,13 @@ export function buildSessionDetail(traj) {
|
|
|
418
676
|
id: s.id,
|
|
419
677
|
meta: {
|
|
420
678
|
spanMs: traj.spanMs,
|
|
679
|
+
activeMs: activeMsFromTrajectory(traj),
|
|
421
680
|
turns: (stats.userTurns ?? 0) + (stats.assistantTurns ?? 0),
|
|
422
681
|
tools: stats.toolCount ?? 0,
|
|
423
682
|
errorCount: traj.errorCount,
|
|
424
683
|
tokens: stats.outputTokens ?? 0,
|
|
425
684
|
costUsd: s.costUsd ?? 0,
|
|
426
|
-
outcome: traj
|
|
685
|
+
outcome: deriveRunOutcome(traj),
|
|
427
686
|
repo,
|
|
428
687
|
agent: s.agent,
|
|
429
688
|
model: s.model ?? 'unknown',
|