@phnx-labs/agents-cli 1.22.58 → 1.22.60
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +260 -0
- package/README.md +29 -0
- package/dist/bootstrap.js +32 -1
- package/dist/commands/monitors.js +187 -23
- package/dist/commands/perf.js +10 -0
- package/dist/commands/routines.test-fixture.js +5 -0
- package/dist/commands/send.d.ts +2 -1
- package/dist/commands/send.js +7 -5
- package/dist/commands/sessions-picker.d.ts +13 -0
- package/dist/commands/sessions-picker.js +17 -8
- package/dist/commands/sessions-stats.js +37 -5
- package/dist/commands/sessions.js +52 -16
- package/dist/commands/ssh.js +12 -1
- package/dist/commands/teams-picker.js +20 -6
- package/dist/commands/teams.d.ts +2 -2
- package/dist/commands/teams.js +77 -21
- package/dist/commands/versions.js +12 -4
- package/dist/commands/view.js +7 -2
- package/dist/lib/accounting/rotate.js +12 -4
- package/dist/lib/accounting/usage-sync.d.ts +12 -2
- package/dist/lib/accounting/usage-sync.js +34 -6
- package/dist/lib/auto-pull-worker.js +7 -2
- package/dist/lib/cloud/rush.d.ts +7 -0
- package/dist/lib/cloud/rush.js +29 -1
- package/dist/lib/daemon/daemon.d.ts +22 -0
- package/dist/lib/daemon/daemon.js +39 -0
- package/dist/lib/daemon/session-index-service.js +9 -1
- package/dist/lib/daemon-ticks.d.ts +15 -0
- package/dist/lib/daemon-ticks.js +26 -0
- package/dist/lib/device-config.d.ts +5 -1
- package/dist/lib/device-config.js +2 -2
- package/dist/lib/devices/health.js +5 -1
- package/dist/lib/devices/pool.d.ts +25 -2
- package/dist/lib/devices/pool.js +32 -2
- package/dist/lib/devices/stats-cache.d.ts +0 -6
- package/dist/lib/devices/stats-cache.js +2 -9
- package/dist/lib/doctor-diff.d.ts +14 -0
- package/dist/lib/doctor-diff.js +43 -2
- package/dist/lib/feed/events.js +4 -0
- package/dist/lib/git.d.ts +38 -0
- package/dist/lib/git.js +58 -0
- package/dist/lib/hosts/ready.d.ts +8 -0
- package/dist/lib/hosts/ready.js +13 -2
- package/dist/lib/installations/versions.d.ts +17 -0
- package/dist/lib/installations/versions.js +53 -2
- package/dist/lib/monitors/config.d.ts +71 -3
- package/dist/lib/monitors/config.js +100 -12
- package/dist/lib/monitors/pid-watch.d.ts +35 -0
- package/dist/lib/monitors/pid-watch.js +45 -0
- package/dist/lib/monitors/remote.d.ts +18 -0
- package/dist/lib/monitors/remote.js +11 -0
- package/dist/lib/perf/db.d.ts +1 -1
- package/dist/lib/perf/db.js +53 -2
- package/dist/lib/perf/types.d.ts +14 -0
- package/dist/lib/permissions.js +7 -2
- package/dist/lib/plugins/plugins.d.ts +17 -3
- package/dist/lib/plugins/plugins.js +84 -9
- package/dist/lib/pty-server.d.ts +14 -0
- package/dist/lib/pty-server.js +49 -5
- package/dist/lib/secrets/drivers/rush.js +5 -0
- package/dist/lib/self-update.d.ts +42 -0
- package/dist/lib/self-update.js +88 -0
- package/dist/lib/session/cloud.js +5 -0
- package/dist/lib/session/db.d.ts +32 -6
- package/dist/lib/session/db.js +128 -12
- package/dist/lib/session/live-metadata.js +3 -3
- package/dist/lib/smart-launch.d.ts +6 -0
- package/dist/lib/smart-launch.js +5 -2
- package/dist/lib/staleness/writers/plugins.js +5 -2
- package/dist/lib/staleness/writers/subagents.js +13 -3
- package/dist/lib/state.d.ts +7 -4
- package/dist/lib/state.js +7 -4
- package/dist/lib/subagents.js +8 -2
- package/dist/lib/teams/api.d.ts +8 -0
- package/dist/lib/teams/api.js +50 -6
- package/dist/lib/teams/delivery.d.ts +14 -4
- package/dist/lib/teams/delivery.js +15 -5
- package/dist/lib/teams/scheduler.d.ts +10 -0
- package/dist/lib/teams/scheduler.js +8 -0
- package/dist/lib/traces/sync.d.ts +113 -6
- package/dist/lib/traces/sync.js +193 -19
- package/dist/lib/view-types.d.ts +12 -0
- package/package.json +2 -2
|
@@ -17,6 +17,10 @@ import { AgentStatus } from './agents.js';
|
|
|
17
17
|
* When the process completed with a `prUrl` and merge is unknown or false,
|
|
18
18
|
* delivery is `pr_open` — pessimistic: assume open until proven merged so an
|
|
19
19
|
* orchestrator never mistakes "agent stopped" for "work on main".
|
|
20
|
+
*
|
|
21
|
+
* When the process completed with no PR and uncommitted changes remain in the
|
|
22
|
+
* worktree, delivery is `stranded` — the work exists only locally and will be
|
|
23
|
+
* lost if the worktree is cleaned up (PHNX-2951).
|
|
20
24
|
*/
|
|
21
25
|
export function resolveTeammateDelivery(opts) {
|
|
22
26
|
const status = String(opts.status);
|
|
@@ -30,8 +34,9 @@ export function resolveTeammateDelivery(opts) {
|
|
|
30
34
|
return 'stopped';
|
|
31
35
|
if (status === AgentStatus.COMPLETED || status === 'completed') {
|
|
32
36
|
const prUrl = opts.prUrl?.trim();
|
|
33
|
-
if (!prUrl)
|
|
34
|
-
return 'no_pr';
|
|
37
|
+
if (!prUrl) {
|
|
38
|
+
return opts.hasUncommittedChanges ? 'stranded' : 'no_pr';
|
|
39
|
+
}
|
|
35
40
|
if (opts.prMerged === true)
|
|
36
41
|
return 'pr_merged';
|
|
37
42
|
return 'pr_open';
|
|
@@ -40,21 +45,26 @@ export function resolveTeammateDelivery(opts) {
|
|
|
40
45
|
}
|
|
41
46
|
/**
|
|
42
47
|
* Human label for `teams status` rows. Replaces bare COMPLETED with PR OPEN
|
|
43
|
-
* when delivery is still pending merge
|
|
48
|
+
* when delivery is still pending merge, and with STRANDED when uncommitted
|
|
49
|
+
* work is stranded in the worktree.
|
|
44
50
|
*/
|
|
45
51
|
export function deliveryDisplayLabel(delivery, processStatus) {
|
|
46
52
|
if (delivery === 'pr_open')
|
|
47
53
|
return 'PR OPEN';
|
|
48
54
|
if (delivery === 'pr_merged')
|
|
49
55
|
return 'COMPLETED';
|
|
56
|
+
if (delivery === 'stranded')
|
|
57
|
+
return 'STRANDED';
|
|
50
58
|
return String(processStatus).toUpperCase();
|
|
51
59
|
}
|
|
52
60
|
/**
|
|
53
|
-
* Color key for statusColor-style switches. `pr_open`
|
|
54
|
-
*
|
|
61
|
+
* Color key for statusColor-style switches. `pr_open` and `stranded` get their
|
|
62
|
+
* own keys so the rows are visually distinct from green COMPLETED.
|
|
55
63
|
*/
|
|
56
64
|
export function deliveryColorKey(delivery, processStatus) {
|
|
57
65
|
if (delivery === 'pr_open')
|
|
58
66
|
return 'pr_open';
|
|
67
|
+
if (delivery === 'stranded')
|
|
68
|
+
return 'stranded';
|
|
59
69
|
return String(processStatus);
|
|
60
70
|
}
|
|
@@ -48,6 +48,16 @@ export interface PlacementOptions {
|
|
|
48
48
|
/** Human label of the requested agent (e.g. `claude@2.1.112`) for the
|
|
49
49
|
* fail-loud message. */
|
|
50
50
|
agentLabel?: string;
|
|
51
|
+
/**
|
|
52
|
+
* Normalized hosts boosted with `auto-launch.preferred` (set by
|
|
53
|
+
* `agents devices prefer <name>`). A preferred device ranks ahead of a
|
|
54
|
+
* non-preferred one among the eligible survivors — after the signed-in tier
|
|
55
|
+
* (a preferred box that can't run the agent is still no use) and before load,
|
|
56
|
+
* so an operator boost overrides load-based ordering without overriding hard
|
|
57
|
+
* health. Empty/undefined leaves the ranking unchanged. See
|
|
58
|
+
* {@link autoLaunchPreferredSet}.
|
|
59
|
+
*/
|
|
60
|
+
preferred?: ReadonlySet<string>;
|
|
51
61
|
}
|
|
52
62
|
/** Why a device was excluded from the viable set, for the fail-loud message. */
|
|
53
63
|
export type ExclusionReason = 'unreachable' | 'overloaded' | 'capped' | 'not-installed';
|
|
@@ -266,6 +266,14 @@ export function pickBestDevice(devices, roster, opts) {
|
|
|
266
266
|
const signedIn = (sa?.signedIn === true ? 0 : 1) - (sb?.signedIn === true ? 0 : 1);
|
|
267
267
|
if (signedIn !== 0)
|
|
268
268
|
return signedIn;
|
|
269
|
+
// (a2) operator-preferred device next — `agents devices prefer <name>`
|
|
270
|
+
// boosts a box above its load-equal peers, overriding load-based order.
|
|
271
|
+
const preferred = opts?.preferred;
|
|
272
|
+
if (preferred && preferred.size > 0) {
|
|
273
|
+
const pref = (preferred.has(a) ? 0 : 1) - (preferred.has(b) ? 0 : 1);
|
|
274
|
+
if (pref !== 0)
|
|
275
|
+
return pref;
|
|
276
|
+
}
|
|
269
277
|
// (b) lower load — coarse headroom tier, then raw load cost.
|
|
270
278
|
const tier = headroomTier(sa?.headroom) - headroomTier(sb?.headroom);
|
|
271
279
|
if (tier !== 0)
|
|
@@ -67,18 +67,46 @@ export interface TracesIndexShard {
|
|
|
67
67
|
owner: string;
|
|
68
68
|
stats: {
|
|
69
69
|
sessionsImported: number;
|
|
70
|
+
/**
|
|
71
|
+
* Median ACTIVE duration (span − idle gaps > 120s), ms — the meaningful figure
|
|
72
|
+
* (PHNX-3457). Same key/shape as before this change, so the fleet-aggregate
|
|
73
|
+
* worker (`worker-template.ts`) keeps weighted-averaging it unchanged; only its
|
|
74
|
+
* VALUE moved from raw span to active time. The raw span stays available per
|
|
75
|
+
* session on `SessionDetail.meta.spanMs`.
|
|
76
|
+
*/
|
|
70
77
|
medianMs: number;
|
|
78
|
+
/** p90 ACTIVE duration, ms. */
|
|
71
79
|
p90Ms: number;
|
|
80
|
+
/**
|
|
81
|
+
* SEGMENTED active-time stats (PHNX-3472). The blended `medianMs`/`p90Ms`
|
|
82
|
+
* above conflate one-shot interactive queries (63% of the corpus, ~15s
|
|
83
|
+
* median) with substantial agent runs (~15min median), so they headline
|
|
84
|
+
* neither. A session is an AGENT run when it made any tool call OR has more
|
|
85
|
+
* than 8 messages; otherwise INTERACTIVE. These segment the same active-time
|
|
86
|
+
* figure so the console can headline agent runs on their own axis. Each is
|
|
87
|
+
* computed only over sessions with a non-null duration.
|
|
88
|
+
*/
|
|
89
|
+
agentMedianMs: number;
|
|
90
|
+
/** p90 ACTIVE duration over AGENT sessions, ms. */
|
|
91
|
+
agentP90Ms: number;
|
|
92
|
+
/** Median ACTIVE duration over INTERACTIVE sessions, ms. */
|
|
93
|
+
interactiveMedianMs: number;
|
|
94
|
+
/** (sessions with a non-null duration) / (total sessions), 0..1 — coverage of the duration stats. */
|
|
95
|
+
measuredFraction: number;
|
|
72
96
|
needAttention: number;
|
|
73
97
|
toolErrorRate: number;
|
|
74
98
|
};
|
|
99
|
+
/**
|
|
100
|
+
* Sessions excluded from the eval corpus as internal utility plumbing (PHNX-3474):
|
|
101
|
+
* single-shot machine calls (no tool call AND ≤2 messages) or a known
|
|
102
|
+
* internal-prompt signature (title generation, watchdog, commit-message, factory
|
|
103
|
+
* worker). Every `stats` figure above, `topics` counts, and `needsAttention` are
|
|
104
|
+
* computed over the AGENT set ONLY — `sessionsImported` is the real agent count,
|
|
105
|
+
* not the raw row count. This is the number that was dropped.
|
|
106
|
+
*/
|
|
107
|
+
utilityCount: number;
|
|
75
108
|
needsAttention: IndexedSession[];
|
|
76
|
-
topics:
|
|
77
|
-
key: string;
|
|
78
|
-
label: string;
|
|
79
|
-
count: number;
|
|
80
|
-
group: TraceTopicGroup;
|
|
81
|
-
}>;
|
|
109
|
+
topics: TopicItem[];
|
|
82
110
|
failures: {
|
|
83
111
|
byToolError: Array<{
|
|
84
112
|
tool: string;
|
|
@@ -104,11 +132,56 @@ export interface IndexedSession {
|
|
|
104
132
|
title: string;
|
|
105
133
|
repo: string;
|
|
106
134
|
device: string;
|
|
135
|
+
/** The harness that produced the session (claude/codex/rush/grok/…). Same as `harness`. */
|
|
107
136
|
agent: string;
|
|
108
137
|
model: string;
|
|
138
|
+
/** Corpus classification (PHNX-3474). Always `'agent'` here — utility rows are excluded. */
|
|
139
|
+
kind: SessionKind;
|
|
109
140
|
severity: number;
|
|
110
141
|
flags: string[];
|
|
111
142
|
}
|
|
143
|
+
/**
|
|
144
|
+
* One example session under a topic tile — the shape the console drill-down consumes.
|
|
145
|
+
* Carries `kind` + `harness` (PHNX-3474) so the console can filter a tile's session
|
|
146
|
+
* list by corpus class and by harness. Refs on a topic tile are always `'agent'`
|
|
147
|
+
* (utility rows never reach a bucket), but the field is explicit for the consumer.
|
|
148
|
+
*/
|
|
149
|
+
export interface TopicSessionRef {
|
|
150
|
+
id: string;
|
|
151
|
+
title: string;
|
|
152
|
+
kind: SessionKind;
|
|
153
|
+
harness: string;
|
|
154
|
+
}
|
|
155
|
+
/**
|
|
156
|
+
* Corpus class of a session (PHNX-3474). `utility` is internal machine plumbing —
|
|
157
|
+
* a single-shot call with no tool use and ≤2 messages, or one whose topic/label
|
|
158
|
+
* matches a known internal-prompt signature (title generation, watchdog,
|
|
159
|
+
* commit-message writer, factory worker). Everything else is `agent`: real agent
|
|
160
|
+
* work the Evals console counts and scores. Utility rows are tagged, never deleted,
|
|
161
|
+
* and excluded from every index statistic.
|
|
162
|
+
*/
|
|
163
|
+
export type SessionKind = 'utility' | 'agent';
|
|
164
|
+
/**
|
|
165
|
+
* Classify a session as internal `utility` plumbing vs real `agent` work (PHNX-3474).
|
|
166
|
+
* `toolCallCount` is the AUTHORITATIVE per-session tool-call count from the loaded
|
|
167
|
+
* `tool_calls` rows (the row's own `tool_call_count` column is a fallback for a row
|
|
168
|
+
* whose calls weren't loaded). A session is `utility` when a known internal-prompt
|
|
169
|
+
* signature matches its topic/label, OR it made no tool call AND has ≤2 messages —
|
|
170
|
+
* the single-shot machine-call shape. Otherwise it is `agent`.
|
|
171
|
+
*/
|
|
172
|
+
export declare function classifySessionKind(row: Pick<SyncRow, 'topic' | 'label' | 'message_count' | 'tool_call_count'>, toolCallCount: number): SessionKind;
|
|
173
|
+
/**
|
|
174
|
+
* One topic bucket in the treemap. `sessions` carries up to {@link TOPIC_SESSION_CAP}
|
|
175
|
+
* example refs so the console can drill from the tile into its session list — a tile
|
|
176
|
+
* with no refs renders display-only (PHNX-3408). `count` stays the true total.
|
|
177
|
+
*/
|
|
178
|
+
export interface TopicItem {
|
|
179
|
+
key: string;
|
|
180
|
+
label: string;
|
|
181
|
+
count: number;
|
|
182
|
+
group: TraceTopicGroup;
|
|
183
|
+
sessions: TopicSessionRef[];
|
|
184
|
+
}
|
|
112
185
|
/** A row from `tool_calls`. `ordinal`/`timestamp` order calls within a session for computeInsights(). */
|
|
113
186
|
export interface ToolCallRow {
|
|
114
187
|
session_id: string;
|
|
@@ -129,6 +202,31 @@ export interface ToolCallRow {
|
|
|
129
202
|
error: string | null;
|
|
130
203
|
parse_error: string | null;
|
|
131
204
|
}
|
|
205
|
+
/**
|
|
206
|
+
* Active time for a session in the index shard: its recorded span minus every idle
|
|
207
|
+
* gap > 120s (PHNX-3457). The index build already holds the ordered `tool_calls`
|
|
208
|
+
* rows for the whole corpus, so idle is derived from them here — no transcript
|
|
209
|
+
* re-parse. A cursor sweeps the span from `sessionStartMs`: each stretch where the
|
|
210
|
+
* cursor sits idle for more than the threshold before the next call starts is
|
|
211
|
+
* subtracted, and idle is measured from a call's END (its own `end_timestamp` when
|
|
212
|
+
* known, else its start) so a call's own blocking duration is never mistaken for
|
|
213
|
+
* idle. Crucially the sweep also books the gaps at the two BOUNDARIES — before the
|
|
214
|
+
* first call and after the last call to the session end — so a session with a lone
|
|
215
|
+
* tool call that was then abandoned and resumed hours later (the case a
|
|
216
|
+
* between-calls-only measure missed entirely, leaving the whole 345h span counted
|
|
217
|
+
* as active) has that trailing idle stripped. A session end is `sessionStartMs +
|
|
218
|
+
* spanMs`, so the two agree by construction.
|
|
219
|
+
*
|
|
220
|
+
* Bounded to `[0, spanMs]`. A session with NO tool calls returns the full span
|
|
221
|
+
* unchanged rather than a fabricated zero: there is no tool-call evidence of idle
|
|
222
|
+
* either way, and treating a chat-only turn as 100% idle would be a worse error
|
|
223
|
+
* than leaving its span uncorrected. Where the full event stream IS available (a
|
|
224
|
+
* per-session `SessionDetail`), {@link activeMsFromTrajectory} is used instead —
|
|
225
|
+
* it sees message events this call-only approximation cannot, so the two are close
|
|
226
|
+
* but not identical by design (the corpus-scale index build cannot afford the
|
|
227
|
+
* per-session parse the detail view does).
|
|
228
|
+
*/
|
|
229
|
+
export declare function sessionActiveMs(spanMs: number, sessionCalls: ToolCallRow[], sessionStartMs: number): number;
|
|
132
230
|
/** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
|
|
133
231
|
export declare function failureDescription(call: ToolCallRow, cause: TraceFailureCause): string;
|
|
134
232
|
/** Build the redacted rich console shard from indexed metadata and derived caches. */
|
|
@@ -138,7 +236,16 @@ export interface SessionDetail {
|
|
|
138
236
|
schema: 1;
|
|
139
237
|
id: string;
|
|
140
238
|
meta: {
|
|
239
|
+
/** Raw wall-clock span (last event − first event), idle time included. */
|
|
141
240
|
spanMs: number;
|
|
241
|
+
/**
|
|
242
|
+
* Active time: `spanMs` minus every idle gap > 120s (PHNX-3457). A session
|
|
243
|
+
* resumed hours later, or left idle mid-turn, inflates `spanMs` with wall-clock
|
|
244
|
+
* the agent did no work in — active time strips those gaps so a duration reads
|
|
245
|
+
* as effort, not calendar span. This is what the console's duration median/p90
|
|
246
|
+
* should trust; `spanMs` stays available as the raw figure.
|
|
247
|
+
*/
|
|
248
|
+
activeMs: number;
|
|
142
249
|
turns: number;
|
|
143
250
|
tools: number;
|
|
144
251
|
errorCount: number;
|
package/dist/lib/traces/sync.js
CHANGED
|
@@ -213,12 +213,109 @@ export async function syncTraces(opts = {}) {
|
|
|
213
213
|
indexError,
|
|
214
214
|
};
|
|
215
215
|
}
|
|
216
|
+
/**
|
|
217
|
+
* Topic/label substrings that identify an internal-prompt session regardless of its
|
|
218
|
+
* message/tool shape. These are the harness-spawned utility prompts the Rush app
|
|
219
|
+
* fires (they run under the `claude` harness): title generation writes the 3–4 word
|
|
220
|
+
* session title, the watchdog polls for stalled agents, the commit-message writer
|
|
221
|
+
* drafts a conventional commit, and factory workers are dispatched sub-agents. The
|
|
222
|
+
* title-generation prompt lives in the `topic` column, the rest can land in either
|
|
223
|
+
* `topic` or `label`, so both are matched.
|
|
224
|
+
*/
|
|
225
|
+
const UTILITY_PROMPT_SIGNATURES = [
|
|
226
|
+
/generate a 3-4 word title/i, // title generation
|
|
227
|
+
/you are a watchdog|watchdog monitoring/i, // watchdog tick
|
|
228
|
+
/conventional[- ]commit/i, // commit-message writer
|
|
229
|
+
/factory worker/i, // dispatched factory worker
|
|
230
|
+
];
|
|
231
|
+
/**
|
|
232
|
+
* Classify a session as internal `utility` plumbing vs real `agent` work (PHNX-3474).
|
|
233
|
+
* `toolCallCount` is the AUTHORITATIVE per-session tool-call count from the loaded
|
|
234
|
+
* `tool_calls` rows (the row's own `tool_call_count` column is a fallback for a row
|
|
235
|
+
* whose calls weren't loaded). A session is `utility` when a known internal-prompt
|
|
236
|
+
* signature matches its topic/label, OR it made no tool call AND has ≤2 messages —
|
|
237
|
+
* the single-shot machine-call shape. Otherwise it is `agent`.
|
|
238
|
+
*/
|
|
239
|
+
export function classifySessionKind(row, toolCallCount) {
|
|
240
|
+
const haystack = `${row.topic ?? ''}\n${row.label ?? ''}`;
|
|
241
|
+
if (UTILITY_PROMPT_SIGNATURES.some((re) => re.test(haystack)))
|
|
242
|
+
return 'utility';
|
|
243
|
+
const hasToolCalls = toolCallCount > 0 || (row.tool_call_count ?? 0) > 0;
|
|
244
|
+
const messages = row.message_count ?? 0;
|
|
245
|
+
if (!hasToolCalls && messages <= 2)
|
|
246
|
+
return 'utility';
|
|
247
|
+
return 'agent';
|
|
248
|
+
}
|
|
216
249
|
function percentile(values, ratio) {
|
|
217
250
|
if (values.length === 0)
|
|
218
251
|
return 0;
|
|
219
252
|
const sorted = [...values].sort((a, b) => a - b);
|
|
220
253
|
return sorted[Math.min(sorted.length - 1, Math.ceil(sorted.length * ratio) - 1)];
|
|
221
254
|
}
|
|
255
|
+
/**
|
|
256
|
+
* A delta between consecutive events longer than this reads as an idle stall, not
|
|
257
|
+
* work — the same threshold the trajectory uses for its gap detection
|
|
258
|
+
* (`DEFAULT_IDLE_THRESHOLD_MS`, trajectory.ts). Kept in lockstep so active time
|
|
259
|
+
* here and the gaps drawn in a session's detail view agree on what "idle" means.
|
|
260
|
+
*/
|
|
261
|
+
const IDLE_GAP_THRESHOLD_MS = 120_000;
|
|
262
|
+
/** Max example session refs carried per topic tile so a tile is drillable (PHNX-3408). */
|
|
263
|
+
const TOPIC_SESSION_CAP = 30;
|
|
264
|
+
/**
|
|
265
|
+
* Active time for a session in the index shard: its recorded span minus every idle
|
|
266
|
+
* gap > 120s (PHNX-3457). The index build already holds the ordered `tool_calls`
|
|
267
|
+
* rows for the whole corpus, so idle is derived from them here — no transcript
|
|
268
|
+
* re-parse. A cursor sweeps the span from `sessionStartMs`: each stretch where the
|
|
269
|
+
* cursor sits idle for more than the threshold before the next call starts is
|
|
270
|
+
* subtracted, and idle is measured from a call's END (its own `end_timestamp` when
|
|
271
|
+
* known, else its start) so a call's own blocking duration is never mistaken for
|
|
272
|
+
* idle. Crucially the sweep also books the gaps at the two BOUNDARIES — before the
|
|
273
|
+
* first call and after the last call to the session end — so a session with a lone
|
|
274
|
+
* tool call that was then abandoned and resumed hours later (the case a
|
|
275
|
+
* between-calls-only measure missed entirely, leaving the whole 345h span counted
|
|
276
|
+
* as active) has that trailing idle stripped. A session end is `sessionStartMs +
|
|
277
|
+
* spanMs`, so the two agree by construction.
|
|
278
|
+
*
|
|
279
|
+
* Bounded to `[0, spanMs]`. A session with NO tool calls returns the full span
|
|
280
|
+
* unchanged rather than a fabricated zero: there is no tool-call evidence of idle
|
|
281
|
+
* either way, and treating a chat-only turn as 100% idle would be a worse error
|
|
282
|
+
* than leaving its span uncorrected. Where the full event stream IS available (a
|
|
283
|
+
* per-session `SessionDetail`), {@link activeMsFromTrajectory} is used instead —
|
|
284
|
+
* it sees message events this call-only approximation cannot, so the two are close
|
|
285
|
+
* but not identical by design (the corpus-scale index build cannot afford the
|
|
286
|
+
* per-session parse the detail view does).
|
|
287
|
+
*/
|
|
288
|
+
export function sessionActiveMs(spanMs, sessionCalls, sessionStartMs) {
|
|
289
|
+
if (spanMs <= 0)
|
|
290
|
+
return Math.max(0, spanMs);
|
|
291
|
+
if (!Number.isFinite(sessionStartMs))
|
|
292
|
+
return spanMs; // can't place calls on the span
|
|
293
|
+
const spanEndMs = sessionStartMs + spanMs;
|
|
294
|
+
const ordered = sessionCalls
|
|
295
|
+
.map((c) => ({ startMs: Date.parse(c.timestamp), endMs: Date.parse(c.end_timestamp ?? c.timestamp) }))
|
|
296
|
+
.filter((c) => Number.isFinite(c.startMs))
|
|
297
|
+
.sort((a, b) => a.startMs - b.startMs);
|
|
298
|
+
if (ordered.length === 0)
|
|
299
|
+
return spanMs; // no tool-call evidence of idle
|
|
300
|
+
let idleMs = 0;
|
|
301
|
+
let cursor = sessionStartMs;
|
|
302
|
+
for (const call of ordered) {
|
|
303
|
+
if (call.startMs > cursor + IDLE_GAP_THRESHOLD_MS)
|
|
304
|
+
idleMs += call.startMs - cursor;
|
|
305
|
+
const endMs = Number.isFinite(call.endMs) ? Math.max(call.endMs, call.startMs) : call.startMs;
|
|
306
|
+
if (endMs > cursor)
|
|
307
|
+
cursor = endMs;
|
|
308
|
+
}
|
|
309
|
+
// Trailing idle: the stretch from the last call's end to the session's end.
|
|
310
|
+
if (spanEndMs > cursor + IDLE_GAP_THRESHOLD_MS)
|
|
311
|
+
idleMs += spanEndMs - cursor;
|
|
312
|
+
return Math.max(0, spanMs - Math.min(idleMs, spanMs));
|
|
313
|
+
}
|
|
314
|
+
/** Active time from an already-built trajectory: span minus its idle gaps (all > threshold). */
|
|
315
|
+
function activeMsFromTrajectory(traj) {
|
|
316
|
+
const idleMs = traj.gaps.reduce((sum, gap) => sum + gap.durationMs, 0);
|
|
317
|
+
return Math.max(0, traj.spanMs - Math.min(idleMs, traj.spanMs));
|
|
318
|
+
}
|
|
222
319
|
/** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
|
|
223
320
|
export function failureDescription(call, cause) {
|
|
224
321
|
if (cause === 'guard')
|
|
@@ -297,6 +394,7 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
297
394
|
}
|
|
298
395
|
const toolMix = new Map();
|
|
299
396
|
const errorCounts = new Map();
|
|
397
|
+
const callsBySession = new Map();
|
|
300
398
|
for (const call of calls) {
|
|
301
399
|
const mix = toolMix.get(call.session_id) ?? {};
|
|
302
400
|
mix[call.tool] = (mix[call.tool] ?? 0) + 1;
|
|
@@ -304,9 +402,27 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
304
402
|
if (call.outcome === 'error') {
|
|
305
403
|
errorCounts.set(call.session_id, (errorCounts.get(call.session_id) ?? 0) + 1);
|
|
306
404
|
}
|
|
405
|
+
const list = callsBySession.get(call.session_id);
|
|
406
|
+
if (list)
|
|
407
|
+
list.push(call);
|
|
408
|
+
else
|
|
409
|
+
callsBySession.set(call.session_id, [call]);
|
|
307
410
|
}
|
|
308
|
-
|
|
309
|
-
|
|
411
|
+
// Classify every row as real `agent` work or internal `utility` plumbing, then run
|
|
412
|
+
// the ENTIRE rest of the shard build over the agent set ONLY (PHNX-3474). Utility
|
|
413
|
+
// rows — title-gen / watchdog / commit-message / factory-worker calls, ~68% of the
|
|
414
|
+
// corpus — are tagged and excluded here, never deleted from sessions.db, so the
|
|
415
|
+
// console counts and scores real agent work: `sessionsImported`, the medians,
|
|
416
|
+
// needs-attention, tool-error-rate, and the topic buckets all measure `agentRows`.
|
|
417
|
+
const kindOf = new Map(rows.map((row) => [row.id, classifySessionKind(row, callsBySession.get(row.id)?.length ?? 0)]));
|
|
418
|
+
const agentRows = rows.filter((row) => kindOf.get(row.id) === 'agent');
|
|
419
|
+
const agentIds = agentRows.map((row) => row.id);
|
|
420
|
+
const utilityCount = rows.length - agentRows.length;
|
|
421
|
+
// Tool calls belonging to utility rows never contribute to the failure/latency
|
|
422
|
+
// stats or the tool-error rate — filter them out at the source alongside the rows.
|
|
423
|
+
const agentCalls = calls.filter((call) => kindOf.get(call.session_id) === 'agent');
|
|
424
|
+
const topics = readSessionTopics(agentIds);
|
|
425
|
+
const missingTopics = agentRows.filter((row) => !topics.has(row.id)).map((row) => {
|
|
310
426
|
const topic = classifyTopic({
|
|
311
427
|
cwd: row.cwd,
|
|
312
428
|
gitBranch: row.git_branch,
|
|
@@ -327,11 +443,11 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
327
443
|
// from an identically-signatured session synced this run purely by *when* each
|
|
328
444
|
// was first seen (PHNX-3327). A cache-miss row is parsed at most once here even
|
|
329
445
|
// when both derivations are missing.
|
|
330
|
-
const insights = readSessionInsights(
|
|
331
|
-
const phenotypes = readSessionPhenotypes(
|
|
446
|
+
const insights = readSessionInsights(agentIds);
|
|
447
|
+
const phenotypes = readSessionPhenotypes(agentIds);
|
|
332
448
|
const missingInsights = [];
|
|
333
449
|
const missingPhenotypes = [];
|
|
334
|
-
for (const row of
|
|
450
|
+
for (const row of agentRows) {
|
|
335
451
|
const needInsights = !insights.has(row.id);
|
|
336
452
|
const needPhenotype = !phenotypes.has(row.id);
|
|
337
453
|
if (!needInsights && !needPhenotype)
|
|
@@ -363,7 +479,7 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
363
479
|
}
|
|
364
480
|
persistDerivedCache('session-insights', () => writeSessionInsights(missingInsights));
|
|
365
481
|
persistDerivedCache('session-phenotypes', () => writeSessionPhenotypes(missingPhenotypes));
|
|
366
|
-
const needsAttention =
|
|
482
|
+
const needsAttention = agentRows.flatMap((row) => {
|
|
367
483
|
const facets = insights.get(row.id);
|
|
368
484
|
const errorCount = errorCounts.get(row.id) ?? 0;
|
|
369
485
|
const flags = attentionFlags(errorCount, facets);
|
|
@@ -378,17 +494,28 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
378
494
|
device,
|
|
379
495
|
agent: row.agent,
|
|
380
496
|
model: row.model ?? 'unknown',
|
|
497
|
+
kind: 'agent',
|
|
381
498
|
severity: errorCount * 2 + friction * 3 + corrections * 2,
|
|
382
499
|
flags,
|
|
383
500
|
}];
|
|
384
501
|
}).sort((a, b) => b.severity - a.severity || a.id.localeCompare(b.id));
|
|
385
502
|
const topicCounts = new Map();
|
|
386
|
-
for (const
|
|
387
|
-
const
|
|
388
|
-
|
|
389
|
-
|
|
503
|
+
for (const row of agentRows) {
|
|
504
|
+
const topic = topics.get(row.id);
|
|
505
|
+
if (!topic)
|
|
506
|
+
continue;
|
|
507
|
+
const bucket = topicCounts.get(topic.key)
|
|
508
|
+
?? { key: topic.key, label: topic.label, group: topic.group, count: 0, refs: [] };
|
|
509
|
+
bucket.count++;
|
|
510
|
+
bucket.refs.push({
|
|
511
|
+
id: row.id,
|
|
512
|
+
title: redactSecrets(row.label ?? row.topic ?? topic.label ?? 'Untitled session', knownSecrets),
|
|
513
|
+
harness: row.agent,
|
|
514
|
+
recencyMs: Date.parse(row.last_activity ?? row.timestamp) || 0,
|
|
515
|
+
});
|
|
516
|
+
topicCounts.set(topic.key, bucket);
|
|
390
517
|
}
|
|
391
|
-
const failedCalls =
|
|
518
|
+
const failedCalls = agentCalls.filter((call) => call.outcome === 'error');
|
|
392
519
|
const byCause = { real: 0, guard: 0, hook: 0 };
|
|
393
520
|
const failureCounts = new Map();
|
|
394
521
|
for (const call of failedCalls) {
|
|
@@ -400,14 +527,38 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
400
527
|
current.count++;
|
|
401
528
|
failureCounts.set(key, current);
|
|
402
529
|
}
|
|
403
|
-
|
|
530
|
+
// Duration stats run over ACTIVE time, not raw span (PHNX-3457): span minus idle
|
|
531
|
+
// gaps > 120s, derived per session from the tool_calls already loaded above. A
|
|
532
|
+
// session resumed after hours, or left idle mid-turn, otherwise inflates the
|
|
533
|
+
// median/p90 with wall-clock the agent did no work in (real corpus max span:
|
|
534
|
+
// 345h). The raw span stays available per session on `SessionDetail.meta.spanMs`.
|
|
535
|
+
const activeDurations = agentRows.flatMap((row) => row.duration_ms == null
|
|
536
|
+
? []
|
|
537
|
+
: [sessionActiveMs(row.duration_ms, callsBySession.get(row.id) ?? [], Date.parse(row.timestamp))]);
|
|
538
|
+
// Segment the same active-time figure into AGENT vs INTERACTIVE runs (PHNX-3472).
|
|
539
|
+
// A session is an AGENT run when it made any tool call OR has more than 8
|
|
540
|
+
// messages; otherwise INTERACTIVE (a one-shot query). Only sessions with a
|
|
541
|
+
// non-null duration contribute to the medians; `measuredFraction` reports how
|
|
542
|
+
// much of the corpus that covers.
|
|
543
|
+
const agentActive = [];
|
|
544
|
+
const interactiveActive = [];
|
|
545
|
+
let measured = 0;
|
|
546
|
+
for (const row of agentRows) {
|
|
547
|
+
if (row.duration_ms == null)
|
|
548
|
+
continue;
|
|
549
|
+
measured++;
|
|
550
|
+
const active = sessionActiveMs(row.duration_ms, callsBySession.get(row.id) ?? [], Date.parse(row.timestamp));
|
|
551
|
+
const isAgent = (callsBySession.get(row.id)?.length ?? 0) > 0 || (row.message_count ?? 0) > 8;
|
|
552
|
+
(isAgent ? agentActive : interactiveActive).push(active);
|
|
553
|
+
}
|
|
554
|
+
const measuredFraction = agentRows.length === 0 ? 0 : measured / agentRows.length;
|
|
404
555
|
// Build today's per-bucket stats for the rolling drift window.
|
|
405
556
|
const todayDate = new Date().toISOString().slice(0, 10);
|
|
406
557
|
const todayStats = [...topicCounts.values()].map(({ key }) => {
|
|
407
558
|
const sessionsInBucket = [...topics.entries()]
|
|
408
559
|
.filter(([, t]) => t.key === key)
|
|
409
560
|
.map(([id]) => id);
|
|
410
|
-
const bucketCalls =
|
|
561
|
+
const bucketCalls = agentCalls.filter((c) => sessionsInBucket.includes(c.session_id));
|
|
411
562
|
const bucketErrors = bucketCalls.filter((c) => c.outcome === 'error').length;
|
|
412
563
|
const errorRate = bucketCalls.length === 0 ? 0 : bucketErrors / bucketCalls.length;
|
|
413
564
|
const stallCount = sessionsInBucket.filter((id) => {
|
|
@@ -421,21 +572,43 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
421
572
|
const prevHistory = prevShard?.bucketHistory ?? [];
|
|
422
573
|
const bucketHistory = [...prevHistory, todayStats].slice(-14);
|
|
423
574
|
const driftSignals = computeDriftSignal(prevHistory, todayStats);
|
|
424
|
-
const patternInsights = computeInsights(
|
|
575
|
+
const patternInsights = computeInsights(agentRows, agentCalls, prevShard, phenotypes);
|
|
425
576
|
return {
|
|
426
577
|
schema: 1,
|
|
427
578
|
device,
|
|
428
579
|
syncedAt: Date.now(),
|
|
429
580
|
owner,
|
|
430
581
|
stats: {
|
|
431
|
-
sessionsImported:
|
|
432
|
-
medianMs: percentile(
|
|
433
|
-
p90Ms: percentile(
|
|
582
|
+
sessionsImported: agentRows.length,
|
|
583
|
+
medianMs: percentile(activeDurations, 0.5),
|
|
584
|
+
p90Ms: percentile(activeDurations, 0.9),
|
|
585
|
+
agentMedianMs: percentile(agentActive, 0.5),
|
|
586
|
+
agentP90Ms: percentile(agentActive, 0.9),
|
|
587
|
+
interactiveMedianMs: percentile(interactiveActive, 0.5),
|
|
588
|
+
measuredFraction,
|
|
434
589
|
needAttention: needsAttention.length,
|
|
435
|
-
toolErrorRate:
|
|
590
|
+
toolErrorRate: agentCalls.length === 0 ? 0 : failedCalls.length / agentCalls.length,
|
|
436
591
|
},
|
|
592
|
+
utilityCount,
|
|
437
593
|
needsAttention,
|
|
438
|
-
topics: [...topicCounts.values()]
|
|
594
|
+
topics: [...topicCounts.values()]
|
|
595
|
+
.sort((a, b) => b.count - a.count || a.key.localeCompare(b.key))
|
|
596
|
+
.map((bucket) => ({
|
|
597
|
+
key: bucket.key,
|
|
598
|
+
label: bucket.label,
|
|
599
|
+
count: bucket.count,
|
|
600
|
+
group: bucket.group,
|
|
601
|
+
// Up to TOPIC_SESSION_CAP most-recent example sessions, so the console can
|
|
602
|
+
// drill from a tile into its session list. Capped to keep the shard small;
|
|
603
|
+
// the tile's `count` remains the true total. These are correct on this
|
|
604
|
+
// per-device shard; the fleet-aggregate `/all` view (worker-template.ts,
|
|
605
|
+
// a must-not-touch R2 worker here) carries only the first device's refs
|
|
606
|
+
// per topic until PHNX-3464 merges them across devices.
|
|
607
|
+
sessions: bucket.refs
|
|
608
|
+
.sort((a, b) => b.recencyMs - a.recencyMs)
|
|
609
|
+
.slice(0, TOPIC_SESSION_CAP)
|
|
610
|
+
.map(({ id, title, harness }) => ({ id, title, kind: 'agent', harness })),
|
|
611
|
+
})),
|
|
439
612
|
failures: {
|
|
440
613
|
byToolError: [...failureCounts.values()].sort((a, b) => b.count - a.count || a.tool.localeCompare(b.tool)),
|
|
441
614
|
byCause,
|
|
@@ -503,6 +676,7 @@ export function buildSessionDetail(traj) {
|
|
|
503
676
|
id: s.id,
|
|
504
677
|
meta: {
|
|
505
678
|
spanMs: traj.spanMs,
|
|
679
|
+
activeMs: activeMsFromTrajectory(traj),
|
|
506
680
|
turns: (stats.userTurns ?? 0) + (stats.assistantTurns ?? 0),
|
|
507
681
|
tools: stats.toolCount ?? 0,
|
|
508
682
|
errorCount: traj.errorCount,
|
package/dist/lib/view-types.d.ts
CHANGED
|
@@ -9,6 +9,18 @@ export interface ViewJsonVersion {
|
|
|
9
9
|
isolated: boolean;
|
|
10
10
|
isIsolatedDefault: boolean;
|
|
11
11
|
signedIn: boolean;
|
|
12
|
+
/**
|
|
13
|
+
* Whether THIS version home can actually spawn a signed-in agent — the strict
|
|
14
|
+
* per-version launch truth (`isLaunchableSignedIn`), not the display `signedIn`
|
|
15
|
+
* above. `signedIn` is true when the version *inherits* the active/global HOME
|
|
16
|
+
* login even with no per-version credential of its own; such a home shows "who
|
|
17
|
+
* is logged in" but dies at spawn once launch isolates HOME to it. Automatic
|
|
18
|
+
* `--device auto` placement gates on THIS field so a remote box is judged by
|
|
19
|
+
* the same launchability the local candidate uses (`collectRunCandidates` →
|
|
20
|
+
* `isLaunchableSignedIn`), closing the local/remote asymmetry (PHNX-3466).
|
|
21
|
+
* Absent on an older remote CLI, whose consumers fall back to `signedIn`.
|
|
22
|
+
*/
|
|
23
|
+
launchable: boolean;
|
|
12
24
|
/** Live cached authentication verdict for this installed version. */
|
|
13
25
|
authVerdict: AuthVerdict | null;
|
|
14
26
|
email: string | null;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@phnx-labs/agents-cli",
|
|
3
|
-
"version": "1.22.
|
|
3
|
+
"version": "1.22.60",
|
|
4
4
|
"description": "One CLI for all your AI coding agents - versions, config, cloud dispatch, sessions, and teams (now with first-class Grok Build CLI support)",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -88,7 +88,7 @@
|
|
|
88
88
|
"dependencies": {
|
|
89
89
|
"@fontsource/inter": "^5.3.0",
|
|
90
90
|
"@fontsource/jetbrains-mono": "^5.3.0",
|
|
91
|
-
"@homebridge/node-pty-prebuilt-multiarch": "0.
|
|
91
|
+
"@homebridge/node-pty-prebuilt-multiarch": "0.14.1",
|
|
92
92
|
"@inquirer/prompts": "8.5.2",
|
|
93
93
|
"@resvg/resvg-wasm": "^2.6.2",
|
|
94
94
|
"@types/proper-lockfile": "4.1.4",
|