@phnx-labs/agents-cli 1.22.56 → 1.22.58
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +70 -0
- package/README.md +4 -4
- package/dist/bootstrap.js +11 -2
- package/dist/cli/command-registry.d.ts +0 -1
- package/dist/cli/command-registry.js +0 -3
- package/dist/commands/accounts.js +7 -3
- package/dist/commands/apply.js +10 -2
- package/dist/commands/exec.js +1 -1
- package/dist/commands/fork.d.ts +23 -10
- package/dist/commands/fork.js +115 -58
- package/dist/commands/hooks.js +4 -4
- package/dist/commands/insights.d.ts +7 -5
- package/dist/commands/insights.js +16 -9
- package/dist/commands/monitors.js +11 -0
- package/dist/commands/perf.d.ts +16 -7
- package/dist/commands/perf.js +29 -20
- package/dist/commands/prune.js +5 -3
- package/dist/commands/routines.d.ts +8 -0
- package/dist/commands/routines.js +57 -3
- package/dist/commands/rules.js +1 -1
- package/dist/commands/sessions-picker.d.ts +11 -0
- package/dist/commands/sessions-picker.js +16 -0
- package/dist/commands/sessions.js +1 -0
- package/dist/commands/share.d.ts +14 -0
- package/dist/commands/share.js +43 -2
- package/dist/commands/ssh.js +24 -14
- package/dist/commands/status.js +1 -1
- package/dist/commands/sync.js +83 -7
- package/dist/commands/traces.js +7 -0
- package/dist/commands/trash.d.ts +2 -2
- package/dist/commands/trash.js +2 -6
- package/dist/commands/versions.d.ts +2 -2
- package/dist/commands/versions.js +1 -10
- package/dist/commands/view.d.ts +2 -2
- package/dist/commands/view.js +7 -6
- package/dist/index.d.ts +1 -0
- package/dist/index.js +14 -0
- package/dist/lib/account-registry.d.ts +5 -1
- package/dist/lib/account-registry.js +47 -14
- package/dist/lib/accounting/capacity.d.ts +18 -7
- package/dist/lib/accounting/capacity.js +19 -8
- package/dist/lib/accounting/usage-ingest.d.ts +1 -0
- package/dist/lib/accounting/usage-ingest.js +75 -0
- package/dist/lib/accounting/usage-sync.d.ts +97 -0
- package/dist/lib/accounting/usage-sync.js +203 -0
- package/dist/lib/accounting/usage.d.ts +48 -2
- package/dist/lib/accounting/usage.js +79 -2
- package/dist/lib/agent-spec/agents.js +1 -1
- package/dist/lib/analytics/mix-commands.d.ts +8 -7
- package/dist/lib/analytics/mix-commands.js +50 -73
- package/dist/lib/auth-mint.d.ts +11 -1
- package/dist/lib/auth-mint.js +21 -6
- package/dist/lib/browser/ipc.d.ts +8 -0
- package/dist/lib/browser/ipc.js +87 -0
- package/dist/lib/browser/service.d.ts +19 -0
- package/dist/lib/browser/service.js +96 -11
- package/dist/lib/browser/sessions-list.js +10 -1
- package/dist/lib/daemon/daemon.js +5 -0
- package/dist/lib/daemon/runner.d.ts +3 -0
- package/dist/lib/daemon/runner.js +95 -53
- package/dist/lib/daemon/usage-sync-service.d.ts +21 -0
- package/dist/lib/daemon/usage-sync-service.js +42 -0
- package/dist/lib/daemon-services.d.ts +1 -1
- package/dist/lib/daemon-services.js +5 -0
- package/dist/lib/device-config.d.ts +17 -6
- package/dist/lib/device-config.js +25 -11
- package/dist/lib/devices/connect.d.ts +17 -8
- package/dist/lib/devices/connect.js +31 -14
- package/dist/lib/devices/pool.d.ts +4 -3
- package/dist/lib/devices/pool.js +13 -5
- package/dist/lib/doctor-diff.js +77 -7
- package/dist/lib/exec.d.ts +6 -41
- package/dist/lib/exec.js +6 -41
- package/dist/lib/fleet/manifest.d.ts +17 -0
- package/dist/lib/fleet/manifest.js +26 -0
- package/dist/lib/git.d.ts +13 -1
- package/dist/lib/git.js +36 -7
- package/dist/lib/harness/adapter.d.ts +7 -7
- package/dist/lib/harness/adapters/claude.js +3 -2
- package/dist/lib/hooks/install.d.ts +27 -11
- package/dist/lib/hooks/install.js +42 -17
- package/dist/lib/hosts/reconnect.d.ts +52 -203
- package/dist/lib/hosts/reconnect.js +64 -284
- package/dist/lib/hosts/remote-cmd.d.ts +9 -0
- package/dist/lib/hosts/remote-cmd.js +22 -0
- package/dist/lib/installations/migrate.d.ts +6 -120
- package/dist/lib/installations/migrate.js +27 -259
- package/dist/lib/installations/shims.d.ts +13 -95
- package/dist/lib/installations/shims.js +22 -139
- package/dist/lib/installations/store.js +1 -1
- package/dist/lib/installations/versions.d.ts +26 -133
- package/dist/lib/installations/versions.js +41 -204
- package/dist/lib/perf/db.d.ts +1 -1
- package/dist/lib/perf/db.js +1 -1
- package/dist/lib/plugins/skills.d.ts +8 -1
- package/dist/lib/plugins/skills.js +18 -2
- package/dist/lib/refresh.d.ts +9 -0
- package/dist/lib/refresh.js +3 -1
- package/dist/lib/routine-readiness.d.ts +15 -1
- package/dist/lib/routine-readiness.js +41 -0
- package/dist/lib/sandbox.d.ts +4 -1
- package/dist/lib/sandbox.js +30 -1
- package/dist/lib/secrets/agent.d.ts +80 -225
- package/dist/lib/secrets/agent.js +139 -401
- package/dist/lib/secrets/bundles.d.ts +73 -222
- package/dist/lib/secrets/bundles.js +168 -467
- package/dist/lib/secrets/reaper.d.ts +28 -70
- package/dist/lib/secrets/reaper.js +30 -85
- package/dist/lib/secrets/remote.d.ts +42 -129
- package/dist/lib/secrets/remote.js +55 -173
- package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
- package/dist/lib/self-heal/checks/install-staging.js +96 -0
- package/dist/lib/self-heal/registry.js +2 -0
- package/dist/lib/self-heal/types.d.ts +1 -1
- package/dist/lib/self-update.d.ts +23 -0
- package/dist/lib/self-update.js +50 -0
- package/dist/lib/session/active.d.ts +16 -32
- package/dist/lib/session/active.js +10 -68
- package/dist/lib/session/db.d.ts +24 -36
- package/dist/lib/session/db.js +143 -44
- package/dist/lib/session/discover.d.ts +6 -58
- package/dist/lib/session/discover.js +5 -43
- package/dist/lib/session/fork.d.ts +45 -26
- package/dist/lib/session/fork.js +32 -95
- package/dist/lib/session/parse.d.ts +1 -19
- package/dist/lib/session/parse.js +2 -15
- package/dist/lib/session/tool-calls.d.ts +43 -1
- package/dist/lib/session/tool-calls.js +74 -44
- package/dist/lib/session/tool-store.d.ts +33 -2
- package/dist/lib/session/tool-store.js +56 -3
- package/dist/lib/staleness/writers/sources.d.ts +5 -0
- package/dist/lib/staleness/writers/sources.js +2 -1
- package/dist/lib/startup/command-registry.d.ts +8 -2
- package/dist/lib/startup/command-registry.js +12 -4
- package/dist/lib/sync-status.d.ts +22 -0
- package/dist/lib/sync-status.js +27 -0
- package/dist/lib/sync-umbrella.d.ts +9 -0
- package/dist/lib/sync-umbrella.js +21 -2
- package/dist/lib/traces/insights.d.ts +47 -14
- package/dist/lib/traces/insights.js +92 -21
- package/dist/lib/traces/phenotype.d.ts +23 -3
- package/dist/lib/traces/phenotype.js +72 -24
- package/dist/lib/traces/sync.d.ts +15 -0
- package/dist/lib/traces/sync.js +104 -19
- package/dist/lib/traces/worker-template.js +154 -1
- package/package.json +1 -1
|
@@ -35,13 +35,33 @@ export interface OutcomeResult {
|
|
|
35
35
|
confidence: 'high' | 'medium' | 'low';
|
|
36
36
|
reason: string;
|
|
37
37
|
}
|
|
38
|
+
/**
|
|
39
|
+
* Did a run that hit tool errors nonetheless *recover and finish*?
|
|
40
|
+
*
|
|
41
|
+
* The causal recovery test: a **substantive** tool step (not a human-facing
|
|
42
|
+
* `AskUserQuestion` / `SendMessage` / `wait`) SUCCEEDED strictly AFTER the last
|
|
43
|
+
* error's ordinal, AND that success resolves the failed work — its
|
|
44
|
+
* {@link workSignature} matches an errored step's. A later success of *unrelated*
|
|
45
|
+
* work does NOT count: a `bun test` failure followed by an incidental `ls` leaves
|
|
46
|
+
* the failed test unresolved, so the run stays errored, even though the `ls`
|
|
47
|
+
* succeeded after it. A punt to a human is excluded twice over — human-facing
|
|
48
|
+
* tools are outside the substantive set, and they never match a failed signature.
|
|
49
|
+
*
|
|
50
|
+
* This is the single source of truth for "did this finish", shared by the
|
|
51
|
+
* false-termination phenotype below and `sync.ts`'s `deriveRunOutcome` — so a
|
|
52
|
+
* run's console outcome and its failure phenotype can never disagree about
|
|
53
|
+
* whether it recovered. It reads only the derived steps, never `meta.outcome`,
|
|
54
|
+
* precisely so `deriveRunOutcome` can call it while it is still *computing*
|
|
55
|
+
* `meta.outcome`.
|
|
56
|
+
*/
|
|
57
|
+
export declare function recoveredAfterErrors(session: Pick<SessionDetail, 'steps'>): boolean;
|
|
38
58
|
/**
|
|
39
59
|
* Classify the failure phenotype of a session from its derived trajectory.
|
|
40
60
|
*
|
|
41
61
|
* Definitions (from agent-failure research):
|
|
42
|
-
* - `false-termination` — stopped with an unresolved error.
|
|
43
|
-
* - `premature-completion` — declared done
|
|
44
|
-
*
|
|
62
|
+
* - `false-termination` — stopped with an unresolved error (did not causally recover).
|
|
63
|
+
* - `premature-completion` — declared done with no verification step for the
|
|
64
|
+
* engineering work.
|
|
45
65
|
* - `out-of-order` — a write/edit step occurred before any read/plan of the
|
|
46
66
|
* target.
|
|
47
67
|
* - `failure-to-act` — stalled or produced no meaningful tool use.
|
|
@@ -156,30 +156,75 @@ function reasonOutOfOrder(session) {
|
|
|
156
156
|
const tool = session.steps.find((s) => WRITE_EDIT_TOOLS.has(s.tool ?? s.lane));
|
|
157
157
|
return `write/edit step ${tool?.tool ?? tool?.lane ?? ''} preceded any read/plan`;
|
|
158
158
|
}
|
|
159
|
+
/**
|
|
160
|
+
* The **work signature** of a step — what work it represents, so a later success
|
|
161
|
+
* can be matched back to the specific failure it resolves. For a shell step this
|
|
162
|
+
* is the effective program (`bun`, `git`, `gh`, …, from `TrajectoryStep.program`),
|
|
163
|
+
* so a failed `bun test` is resolved by a later `bun test` but NOT by an incidental
|
|
164
|
+
* `ls`. For any other tool it is the tool identity, so a failed `Edit` is resolved
|
|
165
|
+
* by a later successful `Edit`, not by an unrelated `Read`. A shell step whose
|
|
166
|
+
* command did not parse (no `program`) degrades to the bare tool name, matching
|
|
167
|
+
* the pre-`program` behavior only for that unparseable minority.
|
|
168
|
+
*/
|
|
169
|
+
function workSignature(step) {
|
|
170
|
+
const tool = step.tool ?? step.lane;
|
|
171
|
+
if (SHELL_TOOLS.has(tool) && step.program)
|
|
172
|
+
return `${tool}:${step.program}`;
|
|
173
|
+
return tool;
|
|
174
|
+
}
|
|
175
|
+
/**
|
|
176
|
+
* Did a run that hit tool errors nonetheless *recover and finish*?
|
|
177
|
+
*
|
|
178
|
+
* The causal recovery test: a **substantive** tool step (not a human-facing
|
|
179
|
+
* `AskUserQuestion` / `SendMessage` / `wait`) SUCCEEDED strictly AFTER the last
|
|
180
|
+
* error's ordinal, AND that success resolves the failed work — its
|
|
181
|
+
* {@link workSignature} matches an errored step's. A later success of *unrelated*
|
|
182
|
+
* work does NOT count: a `bun test` failure followed by an incidental `ls` leaves
|
|
183
|
+
* the failed test unresolved, so the run stays errored, even though the `ls`
|
|
184
|
+
* succeeded after it. A punt to a human is excluded twice over — human-facing
|
|
185
|
+
* tools are outside the substantive set, and they never match a failed signature.
|
|
186
|
+
*
|
|
187
|
+
* This is the single source of truth for "did this finish", shared by the
|
|
188
|
+
* false-termination phenotype below and `sync.ts`'s `deriveRunOutcome` — so a
|
|
189
|
+
* run's console outcome and its failure phenotype can never disagree about
|
|
190
|
+
* whether it recovered. It reads only the derived steps, never `meta.outcome`,
|
|
191
|
+
* precisely so `deriveRunOutcome` can call it while it is still *computing*
|
|
192
|
+
* `meta.outcome`.
|
|
193
|
+
*/
|
|
194
|
+
export function recoveredAfterErrors(session) {
|
|
195
|
+
const substantive = substantiveSteps(session);
|
|
196
|
+
if (substantive.length === 0)
|
|
197
|
+
return false;
|
|
198
|
+
const last = substantive[substantive.length - 1];
|
|
199
|
+
if (last.outcome === 'error')
|
|
200
|
+
return false;
|
|
201
|
+
const lastErrorOrdinal = lastStepOrdinalOf(session, (s) => s.outcome === 'error');
|
|
202
|
+
if (lastErrorOrdinal === undefined)
|
|
203
|
+
return true;
|
|
204
|
+
const failedSignatures = new Set();
|
|
205
|
+
for (const s of session.steps) {
|
|
206
|
+
if (s.outcome === 'error')
|
|
207
|
+
failedSignatures.add(workSignature(s));
|
|
208
|
+
}
|
|
209
|
+
return substantive.some((s) => s.ordinal > lastErrorOrdinal &&
|
|
210
|
+
s.outcome === 'ok' &&
|
|
211
|
+
!HUMAN_FACING_TOOLS.has(s.tool ?? s.lane) &&
|
|
212
|
+
failedSignatures.has(workSignature(s)));
|
|
213
|
+
}
|
|
159
214
|
/**
|
|
160
215
|
* False-termination: the session stopped with an unresolved error.
|
|
161
216
|
*
|
|
162
217
|
* Rubric:
|
|
163
218
|
* - meta outcome is `errored`, and
|
|
164
219
|
* - at least one step outcome is `error`, and
|
|
165
|
-
* - the
|
|
220
|
+
* - the run did not causally recover ({@link recoveredAfterErrors}).
|
|
166
221
|
*/
|
|
167
222
|
function isFalseTermination(session) {
|
|
168
223
|
if (session.meta.outcome !== 'errored')
|
|
169
224
|
return false;
|
|
170
225
|
if (session.meta.errorCount === 0)
|
|
171
226
|
return false;
|
|
172
|
-
|
|
173
|
-
if (substantive.length === 0)
|
|
174
|
-
return true;
|
|
175
|
-
const last = substantive[substantive.length - 1];
|
|
176
|
-
if (last.outcome === 'error')
|
|
177
|
-
return true;
|
|
178
|
-
const lastErrorOrdinal = lastStepOrdinalOf(session, (s) => s.outcome === 'error');
|
|
179
|
-
if (lastErrorOrdinal === undefined)
|
|
180
|
-
return false;
|
|
181
|
-
const recoveryAfter = substantive.some((s) => s.ordinal > lastErrorOrdinal && s.outcome === 'ok' && !HUMAN_FACING_TOOLS.has(s.tool ?? s.lane));
|
|
182
|
-
return !recoveryAfter;
|
|
227
|
+
return !recoveredAfterErrors(session);
|
|
183
228
|
}
|
|
184
229
|
function reasonFalseTermination(session) {
|
|
185
230
|
const substantive = substantiveSteps(session);
|
|
@@ -192,13 +237,21 @@ function reasonFalseTermination(session) {
|
|
|
192
237
|
return 'errored with no successful recovery after the last error';
|
|
193
238
|
}
|
|
194
239
|
/**
|
|
195
|
-
* Premature-completion: declared done
|
|
196
|
-
*
|
|
240
|
+
* Premature-completion: declared done without a verification step for an
|
|
241
|
+
* engineering task.
|
|
197
242
|
*
|
|
198
243
|
* Rubric:
|
|
199
244
|
* - meta outcome is `completed`, and
|
|
200
245
|
* - the session performed write/edit work, and
|
|
201
|
-
* -
|
|
246
|
+
* - no test/build/lint verification step ran.
|
|
247
|
+
*
|
|
248
|
+
* `errorCount` is deliberately NOT a signal here. Under truthful outcomes
|
|
249
|
+
* (PHNX-3387) a `completed` run CAN carry `errorCount > 0` — that is precisely a
|
|
250
|
+
* recover-then-succeed run, where the errors were *resolved*, not left hanging.
|
|
251
|
+
* Keying prematurity off `errorCount > 0` would mislabel every such recovery as
|
|
252
|
+
* premature. (Under the pre-PHNX-3387 `errorCount > 0 ? errored : completed`
|
|
253
|
+
* derivation this branch was unreachable — a `completed` run always had
|
|
254
|
+
* `errorCount === 0` — so dropping it changes nothing for a clean-completed run.)
|
|
202
255
|
*/
|
|
203
256
|
function isPrematureCompletion(session) {
|
|
204
257
|
if (session.meta.outcome !== 'completed')
|
|
@@ -206,8 +259,6 @@ function isPrematureCompletion(session) {
|
|
|
206
259
|
const didWriteEdit = session.steps.some((s) => WRITE_EDIT_TOOLS.has(s.tool ?? s.lane));
|
|
207
260
|
if (!didWriteEdit)
|
|
208
261
|
return false;
|
|
209
|
-
if (session.meta.errorCount > 0)
|
|
210
|
-
return true;
|
|
211
262
|
const verified = session.steps.some((s) => {
|
|
212
263
|
if (!SHELL_TOOLS.has(s.tool ?? s.lane))
|
|
213
264
|
return false;
|
|
@@ -216,10 +267,7 @@ function isPrematureCompletion(session) {
|
|
|
216
267
|
});
|
|
217
268
|
return !verified;
|
|
218
269
|
}
|
|
219
|
-
function reasonPrematureCompletion(
|
|
220
|
-
if (session.meta.errorCount > 0) {
|
|
221
|
-
return `declared completed with ${session.meta.errorCount} unresolved error(s)`;
|
|
222
|
-
}
|
|
270
|
+
function reasonPrematureCompletion(_session) {
|
|
223
271
|
return 'engineering work completed without a test/build/lint verification step';
|
|
224
272
|
}
|
|
225
273
|
/** Ordered rubric: the first matching phenotype wins. */
|
|
@@ -391,9 +439,9 @@ const OUTCOME_RULES = [
|
|
|
391
439
|
* Classify the failure phenotype of a session from its derived trajectory.
|
|
392
440
|
*
|
|
393
441
|
* Definitions (from agent-failure research):
|
|
394
|
-
* - `false-termination` — stopped with an unresolved error.
|
|
395
|
-
* - `premature-completion` — declared done
|
|
396
|
-
*
|
|
442
|
+
* - `false-termination` — stopped with an unresolved error (did not causally recover).
|
|
443
|
+
* - `premature-completion` — declared done with no verification step for the
|
|
444
|
+
* engineering work.
|
|
397
445
|
* - `out-of-order` — a write/edit step occurred before any read/plan of the
|
|
398
446
|
* target.
|
|
399
447
|
* - `failure-to-act` — stalled or produced no meaningful tool use.
|
|
@@ -49,6 +49,14 @@ export interface SyncResult {
|
|
|
49
49
|
parseFailed: number;
|
|
50
50
|
/** Parsed fine but the upload PUT failed (network/5xx) — retried on the next sync. */
|
|
51
51
|
uploadFailed: number;
|
|
52
|
+
/**
|
|
53
|
+
* The index shard build or upload failed. Undefined on success. The per-session
|
|
54
|
+
* data still uploaded (that loop runs first), but the aggregated console shard —
|
|
55
|
+
* stats, needs-attention, failure clusters, latency — was NOT refreshed, so the
|
|
56
|
+
* console keeps serving the last good index. Surfaced (not swallowed) so a stale
|
|
57
|
+
* console is diagnosable instead of looking like a clean sync. See PHNX-3401.
|
|
58
|
+
*/
|
|
59
|
+
indexError?: string;
|
|
52
60
|
}
|
|
53
61
|
/** Push derived, redacted trajectories for this device to the traces store. */
|
|
54
62
|
export declare function syncTraces(opts?: SyncOpts): Promise<SyncResult>;
|
|
@@ -106,6 +114,13 @@ export interface ToolCallRow {
|
|
|
106
114
|
session_id: string;
|
|
107
115
|
ordinal: number;
|
|
108
116
|
timestamp: string;
|
|
117
|
+
/**
|
|
118
|
+
* When the call's result arrived — its own end time (PHNX-3437). Optional
|
|
119
|
+
* because rows an older extractor stored, and calls that never produced a
|
|
120
|
+
* result, carry NULL; `computeInsights` falls back to the bounded inter-call
|
|
121
|
+
* gap when it is absent.
|
|
122
|
+
*/
|
|
123
|
+
end_timestamp?: string | null;
|
|
109
124
|
tool: string;
|
|
110
125
|
outcome: string;
|
|
111
126
|
exit_code: number | null;
|
package/dist/lib/traces/sync.js
CHANGED
|
@@ -22,7 +22,7 @@
|
|
|
22
22
|
import fs from 'node:fs';
|
|
23
23
|
import os from 'node:os';
|
|
24
24
|
import path from 'node:path';
|
|
25
|
-
import { getDB, readSessionInsights, readSessionTopics, writeSessionInsights, writeSessionTopics, } from '../session/db.js';
|
|
25
|
+
import { getDB, readSessionInsights, readSessionPhenotypes, readSessionTopics, writeSessionInsights, writeSessionPhenotypes, writeSessionTopics, } from '../session/db.js';
|
|
26
26
|
import { parseSession } from '../session/parse.js';
|
|
27
27
|
import { buildTrajectory } from '../session/trajectory.js';
|
|
28
28
|
import { computeInsightFacets } from '../session/insights.js';
|
|
@@ -31,6 +31,7 @@ import { getRuntimeStateDir } from '../state.js';
|
|
|
31
31
|
import { resolveTracesBackend } from './backend.js';
|
|
32
32
|
import { classifyCause, classifyTopic, computeDriftSignal, } from './classify.js';
|
|
33
33
|
import { computeInsights } from './insights.js';
|
|
34
|
+
import { classifyPhenotype, recoveredAfterErrors } from './phenotype.js';
|
|
34
35
|
/** Push derived, redacted trajectories for this device to the traces store. */
|
|
35
36
|
export async function syncTraces(opts = {}) {
|
|
36
37
|
const dryRun = opts.dryRun === true;
|
|
@@ -147,6 +148,7 @@ export async function syncTraces(opts = {}) {
|
|
|
147
148
|
recordFailure(row, 'upload-failed', err);
|
|
148
149
|
}
|
|
149
150
|
}
|
|
151
|
+
let indexError;
|
|
150
152
|
if (!opts.skipIndex) {
|
|
151
153
|
try {
|
|
152
154
|
const allRows = db
|
|
@@ -172,8 +174,13 @@ export async function syncTraces(opts = {}) {
|
|
|
172
174
|
await putIndexShard(backend, device, owner, shard);
|
|
173
175
|
}
|
|
174
176
|
}
|
|
175
|
-
catch {
|
|
176
|
-
//
|
|
177
|
+
catch (err) {
|
|
178
|
+
// Not fatal to the per-session upload (that loop already ran), but it DOES
|
|
179
|
+
// mean the console shard is now stale. Record it so the caller can surface a
|
|
180
|
+
// warning instead of reporting a clean, green sync — a silent swallow here is
|
|
181
|
+
// exactly what let a 59h-stale, insight-less index hide in plain sight
|
|
182
|
+
// (PHNX-3401). Do NOT re-throw: the session data is durable and worth keeping.
|
|
183
|
+
indexError = err instanceof Error ? err.message : String(err);
|
|
177
184
|
}
|
|
178
185
|
}
|
|
179
186
|
// A dry-run never advances the incremental watermark: it is a read-only export.
|
|
@@ -203,6 +210,7 @@ export async function syncTraces(opts = {}) {
|
|
|
203
210
|
transcriptUnavailable,
|
|
204
211
|
parseFailed,
|
|
205
212
|
uploadFailed,
|
|
213
|
+
indexError,
|
|
206
214
|
};
|
|
207
215
|
}
|
|
208
216
|
function percentile(values, ratio) {
|
|
@@ -247,6 +255,32 @@ function attentionFlags(errorCount, facets) {
|
|
|
247
255
|
flags.push(`${correctionCount} correction${correctionCount === 1 ? '' : 's'}`);
|
|
248
256
|
return flags;
|
|
249
257
|
}
|
|
258
|
+
/**
|
|
259
|
+
* Persist a derived-cache warm-up (topics / insights) without letting a
|
|
260
|
+
* contended DB take down the whole index build.
|
|
261
|
+
*
|
|
262
|
+
* These write-backs only speed up the NEXT sync — the shard about to be built
|
|
263
|
+
* reads from the in-memory `topics` / `insights` maps that were already
|
|
264
|
+
* populated above, never from what this write persists. So the write is
|
|
265
|
+
* genuinely optional to the shard's correctness.
|
|
266
|
+
*
|
|
267
|
+
* Yet it was the single point that broke the console: on an active machine the
|
|
268
|
+
* Rush app holds `sessions.db`, this `BEGIN IMMEDIATE` waits out the 30s
|
|
269
|
+
* `busy_timeout` and throws `SQLITE_BUSY`, the throw escaped `buildIndexShard`,
|
|
270
|
+
* and `syncTraces` swallowed it — so the index (with wasted-time / failure
|
|
271
|
+
* clusters / latency) never re-uploaded and the dashboard sat 59h stale
|
|
272
|
+
* (PHNX-3401). Isolating the failure here keeps the index building; the warning
|
|
273
|
+
* makes the degraded cache visible instead of silent.
|
|
274
|
+
*/
|
|
275
|
+
function persistDerivedCache(label, write) {
|
|
276
|
+
try {
|
|
277
|
+
write();
|
|
278
|
+
}
|
|
279
|
+
catch (err) {
|
|
280
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
281
|
+
console.warn(`traces: ${label} cache warm-up skipped (${msg}) — index still built`);
|
|
282
|
+
}
|
|
283
|
+
}
|
|
250
284
|
/** Build the redacted rich console shard from indexed metadata and derived caches. */
|
|
251
285
|
export function buildIndexShard(rows, device, owner, prevShard) {
|
|
252
286
|
const db = getDB();
|
|
@@ -256,7 +290,7 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
256
290
|
for (let i = 0; i < ids.length; i += 400) {
|
|
257
291
|
const chunk = ids.slice(i, i + 400);
|
|
258
292
|
calls.push(...db.prepare(`
|
|
259
|
-
SELECT session_id, ordinal, timestamp, tool, outcome, exit_code, status_code, error_code, error, parse_error
|
|
293
|
+
SELECT session_id, ordinal, timestamp, end_timestamp, tool, outcome, exit_code, status_code, error_code, error, parse_error
|
|
260
294
|
FROM tool_calls
|
|
261
295
|
WHERE session_id IN (${chunk.map(() => '?').join(',')})
|
|
262
296
|
`).all(...chunk));
|
|
@@ -283,26 +317,52 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
283
317
|
topics.set(row.id, topic);
|
|
284
318
|
return { id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, topic };
|
|
285
319
|
});
|
|
286
|
-
writeSessionTopics(missingTopics);
|
|
320
|
+
persistDerivedCache('session-topics', () => writeSessionTopics(missingTopics));
|
|
321
|
+
// Insights (frictionSignals) and phenotype (false-termination / …) both need
|
|
322
|
+
// the parsed transcript, which flat tool_calls rows don't carry — so both are
|
|
323
|
+
// lazily derived per-session and cached by transcript mtime+size, then read for
|
|
324
|
+
// the WHOLE corpus (`rows` = allRows) every sync. That full-corpus read is what
|
|
325
|
+
// keeps the phenotype grouping dimension consistent: a session synced weeks ago
|
|
326
|
+
// still contributes its real phenotype from cache, so it can never fragment away
|
|
327
|
+
// from an identically-signatured session synced this run purely by *when* each
|
|
328
|
+
// was first seen (PHNX-3327). A cache-miss row is parsed at most once here even
|
|
329
|
+
// when both derivations are missing.
|
|
287
330
|
const insights = readSessionInsights(ids);
|
|
331
|
+
const phenotypes = readSessionPhenotypes(ids);
|
|
288
332
|
const missingInsights = [];
|
|
289
|
-
|
|
333
|
+
const missingPhenotypes = [];
|
|
334
|
+
for (const row of rows) {
|
|
335
|
+
const needInsights = !insights.has(row.id);
|
|
336
|
+
const needPhenotype = !phenotypes.has(row.id);
|
|
337
|
+
if (!needInsights && !needPhenotype)
|
|
338
|
+
continue;
|
|
339
|
+
let events;
|
|
290
340
|
try {
|
|
291
|
-
|
|
292
|
-
const facets = computeInsightFacets(events);
|
|
293
|
-
insights.set(row.id, facets);
|
|
294
|
-
missingInsights.push({
|
|
295
|
-
id: row.id,
|
|
296
|
-
fileMtimeMs: row.file_mtime_ms,
|
|
297
|
-
fileSize: row.file_size,
|
|
298
|
-
facets,
|
|
299
|
-
});
|
|
341
|
+
events = parseSession(row.file_path, row.agent);
|
|
300
342
|
}
|
|
301
343
|
catch {
|
|
302
|
-
continue;
|
|
344
|
+
continue; // gone/unreadable transcript — leave both uncached, same as before
|
|
345
|
+
}
|
|
346
|
+
if (needInsights) {
|
|
347
|
+
try {
|
|
348
|
+
const facets = computeInsightFacets(events);
|
|
349
|
+
insights.set(row.id, facets);
|
|
350
|
+
missingInsights.push({ id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, facets });
|
|
351
|
+
}
|
|
352
|
+
catch { /* leave this session's facets uncached; recompute next sync */ }
|
|
353
|
+
}
|
|
354
|
+
if (needPhenotype) {
|
|
355
|
+
try {
|
|
356
|
+
const traj = buildTrajectory(events, rowToMeta(row), { redact: true, knownSecrets });
|
|
357
|
+
const phenotype = classifyPhenotype(buildSessionDetail(traj));
|
|
358
|
+
phenotypes.set(row.id, phenotype);
|
|
359
|
+
missingPhenotypes.push({ id: row.id, fileMtimeMs: row.file_mtime_ms, fileSize: row.file_size, phenotype });
|
|
360
|
+
}
|
|
361
|
+
catch { /* leave this session's phenotype uncached; recompute next sync */ }
|
|
303
362
|
}
|
|
304
363
|
}
|
|
305
|
-
writeSessionInsights(missingInsights);
|
|
364
|
+
persistDerivedCache('session-insights', () => writeSessionInsights(missingInsights));
|
|
365
|
+
persistDerivedCache('session-phenotypes', () => writeSessionPhenotypes(missingPhenotypes));
|
|
306
366
|
const needsAttention = rows.flatMap((row) => {
|
|
307
367
|
const facets = insights.get(row.id);
|
|
308
368
|
const errorCount = errorCounts.get(row.id) ?? 0;
|
|
@@ -361,7 +421,7 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
361
421
|
const prevHistory = prevShard?.bucketHistory ?? [];
|
|
362
422
|
const bucketHistory = [...prevHistory, todayStats].slice(-14);
|
|
363
423
|
const driftSignals = computeDriftSignal(prevHistory, todayStats);
|
|
364
|
-
const patternInsights = computeInsights(rows, calls, prevShard);
|
|
424
|
+
const patternInsights = computeInsights(rows, calls, prevShard, phenotypes);
|
|
365
425
|
return {
|
|
366
426
|
schema: 1,
|
|
367
427
|
device,
|
|
@@ -404,6 +464,31 @@ function buildWhereItWentWrong(traj) {
|
|
|
404
464
|
return null;
|
|
405
465
|
return `This run hit ${parts.join('; ')}.`;
|
|
406
466
|
}
|
|
467
|
+
/**
|
|
468
|
+
* Truthful run-level outcome (PHNX-3387).
|
|
469
|
+
*
|
|
470
|
+
* A run with zero tool errors `completed`. A run that hit tool errors is
|
|
471
|
+
* `completed` ONLY when it *causally recovered* — a substantive, non-human-facing
|
|
472
|
+
* tool step succeeded strictly after the last error AND resolved the failed work
|
|
473
|
+
* (its work signature matches an errored step's), the exact predicate the
|
|
474
|
+
* false-termination phenotype uses ({@link recoveredAfterErrors}). A run whose
|
|
475
|
+
* last substantive step is the error, whose only post-error steps are human-facing
|
|
476
|
+
* (a punt to `AskUserQuestion` — the case the broken "last tool call ok" heuristic
|
|
477
|
+
* mislabeled `completed`), or whose only post-error success is unrelated work (a
|
|
478
|
+
* failed `bun test` followed by an incidental `ls`) stays `errored`.
|
|
479
|
+
*
|
|
480
|
+
* This is what makes `surfacedToolFailures` on a `completed` run honest: those
|
|
481
|
+
* are failures the run recovered from, not a green status hiding an unresolved
|
|
482
|
+
* failure. It never flips a run that ended unresolved to `completed` (no
|
|
483
|
+
* regression vs the old `errorCount > 0 ? errored : completed`), and it does not
|
|
484
|
+
* flip a run whose failed work was never resolved just because some later,
|
|
485
|
+
* unrelated call happened to succeed.
|
|
486
|
+
*/
|
|
487
|
+
function deriveRunOutcome(traj) {
|
|
488
|
+
if (traj.errorCount === 0)
|
|
489
|
+
return 'completed';
|
|
490
|
+
return recoveredAfterErrors({ steps: traj.steps }) ? 'completed' : 'errored';
|
|
491
|
+
}
|
|
407
492
|
/**
|
|
408
493
|
* Map the derived trajectory to the console's SessionDetail shape, stripping
|
|
409
494
|
* local-machine PII (full cwd, account) that would expose filesystem paths if
|
|
@@ -423,7 +508,7 @@ export function buildSessionDetail(traj) {
|
|
|
423
508
|
errorCount: traj.errorCount,
|
|
424
509
|
tokens: stats.outputTokens ?? 0,
|
|
425
510
|
costUsd: s.costUsd ?? 0,
|
|
426
|
-
outcome: traj
|
|
511
|
+
outcome: deriveRunOutcome(traj),
|
|
427
512
|
repo,
|
|
428
513
|
agent: s.agent,
|
|
429
514
|
model: s.model ?? 'unknown',
|
|
@@ -8,9 +8,19 @@
|
|
|
8
8
|
// Routes (all require Phoenix bearer + userId owner match):
|
|
9
9
|
// PUT /<userId>/<device>/index.json — per-device stats shard
|
|
10
10
|
// PUT /<userId>/<device>/sessions/<id>.json — per-session SessionDetail
|
|
11
|
-
// GET /<userId>/...
|
|
11
|
+
// GET /<userId>/<device>/... — returns the stored object
|
|
12
|
+
// GET /<userId>/all/index.json — cross-device merge (read-side)
|
|
13
|
+
// GET /<userId>/all/sessions/<id>.json — first device that holds <id>
|
|
12
14
|
// GET / — 200 description (no data)
|
|
13
15
|
//
|
|
16
|
+
// The CLI writes per-device (`localDevice()` == hostname); the console asks for
|
|
17
|
+
// the "all agents" view (`/all/`). `all` is not a real device — the worker
|
|
18
|
+
// synthesizes it on read by listing this owner's device prefixes and merging.
|
|
19
|
+
// A single device is an exact passthrough; cross-device stats that need the raw
|
|
20
|
+
// sample (median, latency percentiles) are session-weighted approximations,
|
|
21
|
+
// documented at the merge site. This keeps the CLI unchanged and uploads nothing
|
|
22
|
+
// extra — the per-device shards already in R2 are the source (PHNX-3397).
|
|
23
|
+
//
|
|
14
24
|
// Emitted as a string so it compiles into dist/** and ships with no
|
|
15
25
|
// package.json#files change. `provisionTraces.ts` uploads this verbatim as an
|
|
16
26
|
// ES-module Worker with a BUCKET (R2) binding + a PHOENIX_ID_BASE secret.
|
|
@@ -79,6 +89,13 @@ async function handleGet(request, env, path) {
|
|
|
79
89
|
const auth = await authorizeRead(request, env, path);
|
|
80
90
|
if (auth.error) return auth.error;
|
|
81
91
|
|
|
92
|
+
const segments = path.split('/').filter(Boolean);
|
|
93
|
+
// The "all agents" view is synthesized, not stored: merge this owner's device
|
|
94
|
+
// shards on read. segments[0] === owner is already enforced by authorizeRead.
|
|
95
|
+
if (segments[1] === 'all') {
|
|
96
|
+
return handleAggregate(request, env, segments[0], segments.slice(2).join('/'));
|
|
97
|
+
}
|
|
98
|
+
|
|
82
99
|
const object = await env.BUCKET.get(path);
|
|
83
100
|
if (!object) return json({ error: 'not found' }, 404);
|
|
84
101
|
|
|
@@ -87,6 +104,142 @@ async function handleGet(request, env, path) {
|
|
|
87
104
|
return new Response(object.body, { headers });
|
|
88
105
|
}
|
|
89
106
|
|
|
107
|
+
// Enumerate this owner's real device names (never 'all') using a delimited list
|
|
108
|
+
// so we read one page of prefixes, not every session object.
|
|
109
|
+
async function listDevices(env, owner) {
|
|
110
|
+
const listed = await env.BUCKET.list({ prefix: owner + '/', delimiter: '/' });
|
|
111
|
+
const prefixes = (listed && listed.delimitedPrefixes) || [];
|
|
112
|
+
const base = owner + '/';
|
|
113
|
+
return prefixes
|
|
114
|
+
.map((p) => (p.startsWith(base) ? p.slice(base.length) : p).replace(/\\/+$/, ''))
|
|
115
|
+
.filter((d) => d && d !== 'all');
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
async function handleAggregate(request, env, owner, rest) {
|
|
119
|
+
const devices = await listDevices(env, owner);
|
|
120
|
+
if (devices.length === 0) return json({ error: 'not found' }, 404);
|
|
121
|
+
|
|
122
|
+
// A single session drill-down: the shard lives under whichever device owns it.
|
|
123
|
+
if (rest.startsWith('sessions/')) {
|
|
124
|
+
for (const device of devices) {
|
|
125
|
+
const object = await env.BUCKET.get(owner + '/' + device + '/' + rest);
|
|
126
|
+
if (object) {
|
|
127
|
+
const headers = new Headers(object.httpMetadata ?? {});
|
|
128
|
+
headers.set('cache-control', 'private, no-store');
|
|
129
|
+
return new Response(object.body, { headers });
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
return json({ error: 'not found' }, 404);
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
if (rest === 'index.json') {
|
|
136
|
+
const shards = [];
|
|
137
|
+
for (const device of devices) {
|
|
138
|
+
const object = await env.BUCKET.get(owner + '/' + device + '/index.json');
|
|
139
|
+
if (!object) continue;
|
|
140
|
+
try {
|
|
141
|
+
shards.push(JSON.parse(await object.text()));
|
|
142
|
+
} catch {
|
|
143
|
+
// A corrupt per-device shard must not sink the whole aggregate.
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
if (shards.length === 0) return json({ error: 'not found' }, 404);
|
|
147
|
+
return json(mergeIndexShards(shards, owner), 200);
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
return json({ error: 'not found' }, 404);
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
// Merge per-device index shards into one "all agents" shard. One device is an
|
|
154
|
+
// exact passthrough (relabelled). For many, counts sum and lists concat+resort;
|
|
155
|
+
// median/latency percentiles cannot be recomputed without the raw per-session
|
|
156
|
+
// sample, so they are session-count-weighted approximations — good enough for an
|
|
157
|
+
// at-a-glance fleet view, and exact the moment you filter to a single device.
|
|
158
|
+
function mergeIndexShards(shards, owner) {
|
|
159
|
+
if (shards.length === 1) {
|
|
160
|
+
return Object.assign({}, shards[0], { device: 'all', owner });
|
|
161
|
+
}
|
|
162
|
+
const sorted = shards.slice().sort((a, b) => (b.syncedAt || 0) - (a.syncedAt || 0));
|
|
163
|
+
const totalSessions = sorted.reduce((n, s) => n + ((s.stats && s.stats.sessionsImported) || 0), 0) || 1;
|
|
164
|
+
const wsum = (pick) => sorted.reduce((n, s) => n + (pick(s) || 0) * ((s.stats && s.stats.sessionsImported) || 0), 0) / totalSessions;
|
|
165
|
+
|
|
166
|
+
const topicCounts = new Map();
|
|
167
|
+
for (const s of sorted) for (const t of s.topics || []) {
|
|
168
|
+
const cur = topicCounts.get(t.key) || Object.assign({}, t, { count: 0 });
|
|
169
|
+
cur.count += t.count || 0;
|
|
170
|
+
topicCounts.set(t.key, cur);
|
|
171
|
+
}
|
|
172
|
+
const byToolError = new Map();
|
|
173
|
+
let real = 0, guard = 0, hook = 0;
|
|
174
|
+
for (const s of sorted) {
|
|
175
|
+
const f = s.failures || {};
|
|
176
|
+
for (const e of f.byToolError || []) {
|
|
177
|
+
const key = e.tool + '\\u0000' + e.desc + '\\u0000' + e.cause;
|
|
178
|
+
const cur = byToolError.get(key) || Object.assign({}, e, { count: 0 });
|
|
179
|
+
cur.count += e.count || 0;
|
|
180
|
+
byToolError.set(key, cur);
|
|
181
|
+
}
|
|
182
|
+
real += (f.byCause && f.byCause.real) || 0;
|
|
183
|
+
guard += (f.byCause && f.byCause.guard) || 0;
|
|
184
|
+
hook += (f.byCause && f.byCause.hook) || 0;
|
|
185
|
+
}
|
|
186
|
+
// Fold failure patterns by their stable id, the same way topics/byToolError fold
|
|
187
|
+
// by key above — a signature that fired on two devices (a rate limit on the laptop
|
|
188
|
+
// AND on CI) is ONE ranked issue with combined counts, not two half-counted rows.
|
|
189
|
+
const patternById = new Map();
|
|
190
|
+
for (const s of sorted) for (const p of s.failurePatterns || []) {
|
|
191
|
+
const cur = patternById.get(p.id);
|
|
192
|
+
if (!cur) {
|
|
193
|
+
patternById.set(p.id, Object.assign({}, p, { exampleSessionIds: (p.exampleSessionIds || []).slice(0, 5) }));
|
|
194
|
+
} else {
|
|
195
|
+
cur.sessions += p.sessions || 0;
|
|
196
|
+
cur.occurrences += p.occurrences || 0;
|
|
197
|
+
cur.wastedMs += p.wastedMs || 0;
|
|
198
|
+
for (const id of p.exampleSessionIds || []) {
|
|
199
|
+
if (cur.exampleSessionIds.length < 5 && !cur.exampleSessionIds.includes(id)) cur.exampleSessionIds.push(id);
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
const latencies = sorted.map((s) => s.latency).filter(Boolean);
|
|
204
|
+
|
|
205
|
+
return {
|
|
206
|
+
schema: 1,
|
|
207
|
+
device: 'all',
|
|
208
|
+
owner,
|
|
209
|
+
syncedAt: sorted[0].syncedAt || 0,
|
|
210
|
+
stats: {
|
|
211
|
+
sessionsImported: sorted.reduce((n, s) => n + ((s.stats && s.stats.sessionsImported) || 0), 0),
|
|
212
|
+
medianMs: Math.round(wsum((s) => s.stats && s.stats.medianMs)),
|
|
213
|
+
p90Ms: Math.max.apply(null, sorted.map((s) => (s.stats && s.stats.p90Ms) || 0)),
|
|
214
|
+
needAttention: sorted.reduce((n, s) => n + ((s.stats && s.stats.needAttention) || 0), 0),
|
|
215
|
+
toolErrorRate: wsum((s) => s.stats && s.stats.toolErrorRate),
|
|
216
|
+
},
|
|
217
|
+
needsAttention: sorted.flatMap((s) => s.needsAttention || [])
|
|
218
|
+
.sort((a, b) => (b.severity || 0) - (a.severity || 0) || String(a.id).localeCompare(String(b.id)))
|
|
219
|
+
.slice(0, 100),
|
|
220
|
+
topics: Array.from(topicCounts.values()).sort((a, b) => b.count - a.count),
|
|
221
|
+
failures: {
|
|
222
|
+
byToolError: Array.from(byToolError.values()).sort((a, b) => b.count - a.count).slice(0, 50),
|
|
223
|
+
byCause: { real, guard, hook },
|
|
224
|
+
},
|
|
225
|
+
failurePatterns: Array.from(patternById.values())
|
|
226
|
+
.sort((a, b) => (b.wastedMs || 0) - (a.wastedMs || 0)).slice(0, 25),
|
|
227
|
+
wastedMsTotal: sorted.reduce((n, s) => n + (s.wastedMsTotal || 0), 0),
|
|
228
|
+
latency: latencies.length ? {
|
|
229
|
+
firstToolMs: {
|
|
230
|
+
p50: Math.round(wsum((s) => s.latency && s.latency.firstToolMs && s.latency.firstToolMs.p50)),
|
|
231
|
+
p90: Math.max.apply(null, latencies.map((l) => (l.firstToolMs && l.firstToolMs.p90) || 0)),
|
|
232
|
+
p99: Math.max.apply(null, latencies.map((l) => (l.firstToolMs && l.firstToolMs.p99) || 0)),
|
|
233
|
+
max: Math.max.apply(null, latencies.map((l) => (l.firstToolMs && l.firstToolMs.max) || 0)),
|
|
234
|
+
},
|
|
235
|
+
} : undefined,
|
|
236
|
+
// Rolling history / drift are per-device time series; the freshest device's
|
|
237
|
+
// are the most representative for the merged view without double-counting.
|
|
238
|
+
bucketHistory: sorted[0].bucketHistory || [],
|
|
239
|
+
driftSignals: sorted[0].driftSignals || [],
|
|
240
|
+
};
|
|
241
|
+
}
|
|
242
|
+
|
|
90
243
|
// Require a Phoenix bearer that owns the path prefix.
|
|
91
244
|
async function authorizeRead(request, env, path) {
|
|
92
245
|
const claims = await hooks.verifyPhoenixToken(request, env);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@phnx-labs/agents-cli",
|
|
3
|
-
"version": "1.22.
|
|
3
|
+
"version": "1.22.58",
|
|
4
4
|
"description": "One CLI for all your AI coding agents - versions, config, cloud dispatch, sessions, and teams (now with first-class Grok Build CLI support)",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|