@phnx-labs/agents-cli 1.22.70 → 1.22.72
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -0
- package/README.md +31 -1
- package/dist/bootstrap.js +4 -4
- package/dist/commands/repo.js +2 -2
- package/dist/commands/sessions-export.d.ts +5 -1
- package/dist/commands/sessions-export.js +100 -24
- package/dist/commands/sessions-import.d.ts +2 -1
- package/dist/commands/sessions-import.js +85 -21
- package/dist/lib/accounting/usage-sync.d.ts +1 -1
- package/dist/lib/accounting/usage-sync.js +3 -3
- package/dist/lib/browser/ipc.d.ts +34 -0
- package/dist/lib/browser/ipc.js +140 -19
- package/dist/lib/browser/types.d.ts +3 -1
- package/dist/lib/daemon/auth-sync-service.js +1 -1
- package/dist/lib/daemon/browser-task-reap-service.js +1 -1
- package/dist/lib/daemon/daemon.js +13 -3
- package/dist/lib/daemon/heartbeat-service.js +3 -3
- package/dist/lib/daemon/keychain-reap-service.js +1 -1
- package/dist/lib/daemon/runner.d.ts +18 -1
- package/dist/lib/daemon/runner.js +231 -78
- package/dist/lib/daemon/self-heal-service.js +13 -3
- package/dist/lib/daemon/self-update-service.d.ts +174 -0
- package/dist/lib/daemon/self-update-service.js +353 -0
- package/dist/lib/daemon/state-dir-check-service.js +3 -3
- package/dist/lib/daemon/usage-sync-service.js +1 -1
- package/dist/lib/daemon/watchdog-service.js +4 -4
- package/dist/lib/daemon-services.d.ts +1 -1
- package/dist/lib/daemon-services.js +5 -0
- package/dist/lib/device-config.d.ts +12 -1
- package/dist/lib/device-config.js +63 -13
- package/dist/lib/exec-bounded.d.ts +52 -0
- package/dist/lib/exec-bounded.js +113 -0
- package/dist/lib/feed/events.d.ts +22 -14
- package/dist/lib/feed/events.js +84 -44
- package/dist/lib/fleet-shared-state.d.ts +12 -5
- package/dist/lib/fleet-shared-state.js +50 -20
- package/dist/lib/fs-atomic.d.ts +11 -0
- package/dist/lib/fs-atomic.js +60 -0
- package/dist/lib/hosts/reconcile.d.ts +11 -4
- package/dist/lib/hosts/reconcile.js +31 -5
- package/dist/lib/project-resources.d.ts +12 -0
- package/dist/lib/project-resources.js +138 -0
- package/dist/lib/routine-process-cleanup.d.ts +2 -2
- package/dist/lib/routine-process-cleanup.js +45 -34
- package/dist/lib/secrets/reaper.d.ts +2 -2
- package/dist/lib/secrets/reaper.js +13 -10
- package/dist/lib/secrets/reserved-sync.d.ts +1 -1
- package/dist/lib/secrets/reserved-sync.js +4 -4
- package/dist/lib/self-update.d.ts +21 -8
- package/dist/lib/self-update.js +54 -31
- package/dist/lib/session/sync/backend.d.ts +61 -0
- package/dist/lib/session/sync/backend.js +89 -0
- package/dist/lib/session/sync/managed-config.d.ts +29 -0
- package/dist/lib/session/sync/managed-config.js +23 -0
- package/dist/lib/session/sync/managed-key.d.ts +45 -0
- package/dist/lib/session/sync/managed-key.js +128 -0
- package/dist/lib/session/sync/net-client.d.ts +65 -0
- package/dist/lib/session/sync/net-client.js +117 -0
- package/dist/lib/session/sync/provision.d.ts +19 -0
- package/dist/lib/session/sync/provision.js +38 -0
- package/dist/lib/session/sync/r2.d.ts +5 -2
- package/dist/lib/session/sync/r2.js +5 -2
- package/dist/lib/session/sync/worker-template.d.ts +6 -0
- package/dist/lib/session/sync/worker-template.js +847 -0
- package/dist/lib/tmux/orphan-reap.js +6 -4
- package/dist/lib/tmux/session.js +4 -1
- package/dist/lib/traces/classify.d.ts +8 -1
- package/dist/lib/traces/insights.d.ts +13 -1
- package/dist/lib/traces/insights.js +78 -3
- package/dist/lib/traces/sync.js +8 -3
- package/dist/lib/traces/worker-template.js +9 -5
- package/package.json +1 -1
|
@@ -99,7 +99,7 @@
|
|
|
99
99
|
* helper); it is not a parsing bug {@link parseTmuxSessionMarker} can fix.
|
|
100
100
|
*/
|
|
101
101
|
import { execFile } from 'child_process';
|
|
102
|
-
import * as
|
|
102
|
+
import * as fsp from 'fs/promises';
|
|
103
103
|
import { promisify } from 'util';
|
|
104
104
|
const execFileAsync = promisify(execFile);
|
|
105
105
|
/** Ceiling on the `ps` snapshot so a wedged `ps` can never stall the daemon tick. */
|
|
@@ -407,7 +407,9 @@ export async function readAgentProcesses(opts = {}) {
|
|
|
407
407
|
if (process.platform === 'win32')
|
|
408
408
|
return [];
|
|
409
409
|
const scope = opts.pids && opts.pids.length > 0 ? ['-p', opts.pids.join(',')] : ['-A'];
|
|
410
|
-
|
|
410
|
+
// Async /proc reads — this runs on the daemon's tmux-reap tick, so a sync scan
|
|
411
|
+
// of the whole process table's environ files would freeze the loop (PHNX-3695).
|
|
412
|
+
const hasProc = await fsp.access('/proc/self/environ').then(() => true, () => false);
|
|
411
413
|
const args = hasProc ? [...scope, '-o', 'pid=,ppid=,args='] : [...scope, '-E', '-o', 'pid=,ppid=,args='];
|
|
412
414
|
let stdout;
|
|
413
415
|
try {
|
|
@@ -426,7 +428,7 @@ export async function readAgentProcesses(opts = {}) {
|
|
|
426
428
|
for (const row of rows) {
|
|
427
429
|
let blob;
|
|
428
430
|
try {
|
|
429
|
-
blob =
|
|
431
|
+
blob = await fsp.readFile(`/proc/${row.pid}/environ`, 'utf8');
|
|
430
432
|
}
|
|
431
433
|
catch {
|
|
432
434
|
continue; // exited between ps and the read, or not ours to inspect
|
|
@@ -450,7 +452,7 @@ export async function readPaneOwners(socket) {
|
|
|
450
452
|
// No socket file at all means no server was ever started here — for the
|
|
451
453
|
// shared agents socket this is reliable (tmux unlinks its own socket on
|
|
452
454
|
// exit), so this is a confident, reliable EMPTY read, not an unknown one.
|
|
453
|
-
if (!
|
|
455
|
+
if (!(await fsp.access(socket).then(() => true, () => false)))
|
|
454
456
|
return { ok: true, owners: new Map(), panePids: new Set() };
|
|
455
457
|
const { runTmux } = await import('./binary.js');
|
|
456
458
|
try {
|
package/dist/lib/tmux/session.js
CHANGED
|
@@ -10,6 +10,7 @@
|
|
|
10
10
|
* `tmux list-sessions` and prunes stale entries on the fly.
|
|
11
11
|
*/
|
|
12
12
|
import * as fs from 'fs';
|
|
13
|
+
import * as fsp from 'fs/promises';
|
|
13
14
|
import * as os from 'os';
|
|
14
15
|
import * as path from 'path';
|
|
15
16
|
import { runTmux, TmuxCommandError } from './binary.js';
|
|
@@ -412,7 +413,9 @@ export async function reapDeadTmuxPanes(socket, opts = {}) {
|
|
|
412
413
|
result.warnings = orphans.warnings;
|
|
413
414
|
result.processes = opts.dryRun ? orphans.candidates.length : orphans.killed;
|
|
414
415
|
result.processDetails = orphans.details;
|
|
415
|
-
|
|
416
|
+
// Async existence check — this runs on the daemon's tmux-reap tick, so a sync
|
|
417
|
+
// `existsSync` would block the shared event loop (PHNX-3695).
|
|
418
|
+
if (!(await fsp.access(sock).then(() => true, () => false)))
|
|
416
419
|
return result;
|
|
417
420
|
const res = await runTmux({
|
|
418
421
|
socket: sock,
|
|
@@ -1,5 +1,12 @@
|
|
|
1
1
|
export type TraceTopicGroup = 'code' | 'research' | 'review' | 'content' | 'ops';
|
|
2
|
-
|
|
2
|
+
/**
|
|
3
|
+
* `real`/`guard`/`hook` are the cause buckets of a FAILED tool call (`classifyCause`).
|
|
4
|
+
* `behavioral` is different in kind: a silent failure with no error code at all — the
|
|
5
|
+
* agent went idle after its last event and a human had to nudge it. It is derived from
|
|
6
|
+
* per-session friction facets (`computeBehavioralPatterns` in `insights.ts`), never from
|
|
7
|
+
* `classifyCause`, so a failed-tool-call classifier never returns it.
|
|
8
|
+
*/
|
|
9
|
+
export type TraceFailureCause = 'real' | 'guard' | 'hook' | 'behavioral';
|
|
3
10
|
/** Per-bucket aggregate stats for one day, stored in the rolling bucketHistory. */
|
|
4
11
|
export interface BucketStats {
|
|
5
12
|
key: string;
|
|
@@ -29,6 +29,7 @@ import { type TraceFailureCause } from './classify.js';
|
|
|
29
29
|
import type { FailurePhenotype } from './phenotype.js';
|
|
30
30
|
import { type LatencyInsight } from './segments.js';
|
|
31
31
|
import { type SyncRow, type ToolCallRow, type TracesIndexShard } from './sync.js';
|
|
32
|
+
import type { InsightFacets } from '../session/insights.js';
|
|
32
33
|
export interface FailureSignature {
|
|
33
34
|
tool: string;
|
|
34
35
|
cause: TraceFailureCause;
|
|
@@ -97,4 +98,15 @@ export declare function normalizeErrorKey(desc: string, raw: string | null): str
|
|
|
97
98
|
* estimate, not ground truth; it is not inflated by folding in ordinary
|
|
98
99
|
* processing time between unrelated calls.
|
|
99
100
|
*/
|
|
100
|
-
export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null, phenotypes?: ReadonlyMap<string, FailurePhenotype | null
|
|
101
|
+
export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null, phenotypes?: ReadonlyMap<string, FailurePhenotype | null>, behavioralPatterns?: readonly FailurePattern[]): ComputedInsights;
|
|
102
|
+
/**
|
|
103
|
+
* Promote the per-session silent-stall friction signals — already computed by
|
|
104
|
+
* `computeInsightFacets` in `session/insights.ts` and keyed `silent stall: <bucket>`
|
|
105
|
+
* — into cross-session behavioral `FailurePattern`s. These are silent failures
|
|
106
|
+
* with NO error code: the agent went idle after its last event and a human had to
|
|
107
|
+
* nudge it. They never surface through `computeInsights` (which only clusters
|
|
108
|
+
* `outcome === 'error'` tool calls) and were previously only a `needsAttention`
|
|
109
|
+
* friction counter, never a ranked issue. Cause is `behavioral`; the signature
|
|
110
|
+
* `tool` is the synthetic `silent-stall` and `key` is the duration bucket.
|
|
111
|
+
*/
|
|
112
|
+
export declare function computeBehavioralPatterns(facetsBySession: ReadonlyMap<string, Pick<InsightFacets, 'frictionSignals'>>, prevShard?: TracesIndexShard | null): FailurePattern[];
|
|
@@ -132,7 +132,7 @@ function labelFor(tool, cause, key) {
|
|
|
132
132
|
* estimate, not ground truth; it is not inflated by folding in ordinary
|
|
133
133
|
* processing time between unrelated calls.
|
|
134
134
|
*/
|
|
135
|
-
export function computeInsights(rows, calls, prevShard, phenotypes) {
|
|
135
|
+
export function computeInsights(rows, calls, prevShard, phenotypes, behavioralPatterns = []) {
|
|
136
136
|
const bySession = new Map();
|
|
137
137
|
for (const call of calls) {
|
|
138
138
|
const list = bySession.get(call.session_id);
|
|
@@ -226,13 +226,88 @@ export function computeInsights(rows, calls, prevShard, phenotypes) {
|
|
|
226
226
|
drift,
|
|
227
227
|
};
|
|
228
228
|
});
|
|
229
|
-
|
|
230
|
-
|
|
229
|
+
// Behavioral patterns (silent stalls — no failed tool call, so absent from the
|
|
230
|
+
// error-anchored clustering above) join the SAME ranking and total, so the top-K
|
|
231
|
+
// is by impact across both kinds and `wastedMsTotal` counts the idle time too.
|
|
232
|
+
const combined = [...allPatterns, ...behavioralPatterns];
|
|
233
|
+
const wastedMsTotal = combined.reduce((sum, p) => sum + p.wastedMs, 0);
|
|
234
|
+
const failurePatterns = [...combined]
|
|
231
235
|
.sort((a, b) => b.wastedMs - a.wastedMs || b.occurrences - a.occurrences || a.id.localeCompare(b.id))
|
|
232
236
|
.slice(0, TOP_K_PATTERNS);
|
|
233
237
|
const latency = computeLatency(firstToolSegments(rows, bySession));
|
|
234
238
|
return { failurePatterns, wastedMsTotal, latency };
|
|
235
239
|
}
|
|
240
|
+
// ---------------------------------------------------------------------------
|
|
241
|
+
// Behavioral patterns — silent failures with no error code
|
|
242
|
+
// ---------------------------------------------------------------------------
|
|
243
|
+
/**
|
|
244
|
+
* Estimated agent-owned idle ms per silent-stall bucket. A bucket's midpoint,
|
|
245
|
+
* capped at MAX_GAP_ATTRIBUTION_MS exactly like the tool-error gap attribution
|
|
246
|
+
* above so one long overnight stall can't book hours of "waste" — the same
|
|
247
|
+
* honesty bound the error path uses. The open-ended buckets are the cap.
|
|
248
|
+
*/
|
|
249
|
+
const SILENT_STALL_WASTED_MS = {
|
|
250
|
+
'5-15m': 10 * 60_000,
|
|
251
|
+
'15-60m': MAX_GAP_ATTRIBUTION_MS,
|
|
252
|
+
'1h+': MAX_GAP_ATTRIBUTION_MS,
|
|
253
|
+
};
|
|
254
|
+
/**
|
|
255
|
+
* Promote the per-session silent-stall friction signals — already computed by
|
|
256
|
+
* `computeInsightFacets` in `session/insights.ts` and keyed `silent stall: <bucket>`
|
|
257
|
+
* — into cross-session behavioral `FailurePattern`s. These are silent failures
|
|
258
|
+
* with NO error code: the agent went idle after its last event and a human had to
|
|
259
|
+
* nudge it. They never surface through `computeInsights` (which only clusters
|
|
260
|
+
* `outcome === 'error'` tool calls) and were previously only a `needsAttention`
|
|
261
|
+
* friction counter, never a ranked issue. Cause is `behavioral`; the signature
|
|
262
|
+
* `tool` is the synthetic `silent-stall` and `key` is the duration bucket.
|
|
263
|
+
*/
|
|
264
|
+
export function computeBehavioralPatterns(facetsBySession, prevShard) {
|
|
265
|
+
const groups = new Map();
|
|
266
|
+
for (const [sessionId, facets] of facetsBySession) {
|
|
267
|
+
for (const [signal, count] of Object.entries(facets.frictionSignals ?? {})) {
|
|
268
|
+
if (count <= 0)
|
|
269
|
+
continue;
|
|
270
|
+
const match = /^silent stall: (.+)$/.exec(signal);
|
|
271
|
+
if (!match)
|
|
272
|
+
continue;
|
|
273
|
+
const bucket = match[1];
|
|
274
|
+
let group = groups.get(bucket);
|
|
275
|
+
if (!group) {
|
|
276
|
+
group = { bucket, sessions: new Set(), occurrences: 0, wastedMs: 0, examples: [] };
|
|
277
|
+
groups.set(bucket, group);
|
|
278
|
+
}
|
|
279
|
+
group.occurrences += count;
|
|
280
|
+
group.sessions.add(sessionId);
|
|
281
|
+
group.wastedMs += (SILENT_STALL_WASTED_MS[bucket] ?? MAX_GAP_ATTRIBUTION_MS) * count;
|
|
282
|
+
if (group.examples.length < MAX_EXAMPLE_SESSIONS && !group.examples.includes(sessionId)) {
|
|
283
|
+
group.examples.push(sessionId);
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
const prevById = new Map((prevShard?.failurePatterns ?? []).map((p) => [p.id, p]));
|
|
288
|
+
return [...groups.values()].map((group) => {
|
|
289
|
+
const id = hashSignature('silent-stall', 'behavioral', group.bucket, null);
|
|
290
|
+
const prev = prevById.get(id);
|
|
291
|
+
const drift = !prev
|
|
292
|
+
? 'up'
|
|
293
|
+
: group.occurrences > prev.occurrences
|
|
294
|
+
? 'up'
|
|
295
|
+
: group.occurrences < prev.occurrences
|
|
296
|
+
? 'down'
|
|
297
|
+
: 'flat';
|
|
298
|
+
return {
|
|
299
|
+
id,
|
|
300
|
+
label: `Agent silent stall (${group.bucket}) — idle until nudged`,
|
|
301
|
+
signature: { tool: 'silent-stall', cause: 'behavioral', key: group.bucket },
|
|
302
|
+
phenotype: null,
|
|
303
|
+
sessions: group.sessions.size,
|
|
304
|
+
occurrences: group.occurrences,
|
|
305
|
+
wastedMs: group.wastedMs,
|
|
306
|
+
exampleSessionIds: group.examples,
|
|
307
|
+
drift,
|
|
308
|
+
};
|
|
309
|
+
});
|
|
310
|
+
}
|
|
236
311
|
/** Synthesize one-step SegmentSessions carrying only the time-to-first-tool offset, for computeLatency() reuse. */
|
|
237
312
|
function firstToolSegments(rows, bySession) {
|
|
238
313
|
return rows.flatMap((row) => {
|
package/dist/lib/traces/sync.js
CHANGED
|
@@ -30,7 +30,7 @@ import { knownSecretValuesFromEnv, redactSecrets } from '../redact.js';
|
|
|
30
30
|
import { getRuntimeStateDir } from '../state.js';
|
|
31
31
|
import { resolveTracesBackend } from './backend.js';
|
|
32
32
|
import { classifyCause, classifyTopic, computeDriftSignal, } from './classify.js';
|
|
33
|
-
import { computeInsights } from './insights.js';
|
|
33
|
+
import { computeBehavioralPatterns, computeInsights } from './insights.js';
|
|
34
34
|
import { classifyPhenotype, recoveredAfterErrors } from './phenotype.js';
|
|
35
35
|
import { buildSessionDetailV2 } from './schema2-build.js';
|
|
36
36
|
/**
|
|
@@ -529,7 +529,9 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
529
529
|
topicCounts.set(topic.key, bucket);
|
|
530
530
|
}
|
|
531
531
|
const failedCalls = agentCalls.filter((call) => call.outcome === 'error');
|
|
532
|
-
|
|
532
|
+
// `behavioral` is not a failed-tool-call cause (classifyCause never returns it),
|
|
533
|
+
// so it stays 0 in this tool-error split; it surfaces as its own failurePatterns.
|
|
534
|
+
const byCause = { real: 0, guard: 0, hook: 0, behavioral: 0 };
|
|
533
535
|
const failureCounts = new Map();
|
|
534
536
|
for (const call of failedCalls) {
|
|
535
537
|
const cause = classifyCause(call);
|
|
@@ -585,7 +587,10 @@ export function buildIndexShard(rows, device, owner, prevShard) {
|
|
|
585
587
|
const prevHistory = prevShard?.bucketHistory ?? [];
|
|
586
588
|
const bucketHistory = [...prevHistory, todayStats].slice(-14);
|
|
587
589
|
const driftSignals = computeDriftSignal(prevHistory, todayStats);
|
|
588
|
-
|
|
590
|
+
// Silent-stall friction (per-session, no failed tool call) becomes cross-session
|
|
591
|
+
// behavioral FailurePatterns, ranked into the same top-K by wasted idle time.
|
|
592
|
+
const behavioralPatterns = computeBehavioralPatterns(insights, prevShard);
|
|
593
|
+
const patternInsights = computeInsights(agentRows, agentCalls, prevShard, phenotypes, behavioralPatterns);
|
|
589
594
|
// Per-session roster (PHNX-3483): one flat scalar row per agent session, the raw
|
|
590
595
|
// material the Rush console filters and re-aggregates client-side. `durationMs`
|
|
591
596
|
// reuses `sessionActiveMs` (the value behind `stats.medianMs`; 0 for a null-duration
|
|
@@ -170,7 +170,11 @@ function mergeIndexShards(shards, owner) {
|
|
|
170
170
|
topicCounts.set(t.key, cur);
|
|
171
171
|
}
|
|
172
172
|
const byToolError = new Map();
|
|
173
|
-
|
|
173
|
+
// Accumulate byCause over WHATEVER cause keys each shard carries (real / guard /
|
|
174
|
+
// hook / behavioral / any future TraceFailureCause) rather than a hand-enumerated
|
|
175
|
+
// set — a hardcoded {real,guard,hook} silently dropped a new member from the /all
|
|
176
|
+
// view and made the cause sum NaN (RUSH-2988).
|
|
177
|
+
const byCause = {};
|
|
174
178
|
for (const s of sorted) {
|
|
175
179
|
const f = s.failures || {};
|
|
176
180
|
for (const e of f.byToolError || []) {
|
|
@@ -179,9 +183,9 @@ function mergeIndexShards(shards, owner) {
|
|
|
179
183
|
cur.count += e.count || 0;
|
|
180
184
|
byToolError.set(key, cur);
|
|
181
185
|
}
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
186
|
+
for (const [cause, n] of Object.entries((f.byCause) || {})) {
|
|
187
|
+
byCause[cause] = (byCause[cause] || 0) + (n || 0);
|
|
188
|
+
}
|
|
185
189
|
}
|
|
186
190
|
// Fold failure patterns by their stable id, the same way topics/byToolError fold
|
|
187
191
|
// by key above — a signature that fired on two devices (a rate limit on the laptop
|
|
@@ -220,7 +224,7 @@ function mergeIndexShards(shards, owner) {
|
|
|
220
224
|
topics: Array.from(topicCounts.values()).sort((a, b) => b.count - a.count),
|
|
221
225
|
failures: {
|
|
222
226
|
byToolError: Array.from(byToolError.values()).sort((a, b) => b.count - a.count).slice(0, 50),
|
|
223
|
-
byCause
|
|
227
|
+
byCause,
|
|
224
228
|
},
|
|
225
229
|
failurePatterns: Array.from(patternById.values())
|
|
226
230
|
.sort((a, b) => (b.wastedMs || 0) - (a.wastedMs || 0)).slice(0, 25),
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@phnx-labs/agents-cli",
|
|
3
|
-
"version": "1.22.
|
|
3
|
+
"version": "1.22.72",
|
|
4
4
|
"description": "One CLI for all your AI coding agents - versions, config, cloud dispatch, sessions, and teams (now with first-class Grok Build CLI support)",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|