@phnx-labs/agents-cli 1.22.57 → 1.22.58
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +56 -0
- package/dist/bootstrap.js +8 -1
- package/dist/commands/accounts.js +7 -3
- package/dist/commands/apply.js +10 -2
- package/dist/commands/fork.d.ts +23 -10
- package/dist/commands/fork.js +115 -58
- package/dist/commands/monitors.js +11 -0
- package/dist/commands/prune.js +5 -3
- package/dist/commands/routines.d.ts +8 -0
- package/dist/commands/routines.js +57 -3
- package/dist/commands/sessions-picker.d.ts +11 -0
- package/dist/commands/sessions-picker.js +16 -0
- package/dist/commands/sessions.js +1 -0
- package/dist/commands/share.d.ts +14 -0
- package/dist/commands/share.js +43 -2
- package/dist/commands/status.js +1 -1
- package/dist/commands/sync.js +83 -7
- package/dist/commands/traces.js +7 -0
- package/dist/index.d.ts +1 -1
- package/dist/index.js +6 -1
- package/dist/lib/account-registry.d.ts +5 -1
- package/dist/lib/account-registry.js +47 -14
- package/dist/lib/accounting/capacity.d.ts +18 -7
- package/dist/lib/accounting/capacity.js +19 -8
- package/dist/lib/accounting/usage-sync.d.ts +29 -1
- package/dist/lib/accounting/usage-sync.js +76 -2
- package/dist/lib/accounting/usage.js +7 -1
- package/dist/lib/auth-mint.d.ts +11 -1
- package/dist/lib/auth-mint.js +21 -6
- package/dist/lib/browser/ipc.d.ts +8 -0
- package/dist/lib/browser/ipc.js +87 -0
- package/dist/lib/browser/service.d.ts +19 -0
- package/dist/lib/browser/service.js +96 -11
- package/dist/lib/browser/sessions-list.js +10 -1
- package/dist/lib/daemon/runner.d.ts +3 -0
- package/dist/lib/daemon/runner.js +86 -45
- package/dist/lib/daemon/usage-sync-service.d.ts +3 -3
- package/dist/lib/daemon/usage-sync-service.js +14 -8
- package/dist/lib/daemon-services.js +1 -1
- package/dist/lib/devices/connect.d.ts +17 -8
- package/dist/lib/devices/connect.js +31 -14
- package/dist/lib/doctor-diff.js +77 -7
- package/dist/lib/fleet/manifest.d.ts +17 -0
- package/dist/lib/fleet/manifest.js +26 -0
- package/dist/lib/hooks/install.d.ts +27 -11
- package/dist/lib/hooks/install.js +42 -17
- package/dist/lib/hosts/reconnect.d.ts +52 -203
- package/dist/lib/hosts/reconnect.js +64 -284
- package/dist/lib/installations/migrate.d.ts +6 -120
- package/dist/lib/installations/migrate.js +27 -259
- package/dist/lib/installations/shims.d.ts +13 -95
- package/dist/lib/installations/shims.js +22 -139
- package/dist/lib/installations/store.js +1 -1
- package/dist/lib/installations/versions.d.ts +26 -133
- package/dist/lib/installations/versions.js +41 -204
- package/dist/lib/plugins/skills.d.ts +8 -1
- package/dist/lib/plugins/skills.js +18 -2
- package/dist/lib/refresh.d.ts +9 -0
- package/dist/lib/refresh.js +3 -1
- package/dist/lib/routine-readiness.d.ts +15 -1
- package/dist/lib/routine-readiness.js +41 -0
- package/dist/lib/sandbox.d.ts +4 -1
- package/dist/lib/sandbox.js +30 -1
- package/dist/lib/secrets/agent.d.ts +80 -225
- package/dist/lib/secrets/agent.js +139 -401
- package/dist/lib/secrets/bundles.d.ts +73 -222
- package/dist/lib/secrets/bundles.js +168 -467
- package/dist/lib/secrets/reaper.d.ts +28 -70
- package/dist/lib/secrets/reaper.js +30 -85
- package/dist/lib/secrets/remote.d.ts +42 -129
- package/dist/lib/secrets/remote.js +55 -173
- package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
- package/dist/lib/self-heal/checks/install-staging.js +96 -0
- package/dist/lib/self-heal/registry.js +2 -0
- package/dist/lib/self-heal/types.d.ts +1 -1
- package/dist/lib/self-update.d.ts +23 -0
- package/dist/lib/self-update.js +50 -0
- package/dist/lib/session/active.d.ts +13 -1
- package/dist/lib/session/active.js +2 -0
- package/dist/lib/session/db.d.ts +20 -1
- package/dist/lib/session/db.js +139 -9
- package/dist/lib/session/fork.d.ts +45 -26
- package/dist/lib/session/fork.js +32 -95
- package/dist/lib/session/tool-calls.d.ts +43 -1
- package/dist/lib/session/tool-calls.js +74 -44
- package/dist/lib/session/tool-store.d.ts +33 -2
- package/dist/lib/session/tool-store.js +56 -3
- package/dist/lib/staleness/writers/sources.d.ts +5 -0
- package/dist/lib/staleness/writers/sources.js +2 -1
- package/dist/lib/sync-status.d.ts +22 -0
- package/dist/lib/sync-status.js +27 -0
- package/dist/lib/sync-umbrella.d.ts +9 -0
- package/dist/lib/sync-umbrella.js +21 -2
- package/dist/lib/traces/insights.d.ts +47 -14
- package/dist/lib/traces/insights.js +92 -21
- package/dist/lib/traces/phenotype.d.ts +23 -3
- package/dist/lib/traces/phenotype.js +72 -24
- package/dist/lib/traces/sync.d.ts +15 -0
- package/dist/lib/traces/sync.js +104 -19
- package/dist/lib/traces/worker-template.js +154 -1
- package/package.json +1 -1
|
@@ -12,7 +12,10 @@ export const TOOL_CHANGED_MAX_CALLS = 10_000;
|
|
|
12
12
|
export const TOOL_INDEX_LIMIT_ORDINAL = Number.MAX_SAFE_INTEGER;
|
|
13
13
|
export const TOOL_TEXT_PROCESSING_MAX_BYTES = 64 * 1024;
|
|
14
14
|
export const TOOL_SHELL_PARSE_MAX_BYTES = 64 * 1024;
|
|
15
|
-
|
|
15
|
+
// Bumped to 8 for the per-call end timestamp (PHNX-3437): the extractor now
|
|
16
|
+
// records when a call's result arrived, so a re-index re-derives it for rows
|
|
17
|
+
// stored by an older extractor.
|
|
18
|
+
export const TOOL_INDEX_VERSION = 8;
|
|
16
19
|
const BASE64_BLOCK = /(?:[A-Za-z0-9+/]{256,}={0,2})/g;
|
|
17
20
|
const SECRET_FIELD = /(?:token|secret|password|authorization|cookie|api[_-]?key|private[_-]?key)$/i;
|
|
18
21
|
const KNOWN_SECRET_VALUES = knownSecretValuesFromEnv();
|
|
@@ -310,7 +313,7 @@ function buildCall(ordinal, timestamp, tool, args, command, sourceCallId) {
|
|
|
310
313
|
}
|
|
311
314
|
export function toolCallEvidenceBytes(call) {
|
|
312
315
|
return Buffer.byteLength([
|
|
313
|
-
call.sourceCallId, call.timestamp, call.tool, call.input, call.errorCode,
|
|
316
|
+
call.sourceCallId, call.timestamp, call.endTimestamp, call.tool, call.input, call.errorCode,
|
|
314
317
|
call.output, call.error, call.parseError, ...call.programs,
|
|
315
318
|
...call.programOccurrences.map((occurrence) => `${occurrence.role}:${occurrence.program}`),
|
|
316
319
|
].filter((value) => typeof value === 'string').join('\0'));
|
|
@@ -363,6 +366,9 @@ export class ToolCallCollector {
|
|
|
363
366
|
const call = this.takePending(args.sourceCallId, tool);
|
|
364
367
|
if (!call)
|
|
365
368
|
return undefined;
|
|
369
|
+
if (typeof args.timestamp === 'string' && args.timestamp.length > 0) {
|
|
370
|
+
call.endTimestamp = sanitizeToolEvidenceText(args.timestamp, 128);
|
|
371
|
+
}
|
|
366
372
|
const outcome = args.outcome === 'ok' || args.outcome === 'error' || args.outcome === 'unknown'
|
|
367
373
|
? args.outcome
|
|
368
374
|
: undefined;
|
|
@@ -471,50 +477,74 @@ export class ToolCallCollector {
|
|
|
471
477
|
return true;
|
|
472
478
|
}
|
|
473
479
|
}
|
|
480
|
+
/** Fold one normalized event into a collector (the full-file harness contract). */
|
|
481
|
+
function foldEventIntoCollector(collector, event) {
|
|
482
|
+
if (event.type === 'tool_use') {
|
|
483
|
+
collector.start({
|
|
484
|
+
timestamp: event.timestamp,
|
|
485
|
+
tool: event.tool || 'unknown',
|
|
486
|
+
input: event.args,
|
|
487
|
+
command: event.command,
|
|
488
|
+
sourceCallId: event.callId,
|
|
489
|
+
});
|
|
490
|
+
}
|
|
491
|
+
else if (event.type === 'tool_result') {
|
|
492
|
+
collector.finish({
|
|
493
|
+
timestamp: event.timestamp,
|
|
494
|
+
sourceCallId: event.callId,
|
|
495
|
+
tool: event.tool,
|
|
496
|
+
success: event.success,
|
|
497
|
+
outcome: event.outcome,
|
|
498
|
+
exitCode: event.exitCode,
|
|
499
|
+
statusCode: event.statusCode,
|
|
500
|
+
errorCode: event.errorCode,
|
|
501
|
+
output: event.output,
|
|
502
|
+
});
|
|
503
|
+
}
|
|
504
|
+
else if (event.type === 'error' && event.tool) {
|
|
505
|
+
const structuredError = event.outcome === 'error' || event.success === false;
|
|
506
|
+
const evidence = event.content || event.output || 'Tool execution failed';
|
|
507
|
+
collector.finish({
|
|
508
|
+
timestamp: event.timestamp,
|
|
509
|
+
sourceCallId: event.callId,
|
|
510
|
+
tool: event.tool,
|
|
511
|
+
success: structuredError ? false : undefined,
|
|
512
|
+
outcome: event.outcome,
|
|
513
|
+
exitCode: event.exitCode,
|
|
514
|
+
statusCode: event.statusCode,
|
|
515
|
+
errorCode: event.errorCode,
|
|
516
|
+
error: structuredError ? evidence : undefined,
|
|
517
|
+
output: structuredError ? undefined : evidence,
|
|
518
|
+
});
|
|
519
|
+
}
|
|
520
|
+
}
|
|
521
|
+
/**
|
|
522
|
+
* Derive tool calls from a full-file harness's normalized events, optionally
|
|
523
|
+
* RESUMING from a prior scan of the same append-only stream.
|
|
524
|
+
*
|
|
525
|
+
* Without `prior` this is a full parse from event 0 (identical to the legacy
|
|
526
|
+
* {@link toolCallsFromEvents}). With `prior`, the collector is seeded from the
|
|
527
|
+
* prior snapshot and only events at or after `prior.eventCount` are folded — so
|
|
528
|
+
* an active session that grew by a few turns re-derives (and re-redacts) only
|
|
529
|
+
* those new tool calls instead of re-sanitizing its entire history on every
|
|
530
|
+
* daemon warm tick (PHNX-3411). The ordinals continue deterministically from the
|
|
531
|
+
* snapshot, so folding [0..k) then [k..n) yields the same index as folding
|
|
532
|
+
* [0..n) once, and the CHANGED set is safe to persist with `mode: 'append'`.
|
|
533
|
+
*
|
|
534
|
+
* The caller is responsible for only supplying `prior` when the stream is still
|
|
535
|
+
* an append of what was scanned before (same source, un-truncated,
|
|
536
|
+
* `prior.eventCount <= events.length`); anything else must full-scan.
|
|
537
|
+
*/
|
|
538
|
+
export function scanEventToolCalls(events, prior) {
|
|
539
|
+
const startIndex = prior ? Math.min(prior.eventCount, events.length) : 0;
|
|
540
|
+
const collector = new ToolCallCollector(prior?.snapshot);
|
|
541
|
+
for (let i = startIndex; i < events.length; i++)
|
|
542
|
+
foldEventIntoCollector(collector, events[i]);
|
|
543
|
+
return { calls: collector.drainChanged(), snapshot: collector.snapshot(), eventCount: events.length };
|
|
544
|
+
}
|
|
474
545
|
/** Build indexed calls from the normalized parser contract used by full-file harnesses. */
|
|
475
546
|
export function toolCallsFromEvents(events) {
|
|
476
|
-
|
|
477
|
-
for (const event of events) {
|
|
478
|
-
if (event.type === 'tool_use') {
|
|
479
|
-
collector.start({
|
|
480
|
-
timestamp: event.timestamp,
|
|
481
|
-
tool: event.tool || 'unknown',
|
|
482
|
-
input: event.args,
|
|
483
|
-
command: event.command,
|
|
484
|
-
sourceCallId: event.callId,
|
|
485
|
-
});
|
|
486
|
-
}
|
|
487
|
-
else if (event.type === 'tool_result') {
|
|
488
|
-
collector.finish({
|
|
489
|
-
timestamp: event.timestamp,
|
|
490
|
-
sourceCallId: event.callId,
|
|
491
|
-
tool: event.tool,
|
|
492
|
-
success: event.success,
|
|
493
|
-
outcome: event.outcome,
|
|
494
|
-
exitCode: event.exitCode,
|
|
495
|
-
statusCode: event.statusCode,
|
|
496
|
-
errorCode: event.errorCode,
|
|
497
|
-
output: event.output,
|
|
498
|
-
});
|
|
499
|
-
}
|
|
500
|
-
else if (event.type === 'error' && event.tool) {
|
|
501
|
-
const structuredError = event.outcome === 'error' || event.success === false;
|
|
502
|
-
const evidence = event.content || event.output || 'Tool execution failed';
|
|
503
|
-
collector.finish({
|
|
504
|
-
timestamp: event.timestamp,
|
|
505
|
-
sourceCallId: event.callId,
|
|
506
|
-
tool: event.tool,
|
|
507
|
-
success: structuredError ? false : undefined,
|
|
508
|
-
outcome: event.outcome,
|
|
509
|
-
exitCode: event.exitCode,
|
|
510
|
-
statusCode: event.statusCode,
|
|
511
|
-
errorCode: event.errorCode,
|
|
512
|
-
error: structuredError ? evidence : undefined,
|
|
513
|
-
output: structuredError ? undefined : evidence,
|
|
514
|
-
});
|
|
515
|
-
}
|
|
516
|
-
}
|
|
517
|
-
return collector.drainChanged();
|
|
547
|
+
return scanEventToolCalls(events).calls;
|
|
518
548
|
}
|
|
519
549
|
export function toolCallKey(sessionId, ordinal) {
|
|
520
550
|
return createHash('sha256').update(`${sessionId}\0${ordinal}`).digest('hex').slice(0, 20);
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type Database from '../sqlite.js';
|
|
2
|
-
import { type IndexedToolCall } from './tool-calls.js';
|
|
2
|
+
import { type IndexedToolCall, type EventToolScanResumePoint } from './tool-calls.js';
|
|
3
3
|
import type { SessionMeta } from './types.js';
|
|
4
4
|
export declare function canonicalToolLedgerPath(filePath: string): string;
|
|
5
5
|
/** Resolve the transcript that actually carries tool events for split-file harnesses. */
|
|
@@ -12,7 +12,14 @@ export declare function purgeMissingToolCallsInDirectory(db: Database.Database,
|
|
|
12
12
|
export interface ToolScanResumePoint {
|
|
13
13
|
/** Serialized ToolCallCollector snapshot at `parsedOffset`. */
|
|
14
14
|
parserState: string;
|
|
15
|
-
/**
|
|
15
|
+
/**
|
|
16
|
+
* Where a later scan may resume. For the streaming (claude/codex) path this is
|
|
17
|
+
* a BYTE offset just past the last complete record consumed. For the full-file
|
|
18
|
+
* harness path (`planEventToolResume`) it is instead the COUNT of normalized
|
|
19
|
+
* events already folded — the two never share a session (a session is one
|
|
20
|
+
* agent, and the streaming path is claude/codex only), so the column carries
|
|
21
|
+
* whichever meaning that agent's resumer wrote.
|
|
22
|
+
*/
|
|
16
23
|
parsedOffset: number;
|
|
17
24
|
}
|
|
18
25
|
export interface PersistToolCallsOptions {
|
|
@@ -37,3 +44,27 @@ export declare function persistToolCalls(db: Database.Database, session: Session
|
|
|
37
44
|
fileMtimeMs: number;
|
|
38
45
|
fileSize: number;
|
|
39
46
|
}, options?: PersistToolCallsOptions): void;
|
|
47
|
+
/**
|
|
48
|
+
* Decide whether a changed full-file-harness session can RESUME its tool index
|
|
49
|
+
* from the last scan instead of re-deriving every call (PHNX-3411).
|
|
50
|
+
*
|
|
51
|
+
* The warm-tick indexer re-derives tool calls for a changed session on every
|
|
52
|
+
* tick, and for an ACTIVE large session that means re-sanitizing tens of
|
|
53
|
+
* thousands of calls each time — the synchronous work that wedged the daemon
|
|
54
|
+
* event loop. Streaming harnesses (claude/codex) already resume from a byte
|
|
55
|
+
* offset; every other harness re-parses the whole file, so this brings them the
|
|
56
|
+
* same benefit at the EVENT level: the ledger stores how many events were folded
|
|
57
|
+
* (`parsed_offset`) plus the collector snapshot (`parser_state`), and a later
|
|
58
|
+
* scan folds only the newly appended events.
|
|
59
|
+
*
|
|
60
|
+
* Returns the prior resume point when it is safe to append, or `null` (full
|
|
61
|
+
* re-scan) whenever the stored prefix may no longer describe the current file:
|
|
62
|
+
* a different extractor, no recorded resume point, a source the ledger row does
|
|
63
|
+
* not describe, a transcript that shrank below what was already parsed (a
|
|
64
|
+
* truncation/rewrite, not an append), more events already folded than the file
|
|
65
|
+
* now yields, or a snapshot that does not read back.
|
|
66
|
+
*/
|
|
67
|
+
export declare function planEventToolResume(db: Database.Database, sessionId: string, sourcePath: string, stamp: {
|
|
68
|
+
fileMtimeMs: number;
|
|
69
|
+
fileSize: number;
|
|
70
|
+
}, currentEventCount: number): EventToolScanResumePoint | null;
|
|
@@ -75,13 +75,14 @@ export function persistToolCalls(db, session, calls, sourceStamp, options = {})
|
|
|
75
75
|
const sourcePath = toolEvidenceSourcePath(session.filePath, session.agent);
|
|
76
76
|
const insertCall = db.prepare(`
|
|
77
77
|
INSERT INTO tool_calls (
|
|
78
|
-
call_key, session_id, ordinal, source_call_id, timestamp, tool, input,
|
|
78
|
+
call_key, session_id, ordinal, source_call_id, timestamp, end_timestamp, tool, input,
|
|
79
79
|
outcome, exit_code, status_code, error_code, output, error, parse_error
|
|
80
80
|
, evidence_bytes
|
|
81
|
-
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
81
|
+
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
82
82
|
ON CONFLICT(session_id, ordinal) DO UPDATE SET
|
|
83
83
|
source_call_id = excluded.source_call_id,
|
|
84
84
|
timestamp = excluded.timestamp,
|
|
85
|
+
end_timestamp = excluded.end_timestamp,
|
|
85
86
|
tool = excluded.tool,
|
|
86
87
|
input = excluded.input,
|
|
87
88
|
outcome = excluded.outcome,
|
|
@@ -197,7 +198,7 @@ export function persistToolCalls(db, session, calls, sourceStamp, options = {})
|
|
|
197
198
|
}
|
|
198
199
|
for (const call of accepted) {
|
|
199
200
|
const key = toolCallKey(session.id, call.ordinal);
|
|
200
|
-
insertCall.run(key, session.id, call.ordinal, call.sourceCallId ?? null, call.timestamp, call.tool, call.input, call.outcome, call.exitCode ?? null, call.statusCode ?? null, call.errorCode ?? null, call.output ?? null, call.error ?? null, call.parseError ?? null, toolCallEvidenceBytes(call));
|
|
201
|
+
insertCall.run(key, session.id, call.ordinal, call.sourceCallId ?? null, call.timestamp, call.endTimestamp ?? null, call.tool, call.input, call.outcome, call.exitCode ?? null, call.statusCode ?? null, call.errorCode ?? null, call.output ?? null, call.error ?? null, call.parseError ?? null, toolCallEvidenceBytes(call));
|
|
201
202
|
// The upsert above preserves the rowid of a call it updated, so this is
|
|
202
203
|
// the same rowid the existing text row (if any) was written under.
|
|
203
204
|
const { rowid } = callRowid.get(key);
|
|
@@ -215,3 +216,55 @@ export function persistToolCalls(db, session, calls, sourceStamp, options = {})
|
|
|
215
216
|
});
|
|
216
217
|
txn();
|
|
217
218
|
}
|
|
219
|
+
/**
|
|
220
|
+
* Decide whether a changed full-file-harness session can RESUME its tool index
|
|
221
|
+
* from the last scan instead of re-deriving every call (PHNX-3411).
|
|
222
|
+
*
|
|
223
|
+
* The warm-tick indexer re-derives tool calls for a changed session on every
|
|
224
|
+
* tick, and for an ACTIVE large session that means re-sanitizing tens of
|
|
225
|
+
* thousands of calls each time — the synchronous work that wedged the daemon
|
|
226
|
+
* event loop. Streaming harnesses (claude/codex) already resume from a byte
|
|
227
|
+
* offset; every other harness re-parses the whole file, so this brings them the
|
|
228
|
+
* same benefit at the EVENT level: the ledger stores how many events were folded
|
|
229
|
+
* (`parsed_offset`) plus the collector snapshot (`parser_state`), and a later
|
|
230
|
+
* scan folds only the newly appended events.
|
|
231
|
+
*
|
|
232
|
+
* Returns the prior resume point when it is safe to append, or `null` (full
|
|
233
|
+
* re-scan) whenever the stored prefix may no longer describe the current file:
|
|
234
|
+
* a different extractor, no recorded resume point, a source the ledger row does
|
|
235
|
+
* not describe, a transcript that shrank below what was already parsed (a
|
|
236
|
+
* truncation/rewrite, not an append), more events already folded than the file
|
|
237
|
+
* now yields, or a snapshot that does not read back.
|
|
238
|
+
*/
|
|
239
|
+
export function planEventToolResume(db, sessionId, sourcePath, stamp, currentEventCount) {
|
|
240
|
+
const row = db.prepare(`
|
|
241
|
+
SELECT file_path, file_size, extractor_version, parsed_offset, parser_state
|
|
242
|
+
FROM tool_scan_ledger WHERE session_id = ?
|
|
243
|
+
`).get(sessionId);
|
|
244
|
+
if (!row)
|
|
245
|
+
return null;
|
|
246
|
+
if (row.extractor_version !== TOOL_INDEX_VERSION)
|
|
247
|
+
return null;
|
|
248
|
+
if (row.parsed_offset === null || !Number.isSafeInteger(row.parsed_offset) || row.parsed_offset < 0)
|
|
249
|
+
return null;
|
|
250
|
+
if (row.file_path !== canonicalToolLedgerPath(sourcePath))
|
|
251
|
+
return null;
|
|
252
|
+
if (stamp.fileSize < row.file_size)
|
|
253
|
+
return null;
|
|
254
|
+
// The file grew but reports fewer events than were already folded — the prefix
|
|
255
|
+
// was rewritten, not appended to. Re-scan from scratch.
|
|
256
|
+
if (row.parsed_offset > currentEventCount)
|
|
257
|
+
return null;
|
|
258
|
+
if (row.parser_state === null)
|
|
259
|
+
return null;
|
|
260
|
+
let snapshot;
|
|
261
|
+
try {
|
|
262
|
+
snapshot = JSON.parse(row.parser_state);
|
|
263
|
+
}
|
|
264
|
+
catch {
|
|
265
|
+
return null;
|
|
266
|
+
}
|
|
267
|
+
if (snapshot?.v !== 1 || !Number.isSafeInteger(snapshot.nextOrdinal))
|
|
268
|
+
return null;
|
|
269
|
+
return { snapshot, eventCount: row.parsed_offset };
|
|
270
|
+
}
|
|
@@ -7,6 +7,11 @@ export type EnabledExtra = {
|
|
|
7
7
|
export declare function trustedSourceBases(): {
|
|
8
8
|
dir: string;
|
|
9
9
|
}[];
|
|
10
|
+
/** Every `plugins/<plugin>/skills` dir across trusted source bases, filtered to an optional plugin/agent scope. */
|
|
11
|
+
export declare function pluginSkillDirs(options?: {
|
|
12
|
+
agent?: AgentId;
|
|
13
|
+
plugins?: Set<string>;
|
|
14
|
+
}): string[];
|
|
10
15
|
/** Find the trusted source for a command markdown by name. */
|
|
11
16
|
export declare function resolveCommandSource(name: string): string | null;
|
|
12
17
|
/** Find the trusted source directory for a skill by name. */
|
|
@@ -54,7 +54,8 @@ function pluginSupportsAgent(manifest, agent) {
|
|
|
54
54
|
return true;
|
|
55
55
|
return !manifest.agents || manifest.agents.length === 0 || manifest.agents.includes(agent);
|
|
56
56
|
}
|
|
57
|
-
|
|
57
|
+
/** Every `plugins/<plugin>/skills` dir across trusted source bases, filtered to an optional plugin/agent scope. */
|
|
58
|
+
export function pluginSkillDirs(options = {}) {
|
|
58
59
|
const dirs = [];
|
|
59
60
|
for (const base of trustedSourceBases()) {
|
|
60
61
|
const pluginsDir = path.join(base.dir, 'plugins');
|
|
@@ -93,6 +93,28 @@ export interface UnifiedSyncStatus {
|
|
|
93
93
|
agentsNeedingSync: number;
|
|
94
94
|
};
|
|
95
95
|
}
|
|
96
|
+
/**
|
|
97
|
+
* Residual drift for a single (agent, version) after a reconcile — the drifted
|
|
98
|
+
* and missing rows that a sync claimed to fix but did not. `orphan` rows are
|
|
99
|
+
* deliberately excluded: sync never removes them (that is `agents prune`'s job),
|
|
100
|
+
* so they are not "unfinished sync". Empty `drifted`+`missing` ⇒ converged.
|
|
101
|
+
*/
|
|
102
|
+
export interface ResidualDrift {
|
|
103
|
+
agent: AgentId;
|
|
104
|
+
version: string;
|
|
105
|
+
rows: ResourceStatusRow[];
|
|
106
|
+
}
|
|
107
|
+
/**
|
|
108
|
+
* Re-check that a version's home now matches its resolved sources after a
|
|
109
|
+
* reconcile. This is the post-write verification the `agents sync` success line
|
|
110
|
+
* depends on (PHNX-3186): the exit line must not read "reconciled" while drift
|
|
111
|
+
* it was asked to fix stays put. Resolves against non-project layers only
|
|
112
|
+
* (`excludeProject: true`), mirroring what the sync writer targets. Returns null
|
|
113
|
+
* when the version converged (no drifted/missing), else the residual rows.
|
|
114
|
+
*/
|
|
115
|
+
export declare function verifyVersionConverged(agent: AgentId, version: string, cwd?: string): ResidualDrift | null;
|
|
116
|
+
/** One-line-per-resource description of residual drift, for the sync exit line. */
|
|
117
|
+
export declare function formatResidualDrift(residual: ResidualDrift[]): string[];
|
|
96
118
|
export interface SyncStatusOptions {
|
|
97
119
|
cwd?: string;
|
|
98
120
|
/** Restrict to specific agent ids; undefined = every supported agent. */
|
package/dist/lib/sync-status.js
CHANGED
|
@@ -27,6 +27,33 @@ import { getSystemAgentsDir, getUserAgentsDir } from './state.js';
|
|
|
27
27
|
import * as fs from 'fs';
|
|
28
28
|
import { isGitRepo, readOriginUrl } from './git.js';
|
|
29
29
|
import { detectConfigDrift } from './config-drift.js';
|
|
30
|
+
/**
|
|
31
|
+
* Re-check that a version's home now matches its resolved sources after a
|
|
32
|
+
* reconcile. This is the post-write verification the `agents sync` success line
|
|
33
|
+
* depends on (PHNX-3186): the exit line must not read "reconciled" while drift
|
|
34
|
+
* it was asked to fix stays put. Resolves against non-project layers only
|
|
35
|
+
* (`excludeProject: true`), mirroring what the sync writer targets. Returns null
|
|
36
|
+
* when the version converged (no drifted/missing), else the residual rows.
|
|
37
|
+
*/
|
|
38
|
+
export function verifyVersionConverged(agent, version, cwd = process.cwd()) {
|
|
39
|
+
const report = diffVersionResources(agent, version, { cwd, excludeProject: true });
|
|
40
|
+
const rows = rowsFromReport(agent, version, report)
|
|
41
|
+
.filter((r) => r.status === 'drifted' || r.status === 'missing');
|
|
42
|
+
if (rows.length === 0)
|
|
43
|
+
return null;
|
|
44
|
+
return { agent, version, rows };
|
|
45
|
+
}
|
|
46
|
+
/** One-line-per-resource description of residual drift, for the sync exit line. */
|
|
47
|
+
export function formatResidualDrift(residual) {
|
|
48
|
+
const lines = [];
|
|
49
|
+
for (const r of residual) {
|
|
50
|
+
for (const row of r.rows) {
|
|
51
|
+
const detail = row.detail ? ` (${row.detail})` : '';
|
|
52
|
+
lines.push(`${r.agent}@${r.version}: ${row.status} ${row.kind} '${row.name}'${detail}`);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
return lines;
|
|
56
|
+
}
|
|
30
57
|
const STATUS_MAP = {
|
|
31
58
|
ok: 'synced',
|
|
32
59
|
diff: 'drifted',
|
|
@@ -64,6 +64,15 @@ export interface UmbrellaResult {
|
|
|
64
64
|
* Empty when nothing was declined — never conflated with "nothing to do".
|
|
65
65
|
*/
|
|
66
66
|
declined: string[];
|
|
67
|
+
/**
|
|
68
|
+
* The (agent, version) pairs the reconcile stage actually wrote into — the set
|
|
69
|
+
* the caller re-verifies for residual drift so the `✓ sync: reconciled` line is
|
|
70
|
+
* never printed while drift it was asked to fix stays put (PHNX-3186).
|
|
71
|
+
*/
|
|
72
|
+
reconciledVersions: Array<{
|
|
73
|
+
agent: string;
|
|
74
|
+
version: string;
|
|
75
|
+
}>;
|
|
67
76
|
}
|
|
68
77
|
export interface RunUmbrellaArgs {
|
|
69
78
|
flags: UmbrellaFlags;
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
* arrives with `agents secrets vault unlock`, #366/#367)
|
|
16
16
|
* reconcile-> refresh({ skipPrompts }) — re-materialize resources into homes
|
|
17
17
|
*/
|
|
18
|
-
import { pullRepo } from './git.js';
|
|
18
|
+
import { pullRepo, adoptUserRepoIfNeeded } from './git.js';
|
|
19
19
|
import { getUserAgentsDir, getEnabledExtraRepos } from './state.js';
|
|
20
20
|
import { listRemoteBundles, pullBundle } from './secrets/sync.js';
|
|
21
21
|
import { SYNC_PASSPHRASE_ENV } from './secrets/sync-passphrase.js';
|
|
@@ -55,7 +55,7 @@ export function planUmbrellaStages(f) {
|
|
|
55
55
|
export async function runUmbrellaSync(args) {
|
|
56
56
|
const { flags, log, yes, passphrase, quiet = false } = args;
|
|
57
57
|
const plan = planUmbrellaStages(flags);
|
|
58
|
-
const result = { plan, reconciled: false, declined: [] };
|
|
58
|
+
const result = { plan, reconciled: false, declined: [], reconciledVersions: [] };
|
|
59
59
|
if (plan.fetchRepos) {
|
|
60
60
|
const dirs = [
|
|
61
61
|
{ alias: 'user', dir: getUserAgentsDir() },
|
|
@@ -64,6 +64,24 @@ export async function runUmbrellaSync(args) {
|
|
|
64
64
|
let pulled = 0;
|
|
65
65
|
const errors = [];
|
|
66
66
|
for (const { alias, dir } of dirs) {
|
|
67
|
+
// A box where `~/.agents` is present but not a git repo (or lost its
|
|
68
|
+
// `.git`) is a partial install: `pullRepo` there silently fails while the
|
|
69
|
+
// umbrella still reported `✓ reconciled`, so fleet dotfiles/resources never
|
|
70
|
+
// propagated and nothing said so (PHNX-3239, m0). Adopt it in place first —
|
|
71
|
+
// the same self-heal `agents sync user` runs (PHNX-3301) — so the pull has a
|
|
72
|
+
// real repo to fast-forward. A box that cannot be adopted (no recorded
|
|
73
|
+
// remote) fails LOUD into `errors` with the reason, never a silent no-op.
|
|
74
|
+
if (alias === 'user') {
|
|
75
|
+
const adopted = await adoptUserRepoIfNeeded(dir);
|
|
76
|
+
if (adopted && !adopted.success) {
|
|
77
|
+
const hint = adopted.needsUrl ? ' — git-back it: agents repo pull user <git-url>' : '';
|
|
78
|
+
errors.push(`${alias}: ${adopted.error}${hint}`);
|
|
79
|
+
continue;
|
|
80
|
+
}
|
|
81
|
+
if (adopted?.success) {
|
|
82
|
+
log(`repos: ${alias} adopted in place → ${adopted.commit} (${adopted.materialized} file(s) materialized)`);
|
|
83
|
+
}
|
|
84
|
+
}
|
|
67
85
|
const r = await pullRepo(dir);
|
|
68
86
|
if (r.success) {
|
|
69
87
|
pulled++;
|
|
@@ -116,6 +134,7 @@ export async function runUmbrellaSync(args) {
|
|
|
116
134
|
const refreshed = await refresh({ skipPrompts: yes, quiet });
|
|
117
135
|
result.reconciled = true;
|
|
118
136
|
result.declined = refreshed.declined;
|
|
137
|
+
result.reconciledVersions = refreshed.reconciled;
|
|
119
138
|
// Keep already-registered devices' reachability current, and surface newly
|
|
120
139
|
// appeared tailnet nodes as "pending" for the menu-bar Register/Ignore gate
|
|
121
140
|
// rather than silently adding them (refresh mode). Soft: a machine without
|
|
@@ -12,14 +12,21 @@
|
|
|
12
12
|
* is enough to reconstruct per-session call order and inter-call gaps without a
|
|
13
13
|
* full `SessionTrajectory` — that is what makes this incremental at scale.
|
|
14
14
|
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
* needs the full derived trajectory
|
|
18
|
-
*
|
|
19
|
-
* `
|
|
20
|
-
* `
|
|
15
|
+
* Failure phenotype (false-termination / out-of-order / premature-completion /
|
|
16
|
+
* failure-to-act, `phenotype.ts`) is folded in as a fourth grouping dimension
|
|
17
|
+
* (PHNX-3327). Classifying it needs the full derived trajectory, so it is NOT
|
|
18
|
+
* computed here — the caller passes a per-session `phenotypes` map that
|
|
19
|
+
* `buildIndexShard` fills from the persisted, mtime+size-keyed
|
|
20
|
+
* `session_phenotypes` cache for the WHOLE corpus. Keying the group on the
|
|
21
|
+
* cached-per-session phenotype (never on this run's incremental batch) is what
|
|
22
|
+
* keeps two identically-signatured sessions in one cluster regardless of when
|
|
23
|
+
* each was synced. The `signature` OUTPUT is unchanged (`{ tool, cause, key }`);
|
|
24
|
+
* phenotype is an added dimension carried alongside it, so callers that never
|
|
25
|
+
* pass a map (the unit tests, a pre-phenotype caller) see the exact prior
|
|
26
|
+
* grouping.
|
|
21
27
|
*/
|
|
22
28
|
import { type TraceFailureCause } from './classify.js';
|
|
29
|
+
import type { FailurePhenotype } from './phenotype.js';
|
|
23
30
|
import { type LatencyInsight } from './segments.js';
|
|
24
31
|
import { type SyncRow, type ToolCallRow, type TracesIndexShard } from './sync.js';
|
|
25
32
|
export interface FailureSignature {
|
|
@@ -29,10 +36,17 @@ export interface FailureSignature {
|
|
|
29
36
|
key: string;
|
|
30
37
|
}
|
|
31
38
|
export interface FailurePattern {
|
|
32
|
-
/** Stable hash of the signature — deep-linkable, unaffected by row order. */
|
|
39
|
+
/** Stable hash of the signature (incl. phenotype) — deep-linkable, unaffected by row order. */
|
|
33
40
|
id: string;
|
|
34
41
|
label: string;
|
|
35
42
|
signature: FailureSignature;
|
|
43
|
+
/**
|
|
44
|
+
* Dominant failure phenotype of the sessions in this cluster, or `null` when
|
|
45
|
+
* none was classifiable. A fourth grouping dimension (PHNX-3327): two failures
|
|
46
|
+
* with the same `(tool, cause, key)` but different phenotypes are distinct
|
|
47
|
+
* patterns.
|
|
48
|
+
*/
|
|
49
|
+
phenotype: FailurePhenotype | null;
|
|
36
50
|
/** Distinct sessions this pattern occurred in. */
|
|
37
51
|
sessions: number;
|
|
38
52
|
/** Total failing calls matching this signature. */
|
|
@@ -57,11 +71,30 @@ export declare function normalizeErrorKey(desc: string, raw: string | null): str
|
|
|
57
71
|
* Cluster failed tool calls into ranked patterns and estimate the wasted time
|
|
58
72
|
* behind each, plus device-wide time-to-first-tool latency.
|
|
59
73
|
*
|
|
60
|
-
* wastedMs attribution
|
|
61
|
-
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
74
|
+
* wastedMs attribution has two parts that sum:
|
|
75
|
+
*
|
|
76
|
+
* (1) The failed call's OWN blocking duration — `end_timestamp - timestamp`
|
|
77
|
+
* (PHNX-3437). A call that hung for minutes and then failed wasted that whole
|
|
78
|
+
* time even if it was the last call in its session or was followed quickly by an
|
|
79
|
+
* unrelated call — the case the gap heuristic alone booked as ~0. This is what
|
|
80
|
+
* makes a fail-fast fix measurable: a channel that stops hanging 5.5min on stdin
|
|
81
|
+
* and instead fails in <1s (PHNX-3407) drops from ~5.5min of attributed waste to
|
|
82
|
+
* ~0. Bounded by MAX_GAP_ATTRIBUTION_MS so a corrupt end timestamp can't dominate.
|
|
83
|
+
*
|
|
84
|
+
* (2) The idle gap AFTER the call, before the NEXT call in the same session,
|
|
85
|
+
* counted when either (a) the next call repeats the same signature (a retry
|
|
86
|
+
* loop) or (b) the gap is a stall (≥60s) before an unrelated next call — each
|
|
87
|
+
* bounded by MAX_GAP_ATTRIBUTION_MS so one failure can't absorb hours of
|
|
88
|
+
* human-away idle in an async channel session, whether as a lone stall or a
|
|
89
|
+
* same-signature re-ask hours later; a genuine active retry loop is many short
|
|
90
|
+
* gaps that each clear the cap and still sum large. When the end timestamp is
|
|
91
|
+
* known, this gap is measured from the call's END, so the blocking time counted
|
|
92
|
+
* in (1) is never double-counted; for a NULL end (rows an older extractor stored,
|
|
93
|
+
* or a call still pending at scan end) it falls back to the original
|
|
94
|
+
* gap-from-START heuristic unchanged — no crash, no NaN, no regression.
|
|
95
|
+
*
|
|
96
|
+
* An idle gap unrelated to a nearby failure is never counted. This is an
|
|
97
|
+
* estimate, not ground truth; it is not inflated by folding in ordinary
|
|
98
|
+
* processing time between unrelated calls.
|
|
66
99
|
*/
|
|
67
|
-
export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null): ComputedInsights;
|
|
100
|
+
export declare function computeInsights(rows: readonly SyncRow[], calls: readonly ToolCallRow[], prevShard?: TracesIndexShard | null, phenotypes?: ReadonlyMap<string, FailurePhenotype | null>): ComputedInsights;
|