@phnx-labs/agents-cli 1.22.58 → 1.22.59
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +238 -0
- package/README.md +29 -0
- package/dist/bootstrap.js +32 -1
- package/dist/commands/monitors.js +187 -23
- package/dist/commands/routines.test-fixture.js +5 -0
- package/dist/commands/send.d.ts +2 -1
- package/dist/commands/send.js +7 -5
- package/dist/commands/sessions-stats.js +37 -5
- package/dist/commands/sessions.js +39 -5
- package/dist/commands/ssh.js +12 -1
- package/dist/commands/versions.js +12 -4
- package/dist/commands/view.js +7 -2
- package/dist/lib/auto-pull-worker.js +7 -2
- package/dist/lib/cloud/rush.d.ts +7 -0
- package/dist/lib/cloud/rush.js +29 -1
- package/dist/lib/daemon/daemon.d.ts +22 -0
- package/dist/lib/daemon/daemon.js +39 -0
- package/dist/lib/daemon/session-index-service.js +9 -1
- package/dist/lib/daemon-ticks.d.ts +15 -0
- package/dist/lib/daemon-ticks.js +26 -0
- package/dist/lib/device-config.d.ts +5 -1
- package/dist/lib/device-config.js +2 -2
- package/dist/lib/devices/health.js +5 -1
- package/dist/lib/devices/pool.d.ts +25 -2
- package/dist/lib/devices/pool.js +32 -2
- package/dist/lib/devices/stats-cache.d.ts +0 -6
- package/dist/lib/devices/stats-cache.js +2 -9
- package/dist/lib/doctor-diff.d.ts +14 -0
- package/dist/lib/doctor-diff.js +43 -2
- package/dist/lib/git.d.ts +38 -0
- package/dist/lib/git.js +58 -0
- package/dist/lib/hosts/ready.d.ts +8 -0
- package/dist/lib/hosts/ready.js +13 -2
- package/dist/lib/installations/versions.d.ts +17 -0
- package/dist/lib/installations/versions.js +53 -2
- package/dist/lib/monitors/config.d.ts +71 -3
- package/dist/lib/monitors/config.js +100 -12
- package/dist/lib/monitors/pid-watch.d.ts +35 -0
- package/dist/lib/monitors/pid-watch.js +45 -0
- package/dist/lib/monitors/remote.d.ts +18 -0
- package/dist/lib/monitors/remote.js +11 -0
- package/dist/lib/permissions.js +7 -2
- package/dist/lib/plugins/plugins.d.ts +17 -3
- package/dist/lib/plugins/plugins.js +84 -9
- package/dist/lib/pty-server.d.ts +14 -0
- package/dist/lib/pty-server.js +49 -5
- package/dist/lib/secrets/drivers/rush.js +5 -0
- package/dist/lib/self-update.d.ts +42 -0
- package/dist/lib/self-update.js +88 -0
- package/dist/lib/session/cloud.js +5 -0
- package/dist/lib/session/db.d.ts +32 -6
- package/dist/lib/session/db.js +128 -12
- package/dist/lib/smart-launch.d.ts +6 -0
- package/dist/lib/smart-launch.js +5 -2
- package/dist/lib/staleness/writers/plugins.js +5 -2
- package/dist/lib/staleness/writers/subagents.js +13 -3
- package/dist/lib/state.d.ts +7 -4
- package/dist/lib/state.js +7 -4
- package/dist/lib/subagents.js +8 -2
- package/dist/lib/teams/scheduler.d.ts +10 -0
- package/dist/lib/teams/scheduler.js +8 -0
- package/dist/lib/traces/sync.d.ts +113 -6
- package/dist/lib/traces/sync.js +193 -19
- package/dist/lib/view-types.d.ts +12 -0
- package/package.json +2 -2
package/dist/lib/session/db.js
CHANGED
|
@@ -27,7 +27,7 @@ const DB_PATH = getSessionsDbPath();
|
|
|
27
27
|
/** Current schema version; bumped when migrations are added. Exported so tests
|
|
28
28
|
* assert against the constant instead of hardcoding a number that every bump
|
|
29
29
|
* then has to chase (docs/sessions.md calls the constant the source of truth). */
|
|
30
|
-
export const SCHEMA_VERSION =
|
|
30
|
+
export const SCHEMA_VERSION = 44;
|
|
31
31
|
/**
|
|
32
32
|
* Bump to force the content extractor (assistant-answer text, alongside the
|
|
33
33
|
* user-prompt text every harness already accumulates) to re-derive on every
|
|
@@ -1205,6 +1205,32 @@ function migrateSchema(db, fromVersion) {
|
|
|
1205
1205
|
if (!cols.has('end_timestamp'))
|
|
1206
1206
|
db.exec(`ALTER TABLE tool_calls ADD COLUMN end_timestamp TEXT`);
|
|
1207
1207
|
}
|
|
1208
|
+
if (fromVersion < 44) {
|
|
1209
|
+
// v43 -> v44: backfill duration_ms for sessions whose harness scan extractor
|
|
1210
|
+
// never derived it (PHNX-3457). rush/grok/kimi/cursor/muse/antigravity/hermes/
|
|
1211
|
+
// openclaw left duration_ms NULL — 52% of the corpus, 100% of rush — so the
|
|
1212
|
+
// console median was computed over only the ~48% that carried it, skewing it
|
|
1213
|
+
// short. Going forward resolveDurationMs() populates it at every upsert; this
|
|
1214
|
+
// repairs already-indexed rows in place from the timestamps they already store
|
|
1215
|
+
// (last_activity, itself resolved from the last-message time else file mtime,
|
|
1216
|
+
// minus the creation timestamp), so no transcript is re-parsed. julianday is
|
|
1217
|
+
// avoided because it does not accept a trailing 'Z'; the arithmetic is done in
|
|
1218
|
+
// JS with the exact Date.parse resolveDurationMs uses, keeping the backfill and
|
|
1219
|
+
// the live path consistent. Only rows with a positive span are touched; a NULL
|
|
1220
|
+
// that cannot be resolved stays NULL rather than becoming a fabricated 0.
|
|
1221
|
+
const nullDurationRows = db.prepare(`SELECT id, timestamp, last_activity FROM sessions
|
|
1222
|
+
WHERE duration_ms IS NULL AND last_activity IS NOT NULL`).all();
|
|
1223
|
+
// Runs inside migrateSchema's own transaction (db.ts:1468), so no nested
|
|
1224
|
+
// db.transaction() here — that would raise "transaction within a transaction".
|
|
1225
|
+
const update = db.prepare(`UPDATE sessions SET duration_ms = ? WHERE id = ?`);
|
|
1226
|
+
for (const row of nullDurationRows) {
|
|
1227
|
+
const startMs = Date.parse(row.timestamp);
|
|
1228
|
+
const lastMs = Date.parse(row.last_activity);
|
|
1229
|
+
if (Number.isFinite(startMs) && Number.isFinite(lastMs) && lastMs > startMs) {
|
|
1230
|
+
update.run(lastMs - startMs, row.id);
|
|
1231
|
+
}
|
|
1232
|
+
}
|
|
1233
|
+
}
|
|
1208
1234
|
}
|
|
1209
1235
|
/**
|
|
1210
1236
|
* Stamp `account_key` / `account_org` / `account` on every Claude row from its
|
|
@@ -2075,7 +2101,7 @@ export function upsertSession(meta, content, scan, assistantContent = '') {
|
|
|
2075
2101
|
cache_write_tokens: meta.cacheWriteTokens ?? null,
|
|
2076
2102
|
cost_usd: meta.costUsd ?? null,
|
|
2077
2103
|
cost_usd_nocache: meta.costUsdNoCache ?? null,
|
|
2078
|
-
duration_ms: meta
|
|
2104
|
+
duration_ms: resolveDurationMs(meta, scan),
|
|
2079
2105
|
model: meta.model ?? null,
|
|
2080
2106
|
tool_call_count: meta.toolCallCount ?? null,
|
|
2081
2107
|
file_path: meta.filePath,
|
|
@@ -2163,6 +2189,23 @@ export function upsertSessionsBatch(entries) {
|
|
|
2163
2189
|
? { ...entry, meta: { ...entry.meta, ...fanOutCounts(entry.events, entry.meta.agent) } }
|
|
2164
2190
|
: entry;
|
|
2165
2191
|
}
|
|
2192
|
+
// Harnesses whose scanners produce no events AND whose parseSession reads a
|
|
2193
|
+
// potentially large flat transcript file (not a compact SQLite DB). Calling
|
|
2194
|
+
// parseSession on the warm tick for an active large session wedges the Node
|
|
2195
|
+
// event loop for seconds, making browser IPC miss its connection window
|
|
2196
|
+
// (PHNX-3411). Defer their tool-call indexing to runDeferredToolIndex, which
|
|
2197
|
+
// uses ensureToolIndex with tool_scan_ledger stamps and byte/file budget caps.
|
|
2198
|
+
// NOT opencode — parseOpenCode issues a targeted SQLite query, so it is fast
|
|
2199
|
+
// even for large sessions and its results populate recentDirectoriesTouched.
|
|
2200
|
+
const LARGE_TRANSCRIPT_AGENTS = new Set(['kimi', 'grok']);
|
|
2201
|
+
if (!entry.events && LARGE_TRANSCRIPT_AGENTS.has(entry.meta.agent)) {
|
|
2202
|
+
return entry;
|
|
2203
|
+
}
|
|
2204
|
+
// Enrich the entry. When the scanner already provided events, use them
|
|
2205
|
+
// directly (no transcript re-read). When it didn't — e.g. OpenCode whose
|
|
2206
|
+
// scanner produces only metadata but whose parseOpenCode is a fast SQLite
|
|
2207
|
+
// query — fall back to parseSession. Agents that would call an expensive
|
|
2208
|
+
// flat-file parse already returned above.
|
|
2166
2209
|
try {
|
|
2167
2210
|
const toolSourcePath = toolEvidenceSourcePath(entry.meta.filePath, entry.meta.agent);
|
|
2168
2211
|
const toolScan = toolSourcePath === entry.meta.filePath
|
|
@@ -2171,9 +2214,6 @@ export function upsertSessionsBatch(entries) {
|
|
|
2171
2214
|
const stat = fs.statSync(toolSourcePath);
|
|
2172
2215
|
return { fileMtimeMs: stat.mtimeMs, fileSize: stat.size };
|
|
2173
2216
|
})();
|
|
2174
|
-
// Some non-resumable scanners already normalized the transcript while
|
|
2175
|
-
// deriving metadata. Reuse those events; scanners that only read summary
|
|
2176
|
-
// metadata fall back to exactly one normalized parse here.
|
|
2177
2217
|
const events = entry.events ?? parseSession(entry.meta.filePath, entry.meta.agent);
|
|
2178
2218
|
writeResourceUsage(entry.meta.id, events, entry.meta.cwd);
|
|
2179
2219
|
// Resume the tool index from the last scan of this append-only stream when
|
|
@@ -2302,7 +2342,7 @@ export function upsertSessionsBatch(entries) {
|
|
|
2302
2342
|
cache_write_tokens: meta.cacheWriteTokens ?? null,
|
|
2303
2343
|
cost_usd: meta.costUsd ?? null,
|
|
2304
2344
|
cost_usd_nocache: meta.costUsdNoCache ?? null,
|
|
2305
|
-
duration_ms: meta
|
|
2345
|
+
duration_ms: resolveDurationMs(meta, scan),
|
|
2306
2346
|
model: meta.model ?? null,
|
|
2307
2347
|
tool_call_count: meta.toolCallCount ?? null,
|
|
2308
2348
|
file_path: meta.filePath,
|
|
@@ -2599,6 +2639,36 @@ function resolveLastActivity(meta, scan) {
|
|
|
2599
2639
|
return new Date(scan.fileMtimeMs).toISOString();
|
|
2600
2640
|
return meta.timestamp;
|
|
2601
2641
|
}
|
|
2642
|
+
/**
|
|
2643
|
+
* The persisted wall-clock span for a session (PHNX-3457).
|
|
2644
|
+
*
|
|
2645
|
+
* `durationMs` is canonically `lastTs − firstTs`. Some harness scan extractors
|
|
2646
|
+
* derive it themselves from per-event timestamps (claude/codex/droid/gemini/
|
|
2647
|
+
* opencode set `meta.durationMs`); the rest (rush/grok/kimi/cursor/muse/
|
|
2648
|
+
* antigravity/hermes/openclaw) never did, so `duration_ms` landed NULL for them —
|
|
2649
|
+
* 52% of the corpus, including 100% of the dominant `rush` usage — and the
|
|
2650
|
+
* console median was computed over only the ~48% that happened to carry it,
|
|
2651
|
+
* skewing it short and misleading.
|
|
2652
|
+
*
|
|
2653
|
+
* This closes the gap at the single write boundary every harness funnels
|
|
2654
|
+
* through, so parity is automatic rather than per-extractor: when the extractor
|
|
2655
|
+
* already computed a precise span, keep it; otherwise derive it from the same
|
|
2656
|
+
* `timestamp` (creation) and `resolveLastActivity` (last event, itself resolved
|
|
2657
|
+
* from the harness's last-message time, else file mtime) that the row already
|
|
2658
|
+
* stores. Returns null only when no positive span can be established (a single
|
|
2659
|
+
* timestamped event, or a clock that runs backwards), which reads as NULL rather
|
|
2660
|
+
* than a fabricated 0.
|
|
2661
|
+
*/
|
|
2662
|
+
function resolveDurationMs(meta, scan) {
|
|
2663
|
+
if (meta.durationMs != null)
|
|
2664
|
+
return meta.durationMs;
|
|
2665
|
+
const startMs = Date.parse(meta.timestamp);
|
|
2666
|
+
const lastMs = Date.parse(resolveLastActivity(meta, scan));
|
|
2667
|
+
if (Number.isFinite(startMs) && Number.isFinite(lastMs) && lastMs > startMs) {
|
|
2668
|
+
return lastMs - startMs;
|
|
2669
|
+
}
|
|
2670
|
+
return null;
|
|
2671
|
+
}
|
|
2602
2672
|
export function isSessionActivityFresh(row, maxAgeMs, nowMs) {
|
|
2603
2673
|
const parsedActivityMs = Date.parse(row.last_activity ?? row.timestamp);
|
|
2604
2674
|
const activityMs = Number.isFinite(parsedActivityMs) ? parsedActivityMs : row.file_mtime_ms ?? undefined;
|
|
@@ -2892,6 +2962,26 @@ export function querySessions(options = {}) {
|
|
|
2892
2962
|
const trimmed = options.limit ? live.slice(0, options.limit) : live;
|
|
2893
2963
|
return trimmed.map(rowToMeta);
|
|
2894
2964
|
}
|
|
2965
|
+
/**
|
|
2966
|
+
* Cheap query for the daemon's deferred tool-index pass (PHNX-3411).
|
|
2967
|
+
*
|
|
2968
|
+
* Returns the most-recently-active sessions whose parseSession reads a large
|
|
2969
|
+
* flat transcript (kimi: wire.jsonl, grok: chat_history.jsonl). Their scanners
|
|
2970
|
+
* produce no events, so upsertSessionsBatch skips them in the warm tick to
|
|
2971
|
+
* avoid wedging the event loop. ensureToolIndex uses tool_scan_ledger stamps
|
|
2972
|
+
* to skip already-current rows and applies byte/file budget caps.
|
|
2973
|
+
*/
|
|
2974
|
+
export function querySessionsForDeferredToolIndex(limit) {
|
|
2975
|
+
const db = getDB();
|
|
2976
|
+
const rows = db.prepare(`
|
|
2977
|
+
SELECT * FROM sessions
|
|
2978
|
+
WHERE file_path IS NOT NULL
|
|
2979
|
+
AND agent IN ('kimi', 'grok')
|
|
2980
|
+
ORDER BY last_activity DESC, timestamp DESC
|
|
2981
|
+
LIMIT ?
|
|
2982
|
+
`).all(limit);
|
|
2983
|
+
return rows.map(rowToMeta);
|
|
2984
|
+
}
|
|
2895
2985
|
/** Count sessions matching the given filter options. */
|
|
2896
2986
|
export function countSessions(options = {}) {
|
|
2897
2987
|
const db = getDB();
|
|
@@ -3333,17 +3423,43 @@ export function queryResourceUsageStats(options) {
|
|
|
3333
3423
|
return db.prepare(sql).all(...allParams);
|
|
3334
3424
|
}
|
|
3335
3425
|
/**
|
|
3336
|
-
* Coverage of the resource-usage signal
|
|
3337
|
-
*
|
|
3338
|
-
*
|
|
3339
|
-
*
|
|
3340
|
-
*
|
|
3426
|
+
* Coverage of the resource-usage signal, as three honest facts:
|
|
3427
|
+
*
|
|
3428
|
+
* - `scanned` — sessions the resource extractor has actually processed, i.e.
|
|
3429
|
+
* those carrying a `resource_scan_ledger` row at the current
|
|
3430
|
+
* `RESOURCE_INDEX_VERSION`. This is the true "has the historical backfill run"
|
|
3431
|
+
* signal: the ledger is stamped for EVERY scanned session, including ones that
|
|
3432
|
+
* invoked nothing (`resource_count = 0`), so `scanned/total` rises to ~1 after
|
|
3433
|
+
* `agents sessions backfill resources` regardless of how sparse explicit
|
|
3434
|
+
* invocations are.
|
|
3435
|
+
* - `covered` — distinct sessions that carry AT LEAST ONE row in
|
|
3436
|
+
* `session_resource_usage`, i.e. that actually recorded an explicit invocation.
|
|
3437
|
+
* This is an ABSOLUTE signal count, not a coverage ratio: it stays small even
|
|
3438
|
+
* at full scan coverage because most sessions invoke no skill/command, and a
|
|
3439
|
+
* non-recording harness contributes none by construction.
|
|
3440
|
+
* - `total` — sessions indexed.
|
|
3441
|
+
*
|
|
3442
|
+
* The two were previously conflated: `covered/total` was framed as coverage and
|
|
3443
|
+
* read ~1.2% even after a full backfill (most sessions genuinely invoke nothing),
|
|
3444
|
+
* so the "run the backfill" hint never cleared. Keying the hint on `scanned/total`
|
|
3445
|
+
* fixes that — see `commands/sessions-stats.ts` (PHNX-2301).
|
|
3341
3446
|
*/
|
|
3342
3447
|
export function resourceUsageCoverage() {
|
|
3343
3448
|
const db = getDB();
|
|
3344
3449
|
const covered = db.prepare(`SELECT COUNT(DISTINCT session_id) AS n FROM session_resource_usage`).get().n;
|
|
3450
|
+
// JOIN sessions so a ledger row for a since-vanished transcript (dropped from
|
|
3451
|
+
// `sessions` but not yet from the ledger) can't inflate scan coverage past the
|
|
3452
|
+
// indexed set. Only rows at the current extractor version count as scanned —
|
|
3453
|
+
// a stale-version row is re-derived on the next backfill, so it is not yet
|
|
3454
|
+
// "covered" for this extractor.
|
|
3455
|
+
const scanned = db.prepare(`
|
|
3456
|
+
SELECT COUNT(*) AS n
|
|
3457
|
+
FROM resource_scan_ledger l
|
|
3458
|
+
JOIN sessions s ON s.id = l.session_id
|
|
3459
|
+
WHERE l.extractor_version = ?
|
|
3460
|
+
`).get(RESOURCE_INDEX_VERSION).n;
|
|
3345
3461
|
const total = db.prepare(`SELECT COUNT(*) AS n FROM sessions`).get().n;
|
|
3346
|
-
return { covered, total };
|
|
3462
|
+
return { covered, scanned, total };
|
|
3347
3463
|
}
|
|
3348
3464
|
/** Has this session's resource usage been derived at the current extractor version for this exact file? */
|
|
3349
3465
|
function needsResourceIndex(db, sessionId, stamp) {
|
|
@@ -95,6 +95,12 @@ export declare function resolveDeviceAuto(agent?: string, opts?: {
|
|
|
95
95
|
/** Route only to devices with a row the interactive account picker can launch. */
|
|
96
96
|
accountPicker?: boolean;
|
|
97
97
|
probe?: (pool: string[], agent?: AgentType) => Promise<Map<string, DevicePlacementSignal>>;
|
|
98
|
+
/**
|
|
99
|
+
* Preferred hosts (`auto-launch.preferred`) that get the ranking boost.
|
|
100
|
+
* Defaults to the stored fleet block resolved over the candidate pool;
|
|
101
|
+
* injectable so a test pins it without touching disk.
|
|
102
|
+
*/
|
|
103
|
+
preferred?: ReadonlySet<string>;
|
|
98
104
|
}): Promise<DeviceAutoPlan>;
|
|
99
105
|
/**
|
|
100
106
|
* Resolve host for `--device auto`. Does NOT pick harness or accounts.
|
package/dist/lib/smart-launch.js
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
import { queryAffinityRollup } from './session/db.js';
|
|
9
9
|
import { localMachineId } from './session/origin-machine.js';
|
|
10
10
|
import { loadDevicesSync } from './devices/registry.js';
|
|
11
|
-
import { describeAutoPool, filterAutoPool, isAutoPoolMember } from './devices/pool.js';
|
|
11
|
+
import { autoLaunchPreferredSet, describeAutoPool, filterAutoPool, isAutoPoolMember } from './devices/pool.js';
|
|
12
12
|
import { normalizeHost } from './machine-id.js';
|
|
13
13
|
import { probePoolSignals } from './teams/placement-probe.js';
|
|
14
14
|
import { pickBestDevice } from './teams/scheduler.js';
|
|
@@ -153,7 +153,10 @@ export async function resolveDeviceAuto(agent, opts = {}) {
|
|
|
153
153
|
if (eligiblePool.length === 0) {
|
|
154
154
|
throw new Error(formatNoHealthyDeviceError(pool, signals, agent));
|
|
155
155
|
}
|
|
156
|
-
|
|
156
|
+
// `agents devices prefer <name>` boosts a device in the ranking — resolved
|
|
157
|
+
// over the same pool so a fleet default reaches a doc-less box.
|
|
158
|
+
const preferred = opts.preferred ?? autoLaunchPreferredSet(pool, { roster: pool });
|
|
159
|
+
const picked = pickBestDevice(eligiblePool, [], { signals, agentLabel: agent, preferred });
|
|
157
160
|
return {
|
|
158
161
|
host: picked === local ? null : picked,
|
|
159
162
|
pickedDeviceKey: picked,
|
|
@@ -8,8 +8,11 @@ function buildPluginsWriter(agent) {
|
|
|
8
8
|
write({ version, versionHome, selection }) {
|
|
9
9
|
const all = discoverPlugins();
|
|
10
10
|
const map = new Map(all.map(p => [p.name, p]));
|
|
11
|
-
// Clean orphan plugin-skills from plugins that no longer exist.
|
|
12
|
-
|
|
11
|
+
// Clean orphan plugin-skills from plugins that no longer exist. Pass the
|
|
12
|
+
// discovered plugins (not just names) so a stale install under one
|
|
13
|
+
// marketplace is trashed even when another marketplace still ships that
|
|
14
|
+
// name — the PHNX-2618 shadow `code` plugin.
|
|
15
|
+
cleanOrphanedPluginSkills(agent, versionHome, all);
|
|
13
16
|
const synced = [];
|
|
14
17
|
for (const name of selection) {
|
|
15
18
|
const plugin = map.get(name);
|
|
@@ -25,10 +25,16 @@ function buildSubagentsWriter(agent) {
|
|
|
25
25
|
const dir = target.dir(versionHome);
|
|
26
26
|
const synced = [];
|
|
27
27
|
const paths = [];
|
|
28
|
+
const errors = [];
|
|
28
29
|
for (const name of selection) {
|
|
29
30
|
const sub = map.get(name);
|
|
30
|
-
if (!sub)
|
|
31
|
+
if (!sub) {
|
|
32
|
+
// Requested but not discoverable as an installed central subagent —
|
|
33
|
+
// e.g. its AGENT.md failed to parse. Say so instead of silently
|
|
34
|
+
// dropping it, which read as an unactionable doctor "hold" (PHNX-3187).
|
|
35
|
+
errors.push(`subagent '${name}': no parseable AGENT.md in ~/.agents/subagents`);
|
|
31
36
|
continue;
|
|
37
|
+
}
|
|
32
38
|
try {
|
|
33
39
|
target.write(dir, sub);
|
|
34
40
|
synced.push(sub.name);
|
|
@@ -39,9 +45,13 @@ function buildSubagentsWriter(agent) {
|
|
|
39
45
|
paths.push(entry.path);
|
|
40
46
|
}
|
|
41
47
|
}
|
|
42
|
-
catch {
|
|
48
|
+
catch (e) {
|
|
49
|
+
// A genuine fs/transform failure must surface with its reason, not
|
|
50
|
+
// vanish behind a bare `catch` (RUSH-2677 / PHNX-3187).
|
|
51
|
+
errors.push(`subagent '${sub.name}': ${e.message}`);
|
|
52
|
+
}
|
|
43
53
|
}
|
|
44
|
-
return { synced, paths };
|
|
54
|
+
return errors.length > 0 ? { synced, paths, errors } : { synced, paths };
|
|
45
55
|
},
|
|
46
56
|
};
|
|
47
57
|
}
|
package/dist/lib/state.d.ts
CHANGED
|
@@ -198,10 +198,13 @@ export declare function getMonitorsDir(): string;
|
|
|
198
198
|
* Path to built-in monitor definitions shipped in the system repo
|
|
199
199
|
* (`~/.agents/.system/monitors/`). Unioned under the user monitors dir by
|
|
200
200
|
* listMonitors()/readMonitor(): a monitor shipped here is available on every
|
|
201
|
-
* install, and a user monitor of the same name overrides it. A built-in
|
|
202
|
-
*
|
|
203
|
-
*
|
|
204
|
-
*
|
|
201
|
+
* install, and a user monitor of the same name overrides it. A built-in is
|
|
202
|
+
* enabled by default like every other system-layer resource — it runs on every
|
|
203
|
+
* install unless the user shadows it with `enabled: false` (via `agents monitors
|
|
204
|
+
* pause`, which materializes a user copy; writes never touch this pull-only
|
|
205
|
+
* mirror). A shared-input built-in still carries its own `device:` owner pin in
|
|
206
|
+
* the shipped YAML so exactly one box fires it (SING-9). The directory need not
|
|
207
|
+
* exist.
|
|
205
208
|
*/
|
|
206
209
|
export declare function getSystemMonitorsDir(): string;
|
|
207
210
|
/** Path to the durable per-monitor state-diff store + fire history
|
package/dist/lib/state.js
CHANGED
|
@@ -491,10 +491,13 @@ export function getMonitorsDir() { return process.env.AGENTS_MONITORS_DIR ?? MON
|
|
|
491
491
|
* Path to built-in monitor definitions shipped in the system repo
|
|
492
492
|
* (`~/.agents/.system/monitors/`). Unioned under the user monitors dir by
|
|
493
493
|
* listMonitors()/readMonitor(): a monitor shipped here is available on every
|
|
494
|
-
* install, and a user monitor of the same name overrides it. A built-in
|
|
495
|
-
*
|
|
496
|
-
*
|
|
497
|
-
*
|
|
494
|
+
* install, and a user monitor of the same name overrides it. A built-in is
|
|
495
|
+
* enabled by default like every other system-layer resource — it runs on every
|
|
496
|
+
* install unless the user shadows it with `enabled: false` (via `agents monitors
|
|
497
|
+
* pause`, which materializes a user copy; writes never touch this pull-only
|
|
498
|
+
* mirror). A shared-input built-in still carries its own `device:` owner pin in
|
|
499
|
+
* the shipped YAML so exactly one box fires it (SING-9). The directory need not
|
|
500
|
+
* exist.
|
|
498
501
|
*/
|
|
499
502
|
export function getSystemMonitorsDir() { return process.env.AGENTS_SYSTEM_MONITORS_DIR ?? SYSTEM_MONITORS_DIR; }
|
|
500
503
|
/** Path to the durable per-monitor state-diff store + fire history
|
package/dist/lib/subagents.js
CHANGED
|
@@ -23,7 +23,12 @@ export function parseSubagentFrontmatter(filePath) {
|
|
|
23
23
|
}
|
|
24
24
|
try {
|
|
25
25
|
const content = fs.readFileSync(filePath, 'utf-8');
|
|
26
|
-
|
|
26
|
+
// Split on CRLF or LF: git checks text files out with CRLF on Windows
|
|
27
|
+
// (core.autocrlf), so a plain split('\n') leaves a trailing '\r' on every
|
|
28
|
+
// line and the frontmatter fence `'---\r' !== '---'` never matches —
|
|
29
|
+
// silently dropping the subagent from discovery so it can never be
|
|
30
|
+
// installed or reconciled (PHNX-3187).
|
|
31
|
+
const lines = content.split(/\r?\n/);
|
|
27
32
|
// Check for YAML frontmatter
|
|
28
33
|
if (lines[0] === '---') {
|
|
29
34
|
const endIndex = lines.slice(1).findIndex((l) => l === '---');
|
|
@@ -52,7 +57,8 @@ export function getSubagentBody(filePath) {
|
|
|
52
57
|
return '';
|
|
53
58
|
}
|
|
54
59
|
const content = fs.readFileSync(filePath, 'utf-8');
|
|
55
|
-
|
|
60
|
+
// CRLF-robust for the same reason as parseSubagentFrontmatter (PHNX-3187).
|
|
61
|
+
const lines = content.split(/\r?\n/);
|
|
56
62
|
// Skip YAML frontmatter
|
|
57
63
|
if (lines[0] === '---') {
|
|
58
64
|
const endIndex = lines.slice(1).findIndex((l) => l === '---');
|
|
@@ -48,6 +48,16 @@ export interface PlacementOptions {
|
|
|
48
48
|
/** Human label of the requested agent (e.g. `claude@2.1.112`) for the
|
|
49
49
|
* fail-loud message. */
|
|
50
50
|
agentLabel?: string;
|
|
51
|
+
/**
|
|
52
|
+
* Normalized hosts boosted with `auto-launch.preferred` (set by
|
|
53
|
+
* `agents devices prefer <name>`). A preferred device ranks ahead of a
|
|
54
|
+
* non-preferred one among the eligible survivors — after the signed-in tier
|
|
55
|
+
* (a preferred box that can't run the agent is still no use) and before load,
|
|
56
|
+
* so an operator boost overrides load-based ordering without overriding hard
|
|
57
|
+
* health. Empty/undefined leaves the ranking unchanged. See
|
|
58
|
+
* {@link autoLaunchPreferredSet}.
|
|
59
|
+
*/
|
|
60
|
+
preferred?: ReadonlySet<string>;
|
|
51
61
|
}
|
|
52
62
|
/** Why a device was excluded from the viable set, for the fail-loud message. */
|
|
53
63
|
export type ExclusionReason = 'unreachable' | 'overloaded' | 'capped' | 'not-installed';
|
|
@@ -266,6 +266,14 @@ export function pickBestDevice(devices, roster, opts) {
|
|
|
266
266
|
const signedIn = (sa?.signedIn === true ? 0 : 1) - (sb?.signedIn === true ? 0 : 1);
|
|
267
267
|
if (signedIn !== 0)
|
|
268
268
|
return signedIn;
|
|
269
|
+
// (a2) operator-preferred device next — `agents devices prefer <name>`
|
|
270
|
+
// boosts a box above its load-equal peers, overriding load-based order.
|
|
271
|
+
const preferred = opts?.preferred;
|
|
272
|
+
if (preferred && preferred.size > 0) {
|
|
273
|
+
const pref = (preferred.has(a) ? 0 : 1) - (preferred.has(b) ? 0 : 1);
|
|
274
|
+
if (pref !== 0)
|
|
275
|
+
return pref;
|
|
276
|
+
}
|
|
269
277
|
// (b) lower load — coarse headroom tier, then raw load cost.
|
|
270
278
|
const tier = headroomTier(sa?.headroom) - headroomTier(sb?.headroom);
|
|
271
279
|
if (tier !== 0)
|
|
@@ -67,18 +67,46 @@ export interface TracesIndexShard {
|
|
|
67
67
|
owner: string;
|
|
68
68
|
stats: {
|
|
69
69
|
sessionsImported: number;
|
|
70
|
+
/**
|
|
71
|
+
* Median ACTIVE duration (span − idle gaps > 120s), ms — the meaningful figure
|
|
72
|
+
* (PHNX-3457). Same key/shape as before this change, so the fleet-aggregate
|
|
73
|
+
* worker (`worker-template.ts`) keeps weighted-averaging it unchanged; only its
|
|
74
|
+
* VALUE moved from raw span to active time. The raw span stays available per
|
|
75
|
+
* session on `SessionDetail.meta.spanMs`.
|
|
76
|
+
*/
|
|
70
77
|
medianMs: number;
|
|
78
|
+
/** p90 ACTIVE duration, ms. */
|
|
71
79
|
p90Ms: number;
|
|
80
|
+
/**
|
|
81
|
+
* SEGMENTED active-time stats (PHNX-3472). The blended `medianMs`/`p90Ms`
|
|
82
|
+
* above conflate one-shot interactive queries (63% of the corpus, ~15s
|
|
83
|
+
* median) with substantial agent runs (~15min median), so they headline
|
|
84
|
+
* neither. A session is an AGENT run when it made any tool call OR has more
|
|
85
|
+
* than 8 messages; otherwise INTERACTIVE. These segment the same active-time
|
|
86
|
+
* figure so the console can headline agent runs on their own axis. Each is
|
|
87
|
+
* computed only over sessions with a non-null duration.
|
|
88
|
+
*/
|
|
89
|
+
agentMedianMs: number;
|
|
90
|
+
/** p90 ACTIVE duration over AGENT sessions, ms. */
|
|
91
|
+
agentP90Ms: number;
|
|
92
|
+
/** Median ACTIVE duration over INTERACTIVE sessions, ms. */
|
|
93
|
+
interactiveMedianMs: number;
|
|
94
|
+
/** (sessions with a non-null duration) / (total sessions), 0..1 — coverage of the duration stats. */
|
|
95
|
+
measuredFraction: number;
|
|
72
96
|
needAttention: number;
|
|
73
97
|
toolErrorRate: number;
|
|
74
98
|
};
|
|
99
|
+
/**
|
|
100
|
+
* Sessions excluded from the eval corpus as internal utility plumbing (PHNX-3474):
|
|
101
|
+
* single-shot machine calls (no tool call AND ≤2 messages) or a known
|
|
102
|
+
* internal-prompt signature (title generation, watchdog, commit-message, factory
|
|
103
|
+
* worker). Every `stats` figure above, `topics` counts, and `needsAttention` are
|
|
104
|
+
* computed over the AGENT set ONLY — `sessionsImported` is the real agent count,
|
|
105
|
+
* not the raw row count. This is the number that was dropped.
|
|
106
|
+
*/
|
|
107
|
+
utilityCount: number;
|
|
75
108
|
needsAttention: IndexedSession[];
|
|
76
|
-
topics:
|
|
77
|
-
key: string;
|
|
78
|
-
label: string;
|
|
79
|
-
count: number;
|
|
80
|
-
group: TraceTopicGroup;
|
|
81
|
-
}>;
|
|
109
|
+
topics: TopicItem[];
|
|
82
110
|
failures: {
|
|
83
111
|
byToolError: Array<{
|
|
84
112
|
tool: string;
|
|
@@ -104,11 +132,56 @@ export interface IndexedSession {
|
|
|
104
132
|
title: string;
|
|
105
133
|
repo: string;
|
|
106
134
|
device: string;
|
|
135
|
+
/** The harness that produced the session (claude/codex/rush/grok/…). Same as `harness`. */
|
|
107
136
|
agent: string;
|
|
108
137
|
model: string;
|
|
138
|
+
/** Corpus classification (PHNX-3474). Always `'agent'` here — utility rows are excluded. */
|
|
139
|
+
kind: SessionKind;
|
|
109
140
|
severity: number;
|
|
110
141
|
flags: string[];
|
|
111
142
|
}
|
|
143
|
+
/**
|
|
144
|
+
* One example session under a topic tile — the shape the console drill-down consumes.
|
|
145
|
+
* Carries `kind` + `harness` (PHNX-3474) so the console can filter a tile's session
|
|
146
|
+
* list by corpus class and by harness. Refs on a topic tile are always `'agent'`
|
|
147
|
+
* (utility rows never reach a bucket), but the field is explicit for the consumer.
|
|
148
|
+
*/
|
|
149
|
+
export interface TopicSessionRef {
|
|
150
|
+
id: string;
|
|
151
|
+
title: string;
|
|
152
|
+
kind: SessionKind;
|
|
153
|
+
harness: string;
|
|
154
|
+
}
|
|
155
|
+
/**
|
|
156
|
+
* Corpus class of a session (PHNX-3474). `utility` is internal machine plumbing —
|
|
157
|
+
* a single-shot call with no tool use and ≤2 messages, or one whose topic/label
|
|
158
|
+
* matches a known internal-prompt signature (title generation, watchdog,
|
|
159
|
+
* commit-message writer, factory worker). Everything else is `agent`: real agent
|
|
160
|
+
* work the Evals console counts and scores. Utility rows are tagged, never deleted,
|
|
161
|
+
* and excluded from every index statistic.
|
|
162
|
+
*/
|
|
163
|
+
export type SessionKind = 'utility' | 'agent';
|
|
164
|
+
/**
|
|
165
|
+
* Classify a session as internal `utility` plumbing vs real `agent` work (PHNX-3474).
|
|
166
|
+
* `toolCallCount` is the AUTHORITATIVE per-session tool-call count from the loaded
|
|
167
|
+
* `tool_calls` rows (the row's own `tool_call_count` column is a fallback for a row
|
|
168
|
+
* whose calls weren't loaded). A session is `utility` when a known internal-prompt
|
|
169
|
+
* signature matches its topic/label, OR it made no tool call AND has ≤2 messages —
|
|
170
|
+
* the single-shot machine-call shape. Otherwise it is `agent`.
|
|
171
|
+
*/
|
|
172
|
+
export declare function classifySessionKind(row: Pick<SyncRow, 'topic' | 'label' | 'message_count' | 'tool_call_count'>, toolCallCount: number): SessionKind;
|
|
173
|
+
/**
|
|
174
|
+
* One topic bucket in the treemap. `sessions` carries up to {@link TOPIC_SESSION_CAP}
|
|
175
|
+
* example refs so the console can drill from the tile into its session list — a tile
|
|
176
|
+
* with no refs renders display-only (PHNX-3408). `count` stays the true total.
|
|
177
|
+
*/
|
|
178
|
+
export interface TopicItem {
|
|
179
|
+
key: string;
|
|
180
|
+
label: string;
|
|
181
|
+
count: number;
|
|
182
|
+
group: TraceTopicGroup;
|
|
183
|
+
sessions: TopicSessionRef[];
|
|
184
|
+
}
|
|
112
185
|
/** A row from `tool_calls`. `ordinal`/`timestamp` order calls within a session for computeInsights(). */
|
|
113
186
|
export interface ToolCallRow {
|
|
114
187
|
session_id: string;
|
|
@@ -129,6 +202,31 @@ export interface ToolCallRow {
|
|
|
129
202
|
error: string | null;
|
|
130
203
|
parse_error: string | null;
|
|
131
204
|
}
|
|
205
|
+
/**
|
|
206
|
+
* Active time for a session in the index shard: its recorded span minus every idle
|
|
207
|
+
* gap > 120s (PHNX-3457). The index build already holds the ordered `tool_calls`
|
|
208
|
+
* rows for the whole corpus, so idle is derived from them here — no transcript
|
|
209
|
+
* re-parse. A cursor sweeps the span from `sessionStartMs`: each stretch where the
|
|
210
|
+
* cursor sits idle for more than the threshold before the next call starts is
|
|
211
|
+
* subtracted, and idle is measured from a call's END (its own `end_timestamp` when
|
|
212
|
+
* known, else its start) so a call's own blocking duration is never mistaken for
|
|
213
|
+
* idle. Crucially the sweep also books the gaps at the two BOUNDARIES — before the
|
|
214
|
+
* first call and after the last call to the session end — so a session with a lone
|
|
215
|
+
* tool call that was then abandoned and resumed hours later (the case a
|
|
216
|
+
* between-calls-only measure missed entirely, leaving the whole 345h span counted
|
|
217
|
+
* as active) has that trailing idle stripped. A session end is `sessionStartMs +
|
|
218
|
+
* spanMs`, so the two agree by construction.
|
|
219
|
+
*
|
|
220
|
+
* Bounded to `[0, spanMs]`. A session with NO tool calls returns the full span
|
|
221
|
+
* unchanged rather than a fabricated zero: there is no tool-call evidence of idle
|
|
222
|
+
* either way, and treating a chat-only turn as 100% idle would be a worse error
|
|
223
|
+
* than leaving its span uncorrected. Where the full event stream IS available (a
|
|
224
|
+
* per-session `SessionDetail`), {@link activeMsFromTrajectory} is used instead —
|
|
225
|
+
* it sees message events this call-only approximation cannot, so the two are close
|
|
226
|
+
* but not identical by design (the corpus-scale index build cannot afford the
|
|
227
|
+
* per-session parse the detail view does).
|
|
228
|
+
*/
|
|
229
|
+
export declare function sessionActiveMs(spanMs: number, sessionCalls: ToolCallRow[], sessionStartMs: number): number;
|
|
132
230
|
/** Human description of a failed call, keyed by (tool, desc, cause) for grouping. Exported for computeInsights(). */
|
|
133
231
|
export declare function failureDescription(call: ToolCallRow, cause: TraceFailureCause): string;
|
|
134
232
|
/** Build the redacted rich console shard from indexed metadata and derived caches. */
|
|
@@ -138,7 +236,16 @@ export interface SessionDetail {
|
|
|
138
236
|
schema: 1;
|
|
139
237
|
id: string;
|
|
140
238
|
meta: {
|
|
239
|
+
/** Raw wall-clock span (last event − first event), idle time included. */
|
|
141
240
|
spanMs: number;
|
|
241
|
+
/**
|
|
242
|
+
* Active time: `spanMs` minus every idle gap > 120s (PHNX-3457). A session
|
|
243
|
+
* resumed hours later, or left idle mid-turn, inflates `spanMs` with wall-clock
|
|
244
|
+
* the agent did no work in — active time strips those gaps so a duration reads
|
|
245
|
+
* as effort, not calendar span. This is what the console's duration median/p90
|
|
246
|
+
* should trust; `spanMs` stays available as the raw figure.
|
|
247
|
+
*/
|
|
248
|
+
activeMs: number;
|
|
142
249
|
turns: number;
|
|
143
250
|
tools: number;
|
|
144
251
|
errorCount: number;
|