@phnx-labs/agents-cli 1.22.55 → 1.22.57

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/CHANGELOG.md +20 -0
  2. package/README.md +4 -4
  3. package/dist/bootstrap.js +3 -1
  4. package/dist/cli/command-registry.d.ts +0 -1
  5. package/dist/cli/command-registry.js +0 -3
  6. package/dist/commands/exec.js +1 -1
  7. package/dist/commands/hooks.js +4 -4
  8. package/dist/commands/insights.d.ts +7 -5
  9. package/dist/commands/insights.js +16 -9
  10. package/dist/commands/perf.d.ts +16 -7
  11. package/dist/commands/perf.js +29 -20
  12. package/dist/commands/rules.js +1 -1
  13. package/dist/commands/share.js +7 -6
  14. package/dist/commands/ssh.js +24 -14
  15. package/dist/commands/trash.d.ts +2 -2
  16. package/dist/commands/trash.js +2 -6
  17. package/dist/commands/versions.d.ts +2 -2
  18. package/dist/commands/versions.js +1 -10
  19. package/dist/commands/view.d.ts +2 -2
  20. package/dist/commands/view.js +7 -6
  21. package/dist/index.d.ts +1 -0
  22. package/dist/index.js +9 -0
  23. package/dist/lib/accounting/usage-ingest.d.ts +1 -0
  24. package/dist/lib/accounting/usage-ingest.js +75 -0
  25. package/dist/lib/accounting/usage-sync.d.ts +69 -0
  26. package/dist/lib/accounting/usage-sync.js +129 -0
  27. package/dist/lib/accounting/usage.d.ts +48 -2
  28. package/dist/lib/accounting/usage.js +72 -1
  29. package/dist/lib/agent-spec/agents.js +1 -1
  30. package/dist/lib/analytics/mix-commands.d.ts +8 -7
  31. package/dist/lib/analytics/mix-commands.js +50 -73
  32. package/dist/lib/daemon/daemon.js +5 -0
  33. package/dist/lib/daemon/runner.js +9 -8
  34. package/dist/lib/daemon/usage-sync-service.d.ts +21 -0
  35. package/dist/lib/daemon/usage-sync-service.js +36 -0
  36. package/dist/lib/daemon-services.d.ts +1 -1
  37. package/dist/lib/daemon-services.js +5 -0
  38. package/dist/lib/device-config.d.ts +17 -6
  39. package/dist/lib/device-config.js +25 -11
  40. package/dist/lib/devices/pool.d.ts +4 -3
  41. package/dist/lib/devices/pool.js +13 -5
  42. package/dist/lib/exec.d.ts +6 -41
  43. package/dist/lib/exec.js +6 -41
  44. package/dist/lib/git.d.ts +13 -1
  45. package/dist/lib/git.js +36 -7
  46. package/dist/lib/harness/adapter.d.ts +7 -7
  47. package/dist/lib/harness/adapters/claude.js +3 -2
  48. package/dist/lib/hosts/remote-cmd.d.ts +9 -0
  49. package/dist/lib/hosts/remote-cmd.js +22 -0
  50. package/dist/lib/overdue.js +7 -37
  51. package/dist/lib/perf/db.d.ts +1 -1
  52. package/dist/lib/perf/db.js +1 -1
  53. package/dist/lib/scheduler.d.ts +21 -2
  54. package/dist/lib/scheduler.js +28 -5
  55. package/dist/lib/scheduling/routines.d.ts +21 -0
  56. package/dist/lib/scheduling/routines.js +56 -0
  57. package/dist/lib/session/active.d.ts +3 -31
  58. package/dist/lib/session/active.js +8 -68
  59. package/dist/lib/session/db.d.ts +4 -35
  60. package/dist/lib/session/db.js +4 -35
  61. package/dist/lib/session/discover.d.ts +6 -58
  62. package/dist/lib/session/discover.js +5 -43
  63. package/dist/lib/session/parse.d.ts +1 -19
  64. package/dist/lib/session/parse.js +2 -15
  65. package/dist/lib/share/provision.d.ts +3 -2
  66. package/dist/lib/share/provision.js +9 -4
  67. package/dist/lib/share/worker-template.d.ts +24 -1
  68. package/dist/lib/share/worker-template.js +65 -84
  69. package/dist/lib/startup/command-registry.d.ts +8 -2
  70. package/dist/lib/startup/command-registry.js +12 -4
  71. package/package.json +1 -1
package/dist/lib/git.js CHANGED
@@ -498,7 +498,7 @@ export async function getCurrentBranch(repoPath) {
498
498
  * repo" as a requested source before adopting it.
499
499
  */
500
500
  export function canonicalGitRemote(url) {
501
- return url
501
+ const canonical = url
502
502
  .trim()
503
503
  .replace(/\/+$/, '') // trailing slashes first, so a trailing-slash-after-.git still strips
504
504
  .replace(/\.git$/i, '')
@@ -506,6 +506,36 @@ export function canonicalGitRemote(url) {
506
506
  .replace(/^[^@/]+@/, '') // strip user@ (git@, ssh user)
507
507
  .replace(':', '/') // scp-style host:owner/repo → host/owner/repo (first colon only)
508
508
  .toLowerCase();
509
+ // Fold a renamed repo's old name onto its new one so both compare equal
510
+ // everywhere (see RENAMED_REMOTE_ALIASES).
511
+ return RENAMED_REMOTE_ALIASES[canonical] ?? canonical;
512
+ }
513
+ /**
514
+ * Git remotes that denote the SAME repository under an old and a new name,
515
+ * keyed by canonical `host/owner/repo`. `phnx-labs/.agents-system` was renamed
516
+ * to `phnx-labs/.agents` on GitHub (PHNX-3394); {@link DEFAULT_SYSTEM_REPO}
517
+ * still points at the pre-rename slug (GitHub's own redirect makes that
518
+ * resolve fine), so folding the new name onto it here means both compare equal
519
+ * everywhere remotes are compared: {@link sameGitRemote} (repo adoption),
520
+ * {@link isSystemRepoRemote} (the system-origin check), and the
521
+ * DotAgents-layer classifier in state.ts.
522
+ */
523
+ const RENAMED_REMOTE_ALIASES = {
524
+ 'github.com/phnx-labs/.agents': 'github.com/phnx-labs/.agents-system',
525
+ };
526
+ /**
527
+ * True when a git remote URL (any transport form: ssh, https, scp-style) points
528
+ * at the system DotAgents repo — {@link DEFAULT_SYSTEM_REPO}'s current slug OR
529
+ * its `phnx-labs/.agents` rename target (PHNX-3394), which
530
+ * {@link canonicalGitRemote} folds onto it via {@link RENAMED_REMOTE_ALIASES}.
531
+ * Pure string check with no git spawn, so it is unit-testable off a live
532
+ * checkout; {@link isSystemRepoOrigin} reads a dir's origin and delegates here.
533
+ */
534
+ export function isSystemRepoRemote(remote) {
535
+ if (!remote)
536
+ return false;
537
+ const c = canonicalGitRemote(remote);
538
+ return c === canonicalGitRemote(`https://github.com/${systemRepoSlug(DEFAULT_SYSTEM_REPO)}`);
509
539
  }
510
540
  /** True when two git remote URLs point at the same repo across transport forms. */
511
541
  export function sameGitRemote(a, b) {
@@ -1062,18 +1092,17 @@ export async function adoptUserRepoIfNeeded(dir, opts = {}) {
1062
1092
  return adoptRepoInPlace(dir, url);
1063
1093
  }
1064
1094
  /**
1065
- * Check if the repo's origin points to the system repo.
1095
+ * Check if the repo's origin points to the system repo — `phnx-labs/.agents-system`
1096
+ * or its GitHub rename target `phnx-labs/.agents` (PHNX-3394), across any
1097
+ * transport form. Reads the dir's origin and delegates the match to the pure
1098
+ * {@link isSystemRepoRemote}.
1066
1099
  */
1067
1100
  export async function isSystemRepoOrigin(dir) {
1068
1101
  try {
1069
1102
  const git = simpleGit(dir);
1070
1103
  const remotes = await git.getRemotes(true);
1071
1104
  const origin = remotes.find(r => r.name === 'origin');
1072
- if (!origin?.refs?.fetch)
1073
- return false;
1074
- const url = origin.refs.fetch.toLowerCase();
1075
- const currentSlug = systemRepoSlug(DEFAULT_SYSTEM_REPO).toLowerCase();
1076
- return url.includes(currentSlug);
1105
+ return isSystemRepoRemote(origin?.refs?.fetch);
1077
1106
  }
1078
1107
  catch {
1079
1108
  /* not a git repo or no remotes */
@@ -44,13 +44,13 @@ export interface ExecConfigEnvCtx {
44
44
  /** resolveInteractive(options) — computed once by the caller. */
45
45
  interactive: boolean;
46
46
  /**
47
- * The role marked on THIS machine (worker | personal | undefined), resolved
48
- * once by the caller from selfConfiguredDeviceRole(). A `personal` device is
49
- * the user's own interactive box: it holds a real per-version login and the
50
- * credential decision MUST defer to it for EVERY run — interactive OR headless
51
- * — never the worker-only setup-token (RUSH-2395). Injected as a plain value
52
- * (not imported) to keep the adapter import-leaf. Absent/undefined is treated
53
- * as non-personal (worker-equivalent).
47
+ * The role marked on THIS machine (worker | personal | desktop | undefined),
48
+ * resolved once by the caller from selfConfiguredDeviceRole(). A headed device
49
+ * (`personal` or `desktop` — see isHeadedDeviceRole) holds a real per-version
50
+ * login and the credential decision MUST defer to it for EVERY run — interactive
51
+ * OR headless — never the worker-only setup-token (RUSH-2395). Injected as a
52
+ * plain value (not imported) to keep the adapter import-leaf. Absent/undefined
53
+ * is treated as non-headed (worker-equivalent).
54
54
  */
55
55
  deviceRole?: ConfiguredDeviceRole;
56
56
  /**
@@ -1,5 +1,6 @@
1
1
  import * as path from 'path';
2
2
  import { stripForeignConfigDir } from '../adapter.js';
3
+ import { isHeadedDeviceRole } from '../../device-config.js';
3
4
  export const claudeAdapter = {
4
5
  id: 'claude',
5
6
  applyExecConfigEnv(result, ctx) {
@@ -47,8 +48,8 @@ export const claudeAdapter = {
47
48
  // path — agents.ts `isClaudeCredentialFileBlank`), so this path defers to
48
49
  // Claude Code, which reads its own ACL-trusted login item without a prompt and
49
50
  // asks a present human to log in only if the login is missing.
50
- const personalDevice = ctx.deviceRole === 'personal';
51
- if (ctx.interactive || personalDevice) {
51
+ const headedDevice = isHeadedDeviceRole(ctx.deviceRole);
52
+ if (ctx.interactive || headedDevice) {
52
53
  // Drop an INHERITED copy of OUR OWN setup-token: a launch from inside a
53
54
  // headless agent's shell inherits that agent's injected value via
54
55
  // sanitizeProcessEnv(process.env) and would keep authenticating as it,
@@ -177,6 +177,15 @@ export declare function buildWindowsAgentsCommand(cmd: WindowsAgentsCommand): st
177
177
  * (Credential Manager, or the headless file store when there's no logon
178
178
  * session), matching a local `agents secrets import`.
179
179
  */
180
+ /**
181
+ * Run `agents <args> --from <tmp>` on a Windows peer, feeding the ssh-piped stdin
182
+ * through a temp file — the `agents.ps1` shim does not forward piped stdin to the
183
+ * node process, so a verb that reads stdin must be handed a file instead. The
184
+ * generic sibling of {@link buildWindowsStdinImportCommand}; the receiving verb
185
+ * MUST accept `--from <path>` (see `usage-ingest.ts`). The temp file is removed in
186
+ * a `finally` so a throw mid-run never leaves the payload behind.
187
+ */
188
+ export declare function buildWindowsStdinAgentsCommand(args: string[]): string;
180
189
  export declare function buildWindowsStdinImportCommand(bundle: string, opts?: {
181
190
  force?: boolean;
182
191
  policyNever?: boolean;
@@ -323,6 +323,28 @@ export function buildWindowsAgentsCommand(cmd) {
323
323
  * (Credential Manager, or the headless file store when there's no logon
324
324
  * session), matching a local `agents secrets import`.
325
325
  */
326
+ /**
327
+ * Run `agents <args> --from <tmp>` on a Windows peer, feeding the ssh-piped stdin
328
+ * through a temp file — the `agents.ps1` shim does not forward piped stdin to the
329
+ * node process, so a verb that reads stdin must be handed a file instead. The
330
+ * generic sibling of {@link buildWindowsStdinImportCommand}; the receiving verb
331
+ * MUST accept `--from <path>` (see `usage-ingest.ts`). The temp file is removed in
332
+ * a `finally` so a throw mid-run never leaves the payload behind.
333
+ */
334
+ export function buildWindowsStdinAgentsCommand(args) {
335
+ const forwarded = args.map(powershellQuote).join(' ');
336
+ const script = [
337
+ POWERSHELL_PROGRESS_SILENCE,
338
+ '$in = [Console]::In.ReadToEnd()',
339
+ '$tmp = $null',
340
+ `try { $tmp = [System.IO.Path]::GetTempFileName(); [System.IO.File]::WriteAllText($tmp, $in); ` +
341
+ `& agents ${forwarded} --from $tmp; $code = $LASTEXITCODE } ` +
342
+ `finally { if ($tmp) { Remove-Item -LiteralPath $tmp -Force -ErrorAction SilentlyContinue } }`,
343
+ 'if ($null -eq $code) { $code = 1 }',
344
+ 'exit $code',
345
+ ].join('; ');
346
+ return `powershell -NoProfile -EncodedCommand ${encodePowershell(script)}`;
347
+ }
326
348
  export function buildWindowsStdinImportCommand(bundle, opts = {}) {
327
349
  const force = opts.force ? ' --force' : '';
328
350
  const policy = opts.policyNever ? ' --policy never --i-understand' : '';
@@ -13,48 +13,18 @@
13
13
  */
14
14
  import * as fs from 'fs';
15
15
  import { Cron } from 'croner';
16
- import { listJobs, getLatestRun, resolveJobFilePath, isPastEndAt, isOneShotRoutine, jobRunsOnThisDevice } from './scheduling/routines.js';
16
+ import { alignedSlotForFire, listJobs, getLatestRun, resolveJobFilePath, isPastEndAt, isOneShotRoutine, jobRunsOnThisDevice } from './scheduling/routines.js';
17
17
  import { notifyDesktop } from './menubar/notify-desktop.js';
18
18
  // Tolerance between "expected fire" and "recorded run start" — accounts for
19
19
  // the small gap between the cron tick and when the runner writes meta.json.
20
20
  const GRACE_MS = 60_000;
21
- const DAY_MS = 24 * 60 * 60 * 1000;
22
- /**
23
- * Lookback windows, narrowest first. A fixed one-week window silently blinded
24
- * detection to any cron whose gap exceeds it: `0 9 1,13,25 * *` has 12-day gaps,
25
- * so `nextRun(now - 7d)` jumped past `now`, the walk returned null, and the
26
- * routine was never flagged overdue on any device — no missed record, no
27
- * catch-up, permanently. Monthly, quarterly and annual routines were all in that
28
- * class.
29
- *
30
- * A wider window is only tried when the narrower one found nothing, so a dense
31
- * schedule (every minute, hourly, daily) never walks more than a week of
32
- * occurrences. A sparse schedule has few occurrences to walk by definition.
33
- */
34
- const LOOKBACK_WINDOWS_MS = [7 * DAY_MS, 32 * DAY_MS, 93 * DAY_MS, 400 * DAY_MS];
35
- /** Compute the most recent fire of `pattern` at or before `now`. Croner's
36
- * `previousRun()` returns the cron instance's own last fire, which is null
37
- * on a freshly-constructed instance — so we walk `nextRun(cursor)` forward
38
- * from a week ago and keep the last fire still ≤ now. */
21
+ /** Compute the most recent fire of `pattern` at or before `now`. Delegates to
22
+ * {@link alignedSlotForFire} so overdue detection (`missedRunId`) and the live
23
+ * forward-timer dispatch (`slotRunId`) key on the SAME occurrence identity — a
24
+ * missed fire and its live twin for one UTC slot then collide by construction
25
+ * (SING-15). */
39
26
  function previousExpectedFire(cron, now) {
40
- for (const window of LOOKBACK_WINDOWS_MS) {
41
- let cursor = new Date(now.getTime() - window);
42
- let last = null;
43
- // Cap iterations: an every-minute schedule yields ≤ 10080 steps over a week;
44
- // 20k is a paranoia bound against pathological patterns. Only a schedule
45
- // that found nothing in the narrower window reaches a wider one, and such a
46
- // schedule is sparse, so the cap is never the binding constraint.
47
- for (let i = 0; i < 20000; i++) {
48
- const next = cron.nextRun(cursor);
49
- if (!next || next.getTime() > now.getTime())
50
- break;
51
- last = next;
52
- cursor = next;
53
- }
54
- if (last)
55
- return last;
56
- }
57
- return null;
27
+ return alignedSlotForFire(cron, now);
58
28
  }
59
29
  /**
60
30
  * When a routine started existing, and therefore the earliest fire it can
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * Disposable performance warehouse — SQLite under ~/.agents/.cache/perf/.
3
3
  *
4
- * Opened only by `agents perf` / `hooks profile` (read path). Writers use
4
+ * Opened only by `agents insights perf` / `hooks profile` (read path). Writers use
5
5
  * {@link recordSample} in `./spool.ts` (NDJSON, no SQLite).
6
6
  */
7
7
  import Database from '../sqlite.js';
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * Disposable performance warehouse — SQLite under ~/.agents/.cache/perf/.
3
3
  *
4
- * Opened only by `agents perf` / `hooks profile` (read path). Writers use
4
+ * Opened only by `agents insights perf` / `hooks profile` (read path). Writers use
5
5
  * {@link recordSample} in `./spool.ts` (NDJSON, no SQLite).
6
6
  */
7
7
  import * as fs from 'fs';
@@ -5,13 +5,32 @@
5
5
  * process creates a single JobScheduler instance that loads enabled jobs
6
6
  * on startup and reloads them on SIGHUP.
7
7
  */
8
+ import { Cron } from 'croner';
8
9
  import type { JobConfig } from './scheduling/routines.js';
9
10
  /** How a fire was triggered, carrying the scheduler's intended UTC slot time. */
10
11
  export interface TriggerContext {
11
- /** The cron slot this callback fires for (croner `currentRun()`), for the
12
- * single-fire claim keyed on (routine, scheduledFor). */
12
+ /** The ALIGNED cron slot this callback fires for, for the single-fire claim
13
+ * keyed on (routine, scheduledFor). Derived by {@link fireSlot} — NOT croner's
14
+ * raw `currentRun()`, which carries wall-clock jitter. */
13
15
  scheduledFor?: Date;
14
16
  }
17
+ /**
18
+ * The aligned occurrence boundary a fire callback belongs to — the value the
19
+ * single-fire `(routine, scheduledFor)` claim keys on.
20
+ *
21
+ * croner's `currentRun()` inside a fire callback is the JITTERED wall-clock
22
+ * trigger instant (it carries milliseconds — verified against croner 10.x), not
23
+ * the aligned schedule boundary. Keying `slotRunId` on it directly minted a
24
+ * distinct run id per delivery, so two callbacks for one occurrence each claimed
25
+ * a different run dir and both launched, and a live fire never collided with its
26
+ * catch-up twin (`missedRunId`, which keys on the aligned boundary). Flooring the
27
+ * fire to its schedule boundary via {@link alignedSlotForFire} makes the claim a
28
+ * structural claim on the occurrence identity (SING-15). Always returns a
29
+ * concrete Date — `currentRun()` falls back to now, and an unresolvable boundary
30
+ * falls back to the fire instant — so the forward path never dispatches without a
31
+ * durable slot key.
32
+ */
33
+ export declare function fireSlot(cron: Cron): Date;
15
34
  /** In-memory cron scheduler that triggers a callback when jobs fire. */
16
35
  export declare class JobScheduler {
17
36
  private jobs;
@@ -6,7 +6,27 @@
6
6
  * on startup and reloads them on SIGHUP.
7
7
  */
8
8
  import { Cron } from 'croner';
9
- import { listJobs, deleteJob, isPastEndAt, isPastOneShotRoutine, isOneShotRoutine, setJobEnabled, shouldPurgeCompletedOneShotRoutine, jobRunsOnThisDevice, hasAmbiguousDevicePin, routineOwnerDevice, } from './scheduling/routines.js';
9
+ import { alignedSlotForFire, listJobs, deleteJob, isPastEndAt, isPastOneShotRoutine, isOneShotRoutine, setJobEnabled, shouldPurgeCompletedOneShotRoutine, jobRunsOnThisDevice, hasAmbiguousDevicePin, routineOwnerDevice, } from './scheduling/routines.js';
10
+ /**
11
+ * The aligned occurrence boundary a fire callback belongs to — the value the
12
+ * single-fire `(routine, scheduledFor)` claim keys on.
13
+ *
14
+ * croner's `currentRun()` inside a fire callback is the JITTERED wall-clock
15
+ * trigger instant (it carries milliseconds — verified against croner 10.x), not
16
+ * the aligned schedule boundary. Keying `slotRunId` on it directly minted a
17
+ * distinct run id per delivery, so two callbacks for one occurrence each claimed
18
+ * a different run dir and both launched, and a live fire never collided with its
19
+ * catch-up twin (`missedRunId`, which keys on the aligned boundary). Flooring the
20
+ * fire to its schedule boundary via {@link alignedSlotForFire} makes the claim a
21
+ * structural claim on the occurrence identity (SING-15). Always returns a
22
+ * concrete Date — `currentRun()` falls back to now, and an unresolvable boundary
23
+ * falls back to the fire instant — so the forward path never dispatches without a
24
+ * durable slot key.
25
+ */
26
+ export function fireSlot(cron) {
27
+ const fire = cron.currentRun() ?? new Date();
28
+ return alignedSlotForFire(cron, fire) ?? fire;
29
+ }
10
30
  /** In-memory cron scheduler that triggers a callback when jobs fire. */
11
31
  export class JobScheduler {
12
32
  jobs = new Map();
@@ -75,10 +95,13 @@ export class JobScheduler {
75
95
  return;
76
96
  }
77
97
  try {
78
- // croner hands the callback its own Cron instance; `currentRun()` is the
79
- // UTC time THIS invocation was scheduled for — the single-fire slot key.
80
- // A duplicate delivery for the same slot resolves to one run downstream.
81
- await this.onTrigger(config, { scheduledFor: self.currentRun() ?? undefined });
98
+ // scheduledFor is the ALIGNED occurrence boundary (fireSlot), not croner's
99
+ // jittered currentRun(): the single-fire claim keys on (routine,
100
+ // scheduledFor), so the key must be the occurrence identity or a live fire
101
+ // and its catch-up twin (missedRunId) won't collide and two deliveries of
102
+ // one slot each mint a distinct id. fireSlot always returns a Date, so the
103
+ // forward path always carries a durable claim (SING-15).
104
+ await this.onTrigger(config, { scheduledFor: fireSlot(self) });
82
105
  }
83
106
  catch (err) {
84
107
  console.error(`Job '${config.name}' failed:`, err.message);
@@ -6,6 +6,7 @@
6
6
  * run metadata persistence, prompt variable expansion, and one-shot "at" time
7
7
  * scheduling.
8
8
  */
9
+ import { Cron } from 'croner';
9
10
  import { type ResolvedExecutionContext, type ProjectResolution, type PlacementMode, type RoutineKind, type ContextFsProbe } from '../routine-context.js';
10
11
  import type { AgentId, RunStrategy } from '../types.js';
11
12
  import type { LoopConfig } from '../loop.js';
@@ -837,6 +838,26 @@ export declare function getRunDir(jobName: string, runId: string): string;
837
838
  * for the same UTC slot are one record.
838
839
  */
839
840
  export declare function slotRunId(scheduledFor: Date | string): string;
841
+ /**
842
+ * The aligned schedule boundary a fire belongs to: the most recent occurrence of
843
+ * `cron` at or before `at`.
844
+ *
845
+ * This is the occurrence IDENTITY that {@link slotRunId} (forward dispatch) and
846
+ * `missedRunId` (catchup.ts) must both key on. croner's `currentRun()` inside a
847
+ * fire callback is the JITTERED wall-clock trigger instant (it carries
848
+ * milliseconds — verified), not the aligned boundary, so keying `slotRunId`
849
+ * directly on it produced a distinct id per delivery: two callbacks for one
850
+ * occurrence each claimed a different run dir and both launched, and a live fire
851
+ * never collided with its catch-up twin (which keys on the aligned
852
+ * `previousExpectedFire`). Flooring both to this boundary is what makes the
853
+ * single-fire claim a structural claim on `(routine, scheduledFor)` (SING-15).
854
+ *
855
+ * croner's `previousRun()` takes no argument and returns null on a freshly
856
+ * constructed instance, so we walk `nextRun(cursor)` forward from a lookback
857
+ * window and keep the last fire still ≤ `at` — the same derivation catchup's
858
+ * overdue detection has always used.
859
+ */
860
+ export declare function alignedSlotForFire(cron: Cron, at: Date): Date | null;
840
861
  /**
841
862
  * Atomically CLAIM a run directory. Returns true on a successful claim, false
842
863
  * when the directory already exists (another caller — even in a separate process
@@ -1518,6 +1518,62 @@ export function slotRunId(scheduledFor) {
1518
1518
  const iso = typeof scheduledFor === 'string' ? scheduledFor : scheduledFor.toISOString();
1519
1519
  return iso.replace(/[:.]/g, '-');
1520
1520
  }
1521
+ /**
1522
+ * Lookback windows for {@link alignedSlotForFire}, narrowest first. A wider
1523
+ * window is tried ONLY when the narrower one found no fire, so:
1524
+ * - a dense schedule (every-minute) resolves in the 1-hour window — ~60 steps,
1525
+ * not ~10080 — which matters because the forward-timer path now runs this on
1526
+ * every fire (a live fire is milliseconds past its boundary, so the narrowest
1527
+ * window always contains it);
1528
+ * - a sparse schedule (`0 9 1,13,25 * *` has 12-day gaps; monthly/quarterly/
1529
+ * annual) still resolves, because a fixed short window silently blinded
1530
+ * overdue detection to any cron whose gap exceeded it.
1531
+ * A narrower window can only ever find the true most-recent fire ≤ `at` or
1532
+ * nothing (never a wrong boundary), so prepending the cheap windows is
1533
+ * behavior-preserving for the sparse-schedule overdue path.
1534
+ */
1535
+ const HOUR_MS = 60 * 60 * 1000;
1536
+ const DAY_MS = 24 * HOUR_MS;
1537
+ const SLOT_LOOKBACK_WINDOWS_MS = [HOUR_MS, DAY_MS, 7 * DAY_MS, 32 * DAY_MS, 93 * DAY_MS, 400 * DAY_MS];
1538
+ /**
1539
+ * The aligned schedule boundary a fire belongs to: the most recent occurrence of
1540
+ * `cron` at or before `at`.
1541
+ *
1542
+ * This is the occurrence IDENTITY that {@link slotRunId} (forward dispatch) and
1543
+ * `missedRunId` (catchup.ts) must both key on. croner's `currentRun()` inside a
1544
+ * fire callback is the JITTERED wall-clock trigger instant (it carries
1545
+ * milliseconds — verified), not the aligned boundary, so keying `slotRunId`
1546
+ * directly on it produced a distinct id per delivery: two callbacks for one
1547
+ * occurrence each claimed a different run dir and both launched, and a live fire
1548
+ * never collided with its catch-up twin (which keys on the aligned
1549
+ * `previousExpectedFire`). Flooring both to this boundary is what makes the
1550
+ * single-fire claim a structural claim on `(routine, scheduledFor)` (SING-15).
1551
+ *
1552
+ * croner's `previousRun()` takes no argument and returns null on a freshly
1553
+ * constructed instance, so we walk `nextRun(cursor)` forward from a lookback
1554
+ * window and keep the last fire still ≤ `at` — the same derivation catchup's
1555
+ * overdue detection has always used.
1556
+ */
1557
+ export function alignedSlotForFire(cron, at) {
1558
+ for (const window of SLOT_LOOKBACK_WINDOWS_MS) {
1559
+ let cursor = new Date(at.getTime() - window);
1560
+ let last = null;
1561
+ // Cap iterations: an every-minute schedule yields ≤ 10080 steps over a week;
1562
+ // 20k is a paranoia bound against pathological patterns. Only a schedule that
1563
+ // found nothing in the narrower window reaches a wider one, and such a
1564
+ // schedule is sparse, so the cap is never the binding constraint.
1565
+ for (let i = 0; i < 20000; i++) {
1566
+ const next = cron.nextRun(cursor);
1567
+ if (!next || next.getTime() > at.getTime())
1568
+ break;
1569
+ last = next;
1570
+ cursor = next;
1571
+ }
1572
+ if (last)
1573
+ return last;
1574
+ }
1575
+ return null;
1576
+ }
1521
1577
  /**
1522
1578
  * Atomically CLAIM a run directory. Returns true on a successful claim, false
1523
1579
  * when the directory already exists (another caller — even in a separate process
@@ -8,14 +8,7 @@ import { type SessionProvenance } from './provenance.js';
8
8
  import { type DeviceRegistry } from '../devices/registry.js';
9
9
  import { type Presence } from './detached.js';
10
10
  import { type HostLink } from './host-link.js';
11
- /**
12
- * The owner (actor id) to show for a session in `--active`. Prefers the actor
13
- * recorded on the live-attribution source (the pid registry / teammate record),
14
- * but falls back to the durable per-session actor sidecar — written at spawn and,
15
- * unlike the pid entry, NOT overwritten by the SessionStart hook's own by-pid
16
- * write. Without this fallback a real `agents run` shows no owner whenever the
17
- * hook's actor-less entry wins the by-pid file (RUSH-2018 fix).
18
- */
11
+ /** Prefer live actor attribution; the durable sidecar survives actor-less hook rewrites. */
19
12
  export declare function resolveOwner(pidActor: string | null | undefined, sessionId: string | undefined): string | undefined;
20
13
  /**
21
14
  * Per-PID `lsof` probes run bounded and staggered rather than as one parallel
@@ -830,18 +823,7 @@ export declare function sessionProcessIsLocal(s: Pick<ActiveSession, 'machine' |
830
823
  */
831
824
  export declare function sessionProcessHost(s: Pick<ActiveSession, 'machine' | 'offloadedFrom'>, self: string): string | undefined;
832
825
  export declare function foldHostLink(rows: ActiveSession[]): void;
833
- /**
834
- * The recap ladder (RUSH-3011): compute a row's shown {@link ActiveSession.title}
835
- * + {@link RecapSource} from the best available source, plus the cleaned first
836
- * prompt (`userPromptClean`/`userPromptKind`) and the `lastAgentLine`. Pure over
837
- * one row; exported for tests and folded in by {@link foldRecap}.
838
- *
839
- * Ladder, best-first: a `/rename`/harness `label` → the last assistant line →
840
- * the first-prompt topic. The `last` rung is agent-derived, so a session that
841
- * produced work stops showing its stale first prompt as the title. (`topic` is
842
- * the row's already-extracted first line, so image detection here is
843
- * path-based; a pure-attachment turn with no first-line text stays on `prompt`.)
844
- */
826
+ /** Labels win, then the last assistant line, then the first-prompt topic. */
845
827
  export declare function deriveSessionRecap(row: Pick<ActiveSession, 'label' | 'topic' | 'tail'>): {
846
828
  title?: string;
847
829
  recapSource?: RecapSource;
@@ -851,17 +833,7 @@ export declare function deriveSessionRecap(row: Pick<ActiveSession, 'label' | 't
851
833
  };
852
834
  /** Fold the recap ladder onto every row (see {@link deriveSessionRecap}). */
853
835
  export declare function foldRecap(rows: ActiveSession[]): void;
854
- /**
855
- * True when a crash-leaked orphan is genuinely DEAD and should be reaped from the
856
- * reconnectable set rather than shown as resumable forever (RUSH-3011 / issue #3b).
857
- *
858
- * The gate is `abandoned` (no transcript write in {@link ABANDONED_STALE_MS}) AND
859
- * a dead pid — exactly "past the stale threshold whose pid is gone". A live pid
860
- * (an idle-but-unfinished session, the highest-risk state) is NEVER reaped, and
861
- * neither is a recently-`closed`/`crashed` session that just exited (still
862
- * resumable). `pidAlive` absent (a cloud row or an older peer that can't prove
863
- * death) also stays un-reaped — reaping is fail-safe, never a guess.
864
- */
836
+ /** Reap only stale sessions with proven-dead pids; unknown or live processes remain recoverable. */
865
837
  export declare function isReapableOrphan(row: Pick<ActiveSession, 'status' | 'pidAlive'>): boolean;
866
838
  /**
867
839
  * Resolve each teams row's `orchestratorLabel` from the orchestrator's own row,
@@ -47,14 +47,7 @@ import { linearIssueUrl } from './linear.js';
47
47
  import { viewingInLabel } from './viewing-in.js';
48
48
  import { claudeProjectDirName } from '../project-key.js';
49
49
  const execFileAsync = promisify(execFile);
50
- /**
51
- * The owner (actor id) to show for a session in `--active`. Prefers the actor
52
- * recorded on the live-attribution source (the pid registry / teammate record),
53
- * but falls back to the durable per-session actor sidecar — written at spawn and,
54
- * unlike the pid entry, NOT overwritten by the SessionStart hook's own by-pid
55
- * write. Without this fallback a real `agents run` shows no owner whenever the
56
- * hook's actor-less entry wins the by-pid file (RUSH-2018 fix).
57
- */
50
+ /** Prefer live actor attribution; the durable sidecar survives actor-less hook rewrites. */
58
51
  export function resolveOwner(pidActor, sessionId) {
59
52
  return pidActor ?? (sessionId ? readSessionActorRecord(sessionId)?.actor : undefined) ?? undefined;
60
53
  }
@@ -476,18 +469,8 @@ export function isPidAlive(pid, startedAtMs) {
476
469
  return true;
477
470
  }
478
471
  /**
479
- * Read the live-terminals registry, dedupe by sessionId.
480
- *
481
- * A pid-alive entry is a live session. A pid-DEAD entry is normally noise — a
482
- * terminal that closed a moment ago, before its window republished — and is
483
- * dropped. But a dead pid whose owning window ALSO stopped republishing is the
484
- * signature of a crash: the window went down hard and never ran the teardown that
485
- * would have removed this entry. Those are KEPT, so the session reaches the
486
- * listing at all — it used to vanish outright, a VS Code crash simply erasing its
487
- * agents from `--active`. Such a row arrives as `closed` (dead pid) carrying the
488
- * stale `windowHeartbeatMs`, which is what {@link foldHostLink} promotes to
489
- * `crashed`. `pidDead` is local to the dedupe below: a live entry must win a dead
490
- * one for the same session.
472
+ * Keep dead entries only when their window heartbeat also stopped, proving a crash;
473
+ * a live duplicate always wins.
491
474
  */
492
475
  function readLiveTerminals() {
493
476
  let raw;
@@ -541,20 +524,8 @@ function readLiveTerminals() {
541
524
  const CLAUDE_SESSION_FILE_CACHE_MAX = 256;
542
525
  const claudeSessionFileCache = new Map();
543
526
  /**
544
- * Locate the active Claude session file for a process. If we know the session
545
- * UUID (from terminal env or team parent), prefer the exact match. Otherwise
546
- * fall back to the most-recent-mtime .jsonl in the project's folder.
547
- *
548
- * Searches EVERY version-home project root, not just the live `~/.claude`
549
- * symlink. `~/.claude` points at the currently-installed agent version; a
550
- * session launched under an EARLIER version keeps its transcript under that
551
- * version's home (`…/.history/versions/claude/<ver>/home/.claude/projects/`).
552
- * Resolving only `~/.claude/projects` meant that the instant a newer version
553
- * was installed, every still-running older-version session lost its transcript
554
- * here — no `sessionFile`, so no start/activity time, so `agents sessions`
555
- * rendered it `unknown` and the watchdog skipped it as "no activity timestamp".
556
- * `getAgentSessionDirs('claude','projects')` is the same version-aware enumerator
557
- * the rest of the CLI uses, so this stays in lockstep with discovery.
527
+ * Search every version home because the live ~/.claude symlink moves after upgrades
528
+ * while older running sessions keep writing to their original home.
558
529
  */
559
530
  function findClaudeSessionFile(cwd, sessionId) {
560
531
  // Only memoize when the exact session UUID is known. Without an id the
@@ -2080,17 +2051,7 @@ export function foldHostLink(rows) {
2080
2051
  }
2081
2052
  if (link === 'host-gone' && s.status === 'closed')
2082
2053
  s.status = 'crashed';
2083
- // Only idle/input_required are promoted — a `running` session keeps its
2084
- // status (SES-18a). Extending this to a running agent with no client was
2085
- // tried and reverted: since RUSH-3125 wraps every remote interactive run in
2086
- // a detached tmux pane, "running with zero attached clients" is the NORMAL
2087
- // steady state between check-ins, and that path writes no detach record, so
2088
- // `deliberatelyDetached` is false for it. Promoting it would relabel every
2089
- // remote agent as orphaned whenever nobody is looking — the over-reporting
2090
- // this file's header calls worthless. Telling a stranded agent from a
2091
- // healthy unattended one needs to know a client was EXPECTED and LOST, which
2092
- // no signal available here carries; that belongs with the peer-side pane
2093
- // ownership work, not this function.
2054
+ // A clientless running remote pane is normal; only stopped work can be called orphaned.
2094
2055
  else if (link === 'no-client' && (s.status === 'idle' || s.status === 'input_required')) {
2095
2056
  s.status = 'orphaned';
2096
2057
  }
@@ -2103,18 +2064,7 @@ function recapLine(s, max = 120) {
2103
2064
  return undefined;
2104
2065
  return t.length > max ? t.slice(0, max - 1).trimEnd() + '…' : t;
2105
2066
  }
2106
- /**
2107
- * The recap ladder (RUSH-3011): compute a row's shown {@link ActiveSession.title}
2108
- * + {@link RecapSource} from the best available source, plus the cleaned first
2109
- * prompt (`userPromptClean`/`userPromptKind`) and the `lastAgentLine`. Pure over
2110
- * one row; exported for tests and folded in by {@link foldRecap}.
2111
- *
2112
- * Ladder, best-first: a `/rename`/harness `label` → the last assistant line →
2113
- * the first-prompt topic. The `last` rung is agent-derived, so a session that
2114
- * produced work stops showing its stale first prompt as the title. (`topic` is
2115
- * the row's already-extracted first line, so image detection here is
2116
- * path-based; a pure-attachment turn with no first-line text stays on `prompt`.)
2117
- */
2067
+ /** Labels win, then the last assistant line, then the first-prompt topic. */
2118
2068
  export function deriveSessionRecap(row) {
2119
2069
  const lastAgentLine = recapLine(row.tail?.length ? row.tail[row.tail.length - 1] : undefined);
2120
2070
  const { clean: userPromptClean, kind: userPromptKind } = classifyUserPrompt(row.topic ?? '');
@@ -2147,17 +2097,7 @@ export function foldRecap(rows) {
2147
2097
  s.lastAgentLine = recap.lastAgentLine;
2148
2098
  }
2149
2099
  }
2150
- /**
2151
- * True when a crash-leaked orphan is genuinely DEAD and should be reaped from the
2152
- * reconnectable set rather than shown as resumable forever (RUSH-3011 / issue #3b).
2153
- *
2154
- * The gate is `abandoned` (no transcript write in {@link ABANDONED_STALE_MS}) AND
2155
- * a dead pid — exactly "past the stale threshold whose pid is gone". A live pid
2156
- * (an idle-but-unfinished session, the highest-risk state) is NEVER reaped, and
2157
- * neither is a recently-`closed`/`crashed` session that just exited (still
2158
- * resumable). `pidAlive` absent (a cloud row or an older peer that can't prove
2159
- * death) also stays un-reaped — reaping is fail-safe, never a guess.
2160
- */
2100
+ /** Reap only stale sessions with proven-dead pids; unknown or live processes remain recoverable. */
2161
2101
  export function isReapableOrphan(row) {
2162
2102
  return row.status === 'abandoned' && row.pidAlive === false;
2163
2103
  }
@@ -567,46 +567,15 @@ interface TopCostSession {
567
567
  * vanished, mirroring querySessions' liveness filter.
568
568
  */
569
569
  export declare function topSessionsByCost(n: number, options?: QueryOptions): TopCostSession[];
570
- /** Look up a single session by its unique ID. */
571
570
  /**
572
- * Batch-resolve session ids to the machine each one runs on, in ONE indexed
573
- * query. `getActiveSessions` needs only this column for every live row, and
574
- * `getSessionById` would re-`prepare` a `SELECT *` and materialize a full
575
- * `SessionMeta` per id to read it — mirrors {@link findSessionsByShortIds}'s
576
- * single-round-trip pattern. Ids absent from the index are simply absent from
577
- * the map. Best-effort: an unavailable DB yields an empty map, so the live view
578
- * still renders (the caller then leaves rows attributed to this box).
571
+ * Read machine attribution in batches without materializing full sessions.
572
+ * Failure is best-effort so the live view can still render local attribution.
579
573
  */
580
574
  export declare function findSessionMachinesByIds(ids: string[]): Map<string, string>;
581
575
  export declare function getSessionById(id: string): SessionMeta | null;
582
- /**
583
- * Resolve a full-or-partial session id against the index, exact-first then
584
- * prefix — the DB-backed equivalent of resolveSessionById() that runs over the
585
- * SQLite table instead of a pre-loaded array. Matches both the full id and the
586
- * short id. An exact hit short-circuits so a complete id never also drags in its
587
- * prefix siblings. `scope` narrows by agent / version / project (cwd) so an
588
- * ambiguous prefix disambiguates against the caller's context.
589
- *
590
- * Routes through the full querySessions existence check (NOT skipExistenceCheck)
591
- * on purpose (RUSH-2436): that check now KEEPS a file-gone session whose user
592
- * turns still live in session_text (flagged archived) and only suppresses a
593
- * contentless phantom — so `agents sessions <id>` resolves an archived session
594
- * instead of failing with "No session found", while a phantom id still misses.
595
- */
576
+ /** Exact ids win over prefixes; the normal existence check preserves archived content but excludes phantoms. */
596
577
  export declare function findSessionsById(idQuery: string, scope?: Pick<QueryOptions, 'agent' | 'version' | 'cwd' | 'project'>): SessionMeta[];
597
- /**
598
- * Batch-resolve many 8-char short ids to their sessions in ONE indexed query.
599
- * The live-scan path (listTmuxAgentSessions) turns every `ag-<agent>-<shortid>`
600
- * tmux pane name back into a full session id this way, so it pays a single
601
- * `short_id IN (…)` round-trip per scan instead of N per-pane lookups.
602
- *
603
- * Returns a map keyed by short_id (lowercased). Short ids are the first 8 chars
604
- * of the lowercase session UUID (deriveShortId), so a lowercased `IN` matches and
605
- * still uses idx_sessions_short_id. When several sessions share a short id — only
606
- * time-ordered ids (ULID/UUIDv7) ever collide; random UUIDv4 short ids are unique
607
- * in practice — the most-recently-active one wins (the caller can further
608
- * disambiguate by cwd).
609
- */
578
+ /** Batch-resolve pane short ids; on collision the most recently active session wins. */
610
579
  export declare function findSessionsByShortIds(shortIds: string[]): Map<string, SessionMeta>;
611
580
  /** A single full-text search result with ranking score. */
612
581
  interface FtsHit {