@skrr-ai/cli 0.1.27 → 0.1.29

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/dist/base-command.d.ts +15 -0
  2. package/dist/base-command.js +49 -0
  3. package/dist/commands/agents/chat.d.ts +2 -0
  4. package/dist/commands/agents/chat.js +40 -2
  5. package/dist/commands/balance/show.d.ts +2 -3
  6. package/dist/commands/balance/show.js +2 -3
  7. package/dist/commands/balance/usage.d.ts +3 -11
  8. package/dist/commands/balance/usage.js +19 -72
  9. package/dist/commands/code/index.d.ts +4 -0
  10. package/dist/commands/code/index.js +9 -0
  11. package/dist/commands/code/install.d.ts +10 -2
  12. package/dist/commands/code/install.js +38 -28
  13. package/dist/commands/harnesses/leases/show.js +7 -4
  14. package/dist/commands/inbox/index.d.ts +15 -0
  15. package/dist/commands/inbox/index.js +52 -20
  16. package/dist/commands/instructions/install.d.ts +12 -0
  17. package/dist/commands/instructions/install.js +59 -14
  18. package/dist/commands/login.js +6 -0
  19. package/dist/commands/machines/dedicated/attach.js +1 -1
  20. package/dist/commands/machines/dedicated/cp.d.ts +1 -0
  21. package/dist/commands/machines/dedicated/cp.js +63 -9
  22. package/dist/commands/machines/dedicated/create.d.ts +1 -1
  23. package/dist/commands/machines/dedicated/create.js +7 -2
  24. package/dist/commands/machines/dedicated/exec.d.ts +23 -1
  25. package/dist/commands/machines/dedicated/exec.js +67 -7
  26. package/dist/commands/machines/dedicated/index.js +2 -0
  27. package/dist/commands/machines/dedicated/restore.d.ts +6 -0
  28. package/dist/commands/machines/dedicated/restore.js +7 -1
  29. package/dist/commands/machines/dedicated/sign-in.d.ts +4 -3
  30. package/dist/commands/machines/dedicated/sign-in.js +4 -3
  31. package/dist/commands/machines/dedicated/terminal.js +1 -1
  32. package/dist/commands/machines/dedicated/update-image.d.ts +15 -0
  33. package/dist/commands/machines/dedicated/update-image.js +38 -0
  34. package/dist/lib/balance.d.ts +2 -2
  35. package/dist/lib/balance.js +6 -4
  36. package/dist/lib/daemon-target.d.ts +103 -0
  37. package/dist/lib/daemon-target.js +110 -0
  38. package/dist/lib/dedicated-copy.d.ts +92 -2
  39. package/dist/lib/dedicated-copy.js +223 -18
  40. package/dist/lib/dedicated-lease-command.d.ts +7 -1
  41. package/dist/lib/dedicated-lease-command.js +16 -3
  42. package/dist/lib/dedicated-machines.d.ts +130 -8
  43. package/dist/lib/dedicated-machines.js +277 -16
  44. package/dist/lib/dedicated-terminal.d.ts +5 -25
  45. package/dist/lib/dedicated-terminal.js +45 -73
  46. package/dist/lib/dedicated-wait.d.ts +10 -0
  47. package/dist/lib/dedicated-wait.js +52 -0
  48. package/dist/lib/device-code.d.ts +12 -1
  49. package/dist/lib/device-code.js +44 -9
  50. package/dist/lib/harnesses.d.ts +13 -0
  51. package/dist/lib/harnesses.js +24 -0
  52. package/dist/lib/login.js +8 -7
  53. package/dist/lib/sky-code-broker.d.ts +46 -5
  54. package/dist/lib/sky-code-broker.js +96 -26
  55. package/dist/lib/sky-code.d.ts +33 -0
  56. package/dist/lib/sky-code.js +56 -8
  57. package/dist/lib/task-instruction-offer.js +12 -0
  58. package/dist/node_modules/@skrr-ai/auth-core/dist/cjs/refresh.d.ts +67 -1
  59. package/dist/node_modules/@skrr-ai/auth-core/dist/cjs/refresh.js +124 -12
  60. package/dist/node_modules/@skrr-ai/auth-core/dist/esm/refresh.d.ts +67 -1
  61. package/dist/node_modules/@skrr-ai/auth-core/dist/esm/refresh.js +123 -11
  62. package/dist/node_modules/@skrr-ai/auth-core/package.json +1 -1
  63. package/dist/node_modules/@skrr-ai/data-provider/index.js +3061 -2876
  64. package/dist/node_modules/@skrr-ai/data-provider/package.json +1 -1
  65. package/dist/node_modules/@skrr-ai/inference-broker/dist/cjs/index.d.ts +1 -1
  66. package/dist/node_modules/@skrr-ai/inference-broker/dist/cjs/index.js +3 -1
  67. package/dist/node_modules/@skrr-ai/inference-broker/dist/cjs/loopback-server.js +64 -1
  68. package/dist/node_modules/@skrr-ai/inference-broker/dist/cjs/managed-inference-broker.d.ts +30 -0
  69. package/dist/node_modules/@skrr-ai/inference-broker/dist/cjs/managed-inference-broker.js +43 -1
  70. package/dist/node_modules/@skrr-ai/inference-broker/dist/esm/index.d.ts +1 -1
  71. package/dist/node_modules/@skrr-ai/inference-broker/dist/esm/index.js +1 -1
  72. package/dist/node_modules/@skrr-ai/inference-broker/dist/esm/loopback-server.js +64 -1
  73. package/dist/node_modules/@skrr-ai/inference-broker/dist/esm/managed-inference-broker.d.ts +30 -0
  74. package/dist/node_modules/@skrr-ai/inference-broker/dist/esm/managed-inference-broker.js +41 -0
  75. package/dist/node_modules/@skrr-ai/inference-broker/package.json +1 -1
  76. package/oclif.manifest.json +3113 -3011
  77. package/package.json +1 -1
@@ -17,7 +17,7 @@ var __importDefault = (this && this.__importDefault) || function (mod) {
17
17
  return (mod && mod.__esModule) ? mod : { "default": mod };
18
18
  };
19
19
  Object.defineProperty(exports, "__esModule", { value: true });
20
- exports.DEDICATED_RUNTIME_ENGINE_MESSAGE = exports.DEDICATED_WORKLOAD_UID_ENV = exports.ENGINE_PATH_ENV = void 0;
20
+ exports.ENGINE_START_REFUSED_EXIT_CODE = exports.EngineStartRefusedError = exports.DEDICATED_RUNTIME_ENGINE_MESSAGE = exports.DEDICATED_WORKLOAD_UID_ENV = exports.ENGINE_PATH_ENV = void 0;
21
21
  exports.engineHome = engineHome;
22
22
  exports.skyCodeInstructionHome = skyCodeInstructionHome;
23
23
  exports.managedEnginePath = managedEnginePath;
@@ -36,6 +36,7 @@ const node_fs_1 = require("node:fs");
36
36
  const node_os_1 = require("node:os");
37
37
  const node_path_1 = __importDefault(require("node:path"));
38
38
  const auth_core_1 = require("@skrr-ai/auth-core");
39
+ const inference_broker_1 = require("@skrr-ai/inference-broker");
39
40
  const config_1 = require("./config");
40
41
  // Deliberately does NOT import the managed marker any more. This module used to
41
42
  // read it to decide whether a run was managed; that decision now arrives as a
@@ -263,7 +264,7 @@ exports.DEDICATED_RUNTIME_ENGINE_MESSAGE = 'On a Dedicated Runtime, coding engin
263
264
  function notInstalledMessage(env = process.env) {
264
265
  if (isDedicatedRuntimeWorkload(env)) {
265
266
  return [
266
- 'The skrr Code engine (`sky-code`) is not on this machine.',
267
+ 'The skrr Code engine is not on this machine.',
267
268
  '',
268
269
  exports.DEDICATED_RUNTIME_ENGINE_MESSAGE,
269
270
  'This machine was built from an image without it; it arrives when the',
@@ -287,6 +288,41 @@ function notInstalledMessage(env = process.env) {
287
288
  'for a published release.',
288
289
  ].join('\n');
289
290
  }
291
+ /**
292
+ * OSK-8822 — the run was REFUSED before the engine started.
293
+ *
294
+ * `prepare` used to have exactly one way to answer: an env map, where `{}` means
295
+ * "run on the operator's own credentials". So when the server refused the
296
+ * ACCOUNT — a plan allowance used up, a model the plan does not include — the
297
+ * refusal could only become a BYOK run, and the engine started with no
298
+ * credential and failed a second time with a generic error, exiting with the
299
+ * same `1` as a task that ran and failed.
300
+ *
301
+ * A distinct type rather than a sentinel env, so the one outcome that must stop
302
+ * the run cannot be produced by accident and cannot be mistaken for an ordinary
303
+ * `prepare` failure. Whoever throws it has already told the user why: it is a
304
+ * control signal, and its message is for logs, not for printing a second time.
305
+ */
306
+ class EngineStartRefusedError extends Error {
307
+ /** The refusal's machine-readable code, e.g. `plan-allowance-exhausted`. */
308
+ code;
309
+ constructor(code, message) {
310
+ super(message);
311
+ this.name = 'EngineStartRefusedError';
312
+ this.code = code;
313
+ }
314
+ }
315
+ exports.EngineStartRefusedError = EngineStartRefusedError;
316
+ /**
317
+ * The exit code of a refused run: sysexits' `EX_NOPERM`, the one that means
318
+ * "not permitted" rather than "not available".
319
+ *
320
+ * CLI-owned, like `127` for a missing engine, because nothing ran to report one.
321
+ * It has to differ from `1`, which is what an engine that started and failed
322
+ * reports — the production case this exists for exited `1`, and CI could not tell
323
+ * "your plan refused this" from "the task failed".
324
+ */
325
+ exports.ENGINE_START_REFUSED_EXIT_CODE = 77;
290
326
  /**
291
327
  * Credential-shaped variables the engine may inherit as its OWN model credential.
292
328
  *
@@ -468,11 +504,20 @@ function engineSpawnEnv(env, extraEnv,
468
504
  // Required, not defaulted: the safe value differs per call and a parameter that
469
505
  // can be forgotten on a security boundary will be.
470
506
  mode) {
471
- return {
507
+ const spawnEnv = {
472
508
  ...(0, auth_core_1.sanitizeSpawnEnv)(env),
473
509
  ...(mode === 'managed' ? {} : inheritedProviderCredentials(env)),
474
510
  ...extraEnv,
475
511
  };
512
+ if (mode !== 'managed')
513
+ return spawnEnv;
514
+ // OSK-8758 — a managed engine reaches the broker at a synthetic host, and must
515
+ // never address that host to a proxy. Behind an egress proxy (a Dedicated
516
+ // Runtime guest) Bun wrote every broker request in proxy form, and a runtime
517
+ // that honoured the proxy over the socket would send the prompt to it. Layered
518
+ // over the env built above, so it extends whatever list the engine would
519
+ // otherwise have received rather than replacing it.
520
+ return { ...spawnEnv, ...(0, inference_broker_1.inferenceBrokerNoProxyEnv)(spawnEnv) };
476
521
  }
477
522
  async function execEngine(args, env = process.env, options = {}) {
478
523
  // Resolve the binary FIRST, before `prepare` runs.
@@ -483,15 +528,18 @@ async function execEngine(args, env = process.env, options = {}) {
483
528
  const { path: bin } = resolveEngine(env);
484
529
  if (!bin)
485
530
  throw new Error('sky-code-not-installed');
486
- const extraEnv = options.prepare ? await options.prepare() : {};
487
- // A copy, so a hook cannot mutate the caller's array as a side effect.
488
- const spawnArgs = options.finalizeArgs ? options.finalizeArgs([...args]) : args;
489
531
  // Assumed until the spawn tells us otherwise, so a teardown that runs after a
490
532
  // spawn ERROR does not claim the run was interrupted.
491
533
  const outcome = { interrupted: false };
492
- // Read AFTER `prepare`, because that is when the hook knows what it acquired.
493
- const mode = options.credentialMode?.() ?? 'byok';
494
534
  try {
535
+ // Inside the `try`, so a `prepare` that refuses the run (OSK-8822) — or one
536
+ // that acquired something and then threw — still reaches `cleanup`. Rejecting
537
+ // is how a hook stops the run: nothing below it, the spawn included, runs.
538
+ const extraEnv = options.prepare ? await options.prepare() : {};
539
+ // A copy, so a hook cannot mutate the caller's array as a side effect.
540
+ const spawnArgs = options.finalizeArgs ? options.finalizeArgs([...args]) : args;
541
+ // Read AFTER `prepare`, because that is when the hook knows what it acquired.
542
+ const mode = options.credentialMode?.() ?? 'byok';
495
543
  return await spawnEngine(bin, spawnArgs, env, extraEnv, mode, outcome);
496
544
  }
497
545
  finally {
@@ -2,6 +2,8 @@
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.maybeOfferTaskManagementInstructions = maybeOfferTaskManagementInstructions;
4
4
  const data_provider_1 = require("@skrr-ai/data-provider");
5
+ const config_1 = require("./config");
6
+ const daemon_target_1 = require("./daemon-target");
5
7
  const prompt_1 = require("./prompt");
6
8
  const TEMPLATE_KEY = 'oversky-task-management';
7
9
  function capabilitiesOf(daemon) {
@@ -64,11 +66,21 @@ async function maybeOfferTaskManagementInstructions(log) {
64
66
  if (shouldSkipOffer())
65
67
  return;
66
68
  try {
69
+ // The offer speaks of "this machine", so on a Dedicated Runtime guest only the
70
+ // guest's own daemon may be the candidate — never the first eligible row of the
71
+ // account, which is the owner's laptop (OSK-8759). A guest whose daemon has
72
+ // published no id gets no candidate. Off a guest nothing is read here and the
73
+ // search below is unchanged.
74
+ const local = (0, daemon_target_1.localMachine)({ baseURL: (0, config_1.loadConfig)().baseURL });
75
+ if (local.dedicatedRuntime && !local.daemonId)
76
+ return;
67
77
  const [daemons, offerResponse] = await Promise.all([
68
78
  data_provider_1.dataService.getDaemons(),
69
79
  data_provider_1.dataService.getManagedInstructionOffer(TEMPLATE_KEY),
70
80
  ]);
71
81
  const candidate = daemons.active.find((daemon) => {
82
+ if (local.dedicatedRuntime && daemon.daemonId !== local.daemonId)
83
+ return false;
72
84
  if (!detectedProviders(daemon).length)
73
85
  return false;
74
86
  return offerResponse.offers.find((offer) => offer.daemonId === daemon.daemonId)?.eligible;
@@ -1,12 +1,78 @@
1
1
  import { type ReauthState, type TokenPair } from './types.js';
2
2
  declare function authLockPath(): string;
3
3
  declare function reauthFlagPath(): string;
4
+ /**
5
+ * How long a process whose own listeners took a fatal signal gets to shut down
6
+ * before it is exited anyway: bounded, so a cleanup that hangs cannot make the
7
+ * process unkillable, and longer than the longest cleanup a command registers.
8
+ *
9
+ * That is `skrr machines dedicated cp` cancelling an upload. It waits out the
10
+ * chunk in flight, which the server bounds at 30 s (a 20 s tool timeout plus a
11
+ * 10 s response grace, `DedicatedRuntimeMachineAccess.js`), and then sends one
12
+ * abort; 45 s leaves 15 s for that abort after a chunk of the full length. An
13
+ * abort that needs longer is going to a machine that has just failed to answer
14
+ * a chunk in time, and the daemon's stale-staging sweep removes what it leaves.
15
+ * `agents chat` waits 10 s for the cancelled run and then reads its record once.
16
+ *
17
+ * One constant rather than a bound each listener declares: every cleanup
18
+ * registered today fits under it, and a registry of bounds would be machinery
19
+ * for a second number nobody has needed.
20
+ */
21
+ export declare const LOCK_SIGNAL_FALLBACK_EXIT_MS = 45000;
22
+ export interface LockExitHandlerOptions {
23
+ /** Overrides LOCK_SIGNAL_FALLBACK_EXIT_MS. For tests. */
24
+ fallbackExitMs?: number;
25
+ }
4
26
  /**
5
27
  * P1-6 — Register best-effort lock-release handlers on process exit
6
28
  * signals so a crash or SIGTERM during a refresh leaves the lock file
7
29
  * in a released state. Without this, a sibling process must wait up to
8
30
  * `LOCK_STALE_MS` (5 s) for `proper-lockfile` to declare the lock stale.
9
31
  *
32
+ * A signal handler here must see the lock released when the process ends, and
33
+ * otherwise get out of the way. It used to release and call `process.exit`
34
+ * synchronously, and because it is installed as soon as a CLI command loads,
35
+ * it ran before — and so instead of — every listener a command registered for
36
+ * its own cleanup: Ctrl-C during `skrr machines dedicated cp` left partial
37
+ * files behind (OSK-8708), `agents chat` never cancelled the run it said it
38
+ * would, and the daemon had to strip these listeners off to reach its own
39
+ * drain at all (OSK-2086). Now:
40
+ *
41
+ * 1. The lock is released only when the process actually ends: by the
42
+ * `exit` listener, and immediately before a re-raise (dying by a signal
43
+ * runs no `exit` listener). Never on the signal itself. A refresh already
44
+ * inside `withAuthLock` keeps running while a command cleans up, and
45
+ * releasing under it would let a sibling `skrr` take the lock and spend
46
+ * the same refresh token — which rotates on use, so one of the two would
47
+ * be signed out.
48
+ * 2. If anything else is listening for the signal — a command's cleanup —
49
+ * that listener owns shutdown, and this handler does not exit.
50
+ * signal-exit's listeners do not count (`signalExitListenerCount`).
51
+ * 3. Otherwise it does what Node does for an unhandled fatal signal: it
52
+ * removes itself and re-raises the signal, so the process dies BY the
53
+ * signal. That is more faithful than `exit(128 + n)`, which only imitates
54
+ * the number: a parent can tell the two apart, and does. bash stops a
55
+ * loop when a child is killed by SIGINT but carries on after one that
56
+ * exited 130, and systemd counts death by SIGTERM as a clean stop but
57
+ * exit status 143 as a failure. (signal-exit, when present, sees the
58
+ * re-raised signal with no other listener and runs its exit callbacks
59
+ * before re-raising it once more.) Not on Windows, where raising a
60
+ * signal at yourself is TerminateProcess with status 1: Ctrl-C would exit
61
+ * 1 and no `exit` listener would run. There it exits `128 + n`.
62
+ * 4. While owners are shutting down, an unref'd fallback timer exits
63
+ * `128 + n` after `fallbackExitMs`, and a second fatal signal exits as
64
+ * soon as that signal's listeners have run — so an owner's handler for a
65
+ * second signal gets its synchronous turn, and nothing more. The forced
66
+ * exits use `process.exit`, not a re-raise, so `exit` listeners (this
67
+ * lock release, a command's last synchronous cleanup) still run.
68
+ *
69
+ * The listener is PREPENDED, so it runs first and counts every other
70
+ * listener — including `once` listeners, which remove themselves when called
71
+ * and would otherwise be invisible to a handler that ran after them. A
72
+ * listener added LATER with `prependOnceListener` is the exception: it runs
73
+ * before this one and has removed itself by the time this one counts, so it
74
+ * is not seen as an owner. Register command cleanup with `on` or `once`.
75
+ *
10
76
  * Guards:
11
77
  * - `_exitHandlersInstalled` (module-level): safe to call multiple
12
78
  * times (e.g., from unit-test re-imports) — handlers are registered
@@ -20,7 +86,7 @@ declare function reauthFlagPath(): string;
20
86
  * next to `auth.lock`. It is synchronous so it runs safely in POSIX
21
87
  * signal handlers and `process.on('exit')`.
22
88
  */
23
- export declare function installLockExitHandlers(): void;
89
+ export declare function installLockExitHandlers(options?: LockExitHandlerOptions): void;
24
90
  /**
25
91
  * Acquire the cross-process auth lock, run the body, release on `finally`.
26
92
  * Throws if the lock can't be acquired within
@@ -3,7 +3,7 @@ var __importDefault = (this && this.__importDefault) || function (mod) {
3
3
  return (mod && mod.__esModule) ? mod : { "default": mod };
4
4
  };
5
5
  Object.defineProperty(exports, "__esModule", { value: true });
6
- exports.__refreshDiagnostics = exports.UNKNOWN_401_BUDGET = void 0;
6
+ exports.__refreshDiagnostics = exports.UNKNOWN_401_BUDGET = exports.LOCK_SIGNAL_FALLBACK_EXIT_MS = void 0;
7
7
  exports.installLockExitHandlers = installLockExitHandlers;
8
8
  exports.withAuthLock = withAuthLock;
9
9
  exports.__resetUnknown401StreaksForTest = __resetUnknown401StreaksForTest;
@@ -88,12 +88,95 @@ function authLockPath() {
88
88
  function reauthFlagPath() {
89
89
  return node_path_1.default.join((0, runtime_js_1.getAuthConfigDir)(), 'needs-reauth.json');
90
90
  }
91
+ /** The fatal signals the lock-release handler listens for, with their numbers. */
92
+ const LOCK_RELEASE_SIGNALS = { SIGHUP: 1, SIGINT: 2, SIGTERM: 15 };
93
+ /**
94
+ * How long a process whose own listeners took a fatal signal gets to shut down
95
+ * before it is exited anyway: bounded, so a cleanup that hangs cannot make the
96
+ * process unkillable, and longer than the longest cleanup a command registers.
97
+ *
98
+ * That is `skrr machines dedicated cp` cancelling an upload. It waits out the
99
+ * chunk in flight, which the server bounds at 30 s (a 20 s tool timeout plus a
100
+ * 10 s response grace, `DedicatedRuntimeMachineAccess.js`), and then sends one
101
+ * abort; 45 s leaves 15 s for that abort after a chunk of the full length. An
102
+ * abort that needs longer is going to a machine that has just failed to answer
103
+ * a chunk in time, and the daemon's stale-staging sweep removes what it leaves.
104
+ * `agents chat` waits 10 s for the cancelled run and then reads its record once.
105
+ *
106
+ * One constant rather than a bound each listener declares: every cleanup
107
+ * registered today fits under it, and a registry of bounds would be machinery
108
+ * for a second number nobody has needed.
109
+ */
110
+ exports.LOCK_SIGNAL_FALLBACK_EXIT_MS = 45_000;
111
+ /**
112
+ * How many of a signal's listeners belong to `signal-exit`.
113
+ *
114
+ * `proper-lockfile` loads signal-exit when it is imported, so every process that
115
+ * imports this module carries its SIGHUP/SIGINT/SIGTERM listeners. signal-exit
116
+ * is deferential by design: it acts only when it is the sole listener, and
117
+ * otherwise leaves the signal to whoever else is listening. Counted as an owner,
118
+ * it and this handler would each wait for the other and nothing would exit.
119
+ * v3 and v4 publish their loaded-instance counts on process-wide globals, which
120
+ * is how they recognise each other (v4 adds v3's count to its own); this reads
121
+ * the same two counts.
122
+ */
123
+ function signalExitListenerCount() {
124
+ const count = (emitter) => typeof emitter?.count === 'number' ? emitter.count : 0;
125
+ const v3 = process
126
+ .__signal_exit_emitter__;
127
+ const v4 = globalThis[Symbol.for('signal-exit emitter')];
128
+ return count(v3) + count(v4);
129
+ }
91
130
  /**
92
131
  * P1-6 — Register best-effort lock-release handlers on process exit
93
132
  * signals so a crash or SIGTERM during a refresh leaves the lock file
94
133
  * in a released state. Without this, a sibling process must wait up to
95
134
  * `LOCK_STALE_MS` (5 s) for `proper-lockfile` to declare the lock stale.
96
135
  *
136
+ * A signal handler here must see the lock released when the process ends, and
137
+ * otherwise get out of the way. It used to release and call `process.exit`
138
+ * synchronously, and because it is installed as soon as a CLI command loads,
139
+ * it ran before — and so instead of — every listener a command registered for
140
+ * its own cleanup: Ctrl-C during `skrr machines dedicated cp` left partial
141
+ * files behind (OSK-8708), `agents chat` never cancelled the run it said it
142
+ * would, and the daemon had to strip these listeners off to reach its own
143
+ * drain at all (OSK-2086). Now:
144
+ *
145
+ * 1. The lock is released only when the process actually ends: by the
146
+ * `exit` listener, and immediately before a re-raise (dying by a signal
147
+ * runs no `exit` listener). Never on the signal itself. A refresh already
148
+ * inside `withAuthLock` keeps running while a command cleans up, and
149
+ * releasing under it would let a sibling `skrr` take the lock and spend
150
+ * the same refresh token — which rotates on use, so one of the two would
151
+ * be signed out.
152
+ * 2. If anything else is listening for the signal — a command's cleanup —
153
+ * that listener owns shutdown, and this handler does not exit.
154
+ * signal-exit's listeners do not count (`signalExitListenerCount`).
155
+ * 3. Otherwise it does what Node does for an unhandled fatal signal: it
156
+ * removes itself and re-raises the signal, so the process dies BY the
157
+ * signal. That is more faithful than `exit(128 + n)`, which only imitates
158
+ * the number: a parent can tell the two apart, and does. bash stops a
159
+ * loop when a child is killed by SIGINT but carries on after one that
160
+ * exited 130, and systemd counts death by SIGTERM as a clean stop but
161
+ * exit status 143 as a failure. (signal-exit, when present, sees the
162
+ * re-raised signal with no other listener and runs its exit callbacks
163
+ * before re-raising it once more.) Not on Windows, where raising a
164
+ * signal at yourself is TerminateProcess with status 1: Ctrl-C would exit
165
+ * 1 and no `exit` listener would run. There it exits `128 + n`.
166
+ * 4. While owners are shutting down, an unref'd fallback timer exits
167
+ * `128 + n` after `fallbackExitMs`, and a second fatal signal exits as
168
+ * soon as that signal's listeners have run — so an owner's handler for a
169
+ * second signal gets its synchronous turn, and nothing more. The forced
170
+ * exits use `process.exit`, not a re-raise, so `exit` listeners (this
171
+ * lock release, a command's last synchronous cleanup) still run.
172
+ *
173
+ * The listener is PREPENDED, so it runs first and counts every other
174
+ * listener — including `once` listeners, which remove themselves when called
175
+ * and would otherwise be invisible to a handler that ran after them. A
176
+ * listener added LATER with `prependOnceListener` is the exception: it runs
177
+ * before this one and has removed itself by the time this one counts, so it
178
+ * is not seen as an owner. Register command cleanup with `on` or `once`.
179
+ *
97
180
  * Guards:
98
181
  * - `_exitHandlersInstalled` (module-level): safe to call multiple
99
182
  * times (e.g., from unit-test re-imports) — handlers are registered
@@ -107,10 +190,11 @@ function reauthFlagPath() {
107
190
  * next to `auth.lock`. It is synchronous so it runs safely in POSIX
108
191
  * signal handlers and `process.on('exit')`.
109
192
  */
110
- function installLockExitHandlers() {
193
+ function installLockExitHandlers(options = {}) {
111
194
  if (_exitHandlersInstalled)
112
195
  return;
113
196
  _exitHandlersInstalled = true;
197
+ const fallbackExitMs = options.fallbackExitMs ?? exports.LOCK_SIGNAL_FALLBACK_EXIT_MS;
114
198
  function tryRelease() {
115
199
  try {
116
200
  proper_lockfile_1.default.unlockSync(authLockPath());
@@ -120,18 +204,46 @@ function installLockExitHandlers() {
120
204
  }
121
205
  }
122
206
  process.on('exit', tryRelease);
123
- process.on('SIGTERM', () => {
124
- tryRelease();
125
- process.exit(0);
126
- });
127
- process.on('SIGINT', () => {
207
+ const exitStatus = (signal) => 128 + LOCK_RELEASE_SIGNALS[signal];
208
+ let deferred = false;
209
+ const listeners = new Map();
210
+ // `process.exit` runs the `exit` listener above, which releases the lock;
211
+ // releasing here as well costs nothing and does not depend on that.
212
+ const exitNow = (signal) => {
128
213
  tryRelease();
129
- process.exit(130);
130
- });
131
- process.on('SIGHUP', () => {
214
+ process.exit(exitStatus(signal));
215
+ };
216
+ const onSignal = (signal) => {
217
+ if (deferred) {
218
+ // A later signal while owners are shutting down: stop waiting for them.
219
+ setImmediate(() => exitNow(signal));
220
+ return;
221
+ }
222
+ const owners = process.listenerCount(signal) - 1 - signalExitListenerCount();
223
+ if (owners > 0) {
224
+ deferred = true;
225
+ setTimeout(() => exitNow(signal), fallbackExitMs).unref();
226
+ return;
227
+ }
228
+ process.removeListener(signal, listeners.get(signal));
229
+ if (process.platform === 'win32') {
230
+ exitNow(signal);
231
+ return;
232
+ }
233
+ // Dying by the signal runs no `exit` listener, so release first.
132
234
  tryRelease();
133
- process.exit(1);
134
- });
235
+ try {
236
+ process.kill(process.pid, signal);
237
+ }
238
+ catch {
239
+ exitNow(signal);
240
+ }
241
+ };
242
+ for (const signal of Object.keys(LOCK_RELEASE_SIGNALS)) {
243
+ const listener = () => onSignal(signal);
244
+ listeners.set(signal, listener);
245
+ process.prependListener(signal, listener);
246
+ }
135
247
  }
136
248
  /**
137
249
  * Acquire the cross-process auth lock, run the body, release on `finally`.
@@ -1,12 +1,78 @@
1
1
  import { type ReauthState, type TokenPair } from './types.js';
2
2
  declare function authLockPath(): string;
3
3
  declare function reauthFlagPath(): string;
4
+ /**
5
+ * How long a process whose own listeners took a fatal signal gets to shut down
6
+ * before it is exited anyway: bounded, so a cleanup that hangs cannot make the
7
+ * process unkillable, and longer than the longest cleanup a command registers.
8
+ *
9
+ * That is `skrr machines dedicated cp` cancelling an upload. It waits out the
10
+ * chunk in flight, which the server bounds at 30 s (a 20 s tool timeout plus a
11
+ * 10 s response grace, `DedicatedRuntimeMachineAccess.js`), and then sends one
12
+ * abort; 45 s leaves 15 s for that abort after a chunk of the full length. An
13
+ * abort that needs longer is going to a machine that has just failed to answer
14
+ * a chunk in time, and the daemon's stale-staging sweep removes what it leaves.
15
+ * `agents chat` waits 10 s for the cancelled run and then reads its record once.
16
+ *
17
+ * One constant rather than a bound each listener declares: every cleanup
18
+ * registered today fits under it, and a registry of bounds would be machinery
19
+ * for a second number nobody has needed.
20
+ */
21
+ export declare const LOCK_SIGNAL_FALLBACK_EXIT_MS = 45000;
22
+ export interface LockExitHandlerOptions {
23
+ /** Overrides LOCK_SIGNAL_FALLBACK_EXIT_MS. For tests. */
24
+ fallbackExitMs?: number;
25
+ }
4
26
  /**
5
27
  * P1-6 — Register best-effort lock-release handlers on process exit
6
28
  * signals so a crash or SIGTERM during a refresh leaves the lock file
7
29
  * in a released state. Without this, a sibling process must wait up to
8
30
  * `LOCK_STALE_MS` (5 s) for `proper-lockfile` to declare the lock stale.
9
31
  *
32
+ * A signal handler here must see the lock released when the process ends, and
33
+ * otherwise get out of the way. It used to release and call `process.exit`
34
+ * synchronously, and because it is installed as soon as a CLI command loads,
35
+ * it ran before — and so instead of — every listener a command registered for
36
+ * its own cleanup: Ctrl-C during `skrr machines dedicated cp` left partial
37
+ * files behind (OSK-8708), `agents chat` never cancelled the run it said it
38
+ * would, and the daemon had to strip these listeners off to reach its own
39
+ * drain at all (OSK-2086). Now:
40
+ *
41
+ * 1. The lock is released only when the process actually ends: by the
42
+ * `exit` listener, and immediately before a re-raise (dying by a signal
43
+ * runs no `exit` listener). Never on the signal itself. A refresh already
44
+ * inside `withAuthLock` keeps running while a command cleans up, and
45
+ * releasing under it would let a sibling `skrr` take the lock and spend
46
+ * the same refresh token — which rotates on use, so one of the two would
47
+ * be signed out.
48
+ * 2. If anything else is listening for the signal — a command's cleanup —
49
+ * that listener owns shutdown, and this handler does not exit.
50
+ * signal-exit's listeners do not count (`signalExitListenerCount`).
51
+ * 3. Otherwise it does what Node does for an unhandled fatal signal: it
52
+ * removes itself and re-raises the signal, so the process dies BY the
53
+ * signal. That is more faithful than `exit(128 + n)`, which only imitates
54
+ * the number: a parent can tell the two apart, and does. bash stops a
55
+ * loop when a child is killed by SIGINT but carries on after one that
56
+ * exited 130, and systemd counts death by SIGTERM as a clean stop but
57
+ * exit status 143 as a failure. (signal-exit, when present, sees the
58
+ * re-raised signal with no other listener and runs its exit callbacks
59
+ * before re-raising it once more.) Not on Windows, where raising a
60
+ * signal at yourself is TerminateProcess with status 1: Ctrl-C would exit
61
+ * 1 and no `exit` listener would run. There it exits `128 + n`.
62
+ * 4. While owners are shutting down, an unref'd fallback timer exits
63
+ * `128 + n` after `fallbackExitMs`, and a second fatal signal exits as
64
+ * soon as that signal's listeners have run — so an owner's handler for a
65
+ * second signal gets its synchronous turn, and nothing more. The forced
66
+ * exits use `process.exit`, not a re-raise, so `exit` listeners (this
67
+ * lock release, a command's last synchronous cleanup) still run.
68
+ *
69
+ * The listener is PREPENDED, so it runs first and counts every other
70
+ * listener — including `once` listeners, which remove themselves when called
71
+ * and would otherwise be invisible to a handler that ran after them. A
72
+ * listener added LATER with `prependOnceListener` is the exception: it runs
73
+ * before this one and has removed itself by the time this one counts, so it
74
+ * is not seen as an owner. Register command cleanup with `on` or `once`.
75
+ *
10
76
  * Guards:
11
77
  * - `_exitHandlersInstalled` (module-level): safe to call multiple
12
78
  * times (e.g., from unit-test re-imports) — handlers are registered
@@ -20,7 +86,7 @@ declare function reauthFlagPath(): string;
20
86
  * next to `auth.lock`. It is synchronous so it runs safely in POSIX
21
87
  * signal handlers and `process.on('exit')`.
22
88
  */
23
- export declare function installLockExitHandlers(): void;
89
+ export declare function installLockExitHandlers(options?: LockExitHandlerOptions): void;
24
90
  /**
25
91
  * Acquire the cross-process auth lock, run the body, release on `finally`.
26
92
  * Throws if the lock can't be acquired within
@@ -72,12 +72,95 @@ function authLockPath() {
72
72
  function reauthFlagPath() {
73
73
  return path.join(getAuthConfigDir(), 'needs-reauth.json');
74
74
  }
75
+ /** The fatal signals the lock-release handler listens for, with their numbers. */
76
+ const LOCK_RELEASE_SIGNALS = { SIGHUP: 1, SIGINT: 2, SIGTERM: 15 };
77
+ /**
78
+ * How long a process whose own listeners took a fatal signal gets to shut down
79
+ * before it is exited anyway: bounded, so a cleanup that hangs cannot make the
80
+ * process unkillable, and longer than the longest cleanup a command registers.
81
+ *
82
+ * That is `skrr machines dedicated cp` cancelling an upload. It waits out the
83
+ * chunk in flight, which the server bounds at 30 s (a 20 s tool timeout plus a
84
+ * 10 s response grace, `DedicatedRuntimeMachineAccess.js`), and then sends one
85
+ * abort; 45 s leaves 15 s for that abort after a chunk of the full length. An
86
+ * abort that needs longer is going to a machine that has just failed to answer
87
+ * a chunk in time, and the daemon's stale-staging sweep removes what it leaves.
88
+ * `agents chat` waits 10 s for the cancelled run and then reads its record once.
89
+ *
90
+ * One constant rather than a bound each listener declares: every cleanup
91
+ * registered today fits under it, and a registry of bounds would be machinery
92
+ * for a second number nobody has needed.
93
+ */
94
+ export const LOCK_SIGNAL_FALLBACK_EXIT_MS = 45_000;
95
+ /**
96
+ * How many of a signal's listeners belong to `signal-exit`.
97
+ *
98
+ * `proper-lockfile` loads signal-exit when it is imported, so every process that
99
+ * imports this module carries its SIGHUP/SIGINT/SIGTERM listeners. signal-exit
100
+ * is deferential by design: it acts only when it is the sole listener, and
101
+ * otherwise leaves the signal to whoever else is listening. Counted as an owner,
102
+ * it and this handler would each wait for the other and nothing would exit.
103
+ * v3 and v4 publish their loaded-instance counts on process-wide globals, which
104
+ * is how they recognise each other (v4 adds v3's count to its own); this reads
105
+ * the same two counts.
106
+ */
107
+ function signalExitListenerCount() {
108
+ const count = (emitter) => typeof emitter?.count === 'number' ? emitter.count : 0;
109
+ const v3 = process
110
+ .__signal_exit_emitter__;
111
+ const v4 = globalThis[Symbol.for('signal-exit emitter')];
112
+ return count(v3) + count(v4);
113
+ }
75
114
  /**
76
115
  * P1-6 — Register best-effort lock-release handlers on process exit
77
116
  * signals so a crash or SIGTERM during a refresh leaves the lock file
78
117
  * in a released state. Without this, a sibling process must wait up to
79
118
  * `LOCK_STALE_MS` (5 s) for `proper-lockfile` to declare the lock stale.
80
119
  *
120
+ * A signal handler here must see the lock released when the process ends, and
121
+ * otherwise get out of the way. It used to release and call `process.exit`
122
+ * synchronously, and because it is installed as soon as a CLI command loads,
123
+ * it ran before — and so instead of — every listener a command registered for
124
+ * its own cleanup: Ctrl-C during `skrr machines dedicated cp` left partial
125
+ * files behind (OSK-8708), `agents chat` never cancelled the run it said it
126
+ * would, and the daemon had to strip these listeners off to reach its own
127
+ * drain at all (OSK-2086). Now:
128
+ *
129
+ * 1. The lock is released only when the process actually ends: by the
130
+ * `exit` listener, and immediately before a re-raise (dying by a signal
131
+ * runs no `exit` listener). Never on the signal itself. A refresh already
132
+ * inside `withAuthLock` keeps running while a command cleans up, and
133
+ * releasing under it would let a sibling `skrr` take the lock and spend
134
+ * the same refresh token — which rotates on use, so one of the two would
135
+ * be signed out.
136
+ * 2. If anything else is listening for the signal — a command's cleanup —
137
+ * that listener owns shutdown, and this handler does not exit.
138
+ * signal-exit's listeners do not count (`signalExitListenerCount`).
139
+ * 3. Otherwise it does what Node does for an unhandled fatal signal: it
140
+ * removes itself and re-raises the signal, so the process dies BY the
141
+ * signal. That is more faithful than `exit(128 + n)`, which only imitates
142
+ * the number: a parent can tell the two apart, and does. bash stops a
143
+ * loop when a child is killed by SIGINT but carries on after one that
144
+ * exited 130, and systemd counts death by SIGTERM as a clean stop but
145
+ * exit status 143 as a failure. (signal-exit, when present, sees the
146
+ * re-raised signal with no other listener and runs its exit callbacks
147
+ * before re-raising it once more.) Not on Windows, where raising a
148
+ * signal at yourself is TerminateProcess with status 1: Ctrl-C would exit
149
+ * 1 and no `exit` listener would run. There it exits `128 + n`.
150
+ * 4. While owners are shutting down, an unref'd fallback timer exits
151
+ * `128 + n` after `fallbackExitMs`, and a second fatal signal exits as
152
+ * soon as that signal's listeners have run — so an owner's handler for a
153
+ * second signal gets its synchronous turn, and nothing more. The forced
154
+ * exits use `process.exit`, not a re-raise, so `exit` listeners (this
155
+ * lock release, a command's last synchronous cleanup) still run.
156
+ *
157
+ * The listener is PREPENDED, so it runs first and counts every other
158
+ * listener — including `once` listeners, which remove themselves when called
159
+ * and would otherwise be invisible to a handler that ran after them. A
160
+ * listener added LATER with `prependOnceListener` is the exception: it runs
161
+ * before this one and has removed itself by the time this one counts, so it
162
+ * is not seen as an owner. Register command cleanup with `on` or `once`.
163
+ *
81
164
  * Guards:
82
165
  * - `_exitHandlersInstalled` (module-level): safe to call multiple
83
166
  * times (e.g., from unit-test re-imports) — handlers are registered
@@ -91,10 +174,11 @@ function reauthFlagPath() {
91
174
  * next to `auth.lock`. It is synchronous so it runs safely in POSIX
92
175
  * signal handlers and `process.on('exit')`.
93
176
  */
94
- export function installLockExitHandlers() {
177
+ export function installLockExitHandlers(options = {}) {
95
178
  if (_exitHandlersInstalled)
96
179
  return;
97
180
  _exitHandlersInstalled = true;
181
+ const fallbackExitMs = options.fallbackExitMs ?? LOCK_SIGNAL_FALLBACK_EXIT_MS;
98
182
  function tryRelease() {
99
183
  try {
100
184
  lockfile.unlockSync(authLockPath());
@@ -104,18 +188,46 @@ export function installLockExitHandlers() {
104
188
  }
105
189
  }
106
190
  process.on('exit', tryRelease);
107
- process.on('SIGTERM', () => {
108
- tryRelease();
109
- process.exit(0);
110
- });
111
- process.on('SIGINT', () => {
191
+ const exitStatus = (signal) => 128 + LOCK_RELEASE_SIGNALS[signal];
192
+ let deferred = false;
193
+ const listeners = new Map();
194
+ // `process.exit` runs the `exit` listener above, which releases the lock;
195
+ // releasing here as well costs nothing and does not depend on that.
196
+ const exitNow = (signal) => {
112
197
  tryRelease();
113
- process.exit(130);
114
- });
115
- process.on('SIGHUP', () => {
198
+ process.exit(exitStatus(signal));
199
+ };
200
+ const onSignal = (signal) => {
201
+ if (deferred) {
202
+ // A later signal while owners are shutting down: stop waiting for them.
203
+ setImmediate(() => exitNow(signal));
204
+ return;
205
+ }
206
+ const owners = process.listenerCount(signal) - 1 - signalExitListenerCount();
207
+ if (owners > 0) {
208
+ deferred = true;
209
+ setTimeout(() => exitNow(signal), fallbackExitMs).unref();
210
+ return;
211
+ }
212
+ process.removeListener(signal, listeners.get(signal));
213
+ if (process.platform === 'win32') {
214
+ exitNow(signal);
215
+ return;
216
+ }
217
+ // Dying by the signal runs no `exit` listener, so release first.
116
218
  tryRelease();
117
- process.exit(1);
118
- });
219
+ try {
220
+ process.kill(process.pid, signal);
221
+ }
222
+ catch {
223
+ exitNow(signal);
224
+ }
225
+ };
226
+ for (const signal of Object.keys(LOCK_RELEASE_SIGNALS)) {
227
+ const listener = () => onSignal(signal);
228
+ listeners.set(signal, listener);
229
+ process.prependListener(signal, listener);
230
+ }
119
231
  }
120
232
  /**
121
233
  * Acquire the cross-process auth lock, run the body, release on `finally`.