shraga 0.1.31 → 0.1.33

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,10 +1,10 @@
1
- import { loadSchedules, saveSchedules, readCompletionMarker, writeCompletionMarker, readRunningMarker, isProcessAlive, loadThrottleState, saveThrottleState } from './storage.ts';
1
+ import { loadSchedules, saveSchedules, readCompletionMarker, writeCompletionMarker, readRunningMarker, isProcessAlive, loadThrottleState, saveThrottleState, acquireRunLock, clearRunningMarker, markRunStarted } from './storage.ts';
2
2
  import { computeNextRun, computePrevRun, validateTrigger } from './timing.ts';
3
3
  import { runSchedule, type ResumeOptions, type EventContext } from './runner.ts';
4
4
  import { backfillScope, ensureBuiltinSchedules } from './builtins.ts';
5
5
  import { emitEvent } from '../events/bus.ts';
6
6
  import { getSessionUrl } from '../shraga-config.ts';
7
- import type { Schedule } from './types.ts';
7
+ import type { Schedule, MissedPolicy, RunOutcome, RunRefusal } from './types.ts';
8
8
 
9
9
  type Broadcast = (data: object) => void;
10
10
 
@@ -26,6 +26,18 @@ interface RuntimeState {
26
26
  }
27
27
 
28
28
  const QUEUE_CAP = 5;
29
+ /** Catch-up fires are delayed so MCP servers finish initializing first. Read per call so tests
30
+ * (and an operator) can shorten it. */
31
+ const catchupDelayMs = () => Number(process.env.SCHEDULER_CATCHUP_DELAY_MS ?? 10_000);
32
+ /** Hard ceiling on how late a missed window may still be replayed, regardless of `onMissed`.
33
+ * A missed window is never more than one period old, so a period-relative bound alone can never
34
+ * suppress anything — yet replaying an 08:00 report at 22:00 is not the job the schedule
35
+ * describes. This absolute cap is therefore the entire rule.
36
+ *
37
+ * Deliberately NOT `min(period, cap)`: a cron missed window comes from `computePrevRun`, so its
38
+ * age is always strictly less than one period — a period-relative term can never suppress
39
+ * anything, and `min(period, cap)` is provably identical to `cap`. */
40
+ const maxMissedAgeMs = () => Number(process.env.SCHEDULER_MAX_MISSED_AGE_MS ?? 6 * 60 * 60 * 1000);
29
41
  /** Max setTimeout delay (2^31-1 ms ≈ 24.8 days); longer delays overflow and fire immediately. */
30
42
  const MAX_TIMER_MS = 2_147_483_647;
31
43
  /** Only the designated instance fires schedules (DATA_SYNC_SCHEDULER_ACTIVE=true).
@@ -71,38 +83,55 @@ export function start(broadcast: Broadcast): void {
71
83
  }
72
84
  }
73
85
  // Catch up missed cron fires (e.g. process was down when cron should have fired)
74
- const catchUps: string[] = [];
86
+ const catchUps: { id: string; window: number }[] = [];
75
87
  for (const s of state.schedules) {
76
88
  if (!s.enabled || s.trigger.kind !== 'cron') continue;
77
89
  const prev = computePrevRun(s.trigger);
78
90
  if (prev === null) continue;
79
91
  const lastAt = s.lastRun?.at;
80
92
  if (lastAt === undefined) continue; // never ran — nothing to catch up
81
- if (lastAt < prev) {
82
- const marker = readCompletionMarker(s.id);
83
- if (marker && marker.completedAt >= prev) {
84
- console.log(`[scheduler] skipping catch-up for ${s.id} — already completed at ${new Date(marker.completedAt).toISOString()} by ${marker.triggeredBy}`);
85
- continue;
86
- }
87
- const running = readRunningMarker(s.id);
88
- if (running && isProcessAlive(running.pid)) {
89
- console.log(`[scheduler] skipping catch-up for ${s.id} — still running (pid ${running.pid}, started ${new Date(running.startedAt).toISOString()})`);
90
- continue;
91
- }
92
- catchUps.push(s.id);
93
+ if (lastAt >= prev) continue;
94
+
95
+ const marker = readCompletionMarker(s.id);
96
+ if (marker && marker.completedAt >= prev) {
97
+ console.log(`[scheduler] skipping catch-up for ${s.id} — already completed at ${new Date(marker.completedAt).toISOString()} by ${marker.triggeredBy}`);
98
+ continue;
99
+ }
100
+ // An ATTEMPT on this window counts too: a run that errored or was killed already had its
101
+ // shot. Replaying it on the next boot is exactly the duplicate-fire this guards.
102
+ if (marker?.attemptWindow !== undefined && marker.attemptWindow >= prev) {
103
+ console.log(`[scheduler] skipping catch-up for ${s.id} — window ${new Date(prev).toISOString()} already attempted (${marker.status ?? 'started'})`);
104
+ continue;
105
+ }
106
+ // Someone is already on it — either another instance, or (far more common) the recovery
107
+ // path that resumed the interrupted run in-place moments ago.
108
+ const running = readRunningMarker(s.id);
109
+ if (running && isProcessAlive(running.pid)) {
110
+ console.log(`[scheduler] skipping catch-up for ${s.id} — still running (pid ${running.pid}, started ${new Date(running.startedAt).toISOString()})`);
111
+ continue;
93
112
  }
113
+
114
+ const verdict = judgeMissed(s, prev);
115
+ if (!verdict.replay) {
116
+ console.log(`[scheduler] not replaying ${s.id} — ${verdict.message}`);
117
+ noteMissed(s, prev, verdict.reason);
118
+ continue;
119
+ }
120
+ catchUps.push({ id: s.id, window: prev });
94
121
  }
95
122
  if (catchUps.length) {
96
- console.log(`[scheduler] catch-up: ${catchUps.join(', ')} (delayed 10s for MCP init)`);
123
+ console.log(`[scheduler] catch-up: ${catchUps.map((c) => c.id).join(', ')} (delayed ${catchupDelayMs()}ms for MCP init)`);
97
124
  setTimeout(() => {
98
- for (const id of catchUps) {
125
+ for (const { id, window } of catchUps) {
99
126
  const s = getSchedule(id);
100
- if (s?.enabled) {
101
- console.log(`[scheduler] catch-up: firing ${id}`);
102
- runNow(id);
103
- }
127
+ if (!s?.enabled) continue;
128
+ console.log(`[scheduler] catch-up: firing ${id}`);
129
+ // NOT runNow(): a catch-up is not a manual run. It must go through the same
130
+ // completion/attempt/lock guards as any timer fire, so that whatever claimed this
131
+ // window first (typically the in-place resume) makes this a no-op.
132
+ enqueueFire(s, window);
104
133
  }
105
- }, 10_000);
134
+ }, catchupDelayMs());
106
135
  }
107
136
 
108
137
  saveSchedules(state.schedules);
@@ -125,6 +154,9 @@ export function getSchedule(id: string): Schedule | undefined {
125
154
  export function upsertSchedule(s: Schedule): { ok: true; schedule: Schedule } | { ok: false; error: string } {
126
155
  const err = validateTrigger(s.trigger);
127
156
  if (err) return { ok: false, error: err };
157
+ if (s.onMissed !== undefined && !MISSED_POLICIES.includes(s.onMissed)) {
158
+ return { ok: false, error: `Invalid onMissed "${s.onMissed}" (expected ${MISSED_POLICIES.join(' | ')})` };
159
+ }
128
160
 
129
161
  s.updatedAt = Date.now();
130
162
  if (s.enabled) {
@@ -180,9 +212,9 @@ export function toggleSchedule(id: string, enabled: boolean): Schedule | null {
180
212
  return s;
181
213
  }
182
214
 
183
- export function runNow(id: string, override?: string): string | null {
215
+ export function runNow(id: string, override?: string): RunOutcome {
184
216
  const s = getSchedule(id);
185
- if (!s) return null;
217
+ if (!s) return refuse('unknown-schedule', `No schedule ${id}`);
186
218
  return enqueueFire(s, Date.now(), override, true);
187
219
  }
188
220
 
@@ -275,22 +307,133 @@ export function cancelRun(id: string): boolean {
275
307
  * side-effects, e.g. a second Slack post). Mirrors the web/slack restart-resume path.
276
308
  * No-op if the schedule is unknown or already running.
277
309
  */
278
- export function resumeRun(scheduleId: string, sessionId: string, prompt: string): string | null {
310
+ export function resumeRun(scheduleId: string, sessionId: string, prompt: string): RunOutcome {
279
311
  const s = getSchedule(scheduleId);
280
312
  if (!s) {
281
313
  console.warn(`[scheduler] resumeRun: unknown schedule ${scheduleId}`);
282
- return null;
314
+ return refuse('unknown-schedule', `No schedule ${scheduleId}`);
283
315
  }
284
316
  if (state.running.has(scheduleId)) {
285
317
  console.log(`[scheduler] resumeRun: ${scheduleId} already running — skipping resume`);
286
- return null;
318
+ return refuse('already-running', `${s.name} is already running`);
319
+ }
320
+ const window = interruptedWindow(s);
321
+ // A resume continues an existing conversation rather than starting a fresh one, so it dodges
322
+ // the duplicate-side-effect problem — but it does NOT dodge the "is this work still the work
323
+ // the schedule asked for" problem. Finishing the 08:00 report at 22:00 is the incident. So the
324
+ // resume path answers to the same onMissed policy and the same staleness ceiling as catch-up —
325
+ // for EVERY trigger kind, not just cron (interval/once/event have no catch-up path at all, so
326
+ // resume is their only gate).
327
+ if (window !== null) {
328
+ // Someone already finished this window — typically a catch-up that won the boot race and has
329
+ // since completed, so the run lock it held is gone. Resuming now would redo work that is
330
+ // already done: the duplicate fire, one step later. (An *attempt* on this window is NOT a
331
+ // refusal — the interrupted run we are resuming recorded one itself.)
332
+ const marker = readCompletionMarker(scheduleId);
333
+ if (marker && marker.completedAt >= window) {
334
+ const msg = `window ${new Date(window).toISOString()} already completed at ${new Date(marker.completedAt).toISOString()} by ${marker.triggeredBy}`;
335
+ console.log(`[scheduler] not resuming ${scheduleId} — ${msg}`);
336
+ return refuse('already-completed', msg);
337
+ }
338
+ const verdict = judgeMissed(s, window);
339
+ if (!verdict.replay) {
340
+ console.log(`[scheduler] not resuming ${scheduleId} — ${verdict.message}`);
341
+ noteMissed(s, window, verdict.reason);
342
+ return refuse(verdict.refusal, verdict.message);
343
+ }
287
344
  }
288
345
  console.log(`[scheduler] resuming ${scheduleId} in-place on session ${sessionId.slice(0, 30)}…`);
289
- return startRun(s, Date.now(), undefined, { sessionId, prompt });
346
+ return startRun(s, window ?? Date.now(), undefined, { sessionId, prompt });
290
347
  }
291
348
 
292
349
  // ── Internals ───────────────────────────────────────────────────────────────
293
350
 
351
+ const MISSED_POLICIES: MissedPolicy[] = ['run', 'skip', 'offer'];
352
+
353
+ function refuse(reason: RunRefusal, message: string): RunOutcome {
354
+ return { ok: false, reason, message };
355
+ }
356
+
357
+ /**
358
+ * The window the interrupted run was covering — what "how late is this?" is measured against.
359
+ *
360
+ * For `cron` the window is derivable from the expression itself (`computePrevRun`), and that is
361
+ * preferred: it is correct even for legacy state written before markers recorded a window.
362
+ *
363
+ * `interval`/`once`/`event` have no schedule-derivable grid — but they are not therefore timeless.
364
+ * An interval run interrupted 14h ago is exactly as stale as a cron one, and it has no catch-up
365
+ * path to be caught by. So for those the window is the one the interrupted run ITSELF claimed,
366
+ * read back off the markers already on disk: the run lock's `window` (stamped by `acquireRunLock`
367
+ * at fire time), else its `startedAt`, else the attempt ledger, else the at-start `lastRun.at`.
368
+ * Only a schedule with no trace of ever having started has no window — and nothing to resume.
369
+ */
370
+ function interruptedWindow(s: Schedule): number | null {
371
+ const prev = computePrevRun(s.trigger);
372
+ if (prev !== null) return prev;
373
+ const running = readRunningMarker(s.id);
374
+ if (running?.window !== undefined) return running.window;
375
+ if (running?.startedAt) return running.startedAt;
376
+ const marker = readCompletionMarker(s.id);
377
+ if (marker?.attemptWindow !== undefined) return marker.attemptWindow;
378
+ if (marker?.lastAttemptAt) return marker.lastAttemptAt;
379
+ return s.lastRun?.at ?? null;
380
+ }
381
+
382
+ /** Persist the terminal outcome of an attempt without touching the last SUCCESSFUL completion.
383
+ * Every non-ok exit — reported failure or unexpected throw — must land here, else the marker
384
+ * stays `started` forever and a crash is indistinguishable from a throw. */
385
+ function recordAttemptOutcome(scheduleId: string, firedAt: number, at: number, status: 'error' | 'aborted' | 'started'): void {
386
+ const prev = readCompletionMarker(scheduleId);
387
+ writeCompletionMarker({
388
+ completedAt: prev?.completedAt ?? 0,
389
+ triggeredBy: 'scheduler',
390
+ scheduleId,
391
+ lastAttemptAt: prev?.lastAttemptAt ?? at,
392
+ attemptWindow: prev?.attemptWindow ?? firedAt,
393
+ status,
394
+ });
395
+ }
396
+
397
+ /** Default is `run`: it is what every existing schedule already does, and schedules.json has no
398
+ * `onMissed` field on any of them — defaulting to anything else would silently change the
399
+ * behaviour of live automations on upgrade. The unbounded-replay hazard that made `run`
400
+ * dangerous is fixed by the staleness ceiling below, which applies to `run` too. */
401
+ function missedPolicy(s: Schedule): MissedPolicy {
402
+ return s.onMissed ?? 'run';
403
+ }
404
+
405
+ type MissedVerdict =
406
+ | { replay: true }
407
+ | { replay: false; reason: 'skip' | 'offer' | 'stale'; refusal: 'policy-skip' | 'policy-offer' | 'stale'; message: string };
408
+
409
+ /**
410
+ * The one decision behind "this window already elapsed — may it still run?", shared by all three
411
+ * replay paths: the boot catch-up scan, the in-place resume, and the late timer fire.
412
+ *
413
+ * `onMissed` first (an explicit skip/offer is a standing instruction, not an age question), then
414
+ * the absolute staleness ceiling, which overrides even `run`. Callers do the recording, because
415
+ * only they know how to refuse (a return value, or `continue`).
416
+ */
417
+ function judgeMissed(s: Schedule, window: number, now: number = Date.now()): MissedVerdict {
418
+ const when = new Date(window).toISOString();
419
+ const policy = missedPolicy(s);
420
+ if (policy !== 'run') {
421
+ return { replay: false, reason: policy, refusal: policy === 'skip' ? 'policy-skip' : 'policy-offer', message: `window ${when} was missed and onMissed=${policy}` };
422
+ }
423
+ const age = now - window;
424
+ const grace = maxMissedAgeMs();
425
+ if (age > grace) {
426
+ return { replay: false, reason: 'stale', refusal: 'stale', message: `window ${when} is ${Math.round(age / 60_000)}m stale (ceiling ${Math.round(grace / 60_000)}m)` };
427
+ }
428
+ return { replay: true };
429
+ }
430
+
431
+ function noteMissed(s: Schedule, at: number, reason: 'skip' | 'offer' | 'stale'): void {
432
+ s.missedRun = { at, reason, noticedAt: Date.now() };
433
+ saveSchedules(state.schedules);
434
+ state.broadcast({ type: 'schedule:updated', schedule: s });
435
+ }
436
+
294
437
  function replan(): void {
295
438
  if (!schedulerActive) return;
296
439
  if (state.timer) { clearTimeout(state.timer); state.timer = null; }
@@ -317,7 +460,24 @@ function fireDue(): void {
317
460
  const now = Date.now();
318
461
  for (const s of state.schedules) {
319
462
  if (!s.enabled || s.nextRun === undefined) continue;
320
- if (s.nextRun <= now) enqueueFire(s, s.nextRun);
463
+ if (s.nextRun > now) continue;
464
+ // A timer fire is normally punctual, so it is NOT treated as a missed window — `onMissed`
465
+ // must never gate healthy operation. But a setTimeout does not survive a wall-clock jump:
466
+ // when the host sleeps (lid closed) or NTP steps the clock, the process is FROZEN, not
467
+ // killed — no restart, so the boot catch-up scan never runs — and the overdue timer fires
468
+ // the elapsed window the moment the machine wakes. Firing an 08:00 job at 21:36 is exactly
469
+ // what the catch-up ceiling exists to prevent, reached by the one path it didn't cover.
470
+ // Past the ceiling we hand the window to the same judgement the other two paths use.
471
+ // Past the ceiling `judgeMissed` can only refuse (it replays a `run` window only while it is
472
+ // within the ceiling), so there is no replay branch here — the shared decision is used for WHAT
473
+ // to record, not whether to fire.
474
+ if (now - s.nextRun > maxMissedAgeMs()) {
475
+ const verdict = judgeMissed(s, s.nextRun, now) as Extract<MissedVerdict, { replay: false }>;
476
+ console.log(`[scheduler] not firing ${s.id} — ${verdict.message} (timer fired late; host suspended or clock jumped)`);
477
+ noteMissed(s, s.nextRun, verdict.reason);
478
+ continue; // the advance loop below still re-arms / retires this schedule
479
+ }
480
+ enqueueFire(s, s.nextRun);
321
481
  }
322
482
  // Advance nextRun for recurring triggers; disable fired `once` triggers
323
483
  for (const s of state.schedules) {
@@ -336,21 +496,22 @@ function fireDue(): void {
336
496
  replan();
337
497
  }
338
498
 
339
- function enqueueFire(s: Schedule, firedAt: number, override?: string, manual = false, eventCtx?: EventContext): string | null {
340
- // Skip if this cron period was already completed or still running from a prior server instance.
341
- // Manual runs (runNow from UI/API) always proceed.
499
+ function enqueueFire(s: Schedule, firedAt: number, override?: string, manual = false, eventCtx?: EventContext): RunOutcome {
500
+ // Skip if this cron period was already completed/attempted. Manual runs (runNow from UI/API)
501
+ // always proceed past the period guard — but never past the run lock in startRun().
342
502
  if (!manual && s.trigger.kind === 'cron') {
343
503
  const prev = computePrevRun(s.trigger, firedAt + 1);
344
504
  if (prev !== null) {
345
505
  const marker = readCompletionMarker(s.id);
346
506
  if (marker && marker.completedAt >= prev) {
347
- console.log(`[scheduler] skipping fire for ${s.id} — already completed this period (at ${new Date(marker.completedAt).toISOString()})`);
348
- return null;
507
+ const msg = `already completed this period (at ${new Date(marker.completedAt).toISOString()})`;
508
+ console.log(`[scheduler] skipping fire for ${s.id} — ${msg}`);
509
+ return refuse('already-completed', msg);
349
510
  }
350
- const running = readRunningMarker(s.id);
351
- if (running && isProcessAlive(running.pid)) {
352
- console.log(`[scheduler] skipping fire for ${s.id} — still running (pid ${running.pid})`);
353
- return null;
511
+ if (marker?.attemptWindow !== undefined && marker.attemptWindow >= prev) {
512
+ const msg = `this period was already attempted (${marker.status ?? 'started'})`;
513
+ console.log(`[scheduler] skipping fire for ${s.id} — ${msg}`);
514
+ return refuse('already-attempted', msg);
354
515
  }
355
516
  }
356
517
  }
@@ -369,10 +530,36 @@ function enqueueFire(s: Schedule, firedAt: number, override?: string, manual = f
369
530
  console.warn(`[scheduler] queue overflow for ${s.id} (cap=${QUEUE_CAP}), dropped fire @ ${dropped?.firedAt}`);
370
531
  }
371
532
  state.queues.set(s.id, q);
372
- return null;
533
+ return { ok: true, sessionId: null, queued: true };
373
534
  }
374
535
 
375
- function startRun(s: Schedule, _firedAt: number, override?: string, resume?: ResumeOptions, eventCtx?: EventContext): string | null {
536
+ /**
537
+ * Start one run, holding the cross-restart run lock for its whole life.
538
+ *
539
+ * Every path that can start a run funnels through here — timer fire, catch-up, event, manual
540
+ * runNow, and the in-place resume from crash recovery — so the lock is the single place that
541
+ * enforces "one live run per schedule". It is also what makes catch-up and recovery mutually
542
+ * exclusive: whichever reaches this first holds a live pid, and the other's acquire fails.
543
+ */
544
+ function startRun(s: Schedule, firedAt: number, override?: string, resume?: ResumeOptions, eventCtx?: EventContext): RunOutcome {
545
+ const lock = acquireRunLock(s.id, firedAt);
546
+ if (!lock) {
547
+ const held = readRunningMarker(s.id);
548
+ const msg = `run lock held by pid ${held?.pid} since ${new Date(held?.startedAt ?? 0).toISOString()}`;
549
+ console.log(`[scheduler] not starting ${s.id} — ${msg}`);
550
+ return refuse('locked', msg);
551
+ }
552
+ // Record the attempt BEFORE running: a crash from here on must not look like "never tried".
553
+ markRunStarted(s.id, firedAt);
554
+ const pre = getSchedule(s.id);
555
+ if (pre) {
556
+ // Advance lastRun at start, not only on success. start() turns a leftover 'running' into
557
+ // 'error' on the next boot, so the distinction survives while the timestamp still blocks a
558
+ // replay of this window.
559
+ pre.lastRun = { at: Date.now(), sessionId: resume?.sessionId ?? '', status: 'running' };
560
+ pre.missedRun = undefined;
561
+ saveSchedules(state.schedules);
562
+ }
376
563
  // Deep-copy task so edits mid-run don't affect the in-flight execution
377
564
  const snapshot: Schedule = JSON.parse(JSON.stringify(s));
378
565
  let sessionId: string | null = null;
@@ -380,6 +567,14 @@ function startRun(s: Schedule, _firedAt: number, override?: string, resume?: Res
380
567
  const register = (sid: string, ac: AbortController) => {
381
568
  sessionId = sid;
382
569
  state.running.set(s.id, ac);
570
+ // Backfill the session link onto the at-start lastRun. Without this a crash mid-run persists
571
+ // an errored run with sessionId '' — no way back to the conversation that was interrupted,
572
+ // which is exactly what the recovery path needs.
573
+ const live = getSchedule(s.id);
574
+ if (live?.lastRun && live.lastRun.status === 'running' && !live.lastRun.sessionId) {
575
+ live.lastRun.sessionId = sid;
576
+ saveSchedules(state.schedules);
577
+ }
383
578
  };
384
579
 
385
580
  state.broadcast({ type: 'schedule:fired', scheduleId: s.id });
@@ -393,8 +588,13 @@ function startRun(s: Schedule, _firedAt: number, override?: string, resume?: Res
393
588
  saveSchedules(state.schedules);
394
589
  state.broadcast({ type: 'schedule:updated', schedule: live });
395
590
  }
591
+ if (summary.status !== 'ok') {
592
+ // Keep the attempt on record with its real outcome — the next boot must see that this
593
+ // window was tried and failed, not that it never ran.
594
+ recordAttemptOutcome(s.id, firedAt, summary.at, summary.status === 'running' ? 'started' : summary.status);
595
+ }
396
596
  if (summary.status === 'ok') {
397
- writeCompletionMarker({ completedAt: summary.at, triggeredBy: 'scheduler', scheduleId: s.id });
597
+ writeCompletionMarker({ completedAt: summary.at, triggeredBy: 'scheduler', scheduleId: s.id, lastAttemptAt: summary.at, attemptWindow: firedAt, status: 'ok' });
398
598
  if (live && live.trigger.kind === 'once') {
399
599
  console.log(`[scheduler] auto-deleting completed once-schedule ${s.id}`);
400
600
  deleteSchedule(s.id);
@@ -422,9 +622,23 @@ function startRun(s: Schedule, _firedAt: number, override?: string, resume?: Res
422
622
  })
423
623
  .catch((err) => {
424
624
  console.error(`[scheduler] unexpected run failure for ${s.id}:`, err);
625
+ // A rejection never reaches the .then above, so without this the attempt marker stays
626
+ // 'started' forever and an unexpected throw is indistinguishable from a power cut.
627
+ const at = Date.now();
628
+ recordAttemptOutcome(s.id, firedAt, at, 'error');
629
+ const live = getSchedule(s.id);
630
+ if (live) {
631
+ live.lastRun = { at, sessionId: live.lastRun?.sessionId ?? '', status: 'error', error: `unexpected run failure: ${err?.message ?? String(err)}` };
632
+ saveSchedules(state.schedules);
633
+ state.broadcast({ type: 'schedule:updated', schedule: live });
634
+ }
425
635
  })
426
636
  .finally(() => {
427
637
  state.running.delete(s.id);
638
+ // Release the lock on EVERY exit path — ok, error, abort, or an unexpected throw. The
639
+ // runner already clears it on its own terminal states; this is idempotent and covers the
640
+ // paths that never reach the runner's finally.
641
+ clearRunningMarker(s.id);
428
642
  const q = state.queues.get(s.id);
429
643
  if (q && q.length > 0) {
430
644
  const next = q.shift()!;
@@ -434,5 +648,5 @@ function startRun(s: Schedule, _firedAt: number, override?: string, resume?: Res
434
648
  }
435
649
  });
436
650
 
437
- return sessionId;
651
+ return { ok: true, sessionId };
438
652
  }
@@ -6,7 +6,7 @@ import { streamChat, type PermissionHandler } from '../claude.ts';
6
6
  import { getMcpConfig } from '../mcp.ts';
7
7
  import { appendMessage, createScheduledSession, updateScheduledSessionStatus, setRunStatus, registerLivePartial, unregisterLivePartial, writePartial, clearPartial, acquireSessionLock, releaseSessionLock, type ConvBlock } from '../sessions.ts';
8
8
  import type { Schedule, ScheduleRunSummary } from './types.ts';
9
- import { writeRunningMarker, clearRunningMarker } from './storage.ts';
9
+ import { updateRunLockPid, clearRunningMarker } from './storage.ts';
10
10
  import { addUnread } from '../unread.ts';
11
11
 
12
12
  export interface RunContext {
@@ -114,8 +114,8 @@ export async function runSchedule(
114
114
  acquireSessionLock(sessionId, 'scheduler', abortController);
115
115
  setRunStatus(sessionId, 'running', 'scheduler');
116
116
  onEvent({ type: 'session_busy', sessionId, busy: true });
117
- // Mark this period as running with the live pid so startup catch-up won't double-fire it.
118
- writeRunningMarker({ pid: process.pid, startedAt: now, scheduleId: schedule.id });
117
+ // The run lock (running marker) is acquired by the engine BEFORE this point — see
118
+ // engine.startRun. Writing it here too would let a direct runSchedule() call bypass the lock.
119
119
 
120
120
  const task = schedule.task;
121
121
  if (task.kind === 'job') {
@@ -414,9 +414,9 @@ function runCommandWithMarker(command: string, abortController: AbortController,
414
414
  stdio: ['ignore', 'pipe', 'pipe'],
415
415
  });
416
416
 
417
- if (child.pid) {
418
- writeRunningMarker({ pid: child.pid, startedAt: Date.now(), scheduleId });
419
- }
417
+ // Re-point the held lock at the child: the job outlives nothing here, but if the server dies
418
+ // the child may still be alive, and a live pid must keep the schedule locked.
419
+ if (child.pid) updateRunLockPid(scheduleId, child.pid);
420
420
 
421
421
  let output = '';
422
422
  const handleData = (chunk: Buffer) => {
@@ -96,3 +96,77 @@ export function clearRunningMarker(scheduleId: string): void {
96
96
  export function isProcessAlive(pid: number): boolean {
97
97
  try { process.kill(pid, 0); return true; } catch { return false; }
98
98
  }
99
+
100
+ /** How long a held run lock is believed, before it is treated as abandoned regardless of whether
101
+ * its pid answers. Bounds the blast radius of pid reuse: after a power cut the OS restarts pid
102
+ * allocation low, so a persisted pid can plausibly be live again under the same uid — and a
103
+ * liveness check alone would then wedge the schedule forever. Generous vs any real run (agent
104
+ * runs are minutes, not hours) while capping the wedge at one window's worth of a daily job. */
105
+ const runLockMaxAgeMs = () => Number(process.env.SCHEDULER_RUN_LOCK_MAX_AGE_MS ?? 6 * 60 * 60 * 1000);
106
+
107
+ /**
108
+ * Claim the single live-run slot for a schedule.
109
+ *
110
+ * The lock IS the running marker — same file, same conventions — so it survives a process
111
+ * restart: after a crash the marker is still on disk but its pid is dead, and a dead pid is
112
+ * reclaimable (otherwise a power cut would wedge the schedule forever). A LIVE pid, ours or
113
+ * another instance's, means someone is already running this schedule: the caller must back off —
114
+ * UNLESS the claim is older than `runLockMaxAgeMs`, which is the escape hatch for a lock wedged
115
+ * by pid reuse. A live, in-ceiling claim is never stolen, not even by a manual run: on this
116
+ * single-active-instance design that pid is a real run (an agent session, or a spawned job the
117
+ * lock was re-pointed at), and starting a second one is the duplicate-fire we are preventing.
118
+ *
119
+ * Single-writer by design (only the DATA_SYNC_SCHEDULER_ACTIVE instance fires), so this is a
120
+ * read-then-write, not an atomic CAS.
121
+ */
122
+ export function acquireRunLock(scheduleId: string, window: number): RunningMarker | null {
123
+ const existing = readRunningMarker(scheduleId);
124
+ if (existing) {
125
+ const alive = isProcessAlive(existing.pid);
126
+ const age = Date.now() - (Number.isFinite(existing.startedAt) ? existing.startedAt : 0);
127
+ const maxAge = runLockMaxAgeMs();
128
+ if (alive && age <= maxAge) return null;
129
+ const why = !alive
130
+ ? `dead pid ${existing.pid}`
131
+ : `held ${Math.round(age / 60_000)}m by live pid ${existing.pid}, past the ${Math.round(maxAge / 60_000)}m lock ceiling — assuming pid reuse`;
132
+ console.log(`[scheduler] reclaiming run lock for ${scheduleId} — ${why} (window ${new Date(existing.window ?? existing.startedAt).toISOString()})`);
133
+ }
134
+ const marker: RunningMarker = { pid: process.pid, startedAt: Date.now(), scheduleId, window };
135
+ writeRunningMarker(marker);
136
+ return marker;
137
+ }
138
+
139
+ /** Re-point a held lock at a spawned child process. Only ever touches a lock THIS process holds,
140
+ * and preserves `startedAt` so re-pointing can't refresh the age ceiling above. */
141
+ export function updateRunLockPid(scheduleId: string, pid: number): void {
142
+ const existing = readRunningMarker(scheduleId);
143
+ if (!existing) {
144
+ console.warn(`[scheduler] updateRunLockPid(${scheduleId}): no lock held — not creating one`);
145
+ return;
146
+ }
147
+ if (existing.pid !== process.pid) {
148
+ console.warn(`[scheduler] updateRunLockPid(${scheduleId}): lock is held by pid ${existing.pid}, not us — leaving it alone`);
149
+ return;
150
+ }
151
+ writeRunningMarker({ ...existing, scheduleId, pid });
152
+ }
153
+
154
+ /**
155
+ * Record that a run STARTED for `window`, before it can succeed or fail.
156
+ *
157
+ * Without this a run that errors or is killed leaves no trace on disk, so the next boot sees an
158
+ * un-completed window and replays it — the double-fire this ledger exists to prevent. The
159
+ * successful-completion timestamp is preserved untouched so "started" stays distinguishable
160
+ * from "completed ok".
161
+ */
162
+ export function markRunStarted(scheduleId: string, window: number, triggeredBy: CompletionMarker['triggeredBy'] = 'scheduler'): void {
163
+ const prev = readCompletionMarker(scheduleId);
164
+ writeCompletionMarker({
165
+ completedAt: prev?.completedAt ?? 0,
166
+ triggeredBy,
167
+ scheduleId,
168
+ lastAttemptAt: Date.now(),
169
+ attemptWindow: window,
170
+ status: 'started',
171
+ });
172
+ }
@@ -34,16 +34,77 @@ export interface ScheduleRunSummary {
34
34
  error?: string;
35
35
  }
36
36
 
37
+ /** What the scheduler did about a window it found already elapsed at boot.
38
+ * - `run` replay it (today's behaviour, and the default — see engine.ts).
39
+ * - `skip` never replay; the window is simply lost.
40
+ * - `offer` don't auto-fire, but record it on the schedule (`missedRun`) so a human can
41
+ * see it was missed and run it on demand. */
42
+ export type MissedPolicy = 'run' | 'skip' | 'offer';
43
+
44
+ /** A window that elapsed while nothing was running and was NOT replayed.
45
+ * Persisted on the schedule and returned verbatim by `GET /api/schedules[/:id]` — that read
46
+ * path IS the affordance `offer` promises: a human (or the agent, via the scheduler skill) sees
47
+ * the missed window and can `POST /api/schedules/:id/run` it on demand. */
48
+ export interface MissedRun {
49
+ /** The window (fire time) that was missed. */
50
+ at: number;
51
+ reason: 'skip' | 'offer' | 'stale';
52
+ /** When the scheduler noticed — i.e. boot time. Read by whoever acts on the miss: `at` alone
53
+ * can't tell "missed 20m ago, still worth running" from "found on a boot two days later". */
54
+ noticedAt: number;
55
+ }
56
+
57
+ /** Why a start request did not start a run. */
58
+ export type RunRefusal =
59
+ | 'unknown-schedule'
60
+ | 'already-running'
61
+ | 'already-completed'
62
+ | 'already-attempted'
63
+ | 'policy-skip'
64
+ | 'policy-offer'
65
+ | 'stale'
66
+ | 'locked';
67
+
68
+ /**
69
+ * Result of asking the engine to start a run (timer fire, catch-up, manual, event, resume).
70
+ * A refusal carries WHY, so the caller can record a truthful error and map a real HTTP status
71
+ * instead of collapsing every outcome into "null".
72
+ */
73
+ export type RunOutcome =
74
+ | { ok: true; sessionId: string | null; queued?: boolean }
75
+ | { ok: false; reason: RunRefusal; message: string };
76
+
77
+ /**
78
+ * Per-schedule attempt/completion ledger. `completedAt` records the last SUCCESSFUL run;
79
+ * `lastAttemptAt`/`attemptWindow` record the last run that STARTED, whether or not it finished.
80
+ * Both are needed: completion drives "this period is done", attempt drives "this period was
81
+ * already tried, don't replay it after a crash".
82
+ */
37
83
  export interface CompletionMarker {
84
+ /** Last successful completion; 0 when the schedule has never completed. */
38
85
  completedAt: number;
39
86
  triggeredBy: 'scheduler' | 'ssh' | 'api' | 'manual';
40
87
  scheduleId: string;
88
+ /** Wall-clock time the last attempt began. */
89
+ lastAttemptAt?: number;
90
+ /** The window (fire time) that attempt was covering. */
91
+ attemptWindow?: number;
92
+ /** Terminal state of the last attempt; `started` means it never reported back (crash). */
93
+ status?: 'started' | 'ok' | 'error' | 'aborted';
41
94
  }
42
95
 
96
+ /**
97
+ * The run lock: at most one live run per scheduleId, across process restarts.
98
+ * Written by `acquireRunLock` before a run starts, removed when it reaches a terminal state.
99
+ * A lock whose `pid` is dead (crash / power loss) is reclaimable — see `acquireRunLock`.
100
+ */
43
101
  export interface RunningMarker {
44
102
  pid: number;
45
103
  startedAt: number;
46
104
  scheduleId: string;
105
+ /** The window (fire time) this run covers — lets a claimer see WHAT is held, not just that
106
+ * something is. */
107
+ window?: number;
47
108
  }
48
109
 
49
110
  export interface Schedule {
@@ -59,6 +120,10 @@ export interface Schedule {
59
120
  nextRun?: number;
60
121
  lastRun?: ScheduleRunSummary;
61
122
  runCount: number;
123
+ /** What to do with a window that elapsed while the process was down. Absent ⇒ 'run'. */
124
+ onMissed?: MissedPolicy;
125
+ /** Last window that elapsed and was deliberately not replayed (policy or staleness). */
126
+ missedRun?: MissedRun;
62
127
  /** Set when a data-plane module owns this schedule (module name). Module reconcile
63
128
  * updates trigger/task; enable/disable snapshots live in the module's state entry. */
64
129
  managedBy?: string;
@@ -21,6 +21,7 @@ import { pipeAgentReply, type AgentEvent, type IngressMessage } from 'mcp-slack-
21
21
  import { makeSlackQuestionHandler } from './questions.ts';
22
22
  import * as contacts from '../contacts.ts';
23
23
  import { getChannelContext, invalidateChannelContext } from './context-cache.ts';
24
+ import { noteSlackSeen } from '../downtime.ts';
24
25
  import { getOrCreateSession, registerThreadAlias, setLastMessageTs, setUseUserToken, findSlackSessionBySessionId, getProactiveOrigin, hasSessionForThread, isSlackBotPlaceholderEmail } from './sessions.ts';
25
26
 
26
27
  const MAX_DOWNLOAD_SIZE = 25 * 1024 * 1024;
@@ -45,6 +46,10 @@ const mdText = (b: ConvBlock): b is { type: 'text'; text: string } => b.type ===
45
46
  // half: agent-channel summon rules + threads the agent already owns. Referenced app session state
46
47
  // (proactive origins, known threads) is why it can't live in the package.
47
48
  export function shouldRespond(msg: IngressMessage): boolean {
49
+ // Advance the per-channel last-seen cursor for EVERY message we're handed, answered or not:
50
+ // "seen" is the honest baseline for downtime backfill, and it is also how we learn which
51
+ // channels exist at all. Purely a bookkeeping write — it never affects the gate below.
52
+ noteSlackSeen(msg.channel, msg.ts);
48
53
  const isAgentChannel = msg.channel === AGENT_CHANNEL;
49
54
  const isAgentOriginatedThread = msg.isThreadReply && !!(msg.rawThreadTs && getProactiveOrigin(msg.channel, msg.rawThreadTs));
50
55
  const isKnownAgentChannelThread = isAgentChannel && msg.isThreadReply && !!(msg.rawThreadTs && hasSessionForThread(msg.channel, msg.rawThreadTs));