shraga 0.1.31 → 0.1.33
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/defaults/skills/mcp-server.md +1 -0
- package/defaults/skills/scheduler.md +105 -1
- package/defaults/skills/self-aware.md +12 -1
- package/package.json +1 -1
- package/src/server/boot.ts +51 -4
- package/src/server/downtime.ts +451 -0
- package/src/server/mcp-server.ts +19 -2
- package/src/server/scheduler/engine.ts +256 -42
- package/src/server/scheduler/runner.ts +6 -6
- package/src/server/scheduler/storage.ts +74 -0
- package/src/server/scheduler/types.ts +65 -0
- package/src/server/slack/bot.ts +5 -0
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import { loadSchedules, saveSchedules, readCompletionMarker, writeCompletionMarker, readRunningMarker, isProcessAlive, loadThrottleState, saveThrottleState } from './storage.ts';
|
|
1
|
+
import { loadSchedules, saveSchedules, readCompletionMarker, writeCompletionMarker, readRunningMarker, isProcessAlive, loadThrottleState, saveThrottleState, acquireRunLock, clearRunningMarker, markRunStarted } from './storage.ts';
|
|
2
2
|
import { computeNextRun, computePrevRun, validateTrigger } from './timing.ts';
|
|
3
3
|
import { runSchedule, type ResumeOptions, type EventContext } from './runner.ts';
|
|
4
4
|
import { backfillScope, ensureBuiltinSchedules } from './builtins.ts';
|
|
5
5
|
import { emitEvent } from '../events/bus.ts';
|
|
6
6
|
import { getSessionUrl } from '../shraga-config.ts';
|
|
7
|
-
import type { Schedule } from './types.ts';
|
|
7
|
+
import type { Schedule, MissedPolicy, RunOutcome, RunRefusal } from './types.ts';
|
|
8
8
|
|
|
9
9
|
type Broadcast = (data: object) => void;
|
|
10
10
|
|
|
@@ -26,6 +26,18 @@ interface RuntimeState {
|
|
|
26
26
|
}
|
|
27
27
|
|
|
28
28
|
const QUEUE_CAP = 5;
|
|
29
|
+
/** Catch-up fires are delayed so MCP servers finish initializing first. Read per call so tests
|
|
30
|
+
* (and an operator) can shorten it. */
|
|
31
|
+
const catchupDelayMs = () => Number(process.env.SCHEDULER_CATCHUP_DELAY_MS ?? 10_000);
|
|
32
|
+
/** Hard ceiling on how late a missed window may still be replayed, regardless of `onMissed`.
|
|
33
|
+
* A missed window is never more than one period old, so a period-relative bound alone can never
|
|
34
|
+
* suppress anything — yet replaying an 08:00 report at 22:00 is not the job the schedule
|
|
35
|
+
* describes. This absolute cap is therefore the entire rule.
|
|
36
|
+
*
|
|
37
|
+
* Deliberately NOT `min(period, cap)`: a cron missed window comes from `computePrevRun`, so its
|
|
38
|
+
* age is always strictly less than one period — a period-relative term can never suppress
|
|
39
|
+
* anything, and `min(period, cap)` is provably identical to `cap`. */
|
|
40
|
+
const maxMissedAgeMs = () => Number(process.env.SCHEDULER_MAX_MISSED_AGE_MS ?? 6 * 60 * 60 * 1000);
|
|
29
41
|
/** Max setTimeout delay (2^31-1 ms ≈ 24.8 days); longer delays overflow and fire immediately. */
|
|
30
42
|
const MAX_TIMER_MS = 2_147_483_647;
|
|
31
43
|
/** Only the designated instance fires schedules (DATA_SYNC_SCHEDULER_ACTIVE=true).
|
|
@@ -71,38 +83,55 @@ export function start(broadcast: Broadcast): void {
|
|
|
71
83
|
}
|
|
72
84
|
}
|
|
73
85
|
// Catch up missed cron fires (e.g. process was down when cron should have fired)
|
|
74
|
-
const catchUps: string[] = [];
|
|
86
|
+
const catchUps: { id: string; window: number }[] = [];
|
|
75
87
|
for (const s of state.schedules) {
|
|
76
88
|
if (!s.enabled || s.trigger.kind !== 'cron') continue;
|
|
77
89
|
const prev = computePrevRun(s.trigger);
|
|
78
90
|
if (prev === null) continue;
|
|
79
91
|
const lastAt = s.lastRun?.at;
|
|
80
92
|
if (lastAt === undefined) continue; // never ran — nothing to catch up
|
|
81
|
-
if (lastAt
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
}
|
|
92
|
-
|
|
93
|
+
if (lastAt >= prev) continue;
|
|
94
|
+
|
|
95
|
+
const marker = readCompletionMarker(s.id);
|
|
96
|
+
if (marker && marker.completedAt >= prev) {
|
|
97
|
+
console.log(`[scheduler] skipping catch-up for ${s.id} — already completed at ${new Date(marker.completedAt).toISOString()} by ${marker.triggeredBy}`);
|
|
98
|
+
continue;
|
|
99
|
+
}
|
|
100
|
+
// An ATTEMPT on this window counts too: a run that errored or was killed already had its
|
|
101
|
+
// shot. Replaying it on the next boot is exactly the duplicate-fire this guards.
|
|
102
|
+
if (marker?.attemptWindow !== undefined && marker.attemptWindow >= prev) {
|
|
103
|
+
console.log(`[scheduler] skipping catch-up for ${s.id} — window ${new Date(prev).toISOString()} already attempted (${marker.status ?? 'started'})`);
|
|
104
|
+
continue;
|
|
105
|
+
}
|
|
106
|
+
// Someone is already on it — either another instance, or (far more common) the recovery
|
|
107
|
+
// path that resumed the interrupted run in-place moments ago.
|
|
108
|
+
const running = readRunningMarker(s.id);
|
|
109
|
+
if (running && isProcessAlive(running.pid)) {
|
|
110
|
+
console.log(`[scheduler] skipping catch-up for ${s.id} — still running (pid ${running.pid}, started ${new Date(running.startedAt).toISOString()})`);
|
|
111
|
+
continue;
|
|
93
112
|
}
|
|
113
|
+
|
|
114
|
+
const verdict = judgeMissed(s, prev);
|
|
115
|
+
if (!verdict.replay) {
|
|
116
|
+
console.log(`[scheduler] not replaying ${s.id} — ${verdict.message}`);
|
|
117
|
+
noteMissed(s, prev, verdict.reason);
|
|
118
|
+
continue;
|
|
119
|
+
}
|
|
120
|
+
catchUps.push({ id: s.id, window: prev });
|
|
94
121
|
}
|
|
95
122
|
if (catchUps.length) {
|
|
96
|
-
console.log(`[scheduler] catch-up: ${catchUps.join(', ')} (delayed
|
|
123
|
+
console.log(`[scheduler] catch-up: ${catchUps.map((c) => c.id).join(', ')} (delayed ${catchupDelayMs()}ms for MCP init)`);
|
|
97
124
|
setTimeout(() => {
|
|
98
|
-
for (const id of catchUps) {
|
|
125
|
+
for (const { id, window } of catchUps) {
|
|
99
126
|
const s = getSchedule(id);
|
|
100
|
-
if (s?.enabled)
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
127
|
+
if (!s?.enabled) continue;
|
|
128
|
+
console.log(`[scheduler] catch-up: firing ${id}`);
|
|
129
|
+
// NOT runNow(): a catch-up is not a manual run. It must go through the same
|
|
130
|
+
// completion/attempt/lock guards as any timer fire, so that whatever claimed this
|
|
131
|
+
// window first (typically the in-place resume) makes this a no-op.
|
|
132
|
+
enqueueFire(s, window);
|
|
104
133
|
}
|
|
105
|
-
},
|
|
134
|
+
}, catchupDelayMs());
|
|
106
135
|
}
|
|
107
136
|
|
|
108
137
|
saveSchedules(state.schedules);
|
|
@@ -125,6 +154,9 @@ export function getSchedule(id: string): Schedule | undefined {
|
|
|
125
154
|
export function upsertSchedule(s: Schedule): { ok: true; schedule: Schedule } | { ok: false; error: string } {
|
|
126
155
|
const err = validateTrigger(s.trigger);
|
|
127
156
|
if (err) return { ok: false, error: err };
|
|
157
|
+
if (s.onMissed !== undefined && !MISSED_POLICIES.includes(s.onMissed)) {
|
|
158
|
+
return { ok: false, error: `Invalid onMissed "${s.onMissed}" (expected ${MISSED_POLICIES.join(' | ')})` };
|
|
159
|
+
}
|
|
128
160
|
|
|
129
161
|
s.updatedAt = Date.now();
|
|
130
162
|
if (s.enabled) {
|
|
@@ -180,9 +212,9 @@ export function toggleSchedule(id: string, enabled: boolean): Schedule | null {
|
|
|
180
212
|
return s;
|
|
181
213
|
}
|
|
182
214
|
|
|
183
|
-
export function runNow(id: string, override?: string):
|
|
215
|
+
export function runNow(id: string, override?: string): RunOutcome {
|
|
184
216
|
const s = getSchedule(id);
|
|
185
|
-
if (!s) return
|
|
217
|
+
if (!s) return refuse('unknown-schedule', `No schedule ${id}`);
|
|
186
218
|
return enqueueFire(s, Date.now(), override, true);
|
|
187
219
|
}
|
|
188
220
|
|
|
@@ -275,22 +307,133 @@ export function cancelRun(id: string): boolean {
|
|
|
275
307
|
* side-effects, e.g. a second Slack post). Mirrors the web/slack restart-resume path.
|
|
276
308
|
* No-op if the schedule is unknown or already running.
|
|
277
309
|
*/
|
|
278
|
-
export function resumeRun(scheduleId: string, sessionId: string, prompt: string):
|
|
310
|
+
export function resumeRun(scheduleId: string, sessionId: string, prompt: string): RunOutcome {
|
|
279
311
|
const s = getSchedule(scheduleId);
|
|
280
312
|
if (!s) {
|
|
281
313
|
console.warn(`[scheduler] resumeRun: unknown schedule ${scheduleId}`);
|
|
282
|
-
return
|
|
314
|
+
return refuse('unknown-schedule', `No schedule ${scheduleId}`);
|
|
283
315
|
}
|
|
284
316
|
if (state.running.has(scheduleId)) {
|
|
285
317
|
console.log(`[scheduler] resumeRun: ${scheduleId} already running — skipping resume`);
|
|
286
|
-
return
|
|
318
|
+
return refuse('already-running', `${s.name} is already running`);
|
|
319
|
+
}
|
|
320
|
+
const window = interruptedWindow(s);
|
|
321
|
+
// A resume continues an existing conversation rather than starting a fresh one, so it dodges
|
|
322
|
+
// the duplicate-side-effect problem — but it does NOT dodge the "is this work still the work
|
|
323
|
+
// the schedule asked for" problem. Finishing the 08:00 report at 22:00 is the incident. So the
|
|
324
|
+
// resume path answers to the same onMissed policy and the same staleness ceiling as catch-up —
|
|
325
|
+
// for EVERY trigger kind, not just cron (interval/once/event have no catch-up path at all, so
|
|
326
|
+
// resume is their only gate).
|
|
327
|
+
if (window !== null) {
|
|
328
|
+
// Someone already finished this window — typically a catch-up that won the boot race and has
|
|
329
|
+
// since completed, so the run lock it held is gone. Resuming now would redo work that is
|
|
330
|
+
// already done: the duplicate fire, one step later. (An *attempt* on this window is NOT a
|
|
331
|
+
// refusal — the interrupted run we are resuming recorded one itself.)
|
|
332
|
+
const marker = readCompletionMarker(scheduleId);
|
|
333
|
+
if (marker && marker.completedAt >= window) {
|
|
334
|
+
const msg = `window ${new Date(window).toISOString()} already completed at ${new Date(marker.completedAt).toISOString()} by ${marker.triggeredBy}`;
|
|
335
|
+
console.log(`[scheduler] not resuming ${scheduleId} — ${msg}`);
|
|
336
|
+
return refuse('already-completed', msg);
|
|
337
|
+
}
|
|
338
|
+
const verdict = judgeMissed(s, window);
|
|
339
|
+
if (!verdict.replay) {
|
|
340
|
+
console.log(`[scheduler] not resuming ${scheduleId} — ${verdict.message}`);
|
|
341
|
+
noteMissed(s, window, verdict.reason);
|
|
342
|
+
return refuse(verdict.refusal, verdict.message);
|
|
343
|
+
}
|
|
287
344
|
}
|
|
288
345
|
console.log(`[scheduler] resuming ${scheduleId} in-place on session ${sessionId.slice(0, 30)}…`);
|
|
289
|
-
return startRun(s, Date.now(), undefined, { sessionId, prompt });
|
|
346
|
+
return startRun(s, window ?? Date.now(), undefined, { sessionId, prompt });
|
|
290
347
|
}
|
|
291
348
|
|
|
292
349
|
// ── Internals ───────────────────────────────────────────────────────────────
|
|
293
350
|
|
|
351
|
+
const MISSED_POLICIES: MissedPolicy[] = ['run', 'skip', 'offer'];
|
|
352
|
+
|
|
353
|
+
function refuse(reason: RunRefusal, message: string): RunOutcome {
|
|
354
|
+
return { ok: false, reason, message };
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
/**
|
|
358
|
+
* The window the interrupted run was covering — what "how late is this?" is measured against.
|
|
359
|
+
*
|
|
360
|
+
* For `cron` the window is derivable from the expression itself (`computePrevRun`), and that is
|
|
361
|
+
* preferred: it is correct even for legacy state written before markers recorded a window.
|
|
362
|
+
*
|
|
363
|
+
* `interval`/`once`/`event` have no schedule-derivable grid — but they are not therefore timeless.
|
|
364
|
+
* An interval run interrupted 14h ago is exactly as stale as a cron one, and it has no catch-up
|
|
365
|
+
* path to be caught by. So for those the window is the one the interrupted run ITSELF claimed,
|
|
366
|
+
* read back off the markers already on disk: the run lock's `window` (stamped by `acquireRunLock`
|
|
367
|
+
* at fire time), else its `startedAt`, else the attempt ledger, else the at-start `lastRun.at`.
|
|
368
|
+
* Only a schedule with no trace of ever having started has no window — and nothing to resume.
|
|
369
|
+
*/
|
|
370
|
+
function interruptedWindow(s: Schedule): number | null {
|
|
371
|
+
const prev = computePrevRun(s.trigger);
|
|
372
|
+
if (prev !== null) return prev;
|
|
373
|
+
const running = readRunningMarker(s.id);
|
|
374
|
+
if (running?.window !== undefined) return running.window;
|
|
375
|
+
if (running?.startedAt) return running.startedAt;
|
|
376
|
+
const marker = readCompletionMarker(s.id);
|
|
377
|
+
if (marker?.attemptWindow !== undefined) return marker.attemptWindow;
|
|
378
|
+
if (marker?.lastAttemptAt) return marker.lastAttemptAt;
|
|
379
|
+
return s.lastRun?.at ?? null;
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
/** Persist the terminal outcome of an attempt without touching the last SUCCESSFUL completion.
|
|
383
|
+
* Every non-ok exit — reported failure or unexpected throw — must land here, else the marker
|
|
384
|
+
* stays `started` forever and a crash is indistinguishable from a throw. */
|
|
385
|
+
function recordAttemptOutcome(scheduleId: string, firedAt: number, at: number, status: 'error' | 'aborted' | 'started'): void {
|
|
386
|
+
const prev = readCompletionMarker(scheduleId);
|
|
387
|
+
writeCompletionMarker({
|
|
388
|
+
completedAt: prev?.completedAt ?? 0,
|
|
389
|
+
triggeredBy: 'scheduler',
|
|
390
|
+
scheduleId,
|
|
391
|
+
lastAttemptAt: prev?.lastAttemptAt ?? at,
|
|
392
|
+
attemptWindow: prev?.attemptWindow ?? firedAt,
|
|
393
|
+
status,
|
|
394
|
+
});
|
|
395
|
+
}
|
|
396
|
+
|
|
397
|
+
/** Default is `run`: it is what every existing schedule already does, and schedules.json has no
|
|
398
|
+
* `onMissed` field on any of them — defaulting to anything else would silently change the
|
|
399
|
+
* behaviour of live automations on upgrade. The unbounded-replay hazard that made `run`
|
|
400
|
+
* dangerous is fixed by the staleness ceiling below, which applies to `run` too. */
|
|
401
|
+
function missedPolicy(s: Schedule): MissedPolicy {
|
|
402
|
+
return s.onMissed ?? 'run';
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
type MissedVerdict =
|
|
406
|
+
| { replay: true }
|
|
407
|
+
| { replay: false; reason: 'skip' | 'offer' | 'stale'; refusal: 'policy-skip' | 'policy-offer' | 'stale'; message: string };
|
|
408
|
+
|
|
409
|
+
/**
|
|
410
|
+
* The one decision behind "this window already elapsed — may it still run?", shared by all three
|
|
411
|
+
* replay paths: the boot catch-up scan, the in-place resume, and the late timer fire.
|
|
412
|
+
*
|
|
413
|
+
* `onMissed` first (an explicit skip/offer is a standing instruction, not an age question), then
|
|
414
|
+
* the absolute staleness ceiling, which overrides even `run`. Callers do the recording, because
|
|
415
|
+
* only they know how to refuse (a return value, or `continue`).
|
|
416
|
+
*/
|
|
417
|
+
function judgeMissed(s: Schedule, window: number, now: number = Date.now()): MissedVerdict {
|
|
418
|
+
const when = new Date(window).toISOString();
|
|
419
|
+
const policy = missedPolicy(s);
|
|
420
|
+
if (policy !== 'run') {
|
|
421
|
+
return { replay: false, reason: policy, refusal: policy === 'skip' ? 'policy-skip' : 'policy-offer', message: `window ${when} was missed and onMissed=${policy}` };
|
|
422
|
+
}
|
|
423
|
+
const age = now - window;
|
|
424
|
+
const grace = maxMissedAgeMs();
|
|
425
|
+
if (age > grace) {
|
|
426
|
+
return { replay: false, reason: 'stale', refusal: 'stale', message: `window ${when} is ${Math.round(age / 60_000)}m stale (ceiling ${Math.round(grace / 60_000)}m)` };
|
|
427
|
+
}
|
|
428
|
+
return { replay: true };
|
|
429
|
+
}
|
|
430
|
+
|
|
431
|
+
function noteMissed(s: Schedule, at: number, reason: 'skip' | 'offer' | 'stale'): void {
|
|
432
|
+
s.missedRun = { at, reason, noticedAt: Date.now() };
|
|
433
|
+
saveSchedules(state.schedules);
|
|
434
|
+
state.broadcast({ type: 'schedule:updated', schedule: s });
|
|
435
|
+
}
|
|
436
|
+
|
|
294
437
|
function replan(): void {
|
|
295
438
|
if (!schedulerActive) return;
|
|
296
439
|
if (state.timer) { clearTimeout(state.timer); state.timer = null; }
|
|
@@ -317,7 +460,24 @@ function fireDue(): void {
|
|
|
317
460
|
const now = Date.now();
|
|
318
461
|
for (const s of state.schedules) {
|
|
319
462
|
if (!s.enabled || s.nextRun === undefined) continue;
|
|
320
|
-
if (s.nextRun
|
|
463
|
+
if (s.nextRun > now) continue;
|
|
464
|
+
// A timer fire is normally punctual, so it is NOT treated as a missed window — `onMissed`
|
|
465
|
+
// must never gate healthy operation. But a setTimeout does not survive a wall-clock jump:
|
|
466
|
+
// when the host sleeps (lid closed) or NTP steps the clock, the process is FROZEN, not
|
|
467
|
+
// killed — no restart, so the boot catch-up scan never runs — and the overdue timer fires
|
|
468
|
+
// the elapsed window the moment the machine wakes. Firing an 08:00 job at 21:36 is exactly
|
|
469
|
+
// what the catch-up ceiling exists to prevent, reached by the one path it didn't cover.
|
|
470
|
+
// Past the ceiling we hand the window to the same judgement the other two paths use.
|
|
471
|
+
// Past the ceiling `judgeMissed` can only refuse (it replays a `run` window only while it is
|
|
472
|
+
// within the ceiling), so there is no replay branch here — the shared decision is used for WHAT
|
|
473
|
+
// to record, not whether to fire.
|
|
474
|
+
if (now - s.nextRun > maxMissedAgeMs()) {
|
|
475
|
+
const verdict = judgeMissed(s, s.nextRun, now) as Extract<MissedVerdict, { replay: false }>;
|
|
476
|
+
console.log(`[scheduler] not firing ${s.id} — ${verdict.message} (timer fired late; host suspended or clock jumped)`);
|
|
477
|
+
noteMissed(s, s.nextRun, verdict.reason);
|
|
478
|
+
continue; // the advance loop below still re-arms / retires this schedule
|
|
479
|
+
}
|
|
480
|
+
enqueueFire(s, s.nextRun);
|
|
321
481
|
}
|
|
322
482
|
// Advance nextRun for recurring triggers; disable fired `once` triggers
|
|
323
483
|
for (const s of state.schedules) {
|
|
@@ -336,21 +496,22 @@ function fireDue(): void {
|
|
|
336
496
|
replan();
|
|
337
497
|
}
|
|
338
498
|
|
|
339
|
-
function enqueueFire(s: Schedule, firedAt: number, override?: string, manual = false, eventCtx?: EventContext):
|
|
340
|
-
// Skip if this cron period was already completed
|
|
341
|
-
//
|
|
499
|
+
function enqueueFire(s: Schedule, firedAt: number, override?: string, manual = false, eventCtx?: EventContext): RunOutcome {
|
|
500
|
+
// Skip if this cron period was already completed/attempted. Manual runs (runNow from UI/API)
|
|
501
|
+
// always proceed past the period guard — but never past the run lock in startRun().
|
|
342
502
|
if (!manual && s.trigger.kind === 'cron') {
|
|
343
503
|
const prev = computePrevRun(s.trigger, firedAt + 1);
|
|
344
504
|
if (prev !== null) {
|
|
345
505
|
const marker = readCompletionMarker(s.id);
|
|
346
506
|
if (marker && marker.completedAt >= prev) {
|
|
347
|
-
|
|
348
|
-
|
|
507
|
+
const msg = `already completed this period (at ${new Date(marker.completedAt).toISOString()})`;
|
|
508
|
+
console.log(`[scheduler] skipping fire for ${s.id} — ${msg}`);
|
|
509
|
+
return refuse('already-completed', msg);
|
|
349
510
|
}
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
console.log(`[scheduler] skipping fire for ${s.id} —
|
|
353
|
-
return
|
|
511
|
+
if (marker?.attemptWindow !== undefined && marker.attemptWindow >= prev) {
|
|
512
|
+
const msg = `this period was already attempted (${marker.status ?? 'started'})`;
|
|
513
|
+
console.log(`[scheduler] skipping fire for ${s.id} — ${msg}`);
|
|
514
|
+
return refuse('already-attempted', msg);
|
|
354
515
|
}
|
|
355
516
|
}
|
|
356
517
|
}
|
|
@@ -369,10 +530,36 @@ function enqueueFire(s: Schedule, firedAt: number, override?: string, manual = f
|
|
|
369
530
|
console.warn(`[scheduler] queue overflow for ${s.id} (cap=${QUEUE_CAP}), dropped fire @ ${dropped?.firedAt}`);
|
|
370
531
|
}
|
|
371
532
|
state.queues.set(s.id, q);
|
|
372
|
-
return null;
|
|
533
|
+
return { ok: true, sessionId: null, queued: true };
|
|
373
534
|
}
|
|
374
535
|
|
|
375
|
-
|
|
536
|
+
/**
|
|
537
|
+
* Start one run, holding the cross-restart run lock for its whole life.
|
|
538
|
+
*
|
|
539
|
+
* Every path that can start a run funnels through here — timer fire, catch-up, event, manual
|
|
540
|
+
* runNow, and the in-place resume from crash recovery — so the lock is the single place that
|
|
541
|
+
* enforces "one live run per schedule". It is also what makes catch-up and recovery mutually
|
|
542
|
+
* exclusive: whichever reaches this first holds a live pid, and the other's acquire fails.
|
|
543
|
+
*/
|
|
544
|
+
function startRun(s: Schedule, firedAt: number, override?: string, resume?: ResumeOptions, eventCtx?: EventContext): RunOutcome {
|
|
545
|
+
const lock = acquireRunLock(s.id, firedAt);
|
|
546
|
+
if (!lock) {
|
|
547
|
+
const held = readRunningMarker(s.id);
|
|
548
|
+
const msg = `run lock held by pid ${held?.pid} since ${new Date(held?.startedAt ?? 0).toISOString()}`;
|
|
549
|
+
console.log(`[scheduler] not starting ${s.id} — ${msg}`);
|
|
550
|
+
return refuse('locked', msg);
|
|
551
|
+
}
|
|
552
|
+
// Record the attempt BEFORE running: a crash from here on must not look like "never tried".
|
|
553
|
+
markRunStarted(s.id, firedAt);
|
|
554
|
+
const pre = getSchedule(s.id);
|
|
555
|
+
if (pre) {
|
|
556
|
+
// Advance lastRun at start, not only on success. start() turns a leftover 'running' into
|
|
557
|
+
// 'error' on the next boot, so the distinction survives while the timestamp still blocks a
|
|
558
|
+
// replay of this window.
|
|
559
|
+
pre.lastRun = { at: Date.now(), sessionId: resume?.sessionId ?? '', status: 'running' };
|
|
560
|
+
pre.missedRun = undefined;
|
|
561
|
+
saveSchedules(state.schedules);
|
|
562
|
+
}
|
|
376
563
|
// Deep-copy task so edits mid-run don't affect the in-flight execution
|
|
377
564
|
const snapshot: Schedule = JSON.parse(JSON.stringify(s));
|
|
378
565
|
let sessionId: string | null = null;
|
|
@@ -380,6 +567,14 @@ function startRun(s: Schedule, _firedAt: number, override?: string, resume?: Res
|
|
|
380
567
|
const register = (sid: string, ac: AbortController) => {
|
|
381
568
|
sessionId = sid;
|
|
382
569
|
state.running.set(s.id, ac);
|
|
570
|
+
// Backfill the session link onto the at-start lastRun. Without this a crash mid-run persists
|
|
571
|
+
// an errored run with sessionId '' — no way back to the conversation that was interrupted,
|
|
572
|
+
// which is exactly what the recovery path needs.
|
|
573
|
+
const live = getSchedule(s.id);
|
|
574
|
+
if (live?.lastRun && live.lastRun.status === 'running' && !live.lastRun.sessionId) {
|
|
575
|
+
live.lastRun.sessionId = sid;
|
|
576
|
+
saveSchedules(state.schedules);
|
|
577
|
+
}
|
|
383
578
|
};
|
|
384
579
|
|
|
385
580
|
state.broadcast({ type: 'schedule:fired', scheduleId: s.id });
|
|
@@ -393,8 +588,13 @@ function startRun(s: Schedule, _firedAt: number, override?: string, resume?: Res
|
|
|
393
588
|
saveSchedules(state.schedules);
|
|
394
589
|
state.broadcast({ type: 'schedule:updated', schedule: live });
|
|
395
590
|
}
|
|
591
|
+
if (summary.status !== 'ok') {
|
|
592
|
+
// Keep the attempt on record with its real outcome — the next boot must see that this
|
|
593
|
+
// window was tried and failed, not that it never ran.
|
|
594
|
+
recordAttemptOutcome(s.id, firedAt, summary.at, summary.status === 'running' ? 'started' : summary.status);
|
|
595
|
+
}
|
|
396
596
|
if (summary.status === 'ok') {
|
|
397
|
-
writeCompletionMarker({ completedAt: summary.at, triggeredBy: 'scheduler', scheduleId: s.id });
|
|
597
|
+
writeCompletionMarker({ completedAt: summary.at, triggeredBy: 'scheduler', scheduleId: s.id, lastAttemptAt: summary.at, attemptWindow: firedAt, status: 'ok' });
|
|
398
598
|
if (live && live.trigger.kind === 'once') {
|
|
399
599
|
console.log(`[scheduler] auto-deleting completed once-schedule ${s.id}`);
|
|
400
600
|
deleteSchedule(s.id);
|
|
@@ -422,9 +622,23 @@ function startRun(s: Schedule, _firedAt: number, override?: string, resume?: Res
|
|
|
422
622
|
})
|
|
423
623
|
.catch((err) => {
|
|
424
624
|
console.error(`[scheduler] unexpected run failure for ${s.id}:`, err);
|
|
625
|
+
// A rejection never reaches the .then above, so without this the attempt marker stays
|
|
626
|
+
// 'started' forever and an unexpected throw is indistinguishable from a power cut.
|
|
627
|
+
const at = Date.now();
|
|
628
|
+
recordAttemptOutcome(s.id, firedAt, at, 'error');
|
|
629
|
+
const live = getSchedule(s.id);
|
|
630
|
+
if (live) {
|
|
631
|
+
live.lastRun = { at, sessionId: live.lastRun?.sessionId ?? '', status: 'error', error: `unexpected run failure: ${err?.message ?? String(err)}` };
|
|
632
|
+
saveSchedules(state.schedules);
|
|
633
|
+
state.broadcast({ type: 'schedule:updated', schedule: live });
|
|
634
|
+
}
|
|
425
635
|
})
|
|
426
636
|
.finally(() => {
|
|
427
637
|
state.running.delete(s.id);
|
|
638
|
+
// Release the lock on EVERY exit path — ok, error, abort, or an unexpected throw. The
|
|
639
|
+
// runner already clears it on its own terminal states; this is idempotent and covers the
|
|
640
|
+
// paths that never reach the runner's finally.
|
|
641
|
+
clearRunningMarker(s.id);
|
|
428
642
|
const q = state.queues.get(s.id);
|
|
429
643
|
if (q && q.length > 0) {
|
|
430
644
|
const next = q.shift()!;
|
|
@@ -434,5 +648,5 @@ function startRun(s: Schedule, _firedAt: number, override?: string, resume?: Res
|
|
|
434
648
|
}
|
|
435
649
|
});
|
|
436
650
|
|
|
437
|
-
return sessionId;
|
|
651
|
+
return { ok: true, sessionId };
|
|
438
652
|
}
|
|
@@ -6,7 +6,7 @@ import { streamChat, type PermissionHandler } from '../claude.ts';
|
|
|
6
6
|
import { getMcpConfig } from '../mcp.ts';
|
|
7
7
|
import { appendMessage, createScheduledSession, updateScheduledSessionStatus, setRunStatus, registerLivePartial, unregisterLivePartial, writePartial, clearPartial, acquireSessionLock, releaseSessionLock, type ConvBlock } from '../sessions.ts';
|
|
8
8
|
import type { Schedule, ScheduleRunSummary } from './types.ts';
|
|
9
|
-
import {
|
|
9
|
+
import { updateRunLockPid, clearRunningMarker } from './storage.ts';
|
|
10
10
|
import { addUnread } from '../unread.ts';
|
|
11
11
|
|
|
12
12
|
export interface RunContext {
|
|
@@ -114,8 +114,8 @@ export async function runSchedule(
|
|
|
114
114
|
acquireSessionLock(sessionId, 'scheduler', abortController);
|
|
115
115
|
setRunStatus(sessionId, 'running', 'scheduler');
|
|
116
116
|
onEvent({ type: 'session_busy', sessionId, busy: true });
|
|
117
|
-
//
|
|
118
|
-
|
|
117
|
+
// The run lock (running marker) is acquired by the engine BEFORE this point — see
|
|
118
|
+
// engine.startRun. Writing it here too would let a direct runSchedule() call bypass the lock.
|
|
119
119
|
|
|
120
120
|
const task = schedule.task;
|
|
121
121
|
if (task.kind === 'job') {
|
|
@@ -414,9 +414,9 @@ function runCommandWithMarker(command: string, abortController: AbortController,
|
|
|
414
414
|
stdio: ['ignore', 'pipe', 'pipe'],
|
|
415
415
|
});
|
|
416
416
|
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
417
|
+
// Re-point the held lock at the child: the job outlives nothing here, but if the server dies
|
|
418
|
+
// the child may still be alive, and a live pid must keep the schedule locked.
|
|
419
|
+
if (child.pid) updateRunLockPid(scheduleId, child.pid);
|
|
420
420
|
|
|
421
421
|
let output = '';
|
|
422
422
|
const handleData = (chunk: Buffer) => {
|
|
@@ -96,3 +96,77 @@ export function clearRunningMarker(scheduleId: string): void {
|
|
|
96
96
|
export function isProcessAlive(pid: number): boolean {
|
|
97
97
|
try { process.kill(pid, 0); return true; } catch { return false; }
|
|
98
98
|
}
|
|
99
|
+
|
|
100
|
+
/** How long a held run lock is believed, before it is treated as abandoned regardless of whether
|
|
101
|
+
* its pid answers. Bounds the blast radius of pid reuse: after a power cut the OS restarts pid
|
|
102
|
+
* allocation low, so a persisted pid can plausibly be live again under the same uid — and a
|
|
103
|
+
* liveness check alone would then wedge the schedule forever. Generous vs any real run (agent
|
|
104
|
+
* runs are minutes, not hours) while capping the wedge at one window's worth of a daily job. */
|
|
105
|
+
const runLockMaxAgeMs = () => Number(process.env.SCHEDULER_RUN_LOCK_MAX_AGE_MS ?? 6 * 60 * 60 * 1000);
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* Claim the single live-run slot for a schedule.
|
|
109
|
+
*
|
|
110
|
+
* The lock IS the running marker — same file, same conventions — so it survives a process
|
|
111
|
+
* restart: after a crash the marker is still on disk but its pid is dead, and a dead pid is
|
|
112
|
+
* reclaimable (otherwise a power cut would wedge the schedule forever). A LIVE pid, ours or
|
|
113
|
+
* another instance's, means someone is already running this schedule: the caller must back off —
|
|
114
|
+
* UNLESS the claim is older than `runLockMaxAgeMs`, which is the escape hatch for a lock wedged
|
|
115
|
+
* by pid reuse. A live, in-ceiling claim is never stolen, not even by a manual run: on this
|
|
116
|
+
* single-active-instance design that pid is a real run (an agent session, or a spawned job the
|
|
117
|
+
* lock was re-pointed at), and starting a second one is the duplicate-fire we are preventing.
|
|
118
|
+
*
|
|
119
|
+
* Single-writer by design (only the DATA_SYNC_SCHEDULER_ACTIVE instance fires), so this is a
|
|
120
|
+
* read-then-write, not an atomic CAS.
|
|
121
|
+
*/
|
|
122
|
+
export function acquireRunLock(scheduleId: string, window: number): RunningMarker | null {
|
|
123
|
+
const existing = readRunningMarker(scheduleId);
|
|
124
|
+
if (existing) {
|
|
125
|
+
const alive = isProcessAlive(existing.pid);
|
|
126
|
+
const age = Date.now() - (Number.isFinite(existing.startedAt) ? existing.startedAt : 0);
|
|
127
|
+
const maxAge = runLockMaxAgeMs();
|
|
128
|
+
if (alive && age <= maxAge) return null;
|
|
129
|
+
const why = !alive
|
|
130
|
+
? `dead pid ${existing.pid}`
|
|
131
|
+
: `held ${Math.round(age / 60_000)}m by live pid ${existing.pid}, past the ${Math.round(maxAge / 60_000)}m lock ceiling — assuming pid reuse`;
|
|
132
|
+
console.log(`[scheduler] reclaiming run lock for ${scheduleId} — ${why} (window ${new Date(existing.window ?? existing.startedAt).toISOString()})`);
|
|
133
|
+
}
|
|
134
|
+
const marker: RunningMarker = { pid: process.pid, startedAt: Date.now(), scheduleId, window };
|
|
135
|
+
writeRunningMarker(marker);
|
|
136
|
+
return marker;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/** Re-point a held lock at a spawned child process. Only ever touches a lock THIS process holds,
|
|
140
|
+
* and preserves `startedAt` so re-pointing can't refresh the age ceiling above. */
|
|
141
|
+
export function updateRunLockPid(scheduleId: string, pid: number): void {
|
|
142
|
+
const existing = readRunningMarker(scheduleId);
|
|
143
|
+
if (!existing) {
|
|
144
|
+
console.warn(`[scheduler] updateRunLockPid(${scheduleId}): no lock held — not creating one`);
|
|
145
|
+
return;
|
|
146
|
+
}
|
|
147
|
+
if (existing.pid !== process.pid) {
|
|
148
|
+
console.warn(`[scheduler] updateRunLockPid(${scheduleId}): lock is held by pid ${existing.pid}, not us — leaving it alone`);
|
|
149
|
+
return;
|
|
150
|
+
}
|
|
151
|
+
writeRunningMarker({ ...existing, scheduleId, pid });
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* Record that a run STARTED for `window`, before it can succeed or fail.
|
|
156
|
+
*
|
|
157
|
+
* Without this a run that errors or is killed leaves no trace on disk, so the next boot sees an
|
|
158
|
+
* un-completed window and replays it — the double-fire this ledger exists to prevent. The
|
|
159
|
+
* successful-completion timestamp is preserved untouched so "started" stays distinguishable
|
|
160
|
+
* from "completed ok".
|
|
161
|
+
*/
|
|
162
|
+
export function markRunStarted(scheduleId: string, window: number, triggeredBy: CompletionMarker['triggeredBy'] = 'scheduler'): void {
|
|
163
|
+
const prev = readCompletionMarker(scheduleId);
|
|
164
|
+
writeCompletionMarker({
|
|
165
|
+
completedAt: prev?.completedAt ?? 0,
|
|
166
|
+
triggeredBy,
|
|
167
|
+
scheduleId,
|
|
168
|
+
lastAttemptAt: Date.now(),
|
|
169
|
+
attemptWindow: window,
|
|
170
|
+
status: 'started',
|
|
171
|
+
});
|
|
172
|
+
}
|
|
@@ -34,16 +34,77 @@ export interface ScheduleRunSummary {
|
|
|
34
34
|
error?: string;
|
|
35
35
|
}
|
|
36
36
|
|
|
37
|
+
/** What the scheduler did about a window it found already elapsed at boot.
|
|
38
|
+
* - `run` replay it (today's behaviour, and the default — see engine.ts).
|
|
39
|
+
* - `skip` never replay; the window is simply lost.
|
|
40
|
+
* - `offer` don't auto-fire, but record it on the schedule (`missedRun`) so a human can
|
|
41
|
+
* see it was missed and run it on demand. */
|
|
42
|
+
export type MissedPolicy = 'run' | 'skip' | 'offer';
|
|
43
|
+
|
|
44
|
+
/** A window that elapsed while nothing was running and was NOT replayed.
|
|
45
|
+
* Persisted on the schedule and returned verbatim by `GET /api/schedules[/:id]` — that read
|
|
46
|
+
* path IS the affordance `offer` promises: a human (or the agent, via the scheduler skill) sees
|
|
47
|
+
* the missed window and can `POST /api/schedules/:id/run` it on demand. */
|
|
48
|
+
export interface MissedRun {
|
|
49
|
+
/** The window (fire time) that was missed. */
|
|
50
|
+
at: number;
|
|
51
|
+
reason: 'skip' | 'offer' | 'stale';
|
|
52
|
+
/** When the scheduler noticed — i.e. boot time. Read by whoever acts on the miss: `at` alone
|
|
53
|
+
* can't tell "missed 20m ago, still worth running" from "found on a boot two days later". */
|
|
54
|
+
noticedAt: number;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/** Why a start request did not start a run. */
|
|
58
|
+
export type RunRefusal =
|
|
59
|
+
| 'unknown-schedule'
|
|
60
|
+
| 'already-running'
|
|
61
|
+
| 'already-completed'
|
|
62
|
+
| 'already-attempted'
|
|
63
|
+
| 'policy-skip'
|
|
64
|
+
| 'policy-offer'
|
|
65
|
+
| 'stale'
|
|
66
|
+
| 'locked';
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Result of asking the engine to start a run (timer fire, catch-up, manual, event, resume).
|
|
70
|
+
* A refusal carries WHY, so the caller can record a truthful error and map a real HTTP status
|
|
71
|
+
* instead of collapsing every outcome into "null".
|
|
72
|
+
*/
|
|
73
|
+
export type RunOutcome =
|
|
74
|
+
| { ok: true; sessionId: string | null; queued?: boolean }
|
|
75
|
+
| { ok: false; reason: RunRefusal; message: string };
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Per-schedule attempt/completion ledger. `completedAt` records the last SUCCESSFUL run;
|
|
79
|
+
* `lastAttemptAt`/`attemptWindow` record the last run that STARTED, whether or not it finished.
|
|
80
|
+
* Both are needed: completion drives "this period is done", attempt drives "this period was
|
|
81
|
+
* already tried, don't replay it after a crash".
|
|
82
|
+
*/
|
|
37
83
|
export interface CompletionMarker {
|
|
84
|
+
/** Last successful completion; 0 when the schedule has never completed. */
|
|
38
85
|
completedAt: number;
|
|
39
86
|
triggeredBy: 'scheduler' | 'ssh' | 'api' | 'manual';
|
|
40
87
|
scheduleId: string;
|
|
88
|
+
/** Wall-clock time the last attempt began. */
|
|
89
|
+
lastAttemptAt?: number;
|
|
90
|
+
/** The window (fire time) that attempt was covering. */
|
|
91
|
+
attemptWindow?: number;
|
|
92
|
+
/** Terminal state of the last attempt; `started` means it never reported back (crash). */
|
|
93
|
+
status?: 'started' | 'ok' | 'error' | 'aborted';
|
|
41
94
|
}
|
|
42
95
|
|
|
96
|
+
/**
|
|
97
|
+
* The run lock: at most one live run per scheduleId, across process restarts.
|
|
98
|
+
* Written by `acquireRunLock` before a run starts, removed when it reaches a terminal state.
|
|
99
|
+
* A lock whose `pid` is dead (crash / power loss) is reclaimable — see `acquireRunLock`.
|
|
100
|
+
*/
|
|
43
101
|
export interface RunningMarker {
|
|
44
102
|
pid: number;
|
|
45
103
|
startedAt: number;
|
|
46
104
|
scheduleId: string;
|
|
105
|
+
/** The window (fire time) this run covers — lets a claimer see WHAT is held, not just that
|
|
106
|
+
* something is. */
|
|
107
|
+
window?: number;
|
|
47
108
|
}
|
|
48
109
|
|
|
49
110
|
export interface Schedule {
|
|
@@ -59,6 +120,10 @@ export interface Schedule {
|
|
|
59
120
|
nextRun?: number;
|
|
60
121
|
lastRun?: ScheduleRunSummary;
|
|
61
122
|
runCount: number;
|
|
123
|
+
/** What to do with a window that elapsed while the process was down. Absent ⇒ 'run'. */
|
|
124
|
+
onMissed?: MissedPolicy;
|
|
125
|
+
/** Last window that elapsed and was deliberately not replayed (policy or staleness). */
|
|
126
|
+
missedRun?: MissedRun;
|
|
62
127
|
/** Set when a data-plane module owns this schedule (module name). Module reconcile
|
|
63
128
|
* updates trigger/task; enable/disable snapshots live in the module's state entry. */
|
|
64
129
|
managedBy?: string;
|
package/src/server/slack/bot.ts
CHANGED
|
@@ -21,6 +21,7 @@ import { pipeAgentReply, type AgentEvent, type IngressMessage } from 'mcp-slack-
|
|
|
21
21
|
import { makeSlackQuestionHandler } from './questions.ts';
|
|
22
22
|
import * as contacts from '../contacts.ts';
|
|
23
23
|
import { getChannelContext, invalidateChannelContext } from './context-cache.ts';
|
|
24
|
+
import { noteSlackSeen } from '../downtime.ts';
|
|
24
25
|
import { getOrCreateSession, registerThreadAlias, setLastMessageTs, setUseUserToken, findSlackSessionBySessionId, getProactiveOrigin, hasSessionForThread, isSlackBotPlaceholderEmail } from './sessions.ts';
|
|
25
26
|
|
|
26
27
|
const MAX_DOWNLOAD_SIZE = 25 * 1024 * 1024;
|
|
@@ -45,6 +46,10 @@ const mdText = (b: ConvBlock): b is { type: 'text'; text: string } => b.type ===
|
|
|
45
46
|
// half: agent-channel summon rules + threads the agent already owns. Referenced app session state
|
|
46
47
|
// (proactive origins, known threads) is why it can't live in the package.
|
|
47
48
|
export function shouldRespond(msg: IngressMessage): boolean {
|
|
49
|
+
// Advance the per-channel last-seen cursor for EVERY message we're handed, answered or not:
|
|
50
|
+
// "seen" is the honest baseline for downtime backfill, and it is also how we learn which
|
|
51
|
+
// channels exist at all. Purely a bookkeeping write — it never affects the gate below.
|
|
52
|
+
noteSlackSeen(msg.channel, msg.ts);
|
|
48
53
|
const isAgentChannel = msg.channel === AGENT_CHANNEL;
|
|
49
54
|
const isAgentOriginatedThread = msg.isThreadReply && !!(msg.rawThreadTs && getProactiveOrigin(msg.channel, msg.rawThreadTs));
|
|
50
55
|
const isKnownAgentChannelThread = isAgentChannel && msg.isThreadReply && !!(msg.rawThreadTs && hasSessionForThread(msg.channel, msg.rawThreadTs));
|