@substrat-run/kernel 0.114.0 → 0.116.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/denial-query.d.ts +3 -1
- package/dist/denial-query.d.ts.map +1 -1
- package/dist/denial-query.js +5 -1
- package/dist/denial-query.js.map +1 -1
- package/dist/index.d.ts +5 -3
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -1
- package/dist/index.js.map +1 -1
- package/dist/job-run.d.ts +493 -0
- package/dist/job-run.d.ts.map +1 -0
- package/dist/job-run.js +655 -0
- package/dist/job-run.js.map +1 -0
- package/dist/platform-sweep.d.ts +57 -0
- package/dist/platform-sweep.d.ts.map +1 -1
- package/dist/platform-sweep.js +85 -0
- package/dist/platform-sweep.js.map +1 -1
- package/dist/scope-host.d.ts +234 -1
- package/dist/scope-host.d.ts.map +1 -1
- package/dist/scope-host.js.map +1 -1
- package/dist/timeline.d.ts.map +1 -1
- package/dist/timeline.js +12 -1
- package/dist/timeline.js.map +1 -1
- package/package.json +2 -2
package/dist/job-run.js
ADDED
|
@@ -0,0 +1,655 @@
|
|
|
1
|
+
import { substratError } from '@substrat-run/contracts';
|
|
2
|
+
import { backoffAt, resolveRetryPolicy } from './scope-host.js';
|
|
3
|
+
/**
|
|
4
|
+
* The fourth driver (#1577): long, resumable, coalesced work.
|
|
5
|
+
*
|
|
6
|
+
* Three drivers already move work off a request, and each is correct for what it
|
|
7
|
+
* is. `registerExecutor` + `drainDue` retries ONE delivery, whole. A declared
|
|
8
|
+
* schedule + `runDueSchedules` fires ONE operation inside a cadence window, which
|
|
9
|
+
* must then finish. `runPlatformSweep` does a pass of per-unit maintenance and
|
|
10
|
+
* reports it. None of them covers the shape this file names:
|
|
11
|
+
*
|
|
12
|
+
* Walk 100 000 objects in an external system. It takes an hour. It must survive a
|
|
13
|
+
* worker eviction, a deploy and a transient upstream failure by CONTINUING WHERE IT
|
|
14
|
+
* STOPPED. Only one walk per source may run at a time, and asking for a second while
|
|
15
|
+
* one is in flight must join the first. When it ends the outcome has to be legible.
|
|
16
|
+
*
|
|
17
|
+
* ## The model, in four words: run, pass, step, cursor
|
|
18
|
+
*
|
|
19
|
+
* A **run** is the durable record — one `_substrat_job_runs` row, in the scope, with
|
|
20
|
+
* its status, its resume cursor, a counter bag, its start/end and its last error. It
|
|
21
|
+
* is what an operator reads afterwards, and it is the thing coalescing joins.
|
|
22
|
+
*
|
|
23
|
+
* A **pass** is one invocation of the job's handler. A pass does a BOUNDED chunk of
|
|
24
|
+
* the walk and hands a cursor forward; the next pass resumes from it. That is the
|
|
25
|
+
* whole of resume at the outer level, and it is why a 100 000-object walk is not a
|
|
26
|
+
* single hour-long call anybody has to keep alive.
|
|
27
|
+
*
|
|
28
|
+
* A **step** is a named unit inside a pass. Its result is committed as it completes,
|
|
29
|
+
* so a step that already succeeded is NOT re-run — not on a retry of the step after
|
|
30
|
+
* it, and not after the process was killed mid-pass. That is resume at the inner
|
|
31
|
+
* level, and it is what "no step before it repeated" means.
|
|
32
|
+
*
|
|
33
|
+
* A **cursor** is whatever the handler says the next pass should resume from — an id,
|
|
34
|
+
* a page token, an offset. Opaque to the driver, stored as JSON, held to the payload
|
|
35
|
+
* rule below because it has to survive the same trip.
|
|
36
|
+
*
|
|
37
|
+
* **The step ledger is per PASS, and that is deliberate.** When a pass commits, its
|
|
38
|
+
* step rows are dropped: the next pass is new work, named by the new cursor, and a
|
|
39
|
+
* ledger that accumulated across an hour-long walk would be the unbounded table this
|
|
40
|
+
* design exists to avoid. The cursor carries progress BETWEEN passes; the ledger
|
|
41
|
+
* carries it WITHIN one.
|
|
42
|
+
*
|
|
43
|
+
* ## The determinism rule, and the half of it that is mechanical
|
|
44
|
+
*
|
|
45
|
+
* Step names must be a pure function of the payload and prior results. If a name
|
|
46
|
+
* varies between passes — a timestamp in it, a random suffix — the memo never hits,
|
|
47
|
+
* every resume replays work that already happened, and resume is a lie told in a
|
|
48
|
+
* green test. That rule cannot be fully checked from here; what CAN be checked is
|
|
49
|
+
* the sharpest way to break it, and is: two `step()` calls under ONE name in ONE
|
|
50
|
+
* pass are refused (`JOB_STEP_REUSED`), because the second would read the first's
|
|
51
|
+
* memo and silently skip its own work.
|
|
52
|
+
*
|
|
53
|
+
* ## Coalescing is the DRIVER's, never a unique index
|
|
54
|
+
*
|
|
55
|
+
* One run in flight per `(module, job, instance)`. `startJobRun` looks for a live row
|
|
56
|
+
* and RETURNS it rather than inserting a second. It is not a `UNIQUE` constraint, and
|
|
57
|
+
* that is the point: a run whose process was killed is still `running`, and it must be
|
|
58
|
+
* restartable — a uniqueness constraint would be "one row ever", which would refuse
|
|
59
|
+
* the re-import that has to happen next year as loudly as it refuses the duplicate.
|
|
60
|
+
* `_substrat_sweep_runs` is a RECEIPT and carries such a constraint; a cursor is not a
|
|
61
|
+
* receipt, and the two must not be re-merged (#1571, #1572).
|
|
62
|
+
*
|
|
63
|
+
* ## One driver per scope at a time — a stated bound, not a mechanism
|
|
64
|
+
*
|
|
65
|
+
* Coalescing stops duplicate RUNS. It does not stop two concurrent callers of
|
|
66
|
+
* `runDueJobs` from picking the same run out of the due read and advancing it at the
|
|
67
|
+
* same time: there is no lease, and a lease is not smuggled in here. The topology the
|
|
68
|
+
* driver is built for has one tick per scope — `runPlatformSweep` enumerates scopes
|
|
69
|
+
* and does one call each, a scope DO's alarm fires for its own scope — so the bound
|
|
70
|
+
* is satisfied by construction rather than defended against.
|
|
71
|
+
*
|
|
72
|
+
* What the overlap would cost, if a deployment did drive one scope twice at once: the
|
|
73
|
+
* step ledger absorbs most of it (a step already committed returns its memo to both),
|
|
74
|
+
* so the exposure is a step neither pass has finished yet, which both would run. That
|
|
75
|
+
* is the same at-least-once residue an executor already has and which handlers already
|
|
76
|
+
* have to absorb — but it is NOT what "only one walk per source at a time" promises,
|
|
77
|
+
* so it is written down rather than implied. Adding a lease is a real design with a
|
|
78
|
+
* real expiry question behind it (a leaked lease is a run nothing will ever touch
|
|
79
|
+
* again), and it wants a consumer's numbers before it gets one.
|
|
80
|
+
*
|
|
81
|
+
* ## What this is NOT
|
|
82
|
+
*
|
|
83
|
+
* Not a workflow engine: no branching, no fan-out, no BPMN, no timers between steps.
|
|
84
|
+
* A linear, resumable, coalesced sequence of steps with a cursor — the smallest thing
|
|
85
|
+
* that makes an hour-long import survivable. A use case that needs branching is a
|
|
86
|
+
* different issue and probably a different answer.
|
|
87
|
+
*/
|
|
88
|
+
/**
|
|
89
|
+
* The run table and its step ledger, as both adapters build them.
|
|
90
|
+
*
|
|
91
|
+
* Shared rather than spelled twice for the reason `IDEMPOTENCY_DDL` and
|
|
92
|
+
* `SCHEDULE_STATE_DDL` are: `lint:spine-ddl` compares what each adapter's
|
|
93
|
+
* `KERNEL_DDL` executes, and one definition is what keeps the self-hosted store
|
|
94
|
+
* and the hosted store the same shape rather than merely the same intention.
|
|
95
|
+
* Spine — kernel-written, `_substrat_*`, never a module migration.
|
|
96
|
+
*/
|
|
97
|
+
export const JOB_RUN_DDL = `
|
|
98
|
+
CREATE TABLE IF NOT EXISTS _substrat_job_runs (
|
|
99
|
+
id TEXT PRIMARY KEY,
|
|
100
|
+
-- The coalescing key: one LIVE run per (module_id, job, instance). Not a UNIQUE
|
|
101
|
+
-- index, deliberately -- see this file's header. instance names WHAT is being
|
|
102
|
+
-- walked (a source id, a mapping version); a job with one walk per scope uses
|
|
103
|
+
-- the 'default' the input schema fills in.
|
|
104
|
+
module_id TEXT NOT NULL,
|
|
105
|
+
job TEXT NOT NULL,
|
|
106
|
+
instance TEXT NOT NULL,
|
|
107
|
+
-- What start() was handed, as JSON. Held to the queue-safety rule
|
|
108
|
+
-- (assertQueueSafe): ids and configuration, never bytes, class instances or
|
|
109
|
+
-- functions, because this value has to survive a queue message unchanged.
|
|
110
|
+
payload TEXT NOT NULL,
|
|
111
|
+
-- 'running' | 'done' | 'failed'. A killed run stays 'running' and is picked up
|
|
112
|
+
-- again by the next drive -- which is exactly what makes it restartable.
|
|
113
|
+
status TEXT NOT NULL,
|
|
114
|
+
-- What the last COMMITTED pass handed forward, as JSON. NULL = no pass has
|
|
115
|
+
-- committed yet, which is a fact ('start from the beginning'), not missing data.
|
|
116
|
+
cursor TEXT,
|
|
117
|
+
-- The run's counter bag (JSON object of numbers), merged on each commit. An
|
|
118
|
+
-- uncommitted pass's counts are discarded with the rest of the pass.
|
|
119
|
+
counters TEXT NOT NULL DEFAULT '{}',
|
|
120
|
+
-- CONSECUTIVE failed passes. Reset to 0 whenever a pass commits, so this reads
|
|
121
|
+
-- as "how stuck is it now", not "how much work has it done".
|
|
122
|
+
attempts INTEGER NOT NULL DEFAULT 0,
|
|
123
|
+
-- The error the last failed pass left. Retained after the run goes 'failed' --
|
|
124
|
+
-- the record is the evidence, so it must still say why.
|
|
125
|
+
last_error TEXT,
|
|
126
|
+
started_at TEXT NOT NULL,
|
|
127
|
+
updated_at TEXT NOT NULL,
|
|
128
|
+
-- When the next pass may run, on the executor's own backoff curve. NULL = now
|
|
129
|
+
-- (or terminal), which is why the due read tests IS NULL as well as <= now.
|
|
130
|
+
next_attempt_at TEXT,
|
|
131
|
+
-- When the run reached 'done' or 'failed'. NULL while it is still running.
|
|
132
|
+
ended_at TEXT
|
|
133
|
+
);
|
|
134
|
+
-- The drive's read: WHERE status = 'running' AND (next_attempt_at IS NULL OR <= ?)
|
|
135
|
+
-- ORDER BY id. Leading with status makes the live runs a seekable range over a
|
|
136
|
+
-- table that RETAINS every finished run, so the cost tracks how much is in flight
|
|
137
|
+
-- rather than how much the scope has ever imported.
|
|
138
|
+
CREATE INDEX IF NOT EXISTS _substrat_job_runs_due ON _substrat_job_runs (status, next_attempt_at, id);
|
|
139
|
+
-- Coalescing's read, and the operator read's filter.
|
|
140
|
+
CREATE INDEX IF NOT EXISTS _substrat_job_runs_key ON _substrat_job_runs (module_id, job, instance, id);
|
|
141
|
+
-- The step ledger of the pass currently in flight. Rows are written as each step
|
|
142
|
+
-- COMPLETES and dropped when the pass COMMITS, so this holds one pass's worth of
|
|
143
|
+
-- steps and never grows with the length of the walk.
|
|
144
|
+
--
|
|
145
|
+
-- A FAILED run keeps its last pass's rows, deliberately: they are the difference
|
|
146
|
+
-- between "it got nowhere" and "it got three quarters of the way and then the
|
|
147
|
+
-- provider went down", which the run row's single last_error cannot say. Still
|
|
148
|
+
-- bounded -- one pass's worth per failed run -- because a restart is a NEW run
|
|
149
|
+
-- with a new id and therefore its own ledger.
|
|
150
|
+
CREATE TABLE IF NOT EXISTS _substrat_job_steps (
|
|
151
|
+
run_id TEXT NOT NULL,
|
|
152
|
+
step TEXT NOT NULL,
|
|
153
|
+
-- The step's return value as JSON. NOT NULL is what MEANS completed: a step that
|
|
154
|
+
-- threw leaves the row with a NULL result and a raised attempts count, so the
|
|
155
|
+
-- next pass runs it again rather than reading a success that never happened. A
|
|
156
|
+
-- step returning nothing stores the JSON text 'null', which is not SQL NULL.
|
|
157
|
+
result TEXT,
|
|
158
|
+
attempts INTEGER NOT NULL DEFAULT 0,
|
|
159
|
+
last_error TEXT,
|
|
160
|
+
recorded_at TEXT NOT NULL,
|
|
161
|
+
PRIMARY KEY (run_id, step)
|
|
162
|
+
);
|
|
163
|
+
`;
|
|
164
|
+
/** Rows one `jobRuns` read returns by default. */
|
|
165
|
+
export const JOB_RUN_LIST_LIMIT = 50;
|
|
166
|
+
/** The most rows one `jobRuns` read will ever return, whatever the caller asks for. */
|
|
167
|
+
export const JOB_RUN_LIST_MAX = 500;
|
|
168
|
+
/**
|
|
169
|
+
* The row budget for one operator read, normalised before it reaches SQL.
|
|
170
|
+
*
|
|
171
|
+
* `JobRunFilter.limit` comes from a caller and was bound straight to `LIMIT`, where
|
|
172
|
+
* SQLite reads a NEGATIVE value as unbounded, refuses a fractional one outright, and
|
|
173
|
+
* honours an oversized one. Since finished runs are retained, "unbounded" means the
|
|
174
|
+
* scope's entire history in one response — a read whose cost grows with retention,
|
|
175
|
+
* reachable by passing `-1`. Clamped here rather than in each adapter so the pure and
|
|
176
|
+
* the hosted read cannot answer the same filter differently.
|
|
177
|
+
*/
|
|
178
|
+
export function jobRunListLimit(limit) {
|
|
179
|
+
if (limit === undefined || !Number.isFinite(limit))
|
|
180
|
+
return JOB_RUN_LIST_LIMIT;
|
|
181
|
+
return Math.min(JOB_RUN_LIST_MAX, Math.max(1, Math.floor(limit)));
|
|
182
|
+
}
|
|
183
|
+
/** Runs one `runDueJobs` call picks up by default. */
|
|
184
|
+
export const JOB_DRIVE_LIMIT = 50;
|
|
185
|
+
/**
|
|
186
|
+
* Rows one `runDueJobs` call will READ while looking for runnable ones.
|
|
187
|
+
*
|
|
188
|
+
* The drive skips runs whose job this host does not register, so "read `limit` rows"
|
|
189
|
+
* and "find `limit` runs to drive" are different numbers, and a scope can hold an
|
|
190
|
+
* arbitrary number of the unrunnable kind. This caps the difference: past it the call
|
|
191
|
+
* drives what it found and returns, rather than scanning a scope's whole history on a
|
|
192
|
+
* maintenance tick.
|
|
193
|
+
*/
|
|
194
|
+
export const JOB_DRIVE_SCAN_MAX = 500;
|
|
195
|
+
/** `conflict` reason: two `step()` calls under one name in one pass. */
|
|
196
|
+
export const JOB_STEP_REUSED = 'job_step_reused';
|
|
197
|
+
const message = (err) => (err instanceof Error ? err.message : String(err));
|
|
198
|
+
/** What a value is, for the refusal message: `a Uint8Array`, `a function`, `a Date`. */
|
|
199
|
+
function describe(value) {
|
|
200
|
+
const t = typeof value;
|
|
201
|
+
if (t === 'function')
|
|
202
|
+
return 'a function';
|
|
203
|
+
if (t === 'symbol')
|
|
204
|
+
return 'a symbol';
|
|
205
|
+
if (t === 'bigint')
|
|
206
|
+
return 'a bigint';
|
|
207
|
+
if (t === 'undefined')
|
|
208
|
+
return 'undefined';
|
|
209
|
+
if (t === 'number')
|
|
210
|
+
return `${String(value)}`;
|
|
211
|
+
const name = value?.constructor?.name;
|
|
212
|
+
return name ? `a ${name}` : 'a non-plain object';
|
|
213
|
+
}
|
|
214
|
+
/**
|
|
215
|
+
* Refuse anything that cannot survive a queue message, naming the path.
|
|
216
|
+
*
|
|
217
|
+
* What may be handed to a run, and handed forward by one, is **ids and
|
|
218
|
+
* configuration**: JSON's own values, nothing else. Bytes, class instances and
|
|
219
|
+
* functions are refused — and so, less obviously, are `undefined` in a nested
|
|
220
|
+
* position, a non-finite number and a cycle, because JSON turns each of those into
|
|
221
|
+
* something else without saying so. A payload silently reshaped on its way to
|
|
222
|
+
* storage is the failure this rule exists to prevent: the run resumes against a
|
|
223
|
+
* value that is not the one it was started with, and nothing anywhere reports it.
|
|
224
|
+
*
|
|
225
|
+
* `validation_failed` with the offending path in `errors`, exactly as an operation's
|
|
226
|
+
* own input failure arrives, so a transport renders it with no special case.
|
|
227
|
+
*
|
|
228
|
+
* Held against the payload at `start` and against the cursor at every commit —
|
|
229
|
+
* the two values that cross the boundary and land on the record. NOT held against a
|
|
230
|
+
* step's result, deliberately: that value is the handler's own, round-tripped to
|
|
231
|
+
* itself on resume, and the common shape of a step done for its effect is to return
|
|
232
|
+
* nothing at all — which this would refuse.
|
|
233
|
+
*/
|
|
234
|
+
export function assertQueueSafe(value, root) {
|
|
235
|
+
const open = new Set();
|
|
236
|
+
const reject = (path, what) => {
|
|
237
|
+
throw substratError('validation_failed', `${root} is not queue-safe: ${path} is ${what} — a run carries ids and configuration, ` +
|
|
238
|
+
'never bytes, class instances or functions', { errors: [{ path, message: `${what} cannot survive a queue message unchanged` }] });
|
|
239
|
+
};
|
|
240
|
+
const walk = (v, path) => {
|
|
241
|
+
if (v === null)
|
|
242
|
+
return;
|
|
243
|
+
const t = typeof v;
|
|
244
|
+
if (t === 'string' || t === 'boolean')
|
|
245
|
+
return;
|
|
246
|
+
if (t === 'number') {
|
|
247
|
+
if (Number.isFinite(v))
|
|
248
|
+
return;
|
|
249
|
+
reject(path, `${describe(v)}, which JSON stores as null`);
|
|
250
|
+
}
|
|
251
|
+
if (t !== 'object')
|
|
252
|
+
reject(path, describe(v));
|
|
253
|
+
const obj = v;
|
|
254
|
+
if (open.has(obj))
|
|
255
|
+
reject(path, 'a cycle back to a value already on this path');
|
|
256
|
+
open.add(obj);
|
|
257
|
+
if (Array.isArray(obj)) {
|
|
258
|
+
// BY INDEX, not `forEach`, and a HOLE is refused. `forEach` skips a sparse
|
|
259
|
+
// slot entirely, so `new Array(3)` walked cleanly and then stored as
|
|
260
|
+
// `[null,null,null]` — a value that passed the queue-safety check and changed
|
|
261
|
+
// on its way to storage, which is the one thing this function exists to stop.
|
|
262
|
+
for (let i = 0; i < obj.length; i += 1) {
|
|
263
|
+
if (!(i in obj))
|
|
264
|
+
reject(`${path}.${i}`, 'a hole in a sparse array, which JSON stores as null');
|
|
265
|
+
walk(obj[i], `${path}.${i}`);
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
else {
|
|
269
|
+
const proto = Object.getPrototypeOf(obj);
|
|
270
|
+
if (proto !== Object.prototype && proto !== null)
|
|
271
|
+
reject(path, describe(obj));
|
|
272
|
+
for (const [k, item] of Object.entries(obj))
|
|
273
|
+
walk(item, `${path}.${k}`);
|
|
274
|
+
}
|
|
275
|
+
open.delete(obj);
|
|
276
|
+
};
|
|
277
|
+
walk(value, root);
|
|
278
|
+
}
|
|
279
|
+
/**
|
|
280
|
+
* A row, decoded into the record an operator reads — TOLERANTLY, and saying so.
|
|
281
|
+
*
|
|
282
|
+
* The status, the attempts, the timestamps and the `last_error` are columns and
|
|
283
|
+
* always readable; only `payload`, `cursor` and `counters` are JSON, and a row whose
|
|
284
|
+
* JSON will not parse still has to be visible. It comes back with those three empty
|
|
285
|
+
* and `decodeError` naming the parse failure — never a bare `null` cursor, which a
|
|
286
|
+
* reader would take for "no pass has committed yet".
|
|
287
|
+
*
|
|
288
|
+
* This is deliberately NOT what the driver does with the same row: a pass cannot run
|
|
289
|
+
* on a payload it cannot decode, so `runJobPass` treats the parse failure as a failed
|
|
290
|
+
* pass and lets the run retry and then fail with the reason on its record. Strict
|
|
291
|
+
* where work happens, tolerant where evidence is read.
|
|
292
|
+
*/
|
|
293
|
+
export function jobRunOf(row) {
|
|
294
|
+
let decodeError = null;
|
|
295
|
+
const parse = (text, fallback) => {
|
|
296
|
+
if (text === null)
|
|
297
|
+
return fallback;
|
|
298
|
+
try {
|
|
299
|
+
return JSON.parse(text);
|
|
300
|
+
}
|
|
301
|
+
catch (err) {
|
|
302
|
+
decodeError ??= message(err);
|
|
303
|
+
return fallback;
|
|
304
|
+
}
|
|
305
|
+
};
|
|
306
|
+
// Every field is decoded before the error is read, so `decodeError` reports the
|
|
307
|
+
// FIRST failure of the row rather than whichever one happened to short-circuit.
|
|
308
|
+
const payload = parse(row.payload, null);
|
|
309
|
+
const cursor = parse(row.cursor, null);
|
|
310
|
+
const counters = parse(row.counters, {});
|
|
311
|
+
return {
|
|
312
|
+
id: row.id,
|
|
313
|
+
moduleId: row.module_id,
|
|
314
|
+
job: row.job,
|
|
315
|
+
instance: row.instance,
|
|
316
|
+
status: row.status,
|
|
317
|
+
payload,
|
|
318
|
+
cursor,
|
|
319
|
+
counters,
|
|
320
|
+
attempts: row.attempts,
|
|
321
|
+
lastError: row.last_error,
|
|
322
|
+
startedAt: row.started_at,
|
|
323
|
+
updatedAt: row.updated_at,
|
|
324
|
+
nextAttemptAt: row.next_attempt_at,
|
|
325
|
+
endedAt: row.ended_at,
|
|
326
|
+
decodeError,
|
|
327
|
+
};
|
|
328
|
+
}
|
|
329
|
+
/**
|
|
330
|
+
* Start a run, or JOIN the one already in flight (#1577's first acceptance).
|
|
331
|
+
*
|
|
332
|
+
* The payload is refused here, before a row exists, so a run is never recorded
|
|
333
|
+
* carrying a value it cannot resume from.
|
|
334
|
+
*
|
|
335
|
+
* The join returns the LIVE run unchanged — its cursor, its counters, its start
|
|
336
|
+
* time. A second caller therefore learns the id of the walk that is already
|
|
337
|
+
* happening and can watch it, which is what "joins the first" has to mean for the
|
|
338
|
+
* caller to be able to do anything with the answer.
|
|
339
|
+
*
|
|
340
|
+
* **The lookup and the insert are ONE store operation** (`startOrJoin`), not two.
|
|
341
|
+
* As two, concurrent starts both find no live run and both insert — and there is no
|
|
342
|
+
* unique index to catch the second, deliberately, because a crashed run must stay
|
|
343
|
+
* restartable. The coalescing guarantee would then be false exactly when it is load
|
|
344
|
+
* bearing: two callers asking at once, which is the case it exists for.
|
|
345
|
+
*/
|
|
346
|
+
export async function startJobRun(store, input, mintId, now) {
|
|
347
|
+
const payload = input.payload ?? null;
|
|
348
|
+
assertQueueSafe(payload, 'payload');
|
|
349
|
+
const key = {
|
|
350
|
+
moduleId: input.moduleId,
|
|
351
|
+
job: input.job,
|
|
352
|
+
instance: input.instance ?? 'default',
|
|
353
|
+
};
|
|
354
|
+
const at = now();
|
|
355
|
+
const row = {
|
|
356
|
+
id: mintId(),
|
|
357
|
+
module_id: key.moduleId,
|
|
358
|
+
job: key.job,
|
|
359
|
+
instance: key.instance,
|
|
360
|
+
payload: JSON.stringify(payload),
|
|
361
|
+
status: 'running',
|
|
362
|
+
cursor: null,
|
|
363
|
+
counters: '{}',
|
|
364
|
+
attempts: 0,
|
|
365
|
+
last_error: null,
|
|
366
|
+
started_at: at,
|
|
367
|
+
updated_at: at,
|
|
368
|
+
next_attempt_at: null,
|
|
369
|
+
ended_at: null,
|
|
370
|
+
};
|
|
371
|
+
// The row is built unconditionally — an id is minted and a start time stamped even
|
|
372
|
+
// when this call turns out to be a join. That is the price of doing the decision in
|
|
373
|
+
// one store operation, and it is cheap: a ULID nobody kept costs nothing, whereas a
|
|
374
|
+
// "look first so we do not waste an id" round trip is the race this exists to close.
|
|
375
|
+
return store.startOrJoin(key, row);
|
|
376
|
+
}
|
|
377
|
+
/** A step that threw, carrying what the driver needs to decide the run's fate. */
|
|
378
|
+
class JobStepFailure extends Error {
|
|
379
|
+
step;
|
|
380
|
+
stepAttempts;
|
|
381
|
+
policy;
|
|
382
|
+
cause;
|
|
383
|
+
constructor(step, stepAttempts, policy, cause) {
|
|
384
|
+
super(`step '${step}' failed: ${cause}`);
|
|
385
|
+
this.step = step;
|
|
386
|
+
this.stepAttempts = stepAttempts;
|
|
387
|
+
this.policy = policy;
|
|
388
|
+
this.cause = cause;
|
|
389
|
+
this.name = 'JobStepFailure';
|
|
390
|
+
}
|
|
391
|
+
}
|
|
392
|
+
/**
|
|
393
|
+
* Run ONE pass of one run, and write what happened.
|
|
394
|
+
*
|
|
395
|
+
* The whole of the contract is here, which is why both adapters call it rather than
|
|
396
|
+
* porting it:
|
|
397
|
+
*
|
|
398
|
+
* - A committed pass writes the new cursor and counters, clears `attempts` and
|
|
399
|
+
* `last_error`, and DROPS the step ledger — a new pass is new work.
|
|
400
|
+
* - A failed pass writes NOTHING the handler produced: the cursor and counters stay
|
|
401
|
+
* where the last commit left them, so a partially-walked chunk is never mistaken
|
|
402
|
+
* for a committed one. The step ledger SURVIVES, which is what stops the next pass
|
|
403
|
+
* repeating the steps that did succeed.
|
|
404
|
+
* - A step at its `maxAttempts` fails the RUN: status `failed`, the step's error on
|
|
405
|
+
* the record, `ended_at` stamped. It is reported, never thrown — a driver that
|
|
406
|
+
* let one run's failure escape would take down every run behind it, which is the
|
|
407
|
+
* failure `runDueSchedules` already refuses for schedules.
|
|
408
|
+
*/
|
|
409
|
+
export async function runJobPass(options) {
|
|
410
|
+
const { store, run, handler, now, openScope } = options;
|
|
411
|
+
const jobPolicy = resolveRetryPolicy(options.retry);
|
|
412
|
+
const usedThisPass = new Set();
|
|
413
|
+
let scope = null;
|
|
414
|
+
try {
|
|
415
|
+
// DECODING THE ROW IS PART OF THE PASS, not a precondition of it.
|
|
416
|
+
//
|
|
417
|
+
// These three `JSON.parse`es sat above the `try` in the first cut of this file,
|
|
418
|
+
// which made a single malformed row the one failure this driver could not
|
|
419
|
+
// contain: the throw left `runJobPass`, left `runDueJobRuns` — which does not
|
|
420
|
+
// wrap the call either — and came out of `runDueJobs` at whoever was holding the
|
|
421
|
+
// tick, taking every due run BEHIND it with it. That is the exact outcome the
|
|
422
|
+
// per-run isolation exists to refuse, arrived at through the driver itself.
|
|
423
|
+
//
|
|
424
|
+
// It is reachable without any forge: `importDump` replays a dump's rows verbatim
|
|
425
|
+
// (preview-and-snapshots.md §3), so a restore from a foreign or hand-edited dump
|
|
426
|
+
// is enough. Inside the try it is an ordinary failed pass — recorded on the run,
|
|
427
|
+
// retried, then terminal with the parse error as its `last_error`, and the runs
|
|
428
|
+
// beside it untouched.
|
|
429
|
+
// The pass's own copy: the catch below writes `run.counters` back — the string
|
|
430
|
+
// the last COMMIT wrote — so a failed pass's counts go with the rest of it.
|
|
431
|
+
const counters = { ...JSON.parse(run.counters) };
|
|
432
|
+
const pass = {
|
|
433
|
+
run: {
|
|
434
|
+
id: run.id,
|
|
435
|
+
moduleId: run.module_id,
|
|
436
|
+
job: run.job,
|
|
437
|
+
instance: run.instance,
|
|
438
|
+
},
|
|
439
|
+
payload: JSON.parse(run.payload),
|
|
440
|
+
cursor: run.cursor === null ? null : JSON.parse(run.cursor),
|
|
441
|
+
counters,
|
|
442
|
+
count: (name, by = 1) => {
|
|
443
|
+
counters[name] = (counters[name] ?? 0) + by;
|
|
444
|
+
},
|
|
445
|
+
scope: () => (scope ??= openScope()),
|
|
446
|
+
step: async (name, fn, retry) => {
|
|
447
|
+
if (usedThisPass.has(name)) {
|
|
448
|
+
// The determinism rule's mechanical half. The second call would read the
|
|
449
|
+
// first's memo and skip its own work — silently, and only in production,
|
|
450
|
+
// because the first pass of a fresh run runs both bodies before either is
|
|
451
|
+
// committed. Refused where it is unambiguous rather than left to a comment.
|
|
452
|
+
throw substratError('conflict', `step '${name}' was already run in this pass — a step name identifies one unit of ` +
|
|
453
|
+
'work, so a second call under it would return the first one\'s result instead of ' +
|
|
454
|
+
'doing anything', { reason: JOB_STEP_REUSED });
|
|
455
|
+
}
|
|
456
|
+
usedThisPass.add(name);
|
|
457
|
+
const policy = resolveRetryPolicy(retry ?? options.retry);
|
|
458
|
+
const prior = await store.step(run.id, name);
|
|
459
|
+
// A NON-NULL result is what means completed: a step that threw left its row
|
|
460
|
+
// with a null result and a raised count, and must run again.
|
|
461
|
+
if (prior && prior.result !== null)
|
|
462
|
+
return JSON.parse(prior.result);
|
|
463
|
+
const attempts = (prior?.attempts ?? 0) + 1;
|
|
464
|
+
// A step whose RECORDED attempts already reached the policy is spent, and
|
|
465
|
+
// calling `fn` again would be one more real request to somebody else's API
|
|
466
|
+
// for a run that is going to fail anyway. Reachable: a stop after
|
|
467
|
+
// `recordStep` wrote the final failed attempt but before the run patch below
|
|
468
|
+
// marked the run terminal leaves exactly this row, and the next drive would
|
|
469
|
+
// otherwise spend one extra attempt discovering what the row already says.
|
|
470
|
+
if (prior && prior.attempts >= policy.maxAttempts) {
|
|
471
|
+
throw new JobStepFailure(name, prior.attempts, policy, prior.last_error ?? 'exhausted');
|
|
472
|
+
}
|
|
473
|
+
let value;
|
|
474
|
+
try {
|
|
475
|
+
value = await fn();
|
|
476
|
+
}
|
|
477
|
+
catch (err) {
|
|
478
|
+
const cause = message(err);
|
|
479
|
+
await store.recordStep(run.id, name, null, attempts, cause, now());
|
|
480
|
+
throw new JobStepFailure(name, attempts, policy, cause);
|
|
481
|
+
}
|
|
482
|
+
// `undefined` becomes the JSON text 'null', not SQL NULL: a step done purely
|
|
483
|
+
// for its effect must still read as completed on the next pass.
|
|
484
|
+
const stored = JSON.stringify(value) ?? 'null';
|
|
485
|
+
await store.recordStep(run.id, name, stored, attempts, null, now());
|
|
486
|
+
// RETURNED THROUGH THE STORED FORM, not as the raw value. The resume path
|
|
487
|
+
// returns `JSON.parse(row.result)`, so returning `value` here would hand the
|
|
488
|
+
// handler a `Date` on the first pass and the string `"2026-01-01T…"` on the
|
|
489
|
+
// replay — the same code taking a different branch depending on whether it
|
|
490
|
+
// was interrupted, which is the determinism failure this driver is most
|
|
491
|
+
// exposed to and the one least likely to be noticed in a test that never
|
|
492
|
+
// resumes. Both paths now return the same shape.
|
|
493
|
+
return JSON.parse(stored);
|
|
494
|
+
},
|
|
495
|
+
};
|
|
496
|
+
const result = (await handler(pass)) ?? {};
|
|
497
|
+
const keepsCursor = !('cursor' in result);
|
|
498
|
+
// NOT `result.cursor ?? null`. An explicitly supplied `cursor: undefined` is a
|
|
499
|
+
// supplied cursor — `'cursor' in result` is true — and coalescing it to null
|
|
500
|
+
// before the check meant it validated cleanly and RESET the walk to the
|
|
501
|
+
// beginning. The difference between "keep going" and "start over" turned on a
|
|
502
|
+
// `??`, silently, in the direction that repeats an hour of work. `assertQueueSafe`
|
|
503
|
+
// already refuses `undefined`; it simply never saw it.
|
|
504
|
+
if (!keepsCursor)
|
|
505
|
+
assertQueueSafe(result.cursor, 'cursor');
|
|
506
|
+
const at = now();
|
|
507
|
+
const done = result.done === true;
|
|
508
|
+
// ONE operation: the run's new state and the dropping of its step ledger commit
|
|
509
|
+
// together or not at all.
|
|
510
|
+
//
|
|
511
|
+
// This was two calls, ordered patch-then-clear, with a comment explaining that a
|
|
512
|
+
// stop in between was harmless because the next pass's step names "derive from
|
|
513
|
+
// the new cursor" and would miss the stale rows. THAT WAS FALSE. The determinism
|
|
514
|
+
// rule binds a step name to the payload and prior results, not to the cursor, so
|
|
515
|
+
// a handler naming its steps `fetch-page` / `write-batch` — legal, and the
|
|
516
|
+
// obvious way to write one — hits the finished pass's memo on the next pass and
|
|
517
|
+
// skips work it never did. The gap was also wrong in the other direction: a
|
|
518
|
+
// throw from the clear landed in the catch below, which wrote the OLD cursor
|
|
519
|
+
// back and filed an already-committed pass as failed, so the record a human
|
|
520
|
+
// reads to recover would have understated the run's own progress.
|
|
521
|
+
await store.commitPass(run.id, {
|
|
522
|
+
status: done ? 'done' : 'running',
|
|
523
|
+
cursor: keepsCursor ? run.cursor : JSON.stringify(result.cursor),
|
|
524
|
+
counters: JSON.stringify(counters),
|
|
525
|
+
attempts: 0,
|
|
526
|
+
lastError: null,
|
|
527
|
+
updatedAt: at,
|
|
528
|
+
nextAttemptAt: null,
|
|
529
|
+
endedAt: done ? at : null,
|
|
530
|
+
});
|
|
531
|
+
return { status: done ? 'completed' : 'advanced' };
|
|
532
|
+
}
|
|
533
|
+
catch (err) {
|
|
534
|
+
const stepFailure = err instanceof JobStepFailure ? err : null;
|
|
535
|
+
const policy = stepFailure?.policy ?? jobPolicy;
|
|
536
|
+
// The RUN's attempts counts consecutive failed passes — what an operator reads as
|
|
537
|
+
// "how stuck is it". The STEP's own count is what decides exhaustion and backoff,
|
|
538
|
+
// because a step's policy is a fact about that step.
|
|
539
|
+
const attempts = run.attempts + 1;
|
|
540
|
+
const against = stepFailure?.stepAttempts ?? attempts;
|
|
541
|
+
const exhausted = against >= policy.maxAttempts;
|
|
542
|
+
const error = message(err);
|
|
543
|
+
const at = now();
|
|
544
|
+
await store.patch(run.id, {
|
|
545
|
+
status: exhausted ? 'failed' : 'running',
|
|
546
|
+
cursor: run.cursor,
|
|
547
|
+
counters: run.counters,
|
|
548
|
+
attempts,
|
|
549
|
+
lastError: error,
|
|
550
|
+
updatedAt: at,
|
|
551
|
+
nextAttemptAt: exhausted ? null : backoffAt(against, policy, new Date(at)),
|
|
552
|
+
endedAt: exhausted ? at : null,
|
|
553
|
+
});
|
|
554
|
+
return { status: exhausted ? 'failed' : 'retrying', error };
|
|
555
|
+
}
|
|
556
|
+
}
|
|
557
|
+
/**
|
|
558
|
+
* Advance every due run on one scope — the driver both adapters expose as
|
|
559
|
+
* `runDueJobs`.
|
|
560
|
+
*
|
|
561
|
+
* Bounded on both axes, because a maintenance tick has a budget: `limit` runs per
|
|
562
|
+
* call, `maxPasses` passes per run. A run that still has work left after its budget
|
|
563
|
+
* is simply left `running` and due, and the next call takes it — the same "reported
|
|
564
|
+
* rather than looped" shape the event drain's `incomplete` has. Default `maxPasses`
|
|
565
|
+
* is 1, so a caller gets one predictable unit of work unless it asks for more.
|
|
566
|
+
*
|
|
567
|
+
* **`limit` counts RUNNABLE runs, not rows read**, and that distinction is a fix
|
|
568
|
+
* rather than a nicety. Runs whose job this host does not register are skipped, and
|
|
569
|
+
* when the budget was applied to the query instead, a scope holding `limit` such rows
|
|
570
|
+
* at the head of the due order returned the same unrunnable batch on every call —
|
|
571
|
+
* nothing newer was ever reached, and the report said `attempted: 0` forever with no
|
|
572
|
+
* indication why. So the read pages past them, bounded by `JOB_DRIVE_SCAN_MAX` rows
|
|
573
|
+
* examined so one scope full of orphans cannot turn a tick into a table scan.
|
|
574
|
+
*
|
|
575
|
+
* The starvation was spotted while re-reading this file, judged unlikely and left
|
|
576
|
+
* alone — and then found independently by a reviewer. The judgement may even have
|
|
577
|
+
* been right; recording it only in the author's head was not, because a decision
|
|
578
|
+
* nobody can see is indistinguishable from an oversight. Hence the fix and hence
|
|
579
|
+
* this paragraph.
|
|
580
|
+
*/
|
|
581
|
+
export async function runDueJobRuns(options) {
|
|
582
|
+
const report = {
|
|
583
|
+
attempted: 0,
|
|
584
|
+
advanced: 0,
|
|
585
|
+
completed: 0,
|
|
586
|
+
retrying: 0,
|
|
587
|
+
failed: 0,
|
|
588
|
+
errors: [],
|
|
589
|
+
};
|
|
590
|
+
const maxPasses = Math.max(1, options.maxPasses ?? 1);
|
|
591
|
+
const want = options.limit ?? JOB_DRIVE_LIMIT;
|
|
592
|
+
// Page the due read until `want` RUNNABLE runs have been gathered, or the scope
|
|
593
|
+
// runs out, or the scan cap is hit. A run whose job this host does not register —
|
|
594
|
+
// another deployment's, or one whose registration was removed — is stepped over
|
|
595
|
+
// and NOT failed: failing it would destroy a resumable run because the wrong
|
|
596
|
+
// process happened to look at it.
|
|
597
|
+
const runnable = [];
|
|
598
|
+
let scanned = 0;
|
|
599
|
+
let afterId;
|
|
600
|
+
while (runnable.length < want && scanned < JOB_DRIVE_SCAN_MAX) {
|
|
601
|
+
const batch = await options.store.due(options.now(), want, afterId);
|
|
602
|
+
if (batch.length === 0)
|
|
603
|
+
break;
|
|
604
|
+
scanned += batch.length;
|
|
605
|
+
for (const row of batch) {
|
|
606
|
+
if (options.handlerFor(row) && runnable.length < want)
|
|
607
|
+
runnable.push(row);
|
|
608
|
+
}
|
|
609
|
+
afterId = batch[batch.length - 1].id;
|
|
610
|
+
// A short batch is the end of the due set; another round trip would read nothing.
|
|
611
|
+
if (batch.length < want)
|
|
612
|
+
break;
|
|
613
|
+
}
|
|
614
|
+
for (const row of runnable) {
|
|
615
|
+
const registered = options.handlerFor(row);
|
|
616
|
+
report.attempted += 1;
|
|
617
|
+
let run = row;
|
|
618
|
+
for (let pass = 0; pass < maxPasses; pass += 1) {
|
|
619
|
+
const outcome = await runJobPass({
|
|
620
|
+
store: options.store,
|
|
621
|
+
run,
|
|
622
|
+
handler: registered.handler,
|
|
623
|
+
retry: registered.retry,
|
|
624
|
+
now: options.now,
|
|
625
|
+
openScope: () => options.openScope(run),
|
|
626
|
+
});
|
|
627
|
+
if (outcome.status === 'completed') {
|
|
628
|
+
report.completed += 1;
|
|
629
|
+
break;
|
|
630
|
+
}
|
|
631
|
+
if (outcome.status === 'failed') {
|
|
632
|
+
report.failed += 1;
|
|
633
|
+
report.errors.push({ runId: run.id, error: outcome.error });
|
|
634
|
+
break;
|
|
635
|
+
}
|
|
636
|
+
if (outcome.status === 'retrying') {
|
|
637
|
+
report.retrying += 1;
|
|
638
|
+
report.errors.push({ runId: run.id, error: outcome.error });
|
|
639
|
+
break;
|
|
640
|
+
}
|
|
641
|
+
report.advanced += 1;
|
|
642
|
+
// A second pass in this call resumes from what the first one COMMITTED, read
|
|
643
|
+
// back rather than reconstructed: the cursor is the handler's value and the
|
|
644
|
+
// store is the only thing that knows it survived the write.
|
|
645
|
+
if (pass + 1 >= maxPasses)
|
|
646
|
+
break;
|
|
647
|
+
const fresh = await options.store.get(run.id);
|
|
648
|
+
if (!fresh || fresh.status !== 'running')
|
|
649
|
+
break;
|
|
650
|
+
run = fresh;
|
|
651
|
+
}
|
|
652
|
+
}
|
|
653
|
+
return report;
|
|
654
|
+
}
|
|
655
|
+
//# sourceMappingURL=job-run.js.map
|