lightflow-engine 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +200 -0
- package/dist/index.js +409 -0
- package/dist/pg-store.js +83 -0
- package/dist/src/index.js +430 -0
- package/dist/src/pg-store.js +194 -0
- package/dist/test/coverage.js +107 -0
- package/dist/test/drain.js +21 -0
- package/dist/test/entry-parity.js +115 -0
- package/dist/test/resume-all.js +14 -0
- package/dist/test/resume.js +12 -0
- package/dist/test/simulate.js +46 -0
- package/dist/test/smoke.js +157 -0
- package/dist/test/workflow-def.js +54 -0
- package/package.json +50 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 lightflow contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
# lightflow
|
|
2
|
+
|
|
3
|
+
**A tiny durable workflow engine for Node.js and Postgres.**
|
|
4
|
+
|
|
5
|
+
Steps that survive crashes. Timers that survive restarts. Streams that replay.
|
|
6
|
+
~700 lines of TypeScript, two Postgres tables, one dependency (`pg`).
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
npm install lightflow-engine
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
## Why
|
|
13
|
+
|
|
14
|
+
If you've ever needed a workflow where *the server can die at any moment* and
|
|
15
|
+
the work must continue correctly — background jobs, agent loops, billing
|
|
16
|
+
pipelines, sandbox lifecycle management — you've had to choose between:
|
|
17
|
+
|
|
18
|
+
- **Temporal / Cadence**: extremely powerful, extremely heavy to operate
|
|
19
|
+
- **Vercel Workflow / Inngest**: excellent — Vercel Workflow is open source
|
|
20
|
+
(Apache-2.0) and self-hostable via `@workflow/world-postgres`; lightflow
|
|
21
|
+
trades its adapter/SDK surface for a single engine you can read end-to-end
|
|
22
|
+
- **BullMQ / plain queues**: a queue is not durable execution — one crash and
|
|
23
|
+
you re-run side effects (and re-charge your LLM provider)
|
|
24
|
+
|
|
25
|
+
**lightflow** is the middle path: a real durable-execution engine you can run
|
|
26
|
+
yourself in an afternoon. It gives you the primitives that matter:
|
|
27
|
+
|
|
28
|
+
| Primitive | What it means |
|
|
29
|
+
|---|---|
|
|
30
|
+
| `step()` | Side effects run **at most once** — memoized in Postgres |
|
|
31
|
+
| `sleep(Date \| ms)` | Timers survive process death; a worker wakes them |
|
|
32
|
+
| `getWritable()` | Ordered streams with **exactly-once** emission and replay |
|
|
33
|
+
| `start()` / `getRun()` | Launch runs, resume by ID, await results |
|
|
34
|
+
| `FatalError` | Errors that must never be retried |
|
|
35
|
+
| `cancel()` | Idempotent cancellation that propagates to the run |
|
|
36
|
+
| Hooks | Durable external callbacks a workflow can await |
|
|
37
|
+
| Reaper | Stuck `running` runs are automatically recovered |
|
|
38
|
+
|
|
39
|
+
## The one rule
|
|
40
|
+
|
|
41
|
+
> **A completed step must never execute again.**
|
|
42
|
+
|
|
43
|
+
That's the correctness contract everything else serves. If a step charged a
|
|
44
|
+
credit card or committed to GitHub, replaying it after a crash is a bug.
|
|
45
|
+
lightflow keys every step by its deterministic call position and memoizes the
|
|
46
|
+
result in Postgres — verified under `SIGKILL`-mid-flight stress tests.
|
|
47
|
+
|
|
48
|
+
## Quickstart
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
# 1. A Postgres database (any instance, any version)
|
|
52
|
+
export LIGHTFLOW_PG_URL=postgres://user:pass@localhost:5432/lightflow
|
|
53
|
+
|
|
54
|
+
# 2. Install
|
|
55
|
+
npm install lightflow-engine
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
```ts
|
|
59
|
+
import { Engine, step, sleep, getWritable, FatalError } from "lightflow-engine";
|
|
60
|
+
import { createPostgresStore } from "lightflow-engine/pg";
|
|
61
|
+
|
|
62
|
+
// 1. Define a workflow — plain async functions
|
|
63
|
+
async function onboardingWorkflow(user: { id: string; email: string }) {
|
|
64
|
+
const writable = getWritable<string>();
|
|
65
|
+
|
|
66
|
+
await writable.write("stage:start");
|
|
67
|
+
|
|
68
|
+
// Side effects are durable: retried on transient failure (3 attempts),
|
|
69
|
+
// never re-executed once completed.
|
|
70
|
+
const token = await step(async () => sendWelcomeEmail(user.email));
|
|
71
|
+
|
|
72
|
+
// Durable timer — the process can die here; a worker wakes the run later.
|
|
73
|
+
await sleep(new Date(Date.now() + 24 * 60 * 60 * 1000));
|
|
74
|
+
|
|
75
|
+
await writable.write("stage:followup-sent");
|
|
76
|
+
await writable.close();
|
|
77
|
+
|
|
78
|
+
return { userId: user.id, token, done: true };
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
// 2. Register + run
|
|
82
|
+
const store = await createPostgresStore(process.env.LIGHTFLOW_PG_URL!);
|
|
83
|
+
const engine = new Engine(store, { pollMs: 300, staleRunMs: 60_000 });
|
|
84
|
+
|
|
85
|
+
engine.register("onboarding", onboardingWorkflow);
|
|
86
|
+
void engine.startWorker(); // background loop: timers + stale-run recovery
|
|
87
|
+
|
|
88
|
+
const { runId } = await engine.start("onboarding", [
|
|
89
|
+
{ id: "u1", email: "ada@example.com" },
|
|
90
|
+
]);
|
|
91
|
+
|
|
92
|
+
// 3. Stream results (replayable — new clients can reconnect mid-run)
|
|
93
|
+
const run = await engine.getRun(runId);
|
|
94
|
+
for await (const chunk of run.getReadable()) {
|
|
95
|
+
console.log(chunk); // "stage:start" ...
|
|
96
|
+
}
|
|
97
|
+
console.log(await run.returnValue);
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
That's the whole API.
|
|
101
|
+
|
|
102
|
+
## Retry classification
|
|
103
|
+
|
|
104
|
+
```ts
|
|
105
|
+
import { step, FatalError } from "lightflow-engine";
|
|
106
|
+
|
|
107
|
+
// Transient errors: retried with exponential backoff (default 3 attempts).
|
|
108
|
+
await step(async () => callFlakyApi());
|
|
109
|
+
|
|
110
|
+
// Permanent errors: never retried, run fails immediately.
|
|
111
|
+
await step(async () => {
|
|
112
|
+
if (badInput) throw new FatalError("cannot process this input");
|
|
113
|
+
return doWork();
|
|
114
|
+
});
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
## Cancellation
|
|
118
|
+
|
|
119
|
+
```ts
|
|
120
|
+
const run = await engine.getRun(runId);
|
|
121
|
+
await run.cancel(); // idempotent; safe to call twice
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Cancellation marks the run terminal and is checked at every suspend/resume
|
|
125
|
+
boundary, so in-flight sleeps and steps wind down cleanly.
|
|
126
|
+
|
|
127
|
+
## Hooks (wait for the outside world)
|
|
128
|
+
|
|
129
|
+
```ts
|
|
130
|
+
// Inside a workflow:
|
|
131
|
+
const { token } = await defineHook().create();
|
|
132
|
+
const approval = await hookResult<boolean>(token); // suspends the run
|
|
133
|
+
|
|
134
|
+
// Outside (HTTP handler, CLI, another service):
|
|
135
|
+
await store.resolveHook(token, { approved: true }); // run wakes up
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
## Crash-recovery, tested
|
|
139
|
+
|
|
140
|
+
```text
|
|
141
|
+
30 concurrent workflows × 20 durable timers each
|
|
142
|
+
SIGKILL the entire process mid-flight
|
|
143
|
+
resume in a fresh process
|
|
144
|
+
|
|
145
|
+
→ 30/30 completed, 0 stuck, 0 duplicate side effects, 0 duplicate chunks
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
Kill-and-resume is a first-class path, not an edge case. The engine includes a
|
|
149
|
+
**reaper** that resumes runs wedged in `running` — a failure mode we hit in
|
|
150
|
+
other engines and designed out here.
|
|
151
|
+
|
|
152
|
+
## Design decisions
|
|
153
|
+
|
|
154
|
+
- **All timestamps are BIGINT epoch milliseconds.** No `timestamptz`, no
|
|
155
|
+
timezone skew. (A naive-UTC bug in another engine cost us 8 hours of silent
|
|
156
|
+
workflow delay; this schema makes that class of bug impossible.)
|
|
157
|
+
- **Plain JSON state only.** Workflows receive and return serializable data.
|
|
158
|
+
No closures across the boundary — reconstruct clients inside steps.
|
|
159
|
+
- **Two tables.** `lightflow_runs` + `lightflow_events`. Event-sourced replay:
|
|
160
|
+
the workflow function is re-executed and step results are restored from the
|
|
161
|
+
event log, so the engine can be debugged with plain SQL.
|
|
162
|
+
|
|
163
|
+
## Comparison
|
|
164
|
+
|
|
165
|
+
| | lightflow | Temporal | Vercel Workflow | BullMQ |
|
|
166
|
+
|---|---|---|---|---|
|
|
167
|
+
| Durable steps | ✅ | ✅ | ✅ | ❌ |
|
|
168
|
+
| Durable timers | ✅ | ✅ | ✅ | delayed jobs |
|
|
169
|
+
| Replayable streams | ✅ | ❌ | ✅ | ❌ |
|
|
170
|
+
| Dependencies | 1 (`pg`) | many | multi-package | redis |
|
|
171
|
+
| Lines of core code | ~700 | 100k+ | large monorepo | ~10k |
|
|
172
|
+
|
|
173
|
+
Vercel Workflow is itself open source (Apache-2.0) and self-hostable via its
|
|
174
|
+
Postgres world (`@workflow/world-postgres`) — lightflow is the alternative for
|
|
175
|
+
when you want a single small engine you can read end-to-end, not an SDK with
|
|
176
|
+
an adapter system.
|
|
177
|
+
|
|
178
|
+
## Status
|
|
179
|
+
|
|
180
|
+
**Pre-1.0 / experimental.** The core is stress-tested (crash recovery,
|
|
181
|
+
concurrency, exactly-once steps and streams). Not yet battle-tested in
|
|
182
|
+
production. Known gaps:
|
|
183
|
+
|
|
184
|
+
- No bundler plugin — `step()` is an explicit call, not a `"use step"` directive
|
|
185
|
+
- Single-writer per run (multi-worker is safe but one worker wins; no
|
|
186
|
+
leader election yet)
|
|
187
|
+
- No Web dashboard (query Postgres directly for now)
|
|
188
|
+
|
|
189
|
+
## Contributing
|
|
190
|
+
|
|
191
|
+
PRs welcome. Run tests with a local Postgres:
|
|
192
|
+
|
|
193
|
+
```bash
|
|
194
|
+
export LIGHTFLOW_PG_URL=postgres://localhost/lightflow_test
|
|
195
|
+
npm test # covers crash recovery + concurrency
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
## License
|
|
199
|
+
|
|
200
|
+
MIT
|
package/dist/index.js
ADDED
|
@@ -0,0 +1,409 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* lightflow — a durable workflow engine, from scratch.
|
|
3
|
+
*
|
|
4
|
+
* Implements every primitive Entry's workflows rely on:
|
|
5
|
+
* - "use workflow" deterministic orchestration, replayed from an event log
|
|
6
|
+
* - "use step" at-least-once, memoized side effects
|
|
7
|
+
* - sleep(ms | Date) durable timers (survive process death)
|
|
8
|
+
* - getWritable() ordered, resumable output stream
|
|
9
|
+
* - start() / getRun() start, resume by id, await returnValue
|
|
10
|
+
* - FatalError non-retryable failure
|
|
11
|
+
* - getWorkflowMetadata() run id inside a workflow
|
|
12
|
+
*
|
|
13
|
+
* Design goals that fix the two defects found in world-postgres:
|
|
14
|
+
* 1. All timestamps are epoch milliseconds (integer). No naive/UTC skew.
|
|
15
|
+
* 2. No spec-version coupling: events are plain JSON rows with a schema
|
|
16
|
+
* version integer that the engine upgrades itself.
|
|
17
|
+
*/
|
|
18
|
+
import { createHash, randomUUID } from "node:crypto";
|
|
19
|
+
/* ------------------------------------------------------------------ */
|
|
20
|
+
/* Errors */
|
|
21
|
+
/* ------------------------------------------------------------------ */
|
|
22
|
+
export class FatalError extends Error {
|
|
23
|
+
fatal = true;
|
|
24
|
+
constructor(message) {
|
|
25
|
+
super(message);
|
|
26
|
+
this.name = "FatalError";
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
const workflows = new Map();
|
|
30
|
+
const steps = new Map();
|
|
31
|
+
export function registerWorkflow(id, fn) {
|
|
32
|
+
workflows.set(id, fn);
|
|
33
|
+
}
|
|
34
|
+
export function registerStep(id, fn) {
|
|
35
|
+
steps.set(id, fn);
|
|
36
|
+
}
|
|
37
|
+
let current = null;
|
|
38
|
+
export function getWorkflowMetadata() {
|
|
39
|
+
if (!current)
|
|
40
|
+
throw new Error("getWorkflowMetadata() outside a workflow");
|
|
41
|
+
return { runId: current.runId };
|
|
42
|
+
}
|
|
43
|
+
/* ------------------------------------------------------------------ */
|
|
44
|
+
/* The step id: derived from call order, so replay is deterministic */
|
|
45
|
+
/* ------------------------------------------------------------------ */
|
|
46
|
+
function nextStepKey() {
|
|
47
|
+
if (!current)
|
|
48
|
+
throw new Error("steps can only be called inside a workflow");
|
|
49
|
+
return `step:${current.stepCalls}`;
|
|
50
|
+
}
|
|
51
|
+
/* ------------------------------------------------------------------ */
|
|
52
|
+
/* Public API inside a workflow */
|
|
53
|
+
/* ------------------------------------------------------------------ */
|
|
54
|
+
/**
|
|
55
|
+
* Run a side effect durably. On replay the memoized result is returned and
|
|
56
|
+
* the function body is NOT re-executed.
|
|
57
|
+
*/
|
|
58
|
+
export async function step(fn) {
|
|
59
|
+
if (!current)
|
|
60
|
+
return fn(); // plain call outside a workflow
|
|
61
|
+
const ctx = current;
|
|
62
|
+
const key = nextStepKey();
|
|
63
|
+
ctx.stepCalls += 1;
|
|
64
|
+
// Replay: memoized result for this call position?
|
|
65
|
+
const done = ctx.log.find((e) => e.type === "step_completed" && e.payload?.key === key);
|
|
66
|
+
if (done)
|
|
67
|
+
return done.payload.value;
|
|
68
|
+
ctx.seq += 1;
|
|
69
|
+
await ctx.store.appendEvent({
|
|
70
|
+
runId: ctx.runId, seq: ctx.seq, type: "step_started",
|
|
71
|
+
payload: { key }, createdAt: ctx.now(),
|
|
72
|
+
});
|
|
73
|
+
const fnId = fn.__lightflowStepId;
|
|
74
|
+
const impl = fnId ? steps.get(fnId) : undefined;
|
|
75
|
+
// Retry with backoff. FatalError is never retried (Entry retries 4x, then
|
|
76
|
+
// bubbles FatalError -- same contract here).
|
|
77
|
+
const maxAttempts = fn
|
|
78
|
+
.__lightflowRetries ?? DEFAULT_STEP_RETRIES;
|
|
79
|
+
let lastErr;
|
|
80
|
+
let value;
|
|
81
|
+
for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
|
|
82
|
+
try {
|
|
83
|
+
value = await (impl ? impl() : fn());
|
|
84
|
+
lastErr = undefined;
|
|
85
|
+
break;
|
|
86
|
+
}
|
|
87
|
+
catch (err) {
|
|
88
|
+
lastErr = err;
|
|
89
|
+
if (err instanceof FatalError)
|
|
90
|
+
break;
|
|
91
|
+
if (attempt < maxAttempts) {
|
|
92
|
+
await new Promise((r) => setTimeout(r, Math.min(100 * 2 ** (attempt - 1), 2000)));
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
if (lastErr instanceof FatalError)
|
|
97
|
+
throw lastErr;
|
|
98
|
+
if (lastErr) {
|
|
99
|
+
ctx.seq += 1;
|
|
100
|
+
await ctx.store.appendEvent({
|
|
101
|
+
runId: ctx.runId, seq: ctx.seq, type: "step_failed",
|
|
102
|
+
payload: { key, error: String(lastErr instanceof Error ? lastErr.message : lastErr) },
|
|
103
|
+
createdAt: ctx.now(),
|
|
104
|
+
});
|
|
105
|
+
throw lastErr;
|
|
106
|
+
}
|
|
107
|
+
ctx.seq += 1;
|
|
108
|
+
await ctx.store.appendEvent({
|
|
109
|
+
runId: ctx.runId, seq: ctx.seq, type: "step_completed",
|
|
110
|
+
payload: { key, value }, createdAt: ctx.now(),
|
|
111
|
+
});
|
|
112
|
+
return value;
|
|
113
|
+
}
|
|
114
|
+
/** Durable sleep. Accepts milliseconds or an absolute Date. */
|
|
115
|
+
export async function sleep(until) {
|
|
116
|
+
if (!current) {
|
|
117
|
+
const ms = until instanceof Date ? until.getTime() - Date.now() : until;
|
|
118
|
+
return new Promise((r) => setTimeout(r, Math.max(0, ms)));
|
|
119
|
+
}
|
|
120
|
+
const ctx = current;
|
|
121
|
+
const wakeAt = until instanceof Date ? until.getTime() : ctx.now() + until;
|
|
122
|
+
const key = `sleep:${ctx.sleepCalls}`;
|
|
123
|
+
ctx.sleepCalls += 1;
|
|
124
|
+
// Replay: already slept?
|
|
125
|
+
const done = ctx.log.find((e) => e.type === "sleep_completed" && e.payload?.key === key);
|
|
126
|
+
if (done)
|
|
127
|
+
return;
|
|
128
|
+
ctx.seq += 1;
|
|
129
|
+
await ctx.store.appendEvent({
|
|
130
|
+
runId: ctx.runId, seq: ctx.seq, type: "sleep_created",
|
|
131
|
+
payload: { key, wakeAt }, createdAt: ctx.now(),
|
|
132
|
+
});
|
|
133
|
+
// Suspend: throw a control-flow signal caught by the runner.
|
|
134
|
+
throw new SuspendSignal(key, wakeAt);
|
|
135
|
+
}
|
|
136
|
+
export class SuspendSignal {
|
|
137
|
+
key;
|
|
138
|
+
wakeAt;
|
|
139
|
+
constructor(key, wakeAt) {
|
|
140
|
+
this.key = key;
|
|
141
|
+
this.wakeAt = wakeAt;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
/** Ordered output stream for a run. Chunks are persisted and replayable. */
|
|
145
|
+
export function getWritable() {
|
|
146
|
+
const ctx = current;
|
|
147
|
+
if (!ctx)
|
|
148
|
+
throw new Error("getWritable() outside a workflow");
|
|
149
|
+
return {
|
|
150
|
+
async write(chunk) {
|
|
151
|
+
const index = ctx.writes;
|
|
152
|
+
ctx.writes += 1;
|
|
153
|
+
const already = ctx.log.some((e) => e.type === "chunk" &&
|
|
154
|
+
e.payload?.index === index);
|
|
155
|
+
if (already)
|
|
156
|
+
return;
|
|
157
|
+
ctx.seq += 1;
|
|
158
|
+
await ctx.store.appendEvent({
|
|
159
|
+
runId: ctx.runId, seq: ctx.seq, type: "chunk",
|
|
160
|
+
payload: { value: chunk, index }, createdAt: ctx.now(),
|
|
161
|
+
});
|
|
162
|
+
},
|
|
163
|
+
async close() {
|
|
164
|
+
const already = ctx.log.some((e) => e.type === "chunk" && e.payload?.done === true);
|
|
165
|
+
if (already)
|
|
166
|
+
return;
|
|
167
|
+
ctx.seq += 1;
|
|
168
|
+
await ctx.store.appendEvent({
|
|
169
|
+
runId: ctx.runId, seq: ctx.seq, type: "chunk",
|
|
170
|
+
payload: { value: null, done: true, index: ctx.writes },
|
|
171
|
+
createdAt: ctx.now(),
|
|
172
|
+
});
|
|
173
|
+
},
|
|
174
|
+
};
|
|
175
|
+
}
|
|
176
|
+
/* ------------------------------------------------------------------ */
|
|
177
|
+
/* Runner */
|
|
178
|
+
/* ------------------------------------------------------------------ */
|
|
179
|
+
const DEFAULT_STEP_RETRIES = 3;
|
|
180
|
+
export class Engine {
|
|
181
|
+
store;
|
|
182
|
+
opts;
|
|
183
|
+
constructor(store, opts = {}) {
|
|
184
|
+
this.store = store;
|
|
185
|
+
this.opts = opts;
|
|
186
|
+
}
|
|
187
|
+
async start(workflowId, args) {
|
|
188
|
+
const runId = `lrun_${randomUUID().replace(/-/g, "").slice(0, 24)}`;
|
|
189
|
+
await this.store.createRun(runId, workflowId, args);
|
|
190
|
+
void this.run(runId, workflowId, args);
|
|
191
|
+
return { runId };
|
|
192
|
+
}
|
|
193
|
+
async getRun(runId) {
|
|
194
|
+
const row = await this.store.getRun(runId);
|
|
195
|
+
if (!row)
|
|
196
|
+
throw new Error(`run not found: ${runId}`);
|
|
197
|
+
return {
|
|
198
|
+
runId,
|
|
199
|
+
status: row.status,
|
|
200
|
+
returnValue: this.waitFor(runId),
|
|
201
|
+
getReadable: () => this.readable(runId),
|
|
202
|
+
/** Entry calls this to kill a duplicate stream (route.ts:172). */
|
|
203
|
+
cancel: async () => {
|
|
204
|
+
await this.store.cancel?.(runId);
|
|
205
|
+
await this.store.setStatus(runId, "failed", { error: "cancelled" });
|
|
206
|
+
},
|
|
207
|
+
};
|
|
208
|
+
}
|
|
209
|
+
async waitFor(runId) {
|
|
210
|
+
const deadline = Date.now() + 10 * 60_000;
|
|
211
|
+
while (Date.now() < deadline) {
|
|
212
|
+
const row = await this.store.getRun(runId);
|
|
213
|
+
if (row && row.status !== "running")
|
|
214
|
+
return row.output ?? null;
|
|
215
|
+
await new Promise((r) => setTimeout(r, 250));
|
|
216
|
+
}
|
|
217
|
+
throw new Error("returnValue timeout");
|
|
218
|
+
}
|
|
219
|
+
readable(runId) {
|
|
220
|
+
const store = this.store;
|
|
221
|
+
let sent = 0;
|
|
222
|
+
return new ReadableStream({
|
|
223
|
+
async pull(controller) {
|
|
224
|
+
const events = await store.getEvents(runId);
|
|
225
|
+
const chunks = events.filter((e) => e.type === "chunk");
|
|
226
|
+
for (const c of chunks.slice(sent)) {
|
|
227
|
+
const p = c.payload;
|
|
228
|
+
if (p.done) {
|
|
229
|
+
controller.close();
|
|
230
|
+
return;
|
|
231
|
+
}
|
|
232
|
+
controller.enqueue(new TextEncoder().encode(String(p.value)));
|
|
233
|
+
}
|
|
234
|
+
sent = chunks.length;
|
|
235
|
+
const row = await store.getRun(runId);
|
|
236
|
+
if (row && row.status !== "running" && sent >= chunks.length) {
|
|
237
|
+
controller.close();
|
|
238
|
+
}
|
|
239
|
+
},
|
|
240
|
+
});
|
|
241
|
+
}
|
|
242
|
+
/** Execute (or resume) a run. Safe to call repeatedly — replay is idempotent. */
|
|
243
|
+
async run(runId, workflowId, args) {
|
|
244
|
+
const fn = workflows.get(workflowId);
|
|
245
|
+
if (!fn)
|
|
246
|
+
throw new Error(`unknown workflow: ${workflowId}`);
|
|
247
|
+
const log = await this.store.getEvents(runId);
|
|
248
|
+
const ctx = {
|
|
249
|
+
runId, seq: log.length ? log.reduce((m, e) => Math.max(m, e.seq), 0) + 1 : 1,
|
|
250
|
+
stepCalls: 0, sleepCalls: 0, writes: 0, hookCalls: 0,
|
|
251
|
+
store: this.store, log, cursor: 0, chunks: [], now: () => Date.now(),
|
|
252
|
+
};
|
|
253
|
+
const prev = current;
|
|
254
|
+
current = ctx;
|
|
255
|
+
try {
|
|
256
|
+
if (await this.store.isCancelled?.(runId)) {
|
|
257
|
+
await this.store.setStatus(runId, "failed", { error: "cancelled" });
|
|
258
|
+
return;
|
|
259
|
+
}
|
|
260
|
+
const output = await fn(...args);
|
|
261
|
+
await this.store.setStatus(runId, "completed", output);
|
|
262
|
+
}
|
|
263
|
+
catch (err) {
|
|
264
|
+
if (err instanceof SuspendSignal)
|
|
265
|
+
return; // waiting on a timer
|
|
266
|
+
if (err instanceof FatalError) {
|
|
267
|
+
await this.store.setStatus(runId, "failed", { error: err.message });
|
|
268
|
+
return;
|
|
269
|
+
}
|
|
270
|
+
await this.store.setStatus(runId, "failed", {
|
|
271
|
+
error: err instanceof Error ? err.message : String(err),
|
|
272
|
+
});
|
|
273
|
+
}
|
|
274
|
+
finally {
|
|
275
|
+
current = prev;
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
/** Worker loop: resume runs whose timers are due. */
|
|
279
|
+
async startWorker(onError) {
|
|
280
|
+
for (;;) {
|
|
281
|
+
try {
|
|
282
|
+
const due = await this.store.dueTimers(Date.now());
|
|
283
|
+
for (const t of due) {
|
|
284
|
+
const ev = await this.store.getEvents(t.runId);
|
|
285
|
+
const created = ev.find((x) => x.seq === t.seq);
|
|
286
|
+
const key = created?.payload?.key
|
|
287
|
+
?? `sleep:${t.seq}`;
|
|
288
|
+
const alreadyDone = ev.some((x) => x.type === "sleep_completed" &&
|
|
289
|
+
x.payload?.key === key);
|
|
290
|
+
if (alreadyDone)
|
|
291
|
+
continue;
|
|
292
|
+
await this.store.appendEvent({
|
|
293
|
+
runId: t.runId, seq: ev.reduce((m, e) => Math.max(m, e.seq), 0) + 1,
|
|
294
|
+
type: "sleep_completed", payload: { key }, createdAt: Date.now(),
|
|
295
|
+
});
|
|
296
|
+
const row = await this.store.getRun(t.runId);
|
|
297
|
+
if (!row || row.status !== "running")
|
|
298
|
+
continue;
|
|
299
|
+
// resume: re-run with replay; sleep is memoized now
|
|
300
|
+
void this.resume(t.runId);
|
|
301
|
+
}
|
|
302
|
+
await this.reapStale(onError);
|
|
303
|
+
}
|
|
304
|
+
catch (e) {
|
|
305
|
+
onError?.(e);
|
|
306
|
+
}
|
|
307
|
+
await new Promise((r) => setTimeout(r, this.opts.pollMs ?? 500));
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
/**
|
|
311
|
+
* Built-in reaper: resume 'running' runs with no event activity in the
|
|
312
|
+
* last `staleMs`. Fixes the orphaned-run failure mode found in testing
|
|
313
|
+
* world-postgres (a run wedged in 'running' forever).
|
|
314
|
+
*/
|
|
315
|
+
async reapStale(onError) {
|
|
316
|
+
const staleMs = this.opts.staleRunMs ?? 60_000;
|
|
317
|
+
try {
|
|
318
|
+
const cutoff = Date.now() - staleMs;
|
|
319
|
+
const stale = await this.store
|
|
320
|
+
.staleRuns?.(cutoff);
|
|
321
|
+
if (!stale)
|
|
322
|
+
return;
|
|
323
|
+
for (const runId of stale) {
|
|
324
|
+
onError?.(new Error(`reaping stale run ${runId}`));
|
|
325
|
+
await this.resume(runId);
|
|
326
|
+
}
|
|
327
|
+
}
|
|
328
|
+
catch (e) {
|
|
329
|
+
onError?.(e);
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
async resume(runId) {
|
|
333
|
+
const log = await this.store.getEvents(runId);
|
|
334
|
+
const name = log[0]?.payload?.name;
|
|
335
|
+
const args = log[0]?.payload?.args ?? [];
|
|
336
|
+
if (!name)
|
|
337
|
+
return;
|
|
338
|
+
// Keep resuming through suspends until the run reaches a terminal state.
|
|
339
|
+
for (let i = 0; i < 500; i += 1) {
|
|
340
|
+
await this.run(runId, name, args);
|
|
341
|
+
const row = await this.store.getRun(runId);
|
|
342
|
+
if (!row || row.status !== "running")
|
|
343
|
+
return;
|
|
344
|
+
const evs = await this.store.getEvents(runId);
|
|
345
|
+
// Find a pending sleep: created but not yet completed.
|
|
346
|
+
const completed = new Set(evs.filter((e) => e.type === "sleep_completed")
|
|
347
|
+
.map((e) => e.payload.key));
|
|
348
|
+
const pending = evs
|
|
349
|
+
.filter((e) => e.type === "sleep_created")
|
|
350
|
+
.find((e) => !completed.has(e.payload.key));
|
|
351
|
+
if (!pending)
|
|
352
|
+
return; // nothing suspended -> genuinely stuck or done
|
|
353
|
+
const wake = pending.payload.wakeAt ?? 0;
|
|
354
|
+
const delay = wake - Date.now();
|
|
355
|
+
if (delay > 0)
|
|
356
|
+
await new Promise((r) => setTimeout(r, Math.min(delay, 2000)));
|
|
357
|
+
await this.store.appendEvent({
|
|
358
|
+
runId, seq: evs.reduce((m, e) => Math.max(m, e.seq), 0) + 1,
|
|
359
|
+
type: "sleep_completed",
|
|
360
|
+
payload: { key: pending.payload.key },
|
|
361
|
+
createdAt: Date.now(),
|
|
362
|
+
});
|
|
363
|
+
}
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
/**
|
|
367
|
+
* Cancel a run (Entry's route.ts calls getRun(id).cancel() to kill duplicate
|
|
368
|
+
* streams). Terminal runs ignore this.
|
|
369
|
+
*/
|
|
370
|
+
export class CancelledError extends Error {
|
|
371
|
+
constructor(runId) {
|
|
372
|
+
super(`run cancelled: ${runId}`);
|
|
373
|
+
this.name = "CancelledError";
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
/** Create a durable hook a workflow can await; resolve it from outside. */
|
|
377
|
+
export function defineHook() {
|
|
378
|
+
return {
|
|
379
|
+
async create() {
|
|
380
|
+
if (!current)
|
|
381
|
+
throw new Error("hooks only inside a workflow");
|
|
382
|
+
const token = `hook_${randomUUID().replace(/-/g, "").slice(0, 20)}`;
|
|
383
|
+
const key = `hook:${current.hookCalls}`;
|
|
384
|
+
current.hookCalls += 1;
|
|
385
|
+
await current.store.createHook?.(current.runId, token, key);
|
|
386
|
+
return { token };
|
|
387
|
+
},
|
|
388
|
+
};
|
|
389
|
+
}
|
|
390
|
+
/** Await a previously created hook until an external caller resolves it. */
|
|
391
|
+
export async function hookResult(token) {
|
|
392
|
+
if (!current)
|
|
393
|
+
throw new Error("hooks only inside a workflow");
|
|
394
|
+
const existing = await current.store.getHook?.(token);
|
|
395
|
+
if (existing && existing.payload?.resolved) {
|
|
396
|
+
return existing.payload.value;
|
|
397
|
+
}
|
|
398
|
+
throw new SuspendSignal(`hook:${token}`, 0);
|
|
399
|
+
}
|
|
400
|
+
/** Durable fetch: memoized per step position, like any other side effect. */
|
|
401
|
+
export async function workflowFetch(input, init) {
|
|
402
|
+
return step(async () => {
|
|
403
|
+
const res = await fetch(input, init);
|
|
404
|
+
return { __lfResponse: true, status: res.status, body: await res.text() };
|
|
405
|
+
});
|
|
406
|
+
}
|
|
407
|
+
export function hashId(s) {
|
|
408
|
+
return createHash("sha256").update(s).digest("hex").slice(0, 16);
|
|
409
|
+
}
|