@cosmicdrift/kumiko-bundled-features 0.230.0 → 0.232.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +9 -9
- package/src/compliance-profiles/__tests__/compliance-profiles.integration.test.ts +1 -1
- package/src/workflow-runner/__tests__/event-wakeup.integration.test.ts +328 -0
- package/src/workflow-runner/__tests__/pending-projection.integration.test.ts +292 -0
- package/src/workflow-runner/__tests__/resume-loop.integration.test.ts +277 -0
- package/src/workflow-runner/__tests__/workflow-runner.integration.test.ts +6 -11
- package/src/workflow-runner/db/queries/due-runs.ts +23 -0
- package/src/workflow-runner/event-subscriber.ts +93 -0
- package/src/workflow-runner/event-trigger.ts +14 -4
- package/src/workflow-runner/feature.ts +62 -5
- package/src/workflow-runner/handlers/resume-run.write.ts +277 -0
- package/src/workflow-runner/pending-projection.ts +148 -0
- package/src/workflow-runner/runner.ts +36 -13
- package/src/workflow-runner/tables.ts +95 -0
- package/src/workflow-runner/workflow-registry.ts +23 -0
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
// resume-run — wakes one suspended workflow step. Dispatched by the
|
|
2
|
+
// resume-due-runs job (framework#2513 Phase 2), never called directly by a
|
|
3
|
+
// user — r.systemScope() + access: { roles: [SYSTEM_ROLE] } enforce that.
|
|
4
|
+
//
|
|
5
|
+
// The job does nothing but SELECT + dispatch; all resume logic lives here,
|
|
6
|
+
// adapted from samples/recipes/workflow-engine/src/resume-loop.ts:
|
|
7
|
+
// 1. Q7 fingerprint check FIRST — before claiming, before any pipeline
|
|
8
|
+
// work. A changed workflow definition fails loud (WORKFLOW_RUN_FAILED,
|
|
9
|
+
// reason "workflow_definition_changed"), never a silent skip.
|
|
10
|
+
// 2. Claim via a savepoint-scoped WORKFLOW_RESUMED append (ctx.tryAppendEvent,
|
|
11
|
+
// not unsafeAppendEvent — a losing VersionConflict must not poison this
|
|
12
|
+
// handler's own transaction). A losing claim means another worker beat
|
|
13
|
+
// us to this row; silent no-op.
|
|
14
|
+
// 3. Re-run the pipeline via runStepList with resumeFrom: the suspended
|
|
15
|
+
// step's own index for a retry (re-enters the step), stepIndex + 1
|
|
16
|
+
// otherwise (wait already wrote its effect; resume past it).
|
|
17
|
+
//
|
|
18
|
+
// The run's ORIGINAL trigger event (the one that started the whole run) is
|
|
19
|
+
// always recovered from the run's own WORKFLOW_RUN_STARTED_TYPE event,
|
|
20
|
+
// never from the pending row — that event always carries it (see
|
|
21
|
+
// WorkflowRunStartedPayload in ./runner). The pending row's own
|
|
22
|
+
// triggerEventType/triggerPayload columns are a DIFFERENT thing: for a
|
|
23
|
+
// waitForEvent suspension they hold the AWAITED event the Phase 3b
|
|
24
|
+
// event-subscriber matched (NULL until then, and NULL forever on a
|
|
25
|
+
// timeout-without-a-match). That payload becomes the resumed pipeline's
|
|
26
|
+
// result for the skipped waitForEvent step itself (see resultKey on
|
|
27
|
+
// steps/wait-for-event.ts) — pre-seeded into stepsAcc below so a
|
|
28
|
+
// subsequent step's resolver can read `ctx.steps[awaits.someKey]`.
|
|
29
|
+
|
|
30
|
+
import { fetchOne } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
31
|
+
import {
|
|
32
|
+
buildPipelineSteps,
|
|
33
|
+
computeDefinitionFingerprint,
|
|
34
|
+
getStep,
|
|
35
|
+
type HandlerContext,
|
|
36
|
+
runStepList,
|
|
37
|
+
type StepInstance,
|
|
38
|
+
SYSTEM_ROLE,
|
|
39
|
+
WORKFLOW_AGGREGATE_TYPE,
|
|
40
|
+
WORKFLOW_RESUMED_TYPE,
|
|
41
|
+
WORKFLOW_RETRY_SCHEDULED_TYPE,
|
|
42
|
+
WORKFLOW_RUN_COMPLETED_TYPE,
|
|
43
|
+
WORKFLOW_RUN_FAILED_TYPE,
|
|
44
|
+
WORKFLOW_RUN_STARTED_TYPE,
|
|
45
|
+
WORKFLOW_WAITING_FOR_EVENT_TYPE,
|
|
46
|
+
type WorkflowDefinition,
|
|
47
|
+
type WriteEvent,
|
|
48
|
+
type WriteHandlerDef,
|
|
49
|
+
} from "@cosmicdrift/kumiko-framework/engine";
|
|
50
|
+
import { InternalError } from "@cosmicdrift/kumiko-framework/errors";
|
|
51
|
+
import { z } from "zod";
|
|
52
|
+
import {
|
|
53
|
+
isResumableSuspension,
|
|
54
|
+
type WorkflowRunCompletedPayload,
|
|
55
|
+
type WorkflowRunFailedPayload,
|
|
56
|
+
type WorkflowRunStartedPayload,
|
|
57
|
+
WorkflowSuspensionUnsupportedError,
|
|
58
|
+
} from "../runner";
|
|
59
|
+
import { workflowRunPendingTable } from "../tables";
|
|
60
|
+
import { getWorkflow } from "../workflow-registry";
|
|
61
|
+
|
|
62
|
+
const resumeRunSchema = z.object({
|
|
63
|
+
runId: z.string().min(1),
|
|
64
|
+
stepIndex: z.number().int().nonnegative(),
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
async function appendRunFailed(
|
|
68
|
+
ctx: HandlerContext,
|
|
69
|
+
runId: string,
|
|
70
|
+
failedPayload: WorkflowRunFailedPayload,
|
|
71
|
+
): Promise<void> {
|
|
72
|
+
await ctx.unsafeAppendEvent({
|
|
73
|
+
aggregateId: runId,
|
|
74
|
+
aggregateType: WORKFLOW_AGGREGATE_TYPE,
|
|
75
|
+
type: WORKFLOW_RUN_FAILED_TYPE,
|
|
76
|
+
payload: failedPayload,
|
|
77
|
+
});
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function checkQ7Fingerprint(
|
|
81
|
+
workflow: WorkflowDefinition,
|
|
82
|
+
workflowName: string,
|
|
83
|
+
runId: string,
|
|
84
|
+
stepIndex: number,
|
|
85
|
+
storedFingerprint: string | null,
|
|
86
|
+
): WorkflowRunFailedPayload | null {
|
|
87
|
+
const currentFingerprint = computeDefinitionFingerprint(workflow);
|
|
88
|
+
const fingerprintChanged = storedFingerprint !== null && currentFingerprint !== storedFingerprint;
|
|
89
|
+
if (!fingerprintChanged) {
|
|
90
|
+
return null;
|
|
91
|
+
}
|
|
92
|
+
return {
|
|
93
|
+
workflowName,
|
|
94
|
+
stepIndex,
|
|
95
|
+
error:
|
|
96
|
+
`Workflow "${workflowName}" definition changed since run ${runId} started ` +
|
|
97
|
+
`(expected ${storedFingerprint?.slice(0, 12)}…, current ${currentFingerprint?.slice(0, 12)}…).`,
|
|
98
|
+
reason: "workflow_definition_changed",
|
|
99
|
+
};
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
async function recoverTriggerEvent(
|
|
103
|
+
ctx: HandlerContext,
|
|
104
|
+
runId: string,
|
|
105
|
+
user: WriteEvent["user"],
|
|
106
|
+
): Promise<WriteEvent> {
|
|
107
|
+
const startedEvents = await ctx.loadAggregate(runId);
|
|
108
|
+
const started = startedEvents.find((e) => e.type === WORKFLOW_RUN_STARTED_TYPE);
|
|
109
|
+
if (!started) {
|
|
110
|
+
throw new InternalError({
|
|
111
|
+
message: `workflow-runner:write:resume-run: run ${runId} has no ${WORKFLOW_RUN_STARTED_TYPE} event — cannot recover its trigger event.`,
|
|
112
|
+
});
|
|
113
|
+
}
|
|
114
|
+
const startedPayload = started.payload as WorkflowRunStartedPayload; // @cast-boundary event-store-payload
|
|
115
|
+
const triggerEvent: WriteEvent = {
|
|
116
|
+
type: startedPayload.triggerEventType,
|
|
117
|
+
payload: startedPayload.triggerPayload,
|
|
118
|
+
user,
|
|
119
|
+
};
|
|
120
|
+
return triggerEvent;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
// The suspended waitForEvent step itself is skipped on resume
|
|
124
|
+
// (resumeFrom = stepIndex + 1) — its run() never re-executes, so its
|
|
125
|
+
// resultKey (args.event, see steps/wait-for-event.ts) never gets
|
|
126
|
+
// populated by the normal runStepList loop. Seed it here from the
|
|
127
|
+
// pending row's own triggerPayload so a subsequent step's resolver
|
|
128
|
+
// sees the matched event via `ctx.steps[awaits.someKey]`, same as any
|
|
129
|
+
// other step result. NULL on a timeout-without-a-match — a later
|
|
130
|
+
// resolver reading it just sees `undefined`, no special-casing needed.
|
|
131
|
+
function seedResumedStepResults(
|
|
132
|
+
pending: { suspensionEventType: string; triggerPayload: unknown | null },
|
|
133
|
+
steps: readonly StepInstance[],
|
|
134
|
+
stepIndex: number,
|
|
135
|
+
): Record<string, unknown> {
|
|
136
|
+
const stepsAcc: Record<string, unknown> = {};
|
|
137
|
+
if (pending.suspensionEventType === WORKFLOW_WAITING_FOR_EVENT_TYPE) {
|
|
138
|
+
const suspendedStep = steps[stepIndex];
|
|
139
|
+
const key = suspendedStep && getStep(suspendedStep.kind)?.resultKey?.(suspendedStep.args);
|
|
140
|
+
if (key !== undefined) {
|
|
141
|
+
stepsAcc[key] = pending.triggerPayload;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
return stepsAcc;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
export const resumeRunHandler: WriteHandlerDef = {
|
|
148
|
+
name: "resume-run",
|
|
149
|
+
schema: resumeRunSchema,
|
|
150
|
+
access: { roles: [SYSTEM_ROLE] },
|
|
151
|
+
handler: async (event, ctx) => {
|
|
152
|
+
const { runId, stepIndex } = event.payload as z.infer<typeof resumeRunSchema>;
|
|
153
|
+
const tenantId = event.user.tenantId;
|
|
154
|
+
|
|
155
|
+
if (!ctx.systemDb) {
|
|
156
|
+
throw new InternalError({
|
|
157
|
+
message:
|
|
158
|
+
"workflow-runner:write:resume-run requires ctx.systemDb — is r.systemScope() still set on the workflow-runner feature?",
|
|
159
|
+
});
|
|
160
|
+
}
|
|
161
|
+
const db = ctx.systemDb.assertTenantMatch(tenantId);
|
|
162
|
+
|
|
163
|
+
const pending = await fetchOne<{
|
|
164
|
+
workflowName: string;
|
|
165
|
+
suspensionEventType: string;
|
|
166
|
+
retryAttempt: number | null;
|
|
167
|
+
definitionFingerprint: string | null;
|
|
168
|
+
triggerPayload: unknown | null;
|
|
169
|
+
}>(db, workflowRunPendingTable, { runId, stepIndex, tenantId });
|
|
170
|
+
|
|
171
|
+
if (!pending) {
|
|
172
|
+
// skip: no pending row for (runId, stepIndex, tenantId) — another
|
|
173
|
+
// worker already resumed it (or the run reached a terminal state)
|
|
174
|
+
// between the job's SELECT and this dispatch; pending-projection.ts
|
|
175
|
+
// already deleted it.
|
|
176
|
+
return { isSuccess: true, data: { outcome: "already-resumed" as const } };
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
const { workflowName } = pending;
|
|
180
|
+
const workflow = getWorkflow(workflowName);
|
|
181
|
+
if (!workflow) {
|
|
182
|
+
const failedPayload: WorkflowRunFailedPayload = {
|
|
183
|
+
workflowName,
|
|
184
|
+
stepIndex,
|
|
185
|
+
error: `Workflow "${workflowName}" is not registered — cannot resume run ${runId}.`,
|
|
186
|
+
reason: "workflow_definition_changed",
|
|
187
|
+
};
|
|
188
|
+
await appendRunFailed(ctx, runId, failedPayload);
|
|
189
|
+
return { isSuccess: true, data: { outcome: "failed" as const } };
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
const fingerprintFailure = checkQ7Fingerprint(
|
|
193
|
+
workflow,
|
|
194
|
+
workflowName,
|
|
195
|
+
runId,
|
|
196
|
+
stepIndex,
|
|
197
|
+
pending.definitionFingerprint,
|
|
198
|
+
);
|
|
199
|
+
if (fingerprintFailure) {
|
|
200
|
+
await appendRunFailed(ctx, runId, fingerprintFailure);
|
|
201
|
+
return { isSuccess: true, data: { outcome: "failed" as const } };
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
const claim = await ctx.tryAppendEvent({
|
|
205
|
+
aggregateId: runId,
|
|
206
|
+
aggregateType: WORKFLOW_AGGREGATE_TYPE,
|
|
207
|
+
type: WORKFLOW_RESUMED_TYPE,
|
|
208
|
+
payload: {
|
|
209
|
+
stepIndex,
|
|
210
|
+
retryAttempt: pending.retryAttempt ?? undefined,
|
|
211
|
+
},
|
|
212
|
+
});
|
|
213
|
+
if (!claim.ok) {
|
|
214
|
+
// skip: lost the claim race against a concurrent resume-run dispatch
|
|
215
|
+
// for the same (runId, stepIndex) — the winner already re-runs the
|
|
216
|
+
// pipeline; nothing left for this call to do.
|
|
217
|
+
return { isSuccess: true, data: { outcome: "already-resumed" as const } };
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
const triggerEvent = await recoverTriggerEvent(ctx, runId, event.user);
|
|
221
|
+
|
|
222
|
+
const resumeFrom =
|
|
223
|
+
pending.suspensionEventType === WORKFLOW_RETRY_SCHEDULED_TYPE ? stepIndex : stepIndex + 1;
|
|
224
|
+
|
|
225
|
+
try {
|
|
226
|
+
const steps = buildPipelineSteps(workflow.pipelineDef, triggerEvent);
|
|
227
|
+
const workflowCtx = {
|
|
228
|
+
runId,
|
|
229
|
+
workflowName,
|
|
230
|
+
stepIndex,
|
|
231
|
+
definitionFingerprint: pending.definitionFingerprint ?? undefined,
|
|
232
|
+
...(pending.retryAttempt !== null && { retryAttempt: pending.retryAttempt + 1 }),
|
|
233
|
+
};
|
|
234
|
+
|
|
235
|
+
const stepsAcc = seedResumedStepResults(pending, steps, stepIndex);
|
|
236
|
+
|
|
237
|
+
const outcome = await runStepList(
|
|
238
|
+
steps,
|
|
239
|
+
triggerEvent,
|
|
240
|
+
ctx,
|
|
241
|
+
stepsAcc,
|
|
242
|
+
{},
|
|
243
|
+
workflowCtx,
|
|
244
|
+
resumeFrom,
|
|
245
|
+
);
|
|
246
|
+
|
|
247
|
+
if (outcome.kind === "suspended") {
|
|
248
|
+
if (!isResumableSuspension(steps, outcome.stepIndex)) {
|
|
249
|
+
throw new WorkflowSuspensionUnsupportedError(workflowName, outcome.stepIndex);
|
|
250
|
+
}
|
|
251
|
+
// Another suspension further down the pipeline — pending-projection.ts
|
|
252
|
+
// already materialised the new row; nothing more to do this pass.
|
|
253
|
+
return { isSuccess: true, data: { outcome: "suspended" as const } };
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
const completedPayload: WorkflowRunCompletedPayload = {
|
|
257
|
+
workflowName,
|
|
258
|
+
stepIndex: steps.length,
|
|
259
|
+
};
|
|
260
|
+
await ctx.unsafeAppendEvent({
|
|
261
|
+
aggregateId: runId,
|
|
262
|
+
aggregateType: WORKFLOW_AGGREGATE_TYPE,
|
|
263
|
+
type: WORKFLOW_RUN_COMPLETED_TYPE,
|
|
264
|
+
payload: completedPayload,
|
|
265
|
+
});
|
|
266
|
+
return { isSuccess: true, data: { outcome: "completed" as const } };
|
|
267
|
+
} catch (error) {
|
|
268
|
+
const failedPayload: WorkflowRunFailedPayload = {
|
|
269
|
+
workflowName,
|
|
270
|
+
stepIndex,
|
|
271
|
+
error: String(error),
|
|
272
|
+
};
|
|
273
|
+
await appendRunFailed(ctx, runId, failedPayload);
|
|
274
|
+
return { isSuccess: true, data: { outcome: "failed" as const } };
|
|
275
|
+
}
|
|
276
|
+
},
|
|
277
|
+
};
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
// pending-projection — MultiStreamProjection that keeps workflow_run_pending
|
|
2
|
+
// in sync with suspension/resume/terminal events on workflow-run streams.
|
|
3
|
+
//
|
|
4
|
+
// Insert (upsert, for idempotent at-least-once redelivery) on every
|
|
5
|
+
// suspension: wait, waitForEvent, retry. Delete on WORKFLOW_RESUMED or a
|
|
6
|
+
// terminal run outcome (completed/failed) — a run only ever has one row
|
|
7
|
+
// pending at a time (D1: a later suspension always targets a fresh
|
|
8
|
+
// stepIndex, since the prior one was resumed away first), so deleting by
|
|
9
|
+
// (tenantId, runId) alone is exact.
|
|
10
|
+
//
|
|
11
|
+
// No resume logic here — framework#2513 Phase 2 owns the job that reads
|
|
12
|
+
// this table and dispatches the actual resume. This projection only writes
|
|
13
|
+
// the rows.
|
|
14
|
+
|
|
15
|
+
import { deleteMany, upsertOnConflict } from "@cosmicdrift/kumiko-framework/bun-db";
|
|
16
|
+
import type { DbRunner } from "@cosmicdrift/kumiko-framework/db";
|
|
17
|
+
import type {
|
|
18
|
+
FeatureRegistrar,
|
|
19
|
+
MultiStreamProjectionDefinition,
|
|
20
|
+
} from "@cosmicdrift/kumiko-framework/engine";
|
|
21
|
+
import {
|
|
22
|
+
WORKFLOW_RESUMED_TYPE,
|
|
23
|
+
WORKFLOW_RETRY_SCHEDULED_TYPE,
|
|
24
|
+
WORKFLOW_RUN_COMPLETED_TYPE,
|
|
25
|
+
WORKFLOW_RUN_FAILED_TYPE,
|
|
26
|
+
WORKFLOW_WAITING_FOR_EVENT_TYPE,
|
|
27
|
+
WORKFLOW_WAITING_TYPE,
|
|
28
|
+
} from "@cosmicdrift/kumiko-framework/engine";
|
|
29
|
+
import { workflowRunPendingTable } from "./tables";
|
|
30
|
+
|
|
31
|
+
type WaitPayload = {
|
|
32
|
+
readonly wakeAt: string;
|
|
33
|
+
readonly stepIndex: number;
|
|
34
|
+
readonly workflowName: string;
|
|
35
|
+
readonly definitionFingerprint?: string;
|
|
36
|
+
};
|
|
37
|
+
|
|
38
|
+
type WaitForEventPayload = {
|
|
39
|
+
readonly eventType: string;
|
|
40
|
+
readonly match?: unknown;
|
|
41
|
+
readonly timeoutAt: string;
|
|
42
|
+
readonly stepIndex: number;
|
|
43
|
+
readonly workflowName: string;
|
|
44
|
+
readonly definitionFingerprint?: string;
|
|
45
|
+
};
|
|
46
|
+
|
|
47
|
+
type RetryScheduledPayload = {
|
|
48
|
+
readonly stepIndex: number;
|
|
49
|
+
readonly attempt: number;
|
|
50
|
+
readonly wakeAt: string;
|
|
51
|
+
readonly workflowName: string;
|
|
52
|
+
readonly definitionFingerprint?: string;
|
|
53
|
+
};
|
|
54
|
+
|
|
55
|
+
type PendingRow = {
|
|
56
|
+
readonly runId: string;
|
|
57
|
+
readonly tenantId: string;
|
|
58
|
+
readonly workflowName: string;
|
|
59
|
+
readonly stepIndex: number;
|
|
60
|
+
readonly suspensionEventType: string;
|
|
61
|
+
readonly wakeAt: string;
|
|
62
|
+
readonly retryAttempt: number | null;
|
|
63
|
+
readonly definitionFingerprint: string | null;
|
|
64
|
+
readonly waitEventType: string | null;
|
|
65
|
+
readonly matchExpr: unknown | null;
|
|
66
|
+
};
|
|
67
|
+
|
|
68
|
+
async function upsertPending(tx: DbRunner, row: PendingRow): Promise<void> {
|
|
69
|
+
await upsertOnConflict(
|
|
70
|
+
tx,
|
|
71
|
+
workflowRunPendingTable,
|
|
72
|
+
{
|
|
73
|
+
...row,
|
|
74
|
+
// Phase 3's event-subscriber writes these when the awaited event
|
|
75
|
+
// arrives — Phase 1 never touches them, insert or update.
|
|
76
|
+
triggerEventType: null,
|
|
77
|
+
triggerPayload: null,
|
|
78
|
+
},
|
|
79
|
+
{ conflictKeys: ["tenantId", "runId", "stepIndex"] },
|
|
80
|
+
);
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
async function deleteRunPending(tx: DbRunner, tenantId: string, runId: string): Promise<void> {
|
|
84
|
+
await deleteMany(tx, workflowRunPendingTable, { tenantId, runId });
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
export function registerWorkflowRunPendingProjection(r: FeatureRegistrar): void {
|
|
88
|
+
r.multiStreamProjection({
|
|
89
|
+
name: "workflow-run-pending",
|
|
90
|
+
table: workflowRunPendingTable,
|
|
91
|
+
apply: {
|
|
92
|
+
[WORKFLOW_WAITING_TYPE]: async (event, tx) => {
|
|
93
|
+
const p = event.payload as WaitPayload; // @cast-boundary engine-payload
|
|
94
|
+
await upsertPending(tx, {
|
|
95
|
+
runId: event.aggregateId,
|
|
96
|
+
tenantId: event.tenantId,
|
|
97
|
+
workflowName: p.workflowName,
|
|
98
|
+
stepIndex: p.stepIndex,
|
|
99
|
+
suspensionEventType: WORKFLOW_WAITING_TYPE,
|
|
100
|
+
wakeAt: p.wakeAt,
|
|
101
|
+
retryAttempt: null,
|
|
102
|
+
definitionFingerprint: p.definitionFingerprint ?? null,
|
|
103
|
+
waitEventType: null,
|
|
104
|
+
matchExpr: null,
|
|
105
|
+
});
|
|
106
|
+
},
|
|
107
|
+
[WORKFLOW_WAITING_FOR_EVENT_TYPE]: async (event, tx) => {
|
|
108
|
+
const p = event.payload as WaitForEventPayload; // @cast-boundary engine-payload
|
|
109
|
+
await upsertPending(tx, {
|
|
110
|
+
runId: event.aggregateId,
|
|
111
|
+
tenantId: event.tenantId,
|
|
112
|
+
workflowName: p.workflowName,
|
|
113
|
+
stepIndex: p.stepIndex,
|
|
114
|
+
suspensionEventType: WORKFLOW_WAITING_FOR_EVENT_TYPE,
|
|
115
|
+
wakeAt: p.timeoutAt,
|
|
116
|
+
retryAttempt: null,
|
|
117
|
+
definitionFingerprint: p.definitionFingerprint ?? null,
|
|
118
|
+
waitEventType: p.eventType,
|
|
119
|
+
matchExpr: p.match ?? null,
|
|
120
|
+
});
|
|
121
|
+
},
|
|
122
|
+
[WORKFLOW_RETRY_SCHEDULED_TYPE]: async (event, tx) => {
|
|
123
|
+
const p = event.payload as RetryScheduledPayload; // @cast-boundary engine-payload
|
|
124
|
+
await upsertPending(tx, {
|
|
125
|
+
runId: event.aggregateId,
|
|
126
|
+
tenantId: event.tenantId,
|
|
127
|
+
workflowName: p.workflowName,
|
|
128
|
+
stepIndex: p.stepIndex,
|
|
129
|
+
suspensionEventType: WORKFLOW_RETRY_SCHEDULED_TYPE,
|
|
130
|
+
wakeAt: p.wakeAt,
|
|
131
|
+
retryAttempt: p.attempt,
|
|
132
|
+
definitionFingerprint: p.definitionFingerprint ?? null,
|
|
133
|
+
waitEventType: null,
|
|
134
|
+
matchExpr: null,
|
|
135
|
+
});
|
|
136
|
+
},
|
|
137
|
+
[WORKFLOW_RESUMED_TYPE]: async (event, tx) => {
|
|
138
|
+
await deleteRunPending(tx, event.tenantId, event.aggregateId);
|
|
139
|
+
},
|
|
140
|
+
[WORKFLOW_RUN_COMPLETED_TYPE]: async (event, tx) => {
|
|
141
|
+
await deleteRunPending(tx, event.tenantId, event.aggregateId);
|
|
142
|
+
},
|
|
143
|
+
[WORKFLOW_RUN_FAILED_TYPE]: async (event, tx) => {
|
|
144
|
+
await deleteRunPending(tx, event.tenantId, event.aggregateId);
|
|
145
|
+
},
|
|
146
|
+
},
|
|
147
|
+
} satisfies MultiStreamProjectionDefinition);
|
|
148
|
+
}
|
|
@@ -1,15 +1,17 @@
|
|
|
1
1
|
// runner — starts a workflow-run by writing run-started, executing the
|
|
2
2
|
// pipeline until first suspension or completion, and writing run-completed.
|
|
3
3
|
//
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
7
|
-
//
|
|
8
|
-
//
|
|
9
|
-
//
|
|
4
|
+
// wait / retry / waitForEvent suspensions are all resumable — the
|
|
5
|
+
// resume-due-runs job + resume-run handler (framework#2513 Phase 2) wake
|
|
6
|
+
// wait/retry via a plain wakeAt timeout; waitForEvent additionally gets
|
|
7
|
+
// woken early by the event-subscriber (Phase 3b, event-subscriber.ts) when
|
|
8
|
+
// its awaited event arrives, and by the same wakeAt timeout otherwise
|
|
9
|
+
// (D1's `wakeAt ?? timeoutAt`). startAndRunWorkflow returns a silent
|
|
10
|
+
// "suspended" outcome for all three instead of throwing.
|
|
10
11
|
|
|
11
12
|
import type {
|
|
12
13
|
HandlerContext,
|
|
14
|
+
StepInstance,
|
|
13
15
|
WorkflowDefinition,
|
|
14
16
|
WriteEvent,
|
|
15
17
|
} from "@cosmicdrift/kumiko-framework/engine";
|
|
@@ -22,6 +24,17 @@ import {
|
|
|
22
24
|
WORKFLOW_RUN_STARTED_TYPE,
|
|
23
25
|
} from "@cosmicdrift/kumiko-framework/engine";
|
|
24
26
|
|
|
27
|
+
// Suspensions the resume loop knows how to wake — every Tier-3 step that
|
|
28
|
+
// can return SUSPEND_SENTINEL. Kept as an explicit allowlist (not "every
|
|
29
|
+
// registered Tier-3 step") so a future suspending step that forgets to
|
|
30
|
+
// wire up its own wake path fails loud via WorkflowSuspensionUnsupportedError
|
|
31
|
+
// instead of hanging a run forever.
|
|
32
|
+
const RESUMABLE_STEP_KINDS = new Set(["workflow.wait", "workflow.retry", "workflow.waitForEvent"]);
|
|
33
|
+
|
|
34
|
+
export function isResumableSuspension(steps: readonly StepInstance[], stepIndex: number): boolean {
|
|
35
|
+
return RESUMABLE_STEP_KINDS.has(steps[stepIndex]?.kind ?? "");
|
|
36
|
+
}
|
|
37
|
+
|
|
25
38
|
export type WorkflowRunStartedPayload = {
|
|
26
39
|
readonly workflowName: string;
|
|
27
40
|
readonly triggerEventType: string;
|
|
@@ -31,11 +44,11 @@ export type WorkflowRunStartedPayload = {
|
|
|
31
44
|
};
|
|
32
45
|
|
|
33
46
|
// Canonical run-completed shape — the sample this was hoisted from had two
|
|
34
|
-
// writers disagree (`{workflowName}` here, `{stepIndex}` in the
|
|
35
|
-
//
|
|
36
|
-
//
|
|
37
|
-
//
|
|
38
|
-
//
|
|
47
|
+
// writers disagree (`{workflowName}` here, `{stepIndex}` in the resume-loop
|
|
48
|
+
// sample). `stepIndex` is the pipeline's top-level step count (sub-lists
|
|
49
|
+
// inside branch/forEach aren't counted). A run reaches this event either in
|
|
50
|
+
// one pass, or across multiple passes when a wait/retry suspension resumed
|
|
51
|
+
// it (resume-run writes this same event on the pass that finally completes).
|
|
39
52
|
export type WorkflowRunCompletedPayload = {
|
|
40
53
|
readonly workflowName: string;
|
|
41
54
|
readonly stepIndex: number;
|
|
@@ -45,6 +58,10 @@ export type WorkflowRunFailedPayload = {
|
|
|
45
58
|
readonly workflowName: string;
|
|
46
59
|
readonly stepIndex: number;
|
|
47
60
|
readonly error: string;
|
|
61
|
+
// Machine-readable failure category — set by resume-run for a Q7
|
|
62
|
+
// fingerprint mismatch ("workflow_definition_changed"); absent for a
|
|
63
|
+
// plain pipeline-step failure (the `error` string is human-readable only).
|
|
64
|
+
readonly reason?: string;
|
|
48
65
|
};
|
|
49
66
|
|
|
50
67
|
export class WorkflowSuspensionUnsupportedError extends Error {
|
|
@@ -74,7 +91,7 @@ export async function startAndRunWorkflow(args: {
|
|
|
74
91
|
readonly triggerEvent: WriteEvent;
|
|
75
92
|
readonly idempotencyKey?: string;
|
|
76
93
|
readonly handlerCtx: HandlerContext;
|
|
77
|
-
}): Promise<{ readonly outcome: "completed" }> {
|
|
94
|
+
}): Promise<{ readonly outcome: "completed" | "suspended" }> {
|
|
78
95
|
const fingerprint = computeDefinitionFingerprint(args.workflow);
|
|
79
96
|
|
|
80
97
|
const startedPayload: WorkflowRunStartedPayload = {
|
|
@@ -109,7 +126,13 @@ export async function startAndRunWorkflow(args: {
|
|
|
109
126
|
);
|
|
110
127
|
|
|
111
128
|
if (outcome.kind === "suspended") {
|
|
112
|
-
|
|
129
|
+
if (!isResumableSuspension(steps, outcome.stepIndex)) {
|
|
130
|
+
throw new WorkflowSuspensionUnsupportedError(args.workflow.name, outcome.stepIndex);
|
|
131
|
+
}
|
|
132
|
+
// pending-projection.ts already materialised the row that
|
|
133
|
+
// resume-due-runs will pick up (or the event-subscriber will wake
|
|
134
|
+
// early, for waitForEvent) — nothing more to do on this pass.
|
|
135
|
+
return { outcome: "suspended" };
|
|
113
136
|
}
|
|
114
137
|
|
|
115
138
|
const completedPayload: WorkflowRunCompletedPayload = {
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
import {
|
|
2
|
+
defineUnmanagedTable,
|
|
3
|
+
type EntityTableMeta,
|
|
4
|
+
index,
|
|
5
|
+
instant,
|
|
6
|
+
integer,
|
|
7
|
+
jsonb,
|
|
8
|
+
table as pgTable,
|
|
9
|
+
primaryKey,
|
|
10
|
+
text,
|
|
11
|
+
uuid,
|
|
12
|
+
} from "@cosmicdrift/kumiko-framework/db";
|
|
13
|
+
|
|
14
|
+
// workflow_run_pending — one row per currently-suspended workflow step.
|
|
15
|
+
// Read-side of the resume loop (framework#2513 Phase 1): the wait /
|
|
16
|
+
// waitForEvent / retry steps already write their suspension onto the
|
|
17
|
+
// workflow-run event stream, but scanning kumiko_events for due runs
|
|
18
|
+
// doesn't scale (see samples/recipes/workflow-engine/postgres-resume-loop.ts,
|
|
19
|
+
// which does exactly that). This table is kept in sync by
|
|
20
|
+
// pending-projection.ts's MultiStreamProjection so Phase 2's resume job can
|
|
21
|
+
// do an indexed `wakeAt < now()` scan instead.
|
|
22
|
+
//
|
|
23
|
+
// **Unmanaged table** — same reasoning as delivery/tables.ts'
|
|
24
|
+
// deliveryAttemptsTable: rows are keyed by the workflow-run's own
|
|
25
|
+
// (runId, stepIndex), not by an ES-aggregate id of their own, and rows are
|
|
26
|
+
// deleted outright on resume/terminal rather than soft-state-transitioned —
|
|
27
|
+
// no audit trail needed for a purely operational pending-set.
|
|
28
|
+
//
|
|
29
|
+
// PK = (tenant_id, run_id, step_index): workflowRunAggregateId() is
|
|
30
|
+
// intentionally tenant-agnostic (uuidv5 over workflowName + idempotency
|
|
31
|
+
// key only, see aggregate-id.ts), so two tenants triggering the same
|
|
32
|
+
// workflow with the same idempotencyKey get the SAME run_id. Without
|
|
33
|
+
// tenant_id in the key, a same-stepIndex suspension from tenant B would
|
|
34
|
+
// upsert onto tenant A's row and silently steal it (framework#2513).
|
|
35
|
+
export const workflowRunPendingTable = pgTable(
|
|
36
|
+
"workflow_run_pending",
|
|
37
|
+
{
|
|
38
|
+
runId: uuid("run_id").notNull(),
|
|
39
|
+
tenantId: uuid("tenant_id").notNull(),
|
|
40
|
+
workflowName: text("workflow_name").notNull(),
|
|
41
|
+
stepIndex: integer("step_index").notNull(),
|
|
42
|
+
suspensionEventType: text("suspension_event_type").notNull(),
|
|
43
|
+
// Unified `wakeAt ?? timeoutAt` from the suspension payload — one
|
|
44
|
+
// column, one index, one predicate for the resume job regardless of
|
|
45
|
+
// which step suspended.
|
|
46
|
+
wakeAt: instant("wake_at").notNull(),
|
|
47
|
+
retryAttempt: integer("retry_attempt"),
|
|
48
|
+
// Q7 snapshot fingerprint — Phase 2 compares this against the current
|
|
49
|
+
// workflow definition's fingerprint before resuming.
|
|
50
|
+
definitionFingerprint: text("definition_fingerprint"),
|
|
51
|
+
// Set only for waitForEvent suspensions; NULL for wait/retry.
|
|
52
|
+
waitEventType: text("wait_event_type"),
|
|
53
|
+
matchExpr: jsonb("match_expr"),
|
|
54
|
+
// Written by the Phase 3 event-subscriber when the awaited event
|
|
55
|
+
// arrives — always NULL in Phase 1/2.
|
|
56
|
+
triggerEventType: text("trigger_event_type"),
|
|
57
|
+
triggerPayload: jsonb("trigger_payload"),
|
|
58
|
+
},
|
|
59
|
+
(t) => [
|
|
60
|
+
primaryKey({
|
|
61
|
+
columns: [t.tenantId, t.runId, t.stepIndex],
|
|
62
|
+
name: "workflow_run_pending_pkey",
|
|
63
|
+
}),
|
|
64
|
+
// Composite, not wake_at-alone: the resume job's due-scan
|
|
65
|
+
// (feature.ts) always filters `tenant_id = $1 AND wake_at < now()`.
|
|
66
|
+
index("workflow_run_pending_tenant_wake_at_idx").on(t.tenantId, t.wakeAt),
|
|
67
|
+
index("workflow_run_pending_wait_event_type_idx").on(t.waitEventType),
|
|
68
|
+
],
|
|
69
|
+
);
|
|
70
|
+
|
|
71
|
+
export const workflowRunPendingTableMeta: EntityTableMeta = defineUnmanagedTable({
|
|
72
|
+
tableName: "workflow_run_pending",
|
|
73
|
+
columns: [
|
|
74
|
+
{ name: "run_id", pgType: "uuid", notNull: true },
|
|
75
|
+
{ name: "tenant_id", pgType: "uuid", notNull: true },
|
|
76
|
+
{ name: "workflow_name", pgType: "text", notNull: true },
|
|
77
|
+
{ name: "step_index", pgType: "integer", notNull: true },
|
|
78
|
+
{ name: "suspension_event_type", pgType: "text", notNull: true },
|
|
79
|
+
{ name: "wake_at", pgType: "timestamptz", notNull: true },
|
|
80
|
+
{ name: "retry_attempt", pgType: "integer", notNull: false },
|
|
81
|
+
{ name: "definition_fingerprint", pgType: "text", notNull: false },
|
|
82
|
+
{ name: "wait_event_type", pgType: "text", notNull: false },
|
|
83
|
+
{ name: "match_expr", pgType: "jsonb", notNull: false },
|
|
84
|
+
{ name: "trigger_event_type", pgType: "text", notNull: false },
|
|
85
|
+
{ name: "trigger_payload", pgType: "jsonb", notNull: false },
|
|
86
|
+
],
|
|
87
|
+
indexes: [
|
|
88
|
+
{ name: "workflow_run_pending_tenant_wake_at_idx", columns: ["tenant_id", "wake_at"] },
|
|
89
|
+
{ name: "workflow_run_pending_wait_event_type_idx", columns: ["wait_event_type"] },
|
|
90
|
+
],
|
|
91
|
+
compositePrimaryKey: {
|
|
92
|
+
name: "workflow_run_pending_pkey",
|
|
93
|
+
columns: ["tenant_id", "run_id", "step_index"],
|
|
94
|
+
},
|
|
95
|
+
});
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
// workflow-registry — process-local lookup from workflow name to its
|
|
2
|
+
// WorkflowDefinition. registerEventTrigger populates it at feature-
|
|
3
|
+
// registration time; resume-run (Phase 2) is the only reader — it has
|
|
4
|
+
// nothing but a workflow name on the pending row and needs the live
|
|
5
|
+
// definition to rebuild the pipeline + recompute the Q7 fingerprint.
|
|
6
|
+
//
|
|
7
|
+
// Overwrite-on-same-name, no throw: unlike defineStep's registry (which
|
|
8
|
+
// throws on a duplicate kind — a real programming error), redefining a
|
|
9
|
+
// workflow under the same name is a legitimate deploy-time occurrence
|
|
10
|
+
// (the whole point of Q7 is detecting exactly that changed-definition case
|
|
11
|
+
// downstream, not preventing the registration).
|
|
12
|
+
|
|
13
|
+
import type { WorkflowDefinition } from "@cosmicdrift/kumiko-framework/engine";
|
|
14
|
+
|
|
15
|
+
const registry = new Map<string, WorkflowDefinition>();
|
|
16
|
+
|
|
17
|
+
export function registerWorkflow(workflow: WorkflowDefinition): void {
|
|
18
|
+
registry.set(workflow.name, workflow);
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export function getWorkflow(name: string): WorkflowDefinition | undefined {
|
|
22
|
+
return registry.get(name);
|
|
23
|
+
}
|