@pikku/core 0.12.130 → 0.12.134
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +101 -0
- package/dist/middleware-runner.d.ts +4 -0
- package/dist/middleware-runner.js +52 -16
- package/dist/services/in-memory-lease-service.d.ts +13 -0
- package/dist/services/in-memory-lease-service.js +49 -0
- package/dist/services/in-memory-workflow-service.d.ts +11 -3
- package/dist/services/in-memory-workflow-service.js +53 -23
- package/dist/services/index.d.ts +2 -0
- package/dist/services/index.js +2 -0
- package/dist/services/lease-service.d.ts +61 -0
- package/dist/services/lease-service.js +110 -0
- package/dist/services/workflow-service.d.ts +1 -1
- package/dist/testing/index.d.ts +1 -0
- package/dist/testing/service-tests/lease-service-tests.d.ts +3 -0
- package/dist/testing/service-tests/lease-service-tests.js +138 -0
- package/dist/testing/service-tests/workflow-fencing-tests.d.ts +18 -0
- package/dist/testing/service-tests/workflow-fencing-tests.js +216 -0
- package/dist/testing/service-tests.d.ts +5 -0
- package/dist/testing/service-tests.js +8 -0
- package/dist/types/core.types.d.ts +12 -8
- package/dist/utils/hmac.d.ts +14 -10
- package/dist/utils/hmac.js +29 -25
- package/dist/wirings/http/pikku-fetch-http-response.js +13 -0
- package/dist/wirings/secret/derive-oauth2-app-secrets.js +1 -0
- package/dist/wirings/secret/secret.types.d.ts +6 -0
- package/dist/wirings/trigger/index.d.ts +2 -2
- package/dist/wirings/trigger/index.js +1 -1
- package/dist/wirings/trigger/webhook-source-runner.js +52 -1
- package/dist/wirings/trigger/webhook-source.types.d.ts +50 -0
- package/dist/wirings/trigger/webhook-source.types.js +2 -0
- package/dist/wirings/workflow/graph/graph-runner.js +33 -14
- package/dist/wirings/workflow/index.d.ts +4 -3
- package/dist/wirings/workflow/index.js +3 -2
- package/dist/wirings/workflow/pikku-workflow-service.d.ts +57 -18
- package/dist/wirings/workflow/pikku-workflow-service.js +143 -140
- package/dist/wirings/workflow/workflow-constants.d.ts +28 -1
- package/dist/wirings/workflow/workflow-constants.js +28 -1
- package/dist/wirings/workflow/workflow-errors.d.ts +21 -0
- package/dist/wirings/workflow/workflow-errors.js +38 -0
- package/dist/wirings/workflow/workflow-queue-routing.d.ts +4 -2
- package/dist/wirings/workflow/workflow-queue-routing.js +9 -1
- package/dist/wirings/workflow/workflow-run-lease.d.ts +10 -0
- package/dist/wirings/workflow/workflow-run-lease.js +25 -0
- package/dist/wirings/workflow/workflow-run-status.d.ts +4 -0
- package/dist/wirings/workflow/workflow-run-status.js +37 -0
- package/dist/wirings/workflow/workflow-status-stream.d.ts +2 -1
- package/dist/wirings/workflow/workflow-status-stream.js +18 -1
- package/dist/wirings/workflow/workflow-step-claim.d.ts +13 -1
- package/dist/wirings/workflow/workflow-step-claim.js +29 -5
- package/dist/wirings/workflow/workflow-step-lease.d.ts +25 -0
- package/dist/wirings/workflow/workflow-step-lease.js +53 -0
- package/dist/wirings/workflow/workflow-step-retry.d.ts +14 -0
- package/dist/wirings/workflow/workflow-step-retry.js +33 -0
- package/dist/wirings/workflow/workflow-version-fallback.d.ts +11 -0
- package/dist/wirings/workflow/workflow-version-fallback.js +30 -0
- package/dist/wirings/workflow/workflow.types.d.ts +17 -0
- package/knowledge/decisions/internals/a-held-run-is-woken-later-not-retried.md +28 -0
- package/knowledge/decisions/internals/index.md +2 -1
- package/knowledge/decisions/internals/the-in-memory-workflow-service-is-inline-only-and-single-process.md +2 -2
- package/knowledge/decisions/internals/workflow-step-lock-is-held-only-to-claim-the-step.md +12 -5
- package/package.json +1 -1
- package/src/public-surface.json +18 -1
|
@@ -109,6 +109,44 @@ addError(WorkflowStepFunctionMismatchError, {
|
|
|
109
109
|
status: 409,
|
|
110
110
|
message: 'Workflow step was dispatched with a different function.',
|
|
111
111
|
});
|
|
112
|
+
/**
|
|
113
|
+
* Every dispatch that took this step lost its worker before finishing it, and
|
|
114
|
+
* the step has no attempts left to hand out. Failing here is the loud end of
|
|
115
|
+
* the loop: without it the step would be re-claimed and abandoned forever.
|
|
116
|
+
*/
|
|
117
|
+
export class WorkflowStepLeaseExpiredError extends PikkuError {
|
|
118
|
+
runId;
|
|
119
|
+
stepName;
|
|
120
|
+
attemptCount;
|
|
121
|
+
constructor(runId, stepName, attemptCount) {
|
|
122
|
+
super(`Workflow step '${stepName}' (run ${runId}) lost its worker on every one of its ${attemptCount} attempts: the last lease expired with the step still running`);
|
|
123
|
+
this.runId = runId;
|
|
124
|
+
this.stepName = stepName;
|
|
125
|
+
this.attemptCount = attemptCount;
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
addError(WorkflowStepLeaseExpiredError, {
|
|
129
|
+
status: 500,
|
|
130
|
+
message: 'Workflow step lost its worker and has no attempts left.',
|
|
131
|
+
});
|
|
132
|
+
/**
|
|
133
|
+
* A write from a dispatch that no longer owns its step: its lease lapsed and
|
|
134
|
+
* another dispatch claimed the step as a newer attempt. The newer attempt owns
|
|
135
|
+
* the outcome, so the stale one is dropped rather than recorded over it.
|
|
136
|
+
*/
|
|
137
|
+
export class WorkflowStepSupersededError extends PikkuError {
|
|
138
|
+
stepId;
|
|
139
|
+
attempt;
|
|
140
|
+
constructor(stepId, attempt) {
|
|
141
|
+
super(`Workflow step ${stepId}: attempt ${attempt} was superseded by a newer claim`);
|
|
142
|
+
this.stepId = stepId;
|
|
143
|
+
this.attempt = attempt;
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
addError(WorkflowStepSupersededError, {
|
|
147
|
+
status: 409,
|
|
148
|
+
message: 'Workflow step was claimed by a newer attempt.',
|
|
149
|
+
});
|
|
112
150
|
export class WorkflowStepNameNotString extends Error {
|
|
113
151
|
constructor(stepName) {
|
|
114
152
|
super(`Workflow step name must be a string. Received: ${typeof stepName}`);
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { JobGroup, JobOptions } from '../queue/queue.types.js';
|
|
1
|
+
import type { JobGroup, JobOptions, QueueService } from '../queue/queue.types.js';
|
|
2
2
|
import type { WorkflowServiceConfig, WorkflowStepOptions } from './workflow.types.js';
|
|
3
3
|
export type WorkflowQueueStrategy = 'per-workflow' | 'shared-groups';
|
|
4
4
|
export declare const resolveWorkflowConfig: () => WorkflowServiceConfig;
|
|
@@ -8,7 +8,7 @@ export declare const stepWorkerQueueName: (strategy: WorkflowQueueStrategy, rpcN
|
|
|
8
8
|
* How a step reaches its worker: on the queue, or here in the orchestrator.
|
|
9
9
|
*
|
|
10
10
|
* A step naming a workflow queues whenever a queue exists, even unmarked. Run
|
|
11
|
-
* here, it holds the parent's run
|
|
11
|
+
* here, it holds the parent's run lease until the
|
|
12
12
|
* child ends, and marks the child inline so the child's own `sleep` degrades
|
|
13
13
|
* from a suspension into a real in-process wait. Workflows cannot opt in
|
|
14
14
|
* through `workflowQueued`: that flag is read off `rpc` meta, and `addWorkflow`
|
|
@@ -23,4 +23,6 @@ export declare const stepWorkerQueueName: (strategy: WorkflowQueueStrategy, rpcN
|
|
|
23
23
|
*/
|
|
24
24
|
export declare const stepDispatchTarget: (rpcName: string, stepName: string, parentIsInline: () => Promise<boolean>) => Promise<'queue' | 'inline'>;
|
|
25
25
|
export declare const jobGroupFor: (strategy: WorkflowQueueStrategy, id?: string) => JobGroup | undefined;
|
|
26
|
+
/** The queue a remote run is driven through, or a clear account of its absence. */
|
|
27
|
+
export declare const requireQueueService: () => QueueService;
|
|
26
28
|
export declare const stepJobOptions: (stepOptions?: WorkflowStepOptions) => JobOptions;
|
|
@@ -29,7 +29,7 @@ export const stepWorkerQueueName = (strategy, rpcName) => dedicatedQueueName('wf
|
|
|
29
29
|
* How a step reaches its worker: on the queue, or here in the orchestrator.
|
|
30
30
|
*
|
|
31
31
|
* A step naming a workflow queues whenever a queue exists, even unmarked. Run
|
|
32
|
-
* here, it holds the parent's run
|
|
32
|
+
* here, it holds the parent's run lease until the
|
|
33
33
|
* child ends, and marks the child inline so the child's own `sleep` degrades
|
|
34
34
|
* from a suspension into a real in-process wait. Workflows cannot opt in
|
|
35
35
|
* through `workflowQueued`: that flag is read off `rpc` meta, and `addWorkflow`
|
|
@@ -61,6 +61,14 @@ export const stepDispatchTarget = async (rpcName, stepName, parentIsInline) => {
|
|
|
61
61
|
return (await parentIsInline()) ? 'inline' : 'queue';
|
|
62
62
|
};
|
|
63
63
|
export const jobGroupFor = (strategy, id) => id && strategy === 'shared-groups' ? { id, tier: id } : undefined;
|
|
64
|
+
/** The queue a remote run is driven through, or a clear account of its absence. */
|
|
65
|
+
export const requireQueueService = () => {
|
|
66
|
+
const queueService = getSingletonServices()?.queueService;
|
|
67
|
+
if (!queueService) {
|
|
68
|
+
throw new Error('QueueService not configured. Remote workflows require a queue service.');
|
|
69
|
+
}
|
|
70
|
+
return queueService;
|
|
71
|
+
};
|
|
64
72
|
export const stepJobOptions = (stepOptions) => {
|
|
65
73
|
const retries = stepOptions?.retries ?? DEFAULT_STEP_RETRIES;
|
|
66
74
|
const retryDelay = stepOptions?.retryDelay;
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import { type LeaseService } from '../../services/lease-service.js';
|
|
2
|
+
/** The run is, or was meanwhile, orchestrated by someone else. */
|
|
3
|
+
export declare const isRunLeaseError: (error: unknown) => boolean;
|
|
4
|
+
/**
|
|
5
|
+
* Run `fn` holding `key` on the lease service a workflow service was built
|
|
6
|
+
* with. It never waits: a key held elsewhere throws `LeaseTakenError`.
|
|
7
|
+
*/
|
|
8
|
+
export declare const holdWorkflowLease: <T>(service: object, leases: LeaseService | undefined, key: string, fn: () => Promise<T>) => Promise<T>;
|
|
9
|
+
/** A step whose lease is held elsewhere belongs to another dispatch. */
|
|
10
|
+
export declare const nullWhenStepHeld: <T>(claim: () => Promise<T>) => Promise<T | null>;
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { holdLease, LeaseLostError, LeaseTakenError, } from '../../services/lease-service.js';
|
|
2
|
+
/** The run is, or was meanwhile, orchestrated by someone else. */
|
|
3
|
+
export const isRunLeaseError = (error) => error instanceof LeaseTakenError || error instanceof LeaseLostError;
|
|
4
|
+
/**
|
|
5
|
+
* Run `fn` holding `key` on the lease service a workflow service was built
|
|
6
|
+
* with. It never waits: a key held elsewhere throws `LeaseTakenError`.
|
|
7
|
+
*/
|
|
8
|
+
export const holdWorkflowLease = (service, leases, key, fn) => {
|
|
9
|
+
if (!leases) {
|
|
10
|
+
throw new Error(`${service.constructor.name} was constructed without a leaseService, so it cannot exclude a second process from a run or step. Pass the app's leaseService to its constructor.`);
|
|
11
|
+
}
|
|
12
|
+
return holdLease(leases, key, () => fn());
|
|
13
|
+
};
|
|
14
|
+
/** A step whose lease is held elsewhere belongs to another dispatch. */
|
|
15
|
+
export const nullWhenStepHeld = async (claim) => {
|
|
16
|
+
try {
|
|
17
|
+
return await claim();
|
|
18
|
+
}
|
|
19
|
+
catch (error) {
|
|
20
|
+
if (error instanceof LeaseTakenError) {
|
|
21
|
+
return null;
|
|
22
|
+
}
|
|
23
|
+
throw error;
|
|
24
|
+
}
|
|
25
|
+
};
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
import type { HistoryEntry } from './run-timeline.js';
|
|
2
|
+
import type { WorkflowRun, WorkflowRunStatus } from './workflow.types.js';
|
|
3
|
+
/** A run's status with each step folded to its latest attempt. */
|
|
4
|
+
export declare function summarizeRunStatus(run: WorkflowRun, history: HistoryEntry[]): WorkflowRunStatus;
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/** A run's status with each step folded to its latest attempt. */
|
|
2
|
+
export function summarizeRunStatus(run, history) {
|
|
3
|
+
const terminalStatuses = new Set(['completed', 'failed', 'cancelled']);
|
|
4
|
+
const stepMap = new Map();
|
|
5
|
+
for (const step of history) {
|
|
6
|
+
const existing = stepMap.get(step.stepName);
|
|
7
|
+
if (!existing || step.updatedAt > existing.completedAt) {
|
|
8
|
+
stepMap.set(step.stepName, {
|
|
9
|
+
status: step.status,
|
|
10
|
+
startedAt: step.runningAt ?? step.createdAt,
|
|
11
|
+
completedAt: step.succeededAt ?? step.failedAt,
|
|
12
|
+
attempts: step.attemptCount,
|
|
13
|
+
});
|
|
14
|
+
}
|
|
15
|
+
}
|
|
16
|
+
const steps = [...stepMap.entries()].map(([name, s]) => ({
|
|
17
|
+
name,
|
|
18
|
+
status: s.status,
|
|
19
|
+
duration: s.startedAt && s.completedAt
|
|
20
|
+
? s.completedAt.getTime() - s.startedAt.getTime()
|
|
21
|
+
: undefined,
|
|
22
|
+
attempts: s.attempts,
|
|
23
|
+
}));
|
|
24
|
+
return {
|
|
25
|
+
id: run.id,
|
|
26
|
+
status: run.status,
|
|
27
|
+
startedAt: run.createdAt,
|
|
28
|
+
completedAt: terminalStatuses.has(run.status) ? run.updatedAt : undefined,
|
|
29
|
+
deterministic: run.deterministic,
|
|
30
|
+
plannedSteps: run.plannedSteps,
|
|
31
|
+
steps,
|
|
32
|
+
output: run.status === 'completed' ? run.output : undefined,
|
|
33
|
+
error: run.error
|
|
34
|
+
? { message: run.error.message ?? 'Unknown error' }
|
|
35
|
+
: undefined,
|
|
36
|
+
};
|
|
37
|
+
}
|
|
@@ -15,7 +15,8 @@ export interface WorkflowStatusStreamParams {
|
|
|
15
15
|
pollIntervalMs?: number;
|
|
16
16
|
}
|
|
17
17
|
/**
|
|
18
|
-
* Streams one run's progress until it reaches a terminal state.
|
|
18
|
+
* Streams one run's progress until it reaches a terminal state. A suspended
|
|
19
|
+
* run is not terminal: it gets a `suspended` frame and the stream stays open.
|
|
19
20
|
*
|
|
20
21
|
* Polled rather than subscribed because a run's steps are written by whichever
|
|
21
22
|
* worker picked them up, in whichever process — there is no in-memory event to
|
|
@@ -14,7 +14,8 @@ const TERMINAL = new Set([
|
|
|
14
14
|
]);
|
|
15
15
|
const DEFAULT_POLL_INTERVAL_MS = 500;
|
|
16
16
|
/**
|
|
17
|
-
* Streams one run's progress until it reaches a terminal state.
|
|
17
|
+
* Streams one run's progress until it reaches a terminal state. A suspended
|
|
18
|
+
* run is not terminal: it gets a `suspended` frame and the stream stays open.
|
|
18
19
|
*
|
|
19
20
|
* Polled rather than subscribed because a run's steps are written by whichever
|
|
20
21
|
* worker picked them up, in whichever process — there is no in-memory event to
|
|
@@ -27,6 +28,7 @@ const DEFAULT_POLL_INTERVAL_MS = 500;
|
|
|
27
28
|
export const streamWorkflowRunStatus = async ({ workflowRunService, runId, channel, session, detailed = false, pollIntervalMs = DEFAULT_POLL_INTERVAL_MS, }) => {
|
|
28
29
|
let lastHash = '';
|
|
29
30
|
let initSent = false;
|
|
31
|
+
let announcedSuspend;
|
|
30
32
|
const poll = async () => {
|
|
31
33
|
const run = await workflowRunService.getRun(runId);
|
|
32
34
|
if (!run) {
|
|
@@ -73,6 +75,21 @@ export const streamWorkflowRunStatus = async ({ workflowRunService, runId, chann
|
|
|
73
75
|
})),
|
|
74
76
|
});
|
|
75
77
|
}
|
|
78
|
+
// A suspended run is waiting for a decision or a signal, not finished, so
|
|
79
|
+
// the stream stays open for the resume. The frame is what tells a client
|
|
80
|
+
// that nothing more arrives until someone acts. `reason` is the string the
|
|
81
|
+
// workflow author gave `suspend()` for a person to read, so it goes to both
|
|
82
|
+
// routes, unlike `error`.
|
|
83
|
+
if (run.status === 'suspended') {
|
|
84
|
+
const reason = run.error?.message ?? 'Workflow suspended';
|
|
85
|
+
if (announcedSuspend !== reason) {
|
|
86
|
+
announcedSuspend = reason;
|
|
87
|
+
await channel.send({ type: 'suspended', reason });
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
else {
|
|
91
|
+
announcedSuspend = undefined;
|
|
92
|
+
}
|
|
76
93
|
if (TERMINAL.has(run.status)) {
|
|
77
94
|
await channel.send({ type: 'done' });
|
|
78
95
|
await channel.close();
|
|
@@ -3,8 +3,20 @@ import type { StepState } from './workflow.types.js';
|
|
|
3
3
|
export type StepClaimStore = {
|
|
4
4
|
getStepState(runId: string, stepName: string): Promise<StepState>;
|
|
5
5
|
setStepRunning(stepId: string): Promise<void>;
|
|
6
|
+
setStepError(stepId: string, error: Error, attempt?: number): Promise<void>;
|
|
7
|
+
refreshStepLease(stepId: string, leaseMs: number | null): Promise<boolean>;
|
|
6
8
|
createRetryAttempt(failedStepId: string, status: 'pending' | 'running'): Promise<StepState>;
|
|
9
|
+
/**
|
|
10
|
+
* Requeue the orchestrator so it sees the failure. Without it a step failed
|
|
11
|
+
* for lease exhaustion leaves the run `running` until a sweep notices.
|
|
12
|
+
*/
|
|
13
|
+
resumeWorkflow?(runId: string): Promise<void>;
|
|
7
14
|
};
|
|
15
|
+
/**
|
|
16
|
+
* Whether a step whose worker vanished may be handed to another one, or has
|
|
17
|
+
* run out of attempts to spend on that.
|
|
18
|
+
*/
|
|
19
|
+
export declare const leaseAttemptsExhausted: (stepState: StepState) => boolean;
|
|
8
20
|
/**
|
|
9
21
|
* Decide whether this dispatch owns the step, by reading its state and then
|
|
10
22
|
* writing it — which only excludes a concurrent dispatch when the caller holds
|
|
@@ -13,4 +25,4 @@ export type StepClaimStore = {
|
|
|
13
25
|
* A store that can express the whole decision as a single conditional write
|
|
14
26
|
* should do that instead of calling this.
|
|
15
27
|
*/
|
|
16
|
-
export declare const claimStepByReadThenWrite: (store: StepClaimStore, runId: string, stepName: string, rpcName: string) => Promise<StepState | null>;
|
|
28
|
+
export declare const claimStepByReadThenWrite: (store: StepClaimStore, runId: string, stepName: string, rpcName: string, leaseMs: number) => Promise<StepState | null>;
|
|
@@ -1,4 +1,10 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { DEFAULT_STEP_RETRIES, isStepLeaseLive } from './workflow-constants.js';
|
|
2
|
+
import { WorkflowStepFunctionMismatchError, WorkflowStepLeaseExpiredError, } from './workflow-errors.js';
|
|
3
|
+
/**
|
|
4
|
+
* Whether a step whose worker vanished may be handed to another one, or has
|
|
5
|
+
* run out of attempts to spend on that.
|
|
6
|
+
*/
|
|
7
|
+
export const leaseAttemptsExhausted = (stepState) => stepState.attemptCount >= (stepState.retries ?? DEFAULT_STEP_RETRIES) + 1;
|
|
2
8
|
/**
|
|
3
9
|
* Decide whether this dispatch owns the step, by reading its state and then
|
|
4
10
|
* writing it — which only excludes a concurrent dispatch when the caller holds
|
|
@@ -7,21 +13,39 @@ import { WorkflowStepFunctionMismatchError } from './workflow-errors.js';
|
|
|
7
13
|
* A store that can express the whole decision as a single conditional write
|
|
8
14
|
* should do that instead of calling this.
|
|
9
15
|
*/
|
|
10
|
-
export const claimStepByReadThenWrite = async (store, runId, stepName, rpcName) => {
|
|
16
|
+
export const claimStepByReadThenWrite = async (store, runId, stepName, rpcName, leaseMs) => {
|
|
11
17
|
const stepState = await store.getStepState(runId, stepName);
|
|
12
18
|
// knowledge: decisions/security/a-step-runs-the-function-the-workflow-dispatched-it-with.md
|
|
13
19
|
if (stepState.rpcName !== undefined &&
|
|
14
20
|
stepState.rpcName !== (rpcName ?? null)) {
|
|
15
21
|
throw new WorkflowStepFunctionMismatchError(runId, stepName);
|
|
16
22
|
}
|
|
17
|
-
if (stepState.status === 'succeeded'
|
|
23
|
+
if (stepState.status === 'succeeded') {
|
|
24
|
+
return null;
|
|
25
|
+
}
|
|
26
|
+
if (stepState.status === 'running') {
|
|
27
|
+
if (isStepLeaseLive(stepState.leaseExpiresAt)) {
|
|
28
|
+
return null;
|
|
29
|
+
}
|
|
30
|
+
if (leaseAttemptsExhausted(stepState)) {
|
|
31
|
+
await store.setStepError(stepState.stepId, new WorkflowStepLeaseExpiredError(runId, stepName, stepState.attemptCount), stepState.attemptCount);
|
|
32
|
+
await store.resumeWorkflow?.(runId);
|
|
33
|
+
return null;
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
// A failed step that has spent every attempt is settled, however it failed;
|
|
37
|
+
// a redelivered message must not buy it another one.
|
|
38
|
+
if (stepState.status === 'failed' && leaseAttemptsExhausted(stepState)) {
|
|
18
39
|
return null;
|
|
19
40
|
}
|
|
20
|
-
if (stepState.status === 'failed') {
|
|
21
|
-
|
|
41
|
+
if (stepState.status === 'failed' || stepState.status === 'running') {
|
|
42
|
+
const attempt = await store.createRetryAttempt(stepState.stepId, 'running');
|
|
43
|
+
await store.refreshStepLease(attempt.stepId, leaseMs);
|
|
44
|
+
return attempt;
|
|
22
45
|
}
|
|
23
46
|
if (stepState.status === 'pending' || stepState.status === 'scheduled') {
|
|
24
47
|
await store.setStepRunning(stepState.stepId);
|
|
48
|
+
await store.refreshStepLease(stepState.stepId, leaseMs);
|
|
25
49
|
}
|
|
26
50
|
return stepState;
|
|
27
51
|
};
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import type { StepState } from './workflow.types.js';
|
|
2
|
+
/**
|
|
3
|
+
* How long a claim on a step dispatched through `queueName` is good for, taken
|
|
4
|
+
* from that queue's own lock so the two never disagree about who owns the job.
|
|
5
|
+
*/
|
|
6
|
+
export declare const stepLeaseMsForQueue: (queueName: string) => number;
|
|
7
|
+
/**
|
|
8
|
+
* Keep a dispatch's claim alive for as long as it is working, and report how to
|
|
9
|
+
* stop once it is not. It renews on the same loop as `holdLease`.
|
|
10
|
+
*
|
|
11
|
+
* Stopping waits for a refresh already in flight, so a caller that releases the
|
|
12
|
+
* lease after stopping cannot have that release overwritten by a late renewal.
|
|
13
|
+
*
|
|
14
|
+
* A refresh that resolves `false` means another dispatch has claimed the step
|
|
15
|
+
* since: renewing stops there, and says so, because the step's outcome will
|
|
16
|
+
* now be refused rather than recorded.
|
|
17
|
+
*/
|
|
18
|
+
export declare const startStepLeaseRefresh: (stepId: string, leaseMs: number, refresh: () => Promise<boolean>) => (() => Promise<void>);
|
|
19
|
+
/**
|
|
20
|
+
* Who a `running` step belongs to, for a resumed run that meets it. `held` is a
|
|
21
|
+
* dispatch still working it. `lapsed` is one that died: the step is dispatched
|
|
22
|
+
* again but left `running`, so the claim counts it as another attempt rather
|
|
23
|
+
* than a first run — a step that kills its worker every time then runs out.
|
|
24
|
+
*/
|
|
25
|
+
export declare const runningStepLease: (stepState: Pick<StepState, 'status' | 'leaseExpiresAt'>) => 'held' | 'lapsed' | undefined;
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import { getSingletonServices, pikkuState } from '../../pikku-state.js';
|
|
2
|
+
import { DEFAULT_STEP_LEASE_MS, STEP_LEASE_REFRESH_MIN_MS, isStepLeaseLive, } from './workflow-constants.js';
|
|
3
|
+
import { keepLeaseAlive, leaseRenewalIntervalMs, RENEWALS_PER_LEASE, } from '../../services/lease-service.js';
|
|
4
|
+
/**
|
|
5
|
+
* How long a claim on a step dispatched through `queueName` is good for, taken
|
|
6
|
+
* from that queue's own lock so the two never disagree about who owns the job.
|
|
7
|
+
*/
|
|
8
|
+
export const stepLeaseMsForQueue = (queueName) => {
|
|
9
|
+
const worker = pikkuState(null, 'queue', 'registrations').get(queueName);
|
|
10
|
+
const { lockDuration, visibilityTimeout } = worker?.config ?? {};
|
|
11
|
+
if (lockDuration !== undefined) {
|
|
12
|
+
return lockDuration;
|
|
13
|
+
}
|
|
14
|
+
if (visibilityTimeout !== undefined) {
|
|
15
|
+
return visibilityTimeout * 1000;
|
|
16
|
+
}
|
|
17
|
+
return DEFAULT_STEP_LEASE_MS;
|
|
18
|
+
};
|
|
19
|
+
/**
|
|
20
|
+
* Keep a dispatch's claim alive for as long as it is working, and report how to
|
|
21
|
+
* stop once it is not. It renews on the same loop as `holdLease`.
|
|
22
|
+
*
|
|
23
|
+
* Stopping waits for a refresh already in flight, so a caller that releases the
|
|
24
|
+
* lease after stopping cannot have that release overwritten by a late renewal.
|
|
25
|
+
*
|
|
26
|
+
* A refresh that resolves `false` means another dispatch has claimed the step
|
|
27
|
+
* since: renewing stops there, and says so, because the step's outcome will
|
|
28
|
+
* now be refused rather than recorded.
|
|
29
|
+
*/
|
|
30
|
+
export const startStepLeaseRefresh = (stepId, leaseMs, refresh) => {
|
|
31
|
+
const interval = leaseRenewalIntervalMs(leaseMs);
|
|
32
|
+
if (interval < STEP_LEASE_REFRESH_MIN_MS) {
|
|
33
|
+
getSingletonServices()?.logger?.warn(`Workflow step ${stepId}: a ${leaseMs}ms lease is refreshed every ${interval}ms. Raise the queue's lockDuration or visibilityTimeout above ${STEP_LEASE_REFRESH_MIN_MS * RENEWALS_PER_LEASE}ms.`);
|
|
34
|
+
}
|
|
35
|
+
const renewal = keepLeaseAlive(`workflow-step:${stepId}`, leaseMs, () => refresh().catch((error) => {
|
|
36
|
+
getSingletonServices()?.logger?.warn(`Workflow step ${stepId}: could not refresh its lease; another worker may take the step`, error);
|
|
37
|
+
throw error;
|
|
38
|
+
}));
|
|
39
|
+
renewal.signal.addEventListener('abort', () => getSingletonServices()?.logger?.warn(`Workflow step ${stepId}: lost its lease, so another dispatch may claim it; this one stops renewing and its outcome will be refused if it does`));
|
|
40
|
+
return renewal.stop;
|
|
41
|
+
};
|
|
42
|
+
/**
|
|
43
|
+
* Who a `running` step belongs to, for a resumed run that meets it. `held` is a
|
|
44
|
+
* dispatch still working it. `lapsed` is one that died: the step is dispatched
|
|
45
|
+
* again but left `running`, so the claim counts it as another attempt rather
|
|
46
|
+
* than a first run — a step that kills its worker every time then runs out.
|
|
47
|
+
*/
|
|
48
|
+
export const runningStepLease = (stepState) => {
|
|
49
|
+
if (stepState.status !== 'running' || stepState.leaseExpiresAt == null) {
|
|
50
|
+
return undefined;
|
|
51
|
+
}
|
|
52
|
+
return isStepLeaseLive(stepState.leaseExpiresAt) ? 'held' : 'lapsed';
|
|
53
|
+
};
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import type { StepState, WorkflowStepOptions } from './workflow.types.js';
|
|
2
|
+
/** What running a step's retries in-process needs from the workflow service. */
|
|
3
|
+
export type StepRetryStore = {
|
|
4
|
+
setStepRunning(stepId: string): Promise<void>;
|
|
5
|
+
setStepResult(stepId: string, result: any): Promise<void>;
|
|
6
|
+
setStepError(stepId: string, error: Error): Promise<void>;
|
|
7
|
+
createRetryAttempt(failedStepId: string, status: 'pending' | 'running'): Promise<StepState>;
|
|
8
|
+
};
|
|
9
|
+
/**
|
|
10
|
+
* Spend a step's retry budget here and now, rather than by handing the run back
|
|
11
|
+
* to the orchestrator between attempts. For an inline run there is nothing to
|
|
12
|
+
* hand it back to.
|
|
13
|
+
*/
|
|
14
|
+
export declare const runInlineRetryLoop: (store: StepRetryStore, stepState: StepState, retries: number, retryDelay: WorkflowStepOptions['retryDelay'], doWork: (currentStepState: StepState) => Promise<any>, onError?: (error: any) => Promise<void>) => Promise<any>;
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import { getDurationInMilliseconds } from '../../time-utils.js';
|
|
2
|
+
/**
|
|
3
|
+
* Spend a step's retry budget here and now, rather than by handing the run back
|
|
4
|
+
* to the orchestrator between attempts. For an inline run there is nothing to
|
|
5
|
+
* hand it back to.
|
|
6
|
+
*/
|
|
7
|
+
export const runInlineRetryLoop = async (store, stepState, retries, retryDelay, doWork, onError) => {
|
|
8
|
+
let currentStepState = stepState;
|
|
9
|
+
while (true) {
|
|
10
|
+
let result;
|
|
11
|
+
try {
|
|
12
|
+
await store.setStepRunning(currentStepState.stepId);
|
|
13
|
+
result = await doWork(currentStepState);
|
|
14
|
+
}
|
|
15
|
+
catch (error) {
|
|
16
|
+
if (onError)
|
|
17
|
+
await onError(error);
|
|
18
|
+
await store.setStepError(currentStepState.stepId, error);
|
|
19
|
+
if (currentStepState.attemptCount >= retries) {
|
|
20
|
+
throw error;
|
|
21
|
+
}
|
|
22
|
+
currentStepState = await store.createRetryAttempt(currentStepState.stepId, 'pending');
|
|
23
|
+
if (retryDelay) {
|
|
24
|
+
await new Promise((resolve) => setTimeout(resolve, getDurationInMilliseconds(retryDelay)));
|
|
25
|
+
}
|
|
26
|
+
continue;
|
|
27
|
+
}
|
|
28
|
+
// Outside the try: the work is done, so failing to record it must not
|
|
29
|
+
// count as a failed attempt and run the work again.
|
|
30
|
+
await store.setStepResult(currentStepState.stepId, result);
|
|
31
|
+
return result;
|
|
32
|
+
}
|
|
33
|
+
};
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { PikkuRPC } from '../rpc/rpc-types.js';
|
|
2
|
+
import type { PikkuWorkflowService } from './pikku-workflow-service.js';
|
|
3
|
+
import type { WorkflowRun } from './workflow.types.js';
|
|
4
|
+
/**
|
|
5
|
+
* Resume a run whose workflow changed since it started. A graph workflow
|
|
6
|
+
* resumes from the version it started on; a complex one, whose inline steps
|
|
7
|
+
* cannot be replayed against a different definition, fails.
|
|
8
|
+
*/
|
|
9
|
+
export declare const runVersionMismatchFallback: (service: PikkuWorkflowService, run: WorkflowRun, currentMeta: {
|
|
10
|
+
source: string;
|
|
11
|
+
}, rpcService: PikkuRPC) => Promise<void>;
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import { runFromMeta } from './graph/graph-runner.js';
|
|
2
|
+
/**
|
|
3
|
+
* Resume a run whose workflow changed since it started. A graph workflow
|
|
4
|
+
* resumes from the version it started on; a complex one, whose inline steps
|
|
5
|
+
* cannot be replayed against a different definition, fails.
|
|
6
|
+
*/
|
|
7
|
+
export const runVersionMismatchFallback = async (service, run, currentMeta, rpcService) => {
|
|
8
|
+
// What can be resumed depends on the version the run started on, not on what
|
|
9
|
+
// the workflow has since become. The current source only stands in when that
|
|
10
|
+
// version was never stored.
|
|
11
|
+
const version = await service.getWorkflowVersion(run.workflow, run.graphHash);
|
|
12
|
+
const source = version?.source ?? currentMeta.source;
|
|
13
|
+
if (source === 'complex') {
|
|
14
|
+
await service.updateRunStatus(run.id, 'failed', undefined, {
|
|
15
|
+
message: `Workflow '${run.workflow}' definition changed. Complex workflows with inline steps cannot be migrated.`,
|
|
16
|
+
stack: '',
|
|
17
|
+
code: 'VERSION_CONFLICT',
|
|
18
|
+
});
|
|
19
|
+
return;
|
|
20
|
+
}
|
|
21
|
+
if (!version) {
|
|
22
|
+
await service.updateRunStatus(run.id, 'failed', undefined, {
|
|
23
|
+
message: `Workflow '${run.workflow}' version '${run.graphHash}' not found. Cannot resume with changed definition.`,
|
|
24
|
+
stack: '',
|
|
25
|
+
code: 'VERSION_NOT_FOUND',
|
|
26
|
+
});
|
|
27
|
+
return;
|
|
28
|
+
}
|
|
29
|
+
await runFromMeta(service, run.id, version.graph, rpcService);
|
|
30
|
+
};
|
|
@@ -2,6 +2,7 @@ import type { CommonWireMeta } from '../../types/core.types.js';
|
|
|
2
2
|
import type { SerializedError } from '../../errors/serialized-error.js';
|
|
3
3
|
import type { CorePikkuFunctionConfig } from '../../function/functions.types.js';
|
|
4
4
|
import type { GroupConcurrencyConfig } from '../queue/queue.types.js';
|
|
5
|
+
import type { LeaseService } from '../../services/lease-service.js';
|
|
5
6
|
export type { WorkflowService } from '../../services/workflow-service.js';
|
|
6
7
|
export type { WorkflowStepOptions, WorkflowWireDoRPC, WorkflowApprovalOptions, ApprovalOutcome, InputSource, OutputBinding, RpcStepMeta, Condition, BranchStepMeta, ParallelGroupStepMeta, FanoutStepMeta, ReturnStepMeta, InlineStepMeta, SleepStepMeta, CancelStepMeta, SuspendStepMeta, ApprovalStepMeta, SetStepMeta, SwitchCaseMeta, SwitchStepMeta, FilterStepMeta, ArrayPredicateStepMeta, WorkflowStepMeta, WorkflowStepWire, PikkuWorkflowWire, } from './dsl/workflow-dsl.types.js';
|
|
7
8
|
import type { WorkflowStepMeta } from './dsl/workflow-dsl.types.js';
|
|
@@ -24,6 +25,15 @@ export interface WorkflowQueueOptions {
|
|
|
24
25
|
queueConcurrency?: number;
|
|
25
26
|
queueGroupConcurrency?: number | GroupConcurrencyConfig;
|
|
26
27
|
}
|
|
28
|
+
/**
|
|
29
|
+
* What a workflow service over a shared store is built with. Every process that
|
|
30
|
+
* can reach the store may orchestrate the same run or claim the same step, so
|
|
31
|
+
* the service serialises both on `leaseService` — the same one the app
|
|
32
|
+
* registers, whatever store it is backed by.
|
|
33
|
+
*/
|
|
34
|
+
export interface WorkflowServiceOptions extends WorkflowQueueOptions {
|
|
35
|
+
leaseService: LeaseService;
|
|
36
|
+
}
|
|
27
37
|
export interface WorkflowPlannedStep {
|
|
28
38
|
stepName: string;
|
|
29
39
|
displayName?: string;
|
|
@@ -65,6 +75,13 @@ export interface StepState {
|
|
|
65
75
|
createdAt: Date;
|
|
66
76
|
updatedAt: Date;
|
|
67
77
|
childRunId?: string;
|
|
78
|
+
/**
|
|
79
|
+
* When the dispatch that claimed this step stops owning it. The holder pushes
|
|
80
|
+
* it forward while it is still working, so a lapsed lease means the worker is
|
|
81
|
+
* gone and the step may be claimed again. `undefined` is a store that keeps no
|
|
82
|
+
* lease, where a `running` step is owned until it moves on its own.
|
|
83
|
+
*/
|
|
84
|
+
leaseExpiresAt?: Date;
|
|
68
85
|
runningAt?: Date;
|
|
69
86
|
scheduledAt?: Date;
|
|
70
87
|
succeededAt?: Date;
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
---
|
|
2
|
+
type: decision
|
|
3
|
+
title: A held run is woken later, not retried
|
|
4
|
+
description: An orchestrator message that finds its run held enqueues a fresh wake-up a second later instead of failing, because the queue's retry budget is for real failures and a long pass outlasts it
|
|
5
|
+
tags: [workflows, leases, queues]
|
|
6
|
+
---
|
|
7
|
+
|
|
8
|
+
# A held run is woken later, not retried
|
|
9
|
+
|
|
10
|
+
Every message on the orchestrator queue carries news the run has to see — a step
|
|
11
|
+
finished, a child ended — so a message that finds the run held by another pass
|
|
12
|
+
cannot simply be dropped. Throwing it back to the queue looks equivalent and is
|
|
13
|
+
not: it spends the retries the queue keeps for genuine failures. pg-boss gives an
|
|
14
|
+
orchestrator message six attempts on an exponential backoff, the last about 55
|
|
15
|
+
seconds in; a pass that holds the run longer than that leaves the message dead
|
|
16
|
+
and the run `running` forever.
|
|
17
|
+
|
|
18
|
+
So a `LeaseTakenError` or `LeaseLostError` enqueues a new orchestrator message
|
|
19
|
+
with a one-second delay (`RUN_LEASE_RETRY_MS`) and the current message succeeds.
|
|
20
|
+
`verifiers/workflows/src/runners/run-lease.runner.ts` holds a run's lease for 90
|
|
21
|
+
seconds and checks the run still completes.
|
|
22
|
+
|
|
23
|
+
The same errors never fail an inline run: its body already wrote its outcome,
|
|
24
|
+
and losing the lease afterwards says only that the outcome was not produced
|
|
25
|
+
alone — the run keeps the status it wrote.
|
|
26
|
+
|
|
27
|
+
**What this rules out:** treating a taken run lease as a queue failure, and
|
|
28
|
+
marking a run failed because its lease was lost.
|
|
@@ -11,6 +11,7 @@ caller is entitled to assume.
|
|
|
11
11
|
|
|
12
12
|
<!-- pikku:knowledge-index -->
|
|
13
13
|
|
|
14
|
+
- [A held run is woken later, not retried](a-held-run-is-woken-later-not-retried.md) — An orchestrator message that finds its run held enqueues a fresh wake-up a second later instead of failing, because the queue's retry budget is for real failures and a long pass outlasts it
|
|
14
15
|
- [A non-streaming agent run registers with aiRunState on the same terms as a streaming one](a-non-streaming-agent-run-registers-with-airunstate-too.md) — Otherwise interruptAIAgent finds the run, passes the ownership check, then cannot stop it — and reports that as if the run were on another host
|
|
15
16
|
- [With scenarios.reset on, the dev seed is the fixture every assertion is written against](a-reset-suite-makes-the-dev-seed-the-fixture.md) — The rollback restores the database as the suite found it, so seed rows are the shared baseline — which makes an absolute date in the seed a test that expires
|
|
16
17
|
- [A resumed turn is as interruptible as the first one](a-resumed-agent-turn-is-as-interruptible-as-the-first.md) — It is the same person listening to the same voice, and after an approval it is where most of the reply actually gets spoken
|
|
@@ -112,7 +113,7 @@ caller is entitled to assume.
|
|
|
112
113
|
- [The dev queue copies prod timing and serialization semantics](the-dev-queue-copies-prod-timing-and-serialization-semantics.md) — InMemoryQueueService dispatches via setTimeout, retries with backoff, and JSON round-trips every payload so dev behaviour matches a real backend
|
|
113
114
|
- [One door per name — the ecosystem tier and the package root are both gone](the-ecosystem-entry-point-carries-the-adapter-surface.md) — the adapter surface was split onto @pikku/core/ecosystem to keep a stability promise at the root; publishing every module twice cost more than the promise was worth, so both the facades and the root barrel were deleted
|
|
114
115
|
- [The embedding model is pinned per service and doc/query embedding is split](the-embedding-model-is-pinned-per-service-and-doc-query-embedding-is-split.md) — AIEmbeddingService fixes its model at construction so index and query share a vector space, and separates embedDocuments from embedQuery for asymmetric models
|
|
115
|
-
- [The in-memory workflow service is inline-only and single-process](the-in-memory-workflow-service-is-inline-only-and-single-process.md) — InMemoryWorkflowService wires no queues and implements
|
|
116
|
+
- [The in-memory workflow service is inline-only and single-process](the-in-memory-workflow-service-is-inline-only-and-single-process.md) — InMemoryWorkflowService wires no queues and implements withRunLease/withStepLock as pass-throughs, because inline execution has no second holder to exclude
|
|
116
117
|
- [The KEK salt is scoped to the key version, not the secret](the-kek-salt-is-scoped-to-the-key-version.md) — One stored salt per key version means N secrets cost one derivation, which is the point of envelope encryption
|
|
117
118
|
- [The MCP handshake is challenged only when every target is gated](the-mcp-handshake-is-challenged-only-when-every-target-is-gated.md) — A client decides whether a server speaks OAuth from the handshake alone, so a fully-gated server that answers initialize with a 200 is detected as needing no sign-in
|
|
118
119
|
- [The middleware resolution cache is deliberately unbounded](the-middleware-resolution-cache-is-deliberately-unbounded.md) — Its keyspace is the set of registered wires, not request traffic, and middleware is dynamic — so eviction would buy nothing and cost the dedupe guarantee
|
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
---
|
|
2
2
|
type: decision
|
|
3
3
|
title: The in-memory workflow service is inline-only and single-process
|
|
4
|
-
description: InMemoryWorkflowService wires no queues and implements
|
|
4
|
+
description: InMemoryWorkflowService wires no queues and implements withRunLease/withStepLock as pass-throughs, because inline execution has no second holder to exclude
|
|
5
5
|
tags: services
|
|
6
6
|
---
|
|
7
7
|
|
|
8
8
|
# The in-memory workflow service is inline-only and single-process
|
|
9
9
|
|
|
10
10
|
`InMemoryWorkflowService` (`packages/core/src/services/in-memory-workflow-service.ts`)
|
|
11
|
-
calls `super({ ...options, wireQueues: false })` and implements `
|
|
11
|
+
calls `super({ ...options, wireQueues: false })` and implements `withRunLease` and
|
|
12
12
|
`withStepLock` as bare `return fn()`. Both look like unfinished work and neither
|
|
13
13
|
is.
|
|
14
14
|
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
type: decision
|
|
3
3
|
title: A workflow step lock is held only to claim the step, never across its execution
|
|
4
|
-
description: Holding the
|
|
4
|
+
description: Holding the step lock across step work exhausted the connection pool and self-deadlocked when it was an advisory lock; it stays claim-only on leases
|
|
5
5
|
tags: workflow
|
|
6
6
|
---
|
|
7
7
|
|
|
@@ -15,10 +15,17 @@ lock is then released, and the actual work plus result persistence run outside
|
|
|
15
15
|
it.
|
|
16
16
|
|
|
17
17
|
The guard is what makes that safe — once a step is `running`, any concurrent
|
|
18
|
-
worker returns early. The alternative was tried and failed: holding the
|
|
19
|
-
lock, and therefore its pooled connection, across
|
|
20
|
-
I/O plus further pool queries) let concurrent steps
|
|
21
|
-
and self-deadlock.
|
|
18
|
+
worker returns early. The alternative was tried and failed: holding the
|
|
19
|
+
then-advisory lock, and therefore its pooled connection, across
|
|
20
|
+
`executeGraphStep` (network I/O plus further pool queries) let concurrent steps
|
|
21
|
+
exhaust the connection pool and self-deadlock.
|
|
22
|
+
|
|
23
|
+
`withStepLock` is now a lease on the app's `leaseService`
|
|
24
|
+
(`workflow-step:<runId>:<step>`), so it pins no connection. It still stays
|
|
25
|
+
claim-only: a lease held across the work would have to outlive it, and a taken
|
|
26
|
+
step lease already means "another dispatch owns it", so the claim answers
|
|
27
|
+
`null`. Kysely and MongoDB claim with a status-guarded conditional update and do
|
|
28
|
+
not take it at all.
|
|
22
29
|
|
|
23
30
|
**What this rules out:** widening the `withStepLock` callback to cover RPC
|
|
24
31
|
invocation, child-workflow start, `setStepResult` or `resumeWorkflow` — the
|