pixivflow 3.2.0 → 3.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -1
- package/dist/commands/SchedulerCommand.js +31 -51
- package/dist/commands/scheduler-runtime.js +47 -7
- package/dist/config/defaults.d.ts +2 -0
- package/dist/config/defaults.js +6 -0
- package/dist/config/types.d.ts +20 -0
- package/dist/config/validation.js +11 -0
- package/dist/delivery/DeliveryDispatcher.d.ts +14 -0
- package/dist/delivery/DeliveryDispatcher.js +14 -0
- package/dist/delivery/EventCallbackDelivery.d.ts +54 -0
- package/dist/delivery/EventCallbackDelivery.js +89 -0
- package/dist/delivery/OutboxWorker.d.ts +11 -0
- package/dist/delivery/OutboxWorker.js +51 -1
- package/dist/package.json +1 -1
- package/dist/scheduler/CandidateSearchParams.d.ts +76 -0
- package/dist/scheduler/CandidateSearchParams.js +94 -0
- package/dist/scheduler/JobCancellation.d.ts +36 -0
- package/dist/scheduler/JobCancellation.js +73 -0
- package/dist/scheduler/JobEventStream.d.ts +94 -0
- package/dist/scheduler/JobEventStream.js +251 -0
- package/dist/scheduler/JobFacade.d.ts +298 -0
- package/dist/scheduler/JobFacade.js +622 -0
- package/dist/scheduler/JobProjection.d.ts +67 -0
- package/dist/scheduler/JobProjection.js +32 -0
- package/dist/scheduler/JobView.d.ts +34 -0
- package/dist/scheduler/JobView.js +82 -0
- package/dist/scheduler/ManualJobAdmission.d.ts +126 -0
- package/dist/scheduler/ManualJobAdmission.js +310 -0
- package/dist/scheduler/ManualJobService.d.ts +57 -0
- package/dist/scheduler/ManualJobService.js +91 -0
- package/dist/scheduler/ManualRefetchAdapter.d.ts +26 -0
- package/dist/scheduler/ManualRefetchAdapter.js +41 -0
- package/dist/scheduler/MultiScheduleManager.d.ts +17 -0
- package/dist/scheduler/MultiScheduleManager.js +68 -6
- package/dist/scheduler/ProtocolErrors.d.ts +68 -0
- package/dist/scheduler/ProtocolErrors.js +94 -0
- package/dist/scheduler/ScheduleTriggerServer.d.ts +56 -7
- package/dist/scheduler/ScheduleTriggerServer.js +166 -1
- package/dist/scheduler/SlotBusinessStatus.d.ts +5 -1
- package/dist/scheduler/SlotBusinessStatus.js +6 -0
- package/dist/scheduler/SlotCoordinator.d.ts +49 -5
- package/dist/scheduler/SlotCoordinator.js +89 -22
- package/dist/scheduler/StallSweep.d.ts +65 -0
- package/dist/scheduler/StallSweep.js +105 -0
- package/dist/scheduler/TargetOutcome.d.ts +39 -1
- package/dist/scheduler/TargetOutcome.js +51 -1
- package/dist/scheduler/ledger-time.d.ts +23 -0
- package/dist/scheduler/ledger-time.js +37 -0
- package/dist/storage/DatabaseMigration.js +48 -0
- package/dist/storage/repositories/DeliveryRepository.d.ts +6 -0
- package/dist/storage/repositories/DeliveryRepository.js +13 -0
- package/dist/storage/repositories/OutboxRepository.d.ts +73 -1
- package/dist/storage/repositories/OutboxRepository.js +142 -0
- package/dist/storage/repositories/SlotRepository.d.ts +34 -0
- package/dist/storage/repositories/SlotRepository.js +61 -2
- package/dist/version.js +1 -1
- package/dist/webui/package.json +1 -1
- package/package.json +1 -1
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* The generic job surface: `$defs/Task` in, `$defs/Job` out.
|
|
4
|
+
*
|
|
5
|
+
* This is an ADAPTER, not a second execution system. Everything it does is a
|
|
6
|
+
* translation on top of the one durable ledger:
|
|
7
|
+
* - admission -> `ManualJobAdmission` (shared with the legacy refetch path)
|
|
8
|
+
* - read model -> `JobView` (`buildJobProjection` + `buildProtocolJob`)
|
|
9
|
+
* - cancellation -> `cancelConsumerJob` (one transaction)
|
|
10
|
+
* - events/ack -> `JobEventStream` (projection of `delivery_events`)
|
|
11
|
+
*
|
|
12
|
+
* Identity is the consumer's idempotency key, which IS the durable
|
|
13
|
+
* `manual_request_id`; `job_id` IS the slot id. A replay from either entry
|
|
14
|
+
* point therefore converges on the same job, and no second slot or delivery can
|
|
15
|
+
* be created for the same key.
|
|
16
|
+
*/
|
|
17
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
18
|
+
exports.ManualJobService = void 0;
|
|
19
|
+
const JobCancellation_1 = require("./JobCancellation");
|
|
20
|
+
const JobEventStream_1 = require("./JobEventStream");
|
|
21
|
+
const JobFacade_1 = require("./JobFacade");
|
|
22
|
+
const JobView_1 = require("./JobView");
|
|
23
|
+
class ManualJobService {
|
|
24
|
+
deps;
|
|
25
|
+
events;
|
|
26
|
+
constructor(deps) {
|
|
27
|
+
this.deps = deps;
|
|
28
|
+
this.events = new JobEventStream_1.JobEventStream(this.viewDeps());
|
|
29
|
+
}
|
|
30
|
+
now() {
|
|
31
|
+
return this.deps.now ? this.deps.now() : Date.now();
|
|
32
|
+
}
|
|
33
|
+
viewDeps() {
|
|
34
|
+
return {
|
|
35
|
+
database: this.deps.database,
|
|
36
|
+
config: this.deps.config,
|
|
37
|
+
now: () => this.now(),
|
|
38
|
+
};
|
|
39
|
+
}
|
|
40
|
+
capabilities() {
|
|
41
|
+
return (0, JobFacade_1.buildCapabilities)(this.deps.config(), this.now());
|
|
42
|
+
}
|
|
43
|
+
/** Create or replay the one work item this idempotency key owns. */
|
|
44
|
+
submitJob(body) {
|
|
45
|
+
const task = (0, JobFacade_1.parseTaskBody)(body);
|
|
46
|
+
const result = this.deps.admission.admit({
|
|
47
|
+
idempotencyKey: task.idempotencyKey,
|
|
48
|
+
...(task.correlationId !== undefined ? { correlationId: task.correlationId } : {}),
|
|
49
|
+
...(task.account !== undefined ? { account: task.account } : {}),
|
|
50
|
+
params: task.params,
|
|
51
|
+
});
|
|
52
|
+
// The admission event (and the declared callback_url) are recorded against
|
|
53
|
+
// the durable slot inside the one shared admission path, so a crash right
|
|
54
|
+
// after admission is repaired by the same reconcile that serves the stream.
|
|
55
|
+
// A replay never rewrites the endpoint the job already declared.
|
|
56
|
+
this.events.reconcile(result.slotId, task.callbackUrl ?? null);
|
|
57
|
+
return { job: this.viewBySlotId(result.slotId), replayed: result.reused };
|
|
58
|
+
}
|
|
59
|
+
jobStatus(jobId) {
|
|
60
|
+
return (0, JobView_1.protocolJobBySlotId)(this.viewDeps(), jobId);
|
|
61
|
+
}
|
|
62
|
+
jobsByIdempotencyKey(idempotencyKey) {
|
|
63
|
+
const slot = this.deps.database.slots.findManualSlotByKey(idempotencyKey);
|
|
64
|
+
return slot ? [(0, JobView_1.protocolJobForSlot)(this.viewDeps(), slot)] : [];
|
|
65
|
+
}
|
|
66
|
+
/** `GET /jobs/:jobId/events` — the durable stream, reconciled from the ledger. */
|
|
67
|
+
jobEvents(jobId, query) {
|
|
68
|
+
return this.events.page(jobId, query);
|
|
69
|
+
}
|
|
70
|
+
/** `POST /jobs/:jobId/events/ack` — O(1) durable cursor write, never a job mutation. */
|
|
71
|
+
ackJobEvents(jobId, body) {
|
|
72
|
+
return this.events.ack(jobId, body);
|
|
73
|
+
}
|
|
74
|
+
/**
|
|
75
|
+
* Idempotent: cancelling an already-terminal job reports its current
|
|
76
|
+
* projection instead of failing. `cancelConsumerJob` owns the one transaction
|
|
77
|
+
* that terminalises the work and stops further deliveries.
|
|
78
|
+
*/
|
|
79
|
+
cancelJob(jobId) {
|
|
80
|
+
(0, JobCancellation_1.cancelConsumerJob)(this.deps.database, jobId, this.now());
|
|
81
|
+
// The cancellation is now durable, so its terminal event must be too — a
|
|
82
|
+
// consumer that polls after cancelling must never see an unterminated stream.
|
|
83
|
+
this.events.reconcile(jobId);
|
|
84
|
+
return this.viewBySlotId(jobId);
|
|
85
|
+
}
|
|
86
|
+
viewBySlotId(slotId) {
|
|
87
|
+
return (0, JobView_1.requireProtocolJobForSlotId)(this.viewDeps(), slotId);
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
exports.ManualJobService = ManualJobService;
|
|
91
|
+
//# sourceMappingURL=ManualJobService.js.map
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Legacy adapters for the pre-protocol `refetch` surface.
|
|
3
|
+
*
|
|
4
|
+
* Workflow Protocol v1 §3.1: the historic endpoint is kept as a *shim* over the
|
|
5
|
+
* one shared admission, so both entry points create and resolve the SAME work
|
|
6
|
+
* item and share one identity space (`requestId` == `idempotency_key`). Nothing
|
|
7
|
+
* here may add behaviour of its own — it only translates shapes and, for the
|
|
8
|
+
* submit route, restores the historic message strings that route maps onto its
|
|
9
|
+
* HTTP statuses (`unknown target` -> 404, `ambiguous target` -> 409).
|
|
10
|
+
*/
|
|
11
|
+
import { Database } from '../storage/Database';
|
|
12
|
+
import { JobStatusProjection } from './JobProjection';
|
|
13
|
+
import { ManualJobAdmission } from './ManualJobAdmission';
|
|
14
|
+
export interface LegacyRefetchSubmitResult {
|
|
15
|
+
slotId: string;
|
|
16
|
+
disposition: string;
|
|
17
|
+
}
|
|
18
|
+
/** `POST /internal/targets/:targetId/refetch` as a thin adapter. */
|
|
19
|
+
export declare function legacyRefetchSubmit(admission: ManualJobAdmission): (targetId: string, requestId: string, correlationId?: string) => Promise<LegacyRefetchSubmitResult>;
|
|
20
|
+
/**
|
|
21
|
+
* `GET /internal/targets/:targetId/refetch/:requestId`. The historic alias
|
|
22
|
+
* fields (`requestId`, `slotId`, `state`, `slotStatus`) are unchanged; the rest
|
|
23
|
+
* of the projection is additive liveness/cause detail.
|
|
24
|
+
*/
|
|
25
|
+
export declare function legacyRefetchStatus(database: Database): (targetId: string, requestId: string) => JobStatusProjection | null;
|
|
26
|
+
//# sourceMappingURL=ManualRefetchAdapter.d.ts.map
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.legacyRefetchSubmit = legacyRefetchSubmit;
|
|
4
|
+
exports.legacyRefetchStatus = legacyRefetchStatus;
|
|
5
|
+
const JobProjection_1 = require("./JobProjection");
|
|
6
|
+
const ProtocolErrors_1 = require("./ProtocolErrors");
|
|
7
|
+
/** `POST /internal/targets/:targetId/refetch` as a thin adapter. */
|
|
8
|
+
function legacyRefetchSubmit(admission) {
|
|
9
|
+
return async (targetId, requestId, correlationId) => {
|
|
10
|
+
try {
|
|
11
|
+
const result = admission.admit({
|
|
12
|
+
targetId,
|
|
13
|
+
idempotencyKey: requestId,
|
|
14
|
+
...(correlationId ? { correlationId } : {}),
|
|
15
|
+
});
|
|
16
|
+
return { slotId: result.slotId, disposition: result.disposition };
|
|
17
|
+
}
|
|
18
|
+
catch (error) {
|
|
19
|
+
// The legacy route classifies by MESSAGE, not by code. Re-throwing the
|
|
20
|
+
// message preserves the deployed contract; the codes stay on the new
|
|
21
|
+
// surface where callers can rely on them.
|
|
22
|
+
if (error instanceof ProtocolErrors_1.ProtocolRequestError) {
|
|
23
|
+
throw new Error(error.body.message ?? error.body.code);
|
|
24
|
+
}
|
|
25
|
+
throw error;
|
|
26
|
+
}
|
|
27
|
+
};
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* `GET /internal/targets/:targetId/refetch/:requestId`. The historic alias
|
|
31
|
+
* fields (`requestId`, `slotId`, `state`, `slotStatus`) are unchanged; the rest
|
|
32
|
+
* of the projection is additive liveness/cause detail.
|
|
33
|
+
*/
|
|
34
|
+
function legacyRefetchStatus(database) {
|
|
35
|
+
return (targetId, requestId) => {
|
|
36
|
+
const slot = database.slots.findManualSlot(requestId, targetId);
|
|
37
|
+
const cell = slot && database.slots.getCell(slot.id, targetId);
|
|
38
|
+
return slot && cell ? (0, JobProjection_1.buildJobProjection)(requestId, slot, cell) : null;
|
|
39
|
+
};
|
|
40
|
+
}
|
|
41
|
+
//# sourceMappingURL=ManualRefetchAdapter.js.map
|
|
@@ -58,6 +58,23 @@ export declare class MultiScheduleManager {
|
|
|
58
58
|
* schedule/target name). All Pixiv-consuming work shares the account id. */
|
|
59
59
|
private resourceKeyForPlan;
|
|
60
60
|
start(initialConfig?: StandaloneConfig): ConfigReloadResult;
|
|
61
|
+
/**
|
|
62
|
+
* Liveness sweep (§liveness): terminalise slots that were admitted but never
|
|
63
|
+
* claimed, or that lost their worker, once they are older than the configured
|
|
64
|
+
* budget.
|
|
65
|
+
*
|
|
66
|
+
* Ordering is load-bearing. Recovery runs FIRST and reports the slots it
|
|
67
|
+
* re-dispatched; those are excluded here. A re-dispatched slot is owned by an
|
|
68
|
+
* asynchronous run that has not claimed its lease yet, so without the
|
|
69
|
+
* exclusion this tick would see "expired lease + stale heartbeat" and fail the
|
|
70
|
+
* very occurrence recovery just saved — and crash-resume (§manual-resume)
|
|
71
|
+
* would silently become "fail everything that was interrupted long enough".
|
|
72
|
+
* The sweep therefore answers the other half of the question: which slots is
|
|
73
|
+
* recovery UNABLE to rescue (no enabled plan, a stopped/failed Scheduler, or a
|
|
74
|
+
* queue that never drains)? Those are the ones that would otherwise stay
|
|
75
|
+
* non-terminal forever.
|
|
76
|
+
*/
|
|
77
|
+
private runStallSweep;
|
|
61
78
|
private startRecoveryLoop;
|
|
62
79
|
/**
|
|
63
80
|
* Re-dispatch occurrences that the durable ledger says are unfinished but that
|
|
@@ -12,6 +12,7 @@ const Scheduler_1 = require("./Scheduler");
|
|
|
12
12
|
const ResourceAdmission_1 = require("./ResourceAdmission");
|
|
13
13
|
const schedules_1 = require("./schedules");
|
|
14
14
|
const OccurrenceResolver_1 = require("./OccurrenceResolver");
|
|
15
|
+
const StallSweep_1 = require("./StallSweep");
|
|
15
16
|
/**
|
|
16
17
|
* How often the manager looks for unfinished occurrences that no live worker
|
|
17
18
|
* owns. Combined with the slot lease TTL (SlotCoordinator), the worst case for a
|
|
@@ -82,11 +83,47 @@ class MultiScheduleManager {
|
|
|
82
83
|
this.updateWatcher(config);
|
|
83
84
|
// Recovery first: it acts on occurrences the ledger already knows about.
|
|
84
85
|
// Catch-up is inference ("cron suggests a fire was missed") and runs after.
|
|
85
|
-
this.recoverInterruptedSlots();
|
|
86
|
+
const rescued = this.recoverInterruptedSlots();
|
|
87
|
+
this.runStallSweep(rescued);
|
|
86
88
|
this.startRecoveryLoop();
|
|
87
89
|
this.catchUpMissedRuns(config);
|
|
88
90
|
return result;
|
|
89
91
|
}
|
|
92
|
+
/**
|
|
93
|
+
* Liveness sweep (§liveness): terminalise slots that were admitted but never
|
|
94
|
+
* claimed, or that lost their worker, once they are older than the configured
|
|
95
|
+
* budget.
|
|
96
|
+
*
|
|
97
|
+
* Ordering is load-bearing. Recovery runs FIRST and reports the slots it
|
|
98
|
+
* re-dispatched; those are excluded here. A re-dispatched slot is owned by an
|
|
99
|
+
* asynchronous run that has not claimed its lease yet, so without the
|
|
100
|
+
* exclusion this tick would see "expired lease + stale heartbeat" and fail the
|
|
101
|
+
* very occurrence recovery just saved — and crash-resume (§manual-resume)
|
|
102
|
+
* would silently become "fail everything that was interrupted long enough".
|
|
103
|
+
* The sweep therefore answers the other half of the question: which slots is
|
|
104
|
+
* recovery UNABLE to rescue (no enabled plan, a stopped/failed Scheduler, or a
|
|
105
|
+
* queue that never drains)? Those are the ones that would otherwise stay
|
|
106
|
+
* non-terminal forever.
|
|
107
|
+
*/
|
|
108
|
+
runStallSweep(rescued = new Set()) {
|
|
109
|
+
const database = this.options.database;
|
|
110
|
+
if (!database)
|
|
111
|
+
return;
|
|
112
|
+
try {
|
|
113
|
+
const result = (0, StallSweep_1.sweepStalledSlots)(database, {
|
|
114
|
+
...(0, StallSweep_1.resolveStallTimeouts)(this.activeConfig?.schedulerRuntime),
|
|
115
|
+
skipSlotIds: rescued,
|
|
116
|
+
});
|
|
117
|
+
if (result.queuedTooLong > 0 || result.stalledNoHeartbeat > 0) {
|
|
118
|
+
logger_1.logger.warn('Liveness sweep terminalised stalled slots', { ...result });
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
catch (error) {
|
|
122
|
+
logger_1.logger.warn('Liveness sweep failed; will retry on the next tick', {
|
|
123
|
+
error: error instanceof Error ? error.message : String(error),
|
|
124
|
+
});
|
|
125
|
+
}
|
|
126
|
+
}
|
|
90
127
|
startRecoveryLoop() {
|
|
91
128
|
if (!this.options.database)
|
|
92
129
|
return;
|
|
@@ -97,7 +134,7 @@ class MultiScheduleManager {
|
|
|
97
134
|
}
|
|
98
135
|
this.recoveryTimer = setInterval(() => {
|
|
99
136
|
try {
|
|
100
|
-
this.recoverInterruptedSlots();
|
|
137
|
+
this.runStallSweep(this.recoverInterruptedSlots());
|
|
101
138
|
}
|
|
102
139
|
catch (error) {
|
|
103
140
|
logger_1.logger.warn('Slot recovery sweep failed; will retry on the next tick', {
|
|
@@ -125,15 +162,16 @@ class MultiScheduleManager {
|
|
|
125
162
|
*/
|
|
126
163
|
recoverInterruptedSlots() {
|
|
127
164
|
const database = this.options.database;
|
|
165
|
+
const rescued = new Set();
|
|
128
166
|
if (!database)
|
|
129
|
-
return;
|
|
167
|
+
return rescued;
|
|
130
168
|
const config = this.activeConfig;
|
|
131
169
|
const plans = new Map((0, schedules_1.resolveSchedules)(config)
|
|
132
170
|
.filter((plan) => plan.enabled)
|
|
133
171
|
.map((plan) => [plan.id, plan]));
|
|
134
172
|
const recovered = database.slots.recoverableSlots();
|
|
135
173
|
if (recovered.length === 0)
|
|
136
|
-
return;
|
|
174
|
+
return rescued;
|
|
137
175
|
let reclaimed = 0;
|
|
138
176
|
let skipped = 0;
|
|
139
177
|
for (const slot of recovered) {
|
|
@@ -147,6 +185,9 @@ class MultiScheduleManager {
|
|
|
147
185
|
slot: slot.id,
|
|
148
186
|
schedule: slot.scheduleId,
|
|
149
187
|
status: slot.status,
|
|
188
|
+
trigger_source: slot.triggerSource,
|
|
189
|
+
manual_request_id: slot.manualRequestId,
|
|
190
|
+
correlation_id: slot.correlationId,
|
|
150
191
|
});
|
|
151
192
|
continue;
|
|
152
193
|
}
|
|
@@ -177,16 +218,37 @@ class MultiScheduleManager {
|
|
|
177
218
|
triggerSource: context.triggerSource,
|
|
178
219
|
slot: context,
|
|
179
220
|
});
|
|
180
|
-
if (admitted)
|
|
221
|
+
if (admitted) {
|
|
181
222
|
reclaimed++;
|
|
182
|
-
|
|
223
|
+
rescued.add(slot.id);
|
|
224
|
+
}
|
|
225
|
+
else {
|
|
183
226
|
skipped++;
|
|
227
|
+
// NOT silent: a re-dispatch that keeps failing is exactly how a slot sat
|
|
228
|
+
// `pending` for days behind a stopped scheduler. Report the reason with
|
|
229
|
+
// the caller's own request id so a stuck manual job is traceable.
|
|
230
|
+
const scheduler = this.schedulers.get(slot.scheduleId);
|
|
231
|
+
const reason = !scheduler
|
|
232
|
+
? 'scheduler_not_instantiated'
|
|
233
|
+
: scheduler.getStats().stopped
|
|
234
|
+
? 'schedule_stopped'
|
|
235
|
+
: 'scheduler_busy';
|
|
236
|
+
logger_1.logger.warn('Cannot recover slot: its schedule could not be admitted', {
|
|
237
|
+
slot: slot.id,
|
|
238
|
+
schedule: slot.scheduleId,
|
|
239
|
+
status: slot.status,
|
|
240
|
+
reason,
|
|
241
|
+
manual_request_id: slot.manualRequestId,
|
|
242
|
+
correlation_id: slot.correlationId,
|
|
243
|
+
});
|
|
244
|
+
}
|
|
184
245
|
}
|
|
185
246
|
logger_1.logger.info('Recovered interrupted slots', {
|
|
186
247
|
stale_slots_found: recovered.length,
|
|
187
248
|
reclaimed,
|
|
188
249
|
skipped,
|
|
189
250
|
});
|
|
251
|
+
return rescued;
|
|
190
252
|
}
|
|
191
253
|
/**
|
|
192
254
|
* Self-healing: if the daemon was down across a cron fire (deploy window,
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The Workflow Protocol v1 error vocabulary, producer side (§5 / §5.1).
|
|
3
|
+
*
|
|
4
|
+
* The protocol's error enum is CLOSED. PixivFlow keeps its own, richer internal
|
|
5
|
+
* `TerminalReasonCode` vocabulary for diagnosis, and translates it exactly once,
|
|
6
|
+
* at the job facade — a consumer never sees an internal code, and this service
|
|
7
|
+
* never invents a protocol code. `protocol/v1/error-mapping.json` (vendored from
|
|
8
|
+
* the deploy repo's SSOT) is the machine-checkable half of that contract; the
|
|
9
|
+
* `retryable` defaults below mirror its `protocol_codes` table, and
|
|
10
|
+
* `src/__tests__/protocol/contract.test.ts` keeps the two in step.
|
|
11
|
+
*
|
|
12
|
+
* Leaf module on purpose: no imports beyond types, so both the scheduler core
|
|
13
|
+
* and the HTTP facade can use it without a cycle.
|
|
14
|
+
*/
|
|
15
|
+
/** The protocol's closed error enum, verbatim from `$defs/Error` / `protocol_codes`. */
|
|
16
|
+
export type ProtocolErrorCode = 'no_candidate' | 'source_error' | 'auth_error' | 'quota_exceeded' | 'resource_busy' | 'queued_too_long' | 'stalled_no_progress' | 'deadline_exceeded' | 'cancelled_by_consumer' | 'idempotency_conflict' | 'unsupported_protocol_version' | 'invalid_params' | 'delivery_failed' | 'delivery_rejected' | 'delivery_abandoned' | 'internal_error';
|
|
17
|
+
/**
|
|
18
|
+
* A consumer/operator stop. Internal (`TerminalReasonCode`) and protocol
|
|
19
|
+
* (`ProtocolErrorCode`) name the same event with the same word, so it is
|
|
20
|
+
* defined once here — the delivery layer must recognise it without importing
|
|
21
|
+
* the scheduler.
|
|
22
|
+
*/
|
|
23
|
+
export declare const CANCELLED_BY_CONSUMER = "cancelled_by_consumer";
|
|
24
|
+
/**
|
|
25
|
+
* `retryable` defaults, copied from `protocol/v1/error-mapping.json`
|
|
26
|
+
* (`protocol_codes[*].retryable`). A payload may override the flag; it may not
|
|
27
|
+
* omit it, so a consumer can always branch on it without guessing.
|
|
28
|
+
*/
|
|
29
|
+
export declare const PROTOCOL_ERROR_RETRYABLE: Record<ProtocolErrorCode, boolean>;
|
|
30
|
+
/** `$defs/Error`: `code` is required; everything else is optional. */
|
|
31
|
+
export interface ProtocolErrorBody {
|
|
32
|
+
code: ProtocolErrorCode;
|
|
33
|
+
message?: string;
|
|
34
|
+
retryable?: boolean;
|
|
35
|
+
detail?: Record<string, unknown>;
|
|
36
|
+
}
|
|
37
|
+
export interface ProtocolErrorOptions {
|
|
38
|
+
message?: string;
|
|
39
|
+
detail?: Record<string, unknown>;
|
|
40
|
+
/** Override the table default (never omit the field: `retryable` is always sent). */
|
|
41
|
+
retryable?: boolean;
|
|
42
|
+
}
|
|
43
|
+
/** Build a `$defs/Error` body with the protocol's `retryable` default applied. */
|
|
44
|
+
export declare function protocolErrorBody(code: ProtocolErrorCode, options?: ProtocolErrorOptions): ProtocolErrorBody;
|
|
45
|
+
/**
|
|
46
|
+
* A request that must be answered with a protocol `Error` body.
|
|
47
|
+
*
|
|
48
|
+
* `message` is the protocol-visible message (`$defs/Error.message`), NOT an
|
|
49
|
+
* internal detail: the legacy manual-work endpoints keep matching on a few
|
|
50
|
+
* stable message strings, so those strings are passed through verbatim here and
|
|
51
|
+
* the protocol body carries the same text. Nothing may branch on message text
|
|
52
|
+
* going forward (§5) — the `code` is the contract.
|
|
53
|
+
*/
|
|
54
|
+
export declare class ProtocolRequestError extends Error {
|
|
55
|
+
readonly status: number;
|
|
56
|
+
readonly body: ProtocolErrorBody;
|
|
57
|
+
constructor(code: ProtocolErrorCode, status: number, options?: ProtocolErrorOptions);
|
|
58
|
+
}
|
|
59
|
+
/**
|
|
60
|
+
* Coerce anything thrown by a handler into an HTTP status + protocol Error body.
|
|
61
|
+
* Unknown failures become `internal_error` (retryable: false) and never leak the
|
|
62
|
+
* internal message to the consumer.
|
|
63
|
+
*/
|
|
64
|
+
export declare function protocolErrorResponse(error: unknown): {
|
|
65
|
+
status: number;
|
|
66
|
+
body: ProtocolErrorBody;
|
|
67
|
+
};
|
|
68
|
+
//# sourceMappingURL=ProtocolErrors.d.ts.map
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* The Workflow Protocol v1 error vocabulary, producer side (§5 / §5.1).
|
|
4
|
+
*
|
|
5
|
+
* The protocol's error enum is CLOSED. PixivFlow keeps its own, richer internal
|
|
6
|
+
* `TerminalReasonCode` vocabulary for diagnosis, and translates it exactly once,
|
|
7
|
+
* at the job facade — a consumer never sees an internal code, and this service
|
|
8
|
+
* never invents a protocol code. `protocol/v1/error-mapping.json` (vendored from
|
|
9
|
+
* the deploy repo's SSOT) is the machine-checkable half of that contract; the
|
|
10
|
+
* `retryable` defaults below mirror its `protocol_codes` table, and
|
|
11
|
+
* `src/__tests__/protocol/contract.test.ts` keeps the two in step.
|
|
12
|
+
*
|
|
13
|
+
* Leaf module on purpose: no imports beyond types, so both the scheduler core
|
|
14
|
+
* and the HTTP facade can use it without a cycle.
|
|
15
|
+
*/
|
|
16
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
17
|
+
exports.ProtocolRequestError = exports.PROTOCOL_ERROR_RETRYABLE = exports.CANCELLED_BY_CONSUMER = void 0;
|
|
18
|
+
exports.protocolErrorBody = protocolErrorBody;
|
|
19
|
+
exports.protocolErrorResponse = protocolErrorResponse;
|
|
20
|
+
/**
|
|
21
|
+
* A consumer/operator stop. Internal (`TerminalReasonCode`) and protocol
|
|
22
|
+
* (`ProtocolErrorCode`) name the same event with the same word, so it is
|
|
23
|
+
* defined once here — the delivery layer must recognise it without importing
|
|
24
|
+
* the scheduler.
|
|
25
|
+
*/
|
|
26
|
+
exports.CANCELLED_BY_CONSUMER = 'cancelled_by_consumer';
|
|
27
|
+
/**
|
|
28
|
+
* `retryable` defaults, copied from `protocol/v1/error-mapping.json`
|
|
29
|
+
* (`protocol_codes[*].retryable`). A payload may override the flag; it may not
|
|
30
|
+
* omit it, so a consumer can always branch on it without guessing.
|
|
31
|
+
*/
|
|
32
|
+
exports.PROTOCOL_ERROR_RETRYABLE = {
|
|
33
|
+
no_candidate: true,
|
|
34
|
+
source_error: true,
|
|
35
|
+
auth_error: false,
|
|
36
|
+
quota_exceeded: true,
|
|
37
|
+
resource_busy: true,
|
|
38
|
+
queued_too_long: true,
|
|
39
|
+
stalled_no_progress: true,
|
|
40
|
+
deadline_exceeded: true,
|
|
41
|
+
cancelled_by_consumer: false,
|
|
42
|
+
idempotency_conflict: false,
|
|
43
|
+
unsupported_protocol_version: false,
|
|
44
|
+
invalid_params: false,
|
|
45
|
+
delivery_failed: true,
|
|
46
|
+
delivery_rejected: false,
|
|
47
|
+
delivery_abandoned: true,
|
|
48
|
+
internal_error: false,
|
|
49
|
+
};
|
|
50
|
+
/** Build a `$defs/Error` body with the protocol's `retryable` default applied. */
|
|
51
|
+
function protocolErrorBody(code, options = {}) {
|
|
52
|
+
return {
|
|
53
|
+
code,
|
|
54
|
+
...(options.message !== undefined ? { message: options.message } : {}),
|
|
55
|
+
retryable: options.retryable ?? exports.PROTOCOL_ERROR_RETRYABLE[code],
|
|
56
|
+
...(options.detail !== undefined ? { detail: options.detail } : {}),
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
/**
|
|
60
|
+
* A request that must be answered with a protocol `Error` body.
|
|
61
|
+
*
|
|
62
|
+
* `message` is the protocol-visible message (`$defs/Error.message`), NOT an
|
|
63
|
+
* internal detail: the legacy manual-work endpoints keep matching on a few
|
|
64
|
+
* stable message strings, so those strings are passed through verbatim here and
|
|
65
|
+
* the protocol body carries the same text. Nothing may branch on message text
|
|
66
|
+
* going forward (§5) — the `code` is the contract.
|
|
67
|
+
*/
|
|
68
|
+
class ProtocolRequestError extends Error {
|
|
69
|
+
status;
|
|
70
|
+
body;
|
|
71
|
+
constructor(code, status, options = {}) {
|
|
72
|
+
const body = protocolErrorBody(code, options);
|
|
73
|
+
super(body.message ?? code);
|
|
74
|
+
this.name = 'ProtocolRequestError';
|
|
75
|
+
this.status = status;
|
|
76
|
+
this.body = body;
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
exports.ProtocolRequestError = ProtocolRequestError;
|
|
80
|
+
/**
|
|
81
|
+
* Coerce anything thrown by a handler into an HTTP status + protocol Error body.
|
|
82
|
+
* Unknown failures become `internal_error` (retryable: false) and never leak the
|
|
83
|
+
* internal message to the consumer.
|
|
84
|
+
*/
|
|
85
|
+
function protocolErrorResponse(error) {
|
|
86
|
+
if (error instanceof ProtocolRequestError) {
|
|
87
|
+
return { status: error.status, body: error.body };
|
|
88
|
+
}
|
|
89
|
+
return {
|
|
90
|
+
status: 500,
|
|
91
|
+
body: protocolErrorBody('internal_error', { message: 'internal error' }),
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
//# sourceMappingURL=ProtocolErrors.js.map
|
|
@@ -1,5 +1,9 @@
|
|
|
1
1
|
import { Request } from 'express';
|
|
2
|
+
import { Server } from 'node:http';
|
|
2
3
|
import { SlotContext } from './SlotCoordinator';
|
|
4
|
+
import { JobStatusProjection } from './JobProjection';
|
|
5
|
+
import { ProtocolAckResult, ProtocolCapabilities, ProtocolEventPage, ProtocolJob } from './JobFacade';
|
|
6
|
+
import type { JobEventQuery } from './JobEventStream';
|
|
3
7
|
import { TriggerSource } from './OccurrenceResolver';
|
|
4
8
|
/**
|
|
5
9
|
* Authenticated HTTP schedule trigger — a DISPATCH endpoint, not a run endpoint.
|
|
@@ -105,12 +109,12 @@ export interface TriggerHandlers {
|
|
|
105
109
|
slotId: string;
|
|
106
110
|
disposition: string;
|
|
107
111
|
}>;
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
112
|
+
/**
|
|
113
|
+
* Generic Job projection for one manual job. The four legacy identity fields
|
|
114
|
+
* keep their exact meaning; the rest is additive liveness + terminal-cause
|
|
115
|
+
* detail (see `JobStatusProjection`). Null means "no such job" -> 404.
|
|
116
|
+
*/
|
|
117
|
+
refetchStatus?(targetId: string, requestId: string): JobStatusProjection | null;
|
|
114
118
|
/**
|
|
115
119
|
* Admit one manual RECOVERY of a failed target (§manual-recovery). `retryMode`
|
|
116
120
|
* selects a SERVER-DEFINED acquisition policy preset ('normal' | 'relaxed');
|
|
@@ -131,6 +135,42 @@ export interface TriggerHandlers {
|
|
|
131
135
|
/** User-facing copy; never leaks internal error names. */
|
|
132
136
|
message?: string;
|
|
133
137
|
} | null;
|
|
138
|
+
/**
|
|
139
|
+
* Optional generic job facade (Workflow Protocol v1). Mounted on the SAME
|
|
140
|
+
* server and guarded by the SAME refetch token as the manual-work family, so
|
|
141
|
+
* there is exactly one HTTP system and one execution path: both this generic
|
|
142
|
+
* surface and the legacy `/internal/targets/:targetId/refetch` shim call the
|
|
143
|
+
* one shared admission function behind `submitJob`.
|
|
144
|
+
*/
|
|
145
|
+
jobs?: JobHandlers;
|
|
146
|
+
}
|
|
147
|
+
/**
|
|
148
|
+
* The producer side of the generic Workflow Protocol v1 surface. Implementations
|
|
149
|
+
* own all business semantics; the server only authenticates, routes and shapes
|
|
150
|
+
* the HTTP response. Every failure is thrown as a `ProtocolRequestError` so the
|
|
151
|
+
* body is always a schema-valid `Error` with a closed-enum `code`.
|
|
152
|
+
*/
|
|
153
|
+
export interface JobHandlers {
|
|
154
|
+
/** `GET /capabilities` body (`$defs/Capabilities`), derived from live config. */
|
|
155
|
+
capabilities(): ProtocolCapabilities;
|
|
156
|
+
/** `POST /jobs`: parse a `$defs/Task`, admit it, project the `$defs/Job`. */
|
|
157
|
+
submitJob(body: unknown): {
|
|
158
|
+
job: ProtocolJob;
|
|
159
|
+
replayed: boolean;
|
|
160
|
+
};
|
|
161
|
+
/** `GET /jobs/{job_id}`: null means 404. */
|
|
162
|
+
jobStatus(jobId: string): ProtocolJob | null;
|
|
163
|
+
/** `GET /jobs?idempotency_key=`: 0 or 1 job. */
|
|
164
|
+
jobsByIdempotencyKey(idempotencyKey: string): ProtocolJob[];
|
|
165
|
+
/** `POST /jobs/{job_id}/cancel`: idempotent; always returns the current job. */
|
|
166
|
+
cancelJob(jobId: string): ProtocolJob;
|
|
167
|
+
/** `GET /jobs/{job_id}/events` (`$defs/EventPage`); throws 404 for an unknown job. */
|
|
168
|
+
jobEvents(jobId: string, query: JobEventQuery): ProtocolEventPage;
|
|
169
|
+
/**
|
|
170
|
+
* `POST /jobs/{job_id}/events/ack` (`$defs/AckResult`): monotonic, idempotent,
|
|
171
|
+
* and never a job mutation. An unknown or older cursor is a no-op.
|
|
172
|
+
*/
|
|
173
|
+
ackJobEvents(jobId: string, body: unknown): ProtocolAckResult;
|
|
134
174
|
}
|
|
135
175
|
export declare class ScheduleTriggerServer {
|
|
136
176
|
private readonly token;
|
|
@@ -140,7 +180,8 @@ export declare class ScheduleTriggerServer {
|
|
|
140
180
|
constructor(token: string | undefined, handlers: TriggerHandlers, refetchToken?: string | undefined);
|
|
141
181
|
/** Token from config or SCHEDULER_TRIGGER_TOKEN env; empty => fail closed. */
|
|
142
182
|
static resolveToken(configured?: string): string | undefined;
|
|
143
|
-
start
|
|
183
|
+
/** `start` returns the listening server so tests/callers can read the port. */
|
|
184
|
+
start(host: string, port: number): Server;
|
|
144
185
|
/**
|
|
145
186
|
* Per-request correlation, installed before every route. It records only the
|
|
146
187
|
* sanitized attempt id and the arrival instant, so each later line can report
|
|
@@ -165,6 +206,14 @@ export declare class ScheduleTriggerServer {
|
|
|
165
206
|
* clock did not send them, so an unidentified clock stays visible as such.
|
|
166
207
|
*/
|
|
167
208
|
private attemptMeta;
|
|
209
|
+
/**
|
|
210
|
+
* Every generic-surface failure is written as a protocol `Error`: a closed-enum
|
|
211
|
+
* `code`, a `retryable` boolean from the published defaults, and a `detail`
|
|
212
|
+
* object. Nothing here judges a failure by its message text.
|
|
213
|
+
*/
|
|
214
|
+
private jobFailure;
|
|
215
|
+
/** Fail closed when the generic surface is not mounted in this runtime. */
|
|
216
|
+
private jobUnavailableResponse;
|
|
168
217
|
private auth;
|
|
169
218
|
private refetchAuth;
|
|
170
219
|
private authenticate;
|