pg-boss 12.33.2 → 12.33.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/attorney.d.ts.map +1 -1
- package/dist/attorney.js +13 -12
- package/dist/bam.d.ts.map +1 -1
- package/dist/bam.js +74 -28
- package/dist/boss.d.ts +1 -1
- package/dist/boss.js +14 -14
- package/dist/claimTimer.d.ts +11 -4
- package/dist/claimTimer.d.ts.map +1 -1
- package/dist/claimTimer.js +14 -7
- package/dist/cli.js +15 -14
- package/dist/contractor.js +2 -2
- package/dist/db.js +2 -2
- package/dist/drifter.js +21 -21
- package/dist/index.d.ts +2 -2
- package/dist/index.js +3 -3
- package/dist/manager.js +18 -18
- package/dist/migrationStore.js +11 -11
- package/dist/plans.d.ts +8 -7
- package/dist/plans.d.ts.map +1 -1
- package/dist/plans.js +117 -70
- package/dist/timekeeper.d.ts +34 -8
- package/dist/timekeeper.d.ts.map +1 -1
- package/dist/timekeeper.js +73 -12
- package/dist/types.d.ts +21 -21
- package/dist/types.d.ts.map +1 -1
- package/dist/worker.js +1 -1
- package/package.json +6 -5
package/README.md
CHANGED
|
@@ -74,7 +74,7 @@ A HTTP proxy is available in the [`@pg-boss/proxy`](https://www.npmjs.com/packag
|
|
|
74
74
|
See the [proxy documentation](https://pgboss.io/proxy) for details.
|
|
75
75
|
|
|
76
76
|
## Requirements
|
|
77
|
-
* Node 22.12 or higher
|
|
77
|
+
* Node 22.12 or higher, or Bun
|
|
78
78
|
* PostgreSQL 13 or higher
|
|
79
79
|
|
|
80
80
|
## Sponsors
|
package/dist/attorney.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"attorney.d.ts","sourceRoot":"","sources":["../src/attorney.ts"],"names":[],"mappings":"AAGA,OAAO,KAAK,KAAK,KAAK,MAAM,YAAY,CAAA;AAExC,QAAA,MAAM,MAAM;;;;CAIX,CAAA;AAKD,QAAA,MAAM,mBAAmB,uQAaf,CAAA;AA2EV,iBAAS,iBAAiB,CAAE,MAAM,GAAE,GAAQ,QAc3C;
|
|
1
|
+
{"version":3,"file":"attorney.d.ts","sourceRoot":"","sources":["../src/attorney.ts"],"names":[],"mappings":"AAGA,OAAO,KAAK,KAAK,KAAK,MAAM,YAAY,CAAA;AAExC,QAAA,MAAM,MAAM;;;;CAIX,CAAA;AAKD,QAAA,MAAM,mBAAmB,uQAaf,CAAA;AA2EV,iBAAS,iBAAiB,CAAE,MAAM,GAAE,GAAQ,QAc3C;AAiBD,iBAAS,mBAAmB,CAAE,KAAK,EAAE,MAAM,UAE1C;AAED,iBAAS,aAAa,CAAE,IAAI,EAAE,GAAG,GAAG,KAAK,CAAC,OAAO,CAgDhD;AAOD,iBAAS,eAAe,CAAE,IAAI,EAAE,GAAG,EAAE,EAAE,MAAc,EAAE;;CAAK,GAAG,KAAK,CAAC,OAAO,CA4D3E;AAED,iBAAS,mBAAmB,CAAE,MAAM,EAAE,GAAG,QAOxC;AAED,iBAAS,gBAAgB,CAAE,IAAI,EAAE,KAAK,CAAC,OAAO,EAAE,QA2D/C;AA2GD,iBAAS,aAAa,CAAE,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,EAAE,GAAG;IAClD,OAAO,EAAE,KAAK,CAAC,mBAAmB,CAAA;IAClC,QAAQ,EAAE,KAAK,CAAC,WAAW,CAAC,GAAG,CAAC,CAAA;CACjC,CAgCA;AAqBD,iBAAS,cAAc,CAAE,IAAI,EAAE,MAAM,EAAE,OAAO,EAAE,GAAG,QAQlD;AAED,iBAAS,SAAS,CAAE,KAAK,EAAE,MAAM,GAAG,KAAK,CAAC,kBAAkB,GAAG,KAAK,CAAC,0BAA0B,CA0B9F;AA8GD,iBAAS,wBAAwB,CAAE,IAAI,EAAE,MAAM,QAiC9C;AAED,iBAAS,eAAe,CAAE,IAAI,EAAE,MAAM,QAIrC;AAED,iBAAS,SAAS,CAAE,GAAG,EAAE,MAAM,QAI9B;AA+LD,OAAO,EACL,SAAS,EACT,mBAAmB,EACnB,wBAAwB,EACxB,eAAe,EACf,cAAc,EACd,aAAa,EACb,eAAe,EACf,aAAa,EACb,SAAS,EACT,mBAAmB,EACnB,MAAM,EACN,gBAAgB,EAChB,mBAAmB,EACnB,iBAAiB,EAClB,CAAA"}
|
package/dist/attorney.js
CHANGED
|
@@ -41,10 +41,10 @@ const BACKEND_PROFILES = {
|
|
|
41
41
|
noAddColumnBackfill: true,
|
|
42
42
|
noListenNotify: true,
|
|
43
43
|
// Online DDL runs as a schema-change job, not the PG CONCURRENTLY path, and
|
|
44
|
-
// pg_stat_progress_create_index isn't available
|
|
44
|
+
// pg_stat_progress_create_index isn't available, so BAM can't use liveness-based reclaim.
|
|
45
45
|
noIndexProgressView: true,
|
|
46
|
-
// REINDEX is rejected in either form
|
|
47
|
-
// CONCURRENTLY, "unimplemented: this syntax" without
|
|
46
|
+
// REINDEX is rejected in either form, "CockroachDB does not require reindexing" with
|
|
47
|
+
// CONCURRENTLY, "unimplemented: this syntax" without, and the bloat check itself cannot run:
|
|
48
48
|
// there is no pg_relation_size(), and reltuples / relpages is an "unsupported binary
|
|
49
49
|
// operator: <float4> / <int4>".
|
|
50
50
|
noReindex: true,
|
|
@@ -74,7 +74,7 @@ const BACKEND_PROFILES = {
|
|
|
74
74
|
// No noIndexProgressView: pg-boss keeps its tables coordinator-local (it never calls
|
|
75
75
|
// create_distributed_table), so CREATE INDEX CONCURRENTLY runs against ordinary local Postgres tables
|
|
76
76
|
// on the coordinator, where pg_stat_progress_create_index is accurate and liveness-based reclaim is
|
|
77
|
-
// valid. This holds ONLY while the tables stay coordinator-local
|
|
77
|
+
// valid. This holds ONLY while the tables stay coordinator-local. If they are ever distributed, the
|
|
78
78
|
// coordinator's progress view would misread in-flight worker builds as dead and BAM could double-build.
|
|
79
79
|
citus: { kind: 'distributed', flags: {} },
|
|
80
80
|
pglite: { kind: 'embedded', flags: {} }
|
|
@@ -103,8 +103,9 @@ function validateQueueArgs(config = {}) {
|
|
|
103
103
|
const ISO_DATE = /^\d{4}-\d{2}-\d{2}/;
|
|
104
104
|
// Matched against the remainder past YYYY-MM-DD, and deliberately a whitelist of the zone-less
|
|
105
105
|
// spellings rather than a test for a missing zone: a trailing '-01' would read as an offset, and
|
|
106
|
-
// forms Postgres resolves on its own
|
|
107
|
-
//
|
|
106
|
+
// forms Postgres resolves on its own must be left exactly as they are rather than pinned to UTC.
|
|
107
|
+
// Those forms are a named zone ('... UTC', '... America/New_York') and a single-digit offset
|
|
108
|
+
// ('+5:30').
|
|
108
109
|
const ZONELESS_TIME = /^(?:[T ]\d{2}(?::\d{2}(?::\d{2}(?:[.,]\d+)?)?)?)?$/;
|
|
109
110
|
function pinZonelessDateTime(value) {
|
|
110
111
|
return ISO_DATE.test(value) && ZONELESS_TIME.test(value.slice(10)) ? value + 'Z' : value;
|
|
@@ -172,14 +173,14 @@ function checkUpdateArgs(args, { upsert = false } = {}) {
|
|
|
172
173
|
assert(typeof options === 'object', 'options should be an object');
|
|
173
174
|
const { id, singletonKey, match } = options;
|
|
174
175
|
// Both update() and upsert() target by exactly one of id or singletonKey. (upsert() may also
|
|
175
|
-
// require a singletonKey at runtime on key_strict_fifo queues
|
|
176
|
-
// knows the policy
|
|
176
|
+
// require a singletonKey at runtime on key_strict_fifo queues, enforced in the manager, which
|
|
177
|
+
// knows the policy, because an insert-on-miss there needs a key.)
|
|
177
178
|
assert((!!id) !== (!!singletonKey), `${verb} requires exactly one of id or singletonKey`);
|
|
178
179
|
assert(!(id && match !== undefined), 'match is only valid when targeting jobs by singletonKey');
|
|
179
180
|
assert(match === undefined || JOB_MATCH_STRATEGIES.includes(match), `match must be one of: ${JOB_MATCH_STRATEGIES.join(', ')}`);
|
|
180
181
|
assert(!('priority' in options) || (Number.isInteger(options.priority)), 'priority must be an integer');
|
|
181
182
|
if ('startAfter' in options) {
|
|
182
|
-
// Unlike send(), update() must honor a numeric startAfter of 0 (or negative)
|
|
183
|
+
// Unlike send(), update() must honor a numeric startAfter of 0 (or negative). The caller is
|
|
183
184
|
// explicitly pulling a deferred job forward to now. The send-style `+startAfter > 0` guard
|
|
184
185
|
// coerced 0 to undefined, which JSON.stringify then dropped, silently making the edit a no-op.
|
|
185
186
|
// Any finite number is passed through as a seconds interval ('0' -> now()); a string is kept.
|
|
@@ -442,7 +443,7 @@ function validateWarningConfig(config) {
|
|
|
442
443
|
assert(!('warningRetentionDays' in config) || config.warningRetentionDays <= POLICY.MAX_RETENTION_DAYS, `configuration assert: warningRetentionDays cannot exceed ${POLICY.MAX_RETENTION_DAYS} days`);
|
|
443
444
|
}
|
|
444
445
|
// Expands config.backend into the internal compatibility flags. The flags are derived
|
|
445
|
-
// solely from the backend profile
|
|
446
|
+
// solely from the backend profile. They are not part of the public input, so a
|
|
446
447
|
// deployment can't end up with an inconsistent combination.
|
|
447
448
|
function resolveBackend(config) {
|
|
448
449
|
const backend = ('backend' in config) ? config.backend : 'postgres';
|
|
@@ -582,7 +583,7 @@ function applyPollingInterval(config) {
|
|
|
582
583
|
: 2000;
|
|
583
584
|
assert(!('notifyPollingIntervalSeconds' in config) || config.notifyPollingIntervalSeconds >= POLICY.MIN_POLLING_INTERVAL_MS / 1000, `configuration assert: notifyPollingIntervalSeconds must be at least every ${POLICY.MIN_POLLING_INTERVAL_MS}ms`);
|
|
584
585
|
// Relaxed backstop poll used only while NOTIFY is active for the queue; falls back to
|
|
585
|
-
// pollingInterval when notify is unavailable. It must never be smaller than the base poll
|
|
586
|
+
// pollingInterval when notify is unavailable. It must never be smaller than the base poll,
|
|
586
587
|
// that would make a notify-active queue poll more aggressively than an idle one, the opposite
|
|
587
588
|
// of the intent. When explicit, reject a value below the base; when defaulted, floor it at the
|
|
588
589
|
// base so bumping pollingIntervalSeconds past 30s can't silently leave notify smaller.
|
|
@@ -629,7 +630,7 @@ function validateReindexConfig(config) {
|
|
|
629
630
|
assert(maxIndexBytes === undefined || (Number.isInteger(maxIndexBytes) && maxIndexBytes > 0), 'configuration assert: reindex.maxIndexBytes must be an integer > 0');
|
|
630
631
|
// force belongs to an explicit supervise() call, not to a background timer that would then
|
|
631
632
|
// rebuild every job index on every interval.
|
|
632
|
-
assert(force === undefined || force === false, 'configuration assert: reindex.force cannot be set in constructor options
|
|
633
|
+
assert(force === undefined || force === false, 'configuration assert: reindex.force cannot be set in constructor options. Pass it to supervise()');
|
|
633
634
|
}
|
|
634
635
|
assert(!('reindexIntervalSeconds' in config) || config.reindexIntervalSeconds >= 1, 'configuration assert: reindexIntervalSeconds must be at least every second');
|
|
635
636
|
config.reindexIntervalSeconds = config.reindexIntervalSeconds || POLICY.MAX_EXPIRATION_HOURS * 60 * 60;
|
package/dist/bam.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"bam.d.ts","sourceRoot":"","sources":["../src/bam.ts"],"names":[],"mappings":"AAAA,OAAO,YAAY,MAAM,aAAa,CAAA;AAItC,OAAO,KAAK,KAAK,MAAM,YAAY,CAAA;
|
|
1
|
+
{"version":3,"file":"bam.d.ts","sourceRoot":"","sources":["../src/bam.ts"],"names":[],"mappings":"AAAA,OAAO,YAAY,MAAM,aAAa,CAAA;AAItC,OAAO,KAAK,KAAK,MAAM,YAAY,CAAA;AAgBnC,cAAM,GAAI,SAAQ,YAAa,YAAW,KAAK,CAAC,WAAW;;IAOzD,MAAM;;;MAAS;gBAGb,EAAE,EAAE,KAAK,CAAC,SAAS,EACnB,MAAM,EAAE,KAAK,CAAC,0BAA0B;IAU1C,IAAI,OAAO,IAAK,OAAO,CAEtB;IAEK,KAAK;IAcL,IAAI;CAwLX;AAED,eAAe,GAAG,CAAA"}
|
package/dist/bam.js
CHANGED
|
@@ -36,10 +36,8 @@ class Bam extends EventEmitter {
|
|
|
36
36
|
if (this.#stopped)
|
|
37
37
|
return;
|
|
38
38
|
this.#stopped = true;
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
this.#pollTimer = undefined;
|
|
42
|
-
}
|
|
39
|
+
this.#pollTimer.stop();
|
|
40
|
+
this.#pollTimer = undefined;
|
|
43
41
|
while (this.#working) {
|
|
44
42
|
await delay(10);
|
|
45
43
|
}
|
|
@@ -75,8 +73,25 @@ class Bam extends EventEmitter {
|
|
|
75
73
|
if (this.#stopped)
|
|
76
74
|
return;
|
|
77
75
|
const entry = await this.#getNextCommand();
|
|
78
|
-
if (!entry
|
|
76
|
+
if (!entry)
|
|
77
|
+
return;
|
|
78
|
+
if (this.#config.__test__delay_bam_claim_ms) {
|
|
79
|
+
await delay(this.#config.__test__delay_bam_claim_ms);
|
|
80
|
+
}
|
|
81
|
+
if (this.#stopped) {
|
|
82
|
+
// The claim landed on the wrong side of a stop. The command has not run, so hand the row back
|
|
83
|
+
// exactly as it was found rather than leaving it in_progress - an in_progress row blocks the
|
|
84
|
+
// whole BAM queue until it goes stale, which is the grace window on native Postgres and 24
|
|
85
|
+
// hours on the timeout-only backends. If the release itself fails the row keeps that claim and
|
|
86
|
+
// recovers on the stale path, so surface it rather than throwing into #onPoll.
|
|
87
|
+
try {
|
|
88
|
+
await this.#releaseCommand(entry);
|
|
89
|
+
}
|
|
90
|
+
catch (err) {
|
|
91
|
+
this.emit(events.error, err);
|
|
92
|
+
}
|
|
79
93
|
return;
|
|
94
|
+
}
|
|
80
95
|
this.emit(events.bam, {
|
|
81
96
|
id: entry.id,
|
|
82
97
|
name: entry.name,
|
|
@@ -85,31 +100,50 @@ class Bam extends EventEmitter {
|
|
|
85
100
|
table: entry.table
|
|
86
101
|
});
|
|
87
102
|
try {
|
|
88
|
-
|
|
89
|
-
//
|
|
90
|
-
// failed
|
|
91
|
-
//
|
|
92
|
-
//
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
if (dropSql) {
|
|
97
|
-
// Only heal an index the previous attempt left INVALID. A re-attempt can also fire for a
|
|
98
|
-
// build that actually SUCCEEDED but whose row was never marked completed (a graceful stop
|
|
99
|
-
// landed between the CREATE and markCompleted) — that index is VALID and in use, so dropping
|
|
100
|
-
// it would tear down a live production index for the whole rebuild window. Probe indisvalid
|
|
101
|
-
// first; skip the drop for a valid (or absent) index and let the command's IF NOT EXISTS re-run
|
|
102
|
-
// no-op it and mark the row done.
|
|
103
|
-
const probeSql = plans.bamHealProbe(this.#config.schema, entry.command);
|
|
103
|
+
let alreadyBuilt = false;
|
|
104
|
+
// A re-attempted command (a stale in_progress reclaim, or a retry of a prior 'failed', including
|
|
105
|
+
// failed rows left by older releases) needs the catalog consulted before the command is re-run,
|
|
106
|
+
// because the row keeps the text it was enqueued with and that text may not be idempotent.
|
|
107
|
+
// Probe indisvalid on every backend: pg_index is readable everywhere, and both outcomes matter.
|
|
108
|
+
if (entry.reattempt) {
|
|
109
|
+
const probeSql = plans.bamHealProbe(this.#config.schema, entry.command);
|
|
110
|
+
if (probeSql) {
|
|
104
111
|
const { rows } = await this.#db.executeSql(probeSql);
|
|
105
112
|
if (rows[0]?.invalid) {
|
|
106
|
-
|
|
113
|
+
// An interrupted or failed CREATE INDEX CONCURRENTLY left an INVALID stub. Drop it
|
|
114
|
+
// (best-effort, IF EXISTS) so the re-run rebuilds cleanly instead of the command's own
|
|
115
|
+
// IF NOT EXISTS skipping over a broken index forever. Only on the liveness path,
|
|
116
|
+
// CockroachDB/YugabyteDB roll interrupted builds back, so there is nothing to heal and
|
|
117
|
+
// DROP ... CONCURRENTLY isn't their model.
|
|
118
|
+
if (!this.#config.noIndexProgressView) {
|
|
119
|
+
// Non-null wherever the probe was: both recognise the same CONCURRENTLY commands.
|
|
120
|
+
await this.#db.executeSql(plans.bamHealDrop(this.#config.schema, entry.command));
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
else if (rows[0]) {
|
|
124
|
+
// The index is VALID, so the build already succeeded and only the row was never marked (a
|
|
125
|
+
// stop landed between the CREATE and markCompleted). Dropping it would tear down a live
|
|
126
|
+
// production index for the whole rebuild window, and re-running is not safe either: the two
|
|
127
|
+
// oldest index commands (job_i7, job_i8) were queued without IF NOT EXISTS, so re-running
|
|
128
|
+
// one against its own valid index fails with "already exists" and the failed row is retried
|
|
129
|
+
// forever. Short-circuit to marking the row done instead. This half is deliberately NOT
|
|
130
|
+
// gated on the liveness path. The timeout-only backends need it most, because their stale
|
|
131
|
+
// window is BAM_STALE_SECONDS (24 hours) rather than the grace window.
|
|
132
|
+
alreadyBuilt = true;
|
|
107
133
|
}
|
|
108
134
|
}
|
|
109
135
|
}
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
136
|
+
if (!alreadyBuilt) {
|
|
137
|
+
await this.#db.executeSql(entry.command);
|
|
138
|
+
}
|
|
139
|
+
// Record the outcome even when a stop landed while the command was running. stop() waits out
|
|
140
|
+
// #working and index.ts closes the pool only after #bam.stop() resolves, so the UPDATE is safe
|
|
141
|
+
// here. Bailing out instead would strand a VALID index behind an unmarked row: on native
|
|
142
|
+
// Postgres the whole BAM queue then waits out BAM_LIVENESS_GRACE_SECONDS, and on the
|
|
143
|
+
// timeout-only backends (CockroachDB/YugabyteDB) every other command is blocked for
|
|
144
|
+
// BAM_STALE_SECONDS - 24 hours - before anything is retried. A failure here falls through to
|
|
145
|
+
// the catch, which marks the row failed - the next attempt's probe finds the VALID index and
|
|
146
|
+
// completes it.
|
|
113
147
|
await this.#markCompleted(entry.id);
|
|
114
148
|
this.emit(events.bam, {
|
|
115
149
|
id: entry.id,
|
|
@@ -120,9 +154,17 @@ class Bam extends EventEmitter {
|
|
|
120
154
|
});
|
|
121
155
|
}
|
|
122
156
|
catch (err) {
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
157
|
+
// Same reasoning as the completed path: a stop must not leave the row in_progress. This write
|
|
158
|
+
// is guarded because it commonly runs on the connection that just failed - a terminated backend
|
|
159
|
+
// or dropped connection fails the command AND the UPDATE that would record it. Letting that
|
|
160
|
+
// throw would replace the real failure with the write's error and lose which command failed,
|
|
161
|
+
// so report both and let the stale path reclaim the row.
|
|
162
|
+
try {
|
|
163
|
+
await this.#markFailed(entry.id, err);
|
|
164
|
+
}
|
|
165
|
+
catch (markErr) {
|
|
166
|
+
this.emit(events.error, markErr);
|
|
167
|
+
}
|
|
126
168
|
this.emit(events.error, err);
|
|
127
169
|
this.emit(events.bam, {
|
|
128
170
|
id: entry.id,
|
|
@@ -139,6 +181,10 @@ class Bam extends EventEmitter {
|
|
|
139
181
|
const { rows } = await this.#db.executeSql(sql);
|
|
140
182
|
return rows[0] || null;
|
|
141
183
|
}
|
|
184
|
+
async #releaseCommand(entry) {
|
|
185
|
+
const sql = plans.releaseBamCommand(this.#config.schema, entry.id, entry.priorStatus, entry.priorStartedOn ?? null, entry.claimedStartedOn);
|
|
186
|
+
await this.#db.executeSql(sql);
|
|
187
|
+
}
|
|
142
188
|
async #markCompleted(id) {
|
|
143
189
|
const sql = plans.setBamCompleted(this.#config.schema, id);
|
|
144
190
|
await this.#db.executeSql(sql);
|
package/dist/boss.d.ts
CHANGED
|
@@ -15,7 +15,7 @@ declare class Boss extends EventEmitter implements types.EventsMixin {
|
|
|
15
15
|
/**
|
|
16
16
|
* The REINDEX statements this instance would run, for installations where it cannot run them
|
|
17
17
|
* itself. Unlike the background pass this applies no ownership filter and no size cap unless one
|
|
18
|
-
* is passed
|
|
18
|
+
* is passed. The commands are for an operator, who may run them as a different role.
|
|
19
19
|
*/
|
|
20
20
|
getReindexCommands(options?: types.ReindexOptions): Promise<string[]>;
|
|
21
21
|
}
|
package/dist/boss.js
CHANGED
|
@@ -68,7 +68,7 @@ function describeXminHolder(source, row) {
|
|
|
68
68
|
return `${who} (${where}) has held a transaction open ${open}`;
|
|
69
69
|
}
|
|
70
70
|
// SQLSTATE 25001 (active_sql_transaction): "REINDEX CONCURRENTLY cannot run inside a transaction
|
|
71
|
-
// block". Raised when a user-supplied adapter wraps executeSql in a transaction
|
|
71
|
+
// block". Raised when a user-supplied adapter wraps executeSql in a transaction. A property of the
|
|
72
72
|
// adapter, not of this pass, so it disables rebuilds for the life of the instance rather than
|
|
73
73
|
// retrying every interval.
|
|
74
74
|
const IN_TRANSACTION_ERROR = '25001';
|
|
@@ -86,7 +86,7 @@ class Boss extends EventEmitter {
|
|
|
86
86
|
// keeps running; only the DDL is abandoned.
|
|
87
87
|
#reindexUnavailable = null;
|
|
88
88
|
// Warn once per bloated index rather than on every pass. An index leaves the set as soon as it
|
|
89
|
-
// stops qualifying
|
|
89
|
+
// stops qualifying (rebuilt, dropped, or refilled) so a later episode warns again.
|
|
90
90
|
#warnedBloat = new Set();
|
|
91
91
|
// Local rate limit for passes that only report bloat, which deliberately leave the shared interval
|
|
92
92
|
// claim to whichever instance can act on it.
|
|
@@ -229,7 +229,7 @@ class Boss extends EventEmitter {
|
|
|
229
229
|
// Ensure today's/tomorrow's partitions exist before any insertQueueStats below. Retention
|
|
230
230
|
// (#maintainWarnings/#maintainQueueStats) runs at the tail. Both live here, in the public
|
|
231
231
|
// supervise() path that also performs the writes, rather than in the timer-only #onSupervise
|
|
232
|
-
// wrapper
|
|
232
|
+
// wrapper, so manual supervise() callers (instances run with the built-in supervisor disabled)
|
|
233
233
|
// get partitions provisioned and old data pruned, not just job retention.
|
|
234
234
|
if (this.#config.persistQueueStats && !this.#config.noTablePartitioning && !this.#stopping) {
|
|
235
235
|
await this.#ensureQueueStatsPartitions();
|
|
@@ -574,7 +574,7 @@ class Boss extends EventEmitter {
|
|
|
574
574
|
holder: describeXminHolder(holder.source, horizon.row),
|
|
575
575
|
holderClass: XMIN_HOLDERS[holder.source],
|
|
576
576
|
// Null unless the holder is a backend this role could read a row for. `self` says whether it
|
|
577
|
-
// shares this connection's application_name
|
|
577
|
+
// shares this connection's application_name. A pg-boss instance pinning its own horizon and an
|
|
578
578
|
// external reporting tool pinning it have opposite fixes.
|
|
579
579
|
holderPid: backend?.pid ?? null,
|
|
580
580
|
holderApplicationName: backend?.applicationName ?? null,
|
|
@@ -598,10 +598,10 @@ class Boss extends EventEmitter {
|
|
|
598
598
|
});
|
|
599
599
|
}
|
|
600
600
|
/**
|
|
601
|
-
* The widest holder in the row. Every source the query returns has already been qualified
|
|
601
|
+
* The widest holder in the row. Every source the query returns has already been qualified. The
|
|
602
602
|
* backends column is filtered server-side to transactions that predate the failed vacuum, and a
|
|
603
603
|
* slot, standby or prepared transaction advertises an xmin only while something is genuinely
|
|
604
|
-
* stuck
|
|
604
|
+
* stuck, so this is a straight maximum. The horizon is pinned to the oldest of them, which makes
|
|
605
605
|
* the widest the one worth naming.
|
|
606
606
|
*/
|
|
607
607
|
#attributeXminHorizon(row) {
|
|
@@ -665,7 +665,7 @@ class Boss extends EventEmitter {
|
|
|
665
665
|
}
|
|
666
666
|
return readable;
|
|
667
667
|
}
|
|
668
|
-
// DDL runs outside the slow-query timer. A REINDEX is expected to take seconds
|
|
668
|
+
// DDL runs outside the slow-query timer. A REINDEX is expected to take seconds, routing it
|
|
669
669
|
// through #executeQuery would emit a bogus slow_query warning on every rebuild.
|
|
670
670
|
async #executeDdl(sql) {
|
|
671
671
|
return unwrapSQLResult(await this.#db.executeSql(sql));
|
|
@@ -686,7 +686,7 @@ class Boss extends EventEmitter {
|
|
|
686
686
|
* Detection still runs where the rebuild cannot: a role that does not own the indexes, an adapter
|
|
687
687
|
* that wraps queries in a transaction, and `reindex: false` all still produce the `index_bloat`
|
|
688
688
|
* warning and can act on getReindexCommands(). The one exception is a backend that stores data
|
|
689
|
-
* outside PostgreSQL's heap
|
|
689
|
+
* outside PostgreSQL's heap. See the noReindex gate below.
|
|
690
690
|
*/
|
|
691
691
|
async #reindex(tables, options) {
|
|
692
692
|
if (this.#stopping)
|
|
@@ -696,7 +696,7 @@ class Boss extends EventEmitter {
|
|
|
696
696
|
// rejects `reltuples / relpages` outright ("unsupported binary operator: <float4> / <int4>"),
|
|
697
697
|
// so running the check would throw once per interval; YugabyteDB answers but reports relpages
|
|
698
698
|
// and pg_relation_size as 0 for every relation, so nothing could ever match. Both store data in
|
|
699
|
-
// an LSM that compacts on its own, and both reject REINDEX in either form
|
|
699
|
+
// an LSM that compacts on its own, and both reject REINDEX in either form. CockroachDB with
|
|
700
700
|
// the hint "CockroachDB does not require reindexing."
|
|
701
701
|
if (this.#config.noReindex)
|
|
702
702
|
return;
|
|
@@ -706,7 +706,7 @@ class Boss extends EventEmitter {
|
|
|
706
706
|
// An explicit force is a request to run now; everything else waits for an interval.
|
|
707
707
|
//
|
|
708
708
|
// Which interval depends on whether this instance can do the work. The shared claim exists so
|
|
709
|
-
// exactly one instance in the cluster rebuilds per window
|
|
709
|
+
// exactly one instance in the cluster rebuilds per window. An instance that is only ever going
|
|
710
710
|
// to report bloat has no business taking it, or a peer configured to rebuild would find the
|
|
711
711
|
// window gone and skip the rebuild for a whole day. Detection-only passes throttle themselves
|
|
712
712
|
// locally instead, on the same interval.
|
|
@@ -753,7 +753,7 @@ class Boss extends EventEmitter {
|
|
|
753
753
|
if (code === IN_TRANSACTION_ERROR) {
|
|
754
754
|
this.#reindexUnavailable = err.message;
|
|
755
755
|
failed.set(target.name, this.#reindexUnavailable);
|
|
756
|
-
// Every remaining index would fail identically
|
|
756
|
+
// Every remaining index would fail identically. The transaction wrapper is a property of
|
|
757
757
|
// the adapter, not of this index.
|
|
758
758
|
break;
|
|
759
759
|
}
|
|
@@ -789,7 +789,7 @@ class Boss extends EventEmitter {
|
|
|
789
789
|
continue;
|
|
790
790
|
// Ownership first: an index the role cannot touch was never a candidate, so it has no entry in
|
|
791
791
|
// `failed` no matter why the pass stopped. #reindexUnavailable comes next and covers the
|
|
792
|
-
// indexes the 25001 giveup skipped without attempting
|
|
792
|
+
// indexes the 25001 giveup skipped without attempting. They are neither failed nor rebuilt,
|
|
793
793
|
// and reporting a size cap they are nowhere near would point at the wrong knob.
|
|
794
794
|
const reason = failed.get(index.name) ??
|
|
795
795
|
(!index.owned
|
|
@@ -805,11 +805,11 @@ class Boss extends EventEmitter {
|
|
|
805
805
|
/**
|
|
806
806
|
* The REINDEX statements this instance would run, for installations where it cannot run them
|
|
807
807
|
* itself. Unlike the background pass this applies no ownership filter and no size cap unless one
|
|
808
|
-
* is passed
|
|
808
|
+
* is passed. The commands are for an operator, who may run them as a different role.
|
|
809
809
|
*/
|
|
810
810
|
async getReindexCommands(options) {
|
|
811
811
|
// The catalog query reads pg_class.relpages and pg_relation_size(), which the heap-less engines
|
|
812
|
-
// either reject outright or answer with zeroes
|
|
812
|
+
// either reject outright or answer with zeroes, same gate as #reindex, and there is nothing to
|
|
813
813
|
// rebuild on them anyway.
|
|
814
814
|
if (this.#config.noReindex)
|
|
815
815
|
return [];
|
package/dist/claimTimer.d.ts
CHANGED
|
@@ -25,7 +25,8 @@ import type { Clock } from './types.ts';
|
|
|
25
25
|
* is nothing for a skew correction to correct, which is why the fix is in when the attempt is made
|
|
26
26
|
* rather than in what it is measured against.
|
|
27
27
|
*
|
|
28
|
-
* `anchor()` is what a pass calls once its claim has been stamped
|
|
28
|
+
* `anchor()` is what a pass calls once its claim has been stamped, and it takes a wait of its own
|
|
29
|
+
* for the pass that knows when the row it was refused by comes due. A pass that returns without ever
|
|
29
30
|
* reaching its claim - stopped, already working, an error on the way in - is re-armed from the end
|
|
30
31
|
* of the callback instead, so the chain cannot die on a path that never anchored it.
|
|
31
32
|
*/
|
|
@@ -36,9 +37,15 @@ export declare class ClaimTimer {
|
|
|
36
37
|
stop(): void;
|
|
37
38
|
/**
|
|
38
39
|
* Re-anchors the next attempt to now, because the claim this timer drives has just been stamped.
|
|
39
|
-
*
|
|
40
|
-
*
|
|
40
|
+
*
|
|
41
|
+
* Called whether the claim was won or lost. A winner needs its next attempt to fall an interval
|
|
42
|
+
* after its own stamp. A loser has two phases it could take, and which one it takes is not a
|
|
43
|
+
* matter of taste: measured from its own failure it comes back an interval later, which is up to
|
|
44
|
+
* a whole interval after the row is next due, and that is the deployment's spacing the moment the
|
|
45
|
+
* instance holding the claim stops. A caller that knows when the row comes due passes that wait
|
|
46
|
+
* in `ms` instead. See Timekeeper.onCron(), which is the one claim where the difference costs
|
|
47
|
+
* work rather than latency.
|
|
41
48
|
*/
|
|
42
|
-
anchor(): void;
|
|
49
|
+
anchor(ms?: number): void;
|
|
43
50
|
}
|
|
44
51
|
//# sourceMappingURL=claimTimer.d.ts.map
|
package/dist/claimTimer.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"claimTimer.d.ts","sourceRoot":"","sources":["../src/claimTimer.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,KAAK,EAAc,MAAM,YAAY,CAAA;AAEnD
|
|
1
|
+
{"version":3,"file":"claimTimer.d.ts","sourceRoot":"","sources":["../src/claimTimer.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,KAAK,EAAc,MAAM,YAAY,CAAA;AAEnD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA8BG;AACH,qBAAa,UAAU;;gBASR,KAAK,EAAE,KAAK,EAAE,OAAO,EAAE,MAAM,EAAE,EAAE,EAAE,MAAM,OAAO,CAAC,IAAI,CAAC;IAMnE,KAAK,IAAK,IAAI;IAOd,IAAI,IAAK,IAAI;IAKb;;;;;;;;;;OAUG;IACH,MAAM,CAAE,EAAE,GAAE,MAAiB,GAAG,IAAI;CAwCrC"}
|
package/dist/claimTimer.js
CHANGED
|
@@ -24,7 +24,8 @@
|
|
|
24
24
|
* is nothing for a skew correction to correct, which is why the fix is in when the attempt is made
|
|
25
25
|
* rather than in what it is measured against.
|
|
26
26
|
*
|
|
27
|
-
* `anchor()` is what a pass calls once its claim has been stamped
|
|
27
|
+
* `anchor()` is what a pass calls once its claim has been stamped, and it takes a wait of its own
|
|
28
|
+
* for the pass that knows when the row it was refused by comes due. A pass that returns without ever
|
|
28
29
|
* reaching its claim - stopped, already working, an error on the way in - is re-armed from the end
|
|
29
30
|
* of the callback instead, so the chain cannot die on a path that never anchored it.
|
|
30
31
|
*/
|
|
@@ -52,20 +53,26 @@ export class ClaimTimer {
|
|
|
52
53
|
}
|
|
53
54
|
/**
|
|
54
55
|
* Re-anchors the next attempt to now, because the claim this timer drives has just been stamped.
|
|
55
|
-
*
|
|
56
|
-
*
|
|
56
|
+
*
|
|
57
|
+
* Called whether the claim was won or lost. A winner needs its next attempt to fall an interval
|
|
58
|
+
* after its own stamp. A loser has two phases it could take, and which one it takes is not a
|
|
59
|
+
* matter of taste: measured from its own failure it comes back an interval later, which is up to
|
|
60
|
+
* a whole interval after the row is next due, and that is the deployment's spacing the moment the
|
|
61
|
+
* instance holding the claim stops. A caller that knows when the row comes due passes that wait
|
|
62
|
+
* in `ms` instead. See Timekeeper.onCron(), which is the one claim where the difference costs
|
|
63
|
+
* work rather than latency.
|
|
57
64
|
*/
|
|
58
|
-
anchor() {
|
|
65
|
+
anchor(ms = this.#ms) {
|
|
59
66
|
if (this.#stopped)
|
|
60
67
|
return;
|
|
61
68
|
this.#anchored = true;
|
|
62
|
-
this.#arm();
|
|
69
|
+
this.#arm(ms);
|
|
63
70
|
}
|
|
64
|
-
#arm() {
|
|
71
|
+
#arm(ms = this.#ms) {
|
|
65
72
|
this.#disarm();
|
|
66
73
|
if (this.#stopped)
|
|
67
74
|
return;
|
|
68
|
-
this.#handle = this.#clock.setTimeout(() => { this.#run(); },
|
|
75
|
+
this.#handle = this.#clock.setTimeout(() => { this.#run(); }, ms);
|
|
69
76
|
}
|
|
70
77
|
#disarm() {
|
|
71
78
|
if (this.#handle !== undefined) {
|
package/dist/cli.js
CHANGED
|
@@ -193,7 +193,7 @@ async function createDb(config) {
|
|
|
193
193
|
return db;
|
|
194
194
|
}
|
|
195
195
|
// Like getConnectionConfig, but returns null instead of exiting when no connection is
|
|
196
|
-
// configured
|
|
196
|
+
// configured, used by commands (e.g. `plans`) where a connection is optional.
|
|
197
197
|
function tryGetConnectionConfig(args) {
|
|
198
198
|
const fileConfig = loadConfigFile(args.config);
|
|
199
199
|
const hasConnection = args.connectionString || process.env.PGBOSS_DATABASE_URL || fileConfig.connectionString ||
|
|
@@ -303,7 +303,7 @@ async function cmdMigrate(args) {
|
|
|
303
303
|
}
|
|
304
304
|
// Render from the DB's actual version so the printed SQL is exactly what `migrate` would run.
|
|
305
305
|
// Offline (or not yet installed) we can't know it, so fall back to the oldest supported starting
|
|
306
|
-
// version
|
|
306
|
+
// version (the full chain) instead of a bogus "from 0" that fails on non-idempotent steps.
|
|
307
307
|
const fromVersion = version ?? migrationStore.getMinVersion(schema);
|
|
308
308
|
const sql = migrationStore.migrate(schema, fromVersion, migrationStore.getAllForConfig(config), config.noAdvisoryLocks, { inlineAsync: true, partitionTables });
|
|
309
309
|
console.log(`-- SQL to migrate pg-boss from version ${fromVersion} to ${schemaVersion}:`);
|
|
@@ -382,7 +382,7 @@ async function cmdDoctor(args) {
|
|
|
382
382
|
}
|
|
383
383
|
console.log(`Schema "${schema}" version ${version} (latest: ${schemaVersion})`);
|
|
384
384
|
if (version < schemaVersion) {
|
|
385
|
-
console.log(`⚠ Migrations pending: ${schemaVersion - version}
|
|
385
|
+
console.log(`⚠ Migrations pending: ${schemaVersion - version}. Run "pg-boss migrate" before trusting drift results`);
|
|
386
386
|
}
|
|
387
387
|
// Reuse the same drift scan the boss.detectSchemaDrift() API runs, rather than duplicating the
|
|
388
388
|
// catalog queries + computeSchemaDrift wiring here (the two copies had already drifted apart and
|
|
@@ -404,7 +404,7 @@ async function cmdDoctor(args) {
|
|
|
404
404
|
await contractor.detectClockOverride();
|
|
405
405
|
if (override) {
|
|
406
406
|
console.log('\njob_now() carries a TestClock override, left behind by a test run that was');
|
|
407
|
-
console.log('killed before releasing its clock. Restoring it
|
|
407
|
+
console.log('killed before releasing its clock. Restoring it, so make sure no instance is');
|
|
408
408
|
console.log('holding a live TestClock against this schema.');
|
|
409
409
|
await contractor.restoreClockFunction();
|
|
410
410
|
console.log(' restored.');
|
|
@@ -416,7 +416,7 @@ async function cmdDoctor(args) {
|
|
|
416
416
|
}
|
|
417
417
|
}
|
|
418
418
|
if (report.building.length) {
|
|
419
|
-
console.log(`\nBuilding (async index build in progress
|
|
419
|
+
console.log(`\nBuilding (async index build in progress, not yet drift) (${report.building.length}):`);
|
|
420
420
|
for (const i of report.building)
|
|
421
421
|
console.log(` ${i.table}.${i.name}`);
|
|
422
422
|
}
|
|
@@ -429,19 +429,19 @@ async function cmdDoctor(args) {
|
|
|
429
429
|
console.log(`\n⚠ INDEX BLOAT (rebuild with "pg-boss reindex") (${bloated.length}):`);
|
|
430
430
|
for (const i of bloated) {
|
|
431
431
|
const mb = Math.round(Number(i.bytes) / 1024 / 1024);
|
|
432
|
-
console.log(` ${i.table}.${i.name}
|
|
432
|
+
console.log(` ${i.table}.${i.name}: ${mb} MB across ${i.pages} pages, ~${i.entries} live entries`);
|
|
433
433
|
if (!i.owned)
|
|
434
434
|
console.log(' note: the connected role does not own this index and cannot reindex it');
|
|
435
435
|
}
|
|
436
436
|
}
|
|
437
437
|
}
|
|
438
438
|
catch {
|
|
439
|
-
// Best effort
|
|
439
|
+
// Best effort. A role without catalog visibility should not fail the whole drift report.
|
|
440
440
|
}
|
|
441
|
-
// Extra indexes are informational (a stale pg-boss index or a user-added one)
|
|
441
|
+
// Extra indexes are informational (a stale pg-boss index or a user-added one). A warning, not
|
|
442
442
|
// drift. Printed regardless of overall status; it never changes the exit code.
|
|
443
443
|
if (report.extraIndexes.length) {
|
|
444
|
-
console.log(`\n⚠ EXTRA INDEXES (present on a managed table but not expected
|
|
444
|
+
console.log(`\n⚠ EXTRA INDEXES (present on a managed table but not expected, harmless) (${report.extraIndexes.length}):`);
|
|
445
445
|
for (const i of report.extraIndexes)
|
|
446
446
|
console.log(` ${i.table}.${i.name}`);
|
|
447
447
|
}
|
|
@@ -465,9 +465,10 @@ async function cmdDoctor(args) {
|
|
|
465
465
|
}
|
|
466
466
|
}
|
|
467
467
|
if (report.invalid.length) {
|
|
468
|
-
// The definition is correct
|
|
469
|
-
// drop and rebuild it, not an expected-vs-actual comparison
|
|
470
|
-
|
|
468
|
+
// The definition is correct, since an invalid index is an interrupted build rather than a
|
|
469
|
+
// wrong shape. So show the DDL to drop and rebuild it, not an expected-vs-actual comparison,
|
|
470
|
+
// which would be identical.
|
|
471
|
+
console.log(`\nINVALID (present but marked invalid, drop and rebuild) (${report.invalid.length}):`);
|
|
471
472
|
for (const i of report.invalid) {
|
|
472
473
|
console.log(` ${i.table}.${i.name}`);
|
|
473
474
|
if (i.definition)
|
|
@@ -596,7 +597,7 @@ async function cmdReindex(args) {
|
|
|
596
597
|
// The check reads pg_class.relpages and pg_relation_size(). CockroachDB has neither (it
|
|
597
598
|
// rejects `reltuples / relpages` as an unsupported binary operator) and YugabyteDB reports
|
|
598
599
|
// zeroes for every relation. --backend catches those up front (above); this stays for a target
|
|
599
|
-
// that was not declared
|
|
600
|
+
// that was not declared, say what happened instead of surfacing a raw catalog error.
|
|
600
601
|
console.error(`Could not read index statistics from schema "${schema}": ${err.message}`);
|
|
601
602
|
console.error('The bloat check reads pg_class.relpages and pg_relation_size(), which CockroachDB and YugabyteDB do not provide.');
|
|
602
603
|
process.exitCode = 1;
|
|
@@ -625,7 +626,7 @@ async function cmdReindex(args) {
|
|
|
625
626
|
}
|
|
626
627
|
for (const index of rows) {
|
|
627
628
|
const mb = Math.round(Number(index.bytes) / 1024 / 1024);
|
|
628
|
-
console.log(` ${index.table}.${index.name}
|
|
629
|
+
console.log(` ${index.table}.${index.name}: ${mb} MB across ${index.pages} pages, ~${index.entries} live entries`);
|
|
629
630
|
}
|
|
630
631
|
// Executing is different from printing: REINDEX needs ownership, so an index this role cannot
|
|
631
632
|
// touch is reported up front rather than attempted for the sake of the server's error message.
|
package/dist/contractor.js
CHANGED
|
@@ -140,7 +140,7 @@ class Contractor {
|
|
|
140
140
|
// Function-body and enum drift are best-effort: pg_get_functiondef is unsupported on some backends
|
|
141
141
|
// (CockroachDB), so a failure here SKIPS the function check rather than aborting the whole scan.
|
|
142
142
|
// `functionsSupported` must be tracked separately from an empty result: an empty `liveFunctions`
|
|
143
|
-
// means "query failed / unsupported", which is NOT the same as "no functions found"
|
|
143
|
+
// means "query failed / unsupported", which is NOT the same as "no functions found", feeding
|
|
144
144
|
// `live: []` to the drift check would report every expected function as missing and flip `ok` to
|
|
145
145
|
// false on every CockroachDB scan. So the check is gated (passed `undefined`) when the query throws.
|
|
146
146
|
let liveFunctions = [];
|
|
@@ -162,7 +162,7 @@ class Contractor {
|
|
|
162
162
|
}
|
|
163
163
|
// Table presence is read from a catalog-only query independent of the column diff below. The column
|
|
164
164
|
// query uses pg_get_expr (unsupported on some backends) and is best-effort; if it throws, the column
|
|
165
|
-
// check is skipped
|
|
165
|
+
// check is skipped, but table presence must NOT collapse to "everything missing", so it comes from
|
|
166
166
|
// its own pg_class probe. pg_class is available everywhere, so this rarely throws; if it somehow
|
|
167
167
|
// does, fall back to the columns-derived set rather than aborting the scan.
|
|
168
168
|
let liveTables = null;
|
package/dist/db.js
CHANGED
|
@@ -114,7 +114,7 @@ class Db extends EventEmitter {
|
|
|
114
114
|
const keepAliveInitialDelay = this.config.notifyKeepAliveInitialDelayMs ?? DEFAULT_LISTEN_KEEP_ALIVE_INITIAL_DELAY_MS;
|
|
115
115
|
// Only self-heal once the listener has been established at least once. If the INITIAL connect
|
|
116
116
|
// fails, the rejection propagates to the caller (Notifier.start), which falls back to
|
|
117
|
-
// polling-only and discards this subscription's close handle
|
|
117
|
+
// polling-only and discards this subscription's close handle, so a reconnect scheduled from
|
|
118
118
|
// the client 'error' handler would be an untracked connection nothing can close, keeping the
|
|
119
119
|
// event loop alive and delivering notifications into a stopped manager.
|
|
120
120
|
let established = false;
|
|
@@ -198,7 +198,7 @@ class Db extends EventEmitter {
|
|
|
198
198
|
});
|
|
199
199
|
// Track the client before connecting so close() can tear down a connect still in flight
|
|
200
200
|
// (e.g. shutdown during a reconnect). If connect or LISTEN then rejects, the catch ends
|
|
201
|
-
// it and rethrows
|
|
201
|
+
// it and rethrows. Without that, a LISTEN that fails after connect() succeeded would
|
|
202
202
|
// leak an open connection. The reconnect .catch below reschedules on failure; an initial
|
|
203
203
|
// failure propagates to the caller.
|
|
204
204
|
client = next;
|