pg-boss 12.35.0 → 12.36.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/dist/attorney.d.ts.map +1 -1
- package/dist/attorney.js +21 -0
- package/dist/boss.d.ts.map +1 -1
- package/dist/boss.js +35 -4
- package/dist/db.d.ts +1 -0
- package/dist/db.d.ts.map +1 -1
- package/dist/db.js +11 -0
- package/dist/index.d.ts +3 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +23 -5
- package/dist/latency.d.ts +15 -0
- package/dist/latency.d.ts.map +1 -0
- package/dist/latency.js +45 -0
- package/dist/manager.d.ts +2 -1
- package/dist/manager.d.ts.map +1 -1
- package/dist/manager.js +216 -73
- package/dist/migrationStore.d.ts.map +1 -1
- package/dist/migrationStore.js +67 -0
- package/dist/nurse.d.ts +20 -0
- package/dist/nurse.d.ts.map +1 -0
- package/dist/nurse.js +287 -0
- package/dist/plans.d.ts +20 -0
- package/dist/plans.d.ts.map +1 -1
- package/dist/plans.js +500 -50
- package/dist/registrar.d.ts +16 -0
- package/dist/registrar.d.ts.map +1 -0
- package/dist/registrar.js +221 -0
- package/dist/schema.json +480 -1
- package/dist/telemetry.d.ts +65 -0
- package/dist/telemetry.d.ts.map +1 -0
- package/dist/telemetry.js +349 -0
- package/dist/types.d.ts +201 -3
- package/dist/types.d.ts.map +1 -1
- package/package.json +17 -8
package/dist/plans.js
CHANGED
|
@@ -159,6 +159,7 @@ export function create(schema, version, options) {
|
|
|
159
159
|
noPartitioning ? '' : ensureQueueStatsPartitions(schema),
|
|
160
160
|
createTableJobDependency(schema),
|
|
161
161
|
createIndexJobDependencyParent(schema),
|
|
162
|
+
createTableInstance(schema),
|
|
162
163
|
createQueueFunction(schema, noPartitioning),
|
|
163
164
|
deleteQueueFunction(schema, noPartitioning),
|
|
164
165
|
insertVersion(schema, version)
|
|
@@ -184,6 +185,7 @@ function createInline(schema, version, options) {
|
|
|
184
185
|
inlineIntoCreateTable(createTableWarning(schema), [createIndexWarning(schema)]),
|
|
185
186
|
inlineIntoCreateTable(createTableQueueStats(schema, true), [createIndexQueueStats(schema, options.noCovering)]),
|
|
186
187
|
inlineIntoCreateTable(createTableJobDependency(schema), [createIndexJobDependencyParent(schema)]),
|
|
188
|
+
createTableInstance(schema),
|
|
187
189
|
createQueueFunction(schema, true),
|
|
188
190
|
deleteQueueFunction(schema, true),
|
|
189
191
|
insertVersion(schema, version)
|
|
@@ -304,7 +306,11 @@ function createTableVersion(schema) {
|
|
|
304
306
|
// try-lock - without capturedOn claiming a freshness the counts do not have.
|
|
305
307
|
// created_delta / completed_delta / failed_delta are not gauges like the counts beside them: they
|
|
306
308
|
// are how many jobs went through between the previous monitor pass and the latest one. delta_on and
|
|
307
|
-
// delta_seconds are the window those three cover, null until a pass counts.
|
|
309
|
+
// delta_seconds are the window those three cover, null until a pass counts. wait_bins and run_bins
|
|
310
|
+
// are the wait and run times of the jobs that finished in that window, as histograms of LATENCY_SLOTS
|
|
311
|
+
// counts (see LATENCY_BINS), and ready_oldest_seconds is how long the oldest job ready to run had
|
|
312
|
+
// waited at the pass. Like the deltas, all of them are null until a pass counts, and a pass that
|
|
313
|
+
// counts writes a value: every slot null when nothing finished, 0 when nothing was waiting.
|
|
308
314
|
/* eslint-disable no-restricted-syntax -- column defaults stay on the real clock: every pg-boss write names its timestamps through job_now() */
|
|
309
315
|
function createTableQueue(schema) {
|
|
310
316
|
return `
|
|
@@ -322,6 +328,7 @@ function createTableQueue(schema) {
|
|
|
322
328
|
partition bool NOT NULL,
|
|
323
329
|
table_name text NOT NULL,
|
|
324
330
|
deferred_count int NOT NULL default 0,
|
|
331
|
+
blocked_count int NOT NULL default 0,
|
|
325
332
|
queued_count int NOT NULL default 0,
|
|
326
333
|
ready_count int NOT NULL default 0,
|
|
327
334
|
warning_queued int NOT NULL default 0,
|
|
@@ -333,6 +340,9 @@ function createTableQueue(schema) {
|
|
|
333
340
|
failed_delta int NOT NULL default 0,
|
|
334
341
|
delta_on timestamp with time zone,
|
|
335
342
|
delta_seconds int,
|
|
343
|
+
wait_bins int[],
|
|
344
|
+
run_bins int[],
|
|
345
|
+
ready_oldest_seconds int,
|
|
336
346
|
ready_history int[] NOT NULL default '{}',
|
|
337
347
|
heartbeat_seconds int,
|
|
338
348
|
notify bool NOT NULL DEFAULT false,
|
|
@@ -435,9 +445,269 @@ export function createTableJobDependency(schema) {
|
|
|
435
445
|
)
|
|
436
446
|
`;
|
|
437
447
|
}
|
|
448
|
+
// An instance is live until it misses this many heartbeats, and its row is deleted once its heartbeat
|
|
449
|
+
// has not moved for this many days.
|
|
450
|
+
export const INSTANCE_QUIET_BEATS = 3;
|
|
451
|
+
export const INSTANCE_RETENTION_DAYS = 7;
|
|
452
|
+
// A crash never sets stopped_on, so a crash loop leaves a quiet row per restart, and a deploy whose
|
|
453
|
+
// processes never call stop() leaves one per replica. Registering keeps the newest of these dead rows
|
|
454
|
+
// for its own name (for its host, when unnamed), and maintenance keeps the newest overall, so neither
|
|
455
|
+
// can grow the table for the whole retention. Keyed on the name rather than the host because a
|
|
456
|
+
// Kubernetes pod that is recreated comes back under a new hostname.
|
|
457
|
+
export const INSTANCE_DEAD_KEPT_PER_NAME = 20;
|
|
458
|
+
export const INSTANCE_DEAD_KEPT = 1000;
|
|
459
|
+
// Stopped, or quiet: the complement of getInstances' live.
|
|
460
|
+
function instanceDead(schema) {
|
|
461
|
+
return `(stopped_on IS NOT NULL OR heartbeat_on < ${schema}.job_now() - heartbeat_seconds * ${INSTANCE_QUIET_BEATS} * interval '1 second')`;
|
|
462
|
+
}
|
|
463
|
+
// Deletes the dead rows ranked past `keep`, newest start first, among those `where` selects. Ranked
|
|
464
|
+
// before locking and locked with SKIP LOCKED, so a registration and maintenance pruning at once
|
|
465
|
+
// cannot deadlock, and neither deletes a row the other's ranking kept. Used on every backend, unlike
|
|
466
|
+
// the fetch's SKIP LOCKED: where it can pass over an unlocked row (CockroachDB), that row is only
|
|
467
|
+
// left for the next prune.
|
|
468
|
+
function pruneDeadInstances(schema, where, keep) {
|
|
469
|
+
return `
|
|
470
|
+
DELETE FROM ${schema}.instance
|
|
471
|
+
WHERE id IN (
|
|
472
|
+
SELECT id FROM ${schema}.instance
|
|
473
|
+
WHERE id IN (
|
|
474
|
+
SELECT id FROM (
|
|
475
|
+
SELECT id, row_number() OVER (ORDER BY started_on DESC, id DESC) as n
|
|
476
|
+
FROM ${schema}.instance
|
|
477
|
+
WHERE ${where} AND ${instanceDead(schema)}
|
|
478
|
+
) ranked
|
|
479
|
+
WHERE n > ${keep}
|
|
480
|
+
)
|
|
481
|
+
FOR UPDATE SKIP LOCKED
|
|
482
|
+
)
|
|
483
|
+
`;
|
|
484
|
+
}
|
|
485
|
+
// Run by an instance after it registers: $1 its id, $2 its name, $3 its host.
|
|
486
|
+
export function pruneInstanceLives(schema, keep) {
|
|
487
|
+
return pruneDeadInstances(schema, 'id <> $1 AND (CASE WHEN $2::text IS NULL THEN name IS NULL AND host = $3 ELSE name = $2 END)', keep);
|
|
488
|
+
}
|
|
489
|
+
export function trimDeadInstances(schema, keep) {
|
|
490
|
+
return pruneDeadInstances(schema, 'true', keep);
|
|
491
|
+
}
|
|
492
|
+
// Crash restarts: how many lives in a row on this name and host ended without stop() before this one
|
|
493
|
+
// started. Worked out from the rows rather than carried from one life to the next, because at
|
|
494
|
+
// registration a row that still reads live is either a sibling process (pm2, a second PgBoss) or a
|
|
495
|
+
// predecessor that crashed seconds ago, and only its next missed heartbeats tell which. So a crashed
|
|
496
|
+
// life counts toward an instance only once it is dead, and only if its last heartbeat came before the
|
|
497
|
+
// instance started, which a live sibling's keeps moving past.
|
|
498
|
+
//
|
|
499
|
+
// $1 this instance's id, $2 its name, $3 its host, $4 its pid, $5 when its process started: a row
|
|
500
|
+
// with this pid whose heartbeat predates that is an earlier process that reused the pid, dead at once
|
|
501
|
+
// (a container restart, pid 1 each time), and a row with this pid started since is another PgBoss
|
|
502
|
+
// object in this same process, never a predecessor.
|
|
503
|
+
function crashSlot(schema) {
|
|
504
|
+
return `
|
|
505
|
+
me AS (SELECT id, started_on FROM ${schema}.instance WHERE id = $1),
|
|
506
|
+
slot AS (
|
|
507
|
+
SELECT i.*
|
|
508
|
+
FROM ${schema}.instance i, me
|
|
509
|
+
WHERE i.id <> me.id
|
|
510
|
+
AND i.host = $3
|
|
511
|
+
AND (i.name = $2 OR (i.name IS NULL AND $2::text IS NULL))
|
|
512
|
+
AND i.started_on < me.started_on
|
|
513
|
+
AND i.heartbeat_on <= me.started_on
|
|
514
|
+
AND NOT (i.pid = $4 AND i.started_on >= $5::timestamptz)
|
|
515
|
+
),
|
|
516
|
+
boundary AS (SELECT max(started_on) as at FROM slot WHERE stopped_on IS NOT NULL),
|
|
517
|
+
streak AS (
|
|
518
|
+
SELECT s.*
|
|
519
|
+
FROM slot s, boundary b
|
|
520
|
+
WHERE s.stopped_on IS NULL
|
|
521
|
+
AND (b.at IS NULL OR s.started_on > b.at)
|
|
522
|
+
)`;
|
|
523
|
+
}
|
|
524
|
+
function crashDead(schema) {
|
|
525
|
+
return `(heartbeat_on < ${schema}.job_now() - heartbeat_seconds * ${INSTANCE_QUIET_BEATS} * interval '1 second'
|
|
526
|
+
OR (pid = $4 AND heartbeat_on < $5::timestamptz))`;
|
|
527
|
+
}
|
|
528
|
+
// Numbers the dead lives of the current streak from the earliest one still kept, whose count stands
|
|
529
|
+
// for any the pruning has taken, and writes each later life's count and this instance's. Rewriting
|
|
530
|
+
// the streak is what keeps the counts exact: a life that crashed before its own recheck is corrected
|
|
531
|
+
// by its successor, so the earliest kept row is always right when the pruning moves past it.
|
|
532
|
+
export function countCrashRestarts(schema) {
|
|
533
|
+
return `
|
|
534
|
+
WITH ${crashSlot(schema)},
|
|
535
|
+
crashed AS (
|
|
536
|
+
SELECT id, crash_restarts, crash_restarts_since, heartbeat_on,
|
|
537
|
+
row_number() OVER (ORDER BY started_on, id) as n,
|
|
538
|
+
count(*) OVER () as k
|
|
539
|
+
FROM streak
|
|
540
|
+
WHERE ${crashDead(schema)}
|
|
541
|
+
),
|
|
542
|
+
head AS (
|
|
543
|
+
SELECT crash_restarts as base, COALESCE(crash_restarts_since, heartbeat_on) as since, k
|
|
544
|
+
FROM crashed WHERE n = 1
|
|
545
|
+
),
|
|
546
|
+
counts AS (
|
|
547
|
+
SELECT c.id, h.base + c.n - 1 as restarts, h.since
|
|
548
|
+
FROM crashed c, head h
|
|
549
|
+
WHERE c.n > 1
|
|
550
|
+
UNION ALL
|
|
551
|
+
SELECT me.id, COALESCE(h.base + h.k, 0), h.since
|
|
552
|
+
FROM me LEFT JOIN head h ON true
|
|
553
|
+
)
|
|
554
|
+
UPDATE ${schema}.instance i
|
|
555
|
+
SET crash_restarts = counts.restarts::int,
|
|
556
|
+
crash_restarts_since = counts.since
|
|
557
|
+
FROM counts
|
|
558
|
+
WHERE i.id = counts.id
|
|
559
|
+
`;
|
|
560
|
+
}
|
|
561
|
+
// When the rows this count could not yet judge would go quiet, or null when there are none: the
|
|
562
|
+
// registrar counts again then.
|
|
563
|
+
export function crashRecountAt(schema) {
|
|
564
|
+
return `
|
|
565
|
+
WITH ${crashSlot(schema)}
|
|
566
|
+
SELECT max(heartbeat_on + heartbeat_seconds * ${INSTANCE_QUIET_BEATS} * interval '1 second') as "recountAt"
|
|
567
|
+
FROM streak
|
|
568
|
+
WHERE NOT ${crashDead(schema)}
|
|
569
|
+
`;
|
|
570
|
+
}
|
|
571
|
+
// Instance registry statements. $1 is the instance id throughout. Registering replaces the row, so a
|
|
572
|
+
// PgBoss object restarted after stop() reads as live again from a new started_on. A heartbeat is an
|
|
573
|
+
// upsert too: a row pruned while its process was paused (a laptop asleep past the retention) comes
|
|
574
|
+
// back on the next beat, keeping the started_on the instance registered with ($20).
|
|
575
|
+
const INSTANCE_COLUMNS = `id, name, host, pid, version, node_version, heartbeat_seconds,
|
|
576
|
+
supervise, schedule, migrate, persist_queue_stats, persist_warnings,
|
|
577
|
+
pool_max, pool_total, pool_idle, pool_waiting, workers, metrics, config, application_name, started_on, heartbeat_on`;
|
|
578
|
+
const INSTANCE_VALUES = `$1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17::text::jsonb, $18::text::jsonb, $19::text::jsonb,
|
|
579
|
+
current_setting('application_name')`;
|
|
580
|
+
const INSTANCE_BEAT = `pool_max = EXCLUDED.pool_max,
|
|
581
|
+
pool_total = EXCLUDED.pool_total,
|
|
582
|
+
pool_idle = EXCLUDED.pool_idle,
|
|
583
|
+
pool_waiting = EXCLUDED.pool_waiting,
|
|
584
|
+
workers = EXCLUDED.workers,
|
|
585
|
+
metrics = EXCLUDED.metrics,
|
|
586
|
+
heartbeat_on = EXCLUDED.heartbeat_on`;
|
|
587
|
+
export function registerInstance(schema) {
|
|
588
|
+
return `
|
|
589
|
+
INSERT INTO ${schema}.instance (${INSTANCE_COLUMNS})
|
|
590
|
+
VALUES (${INSTANCE_VALUES}, ${schema}.job_now(), ${schema}.job_now())
|
|
591
|
+
ON CONFLICT (id) DO UPDATE SET
|
|
592
|
+
name = EXCLUDED.name,
|
|
593
|
+
host = EXCLUDED.host,
|
|
594
|
+
pid = EXCLUDED.pid,
|
|
595
|
+
version = EXCLUDED.version,
|
|
596
|
+
node_version = EXCLUDED.node_version,
|
|
597
|
+
heartbeat_seconds = EXCLUDED.heartbeat_seconds,
|
|
598
|
+
supervise = EXCLUDED.supervise,
|
|
599
|
+
schedule = EXCLUDED.schedule,
|
|
600
|
+
migrate = EXCLUDED.migrate,
|
|
601
|
+
persist_queue_stats = EXCLUDED.persist_queue_stats,
|
|
602
|
+
persist_warnings = EXCLUDED.persist_warnings,
|
|
603
|
+
application_name = EXCLUDED.application_name,
|
|
604
|
+
config = EXCLUDED.config,
|
|
605
|
+
started_on = EXCLUDED.started_on,
|
|
606
|
+
stopped_on = NULL,
|
|
607
|
+
crash_restarts = 0,
|
|
608
|
+
crash_restarts_since = NULL,
|
|
609
|
+
${INSTANCE_BEAT}
|
|
610
|
+
RETURNING started_on as "startedOn"
|
|
611
|
+
`;
|
|
612
|
+
}
|
|
613
|
+
export function heartbeatInstance(schema) {
|
|
614
|
+
return `
|
|
615
|
+
INSERT INTO ${schema}.instance (${INSTANCE_COLUMNS})
|
|
616
|
+
VALUES (${INSTANCE_VALUES}, $20::timestamptz, ${schema}.job_now())
|
|
617
|
+
ON CONFLICT (id) DO UPDATE SET
|
|
618
|
+
${INSTANCE_BEAT}
|
|
619
|
+
`;
|
|
620
|
+
}
|
|
621
|
+
export function stopInstance(schema) {
|
|
622
|
+
return `UPDATE ${schema}.instance SET stopped_on = ${schema}.job_now() WHERE id = $1`;
|
|
623
|
+
}
|
|
624
|
+
// Rows whose heartbeat has not moved for the retention, stopped or quiet alike. A live instance's
|
|
625
|
+
// row never qualifies, since its heartbeat keeps moving.
|
|
626
|
+
export function deleteOldInstances(schema, days) {
|
|
627
|
+
return `
|
|
628
|
+
DELETE FROM ${schema}.instance
|
|
629
|
+
WHERE heartbeat_on < ${schema}.job_now() - interval '${days} days'
|
|
630
|
+
`;
|
|
631
|
+
}
|
|
632
|
+
export function getInstances(schema) {
|
|
633
|
+
return `
|
|
634
|
+
SELECT
|
|
635
|
+
id,
|
|
636
|
+
name,
|
|
637
|
+
host,
|
|
638
|
+
pid,
|
|
639
|
+
version,
|
|
640
|
+
node_version as "nodeVersion",
|
|
641
|
+
application_name as "applicationName",
|
|
642
|
+
heartbeat_seconds as "heartbeatSeconds",
|
|
643
|
+
supervise,
|
|
644
|
+
schedule,
|
|
645
|
+
migrate,
|
|
646
|
+
persist_queue_stats as "persistQueueStats",
|
|
647
|
+
persist_warnings as "persistWarnings",
|
|
648
|
+
pool_max as "poolMax",
|
|
649
|
+
pool_total as "poolTotal",
|
|
650
|
+
pool_idle as "poolIdle",
|
|
651
|
+
pool_waiting as "poolWaiting",
|
|
652
|
+
workers,
|
|
653
|
+
metrics,
|
|
654
|
+
config,
|
|
655
|
+
crash_restarts as "crashRestarts",
|
|
656
|
+
crash_restarts_since as "crashRestartsSince",
|
|
657
|
+
started_on as "startedOn",
|
|
658
|
+
heartbeat_on as "heartbeatOn",
|
|
659
|
+
stopped_on as "stoppedOn",
|
|
660
|
+
stopped_on IS NULL AND heartbeat_on >= ${schema}.job_now() - heartbeat_seconds * ${INSTANCE_QUIET_BEATS} * interval '1 second' as live
|
|
661
|
+
FROM ${schema}.instance
|
|
662
|
+
ORDER BY started_on, id
|
|
663
|
+
`;
|
|
664
|
+
}
|
|
438
665
|
export function createIndexJobDependencyParent(schema) {
|
|
439
666
|
return `CREATE INDEX IF NOT EXISTS job_dep_parent_idx ON ${schema}.job_dependency (parent_name, parent_id)`;
|
|
440
667
|
}
|
|
668
|
+
/* eslint-disable no-restricted-syntax -- column defaults stay on the real clock: every pg-boss write names its timestamps through job_now() */
|
|
669
|
+
// One row per PgBoss object, written by the object itself at start() and on each heartbeat, so a
|
|
670
|
+
// database can say which instances share it and what each is doing. A crashed process never sets
|
|
671
|
+
// stopped_on; it goes quiet instead, when heartbeat_on stops moving, and heartbeat_seconds says how
|
|
672
|
+
// long that takes for this row, since the interval is per instance. workers is one entry per work()
|
|
673
|
+
// call. The pool columns are node-postgres's counts at the last heartbeat, null for a pool pg-boss
|
|
674
|
+
// was handed and cannot read. metrics is the process's CPU, memory and event loop at the last
|
|
675
|
+
// heartbeat against its container's limits (see nurse.ts), null until the first sample lands.
|
|
676
|
+
// config is the options it runs with, as the registrar's allowlist picks them, for comparing instances.
|
|
677
|
+
// application_name is the one the registering session carried, so pg_stat_activity joins to the row
|
|
678
|
+
// wherever it is unique.
|
|
679
|
+
export function createTableInstance(schema) {
|
|
680
|
+
return `
|
|
681
|
+
CREATE TABLE ${schema}.instance (
|
|
682
|
+
id uuid PRIMARY KEY,
|
|
683
|
+
name text,
|
|
684
|
+
host text NOT NULL,
|
|
685
|
+
pid int NOT NULL,
|
|
686
|
+
version text NOT NULL,
|
|
687
|
+
node_version text NOT NULL,
|
|
688
|
+
application_name text,
|
|
689
|
+
heartbeat_seconds int NOT NULL,
|
|
690
|
+
supervise bool NOT NULL,
|
|
691
|
+
schedule bool NOT NULL,
|
|
692
|
+
migrate bool NOT NULL,
|
|
693
|
+
persist_queue_stats bool NOT NULL,
|
|
694
|
+
persist_warnings bool NOT NULL,
|
|
695
|
+
pool_max int,
|
|
696
|
+
pool_total int,
|
|
697
|
+
pool_idle int,
|
|
698
|
+
pool_waiting int,
|
|
699
|
+
workers jsonb NOT NULL DEFAULT '[]'::jsonb,
|
|
700
|
+
metrics jsonb,
|
|
701
|
+
config jsonb NOT NULL DEFAULT '{}'::jsonb,
|
|
702
|
+
crash_restarts int NOT NULL DEFAULT 0,
|
|
703
|
+
crash_restarts_since timestamptz,
|
|
704
|
+
started_on timestamptz NOT NULL DEFAULT now(),
|
|
705
|
+
heartbeat_on timestamptz NOT NULL DEFAULT now(),
|
|
706
|
+
stopped_on timestamptz
|
|
707
|
+
)
|
|
708
|
+
`;
|
|
709
|
+
}
|
|
710
|
+
/* eslint-enable no-restricted-syntax */
|
|
441
711
|
// Anchored so a schema name that itself contains these substrings (e.g. `job_intake`) isn't
|
|
442
712
|
// mangled: `\.job\y` matches only the base table reference (`schema.job`, not `schema.job_i5` whose
|
|
443
713
|
// `job` is followed by `_`, nor `.job_dependency`), and `\yjob_i(\d+)` matches only the bare
|
|
@@ -583,7 +853,8 @@ function createTableJob(schema, noPartitioning = false) {
|
|
|
583
853
|
source_created_on timestamp with time zone,
|
|
584
854
|
source_retry_count int,
|
|
585
855
|
source_output jsonb,
|
|
586
|
-
source_root_id uuid
|
|
856
|
+
source_root_id uuid,
|
|
857
|
+
trace_context jsonb
|
|
587
858
|
) ${partitionClause}
|
|
588
859
|
`;
|
|
589
860
|
}
|
|
@@ -1178,6 +1449,9 @@ export function updateQueue(schema) {
|
|
|
1178
1449
|
WHERE name = $1
|
|
1179
1450
|
`;
|
|
1180
1451
|
}
|
|
1452
|
+
export function currentDatabase() {
|
|
1453
|
+
return 'SELECT current_database() AS name';
|
|
1454
|
+
}
|
|
1181
1455
|
export function getQueues(schema, names) {
|
|
1182
1456
|
const hasNames = names && names.length > 0;
|
|
1183
1457
|
return {
|
|
@@ -1197,6 +1471,7 @@ export function getQueues(schema, names) {
|
|
|
1197
1471
|
q.notify,
|
|
1198
1472
|
q.dead_letter as "deadLetter",
|
|
1199
1473
|
q.deferred_count as "deferredCount",
|
|
1474
|
+
q.blocked_count as "blockedCount",
|
|
1200
1475
|
q.warning_queued as "warningQueueSize",
|
|
1201
1476
|
q.queued_count as "queuedCount",
|
|
1202
1477
|
q.ready_count as "readyCount",
|
|
@@ -1208,6 +1483,9 @@ export function getQueues(schema, names) {
|
|
|
1208
1483
|
q.failed_delta as "failedDelta",
|
|
1209
1484
|
q.delta_seconds as "deltaSeconds",
|
|
1210
1485
|
q.delta_on as "deltaOn",
|
|
1486
|
+
q.wait_bins as "waitBins",
|
|
1487
|
+
q.run_bins as "runBins",
|
|
1488
|
+
q.ready_oldest_seconds as "readyOldestSeconds",
|
|
1211
1489
|
q.singletons_active as "singletonsActive",
|
|
1212
1490
|
q.table_name as "table",
|
|
1213
1491
|
q.created_on as "createdOn",
|
|
@@ -1239,6 +1517,30 @@ export function deleteStoredJobs(schema, table) {
|
|
|
1239
1517
|
export function truncateTable(schema, table) {
|
|
1240
1518
|
return `TRUNCATE ${schema}.${table}`;
|
|
1241
1519
|
}
|
|
1520
|
+
// The cached counts of a queue whose table was just truncated, written without counting: they are
|
|
1521
|
+
// zero. Only the truncate paths use it. A DELETE leaves concurrent sends and the other states in
|
|
1522
|
+
// place, so its counts can only come from a recount, whose cost the delete cannot bound. The
|
|
1523
|
+
// throughput counters are the monitor's and are left alone. With `one`, $1 is the queue's name.
|
|
1524
|
+
export function zeroQueueStats(schema, one) {
|
|
1525
|
+
return `
|
|
1526
|
+
UPDATE ${schema}.queue SET
|
|
1527
|
+
deferred_count = 0,
|
|
1528
|
+
blocked_count = 0,
|
|
1529
|
+
queued_count = 0,
|
|
1530
|
+
ready_count = 0,
|
|
1531
|
+
active_count = 0,
|
|
1532
|
+
failed_count = 0,
|
|
1533
|
+
total_count = 0,
|
|
1534
|
+
singletons_active = NULL,
|
|
1535
|
+
monitor_on = ${schema}.job_now()
|
|
1536
|
+
FROM (
|
|
1537
|
+
SELECT name
|
|
1538
|
+
FROM ${schema}.queue${one ? '\n WHERE name = $1' : ''}
|
|
1539
|
+
${queueRowLock()}
|
|
1540
|
+
) q
|
|
1541
|
+
WHERE queue.name = q.name
|
|
1542
|
+
`;
|
|
1543
|
+
}
|
|
1242
1544
|
export function deleteAllJobs(schema, table) {
|
|
1243
1545
|
return `DELETE from ${schema}.${table} WHERE name = $1`;
|
|
1244
1546
|
}
|
|
@@ -1285,7 +1587,7 @@ export function setScheduleLastJobIds(schema) {
|
|
|
1285
1587
|
return `
|
|
1286
1588
|
UPDATE ${schema}.schedule s
|
|
1287
1589
|
SET last_job_id = x."jobId"
|
|
1288
|
-
FROM json_to_recordset($1::json) AS x (name text, key text, "jobId" uuid)
|
|
1590
|
+
FROM json_to_recordset($1::text::json) AS x (name text, key text, "jobId" uuid)
|
|
1289
1591
|
WHERE s.name = x.name
|
|
1290
1592
|
AND COALESCE(s.key, '') = x.key
|
|
1291
1593
|
`;
|
|
@@ -1305,7 +1607,7 @@ export function setScheduleLastJobIds(schema) {
|
|
|
1305
1607
|
export function setScheduleKinds(schema) {
|
|
1306
1608
|
return `
|
|
1307
1609
|
UPDATE ${schema}.schedule s SET kind = k.kind
|
|
1308
|
-
FROM json_to_recordset($1::json) as k (name text, key text, kind text, cron text)
|
|
1610
|
+
FROM json_to_recordset($1::text::json) as k (name text, key text, kind text, cron text)
|
|
1309
1611
|
WHERE s.name = k.name
|
|
1310
1612
|
AND COALESCE(s.key, '') = k.key
|
|
1311
1613
|
AND s.cron = k.cron
|
|
@@ -1359,7 +1661,7 @@ export function getTime(schema) {
|
|
|
1359
1661
|
export function insertWarning(schema) {
|
|
1360
1662
|
return `
|
|
1361
1663
|
INSERT INTO ${schema}.warning (type, message, data, created_on)
|
|
1362
|
-
VALUES ($1, $2, $3, ${schema}.job_now())
|
|
1664
|
+
VALUES ($1, $2, $3::text::jsonb, ${schema}.job_now())
|
|
1363
1665
|
`;
|
|
1364
1666
|
}
|
|
1365
1667
|
export function getWarnings(schema) {
|
|
@@ -1410,6 +1712,9 @@ export function createTableQueueStats(schema, noPartitioning = false) {
|
|
|
1410
1712
|
failed_delta int,
|
|
1411
1713
|
delta_seconds int,
|
|
1412
1714
|
delta_on timestamptz,
|
|
1715
|
+
wait_bins int[],
|
|
1716
|
+
run_bins int[],
|
|
1717
|
+
ready_oldest_seconds int,
|
|
1413
1718
|
captured_on timestamptz NOT NULL DEFAULT now(),
|
|
1414
1719
|
${noPartitioning ? 'PRIMARY KEY (id)' : 'PRIMARY KEY (id, captured_on)'}
|
|
1415
1720
|
) ${noPartitioning ? '' : 'PARTITION BY RANGE (captured_on)'}
|
|
@@ -1502,9 +1807,11 @@ export function insertQueueStats(schema, queues, noAdvisoryLocks) {
|
|
|
1502
1807
|
const sql = `
|
|
1503
1808
|
INSERT INTO ${schema}.queue_stats
|
|
1504
1809
|
(name, deferred_count, queued_count, ready_count, active_count, failed_count, total_count,
|
|
1505
|
-
created_delta, completed_delta, failed_delta, delta_seconds, delta_on,
|
|
1810
|
+
created_delta, completed_delta, failed_delta, delta_seconds, delta_on,
|
|
1811
|
+
wait_bins, run_bins, ready_oldest_seconds, captured_on)
|
|
1506
1812
|
SELECT name, deferred_count, queued_count, ready_count, active_count, failed_count, total_count,
|
|
1507
|
-
created_delta, completed_delta, failed_delta, delta_seconds, delta_on,
|
|
1813
|
+
created_delta, completed_delta, failed_delta, delta_seconds, delta_on,
|
|
1814
|
+
wait_bins, run_bins, ready_oldest_seconds, ${schema}.job_now()
|
|
1508
1815
|
FROM ${schema}.queue
|
|
1509
1816
|
WHERE name = ANY(${serializeArrayParam(queues)})
|
|
1510
1817
|
`;
|
|
@@ -1537,6 +1844,9 @@ export function getQueueStatsCache(schema) {
|
|
|
1537
1844
|
failed_delta as "failedDelta",
|
|
1538
1845
|
delta_seconds as "deltaSeconds",
|
|
1539
1846
|
delta_on as "deltaOn",
|
|
1847
|
+
wait_bins as "waitBins",
|
|
1848
|
+
run_bins as "runBins",
|
|
1849
|
+
ready_oldest_seconds as "readyOldestSeconds",
|
|
1540
1850
|
table_name as "table",
|
|
1541
1851
|
monitor_on as "capturedOn",
|
|
1542
1852
|
(extract(epoch from (${schema}.job_now() - monitor_on)) * 1000)::float8 as "cacheAgeMs",
|
|
@@ -1560,6 +1870,9 @@ export function getQueueStatsHistory(schema) {
|
|
|
1560
1870
|
failed_delta as "failedDelta",
|
|
1561
1871
|
delta_seconds as "deltaSeconds",
|
|
1562
1872
|
delta_on as "deltaOn",
|
|
1873
|
+
wait_bins as "waitBins",
|
|
1874
|
+
run_bins as "runBins",
|
|
1875
|
+
ready_oldest_seconds as "readyOldestSeconds",
|
|
1563
1876
|
captured_on as "capturedOn"
|
|
1564
1877
|
FROM ${schema}.queue_stats
|
|
1565
1878
|
WHERE name = $1
|
|
@@ -1613,6 +1926,11 @@ const STATS_AGG = {
|
|
|
1613
1926
|
// YugabyteDB, none of which can rely on it. to_timestamp / extract(epoch) / floor exist on all of
|
|
1614
1927
|
// them (extract returns double on PG13, numeric on PG14+; floor/division handle both identically),
|
|
1615
1928
|
// and buckets align to the Unix epoch so their boundaries are stable across calls.
|
|
1929
|
+
//
|
|
1930
|
+
// Wait and run histograms are added up per returned bucket in SQL (passes, slots, histograms), so a
|
|
1931
|
+
// bucket comes back as one histogram however many passes it covers. Each pass lands in the bucket
|
|
1932
|
+
// its counters were placed in, then its counts are summed per bucket and slot, a slot with no job in
|
|
1933
|
+
// any pass as 0. A bucket no pass measured has no histogram row and comes back null.
|
|
1616
1934
|
export function getQueueStatsHistoryBucketed(schema, aggregate, mode) {
|
|
1617
1935
|
const agg = STATS_AGG[aggregate];
|
|
1618
1936
|
const widthCte = mode === 'auto'
|
|
@@ -1628,7 +1946,7 @@ export function getQueueStatsHistoryBucketed(schema, aggregate, mode) {
|
|
|
1628
1946
|
FROM extent
|
|
1629
1947
|
),
|
|
1630
1948
|
w AS (
|
|
1631
|
-
SELECT greatest(1, ceil(extract(epoch from (hi - lo))::float8 / greatest($5, 1)::float8)::bigint)
|
|
1949
|
+
SELECT greatest(1, ceil(extract(epoch from (hi - lo))::float8 / greatest($5, 1)::float8)::bigint) AS secs
|
|
1632
1950
|
FROM bounds
|
|
1633
1951
|
)`
|
|
1634
1952
|
: 'WITH w AS (SELECT greatest($5, 1)::bigint AS secs)';
|
|
@@ -1663,7 +1981,8 @@ export function getQueueStatsHistoryBucketed(schema, aggregate, mode) {
|
|
|
1663
1981
|
sum(completed_delta)::int as "completedDelta",
|
|
1664
1982
|
sum(failed_delta)::int as "failedDelta",
|
|
1665
1983
|
sum(delta_seconds)::int as "deltaSeconds",
|
|
1666
|
-
max(delta_on) as "deltaOn"
|
|
1984
|
+
max(delta_on) as "deltaOn",
|
|
1985
|
+
max(ready_oldest_seconds) as "readyOldestSeconds"
|
|
1667
1986
|
FROM ${schema}.queue_stats, w
|
|
1668
1987
|
WHERE name = $1
|
|
1669
1988
|
AND delta_on IS NOT NULL
|
|
@@ -1687,10 +2006,37 @@ export function getQueueStatsHistoryBucketed(schema, aggregate, mode) {
|
|
|
1687
2006
|
c."completedDelta",
|
|
1688
2007
|
c."failedDelta",
|
|
1689
2008
|
c."deltaSeconds",
|
|
1690
|
-
c."deltaOn"
|
|
2009
|
+
c."deltaOn",
|
|
2010
|
+
c."readyOldestSeconds",
|
|
2011
|
+
c.bucket as "counterBucket"
|
|
1691
2012
|
FROM gauges g
|
|
1692
2013
|
FULL JOIN counters c ON c.bucket = g.bucket
|
|
1693
|
-
)
|
|
2014
|
+
),
|
|
2015
|
+
passes AS (
|
|
2016
|
+
SELECT ${bucket('delta_on')} as "counterBucket", wait_bins, run_bins
|
|
2017
|
+
FROM ${schema}.queue_stats, w
|
|
2018
|
+
WHERE name = $1
|
|
2019
|
+
AND delta_on IS NOT NULL
|
|
2020
|
+
AND wait_bins IS NOT NULL
|
|
2021
|
+
AND ($2::timestamptz IS NULL OR delta_on >= $2)
|
|
2022
|
+
AND ($3::timestamptz IS NULL OR delta_on <= $3)
|
|
2023
|
+
),
|
|
2024
|
+
slots AS (
|
|
2025
|
+
SELECT p.bucket, u.slot, coalesce(sum(u.w), 0)::int AS w, coalesce(sum(u.r), 0)::int AS r
|
|
2026
|
+
FROM (SELECT DISTINCT bucket, "counterBucket" FROM placed) p
|
|
2027
|
+
JOIN passes ps ON ps."counterBucket" = p."counterBucket",
|
|
2028
|
+
unnest(ps.wait_bins, ps.run_bins) WITH ORDINALITY AS u(w, r, slot)
|
|
2029
|
+
GROUP BY 1, 2
|
|
2030
|
+
),
|
|
2031
|
+
histograms AS (
|
|
2032
|
+
SELECT
|
|
2033
|
+
bucket,
|
|
2034
|
+
array_agg(w ORDER BY slot) as "waitBins",
|
|
2035
|
+
array_agg(r ORDER BY slot) as "runBins"
|
|
2036
|
+
FROM slots
|
|
2037
|
+
GROUP BY 1
|
|
2038
|
+
),
|
|
2039
|
+
folded AS (
|
|
1694
2040
|
SELECT
|
|
1695
2041
|
bucket as "capturedOn",
|
|
1696
2042
|
max("deferredCount")::int as "deferredCount",
|
|
@@ -1703,11 +2049,16 @@ export function getQueueStatsHistoryBucketed(schema, aggregate, mode) {
|
|
|
1703
2049
|
sum("completedDelta")::int as "completedDelta",
|
|
1704
2050
|
sum("failedDelta")::int as "failedDelta",
|
|
1705
2051
|
sum("deltaSeconds")::int as "deltaSeconds",
|
|
1706
|
-
max("deltaOn") as "deltaOn"
|
|
2052
|
+
max("deltaOn") as "deltaOn",
|
|
2053
|
+
max("readyOldestSeconds")::int as "readyOldestSeconds"
|
|
1707
2054
|
FROM placed
|
|
1708
2055
|
WHERE bucket IS NOT NULL
|
|
1709
2056
|
GROUP BY 1
|
|
1710
|
-
|
|
2057
|
+
)
|
|
2058
|
+
SELECT f.*, h."waitBins", h."runBins"
|
|
2059
|
+
FROM folded f
|
|
2060
|
+
LEFT JOIN histograms h ON h.bucket = f."capturedOn"
|
|
2061
|
+
ORDER BY f."capturedOn" DESC
|
|
1711
2062
|
LIMIT ${limit}
|
|
1712
2063
|
`;
|
|
1713
2064
|
}
|
|
@@ -1814,7 +2165,7 @@ function buildFetchParams(options) {
|
|
|
1814
2165
|
* exceeds fetch time.
|
|
1815
2166
|
*/
|
|
1816
2167
|
export function fetchNextJob(options, noSkipLocked = false) {
|
|
1817
|
-
const { schema, table, name, policy, limit, includeMetadata, ignoreStartAfter = false, groupConcurrency, minPriority, maxPriority } = options;
|
|
2168
|
+
const { schema, table, name, policy, limit, includeMetadata, includeTraceContext = false, ignoreStartAfter = false, groupConcurrency, minPriority, maxPriority } = options;
|
|
1818
2169
|
const keyStrictFifo = policy === QUEUE_POLICIES.key_strict_fifo;
|
|
1819
2170
|
const singletonFetch = limit > 1 && (policy === QUEUE_POLICIES.singleton || policy === QUEUE_POLICIES.stately);
|
|
1820
2171
|
const hasIgnoreSingletons = options.ignoreSingletons != null && options.ignoreSingletons.length > 0;
|
|
@@ -2002,7 +2353,7 @@ export function fetchNextJob(options, noSkipLocked = false) {
|
|
|
2002
2353
|
WHERE name = '${name}' AND ${updateMatch}
|
|
2003
2354
|
${singletonFetch && !hasGroupConcurrency ? 'AND singleton_rn = 1' : ''}
|
|
2004
2355
|
${distributedStateCheck}
|
|
2005
|
-
RETURNING j.${includeMetadata ? JOB_COLUMNS_ALL : JOB_COLUMNS_MIN}
|
|
2356
|
+
RETURNING j.${includeMetadata ? JOB_COLUMNS_ALL : JOB_COLUMNS_MIN}${includeTraceContext ? ', j.trace_context as "__traceContext"' : ''}
|
|
2006
2357
|
`,
|
|
2007
2358
|
values: params.values
|
|
2008
2359
|
};
|
|
@@ -2084,10 +2435,16 @@ function lockedChildrenCte(schema) {
|
|
|
2084
2435
|
FOR UPDATE OF j
|
|
2085
2436
|
)`;
|
|
2086
2437
|
}
|
|
2438
|
+
// A child released by its last parent has its start_after moved up to the release, so its wait (in
|
|
2439
|
+
// the monitor's histograms and ready_oldest_seconds) counts from when it could first run rather than
|
|
2440
|
+
// from when the flow was sent. A start_after still in the future is kept.
|
|
2087
2441
|
function unblockChildrenUpdate(schema) {
|
|
2088
2442
|
return `UPDATE ${schema}.job j
|
|
2089
2443
|
SET pending_dependencies = GREATEST(j.pending_dependencies - lc.n, 0),
|
|
2090
|
-
blocked = GREATEST(j.pending_dependencies - lc.n, 0) > 0
|
|
2444
|
+
blocked = GREATEST(j.pending_dependencies - lc.n, 0) > 0,
|
|
2445
|
+
start_after = CASE WHEN GREATEST(j.pending_dependencies - lc.n, 0) = 0
|
|
2446
|
+
THEN GREATEST(j.start_after, ${schema}.job_now())
|
|
2447
|
+
ELSE j.start_after END
|
|
2091
2448
|
FROM locked_children lc
|
|
2092
2449
|
WHERE j.name = lc.name
|
|
2093
2450
|
AND j.id = lc.id`;
|
|
@@ -2161,12 +2518,16 @@ export function cancelJobs(schema, table, fenced) {
|
|
|
2161
2518
|
${settledCountAndIds()}
|
|
2162
2519
|
`;
|
|
2163
2520
|
}
|
|
2521
|
+
// A resumed job's start_after moves up to now, as a released flow child's does, so its wait (in the
|
|
2522
|
+
// monitor's histograms and ready_oldest_seconds) counts from when it could run again rather than
|
|
2523
|
+
// from when it was first sent. A start_after still in the future is kept.
|
|
2164
2524
|
export function resumeJobs(schema, table) {
|
|
2165
2525
|
return `
|
|
2166
2526
|
WITH results as (
|
|
2167
2527
|
UPDATE ${schema}.${table}
|
|
2168
2528
|
SET completed_on = NULL,
|
|
2169
|
-
state = '${JOB_STATES.created}'
|
|
2529
|
+
state = '${JOB_STATES.created}',
|
|
2530
|
+
start_after = GREATEST(start_after, ${schema}.job_now())
|
|
2170
2531
|
WHERE name = $1
|
|
2171
2532
|
AND id = ANY($2::uuid[])
|
|
2172
2533
|
AND state = '${JOB_STATES.cancelled}'
|
|
@@ -2241,7 +2602,8 @@ export function insertJobs(schema, { table, name, returnId = true, notify = fals
|
|
|
2241
2602
|
heartbeat_seconds,
|
|
2242
2603
|
blocked,
|
|
2243
2604
|
blocking,
|
|
2244
|
-
pending_dependencies
|
|
2605
|
+
pending_dependencies,
|
|
2606
|
+
trace_context
|
|
2245
2607
|
)
|
|
2246
2608
|
SELECT
|
|
2247
2609
|
COALESCE(id, gen_random_uuid()) as id,
|
|
@@ -2270,7 +2632,8 @@ export function insertJobs(schema, { table, name, returnId = true, notify = fals
|
|
|
2270
2632
|
COALESCE("heartbeatSeconds", q.heartbeat_seconds) as heartbeat_seconds,
|
|
2271
2633
|
COALESCE(blocked, false) as blocked,
|
|
2272
2634
|
COALESCE(blocking, false) as blocking,
|
|
2273
|
-
COALESCE("pendingDependencies", 0) as pending_dependencies
|
|
2635
|
+
COALESCE("pendingDependencies", 0) as pending_dependencies,
|
|
2636
|
+
"__traceContext" as trace_context
|
|
2274
2637
|
FROM (
|
|
2275
2638
|
SELECT *,
|
|
2276
2639
|
CASE
|
|
@@ -2299,7 +2662,8 @@ export function insertJobs(schema, { table, name, returnId = true, notify = fals
|
|
|
2299
2662
|
"heartbeatSeconds" integer,
|
|
2300
2663
|
blocked boolean,
|
|
2301
2664
|
blocking boolean,
|
|
2302
|
-
"pendingDependencies" integer
|
|
2665
|
+
"pendingDependencies" integer,
|
|
2666
|
+
"__traceContext" jsonb
|
|
2303
2667
|
)
|
|
2304
2668
|
) j
|
|
2305
2669
|
JOIN ${schema}.queue q ON q.name = '${name}'
|
|
@@ -2455,7 +2819,8 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
|
|
|
2455
2819
|
source_created_on,
|
|
2456
2820
|
source_retry_count,
|
|
2457
2821
|
source_output,
|
|
2458
|
-
source_root_id
|
|
2822
|
+
source_root_id,
|
|
2823
|
+
trace_context
|
|
2459
2824
|
)
|
|
2460
2825
|
SELECT
|
|
2461
2826
|
id,
|
|
@@ -2501,7 +2866,8 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
|
|
|
2501
2866
|
source_created_on,
|
|
2502
2867
|
source_retry_count,
|
|
2503
2868
|
source_output,
|
|
2504
|
-
source_root_id
|
|
2869
|
+
source_root_id,
|
|
2870
|
+
trace_context
|
|
2505
2871
|
FROM deleted_jobs
|
|
2506
2872
|
ON CONFLICT DO NOTHING
|
|
2507
2873
|
RETURNING *
|
|
@@ -2542,7 +2908,8 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
|
|
|
2542
2908
|
source_created_on,
|
|
2543
2909
|
source_retry_count,
|
|
2544
2910
|
source_output,
|
|
2545
|
-
source_root_id
|
|
2911
|
+
source_root_id,
|
|
2912
|
+
trace_context
|
|
2546
2913
|
)
|
|
2547
2914
|
SELECT
|
|
2548
2915
|
id,
|
|
@@ -2579,7 +2946,8 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
|
|
|
2579
2946
|
source_created_on,
|
|
2580
2947
|
source_retry_count,
|
|
2581
2948
|
source_output,
|
|
2582
|
-
source_root_id
|
|
2949
|
+
source_root_id,
|
|
2950
|
+
trace_context
|
|
2583
2951
|
FROM deleted_jobs
|
|
2584
2952
|
WHERE id NOT IN (SELECT id from retried_jobs)
|
|
2585
2953
|
RETURNING *
|
|
@@ -2592,7 +2960,7 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
|
|
|
2592
2960
|
dlq_jobs as (
|
|
2593
2961
|
INSERT INTO ${schema}.job (name, priority, data, retry_limit, retry_backoff, retry_delay, start_after, created_on, keep_until, deletion_seconds,
|
|
2594
2962
|
expire_seconds, singleton_key, group_id, group_tier, heartbeat_seconds,
|
|
2595
|
-
source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id)
|
|
2963
|
+
source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id, trace_context)
|
|
2596
2964
|
SELECT
|
|
2597
2965
|
r.dead_letter,
|
|
2598
2966
|
r.priority,
|
|
@@ -2614,7 +2982,8 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
|
|
|
2614
2982
|
r.created_on,
|
|
2615
2983
|
r.retry_count,
|
|
2616
2984
|
r.output,
|
|
2617
|
-
COALESCE(r.source_root_id, r.id)
|
|
2985
|
+
COALESCE(r.source_root_id, r.id),
|
|
2986
|
+
r.trace_context
|
|
2618
2987
|
FROM results r
|
|
2619
2988
|
JOIN ${schema}.queue q ON q.name = r.dead_letter
|
|
2620
2989
|
WHERE state = '${JOB_STATES.failed}'
|
|
@@ -2815,10 +3184,10 @@ export function insertRetryJob(schema, table) {
|
|
|
2815
3184
|
group_id, group_tier, expire_seconds, deletion_seconds, created_on, completed_on,
|
|
2816
3185
|
keep_until, policy, output, dead_letter,
|
|
2817
3186
|
heartbeat_on, heartbeat_seconds, blocked, blocking, pending_dependencies,
|
|
2818
|
-
source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id
|
|
3187
|
+
source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id, trace_context
|
|
2819
3188
|
) VALUES (
|
|
2820
|
-
$1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22,
|
|
2821
|
-
$25, $26, $27, $28, $29, $30, $31, $32, $33, $34, $35
|
|
3189
|
+
$1, $2, $3, $4::text::jsonb, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22,
|
|
3190
|
+
$23::text::jsonb, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34::text::jsonb, $35, $36::text::jsonb
|
|
2822
3191
|
) ON CONFLICT DO NOTHING
|
|
2823
3192
|
RETURNING id
|
|
2824
3193
|
`;
|
|
@@ -2827,10 +3196,10 @@ export function insertDeadLetterJob(schema) {
|
|
|
2827
3196
|
return `
|
|
2828
3197
|
INSERT INTO ${schema}.job (name, data, priority, retry_limit, retry_backoff, retry_delay, start_after, created_on, keep_until, deletion_seconds,
|
|
2829
3198
|
expire_seconds, singleton_key, group_id, group_tier, heartbeat_seconds,
|
|
2830
|
-
source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id)
|
|
2831
|
-
SELECT $1, $2, $9, q.retry_limit, q.retry_backoff, q.retry_delay, ${schema}.job_now(), ${schema}.job_now(), ${schema}.job_now() + q.retention_seconds * interval '1s', q.deletion_seconds,
|
|
3199
|
+
source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id, trace_context)
|
|
3200
|
+
SELECT $1, $2::text::jsonb, $9, q.retry_limit, q.retry_backoff, q.retry_delay, ${schema}.job_now(), ${schema}.job_now(), ${schema}.job_now() + q.retention_seconds * interval '1s', q.deletion_seconds,
|
|
2832
3201
|
q.expire_seconds, $8, $10, $11, q.heartbeat_seconds,
|
|
2833
|
-
$4, $5, $6, $7, $3, COALESCE($12::uuid, $5::uuid)
|
|
3202
|
+
$4, $5, $6, $7, $3::text::jsonb, COALESCE($12::uuid, $5::uuid), $13::text::jsonb
|
|
2834
3203
|
FROM ${schema}.queue q WHERE q.name = $1
|
|
2835
3204
|
`;
|
|
2836
3205
|
}
|
|
@@ -2855,7 +3224,7 @@ function redriveWhere(schema, table) {
|
|
|
2855
3224
|
AND k.state IN ('${JOB_STATES.active}', '${JOB_STATES.retry}', '${JOB_STATES.failed}')
|
|
2856
3225
|
)
|
|
2857
3226
|
AND ($3::text IS NULL OR j.source_name = $3)
|
|
2858
|
-
AND ($4::jsonb IS NULL OR j.data @> $4::jsonb)
|
|
3227
|
+
AND ($4::text::jsonb IS NULL OR j.data @> $4::text::jsonb)
|
|
2859
3228
|
AND ($5::timestamptz IS NULL OR j.created_on < $5)
|
|
2860
3229
|
AND ($6::uuid[] IS NULL OR j.id = ANY($6::uuid[]))`;
|
|
2861
3230
|
}
|
|
@@ -2869,13 +3238,13 @@ function redriveWhere(schema, table) {
|
|
|
2869
3238
|
// Job-identity columns (priority, singleton_key, group_id, group_tier) are carried over instead.
|
|
2870
3239
|
const REDRIVE_INSERT_COLUMNS = `(id, name, data, priority, retry_limit, retry_backoff, retry_delay, retry_delay_max,
|
|
2871
3240
|
expire_seconds, start_after, created_on, keep_until, deletion_seconds, policy, singleton_key, group_id, group_tier,
|
|
2872
|
-
heartbeat_seconds, dead_letter, source_root_id)`;
|
|
3241
|
+
heartbeat_seconds, dead_letter, source_root_id, trace_context)`;
|
|
2873
3242
|
function redriveInsertValues(schema, newId, destination) {
|
|
2874
3243
|
return `${newId}, COALESCE(${destination}, m.source_name), m.data, m.priority, q.retry_limit, q.retry_backoff,
|
|
2875
3244
|
q.retry_delay, q.retry_delay_max, q.expire_seconds, ${schema}.job_now(), ${schema}.job_now(),
|
|
2876
3245
|
${schema}.job_now() + q.retention_seconds * interval '1s', q.deletion_seconds, q.policy,
|
|
2877
3246
|
m.singleton_key, m.group_id, m.group_tier, q.heartbeat_seconds, q.dead_letter,
|
|
2878
|
-
COALESCE(m.source_root_id, m.source_id)`;
|
|
3247
|
+
COALESCE(m.source_root_id, m.source_id), m.trace_context`;
|
|
2879
3248
|
}
|
|
2880
3249
|
// What a job that could not be re-created becomes: failed, in place, in the dead letter queue, with
|
|
2881
3250
|
// the reason as its output. Never deleted: before 12.35 it was, and the job was simply gone. As a
|
|
@@ -3042,13 +3411,15 @@ export function deletion(schema, table, queues, noAdvisoryLocks) {
|
|
|
3042
3411
|
`;
|
|
3043
3412
|
return locked(schema, sql, table + 'deletion', noAdvisoryLocks);
|
|
3044
3413
|
}
|
|
3414
|
+
// start_after moves up to now, as in resumeJobs.
|
|
3045
3415
|
export function retryJobs(schema, table) {
|
|
3046
3416
|
return `
|
|
3047
3417
|
WITH results as (
|
|
3048
3418
|
UPDATE ${schema}.job
|
|
3049
3419
|
SET state = '${JOB_STATES.retry}',
|
|
3050
3420
|
retry_limit = retry_limit + 1,
|
|
3051
|
-
completed_on = NULL
|
|
3421
|
+
completed_on = NULL,
|
|
3422
|
+
start_after = GREATEST(start_after, ${schema}.job_now())
|
|
3052
3423
|
WHERE name = $1
|
|
3053
3424
|
AND id = ANY($2::uuid[])
|
|
3054
3425
|
AND state = '${JOB_STATES.failed}'
|
|
@@ -3199,7 +3570,10 @@ function throughputAssignments(end, resetMax) {
|
|
|
3199
3570
|
completed_delta = COALESCE(stats."completedDelta", 0),
|
|
3200
3571
|
failed_delta = COALESCE(stats."failedDelta", 0),
|
|
3201
3572
|
delta_seconds = CASE WHEN ${seconds} < 0 THEN 0 ELSE ${seconds} END,
|
|
3202
|
-
delta_on = GREATEST(queue.delta_on, ${end})
|
|
3573
|
+
delta_on = GREATEST(queue.delta_on, ${end}),
|
|
3574
|
+
wait_bins = COALESCE(stats."waitBins", ${EMPTY_BINS}),
|
|
3575
|
+
run_bins = COALESCE(stats."runBins", ${EMPTY_BINS}),
|
|
3576
|
+
ready_oldest_seconds = COALESCE(stats."readyOldestSeconds", 0),`;
|
|
3203
3577
|
}
|
|
3204
3578
|
// The windows a true-up may still revise, per queue: every recorded snapshot whose window ends
|
|
3205
3579
|
// after the anchor `h`, up to the newest one, `top`, with the counters they hold between them.
|
|
@@ -3363,6 +3737,61 @@ export function trueUpQueueStats(schema, table, queues, noAdvisoryLocks, window
|
|
|
3363
3737
|
`;
|
|
3364
3738
|
return transaction(sql);
|
|
3365
3739
|
}
|
|
3740
|
+
// Wait and run times, as the monitor records them: a histogram per counted pass, of the jobs that
|
|
3741
|
+
// finished in its window. Slot 0 holds times under 10 ms, slots 1 to LATENCY_BINS bins that each
|
|
3742
|
+
// grow by √2 (slot k runs from 10 ms · √2^(k-1) to 10 ms · √2^k), and the last slot everything past
|
|
3743
|
+
// about 23 hours. Log-spaced because the times span six orders of magnitude, and a percentile read
|
|
3744
|
+
// from them lies in the same bin as the exact one, a factor of √2 at most. Histograms rather than
|
|
3745
|
+
// percentiles, because histograms add: a reader sums them across passes, buckets or queues and
|
|
3746
|
+
// reads any percentile from the sum.
|
|
3747
|
+
// Stored as 48 slots in slot order, null where no job landed: Postgres keeps a null as one bit
|
|
3748
|
+
// rather than four bytes, and most slots are empty. getQueueStats() hands them out with nulls as 0.
|
|
3749
|
+
// Readers that add slots in SQL coalesce, since a null plus a count is null.
|
|
3750
|
+
export const LATENCY_BINS = 46;
|
|
3751
|
+
export const LATENCY_SLOTS = LATENCY_BINS + 2;
|
|
3752
|
+
export const LATENCY_MIN_SECONDS = 0.01;
|
|
3753
|
+
// A measured histogram in which no job finished: every slot null, not a null array.
|
|
3754
|
+
const EMPTY_BINS = `'{${new Array(LATENCY_SLOTS).fill('NULL').join(',')}}'::int[]`;
|
|
3755
|
+
// Both bins travel packed in one integer (wait * LATENCY_PACK + run), so the aggregate needs one
|
|
3756
|
+
// array and one filter. Two arrays, each with its own filter evaluated on every row of the table,
|
|
3757
|
+
// cost twice as much: measured on 2.5M rows, +55 ms on a ~560 ms pass packed, +80 to 120 ms apart.
|
|
3758
|
+
const LATENCY_PACK = 64;
|
|
3759
|
+
// date_part rather than extract: extract returns numeric on PostgreSQL 14 and later, and the numeric
|
|
3760
|
+
// arithmetic on every finished job was a large share of the histograms' cost. date_part is float8
|
|
3761
|
+
// on every supported backend, as the bin math needs anyway.
|
|
3762
|
+
const waitSeconds = "date_part('epoch', (j.started_on - GREATEST(j.created_on, j.start_after)))";
|
|
3763
|
+
const runSeconds = "date_part('epoch', (j.completed_on - j.started_on))";
|
|
3764
|
+
// The slot is width_bucket's, written out: CockroachDB's width_bucket takes decimals, not float8.
|
|
3765
|
+
// ln(t / 10 ms) over ln(√2), plus one, clamped to slot 0 below 10 ms and the last slot past the end.
|
|
3766
|
+
// A duration of zero or less (stamps from two clocks) lands in slot 0.
|
|
3767
|
+
function latencyBin(seconds) {
|
|
3768
|
+
const lo = Math.log(LATENCY_MIN_SECONDS);
|
|
3769
|
+
const width = Math.log(2) / 2;
|
|
3770
|
+
const raw = `floor((ln(GREATEST((${seconds})::float8, 0.001::float8)) - ${lo}::float8) / ${width}::float8)::int + 1`;
|
|
3771
|
+
return `LEAST(GREATEST(${raw}, 0), ${LATENCY_SLOTS - 1})`;
|
|
3772
|
+
}
|
|
3773
|
+
// The aggregate's array is unnested once per queue, in a lateral join after the aggregate, and
|
|
3774
|
+
// counted by packed value: at most LATENCY_SLOTS² rows (ps, ns) however many jobs finished. Both
|
|
3775
|
+
// histograms are read from those, so the per-job array is walked once rather than once each.
|
|
3776
|
+
function latencyCounts(packed) {
|
|
3777
|
+
return `LEFT JOIN LATERAL (
|
|
3778
|
+
SELECT array_agg(g.p) AS ps, array_agg(g.n) AS ns
|
|
3779
|
+
FROM (SELECT p, count(*)::int AS n FROM unnest(${packed}) AS u(p) GROUP BY p) g
|
|
3780
|
+
) latency ON true`;
|
|
3781
|
+
}
|
|
3782
|
+
// One histogram from latencyCounts: every slot in slot order, null where no job landed. A pass in
|
|
3783
|
+
// which nothing finished still writes the 48 slots (unnest of a null array is no rows, and the left
|
|
3784
|
+
// join keeps every slot), so a pass that counted says so, as its deltas do. floor() of a float
|
|
3785
|
+
// division rather than integer division, which CockroachDB answers in decimal.
|
|
3786
|
+
function latencySlotOf(which) {
|
|
3787
|
+
return which === 'wait' ? `floor(p / ${LATENCY_PACK}.0)::int` : `(p % ${LATENCY_PACK})::int`;
|
|
3788
|
+
}
|
|
3789
|
+
function latencyHistogram(which) {
|
|
3790
|
+
return `(SELECT array_agg(c.n ORDER BY s.slot)
|
|
3791
|
+
FROM generate_series(0, ${LATENCY_SLOTS - 1}) AS s(slot)
|
|
3792
|
+
LEFT JOIN (SELECT ${latencySlotOf(which)} AS slot, sum(n)::int AS n
|
|
3793
|
+
FROM unnest(latency.ps, latency.ns) AS u(p, n) GROUP BY 1) c ON c.slot = s.slot)`;
|
|
3794
|
+
}
|
|
3366
3795
|
// Every count the monitor keeps, from one pass over the queue's table.
|
|
3367
3796
|
//
|
|
3368
3797
|
// Six of them are gauges — what the queue looks like right now. Three are not:
|
|
@@ -3386,6 +3815,13 @@ export function trueUpQueueStats(schema, table, queues, noAdvisoryLocks, window
|
|
|
3386
3815
|
// filtered on completed_on, would be a whole extra scan of the largest table in
|
|
3387
3816
|
// the schema, and there is no index on that column to make it cheaper.
|
|
3388
3817
|
export function getQueueStats(schema, table, queues, throughput = false, window = {}) {
|
|
3818
|
+
// A queued job is exactly one of blocked (waiting on a flow parent), deferred (start_after still
|
|
3819
|
+
// ahead) or ready, so the three add up to queuedCount. Blocked wins over deferred: a job whose
|
|
3820
|
+
// parent has not finished cannot run when its start_after comes round.
|
|
3821
|
+
const queued = `j.state < '${JOB_STATES.active}'`;
|
|
3822
|
+
const blocked = `${queued} AND j.blocked`;
|
|
3823
|
+
const deferred = `${queued} AND NOT j.blocked AND j.start_after > ${schema}.job_now()`;
|
|
3824
|
+
const ready = `${queued} AND NOT j.blocked AND j.start_after <= ${schema}.job_now()`;
|
|
3389
3825
|
// Counted only with persistQueueStats. Otherwise the aggregate does what it did before throughput
|
|
3390
3826
|
// existed: no join, no extra counts, no cost. The measured price is in the `persistQueueStats` docs.
|
|
3391
3827
|
const end = deltaWindowEnd(schema, window.lag);
|
|
@@ -3400,11 +3836,18 @@ export function getQueueStats(schema, table, queues, throughput = false, window
|
|
|
3400
3836
|
"createdDelta",
|
|
3401
3837
|
"completedDelta",
|
|
3402
3838
|
"failedDelta",
|
|
3839
|
+
${latencyHistogram('wait')} as "waitBins",
|
|
3840
|
+
${latencyHistogram('run')} as "runBins",
|
|
3841
|
+
"readyOldestSeconds",
|
|
3403
3842
|
COALESCE("recount" > "settled", false) as "trueUp",`,
|
|
3404
3843
|
counts: `
|
|
3405
3844
|
(count(*) FILTER (WHERE ${inWindow('created_on')}))::int as "createdDelta",
|
|
3406
3845
|
(count(*) FILTER (WHERE j.state = '${JOB_STATES.completed}' AND ${inWindow('completed_on')}))::int as "completedDelta",
|
|
3407
3846
|
(count(*) FILTER (WHERE j.state = '${JOB_STATES.failed}' AND ${inWindow('completed_on')}))::int as "failedDelta",
|
|
3847
|
+
array_agg(${latencyBin(waitSeconds)} * ${LATENCY_PACK} + ${latencyBin(runSeconds)})
|
|
3848
|
+
FILTER (WHERE j.state IN ('${JOB_STATES.completed}', '${JOB_STATES.failed}') AND j.started_on IS NOT NULL AND ${inWindow('completed_on')}) as "latencyBins",
|
|
3849
|
+
round(extract(epoch from (${schema}.job_now() - min(GREATEST(j.created_on, j.start_after))
|
|
3850
|
+
FILTER (WHERE ${ready}))))::int as "readyOldestSeconds",
|
|
3408
3851
|
sum(
|
|
3409
3852
|
CASE WHEN ${settled('created_on')} THEN 1 ELSE 0 END +
|
|
3410
3853
|
CASE WHEN ${settled('completed_on')} AND j.state IN ('${JOB_STATES.completed}', '${JOB_STATES.failed}') THEN 1 ELSE 0 END
|
|
@@ -3414,16 +3857,18 @@ export function getQueueStats(schema, table, queues, throughput = false, window
|
|
|
3414
3857
|
SELECT q.name, q.delta_on, a.h, t.top, t.settled
|
|
3415
3858
|
FROM ${schema}.queue q${trueUpSettled(schema, 'q', window.trueUpMax)}
|
|
3416
3859
|
WHERE q.name = ANY($1::text[])
|
|
3417
|
-
) q ON q.name = j.name
|
|
3860
|
+
) q ON q.name = j.name`,
|
|
3861
|
+
lateral: latencyCounts('stats."latencyBins"')
|
|
3418
3862
|
}
|
|
3419
|
-
: { select: '', counts: '', join: '' };
|
|
3863
|
+
: { select: '', counts: '', join: '', lateral: '' };
|
|
3420
3864
|
return {
|
|
3421
3865
|
text: `
|
|
3422
3866
|
SELECT
|
|
3423
3867
|
name,
|
|
3424
3868
|
"deferredCount",
|
|
3425
3869
|
"queuedCount",
|
|
3426
|
-
|
|
3870
|
+
"readyCount",
|
|
3871
|
+
"blockedCount",
|
|
3427
3872
|
"activeCount",
|
|
3428
3873
|
"failedCount",
|
|
3429
3874
|
"totalCount",${counters.select}
|
|
@@ -3431,8 +3876,10 @@ export function getQueueStats(schema, table, queues, throughput = false, window
|
|
|
3431
3876
|
FROM (
|
|
3432
3877
|
SELECT
|
|
3433
3878
|
j.name,
|
|
3434
|
-
(count(*) FILTER (WHERE
|
|
3435
|
-
(count(*) FILTER (WHERE
|
|
3879
|
+
(count(*) FILTER (WHERE ${deferred}))::int as "deferredCount",
|
|
3880
|
+
(count(*) FILTER (WHERE ${blocked}))::int as "blockedCount",
|
|
3881
|
+
(count(*) FILTER (WHERE ${ready}))::int as "readyCount",
|
|
3882
|
+
(count(*) FILTER (WHERE ${queued}))::int as "queuedCount",
|
|
3436
3883
|
(count(*) FILTER (WHERE j.state = '${JOB_STATES.active}'))::int as "activeCount",
|
|
3437
3884
|
(count(*) FILTER (WHERE j.state = '${JOB_STATES.failed}'))::int as "failedCount",
|
|
3438
3885
|
count(*)::int as "totalCount",${counters.counts}
|
|
@@ -3442,6 +3889,7 @@ export function getQueueStats(schema, table, queues, throughput = false, window
|
|
|
3442
3889
|
WHERE j.name = ANY($1::text[])
|
|
3443
3890
|
GROUP BY 1
|
|
3444
3891
|
) stats
|
|
3892
|
+
${counters.lateral}
|
|
3445
3893
|
`,
|
|
3446
3894
|
values: [queues]
|
|
3447
3895
|
};
|
|
@@ -3491,6 +3939,7 @@ export function cacheQueueStats(schema, table, queues, noAdvisoryLocks, throughp
|
|
|
3491
3939
|
WITH ${lock.cte}stats AS (SELECT * FROM (${statsText}) agg WHERE true${lock.guard})
|
|
3492
3940
|
UPDATE ${schema}.queue SET
|
|
3493
3941
|
deferred_count = COALESCE(stats."deferredCount", 0),
|
|
3942
|
+
blocked_count = COALESCE(stats."blockedCount", 0),
|
|
3494
3943
|
queued_count = COALESCE(stats."queuedCount", 0),
|
|
3495
3944
|
ready_count = COALESCE(stats."readyCount", 0),
|
|
3496
3945
|
active_count = COALESCE(stats."activeCount", 0),
|
|
@@ -3566,6 +4015,7 @@ export function refreshQueueStats(schema, table, name, options = {}) {
|
|
|
3566
4015
|
WITH ${lock.cte}stats AS (SELECT * FROM (${statsText}) agg WHERE true${lock.guard})
|
|
3567
4016
|
UPDATE ${schema}.queue SET
|
|
3568
4017
|
deferred_count = COALESCE(stats."deferredCount", 0),
|
|
4018
|
+
blocked_count = COALESCE(stats."blockedCount", 0),
|
|
3569
4019
|
queued_count = COALESCE(stats."queuedCount", 0),
|
|
3570
4020
|
ready_count = COALESCE(stats."readyCount", 0),
|
|
3571
4021
|
active_count = COALESCE(stats."activeCount", 0),
|
|
@@ -3676,7 +4126,7 @@ export function findJobs(schema, table, options) {
|
|
|
3676
4126
|
}
|
|
3677
4127
|
if (byData) {
|
|
3678
4128
|
++paramIndex;
|
|
3679
|
-
whereConditions.push(`AND data @> $${paramIndex}`);
|
|
4129
|
+
whereConditions.push(`AND data @> $${paramIndex}::text::jsonb`);
|
|
3680
4130
|
}
|
|
3681
4131
|
if (queued) {
|
|
3682
4132
|
whereConditions.push(`AND state < '${JOB_STATES.active}'`);
|
|
@@ -4002,7 +4452,7 @@ const POLICY_JOB_INDEXES = {
|
|
|
4002
4452
|
// missing: the fetch index was replaced in v40 and the retired number is not reused.
|
|
4003
4453
|
const BASE_JOB_INDEXES = [4, 7, 9, 11, 12];
|
|
4004
4454
|
// The fixed (non-job) managed tables; job/job_common/partitions are handled separately.
|
|
4005
|
-
const FIXED_MANAGED_TABLES = ['version', 'queue', 'schedule', 'subscription', 'bam', 'warning', 'queue_stats', 'job_dependency'];
|
|
4455
|
+
const FIXED_MANAGED_TABLES = ['version', 'queue', 'schedule', 'subscription', 'bam', 'warning', 'queue_stats', 'job_dependency', 'instance'];
|
|
4006
4456
|
// Selects the manifest section for the live architecture, and substitutes the real schema name back in
|
|
4007
4457
|
// for the placeholder the manifest stores.
|
|
4008
4458
|
function manifestSection(partitioned) {
|
|
@@ -4390,12 +4840,12 @@ export function getXminHorizon(lastVacuum, sources = XMIN_HORIZON_QUERY_SOURCES)
|
|
|
4390
4840
|
// query it is follows from pid, application_name and role, looked up live where the catalog's own
|
|
4391
4841
|
// privilege rules still apply.
|
|
4392
4842
|
//
|
|
4393
|
-
// selfApplicationName is what makes "ours or theirs" answerable.
|
|
4394
|
-
// 'pgboss'
|
|
4395
|
-
// it to itself - the monitor's own aggregate, most likely - which has a
|
|
4396
|
-
// from an external reporting tool holding a transaction open. It is
|
|
4397
|
-
// because an adapter-supplied pool sets whatever the host app
|
|
4398
|
-
// definitely pg-boss would be a guess.
|
|
4843
|
+
// selfApplicationName is what makes "ours or theirs" answerable. pg-boss names the pool it owns
|
|
4844
|
+
// 'pgboss', or 'pgboss:<id>' for a registered instance, so a holder matching this connection's own
|
|
4845
|
+
// value is pg-boss doing it to itself - the monitor's own aggregate, most likely - which has a
|
|
4846
|
+
// completely different fix from an external reporting tool holding a transaction open. It is
|
|
4847
|
+
// compared rather than hardcoded because an adapter-supplied pool sets whatever the host app
|
|
4848
|
+
// chose, and claiming that is definitely pg-boss would be a guess.
|
|
4399
4849
|
const backendIdentity = sources.includes('backends')
|
|
4400
4850
|
? `,
|
|
4401
4851
|
(SELECT to_jsonb(h) FROM (
|