pg-boss 12.35.0 → 12.36.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/plans.js CHANGED
@@ -159,6 +159,7 @@ export function create(schema, version, options) {
159
159
  noPartitioning ? '' : ensureQueueStatsPartitions(schema),
160
160
  createTableJobDependency(schema),
161
161
  createIndexJobDependencyParent(schema),
162
+ createTableInstance(schema),
162
163
  createQueueFunction(schema, noPartitioning),
163
164
  deleteQueueFunction(schema, noPartitioning),
164
165
  insertVersion(schema, version)
@@ -184,6 +185,7 @@ function createInline(schema, version, options) {
184
185
  inlineIntoCreateTable(createTableWarning(schema), [createIndexWarning(schema)]),
185
186
  inlineIntoCreateTable(createTableQueueStats(schema, true), [createIndexQueueStats(schema, options.noCovering)]),
186
187
  inlineIntoCreateTable(createTableJobDependency(schema), [createIndexJobDependencyParent(schema)]),
188
+ createTableInstance(schema),
187
189
  createQueueFunction(schema, true),
188
190
  deleteQueueFunction(schema, true),
189
191
  insertVersion(schema, version)
@@ -304,7 +306,11 @@ function createTableVersion(schema) {
304
306
  // try-lock - without capturedOn claiming a freshness the counts do not have.
305
307
  // created_delta / completed_delta / failed_delta are not gauges like the counts beside them: they
306
308
  // are how many jobs went through between the previous monitor pass and the latest one. delta_on and
307
- // delta_seconds are the window those three cover, null until a pass counts.
309
+ // delta_seconds are the window those three cover, null until a pass counts. wait_bins and run_bins
310
+ // are the wait and run times of the jobs that finished in that window, as histograms of LATENCY_SLOTS
311
+ // counts (see LATENCY_BINS), and ready_oldest_seconds is how long the oldest job ready to run had
312
+ // waited at the pass. Like the deltas, all of them are null until a pass counts, and a pass that
313
+ // counts writes a value: every slot null when nothing finished, 0 when nothing was waiting.
308
314
  /* eslint-disable no-restricted-syntax -- column defaults stay on the real clock: every pg-boss write names its timestamps through job_now() */
309
315
  function createTableQueue(schema) {
310
316
  return `
@@ -322,6 +328,7 @@ function createTableQueue(schema) {
322
328
  partition bool NOT NULL,
323
329
  table_name text NOT NULL,
324
330
  deferred_count int NOT NULL default 0,
331
+ blocked_count int NOT NULL default 0,
325
332
  queued_count int NOT NULL default 0,
326
333
  ready_count int NOT NULL default 0,
327
334
  warning_queued int NOT NULL default 0,
@@ -333,6 +340,9 @@ function createTableQueue(schema) {
333
340
  failed_delta int NOT NULL default 0,
334
341
  delta_on timestamp with time zone,
335
342
  delta_seconds int,
343
+ wait_bins int[],
344
+ run_bins int[],
345
+ ready_oldest_seconds int,
336
346
  ready_history int[] NOT NULL default '{}',
337
347
  heartbeat_seconds int,
338
348
  notify bool NOT NULL DEFAULT false,
@@ -435,9 +445,269 @@ export function createTableJobDependency(schema) {
435
445
  )
436
446
  `;
437
447
  }
448
+ // An instance is live until it misses this many heartbeats, and its row is deleted once its heartbeat
449
+ // has not moved for this many days.
450
+ export const INSTANCE_QUIET_BEATS = 3;
451
+ export const INSTANCE_RETENTION_DAYS = 7;
452
+ // A crash never sets stopped_on, so a crash loop leaves a quiet row per restart, and a deploy whose
453
+ // processes never call stop() leaves one per replica. Registering keeps the newest of these dead rows
454
+ // for its own name (for its host, when unnamed), and maintenance keeps the newest overall, so neither
455
+ // can grow the table for the whole retention. Keyed on the name rather than the host because a
456
+ // Kubernetes pod that is recreated comes back under a new hostname.
457
+ export const INSTANCE_DEAD_KEPT_PER_NAME = 20;
458
+ export const INSTANCE_DEAD_KEPT = 1000;
459
+ // Stopped, or quiet: the complement of getInstances' live.
460
+ function instanceDead(schema) {
461
+ return `(stopped_on IS NOT NULL OR heartbeat_on < ${schema}.job_now() - heartbeat_seconds * ${INSTANCE_QUIET_BEATS} * interval '1 second')`;
462
+ }
463
+ // Deletes the dead rows ranked past `keep`, newest start first, among those `where` selects. Ranked
464
+ // before locking and locked with SKIP LOCKED, so a registration and maintenance pruning at once
465
+ // cannot deadlock, and neither deletes a row the other's ranking kept. Used on every backend, unlike
466
+ // the fetch's SKIP LOCKED: where it can pass over an unlocked row (CockroachDB), that row is only
467
+ // left for the next prune.
468
+ function pruneDeadInstances(schema, where, keep) {
469
+ return `
470
+ DELETE FROM ${schema}.instance
471
+ WHERE id IN (
472
+ SELECT id FROM ${schema}.instance
473
+ WHERE id IN (
474
+ SELECT id FROM (
475
+ SELECT id, row_number() OVER (ORDER BY started_on DESC, id DESC) as n
476
+ FROM ${schema}.instance
477
+ WHERE ${where} AND ${instanceDead(schema)}
478
+ ) ranked
479
+ WHERE n > ${keep}
480
+ )
481
+ FOR UPDATE SKIP LOCKED
482
+ )
483
+ `;
484
+ }
485
+ // Run by an instance after it registers: $1 its id, $2 its name, $3 its host.
486
+ export function pruneInstanceLives(schema, keep) {
487
+ return pruneDeadInstances(schema, 'id <> $1 AND (CASE WHEN $2::text IS NULL THEN name IS NULL AND host = $3 ELSE name = $2 END)', keep);
488
+ }
489
+ export function trimDeadInstances(schema, keep) {
490
+ return pruneDeadInstances(schema, 'true', keep);
491
+ }
492
+ // Crash restarts: how many lives in a row on this name and host ended without stop() before this one
493
+ // started. Worked out from the rows rather than carried from one life to the next, because at
494
+ // registration a row that still reads live is either a sibling process (pm2, a second PgBoss) or a
495
+ // predecessor that crashed seconds ago, and only its next missed heartbeats tell which. So a crashed
496
+ // life counts toward an instance only once it is dead, and only if its last heartbeat came before the
497
+ // instance started, which a live sibling's keeps moving past.
498
+ //
499
+ // $1 this instance's id, $2 its name, $3 its host, $4 its pid, $5 when its process started: a row
500
+ // with this pid whose heartbeat predates that is an earlier process that reused the pid, dead at once
501
+ // (a container restart, pid 1 each time), and a row with this pid started since is another PgBoss
502
+ // object in this same process, never a predecessor.
503
+ function crashSlot(schema) {
504
+ return `
505
+ me AS (SELECT id, started_on FROM ${schema}.instance WHERE id = $1),
506
+ slot AS (
507
+ SELECT i.*
508
+ FROM ${schema}.instance i, me
509
+ WHERE i.id <> me.id
510
+ AND i.host = $3
511
+ AND (i.name = $2 OR (i.name IS NULL AND $2::text IS NULL))
512
+ AND i.started_on < me.started_on
513
+ AND i.heartbeat_on <= me.started_on
514
+ AND NOT (i.pid = $4 AND i.started_on >= $5::timestamptz)
515
+ ),
516
+ boundary AS (SELECT max(started_on) as at FROM slot WHERE stopped_on IS NOT NULL),
517
+ streak AS (
518
+ SELECT s.*
519
+ FROM slot s, boundary b
520
+ WHERE s.stopped_on IS NULL
521
+ AND (b.at IS NULL OR s.started_on > b.at)
522
+ )`;
523
+ }
524
+ function crashDead(schema) {
525
+ return `(heartbeat_on < ${schema}.job_now() - heartbeat_seconds * ${INSTANCE_QUIET_BEATS} * interval '1 second'
526
+ OR (pid = $4 AND heartbeat_on < $5::timestamptz))`;
527
+ }
528
+ // Numbers the dead lives of the current streak from the earliest one still kept, whose count stands
529
+ // for any the pruning has taken, and writes each later life's count and this instance's. Rewriting
530
+ // the streak is what keeps the counts exact: a life that crashed before its own recheck is corrected
531
+ // by its successor, so the earliest kept row is always right when the pruning moves past it.
532
+ export function countCrashRestarts(schema) {
533
+ return `
534
+ WITH ${crashSlot(schema)},
535
+ crashed AS (
536
+ SELECT id, crash_restarts, crash_restarts_since, heartbeat_on,
537
+ row_number() OVER (ORDER BY started_on, id) as n,
538
+ count(*) OVER () as k
539
+ FROM streak
540
+ WHERE ${crashDead(schema)}
541
+ ),
542
+ head AS (
543
+ SELECT crash_restarts as base, COALESCE(crash_restarts_since, heartbeat_on) as since, k
544
+ FROM crashed WHERE n = 1
545
+ ),
546
+ counts AS (
547
+ SELECT c.id, h.base + c.n - 1 as restarts, h.since
548
+ FROM crashed c, head h
549
+ WHERE c.n > 1
550
+ UNION ALL
551
+ SELECT me.id, COALESCE(h.base + h.k, 0), h.since
552
+ FROM me LEFT JOIN head h ON true
553
+ )
554
+ UPDATE ${schema}.instance i
555
+ SET crash_restarts = counts.restarts::int,
556
+ crash_restarts_since = counts.since
557
+ FROM counts
558
+ WHERE i.id = counts.id
559
+ `;
560
+ }
561
+ // When the rows this count could not yet judge would go quiet, or null when there are none: the
562
+ // registrar counts again then.
563
+ export function crashRecountAt(schema) {
564
+ return `
565
+ WITH ${crashSlot(schema)}
566
+ SELECT max(heartbeat_on + heartbeat_seconds * ${INSTANCE_QUIET_BEATS} * interval '1 second') as "recountAt"
567
+ FROM streak
568
+ WHERE NOT ${crashDead(schema)}
569
+ `;
570
+ }
571
+ // Instance registry statements. $1 is the instance id throughout. Registering replaces the row, so a
572
+ // PgBoss object restarted after stop() reads as live again from a new started_on. A heartbeat is an
573
+ // upsert too: a row pruned while its process was paused (a laptop asleep past the retention) comes
574
+ // back on the next beat, keeping the started_on the instance registered with ($20).
575
+ const INSTANCE_COLUMNS = `id, name, host, pid, version, node_version, heartbeat_seconds,
576
+ supervise, schedule, migrate, persist_queue_stats, persist_warnings,
577
+ pool_max, pool_total, pool_idle, pool_waiting, workers, metrics, config, application_name, started_on, heartbeat_on`;
578
+ const INSTANCE_VALUES = `$1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17::text::jsonb, $18::text::jsonb, $19::text::jsonb,
579
+ current_setting('application_name')`;
580
+ const INSTANCE_BEAT = `pool_max = EXCLUDED.pool_max,
581
+ pool_total = EXCLUDED.pool_total,
582
+ pool_idle = EXCLUDED.pool_idle,
583
+ pool_waiting = EXCLUDED.pool_waiting,
584
+ workers = EXCLUDED.workers,
585
+ metrics = EXCLUDED.metrics,
586
+ heartbeat_on = EXCLUDED.heartbeat_on`;
587
+ export function registerInstance(schema) {
588
+ return `
589
+ INSERT INTO ${schema}.instance (${INSTANCE_COLUMNS})
590
+ VALUES (${INSTANCE_VALUES}, ${schema}.job_now(), ${schema}.job_now())
591
+ ON CONFLICT (id) DO UPDATE SET
592
+ name = EXCLUDED.name,
593
+ host = EXCLUDED.host,
594
+ pid = EXCLUDED.pid,
595
+ version = EXCLUDED.version,
596
+ node_version = EXCLUDED.node_version,
597
+ heartbeat_seconds = EXCLUDED.heartbeat_seconds,
598
+ supervise = EXCLUDED.supervise,
599
+ schedule = EXCLUDED.schedule,
600
+ migrate = EXCLUDED.migrate,
601
+ persist_queue_stats = EXCLUDED.persist_queue_stats,
602
+ persist_warnings = EXCLUDED.persist_warnings,
603
+ application_name = EXCLUDED.application_name,
604
+ config = EXCLUDED.config,
605
+ started_on = EXCLUDED.started_on,
606
+ stopped_on = NULL,
607
+ crash_restarts = 0,
608
+ crash_restarts_since = NULL,
609
+ ${INSTANCE_BEAT}
610
+ RETURNING started_on as "startedOn"
611
+ `;
612
+ }
613
+ export function heartbeatInstance(schema) {
614
+ return `
615
+ INSERT INTO ${schema}.instance (${INSTANCE_COLUMNS})
616
+ VALUES (${INSTANCE_VALUES}, $20::timestamptz, ${schema}.job_now())
617
+ ON CONFLICT (id) DO UPDATE SET
618
+ ${INSTANCE_BEAT}
619
+ `;
620
+ }
621
+ export function stopInstance(schema) {
622
+ return `UPDATE ${schema}.instance SET stopped_on = ${schema}.job_now() WHERE id = $1`;
623
+ }
624
+ // Rows whose heartbeat has not moved for the retention, stopped or quiet alike. A live instance's
625
+ // row never qualifies, since its heartbeat keeps moving.
626
+ export function deleteOldInstances(schema, days) {
627
+ return `
628
+ DELETE FROM ${schema}.instance
629
+ WHERE heartbeat_on < ${schema}.job_now() - interval '${days} days'
630
+ `;
631
+ }
632
+ export function getInstances(schema) {
633
+ return `
634
+ SELECT
635
+ id,
636
+ name,
637
+ host,
638
+ pid,
639
+ version,
640
+ node_version as "nodeVersion",
641
+ application_name as "applicationName",
642
+ heartbeat_seconds as "heartbeatSeconds",
643
+ supervise,
644
+ schedule,
645
+ migrate,
646
+ persist_queue_stats as "persistQueueStats",
647
+ persist_warnings as "persistWarnings",
648
+ pool_max as "poolMax",
649
+ pool_total as "poolTotal",
650
+ pool_idle as "poolIdle",
651
+ pool_waiting as "poolWaiting",
652
+ workers,
653
+ metrics,
654
+ config,
655
+ crash_restarts as "crashRestarts",
656
+ crash_restarts_since as "crashRestartsSince",
657
+ started_on as "startedOn",
658
+ heartbeat_on as "heartbeatOn",
659
+ stopped_on as "stoppedOn",
660
+ stopped_on IS NULL AND heartbeat_on >= ${schema}.job_now() - heartbeat_seconds * ${INSTANCE_QUIET_BEATS} * interval '1 second' as live
661
+ FROM ${schema}.instance
662
+ ORDER BY started_on, id
663
+ `;
664
+ }
438
665
  export function createIndexJobDependencyParent(schema) {
439
666
  return `CREATE INDEX IF NOT EXISTS job_dep_parent_idx ON ${schema}.job_dependency (parent_name, parent_id)`;
440
667
  }
668
+ /* eslint-disable no-restricted-syntax -- column defaults stay on the real clock: every pg-boss write names its timestamps through job_now() */
669
+ // One row per PgBoss object, written by the object itself at start() and on each heartbeat, so a
670
+ // database can say which instances share it and what each is doing. A crashed process never sets
671
+ // stopped_on; it goes quiet instead, when heartbeat_on stops moving, and heartbeat_seconds says how
672
+ // long that takes for this row, since the interval is per instance. workers is one entry per work()
673
+ // call. The pool columns are node-postgres's counts at the last heartbeat, null for a pool pg-boss
674
+ // was handed and cannot read. metrics is the process's CPU, memory and event loop at the last
675
+ // heartbeat against its container's limits (see nurse.ts), null until the first sample lands.
676
+ // config is the options it runs with, as the registrar's allowlist picks them, for comparing instances.
677
+ // application_name is the one the registering session carried, so pg_stat_activity joins to the row
678
+ // wherever it is unique.
679
+ export function createTableInstance(schema) {
680
+ return `
681
+ CREATE TABLE ${schema}.instance (
682
+ id uuid PRIMARY KEY,
683
+ name text,
684
+ host text NOT NULL,
685
+ pid int NOT NULL,
686
+ version text NOT NULL,
687
+ node_version text NOT NULL,
688
+ application_name text,
689
+ heartbeat_seconds int NOT NULL,
690
+ supervise bool NOT NULL,
691
+ schedule bool NOT NULL,
692
+ migrate bool NOT NULL,
693
+ persist_queue_stats bool NOT NULL,
694
+ persist_warnings bool NOT NULL,
695
+ pool_max int,
696
+ pool_total int,
697
+ pool_idle int,
698
+ pool_waiting int,
699
+ workers jsonb NOT NULL DEFAULT '[]'::jsonb,
700
+ metrics jsonb,
701
+ config jsonb NOT NULL DEFAULT '{}'::jsonb,
702
+ crash_restarts int NOT NULL DEFAULT 0,
703
+ crash_restarts_since timestamptz,
704
+ started_on timestamptz NOT NULL DEFAULT now(),
705
+ heartbeat_on timestamptz NOT NULL DEFAULT now(),
706
+ stopped_on timestamptz
707
+ )
708
+ `;
709
+ }
710
+ /* eslint-enable no-restricted-syntax */
441
711
  // Anchored so a schema name that itself contains these substrings (e.g. `job_intake`) isn't
442
712
  // mangled: `\.job\y` matches only the base table reference (`schema.job`, not `schema.job_i5` whose
443
713
  // `job` is followed by `_`, nor `.job_dependency`), and `\yjob_i(\d+)` matches only the bare
@@ -583,7 +853,8 @@ function createTableJob(schema, noPartitioning = false) {
583
853
  source_created_on timestamp with time zone,
584
854
  source_retry_count int,
585
855
  source_output jsonb,
586
- source_root_id uuid
856
+ source_root_id uuid,
857
+ trace_context jsonb
587
858
  ) ${partitionClause}
588
859
  `;
589
860
  }
@@ -1178,6 +1449,9 @@ export function updateQueue(schema) {
1178
1449
  WHERE name = $1
1179
1450
  `;
1180
1451
  }
1452
+ export function currentDatabase() {
1453
+ return 'SELECT current_database() AS name';
1454
+ }
1181
1455
  export function getQueues(schema, names) {
1182
1456
  const hasNames = names && names.length > 0;
1183
1457
  return {
@@ -1197,6 +1471,7 @@ export function getQueues(schema, names) {
1197
1471
  q.notify,
1198
1472
  q.dead_letter as "deadLetter",
1199
1473
  q.deferred_count as "deferredCount",
1474
+ q.blocked_count as "blockedCount",
1200
1475
  q.warning_queued as "warningQueueSize",
1201
1476
  q.queued_count as "queuedCount",
1202
1477
  q.ready_count as "readyCount",
@@ -1208,6 +1483,9 @@ export function getQueues(schema, names) {
1208
1483
  q.failed_delta as "failedDelta",
1209
1484
  q.delta_seconds as "deltaSeconds",
1210
1485
  q.delta_on as "deltaOn",
1486
+ q.wait_bins as "waitBins",
1487
+ q.run_bins as "runBins",
1488
+ q.ready_oldest_seconds as "readyOldestSeconds",
1211
1489
  q.singletons_active as "singletonsActive",
1212
1490
  q.table_name as "table",
1213
1491
  q.created_on as "createdOn",
@@ -1239,6 +1517,30 @@ export function deleteStoredJobs(schema, table) {
1239
1517
  export function truncateTable(schema, table) {
1240
1518
  return `TRUNCATE ${schema}.${table}`;
1241
1519
  }
1520
+ // The cached counts of a queue whose table was just truncated, written without counting: they are
1521
+ // zero. Only the truncate paths use it. A DELETE leaves concurrent sends and the other states in
1522
+ // place, so its counts can only come from a recount, whose cost the delete cannot bound. The
1523
+ // throughput counters are the monitor's and are left alone. With `one`, $1 is the queue's name.
1524
+ export function zeroQueueStats(schema, one) {
1525
+ return `
1526
+ UPDATE ${schema}.queue SET
1527
+ deferred_count = 0,
1528
+ blocked_count = 0,
1529
+ queued_count = 0,
1530
+ ready_count = 0,
1531
+ active_count = 0,
1532
+ failed_count = 0,
1533
+ total_count = 0,
1534
+ singletons_active = NULL,
1535
+ monitor_on = ${schema}.job_now()
1536
+ FROM (
1537
+ SELECT name
1538
+ FROM ${schema}.queue${one ? '\n WHERE name = $1' : ''}
1539
+ ${queueRowLock()}
1540
+ ) q
1541
+ WHERE queue.name = q.name
1542
+ `;
1543
+ }
1242
1544
  export function deleteAllJobs(schema, table) {
1243
1545
  return `DELETE from ${schema}.${table} WHERE name = $1`;
1244
1546
  }
@@ -1285,7 +1587,7 @@ export function setScheduleLastJobIds(schema) {
1285
1587
  return `
1286
1588
  UPDATE ${schema}.schedule s
1287
1589
  SET last_job_id = x."jobId"
1288
- FROM json_to_recordset($1::json) AS x (name text, key text, "jobId" uuid)
1590
+ FROM json_to_recordset($1::text::json) AS x (name text, key text, "jobId" uuid)
1289
1591
  WHERE s.name = x.name
1290
1592
  AND COALESCE(s.key, '') = x.key
1291
1593
  `;
@@ -1305,7 +1607,7 @@ export function setScheduleLastJobIds(schema) {
1305
1607
  export function setScheduleKinds(schema) {
1306
1608
  return `
1307
1609
  UPDATE ${schema}.schedule s SET kind = k.kind
1308
- FROM json_to_recordset($1::json) as k (name text, key text, kind text, cron text)
1610
+ FROM json_to_recordset($1::text::json) as k (name text, key text, kind text, cron text)
1309
1611
  WHERE s.name = k.name
1310
1612
  AND COALESCE(s.key, '') = k.key
1311
1613
  AND s.cron = k.cron
@@ -1359,7 +1661,7 @@ export function getTime(schema) {
1359
1661
  export function insertWarning(schema) {
1360
1662
  return `
1361
1663
  INSERT INTO ${schema}.warning (type, message, data, created_on)
1362
- VALUES ($1, $2, $3, ${schema}.job_now())
1664
+ VALUES ($1, $2, $3::text::jsonb, ${schema}.job_now())
1363
1665
  `;
1364
1666
  }
1365
1667
  export function getWarnings(schema) {
@@ -1410,6 +1712,9 @@ export function createTableQueueStats(schema, noPartitioning = false) {
1410
1712
  failed_delta int,
1411
1713
  delta_seconds int,
1412
1714
  delta_on timestamptz,
1715
+ wait_bins int[],
1716
+ run_bins int[],
1717
+ ready_oldest_seconds int,
1413
1718
  captured_on timestamptz NOT NULL DEFAULT now(),
1414
1719
  ${noPartitioning ? 'PRIMARY KEY (id)' : 'PRIMARY KEY (id, captured_on)'}
1415
1720
  ) ${noPartitioning ? '' : 'PARTITION BY RANGE (captured_on)'}
@@ -1502,9 +1807,11 @@ export function insertQueueStats(schema, queues, noAdvisoryLocks) {
1502
1807
  const sql = `
1503
1808
  INSERT INTO ${schema}.queue_stats
1504
1809
  (name, deferred_count, queued_count, ready_count, active_count, failed_count, total_count,
1505
- created_delta, completed_delta, failed_delta, delta_seconds, delta_on, captured_on)
1810
+ created_delta, completed_delta, failed_delta, delta_seconds, delta_on,
1811
+ wait_bins, run_bins, ready_oldest_seconds, captured_on)
1506
1812
  SELECT name, deferred_count, queued_count, ready_count, active_count, failed_count, total_count,
1507
- created_delta, completed_delta, failed_delta, delta_seconds, delta_on, ${schema}.job_now()
1813
+ created_delta, completed_delta, failed_delta, delta_seconds, delta_on,
1814
+ wait_bins, run_bins, ready_oldest_seconds, ${schema}.job_now()
1508
1815
  FROM ${schema}.queue
1509
1816
  WHERE name = ANY(${serializeArrayParam(queues)})
1510
1817
  `;
@@ -1537,6 +1844,9 @@ export function getQueueStatsCache(schema) {
1537
1844
  failed_delta as "failedDelta",
1538
1845
  delta_seconds as "deltaSeconds",
1539
1846
  delta_on as "deltaOn",
1847
+ wait_bins as "waitBins",
1848
+ run_bins as "runBins",
1849
+ ready_oldest_seconds as "readyOldestSeconds",
1540
1850
  table_name as "table",
1541
1851
  monitor_on as "capturedOn",
1542
1852
  (extract(epoch from (${schema}.job_now() - monitor_on)) * 1000)::float8 as "cacheAgeMs",
@@ -1560,6 +1870,9 @@ export function getQueueStatsHistory(schema) {
1560
1870
  failed_delta as "failedDelta",
1561
1871
  delta_seconds as "deltaSeconds",
1562
1872
  delta_on as "deltaOn",
1873
+ wait_bins as "waitBins",
1874
+ run_bins as "runBins",
1875
+ ready_oldest_seconds as "readyOldestSeconds",
1563
1876
  captured_on as "capturedOn"
1564
1877
  FROM ${schema}.queue_stats
1565
1878
  WHERE name = $1
@@ -1613,6 +1926,11 @@ const STATS_AGG = {
1613
1926
  // YugabyteDB, none of which can rely on it. to_timestamp / extract(epoch) / floor exist on all of
1614
1927
  // them (extract returns double on PG13, numeric on PG14+; floor/division handle both identically),
1615
1928
  // and buckets align to the Unix epoch so their boundaries are stable across calls.
1929
+ //
1930
+ // Wait and run histograms are added up per returned bucket in SQL (passes, slots, histograms), so a
1931
+ // bucket comes back as one histogram however many passes it covers. Each pass lands in the bucket
1932
+ // its counters were placed in, then its counts are summed per bucket and slot, a slot with no job in
1933
+ // any pass as 0. A bucket no pass measured has no histogram row and comes back null.
1616
1934
  export function getQueueStatsHistoryBucketed(schema, aggregate, mode) {
1617
1935
  const agg = STATS_AGG[aggregate];
1618
1936
  const widthCte = mode === 'auto'
@@ -1628,7 +1946,7 @@ export function getQueueStatsHistoryBucketed(schema, aggregate, mode) {
1628
1946
  FROM extent
1629
1947
  ),
1630
1948
  w AS (
1631
- SELECT greatest(1, ceil(extract(epoch from (hi - lo))::float8 / greatest($5, 1)::float8)::bigint)::bigint AS secs
1949
+ SELECT greatest(1, ceil(extract(epoch from (hi - lo))::float8 / greatest($5, 1)::float8)::bigint) AS secs
1632
1950
  FROM bounds
1633
1951
  )`
1634
1952
  : 'WITH w AS (SELECT greatest($5, 1)::bigint AS secs)';
@@ -1663,7 +1981,8 @@ export function getQueueStatsHistoryBucketed(schema, aggregate, mode) {
1663
1981
  sum(completed_delta)::int as "completedDelta",
1664
1982
  sum(failed_delta)::int as "failedDelta",
1665
1983
  sum(delta_seconds)::int as "deltaSeconds",
1666
- max(delta_on) as "deltaOn"
1984
+ max(delta_on) as "deltaOn",
1985
+ max(ready_oldest_seconds) as "readyOldestSeconds"
1667
1986
  FROM ${schema}.queue_stats, w
1668
1987
  WHERE name = $1
1669
1988
  AND delta_on IS NOT NULL
@@ -1687,10 +2006,37 @@ export function getQueueStatsHistoryBucketed(schema, aggregate, mode) {
1687
2006
  c."completedDelta",
1688
2007
  c."failedDelta",
1689
2008
  c."deltaSeconds",
1690
- c."deltaOn"
2009
+ c."deltaOn",
2010
+ c."readyOldestSeconds",
2011
+ c.bucket as "counterBucket"
1691
2012
  FROM gauges g
1692
2013
  FULL JOIN counters c ON c.bucket = g.bucket
1693
- )
2014
+ ),
2015
+ passes AS (
2016
+ SELECT ${bucket('delta_on')} as "counterBucket", wait_bins, run_bins
2017
+ FROM ${schema}.queue_stats, w
2018
+ WHERE name = $1
2019
+ AND delta_on IS NOT NULL
2020
+ AND wait_bins IS NOT NULL
2021
+ AND ($2::timestamptz IS NULL OR delta_on >= $2)
2022
+ AND ($3::timestamptz IS NULL OR delta_on <= $3)
2023
+ ),
2024
+ slots AS (
2025
+ SELECT p.bucket, u.slot, coalesce(sum(u.w), 0)::int AS w, coalesce(sum(u.r), 0)::int AS r
2026
+ FROM (SELECT DISTINCT bucket, "counterBucket" FROM placed) p
2027
+ JOIN passes ps ON ps."counterBucket" = p."counterBucket",
2028
+ unnest(ps.wait_bins, ps.run_bins) WITH ORDINALITY AS u(w, r, slot)
2029
+ GROUP BY 1, 2
2030
+ ),
2031
+ histograms AS (
2032
+ SELECT
2033
+ bucket,
2034
+ array_agg(w ORDER BY slot) as "waitBins",
2035
+ array_agg(r ORDER BY slot) as "runBins"
2036
+ FROM slots
2037
+ GROUP BY 1
2038
+ ),
2039
+ folded AS (
1694
2040
  SELECT
1695
2041
  bucket as "capturedOn",
1696
2042
  max("deferredCount")::int as "deferredCount",
@@ -1703,11 +2049,16 @@ export function getQueueStatsHistoryBucketed(schema, aggregate, mode) {
1703
2049
  sum("completedDelta")::int as "completedDelta",
1704
2050
  sum("failedDelta")::int as "failedDelta",
1705
2051
  sum("deltaSeconds")::int as "deltaSeconds",
1706
- max("deltaOn") as "deltaOn"
2052
+ max("deltaOn") as "deltaOn",
2053
+ max("readyOldestSeconds")::int as "readyOldestSeconds"
1707
2054
  FROM placed
1708
2055
  WHERE bucket IS NOT NULL
1709
2056
  GROUP BY 1
1710
- ORDER BY 1 DESC
2057
+ )
2058
+ SELECT f.*, h."waitBins", h."runBins"
2059
+ FROM folded f
2060
+ LEFT JOIN histograms h ON h.bucket = f."capturedOn"
2061
+ ORDER BY f."capturedOn" DESC
1711
2062
  LIMIT ${limit}
1712
2063
  `;
1713
2064
  }
@@ -1814,7 +2165,7 @@ function buildFetchParams(options) {
1814
2165
  * exceeds fetch time.
1815
2166
  */
1816
2167
  export function fetchNextJob(options, noSkipLocked = false) {
1817
- const { schema, table, name, policy, limit, includeMetadata, ignoreStartAfter = false, groupConcurrency, minPriority, maxPriority } = options;
2168
+ const { schema, table, name, policy, limit, includeMetadata, includeTraceContext = false, ignoreStartAfter = false, groupConcurrency, minPriority, maxPriority } = options;
1818
2169
  const keyStrictFifo = policy === QUEUE_POLICIES.key_strict_fifo;
1819
2170
  const singletonFetch = limit > 1 && (policy === QUEUE_POLICIES.singleton || policy === QUEUE_POLICIES.stately);
1820
2171
  const hasIgnoreSingletons = options.ignoreSingletons != null && options.ignoreSingletons.length > 0;
@@ -2002,7 +2353,7 @@ export function fetchNextJob(options, noSkipLocked = false) {
2002
2353
  WHERE name = '${name}' AND ${updateMatch}
2003
2354
  ${singletonFetch && !hasGroupConcurrency ? 'AND singleton_rn = 1' : ''}
2004
2355
  ${distributedStateCheck}
2005
- RETURNING j.${includeMetadata ? JOB_COLUMNS_ALL : JOB_COLUMNS_MIN}
2356
+ RETURNING j.${includeMetadata ? JOB_COLUMNS_ALL : JOB_COLUMNS_MIN}${includeTraceContext ? ', j.trace_context as "__traceContext"' : ''}
2006
2357
  `,
2007
2358
  values: params.values
2008
2359
  };
@@ -2084,10 +2435,16 @@ function lockedChildrenCte(schema) {
2084
2435
  FOR UPDATE OF j
2085
2436
  )`;
2086
2437
  }
2438
+ // A child released by its last parent has its start_after moved up to the release, so its wait (in
2439
+ // the monitor's histograms and ready_oldest_seconds) counts from when it could first run rather than
2440
+ // from when the flow was sent. A start_after still in the future is kept.
2087
2441
  function unblockChildrenUpdate(schema) {
2088
2442
  return `UPDATE ${schema}.job j
2089
2443
  SET pending_dependencies = GREATEST(j.pending_dependencies - lc.n, 0),
2090
- blocked = GREATEST(j.pending_dependencies - lc.n, 0) > 0
2444
+ blocked = GREATEST(j.pending_dependencies - lc.n, 0) > 0,
2445
+ start_after = CASE WHEN GREATEST(j.pending_dependencies - lc.n, 0) = 0
2446
+ THEN GREATEST(j.start_after, ${schema}.job_now())
2447
+ ELSE j.start_after END
2091
2448
  FROM locked_children lc
2092
2449
  WHERE j.name = lc.name
2093
2450
  AND j.id = lc.id`;
@@ -2161,12 +2518,16 @@ export function cancelJobs(schema, table, fenced) {
2161
2518
  ${settledCountAndIds()}
2162
2519
  `;
2163
2520
  }
2521
+ // A resumed job's start_after moves up to now, as a released flow child's does, so its wait (in the
2522
+ // monitor's histograms and ready_oldest_seconds) counts from when it could run again rather than
2523
+ // from when it was first sent. A start_after still in the future is kept.
2164
2524
  export function resumeJobs(schema, table) {
2165
2525
  return `
2166
2526
  WITH results as (
2167
2527
  UPDATE ${schema}.${table}
2168
2528
  SET completed_on = NULL,
2169
- state = '${JOB_STATES.created}'
2529
+ state = '${JOB_STATES.created}',
2530
+ start_after = GREATEST(start_after, ${schema}.job_now())
2170
2531
  WHERE name = $1
2171
2532
  AND id = ANY($2::uuid[])
2172
2533
  AND state = '${JOB_STATES.cancelled}'
@@ -2241,7 +2602,8 @@ export function insertJobs(schema, { table, name, returnId = true, notify = fals
2241
2602
  heartbeat_seconds,
2242
2603
  blocked,
2243
2604
  blocking,
2244
- pending_dependencies
2605
+ pending_dependencies,
2606
+ trace_context
2245
2607
  )
2246
2608
  SELECT
2247
2609
  COALESCE(id, gen_random_uuid()) as id,
@@ -2270,7 +2632,8 @@ export function insertJobs(schema, { table, name, returnId = true, notify = fals
2270
2632
  COALESCE("heartbeatSeconds", q.heartbeat_seconds) as heartbeat_seconds,
2271
2633
  COALESCE(blocked, false) as blocked,
2272
2634
  COALESCE(blocking, false) as blocking,
2273
- COALESCE("pendingDependencies", 0) as pending_dependencies
2635
+ COALESCE("pendingDependencies", 0) as pending_dependencies,
2636
+ "__traceContext" as trace_context
2274
2637
  FROM (
2275
2638
  SELECT *,
2276
2639
  CASE
@@ -2299,7 +2662,8 @@ export function insertJobs(schema, { table, name, returnId = true, notify = fals
2299
2662
  "heartbeatSeconds" integer,
2300
2663
  blocked boolean,
2301
2664
  blocking boolean,
2302
- "pendingDependencies" integer
2665
+ "pendingDependencies" integer,
2666
+ "__traceContext" jsonb
2303
2667
  )
2304
2668
  ) j
2305
2669
  JOIN ${schema}.queue q ON q.name = '${name}'
@@ -2455,7 +2819,8 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
2455
2819
  source_created_on,
2456
2820
  source_retry_count,
2457
2821
  source_output,
2458
- source_root_id
2822
+ source_root_id,
2823
+ trace_context
2459
2824
  )
2460
2825
  SELECT
2461
2826
  id,
@@ -2501,7 +2866,8 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
2501
2866
  source_created_on,
2502
2867
  source_retry_count,
2503
2868
  source_output,
2504
- source_root_id
2869
+ source_root_id,
2870
+ trace_context
2505
2871
  FROM deleted_jobs
2506
2872
  ON CONFLICT DO NOTHING
2507
2873
  RETURNING *
@@ -2542,7 +2908,8 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
2542
2908
  source_created_on,
2543
2909
  source_retry_count,
2544
2910
  source_output,
2545
- source_root_id
2911
+ source_root_id,
2912
+ trace_context
2546
2913
  )
2547
2914
  SELECT
2548
2915
  id,
@@ -2579,7 +2946,8 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
2579
2946
  source_created_on,
2580
2947
  source_retry_count,
2581
2948
  source_output,
2582
- source_root_id
2949
+ source_root_id,
2950
+ trace_context
2583
2951
  FROM deleted_jobs
2584
2952
  WHERE id NOT IN (SELECT id from retried_jobs)
2585
2953
  RETURNING *
@@ -2592,7 +2960,7 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
2592
2960
  dlq_jobs as (
2593
2961
  INSERT INTO ${schema}.job (name, priority, data, retry_limit, retry_backoff, retry_delay, start_after, created_on, keep_until, deletion_seconds,
2594
2962
  expire_seconds, singleton_key, group_id, group_tier, heartbeat_seconds,
2595
- source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id)
2963
+ source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id, trace_context)
2596
2964
  SELECT
2597
2965
  r.dead_letter,
2598
2966
  r.priority,
@@ -2614,7 +2982,8 @@ function failJobsBody(schema, table, where, output, forceTerminal = false) {
2614
2982
  r.created_on,
2615
2983
  r.retry_count,
2616
2984
  r.output,
2617
- COALESCE(r.source_root_id, r.id)
2985
+ COALESCE(r.source_root_id, r.id),
2986
+ r.trace_context
2618
2987
  FROM results r
2619
2988
  JOIN ${schema}.queue q ON q.name = r.dead_letter
2620
2989
  WHERE state = '${JOB_STATES.failed}'
@@ -2815,10 +3184,10 @@ export function insertRetryJob(schema, table) {
2815
3184
  group_id, group_tier, expire_seconds, deletion_seconds, created_on, completed_on,
2816
3185
  keep_until, policy, output, dead_letter,
2817
3186
  heartbeat_on, heartbeat_seconds, blocked, blocking, pending_dependencies,
2818
- source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id
3187
+ source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id, trace_context
2819
3188
  ) VALUES (
2820
- $1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24,
2821
- $25, $26, $27, $28, $29, $30, $31, $32, $33, $34, $35
3189
+ $1, $2, $3, $4::text::jsonb, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22,
3190
+ $23::text::jsonb, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34::text::jsonb, $35, $36::text::jsonb
2822
3191
  ) ON CONFLICT DO NOTHING
2823
3192
  RETURNING id
2824
3193
  `;
@@ -2827,10 +3196,10 @@ export function insertDeadLetterJob(schema) {
2827
3196
  return `
2828
3197
  INSERT INTO ${schema}.job (name, data, priority, retry_limit, retry_backoff, retry_delay, start_after, created_on, keep_until, deletion_seconds,
2829
3198
  expire_seconds, singleton_key, group_id, group_tier, heartbeat_seconds,
2830
- source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id)
2831
- SELECT $1, $2, $9, q.retry_limit, q.retry_backoff, q.retry_delay, ${schema}.job_now(), ${schema}.job_now(), ${schema}.job_now() + q.retention_seconds * interval '1s', q.deletion_seconds,
3199
+ source_name, source_id, source_created_on, source_retry_count, source_output, source_root_id, trace_context)
3200
+ SELECT $1, $2::text::jsonb, $9, q.retry_limit, q.retry_backoff, q.retry_delay, ${schema}.job_now(), ${schema}.job_now(), ${schema}.job_now() + q.retention_seconds * interval '1s', q.deletion_seconds,
2832
3201
  q.expire_seconds, $8, $10, $11, q.heartbeat_seconds,
2833
- $4, $5, $6, $7, $3, COALESCE($12::uuid, $5::uuid)
3202
+ $4, $5, $6, $7, $3::text::jsonb, COALESCE($12::uuid, $5::uuid), $13::text::jsonb
2834
3203
  FROM ${schema}.queue q WHERE q.name = $1
2835
3204
  `;
2836
3205
  }
@@ -2855,7 +3224,7 @@ function redriveWhere(schema, table) {
2855
3224
  AND k.state IN ('${JOB_STATES.active}', '${JOB_STATES.retry}', '${JOB_STATES.failed}')
2856
3225
  )
2857
3226
  AND ($3::text IS NULL OR j.source_name = $3)
2858
- AND ($4::jsonb IS NULL OR j.data @> $4::jsonb)
3227
+ AND ($4::text::jsonb IS NULL OR j.data @> $4::text::jsonb)
2859
3228
  AND ($5::timestamptz IS NULL OR j.created_on < $5)
2860
3229
  AND ($6::uuid[] IS NULL OR j.id = ANY($6::uuid[]))`;
2861
3230
  }
@@ -2869,13 +3238,13 @@ function redriveWhere(schema, table) {
2869
3238
  // Job-identity columns (priority, singleton_key, group_id, group_tier) are carried over instead.
2870
3239
  const REDRIVE_INSERT_COLUMNS = `(id, name, data, priority, retry_limit, retry_backoff, retry_delay, retry_delay_max,
2871
3240
  expire_seconds, start_after, created_on, keep_until, deletion_seconds, policy, singleton_key, group_id, group_tier,
2872
- heartbeat_seconds, dead_letter, source_root_id)`;
3241
+ heartbeat_seconds, dead_letter, source_root_id, trace_context)`;
2873
3242
  function redriveInsertValues(schema, newId, destination) {
2874
3243
  return `${newId}, COALESCE(${destination}, m.source_name), m.data, m.priority, q.retry_limit, q.retry_backoff,
2875
3244
  q.retry_delay, q.retry_delay_max, q.expire_seconds, ${schema}.job_now(), ${schema}.job_now(),
2876
3245
  ${schema}.job_now() + q.retention_seconds * interval '1s', q.deletion_seconds, q.policy,
2877
3246
  m.singleton_key, m.group_id, m.group_tier, q.heartbeat_seconds, q.dead_letter,
2878
- COALESCE(m.source_root_id, m.source_id)`;
3247
+ COALESCE(m.source_root_id, m.source_id), m.trace_context`;
2879
3248
  }
2880
3249
  // What a job that could not be re-created becomes: failed, in place, in the dead letter queue, with
2881
3250
  // the reason as its output. Never deleted: before 12.35 it was, and the job was simply gone. As a
@@ -3042,13 +3411,15 @@ export function deletion(schema, table, queues, noAdvisoryLocks) {
3042
3411
  `;
3043
3412
  return locked(schema, sql, table + 'deletion', noAdvisoryLocks);
3044
3413
  }
3414
+ // start_after moves up to now, as in resumeJobs.
3045
3415
  export function retryJobs(schema, table) {
3046
3416
  return `
3047
3417
  WITH results as (
3048
3418
  UPDATE ${schema}.job
3049
3419
  SET state = '${JOB_STATES.retry}',
3050
3420
  retry_limit = retry_limit + 1,
3051
- completed_on = NULL
3421
+ completed_on = NULL,
3422
+ start_after = GREATEST(start_after, ${schema}.job_now())
3052
3423
  WHERE name = $1
3053
3424
  AND id = ANY($2::uuid[])
3054
3425
  AND state = '${JOB_STATES.failed}'
@@ -3199,7 +3570,10 @@ function throughputAssignments(end, resetMax) {
3199
3570
  completed_delta = COALESCE(stats."completedDelta", 0),
3200
3571
  failed_delta = COALESCE(stats."failedDelta", 0),
3201
3572
  delta_seconds = CASE WHEN ${seconds} < 0 THEN 0 ELSE ${seconds} END,
3202
- delta_on = GREATEST(queue.delta_on, ${end}),`;
3573
+ delta_on = GREATEST(queue.delta_on, ${end}),
3574
+ wait_bins = COALESCE(stats."waitBins", ${EMPTY_BINS}),
3575
+ run_bins = COALESCE(stats."runBins", ${EMPTY_BINS}),
3576
+ ready_oldest_seconds = COALESCE(stats."readyOldestSeconds", 0),`;
3203
3577
  }
3204
3578
  // The windows a true-up may still revise, per queue: every recorded snapshot whose window ends
3205
3579
  // after the anchor `h`, up to the newest one, `top`, with the counters they hold between them.
@@ -3363,6 +3737,61 @@ export function trueUpQueueStats(schema, table, queues, noAdvisoryLocks, window
3363
3737
  `;
3364
3738
  return transaction(sql);
3365
3739
  }
3740
+ // Wait and run times, as the monitor records them: a histogram per counted pass, of the jobs that
3741
+ // finished in its window. Slot 0 holds times under 10 ms, slots 1 to LATENCY_BINS bins that each
3742
+ // grow by √2 (slot k runs from 10 ms · √2^(k-1) to 10 ms · √2^k), and the last slot everything past
3743
+ // about 23 hours. Log-spaced because the times span six orders of magnitude, and a percentile read
3744
+ // from them lies in the same bin as the exact one, a factor of √2 at most. Histograms rather than
3745
+ // percentiles, because histograms add: a reader sums them across passes, buckets or queues and
3746
+ // reads any percentile from the sum.
3747
+ // Stored as 48 slots in slot order, null where no job landed: Postgres keeps a null as one bit
3748
+ // rather than four bytes, and most slots are empty. getQueueStats() hands them out with nulls as 0.
3749
+ // Readers that add slots in SQL coalesce, since a null plus a count is null.
3750
+ export const LATENCY_BINS = 46;
3751
+ export const LATENCY_SLOTS = LATENCY_BINS + 2;
3752
+ export const LATENCY_MIN_SECONDS = 0.01;
3753
+ // A measured histogram in which no job finished: every slot null, not a null array.
3754
+ const EMPTY_BINS = `'{${new Array(LATENCY_SLOTS).fill('NULL').join(',')}}'::int[]`;
3755
+ // Both bins travel packed in one integer (wait * LATENCY_PACK + run), so the aggregate needs one
3756
+ // array and one filter. Two arrays, each with its own filter evaluated on every row of the table,
3757
+ // cost twice as much: measured on 2.5M rows, +55 ms on a ~560 ms pass packed, +80 to 120 ms apart.
3758
+ const LATENCY_PACK = 64;
3759
+ // date_part rather than extract: extract returns numeric on PostgreSQL 14 and later, and the numeric
3760
+ // arithmetic on every finished job was a large share of the histograms' cost. date_part is float8
3761
+ // on every supported backend, as the bin math needs anyway.
3762
+ const waitSeconds = "date_part('epoch', (j.started_on - GREATEST(j.created_on, j.start_after)))";
3763
+ const runSeconds = "date_part('epoch', (j.completed_on - j.started_on))";
3764
+ // The slot is width_bucket's, written out: CockroachDB's width_bucket takes decimals, not float8.
3765
+ // ln(t / 10 ms) over ln(√2), plus one, clamped to slot 0 below 10 ms and the last slot past the end.
3766
+ // A duration of zero or less (stamps from two clocks) lands in slot 0.
3767
+ function latencyBin(seconds) {
3768
+ const lo = Math.log(LATENCY_MIN_SECONDS);
3769
+ const width = Math.log(2) / 2;
3770
+ const raw = `floor((ln(GREATEST((${seconds})::float8, 0.001::float8)) - ${lo}::float8) / ${width}::float8)::int + 1`;
3771
+ return `LEAST(GREATEST(${raw}, 0), ${LATENCY_SLOTS - 1})`;
3772
+ }
3773
+ // The aggregate's array is unnested once per queue, in a lateral join after the aggregate, and
3774
+ // counted by packed value: at most LATENCY_SLOTS² rows (ps, ns) however many jobs finished. Both
3775
+ // histograms are read from those, so the per-job array is walked once rather than once each.
3776
+ function latencyCounts(packed) {
3777
+ return `LEFT JOIN LATERAL (
3778
+ SELECT array_agg(g.p) AS ps, array_agg(g.n) AS ns
3779
+ FROM (SELECT p, count(*)::int AS n FROM unnest(${packed}) AS u(p) GROUP BY p) g
3780
+ ) latency ON true`;
3781
+ }
3782
+ // One histogram from latencyCounts: every slot in slot order, null where no job landed. A pass in
3783
+ // which nothing finished still writes the 48 slots (unnest of a null array is no rows, and the left
3784
+ // join keeps every slot), so a pass that counted says so, as its deltas do. floor() of a float
3785
+ // division rather than integer division, which CockroachDB answers in decimal.
3786
+ function latencySlotOf(which) {
3787
+ return which === 'wait' ? `floor(p / ${LATENCY_PACK}.0)::int` : `(p % ${LATENCY_PACK})::int`;
3788
+ }
3789
+ function latencyHistogram(which) {
3790
+ return `(SELECT array_agg(c.n ORDER BY s.slot)
3791
+ FROM generate_series(0, ${LATENCY_SLOTS - 1}) AS s(slot)
3792
+ LEFT JOIN (SELECT ${latencySlotOf(which)} AS slot, sum(n)::int AS n
3793
+ FROM unnest(latency.ps, latency.ns) AS u(p, n) GROUP BY 1) c ON c.slot = s.slot)`;
3794
+ }
3366
3795
  // Every count the monitor keeps, from one pass over the queue's table.
3367
3796
  //
3368
3797
  // Six of them are gauges — what the queue looks like right now. Three are not:
@@ -3386,6 +3815,13 @@ export function trueUpQueueStats(schema, table, queues, noAdvisoryLocks, window
3386
3815
  // filtered on completed_on, would be a whole extra scan of the largest table in
3387
3816
  // the schema, and there is no index on that column to make it cheaper.
3388
3817
  export function getQueueStats(schema, table, queues, throughput = false, window = {}) {
3818
+ // A queued job is exactly one of blocked (waiting on a flow parent), deferred (start_after still
3819
+ // ahead) or ready, so the three add up to queuedCount. Blocked wins over deferred: a job whose
3820
+ // parent has not finished cannot run when its start_after comes round.
3821
+ const queued = `j.state < '${JOB_STATES.active}'`;
3822
+ const blocked = `${queued} AND j.blocked`;
3823
+ const deferred = `${queued} AND NOT j.blocked AND j.start_after > ${schema}.job_now()`;
3824
+ const ready = `${queued} AND NOT j.blocked AND j.start_after <= ${schema}.job_now()`;
3389
3825
  // Counted only with persistQueueStats. Otherwise the aggregate does what it did before throughput
3390
3826
  // existed: no join, no extra counts, no cost. The measured price is in the `persistQueueStats` docs.
3391
3827
  const end = deltaWindowEnd(schema, window.lag);
@@ -3400,11 +3836,18 @@ export function getQueueStats(schema, table, queues, throughput = false, window
3400
3836
  "createdDelta",
3401
3837
  "completedDelta",
3402
3838
  "failedDelta",
3839
+ ${latencyHistogram('wait')} as "waitBins",
3840
+ ${latencyHistogram('run')} as "runBins",
3841
+ "readyOldestSeconds",
3403
3842
  COALESCE("recount" > "settled", false) as "trueUp",`,
3404
3843
  counts: `
3405
3844
  (count(*) FILTER (WHERE ${inWindow('created_on')}))::int as "createdDelta",
3406
3845
  (count(*) FILTER (WHERE j.state = '${JOB_STATES.completed}' AND ${inWindow('completed_on')}))::int as "completedDelta",
3407
3846
  (count(*) FILTER (WHERE j.state = '${JOB_STATES.failed}' AND ${inWindow('completed_on')}))::int as "failedDelta",
3847
+ array_agg(${latencyBin(waitSeconds)} * ${LATENCY_PACK} + ${latencyBin(runSeconds)})
3848
+ FILTER (WHERE j.state IN ('${JOB_STATES.completed}', '${JOB_STATES.failed}') AND j.started_on IS NOT NULL AND ${inWindow('completed_on')}) as "latencyBins",
3849
+ round(extract(epoch from (${schema}.job_now() - min(GREATEST(j.created_on, j.start_after))
3850
+ FILTER (WHERE ${ready}))))::int as "readyOldestSeconds",
3408
3851
  sum(
3409
3852
  CASE WHEN ${settled('created_on')} THEN 1 ELSE 0 END +
3410
3853
  CASE WHEN ${settled('completed_on')} AND j.state IN ('${JOB_STATES.completed}', '${JOB_STATES.failed}') THEN 1 ELSE 0 END
@@ -3414,16 +3857,18 @@ export function getQueueStats(schema, table, queues, throughput = false, window
3414
3857
  SELECT q.name, q.delta_on, a.h, t.top, t.settled
3415
3858
  FROM ${schema}.queue q${trueUpSettled(schema, 'q', window.trueUpMax)}
3416
3859
  WHERE q.name = ANY($1::text[])
3417
- ) q ON q.name = j.name`
3860
+ ) q ON q.name = j.name`,
3861
+ lateral: latencyCounts('stats."latencyBins"')
3418
3862
  }
3419
- : { select: '', counts: '', join: '' };
3863
+ : { select: '', counts: '', join: '', lateral: '' };
3420
3864
  return {
3421
3865
  text: `
3422
3866
  SELECT
3423
3867
  name,
3424
3868
  "deferredCount",
3425
3869
  "queuedCount",
3426
- GREATEST("queuedCount" - "deferredCount", 0) as "readyCount",
3870
+ "readyCount",
3871
+ "blockedCount",
3427
3872
  "activeCount",
3428
3873
  "failedCount",
3429
3874
  "totalCount",${counters.select}
@@ -3431,8 +3876,10 @@ export function getQueueStats(schema, table, queues, throughput = false, window
3431
3876
  FROM (
3432
3877
  SELECT
3433
3878
  j.name,
3434
- (count(*) FILTER (WHERE j.start_after > ${schema}.job_now() AND j.state < '${JOB_STATES.active}'))::int as "deferredCount",
3435
- (count(*) FILTER (WHERE j.state < '${JOB_STATES.active}'))::int as "queuedCount",
3879
+ (count(*) FILTER (WHERE ${deferred}))::int as "deferredCount",
3880
+ (count(*) FILTER (WHERE ${blocked}))::int as "blockedCount",
3881
+ (count(*) FILTER (WHERE ${ready}))::int as "readyCount",
3882
+ (count(*) FILTER (WHERE ${queued}))::int as "queuedCount",
3436
3883
  (count(*) FILTER (WHERE j.state = '${JOB_STATES.active}'))::int as "activeCount",
3437
3884
  (count(*) FILTER (WHERE j.state = '${JOB_STATES.failed}'))::int as "failedCount",
3438
3885
  count(*)::int as "totalCount",${counters.counts}
@@ -3442,6 +3889,7 @@ export function getQueueStats(schema, table, queues, throughput = false, window
3442
3889
  WHERE j.name = ANY($1::text[])
3443
3890
  GROUP BY 1
3444
3891
  ) stats
3892
+ ${counters.lateral}
3445
3893
  `,
3446
3894
  values: [queues]
3447
3895
  };
@@ -3491,6 +3939,7 @@ export function cacheQueueStats(schema, table, queues, noAdvisoryLocks, throughp
3491
3939
  WITH ${lock.cte}stats AS (SELECT * FROM (${statsText}) agg WHERE true${lock.guard})
3492
3940
  UPDATE ${schema}.queue SET
3493
3941
  deferred_count = COALESCE(stats."deferredCount", 0),
3942
+ blocked_count = COALESCE(stats."blockedCount", 0),
3494
3943
  queued_count = COALESCE(stats."queuedCount", 0),
3495
3944
  ready_count = COALESCE(stats."readyCount", 0),
3496
3945
  active_count = COALESCE(stats."activeCount", 0),
@@ -3566,6 +4015,7 @@ export function refreshQueueStats(schema, table, name, options = {}) {
3566
4015
  WITH ${lock.cte}stats AS (SELECT * FROM (${statsText}) agg WHERE true${lock.guard})
3567
4016
  UPDATE ${schema}.queue SET
3568
4017
  deferred_count = COALESCE(stats."deferredCount", 0),
4018
+ blocked_count = COALESCE(stats."blockedCount", 0),
3569
4019
  queued_count = COALESCE(stats."queuedCount", 0),
3570
4020
  ready_count = COALESCE(stats."readyCount", 0),
3571
4021
  active_count = COALESCE(stats."activeCount", 0),
@@ -3676,7 +4126,7 @@ export function findJobs(schema, table, options) {
3676
4126
  }
3677
4127
  if (byData) {
3678
4128
  ++paramIndex;
3679
- whereConditions.push(`AND data @> $${paramIndex}`);
4129
+ whereConditions.push(`AND data @> $${paramIndex}::text::jsonb`);
3680
4130
  }
3681
4131
  if (queued) {
3682
4132
  whereConditions.push(`AND state < '${JOB_STATES.active}'`);
@@ -4002,7 +4452,7 @@ const POLICY_JOB_INDEXES = {
4002
4452
  // missing: the fetch index was replaced in v40 and the retired number is not reused.
4003
4453
  const BASE_JOB_INDEXES = [4, 7, 9, 11, 12];
4004
4454
  // The fixed (non-job) managed tables; job/job_common/partitions are handled separately.
4005
- const FIXED_MANAGED_TABLES = ['version', 'queue', 'schedule', 'subscription', 'bam', 'warning', 'queue_stats', 'job_dependency'];
4455
+ const FIXED_MANAGED_TABLES = ['version', 'queue', 'schedule', 'subscription', 'bam', 'warning', 'queue_stats', 'job_dependency', 'instance'];
4006
4456
  // Selects the manifest section for the live architecture, and substitutes the real schema name back in
4007
4457
  // for the placeholder the manifest stores.
4008
4458
  function manifestSection(partitioned) {
@@ -4390,12 +4840,12 @@ export function getXminHorizon(lastVacuum, sources = XMIN_HORIZON_QUERY_SOURCES)
4390
4840
  // query it is follows from pid, application_name and role, looked up live where the catalog's own
4391
4841
  // privilege rules still apply.
4392
4842
  //
4393
- // selfApplicationName is what makes "ours or theirs" answerable. Db sets application_name to
4394
- // 'pgboss' on the pool it owns, so a holder matching this connection's own value is pg-boss doing
4395
- // it to itself - the monitor's own aggregate, most likely - which has a completely different fix
4396
- // from an external reporting tool holding a transaction open. It is compared rather than hardcoded
4397
- // because an adapter-supplied pool sets whatever the host app chose, and claiming that is
4398
- // definitely pg-boss would be a guess.
4843
+ // selfApplicationName is what makes "ours or theirs" answerable. pg-boss names the pool it owns
4844
+ // 'pgboss', or 'pgboss:<id>' for a registered instance, so a holder matching this connection's own
4845
+ // value is pg-boss doing it to itself - the monitor's own aggregate, most likely - which has a
4846
+ // completely different fix from an external reporting tool holding a transaction open. It is
4847
+ // compared rather than hardcoded because an adapter-supplied pool sets whatever the host app
4848
+ // chose, and claiming that is definitely pg-boss would be a guess.
4399
4849
  const backendIdentity = sources.includes('backends')
4400
4850
  ? `,
4401
4851
  (SELECT to_jsonb(h) FROM (