bullpane 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/LICENSE +35 -0
  2. package/LICENSE-ee +46 -0
  3. package/README.md +60 -0
  4. package/bin/bullpane.mjs +99 -0
  5. package/dist/lua/getGroups.lua +65 -0
  6. package/dist/lua/getJob.lua +61 -0
  7. package/dist/lua/getJobs.lua +156 -0
  8. package/dist/lua/getSchedulers.lua +76 -0
  9. package/dist/lua/getTreeNode.lua +105 -0
  10. package/dist/lua/queueSetup.lua +47 -0
  11. package/dist/lua/queueStats.lua +140 -0
  12. package/dist/lua/sampleParents.lua +65 -0
  13. package/dist/lua/searchJobs.lua +146 -0
  14. package/dist/lua/windowMetrics.lua +167 -0
  15. package/dist/server.mjs +8703 -0
  16. package/migrations/mysql/0001_init.sql +102 -0
  17. package/migrations/mysql/0002_alert_scopes.sql +9 -0
  18. package/migrations/mysql/0003_hidden_queues.sql +21 -0
  19. package/migrations/mysql/0004_audit_log.sql +57 -0
  20. package/migrations/mysql/0005_sso.sql +48 -0
  21. package/migrations/mysql/0006_connection_position.sql +29 -0
  22. package/migrations/mysql/0007_user_disabled.sql +12 -0
  23. package/migrations/mysql/0008_alert_wide_scopes.sql +18 -0
  24. package/migrations/mysql/0009_mcp.sql +62 -0
  25. package/migrations/sqlite/0001_init.sql +158 -0
  26. package/migrations/sqlite/0002_mcp.sql +38 -0
  27. package/package.json +56 -0
  28. package/web/assets/FlowsPage-CxMz_9Et.js +6 -0
  29. package/web/assets/JobTreePage-BT79W1zX.js +6 -0
  30. package/web/assets/McpConsentPage-B2YnL6fN.js +1 -0
  31. package/web/assets/flowLayout-BnuhLJ6X.css +1 -0
  32. package/web/assets/flowLayout-DoC2dCG0.js +23 -0
  33. package/web/assets/index-C2gL_lbF.css +1 -0
  34. package/web/assets/index-DuAopWKm.js +543 -0
  35. package/web/index.html +18 -0
@@ -0,0 +1,105 @@
1
+ --[[
2
+ One node of a flow tree: the job's own fields, its state, and the KEYS of its
3
+ children — in one round trip, without shipping logs or return values.
4
+
5
+ Why a script separate from getJob.lua: the tree walk reads many jobs and only
6
+ needs a handful of fields plus the child keys. getJob HGETALLs the whole hash
7
+ and LRANGEs the logs, which on a 300-node tree is a lot of bytes for data the
8
+ graph never renders.
9
+
10
+ KEYS[1] job hash
11
+ KEYS[2] `${id}:dependencies` set (unprocessed children, members are full job keys)
12
+ KEYS[3] `${id}:processed` hash (processed children, fields are full job keys)
13
+ KEYS[4..11] state keys in STATE_ORDER:
14
+ wait, active, completed, failed, delayed, prioritized, paused, waiting-children
15
+
16
+ ARGV[1] job id
17
+ ARGV[2] max child keys to return (the rest are counted, not listed)
18
+
19
+ Returns nil when the hash does not exist, otherwise
20
+ { fieldsFlat, state, unprocessedCount, processedCount, childKeys, childrenTruncated }
21
+
22
+ `childKeys` mixes unprocessed and processed children; the caller does not need
23
+ to tell them apart from this list because each child reports its own state.
24
+ Only this job's own keys are touched, so the call is safe to pipeline per queue
25
+ on a cluster. Children living in another queue are walked by the caller in a
26
+ separate, per-queue pipeline.
27
+ ]]
28
+ local rcall = redis.call
29
+ local jobKey = KEYS[1]
30
+
31
+ if rcall("EXISTS", jobKey) == 0 then
32
+ return nil
33
+ end
34
+
35
+ -- Only the fields the tree renders. HMGET keeps a fat `data` payload in Redis.
36
+ local fields = rcall("HMGET", jobKey,
37
+ "name", "timestamp", "finishedOn", "processedOn",
38
+ "attemptsMade", "failedReason", "progress", "parentKey", "parent", "opts")
39
+
40
+ local id = ARGV[1]
41
+ local limit = tonumber(ARGV[2]) or 100
42
+
43
+ local state = "unknown"
44
+ if rcall("ZSCORE", KEYS[6], id) then state = "completed"
45
+ elseif rcall("ZSCORE", KEYS[7], id) then state = "failed"
46
+ elseif rcall("ZSCORE", KEYS[8], id) then state = "delayed"
47
+ elseif rcall("ZSCORE", KEYS[9], id) then state = "prioritized"
48
+ elseif rcall("ZSCORE", KEYS[11], id) then state = "waiting-children"
49
+ else
50
+ local function inList(key)
51
+ local ok, pos = pcall(rcall, "LPOS", key, id)
52
+ if ok and pos then return true end
53
+ return false
54
+ end
55
+ if inList(KEYS[5]) then state = "active"
56
+ elseif inList(KEYS[4]) then state = "waiting"
57
+ elseif inList(KEYS[10]) then state = "paused"
58
+ end
59
+ end
60
+
61
+ local unprocessed = rcall("SCARD", KEYS[2])
62
+ local processed = rcall("HLEN", KEYS[3])
63
+
64
+ -- SSCAN/HSCAN with COUNT rather than SMEMBERS/HKEYS: a fan-out parent can have
65
+ -- tens of thousands of children and we only ever draw `limit` of them.
66
+ local children = {}
67
+ local truncated = 0
68
+
69
+ if limit > 0 and unprocessed > 0 then
70
+ local cursor = "0"
71
+ repeat
72
+ local res = rcall("SSCAN", KEYS[2], cursor, "COUNT", 200)
73
+ cursor = res[1]
74
+ for _, member in ipairs(res[2]) do
75
+ if #children >= limit then
76
+ truncated = 1
77
+ break
78
+ end
79
+ children[#children + 1] = member
80
+ end
81
+ until cursor == "0" or truncated == 1
82
+ end
83
+
84
+ if limit > 0 and processed > 0 and truncated == 0 then
85
+ local cursor = "0"
86
+ repeat
87
+ local res = rcall("HSCAN", KEYS[3], cursor, "COUNT", 200)
88
+ cursor = res[1]
89
+ -- HSCAN returns field, value, field, value...; the value is the child's
90
+ -- return value, which we deliberately do not ship.
91
+ for i = 1, #res[2], 2 do
92
+ if #children >= limit then
93
+ truncated = 1
94
+ break
95
+ end
96
+ children[#children + 1] = res[2][i]
97
+ end
98
+ until cursor == "0" or truncated == 1
99
+ end
100
+
101
+ if (unprocessed + processed) > #children then
102
+ truncated = 1
103
+ end
104
+
105
+ return { fields, state, unprocessed, processed, children, truncated }
@@ -0,0 +1,47 @@
1
+ --[[
2
+ What Redis knows about how ONE queue is configured, in a single round trip.
3
+
4
+ KEYS[1] meta hash
5
+ KEYS[2] `limiter` (worker rate limiter; PTTL > 0 => throttling right now)
6
+ KEYS[3] Pro `groups` zset, status "waiting"
7
+ KEYS[4] Pro `groups:limit` zset, status "limited"
8
+ KEYS[5] Pro `groups:max` zset, status "maxed"
9
+ KEYS[6] Pro `groups:paused` zset, status "paused"
10
+ KEYS[7] Pro `groups:active:count` hash gid -> active jobs (only when the worker sets group.concurrency)
11
+ KEYS[8] Pro `groups:metas` zset gids with per-group overrides (concurrency / rate limit)
12
+ KEYS[9] metrics:completed hash (exists => the worker collects metrics)
13
+
14
+ Returns:
15
+ [1] HGETALL meta (flat field/value array)
16
+ [2] PTTL limiter (-2 when the key does not exist)
17
+ [3..6] ZCARD of the four status zsets (waiting, limited, maxed, paused)
18
+ [7] HLEN groups:active:count
19
+ [8] ZCARD groups:metas
20
+ [9] EXISTS metrics:completed
21
+
22
+ Everything here is O(1) except HGETALL on the tiny meta hash. Cluster safe: one queue.
23
+ Worker-side options (worker concurrency, batch size) are NOT in Redis; the caller
24
+ reports them as unknown instead of guessing.
25
+ ]]
26
+ local rcall = redis.call
27
+
28
+ -- A layout change in Pro must degrade to 0, never to a WRONGTYPE error.
29
+ local function card(key)
30
+ local ok, n = pcall(rcall, "ZCARD", key)
31
+ if ok then return n end
32
+ local ok2, m = pcall(rcall, "SCARD", key)
33
+ if ok2 then return m end
34
+ return 0
35
+ end
36
+
37
+ local out = {}
38
+ out[1] = rcall("HGETALL", KEYS[1])
39
+ out[2] = rcall("PTTL", KEYS[2])
40
+ out[3] = card(KEYS[3])
41
+ out[4] = card(KEYS[4])
42
+ out[5] = card(KEYS[5])
43
+ out[6] = card(KEYS[6])
44
+ out[7] = rcall("HLEN", KEYS[7])
45
+ out[8] = card(KEYS[8])
46
+ out[9] = rcall("EXISTS", KEYS[9])
47
+ return out
@@ -0,0 +1,140 @@
1
+ --[[
2
+ Counts + flags for ONE queue in a single round trip.
3
+
4
+ KEYS[1..8] state keys in STATE_ORDER:
5
+ wait, active, completed, failed, delayed, prioritized, paused, waiting-children
6
+ KEYS[9] meta hash
7
+ KEYS[10] Pro `groups` zset
8
+ KEYS[11] metrics:completed:data list
9
+ KEYS[12] metrics:failed:data list
10
+ KEYS[13] metrics:completed hash (field `count`, cumulative since forever)
11
+ KEYS[14] metrics:failed hash
12
+ KEYS[15] `repeat` zset (job schedulers)
13
+ KEYS[16] `stalled` SET (ids marked as stalled by the StalledCheck)
14
+
15
+ ARGV[1] withMetrics "1" | "0"
16
+ ARGV[2] number of metric points (newest N minutes)
17
+ ARGV[3] window start (unix ms) for the success/failure rate ZCOUNTs
18
+ ARGV[4] `${prefix}:${queue}:` (to read a job's opts and detect retention)
19
+
20
+ Returns a flat array:
21
+ [1..8] counts (LLEN for lists, ZCARD for zsets)
22
+ [9] 1 when the queue is paused (meta.paused exists)
23
+ [10] 1 when the queue shows a BullMQ Pro signal (any group status zset,
24
+ groups:metas, or meta.version "bullmq-pro:x")
25
+ [11] groups with jobs = sum of the four status zsets (see keys.ts)
26
+ [12] completed metric points, newest first (empty when withMetrics = 0)
27
+ [13] failed metric points, newest first
28
+ [14] ZCOUNT completed since ARGV[3] (scores are finishedOn timestamps; O(log N))
29
+ [15] ZCOUNT failed since ARGV[3]
30
+ [16] meta.version ("bullmq:5.x" / "bullmq-pro:7.x") or false
31
+ [17] `opts` of the newest job in completed, or false. Only used to find out
32
+ whether the queue uses removeOnComplete: with aggressive retention the
33
+ zset counts lie, and the caller has to warn the user.
34
+ [20] ZCARD `repeat` — how many job schedulers the queue has. A ZCARD is O(1),
35
+ so it fits here and the Schedulers tab badge costs no extra round trip.
36
+ [21] SCARD `stalled` — jobs that lost their lock. Cost: ONE more O(1) command
37
+ in the same EVALSHA (no extra round trip to Redis). Worth the budget
38
+ because it is the only way for the `active` tab to say "3 of these hung";
39
+ without it the operator sees "active 8" and cannot tell half are dead.
40
+
41
+ Cluster safe: every key belongs to the same queue (same hash tag).
42
+ Read only: the legacy "0:" wait-list marker is skipped, never popped.
43
+ ]]
44
+ local rcall = redis.call
45
+
46
+ -- Lists may carry a deprecated "0:<ts>" marker at the tail (BullMQ v4 -> v5 migration).
47
+ -- It is not a job, so it is excluded from the count.
48
+ local function listCount(key)
49
+ local n = rcall("LLEN", key)
50
+ if n > 0 then
51
+ local last = rcall("LINDEX", key, -1)
52
+ if last and string.sub(last, 1, 2) == "0:" then
53
+ n = n - 1
54
+ end
55
+ end
56
+ return n
57
+ end
58
+
59
+ local out = {}
60
+ out[1] = listCount(KEYS[1]) -- wait
61
+ out[2] = rcall("LLEN", KEYS[2]) -- active
62
+ out[3] = rcall("ZCARD", KEYS[3]) -- completed
63
+ out[4] = rcall("ZCARD", KEYS[4]) -- failed
64
+ out[5] = rcall("ZCARD", KEYS[5]) -- delayed
65
+ out[6] = rcall("ZCARD", KEYS[6]) -- prioritized
66
+ out[7] = listCount(KEYS[7]) -- paused
67
+ out[8] = rcall("ZCARD", KEYS[8]) -- waiting-children
68
+
69
+ out[9] = rcall("HEXISTS", KEYS[9], "paused")
70
+
71
+ -- Pro detection + group count. A group sits in exactly ONE of four status zsets
72
+ -- (`groups` = waiting, groups:limit, groups:max, groups:paused — see keys.ts), so
73
+ -- "how many groups" is the sum of their ZCARDs and a queue whose groups are all
74
+ -- maxed has no `groups` key at all. groups:metas (per-group overrides) and
75
+ -- meta.version = "bullmq-pro:x" also mark a Pro queue, so an idle one still shows
76
+ -- as Pro. The three extra zsets hang off ARGV[4] (same queue, same hash tag):
77
+ -- three more O(1) commands in the same EVALSHA, no extra round trip.
78
+ local function card(key)
79
+ local ok, n = pcall(rcall, "ZCARD", key)
80
+ if ok then return n end
81
+ return 0
82
+ end
83
+ local version = rcall("HGET", KEYS[9], "version")
84
+ local groupsCount = card(KEYS[10])
85
+ + card(ARGV[4] .. "groups:limit")
86
+ + card(ARGV[4] .. "groups:max")
87
+ + card(ARGV[4] .. "groups:paused")
88
+ local isPro = groupsCount > 0
89
+ or rcall("EXISTS", ARGV[4] .. "groups:metas") == 1
90
+ or (version and string.sub(version, 1, 10) == "bullmq-pro") or false
91
+ out[10] = isPro and 1 or 0
92
+ out[11] = groupsCount
93
+
94
+ if ARGV[1] == "1" then
95
+ local points = tonumber(ARGV[2]) or 60
96
+ -- metrics data is LPUSHed by BullMQ, so index 0 is the newest minute.
97
+ out[12] = rcall("LRANGE", KEYS[11], 0, points - 1)
98
+ out[13] = rcall("LRANGE", KEYS[12], 0, points - 1)
99
+ else
100
+ out[12] = {}
101
+ out[13] = {}
102
+ end
103
+
104
+ -- Success / failure over a trailing window. completed/failed zset scores are the
105
+ -- finishedOn timestamp, so ZCOUNT is O(log N) regardless of queue size.
106
+ local since = tonumber(ARGV[3]) or 0
107
+ out[14] = rcall("ZCOUNT", KEYS[3], since, "+inf")
108
+ out[15] = rcall("ZCOUNT", KEYS[4], since, "+inf")
109
+ out[16] = version or false
110
+
111
+ -- Retention: `removeOnComplete` lives in EACH job's opts, not in meta. We read the
112
+ -- opts of ONE job (an HGET on a small hash, O(1)) just to know whether the queue
113
+ -- prunes completed jobs. Without it there is no way to tell "99% success" apart
114
+ -- from "aggressive pruning making the ratio lie".
115
+ -- ARGV[4] = "${prefix}:${queue}:" to build the job hash key.
116
+ out[17] = false
117
+ local newest = rcall("ZREVRANGE", KEYS[3], 0, 0)
118
+ if newest and newest[1] then
119
+ out[17] = rcall("HGET", ARGV[4] .. newest[1], "opts") or false
120
+ end
121
+
122
+ -- BullMQ's cumulative counters. They only exist when the Worker was created with
123
+ -- `metrics: { maxDataPoints }`, but when they do they are the ONLY correct source:
124
+ -- they are incremented as each job finishes and never decremented, so
125
+ -- removeOnComplete cannot touch them. The :data list only gains a point when the
126
+ -- minute rolls over, which is why we read the hash and not just the list.
127
+ out[18] = rcall("HGET", KEYS[13], "count") or false
128
+ out[19] = rcall("HGET", KEYS[14], "count") or false
129
+
130
+ -- Job schedulers (repeatable jobs). They live outside the 8 states, in the `repeat`
131
+ -- zset. ZCARD is O(1), so the count rides along with everything else and the
132
+ -- Schedulers tab starts with the right number without a second round trip to Redis.
133
+ out[20] = rcall("ZCARD", KEYS[15])
134
+
135
+ -- `stalled` is an auxiliary SET, NOT a state: BullMQ does not expose it in
136
+ -- getState() (a stalled job answers `active`) and it only exists while something is
137
+ -- hung. SCARD is O(1), so the count rides along for free in this script.
138
+ out[21] = rcall("SCARD", KEYS[16])
139
+
140
+ return out
@@ -0,0 +1,65 @@
1
+ --[[
2
+ Flow detection: sample the newest N jobs of several states and aggregate the
3
+ queue key of their parent (BullMQ flow children carry `parentKey` and a
4
+ `parent` JSON {id, queueKey} in their hash).
5
+
6
+ KEYS[1..n] state keys to sample
7
+ ARGV[1] per-state sample size N
8
+ ARGV[2] queue key prefix `${prefix}:${queue}:`
9
+ ARGV[3..] "list" | "zset" for each KEYS[i] (ARGV[2 + i])
10
+
11
+ Returns { sampled, pairs } where pairs = { parentQueueKey1, count1, parentQueueKey2, count2, ... }
12
+
13
+ Cost: at most n * N HMGETs, all inside Redis, one round trip. The caller caches.
14
+ ]]
15
+ local rcall = redis.call
16
+ local n = tonumber(ARGV[1])
17
+ local qprefix = ARGV[2]
18
+
19
+ local counts = {}
20
+ local order = {}
21
+ local sampled = 0
22
+
23
+ for i = 1, #KEYS do
24
+ local kind = ARGV[2 + i]
25
+ local ids
26
+ if kind == "list" then
27
+ ids = rcall("LRANGE", KEYS[i], 0, n - 1)
28
+ else
29
+ ids = rcall("ZREVRANGE", KEYS[i], 0, n - 1)
30
+ end
31
+ for _, id in ipairs(ids) do
32
+ if string.sub(id, 1, 2) ~= "0:" then
33
+ local vals = rcall("HMGET", qprefix .. id, "parentKey", "parent")
34
+ sampled = sampled + 1
35
+ local qk = nil
36
+ if vals[2] then
37
+ -- preferred: parent JSON carries the queue key verbatim
38
+ local ok, parsed = pcall(cjson.decode, vals[2])
39
+ if ok and type(parsed) == "table" and type(parsed.queueKey) == "string" then
40
+ qk = parsed.queueKey
41
+ end
42
+ end
43
+ if not qk and vals[1] then
44
+ -- fallback: parentKey is `${prefix}:${queue}:${id}`; drop the trailing `:id`
45
+ local cut = string.match(vals[1], "^(.*):[^:]+$")
46
+ if cut then qk = cut end
47
+ end
48
+ if qk then
49
+ if not counts[qk] then
50
+ counts[qk] = 0
51
+ order[#order + 1] = qk
52
+ end
53
+ counts[qk] = counts[qk] + 1
54
+ end
55
+ end
56
+ end
57
+ end
58
+
59
+ local out = {}
60
+ for _, qk in ipairs(order) do
61
+ out[#out + 1] = qk
62
+ out[#out + 1] = counts[qk]
63
+ end
64
+
65
+ return { sampled, out }
@@ -0,0 +1,146 @@
1
+ --[[
2
+ Bounded, resumable substring search over the jobs of one state.
3
+
4
+ KEYS[1] the state key (list or zset)
5
+
6
+ ARGV[1] "list" | "zset"
7
+ ARGV[2] cursor: index (newest = 0) of the first job to inspect
8
+ ARGV[3] batch: max number of job hashes inspected in THIS call (maxScanPerCall)
9
+ ARGV[4] needle, already lower-cased by the caller
10
+ ARGV[5] limit: stop early once this many matches are found
11
+ ARGV[6] queue key prefix `${prefix}:${queue}:`
12
+ ARGV[7] previewBytes (same truncation as getJobs)
13
+ ARGV[8] maxFieldBytes: a `data` field longer than this is NOT read or searched
14
+ (HSTRLEN first). The job still matches on id / name / failedReason
15
+ and is counted in `skipped` so the UI can say so.
16
+ ARGV[9] byteBudget: stop the call once this many payload bytes were copied
17
+ into Lua, even if `batch` hashes were not reached. The cursor comes
18
+ back as usual and the UI keeps streaming.
19
+ ARGV[10..] hash fields to HMGET (must include name, data, failedReason)
20
+
21
+ Returns { matches, nextCursor, scanned, total, skipped }
22
+ matches rows shaped exactly like getJobs.lua rows (…, dataTruncated, dataBytes)
23
+ nextCursor index to pass back next time, or -1 when the state is exhausted
24
+ scanned hashes inspected in this call
25
+ skipped hashes whose data was over maxFieldBytes and not searched
26
+
27
+ The match is a plain (non-pattern) case-insensitive string.find over id, name,
28
+ data and failedReason, each field on its own (no concatenated haystack: every
29
+ `..` on a 1 MB string is another 1 MB copy). Two bounds make the cost of ONE
30
+ call independent of both queue size and payload size: at most `batch` hashes
31
+ and at most `byteBudget` payload bytes. Before them a search over 1000 × 1 MB
32
+ jobs kept Redis busy for 13 s; with them a call is a few milliseconds.
33
+ ]]
34
+ local rcall = redis.call
35
+ local key = KEYS[1]
36
+ local kind = ARGV[1]
37
+ local cursor = tonumber(ARGV[2])
38
+ local batch = tonumber(ARGV[3])
39
+ local needle = ARGV[4]
40
+ local limit = tonumber(ARGV[5])
41
+ local qprefix = ARGV[6]
42
+ local previewBytes = tonumber(ARGV[7])
43
+ local maxFieldBytes = tonumber(ARGV[8])
44
+ local byteBudget = tonumber(ARGV[9])
45
+
46
+ local fields = {}
47
+ for i = 10, #ARGV do fields[#fields + 1] = ARGV[i] end
48
+ local SKIP = "\0skip"
49
+
50
+ local dataIdx, retIdx, nameIdx, reasonIdx = nil, nil, nil, nil
51
+ for i, f in ipairs(fields) do
52
+ if f == "data" then dataIdx = i end
53
+ if f == "returnvalue" then retIdx = i end
54
+ if f == "name" then nameIdx = i end
55
+ if f == "failedReason" then reasonIdx = i end
56
+ end
57
+
58
+ local total
59
+ local ids
60
+ -- id -> raw score string, zsets only (delayed: when the job becomes runnable)
61
+ local scores = {}
62
+ if kind == "list" then
63
+ total = rcall("LLEN", key)
64
+ ids = rcall("LRANGE", key, cursor, cursor + batch - 1) -- head = newest
65
+ else
66
+ total = rcall("ZCARD", key)
67
+ local flat = rcall("ZREVRANGE", key, cursor, cursor + batch - 1, "WITHSCORES") -- highest score = newest
68
+ ids = {}
69
+ for i = 1, #flat, 2 do
70
+ ids[#ids + 1] = flat[i]
71
+ scores[flat[i]] = flat[i + 1]
72
+ end
73
+ end
74
+
75
+ local matches = {}
76
+ local scanned = 0
77
+ local found = 0
78
+ local skipped = 0
79
+ local bytes = 0
80
+ local stoppedEarly = false
81
+
82
+ local function has(hay, needle)
83
+ return hay and string.find(string.lower(hay), needle, 1, true) ~= nil
84
+ end
85
+
86
+ for _, id in ipairs(ids) do
87
+ scanned = scanned + 1
88
+ if string.sub(id, 1, 2) ~= "0:" then
89
+ local hkey = qprefix .. id
90
+ local dataBytes = dataIdx and rcall("HSTRLEN", hkey, "data") or 0
91
+ local retBytes = retIdx and rcall("HSTRLEN", hkey, "returnvalue") or 0
92
+ local skipData = dataIdx and dataBytes > maxFieldBytes
93
+ local want = fields
94
+ if skipData or (retIdx and retBytes > maxFieldBytes) then
95
+ want = {}
96
+ for i = 1, #fields do want[i] = fields[i] end
97
+ if skipData then want[dataIdx] = SKIP end
98
+ if retIdx and retBytes > maxFieldBytes then want[retIdx] = SKIP end
99
+ end
100
+ local vals = rcall("HMGET", hkey, unpack(want))
101
+ local alive = false
102
+ for i = 1, #vals do
103
+ if vals[i] then alive = true break end
104
+ end
105
+ if alive then
106
+ if skipData then skipped = skipped + 1 else bytes = bytes + dataBytes end
107
+ local hit = has(id, needle)
108
+ or (nameIdx and has(vals[nameIdx], needle))
109
+ or (dataIdx and not skipData and has(vals[dataIdx], needle))
110
+ or (reasonIdx and has(vals[reasonIdx], needle))
111
+ if hit then
112
+ local truncated = 0
113
+ if dataIdx and dataBytes > previewBytes then
114
+ if vals[dataIdx] then vals[dataIdx] = string.sub(vals[dataIdx], 1, previewBytes) end
115
+ truncated = 1
116
+ end
117
+ if retIdx and vals[retIdx] and #vals[retIdx] > previewBytes then
118
+ vals[retIdx] = string.sub(vals[retIdx], 1, previewBytes)
119
+ end
120
+ local row = { id }
121
+ for i = 1, #vals do row[#row + 1] = vals[i] end
122
+ row[#row + 1] = truncated
123
+ row[#row + 1] = dataBytes
124
+ row[#row + 1] = scores[id] or false
125
+ matches[#matches + 1] = row
126
+ found = found + 1
127
+ if found >= limit then
128
+ stoppedEarly = true
129
+ break
130
+ end
131
+ end
132
+ if bytes >= byteBudget then
133
+ stoppedEarly = true
134
+ break
135
+ end
136
+ end
137
+ end
138
+ end
139
+
140
+ local nextCursor = cursor + scanned
141
+ -- Exhausted when we walked past the end of the state and did not stop early.
142
+ if not stoppedEarly and (#ids < batch or nextCursor >= total) then
143
+ nextCursor = -1
144
+ end
145
+
146
+ return { matches, nextCursor, scanned, total, skipped }
@@ -0,0 +1,167 @@
1
+ --[[
2
+ "What finished in the last W minutes" for ONE queue, from BullMQ's own
3
+ metrics, plus a bounded processing-time sample. One round trip, read only.
4
+
5
+ WHY THE :data LISTS AND NOT ZCOUNT. The completed/failed zsets only hold what
6
+ retention left behind: with `removeOnComplete` a healthy queue reads as a
7
+ failing one (300 ok / 15 failed = 4.8% real reads as 23.1% by ZCOUNT).
8
+ BullMQ's metrics are written as each job finishes and pruning never touches
9
+ them.
10
+
11
+ HOW BULLMQ WRITES THEM (commands/includes/collectMetrics.lua). Per side
12
+ (completed / failed) there is a hash { count, prevTS, prevCount } and a list:
13
+ - every finished job does HINCRBY count
14
+ - when a job finishes in a LATER minute than prevTS, the jobs counted since
15
+ prevTS (count - prevCount) are LPUSHed, followed by one 0 per idle minute,
16
+ and prevTS/prevCount move to now
17
+ So, with m0 = floor(prevTS / 60000):
18
+ - `count - prevCount` jobs finished in minute m0 and are not in the list yet
19
+ - list index i holds minute m0 - 1 - i
20
+ - minutes after m0, up to now, had no finished job (a finish would have
21
+ flushed), so they are known zeros
22
+ Nothing is extrapolated and no history is kept on our side: the window is
23
+ exact to the minute from the first read, including right after a restart.
24
+
25
+ KEYS[1] metrics:completed hash
26
+ KEYS[2] metrics:failed hash
27
+ KEYS[3] metrics:completed:data list
28
+ KEYS[4] metrics:failed:data list
29
+ KEYS[5] completed zset (score = finishedOn)
30
+
31
+ ARGV[1] now, unix ms (the server's clock; minute granularity absorbs skew)
32
+ ARGV[2] rate windows in minutes, comma separated ("" = none)
33
+ ARGV[3] duration windows in minutes, comma separated ("" = none)
34
+ ARGV[4] max completed jobs to read for durations (bounded by the caller)
35
+ ARGV[5] `${prefix}:${queue}:` to build job hash keys
36
+
37
+ Returns:
38
+ [1] 1 when either metrics hash exists, else 0 (the Worker has no `metrics`)
39
+ [2] flat { completed, failed, coveredMinutes } per rate window, in ARGV[2] order
40
+ [3] flat { sampled, p50Ms, p95Ms } per duration window, in ARGV[3] order
41
+ (p50/p95 are -1 when nothing was sampled)
42
+
43
+ Cost: 2 HMGET + at most 2 LRANGE of max(window) small integers, and for
44
+ durations 1 ZREVRANGEBYSCORE + at most ARGV[4] HMGETs of two fields. Every
45
+ key belongs to this queue (same hash tag), so it is cluster safe.
46
+ ]]
47
+ local rcall = redis.call
48
+
49
+ local function csv(s)
50
+ local out = {}
51
+ for part in string.gmatch(s or "", "[^,]+") do
52
+ local n = tonumber(part)
53
+ if n and n > 0 then out[#out + 1] = math.floor(n) end
54
+ end
55
+ return out
56
+ end
57
+
58
+ local now = tonumber(ARGV[1])
59
+ local nowMin = math.floor(now / 60000)
60
+ local rateWindows = csv(ARGV[2])
61
+ local durationWindows = csv(ARGV[3])
62
+
63
+ local maxWindow = 0
64
+ for _, w in ipairs(rateWindows) do
65
+ if w > maxWindow then maxWindow = w end
66
+ end
67
+
68
+ -- One side (completed or failed). Returns a reader: window -> total, covered.
69
+ local function side(hashKey, listKey)
70
+ local h = rcall("HMGET", hashKey, "count", "prevTS", "prevCount")
71
+ local count = tonumber(h[1])
72
+ if not count then
73
+ -- Hash absent. For `failed` this is normal on a queue that never failed
74
+ -- (BullMQ creates it on the first failure), so it reads as zero, fully
75
+ -- covered; the caller decides "no metrics" from BOTH sides being absent.
76
+ return false, function(w) return 0, w end
77
+ end
78
+ local prevTS = tonumber(h[2])
79
+ local prevCount = tonumber(h[3]) or 0
80
+ if not prevTS then
81
+ -- Written by a BullMQ that sets count before prevTS: all of it is "now".
82
+ return true, function(w) return count, w end
83
+ end
84
+ local m0 = math.floor(prevTS / 60000)
85
+ local ref = nowMin
86
+ if m0 > ref then ref = m0 end -- a worker clock ahead of ours
87
+ local pending = count - prevCount
88
+
89
+ -- Read the list once, for the widest window.
90
+ local points = {}
91
+ local need = m0 - (ref - maxWindow + 1)
92
+ if need > 0 then points = rcall("LRANGE", listKey, 0, need - 1) end
93
+
94
+ return true, function(w)
95
+ local startMin = ref - w + 1
96
+ if m0 < startMin then return 0, w end -- nothing finished inside the window
97
+ local total = pending
98
+ local covered = ref - m0 + 1
99
+ local n = m0 - startMin
100
+ if n > #points then n = #points end
101
+ for i = 1, n do
102
+ total = total + (tonumber(points[i]) or 0)
103
+ end
104
+ covered = covered + n
105
+ if covered > w then covered = w end
106
+ return total, covered
107
+ end
108
+ end
109
+
110
+ local hasCompleted, completedIn = side(KEYS[1], KEYS[3])
111
+ local hasFailed, failedIn = side(KEYS[2], KEYS[4])
112
+
113
+ local rates = {}
114
+ for _, w in ipairs(rateWindows) do
115
+ local c, cc = completedIn(w)
116
+ local f, fc = failedIn(w)
117
+ local covered = cc
118
+ if fc < covered then covered = fc end
119
+ rates[#rates + 1] = c
120
+ rates[#rates + 1] = f
121
+ rates[#rates + 1] = covered
122
+ end
123
+
124
+ -- Processing time: newest completed jobs inside the widest duration window.
125
+ local durations = {}
126
+ if #durationWindows > 0 then
127
+ local widest = 0
128
+ for _, w in ipairs(durationWindows) do
129
+ if w > widest then widest = w end
130
+ end
131
+ local limit = tonumber(ARGV[4]) or 100
132
+ local ids = rcall("ZREVRANGEBYSCORE", KEYS[5], "+inf", now - widest * 60000, "LIMIT", 0, limit)
133
+ -- { finishedOn, duration } newest first
134
+ local sampled = {}
135
+ for _, id in ipairs(ids) do
136
+ local v = rcall("HMGET", ARGV[5] .. id, "processedOn", "finishedOn")
137
+ local p, f = tonumber(v[1]), tonumber(v[2])
138
+ if p and f and f >= p then sampled[#sampled + 1] = { f, f - p } end
139
+ end
140
+
141
+ local function pct(sorted, q)
142
+ local idx = math.ceil(q / 100 * #sorted)
143
+ if idx < 1 then idx = 1 end
144
+ return sorted[idx]
145
+ end
146
+
147
+ for _, w in ipairs(durationWindows) do
148
+ local since = now - w * 60000
149
+ local ds = {}
150
+ for _, s in ipairs(sampled) do
151
+ if s[1] >= since then ds[#ds + 1] = s[2] end
152
+ end
153
+ table.sort(ds)
154
+ durations[#durations + 1] = #ds
155
+ if #ds > 0 then
156
+ durations[#durations + 1] = pct(ds, 50)
157
+ durations[#durations + 1] = pct(ds, 95)
158
+ else
159
+ durations[#durations + 1] = -1
160
+ durations[#durations + 1] = -1
161
+ end
162
+ end
163
+ end
164
+
165
+ local has = 0
166
+ if hasCompleted or hasFailed then has = 1 end
167
+ return { has, rates, durations }