bullpane 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +35 -0
- package/LICENSE-ee +46 -0
- package/README.md +60 -0
- package/bin/bullpane.mjs +99 -0
- package/dist/lua/getGroups.lua +65 -0
- package/dist/lua/getJob.lua +61 -0
- package/dist/lua/getJobs.lua +156 -0
- package/dist/lua/getSchedulers.lua +76 -0
- package/dist/lua/getTreeNode.lua +105 -0
- package/dist/lua/queueSetup.lua +47 -0
- package/dist/lua/queueStats.lua +140 -0
- package/dist/lua/sampleParents.lua +65 -0
- package/dist/lua/searchJobs.lua +146 -0
- package/dist/lua/windowMetrics.lua +167 -0
- package/dist/server.mjs +8703 -0
- package/migrations/mysql/0001_init.sql +102 -0
- package/migrations/mysql/0002_alert_scopes.sql +9 -0
- package/migrations/mysql/0003_hidden_queues.sql +21 -0
- package/migrations/mysql/0004_audit_log.sql +57 -0
- package/migrations/mysql/0005_sso.sql +48 -0
- package/migrations/mysql/0006_connection_position.sql +29 -0
- package/migrations/mysql/0007_user_disabled.sql +12 -0
- package/migrations/mysql/0008_alert_wide_scopes.sql +18 -0
- package/migrations/mysql/0009_mcp.sql +62 -0
- package/migrations/sqlite/0001_init.sql +158 -0
- package/migrations/sqlite/0002_mcp.sql +38 -0
- package/package.json +56 -0
- package/web/assets/FlowsPage-CxMz_9Et.js +6 -0
- package/web/assets/JobTreePage-BT79W1zX.js +6 -0
- package/web/assets/McpConsentPage-B2YnL6fN.js +1 -0
- package/web/assets/flowLayout-BnuhLJ6X.css +1 -0
- package/web/assets/flowLayout-DoC2dCG0.js +23 -0
- package/web/assets/index-C2gL_lbF.css +1 -0
- package/web/assets/index-DuAopWKm.js +543 -0
- package/web/index.html +18 -0
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
--[[
|
|
2
|
+
One node of a flow tree: the job's own fields, its state, and the KEYS of its
|
|
3
|
+
children — in one round trip, without shipping logs or return values.
|
|
4
|
+
|
|
5
|
+
Why a script separate from getJob.lua: the tree walk reads many jobs and only
|
|
6
|
+
needs a handful of fields plus the child keys. getJob HGETALLs the whole hash
|
|
7
|
+
and LRANGEs the logs, which on a 300-node tree is a lot of bytes for data the
|
|
8
|
+
graph never renders.
|
|
9
|
+
|
|
10
|
+
KEYS[1] job hash
|
|
11
|
+
KEYS[2] `${id}:dependencies` set (unprocessed children, members are full job keys)
|
|
12
|
+
KEYS[3] `${id}:processed` hash (processed children, fields are full job keys)
|
|
13
|
+
KEYS[4..11] state keys in STATE_ORDER:
|
|
14
|
+
wait, active, completed, failed, delayed, prioritized, paused, waiting-children
|
|
15
|
+
|
|
16
|
+
ARGV[1] job id
|
|
17
|
+
ARGV[2] max child keys to return (the rest are counted, not listed)
|
|
18
|
+
|
|
19
|
+
Returns nil when the hash does not exist, otherwise
|
|
20
|
+
{ fieldsFlat, state, unprocessedCount, processedCount, childKeys, childrenTruncated }
|
|
21
|
+
|
|
22
|
+
`childKeys` mixes unprocessed and processed children; the caller does not need
|
|
23
|
+
to tell them apart from this list because each child reports its own state.
|
|
24
|
+
Only this job's own keys are touched, so the call is safe to pipeline per queue
|
|
25
|
+
on a cluster. Children living in another queue are walked by the caller in a
|
|
26
|
+
separate, per-queue pipeline.
|
|
27
|
+
]]
|
|
28
|
+
local rcall = redis.call
|
|
29
|
+
local jobKey = KEYS[1]
|
|
30
|
+
|
|
31
|
+
if rcall("EXISTS", jobKey) == 0 then
|
|
32
|
+
return nil
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
-- Only the fields the tree renders. HMGET keeps a fat `data` payload in Redis.
|
|
36
|
+
local fields = rcall("HMGET", jobKey,
|
|
37
|
+
"name", "timestamp", "finishedOn", "processedOn",
|
|
38
|
+
"attemptsMade", "failedReason", "progress", "parentKey", "parent", "opts")
|
|
39
|
+
|
|
40
|
+
local id = ARGV[1]
|
|
41
|
+
local limit = tonumber(ARGV[2]) or 100
|
|
42
|
+
|
|
43
|
+
local state = "unknown"
|
|
44
|
+
if rcall("ZSCORE", KEYS[6], id) then state = "completed"
|
|
45
|
+
elseif rcall("ZSCORE", KEYS[7], id) then state = "failed"
|
|
46
|
+
elseif rcall("ZSCORE", KEYS[8], id) then state = "delayed"
|
|
47
|
+
elseif rcall("ZSCORE", KEYS[9], id) then state = "prioritized"
|
|
48
|
+
elseif rcall("ZSCORE", KEYS[11], id) then state = "waiting-children"
|
|
49
|
+
else
|
|
50
|
+
local function inList(key)
|
|
51
|
+
local ok, pos = pcall(rcall, "LPOS", key, id)
|
|
52
|
+
if ok and pos then return true end
|
|
53
|
+
return false
|
|
54
|
+
end
|
|
55
|
+
if inList(KEYS[5]) then state = "active"
|
|
56
|
+
elseif inList(KEYS[4]) then state = "waiting"
|
|
57
|
+
elseif inList(KEYS[10]) then state = "paused"
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
local unprocessed = rcall("SCARD", KEYS[2])
|
|
62
|
+
local processed = rcall("HLEN", KEYS[3])
|
|
63
|
+
|
|
64
|
+
-- SSCAN/HSCAN with COUNT rather than SMEMBERS/HKEYS: a fan-out parent can have
|
|
65
|
+
-- tens of thousands of children and we only ever draw `limit` of them.
|
|
66
|
+
local children = {}
|
|
67
|
+
local truncated = 0
|
|
68
|
+
|
|
69
|
+
if limit > 0 and unprocessed > 0 then
|
|
70
|
+
local cursor = "0"
|
|
71
|
+
repeat
|
|
72
|
+
local res = rcall("SSCAN", KEYS[2], cursor, "COUNT", 200)
|
|
73
|
+
cursor = res[1]
|
|
74
|
+
for _, member in ipairs(res[2]) do
|
|
75
|
+
if #children >= limit then
|
|
76
|
+
truncated = 1
|
|
77
|
+
break
|
|
78
|
+
end
|
|
79
|
+
children[#children + 1] = member
|
|
80
|
+
end
|
|
81
|
+
until cursor == "0" or truncated == 1
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
if limit > 0 and processed > 0 and truncated == 0 then
|
|
85
|
+
local cursor = "0"
|
|
86
|
+
repeat
|
|
87
|
+
local res = rcall("HSCAN", KEYS[3], cursor, "COUNT", 200)
|
|
88
|
+
cursor = res[1]
|
|
89
|
+
-- HSCAN returns field, value, field, value...; the value is the child's
|
|
90
|
+
-- return value, which we deliberately do not ship.
|
|
91
|
+
for i = 1, #res[2], 2 do
|
|
92
|
+
if #children >= limit then
|
|
93
|
+
truncated = 1
|
|
94
|
+
break
|
|
95
|
+
end
|
|
96
|
+
children[#children + 1] = res[2][i]
|
|
97
|
+
end
|
|
98
|
+
until cursor == "0" or truncated == 1
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
if (unprocessed + processed) > #children then
|
|
102
|
+
truncated = 1
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
return { fields, state, unprocessed, processed, children, truncated }
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
--[[
|
|
2
|
+
What Redis knows about how ONE queue is configured, in a single round trip.
|
|
3
|
+
|
|
4
|
+
KEYS[1] meta hash
|
|
5
|
+
KEYS[2] `limiter` (worker rate limiter; PTTL > 0 => throttling right now)
|
|
6
|
+
KEYS[3] Pro `groups` zset, status "waiting"
|
|
7
|
+
KEYS[4] Pro `groups:limit` zset, status "limited"
|
|
8
|
+
KEYS[5] Pro `groups:max` zset, status "maxed"
|
|
9
|
+
KEYS[6] Pro `groups:paused` zset, status "paused"
|
|
10
|
+
KEYS[7] Pro `groups:active:count` hash gid -> active jobs (only when the worker sets group.concurrency)
|
|
11
|
+
KEYS[8] Pro `groups:metas` zset gids with per-group overrides (concurrency / rate limit)
|
|
12
|
+
KEYS[9] metrics:completed hash (exists => the worker collects metrics)
|
|
13
|
+
|
|
14
|
+
Returns:
|
|
15
|
+
[1] HGETALL meta (flat field/value array)
|
|
16
|
+
[2] PTTL limiter (-2 when the key does not exist)
|
|
17
|
+
[3..6] ZCARD of the four status zsets (waiting, limited, maxed, paused)
|
|
18
|
+
[7] HLEN groups:active:count
|
|
19
|
+
[8] ZCARD groups:metas
|
|
20
|
+
[9] EXISTS metrics:completed
|
|
21
|
+
|
|
22
|
+
Everything here is O(1) except HGETALL on the tiny meta hash. Cluster safe: one queue.
|
|
23
|
+
Worker-side options (worker concurrency, batch size) are NOT in Redis; the caller
|
|
24
|
+
reports them as unknown instead of guessing.
|
|
25
|
+
]]
|
|
26
|
+
local rcall = redis.call
|
|
27
|
+
|
|
28
|
+
-- A layout change in Pro must degrade to 0, never to a WRONGTYPE error.
|
|
29
|
+
local function card(key)
|
|
30
|
+
local ok, n = pcall(rcall, "ZCARD", key)
|
|
31
|
+
if ok then return n end
|
|
32
|
+
local ok2, m = pcall(rcall, "SCARD", key)
|
|
33
|
+
if ok2 then return m end
|
|
34
|
+
return 0
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
local out = {}
|
|
38
|
+
out[1] = rcall("HGETALL", KEYS[1])
|
|
39
|
+
out[2] = rcall("PTTL", KEYS[2])
|
|
40
|
+
out[3] = card(KEYS[3])
|
|
41
|
+
out[4] = card(KEYS[4])
|
|
42
|
+
out[5] = card(KEYS[5])
|
|
43
|
+
out[6] = card(KEYS[6])
|
|
44
|
+
out[7] = rcall("HLEN", KEYS[7])
|
|
45
|
+
out[8] = card(KEYS[8])
|
|
46
|
+
out[9] = rcall("EXISTS", KEYS[9])
|
|
47
|
+
return out
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
--[[
|
|
2
|
+
Counts + flags for ONE queue in a single round trip.
|
|
3
|
+
|
|
4
|
+
KEYS[1..8] state keys in STATE_ORDER:
|
|
5
|
+
wait, active, completed, failed, delayed, prioritized, paused, waiting-children
|
|
6
|
+
KEYS[9] meta hash
|
|
7
|
+
KEYS[10] Pro `groups` zset
|
|
8
|
+
KEYS[11] metrics:completed:data list
|
|
9
|
+
KEYS[12] metrics:failed:data list
|
|
10
|
+
KEYS[13] metrics:completed hash (field `count`, cumulative since forever)
|
|
11
|
+
KEYS[14] metrics:failed hash
|
|
12
|
+
KEYS[15] `repeat` zset (job schedulers)
|
|
13
|
+
KEYS[16] `stalled` SET (ids marked as stalled by the StalledCheck)
|
|
14
|
+
|
|
15
|
+
ARGV[1] withMetrics "1" | "0"
|
|
16
|
+
ARGV[2] number of metric points (newest N minutes)
|
|
17
|
+
ARGV[3] window start (unix ms) for the success/failure rate ZCOUNTs
|
|
18
|
+
ARGV[4] `${prefix}:${queue}:` (to read a job's opts and detect retention)
|
|
19
|
+
|
|
20
|
+
Returns a flat array:
|
|
21
|
+
[1..8] counts (LLEN for lists, ZCARD for zsets)
|
|
22
|
+
[9] 1 when the queue is paused (meta.paused exists)
|
|
23
|
+
[10] 1 when the queue shows a BullMQ Pro signal (any group status zset,
|
|
24
|
+
groups:metas, or meta.version "bullmq-pro:x")
|
|
25
|
+
[11] groups with jobs = sum of the four status zsets (see keys.ts)
|
|
26
|
+
[12] completed metric points, newest first (empty when withMetrics = 0)
|
|
27
|
+
[13] failed metric points, newest first
|
|
28
|
+
[14] ZCOUNT completed since ARGV[3] (scores are finishedOn timestamps; O(log N))
|
|
29
|
+
[15] ZCOUNT failed since ARGV[3]
|
|
30
|
+
[16] meta.version ("bullmq:5.x" / "bullmq-pro:7.x") or false
|
|
31
|
+
[17] `opts` of the newest job in completed, or false. Only used to find out
|
|
32
|
+
whether the queue uses removeOnComplete: with aggressive retention the
|
|
33
|
+
zset counts lie, and the caller has to warn the user.
|
|
34
|
+
[20] ZCARD `repeat` — how many job schedulers the queue has. A ZCARD is O(1),
|
|
35
|
+
so it fits here and the Schedulers tab badge costs no extra round trip.
|
|
36
|
+
[21] SCARD `stalled` — jobs that lost their lock. Cost: ONE more O(1) command
|
|
37
|
+
in the same EVALSHA (no extra round trip to Redis). Worth the budget
|
|
38
|
+
because it is the only way for the `active` tab to say "3 of these hung";
|
|
39
|
+
without it the operator sees "active 8" and cannot tell half are dead.
|
|
40
|
+
|
|
41
|
+
Cluster safe: every key belongs to the same queue (same hash tag).
|
|
42
|
+
Read only: the legacy "0:" wait-list marker is skipped, never popped.
|
|
43
|
+
]]
|
|
44
|
+
local rcall = redis.call
|
|
45
|
+
|
|
46
|
+
-- Lists may carry a deprecated "0:<ts>" marker at the tail (BullMQ v4 -> v5 migration).
|
|
47
|
+
-- It is not a job, so it is excluded from the count.
|
|
48
|
+
local function listCount(key)
|
|
49
|
+
local n = rcall("LLEN", key)
|
|
50
|
+
if n > 0 then
|
|
51
|
+
local last = rcall("LINDEX", key, -1)
|
|
52
|
+
if last and string.sub(last, 1, 2) == "0:" then
|
|
53
|
+
n = n - 1
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
return n
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
local out = {}
|
|
60
|
+
out[1] = listCount(KEYS[1]) -- wait
|
|
61
|
+
out[2] = rcall("LLEN", KEYS[2]) -- active
|
|
62
|
+
out[3] = rcall("ZCARD", KEYS[3]) -- completed
|
|
63
|
+
out[4] = rcall("ZCARD", KEYS[4]) -- failed
|
|
64
|
+
out[5] = rcall("ZCARD", KEYS[5]) -- delayed
|
|
65
|
+
out[6] = rcall("ZCARD", KEYS[6]) -- prioritized
|
|
66
|
+
out[7] = listCount(KEYS[7]) -- paused
|
|
67
|
+
out[8] = rcall("ZCARD", KEYS[8]) -- waiting-children
|
|
68
|
+
|
|
69
|
+
out[9] = rcall("HEXISTS", KEYS[9], "paused")
|
|
70
|
+
|
|
71
|
+
-- Pro detection + group count. A group sits in exactly ONE of four status zsets
|
|
72
|
+
-- (`groups` = waiting, groups:limit, groups:max, groups:paused — see keys.ts), so
|
|
73
|
+
-- "how many groups" is the sum of their ZCARDs and a queue whose groups are all
|
|
74
|
+
-- maxed has no `groups` key at all. groups:metas (per-group overrides) and
|
|
75
|
+
-- meta.version = "bullmq-pro:x" also mark a Pro queue, so an idle one still shows
|
|
76
|
+
-- as Pro. The three extra zsets hang off ARGV[4] (same queue, same hash tag):
|
|
77
|
+
-- three more O(1) commands in the same EVALSHA, no extra round trip.
|
|
78
|
+
local function card(key)
|
|
79
|
+
local ok, n = pcall(rcall, "ZCARD", key)
|
|
80
|
+
if ok then return n end
|
|
81
|
+
return 0
|
|
82
|
+
end
|
|
83
|
+
local version = rcall("HGET", KEYS[9], "version")
|
|
84
|
+
local groupsCount = card(KEYS[10])
|
|
85
|
+
+ card(ARGV[4] .. "groups:limit")
|
|
86
|
+
+ card(ARGV[4] .. "groups:max")
|
|
87
|
+
+ card(ARGV[4] .. "groups:paused")
|
|
88
|
+
local isPro = groupsCount > 0
|
|
89
|
+
or rcall("EXISTS", ARGV[4] .. "groups:metas") == 1
|
|
90
|
+
or (version and string.sub(version, 1, 10) == "bullmq-pro") or false
|
|
91
|
+
out[10] = isPro and 1 or 0
|
|
92
|
+
out[11] = groupsCount
|
|
93
|
+
|
|
94
|
+
if ARGV[1] == "1" then
|
|
95
|
+
local points = tonumber(ARGV[2]) or 60
|
|
96
|
+
-- metrics data is LPUSHed by BullMQ, so index 0 is the newest minute.
|
|
97
|
+
out[12] = rcall("LRANGE", KEYS[11], 0, points - 1)
|
|
98
|
+
out[13] = rcall("LRANGE", KEYS[12], 0, points - 1)
|
|
99
|
+
else
|
|
100
|
+
out[12] = {}
|
|
101
|
+
out[13] = {}
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
-- Success / failure over a trailing window. completed/failed zset scores are the
|
|
105
|
+
-- finishedOn timestamp, so ZCOUNT is O(log N) regardless of queue size.
|
|
106
|
+
local since = tonumber(ARGV[3]) or 0
|
|
107
|
+
out[14] = rcall("ZCOUNT", KEYS[3], since, "+inf")
|
|
108
|
+
out[15] = rcall("ZCOUNT", KEYS[4], since, "+inf")
|
|
109
|
+
out[16] = version or false
|
|
110
|
+
|
|
111
|
+
-- Retention: `removeOnComplete` lives in EACH job's opts, not in meta. We read the
|
|
112
|
+
-- opts of ONE job (an HGET on a small hash, O(1)) just to know whether the queue
|
|
113
|
+
-- prunes completed jobs. Without it there is no way to tell "99% success" apart
|
|
114
|
+
-- from "aggressive pruning making the ratio lie".
|
|
115
|
+
-- ARGV[4] = "${prefix}:${queue}:" to build the job hash key.
|
|
116
|
+
out[17] = false
|
|
117
|
+
local newest = rcall("ZREVRANGE", KEYS[3], 0, 0)
|
|
118
|
+
if newest and newest[1] then
|
|
119
|
+
out[17] = rcall("HGET", ARGV[4] .. newest[1], "opts") or false
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
-- BullMQ's cumulative counters. They only exist when the Worker was created with
|
|
123
|
+
-- `metrics: { maxDataPoints }`, but when they do they are the ONLY correct source:
|
|
124
|
+
-- they are incremented as each job finishes and never decremented, so
|
|
125
|
+
-- removeOnComplete cannot touch them. The :data list only gains a point when the
|
|
126
|
+
-- minute rolls over, which is why we read the hash and not just the list.
|
|
127
|
+
out[18] = rcall("HGET", KEYS[13], "count") or false
|
|
128
|
+
out[19] = rcall("HGET", KEYS[14], "count") or false
|
|
129
|
+
|
|
130
|
+
-- Job schedulers (repeatable jobs). They live outside the 8 states, in the `repeat`
|
|
131
|
+
-- zset. ZCARD is O(1), so the count rides along with everything else and the
|
|
132
|
+
-- Schedulers tab starts with the right number without a second round trip to Redis.
|
|
133
|
+
out[20] = rcall("ZCARD", KEYS[15])
|
|
134
|
+
|
|
135
|
+
-- `stalled` is an auxiliary SET, NOT a state: BullMQ does not expose it in
|
|
136
|
+
-- getState() (a stalled job answers `active`) and it only exists while something is
|
|
137
|
+
-- hung. SCARD is O(1), so the count rides along for free in this script.
|
|
138
|
+
out[21] = rcall("SCARD", KEYS[16])
|
|
139
|
+
|
|
140
|
+
return out
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
--[[
|
|
2
|
+
Flow detection: sample the newest N jobs of several states and aggregate the
|
|
3
|
+
queue key of their parent (BullMQ flow children carry `parentKey` and a
|
|
4
|
+
`parent` JSON {id, queueKey} in their hash).
|
|
5
|
+
|
|
6
|
+
KEYS[1..n] state keys to sample
|
|
7
|
+
ARGV[1] per-state sample size N
|
|
8
|
+
ARGV[2] queue key prefix `${prefix}:${queue}:`
|
|
9
|
+
ARGV[3..] "list" | "zset" for each KEYS[i] (ARGV[2 + i])
|
|
10
|
+
|
|
11
|
+
Returns { sampled, pairs } where pairs = { parentQueueKey1, count1, parentQueueKey2, count2, ... }
|
|
12
|
+
|
|
13
|
+
Cost: at most n * N HMGETs, all inside Redis, one round trip. The caller caches.
|
|
14
|
+
]]
|
|
15
|
+
local rcall = redis.call
|
|
16
|
+
local n = tonumber(ARGV[1])
|
|
17
|
+
local qprefix = ARGV[2]
|
|
18
|
+
|
|
19
|
+
local counts = {}
|
|
20
|
+
local order = {}
|
|
21
|
+
local sampled = 0
|
|
22
|
+
|
|
23
|
+
for i = 1, #KEYS do
|
|
24
|
+
local kind = ARGV[2 + i]
|
|
25
|
+
local ids
|
|
26
|
+
if kind == "list" then
|
|
27
|
+
ids = rcall("LRANGE", KEYS[i], 0, n - 1)
|
|
28
|
+
else
|
|
29
|
+
ids = rcall("ZREVRANGE", KEYS[i], 0, n - 1)
|
|
30
|
+
end
|
|
31
|
+
for _, id in ipairs(ids) do
|
|
32
|
+
if string.sub(id, 1, 2) ~= "0:" then
|
|
33
|
+
local vals = rcall("HMGET", qprefix .. id, "parentKey", "parent")
|
|
34
|
+
sampled = sampled + 1
|
|
35
|
+
local qk = nil
|
|
36
|
+
if vals[2] then
|
|
37
|
+
-- preferred: parent JSON carries the queue key verbatim
|
|
38
|
+
local ok, parsed = pcall(cjson.decode, vals[2])
|
|
39
|
+
if ok and type(parsed) == "table" and type(parsed.queueKey) == "string" then
|
|
40
|
+
qk = parsed.queueKey
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
if not qk and vals[1] then
|
|
44
|
+
-- fallback: parentKey is `${prefix}:${queue}:${id}`; drop the trailing `:id`
|
|
45
|
+
local cut = string.match(vals[1], "^(.*):[^:]+$")
|
|
46
|
+
if cut then qk = cut end
|
|
47
|
+
end
|
|
48
|
+
if qk then
|
|
49
|
+
if not counts[qk] then
|
|
50
|
+
counts[qk] = 0
|
|
51
|
+
order[#order + 1] = qk
|
|
52
|
+
end
|
|
53
|
+
counts[qk] = counts[qk] + 1
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
local out = {}
|
|
60
|
+
for _, qk in ipairs(order) do
|
|
61
|
+
out[#out + 1] = qk
|
|
62
|
+
out[#out + 1] = counts[qk]
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
return { sampled, out }
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
--[[
|
|
2
|
+
Bounded, resumable substring search over the jobs of one state.
|
|
3
|
+
|
|
4
|
+
KEYS[1] the state key (list or zset)
|
|
5
|
+
|
|
6
|
+
ARGV[1] "list" | "zset"
|
|
7
|
+
ARGV[2] cursor: index (newest = 0) of the first job to inspect
|
|
8
|
+
ARGV[3] batch: max number of job hashes inspected in THIS call (maxScanPerCall)
|
|
9
|
+
ARGV[4] needle, already lower-cased by the caller
|
|
10
|
+
ARGV[5] limit: stop early once this many matches are found
|
|
11
|
+
ARGV[6] queue key prefix `${prefix}:${queue}:`
|
|
12
|
+
ARGV[7] previewBytes (same truncation as getJobs)
|
|
13
|
+
ARGV[8] maxFieldBytes: a `data` field longer than this is NOT read or searched
|
|
14
|
+
(HSTRLEN first). The job still matches on id / name / failedReason
|
|
15
|
+
and is counted in `skipped` so the UI can say so.
|
|
16
|
+
ARGV[9] byteBudget: stop the call once this many payload bytes were copied
|
|
17
|
+
into Lua, even if `batch` hashes were not reached. The cursor comes
|
|
18
|
+
back as usual and the UI keeps streaming.
|
|
19
|
+
ARGV[10..] hash fields to HMGET (must include name, data, failedReason)
|
|
20
|
+
|
|
21
|
+
Returns { matches, nextCursor, scanned, total, skipped }
|
|
22
|
+
matches rows shaped exactly like getJobs.lua rows (…, dataTruncated, dataBytes)
|
|
23
|
+
nextCursor index to pass back next time, or -1 when the state is exhausted
|
|
24
|
+
scanned hashes inspected in this call
|
|
25
|
+
skipped hashes whose data was over maxFieldBytes and not searched
|
|
26
|
+
|
|
27
|
+
The match is a plain (non-pattern) case-insensitive string.find over id, name,
|
|
28
|
+
data and failedReason, each field on its own (no concatenated haystack: every
|
|
29
|
+
`..` on a 1 MB string is another 1 MB copy). Two bounds make the cost of ONE
|
|
30
|
+
call independent of both queue size and payload size: at most `batch` hashes
|
|
31
|
+
and at most `byteBudget` payload bytes. Before them a search over 1000 × 1 MB
|
|
32
|
+
jobs kept Redis busy for 13 s; with them a call is a few milliseconds.
|
|
33
|
+
]]
|
|
34
|
+
local rcall = redis.call
|
|
35
|
+
local key = KEYS[1]
|
|
36
|
+
local kind = ARGV[1]
|
|
37
|
+
local cursor = tonumber(ARGV[2])
|
|
38
|
+
local batch = tonumber(ARGV[3])
|
|
39
|
+
local needle = ARGV[4]
|
|
40
|
+
local limit = tonumber(ARGV[5])
|
|
41
|
+
local qprefix = ARGV[6]
|
|
42
|
+
local previewBytes = tonumber(ARGV[7])
|
|
43
|
+
local maxFieldBytes = tonumber(ARGV[8])
|
|
44
|
+
local byteBudget = tonumber(ARGV[9])
|
|
45
|
+
|
|
46
|
+
local fields = {}
|
|
47
|
+
for i = 10, #ARGV do fields[#fields + 1] = ARGV[i] end
|
|
48
|
+
local SKIP = "\0skip"
|
|
49
|
+
|
|
50
|
+
local dataIdx, retIdx, nameIdx, reasonIdx = nil, nil, nil, nil
|
|
51
|
+
for i, f in ipairs(fields) do
|
|
52
|
+
if f == "data" then dataIdx = i end
|
|
53
|
+
if f == "returnvalue" then retIdx = i end
|
|
54
|
+
if f == "name" then nameIdx = i end
|
|
55
|
+
if f == "failedReason" then reasonIdx = i end
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
local total
|
|
59
|
+
local ids
|
|
60
|
+
-- id -> raw score string, zsets only (delayed: when the job becomes runnable)
|
|
61
|
+
local scores = {}
|
|
62
|
+
if kind == "list" then
|
|
63
|
+
total = rcall("LLEN", key)
|
|
64
|
+
ids = rcall("LRANGE", key, cursor, cursor + batch - 1) -- head = newest
|
|
65
|
+
else
|
|
66
|
+
total = rcall("ZCARD", key)
|
|
67
|
+
local flat = rcall("ZREVRANGE", key, cursor, cursor + batch - 1, "WITHSCORES") -- highest score = newest
|
|
68
|
+
ids = {}
|
|
69
|
+
for i = 1, #flat, 2 do
|
|
70
|
+
ids[#ids + 1] = flat[i]
|
|
71
|
+
scores[flat[i]] = flat[i + 1]
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
local matches = {}
|
|
76
|
+
local scanned = 0
|
|
77
|
+
local found = 0
|
|
78
|
+
local skipped = 0
|
|
79
|
+
local bytes = 0
|
|
80
|
+
local stoppedEarly = false
|
|
81
|
+
|
|
82
|
+
local function has(hay, needle)
|
|
83
|
+
return hay and string.find(string.lower(hay), needle, 1, true) ~= nil
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
for _, id in ipairs(ids) do
|
|
87
|
+
scanned = scanned + 1
|
|
88
|
+
if string.sub(id, 1, 2) ~= "0:" then
|
|
89
|
+
local hkey = qprefix .. id
|
|
90
|
+
local dataBytes = dataIdx and rcall("HSTRLEN", hkey, "data") or 0
|
|
91
|
+
local retBytes = retIdx and rcall("HSTRLEN", hkey, "returnvalue") or 0
|
|
92
|
+
local skipData = dataIdx and dataBytes > maxFieldBytes
|
|
93
|
+
local want = fields
|
|
94
|
+
if skipData or (retIdx and retBytes > maxFieldBytes) then
|
|
95
|
+
want = {}
|
|
96
|
+
for i = 1, #fields do want[i] = fields[i] end
|
|
97
|
+
if skipData then want[dataIdx] = SKIP end
|
|
98
|
+
if retIdx and retBytes > maxFieldBytes then want[retIdx] = SKIP end
|
|
99
|
+
end
|
|
100
|
+
local vals = rcall("HMGET", hkey, unpack(want))
|
|
101
|
+
local alive = false
|
|
102
|
+
for i = 1, #vals do
|
|
103
|
+
if vals[i] then alive = true break end
|
|
104
|
+
end
|
|
105
|
+
if alive then
|
|
106
|
+
if skipData then skipped = skipped + 1 else bytes = bytes + dataBytes end
|
|
107
|
+
local hit = has(id, needle)
|
|
108
|
+
or (nameIdx and has(vals[nameIdx], needle))
|
|
109
|
+
or (dataIdx and not skipData and has(vals[dataIdx], needle))
|
|
110
|
+
or (reasonIdx and has(vals[reasonIdx], needle))
|
|
111
|
+
if hit then
|
|
112
|
+
local truncated = 0
|
|
113
|
+
if dataIdx and dataBytes > previewBytes then
|
|
114
|
+
if vals[dataIdx] then vals[dataIdx] = string.sub(vals[dataIdx], 1, previewBytes) end
|
|
115
|
+
truncated = 1
|
|
116
|
+
end
|
|
117
|
+
if retIdx and vals[retIdx] and #vals[retIdx] > previewBytes then
|
|
118
|
+
vals[retIdx] = string.sub(vals[retIdx], 1, previewBytes)
|
|
119
|
+
end
|
|
120
|
+
local row = { id }
|
|
121
|
+
for i = 1, #vals do row[#row + 1] = vals[i] end
|
|
122
|
+
row[#row + 1] = truncated
|
|
123
|
+
row[#row + 1] = dataBytes
|
|
124
|
+
row[#row + 1] = scores[id] or false
|
|
125
|
+
matches[#matches + 1] = row
|
|
126
|
+
found = found + 1
|
|
127
|
+
if found >= limit then
|
|
128
|
+
stoppedEarly = true
|
|
129
|
+
break
|
|
130
|
+
end
|
|
131
|
+
end
|
|
132
|
+
if bytes >= byteBudget then
|
|
133
|
+
stoppedEarly = true
|
|
134
|
+
break
|
|
135
|
+
end
|
|
136
|
+
end
|
|
137
|
+
end
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
local nextCursor = cursor + scanned
|
|
141
|
+
-- Exhausted when we walked past the end of the state and did not stop early.
|
|
142
|
+
if not stoppedEarly and (#ids < batch or nextCursor >= total) then
|
|
143
|
+
nextCursor = -1
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
return { matches, nextCursor, scanned, total, skipped }
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
--[[
|
|
2
|
+
"What finished in the last W minutes" for ONE queue, from BullMQ's own
|
|
3
|
+
metrics, plus a bounded processing-time sample. One round trip, read only.
|
|
4
|
+
|
|
5
|
+
WHY THE :data LISTS AND NOT ZCOUNT. The completed/failed zsets only hold what
|
|
6
|
+
retention left behind: with `removeOnComplete` a healthy queue reads as a
|
|
7
|
+
failing one (300 ok / 15 failed = 4.8% real reads as 23.1% by ZCOUNT).
|
|
8
|
+
BullMQ's metrics are written as each job finishes and pruning never touches
|
|
9
|
+
them.
|
|
10
|
+
|
|
11
|
+
HOW BULLMQ WRITES THEM (commands/includes/collectMetrics.lua). Per side
|
|
12
|
+
(completed / failed) there is a hash { count, prevTS, prevCount } and a list:
|
|
13
|
+
- every finished job does HINCRBY count
|
|
14
|
+
- when a job finishes in a LATER minute than prevTS, the jobs counted since
|
|
15
|
+
prevTS (count - prevCount) are LPUSHed, followed by one 0 per idle minute,
|
|
16
|
+
and prevTS/prevCount move to now
|
|
17
|
+
So, with m0 = floor(prevTS / 60000):
|
|
18
|
+
- `count - prevCount` jobs finished in minute m0 and are not in the list yet
|
|
19
|
+
- list index i holds minute m0 - 1 - i
|
|
20
|
+
- minutes after m0, up to now, had no finished job (a finish would have
|
|
21
|
+
flushed), so they are known zeros
|
|
22
|
+
Nothing is extrapolated and no history is kept on our side: the window is
|
|
23
|
+
exact to the minute from the first read, including right after a restart.
|
|
24
|
+
|
|
25
|
+
KEYS[1] metrics:completed hash
|
|
26
|
+
KEYS[2] metrics:failed hash
|
|
27
|
+
KEYS[3] metrics:completed:data list
|
|
28
|
+
KEYS[4] metrics:failed:data list
|
|
29
|
+
KEYS[5] completed zset (score = finishedOn)
|
|
30
|
+
|
|
31
|
+
ARGV[1] now, unix ms (the server's clock; minute granularity absorbs skew)
|
|
32
|
+
ARGV[2] rate windows in minutes, comma separated ("" = none)
|
|
33
|
+
ARGV[3] duration windows in minutes, comma separated ("" = none)
|
|
34
|
+
ARGV[4] max completed jobs to read for durations (bounded by the caller)
|
|
35
|
+
ARGV[5] `${prefix}:${queue}:` to build job hash keys
|
|
36
|
+
|
|
37
|
+
Returns:
|
|
38
|
+
[1] 1 when either metrics hash exists, else 0 (the Worker has no `metrics`)
|
|
39
|
+
[2] flat { completed, failed, coveredMinutes } per rate window, in ARGV[2] order
|
|
40
|
+
[3] flat { sampled, p50Ms, p95Ms } per duration window, in ARGV[3] order
|
|
41
|
+
(p50/p95 are -1 when nothing was sampled)
|
|
42
|
+
|
|
43
|
+
Cost: 2 HMGET + at most 2 LRANGE of max(window) small integers, and for
|
|
44
|
+
durations 1 ZREVRANGEBYSCORE + at most ARGV[4] HMGETs of two fields. Every
|
|
45
|
+
key belongs to this queue (same hash tag), so it is cluster safe.
|
|
46
|
+
]]
|
|
47
|
+
local rcall = redis.call
|
|
48
|
+
|
|
49
|
+
local function csv(s)
|
|
50
|
+
local out = {}
|
|
51
|
+
for part in string.gmatch(s or "", "[^,]+") do
|
|
52
|
+
local n = tonumber(part)
|
|
53
|
+
if n and n > 0 then out[#out + 1] = math.floor(n) end
|
|
54
|
+
end
|
|
55
|
+
return out
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
local now = tonumber(ARGV[1])
|
|
59
|
+
local nowMin = math.floor(now / 60000)
|
|
60
|
+
local rateWindows = csv(ARGV[2])
|
|
61
|
+
local durationWindows = csv(ARGV[3])
|
|
62
|
+
|
|
63
|
+
local maxWindow = 0
|
|
64
|
+
for _, w in ipairs(rateWindows) do
|
|
65
|
+
if w > maxWindow then maxWindow = w end
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
-- One side (completed or failed). Returns a reader: window -> total, covered.
|
|
69
|
+
local function side(hashKey, listKey)
|
|
70
|
+
local h = rcall("HMGET", hashKey, "count", "prevTS", "prevCount")
|
|
71
|
+
local count = tonumber(h[1])
|
|
72
|
+
if not count then
|
|
73
|
+
-- Hash absent. For `failed` this is normal on a queue that never failed
|
|
74
|
+
-- (BullMQ creates it on the first failure), so it reads as zero, fully
|
|
75
|
+
-- covered; the caller decides "no metrics" from BOTH sides being absent.
|
|
76
|
+
return false, function(w) return 0, w end
|
|
77
|
+
end
|
|
78
|
+
local prevTS = tonumber(h[2])
|
|
79
|
+
local prevCount = tonumber(h[3]) or 0
|
|
80
|
+
if not prevTS then
|
|
81
|
+
-- Written by a BullMQ that sets count before prevTS: all of it is "now".
|
|
82
|
+
return true, function(w) return count, w end
|
|
83
|
+
end
|
|
84
|
+
local m0 = math.floor(prevTS / 60000)
|
|
85
|
+
local ref = nowMin
|
|
86
|
+
if m0 > ref then ref = m0 end -- a worker clock ahead of ours
|
|
87
|
+
local pending = count - prevCount
|
|
88
|
+
|
|
89
|
+
-- Read the list once, for the widest window.
|
|
90
|
+
local points = {}
|
|
91
|
+
local need = m0 - (ref - maxWindow + 1)
|
|
92
|
+
if need > 0 then points = rcall("LRANGE", listKey, 0, need - 1) end
|
|
93
|
+
|
|
94
|
+
return true, function(w)
|
|
95
|
+
local startMin = ref - w + 1
|
|
96
|
+
if m0 < startMin then return 0, w end -- nothing finished inside the window
|
|
97
|
+
local total = pending
|
|
98
|
+
local covered = ref - m0 + 1
|
|
99
|
+
local n = m0 - startMin
|
|
100
|
+
if n > #points then n = #points end
|
|
101
|
+
for i = 1, n do
|
|
102
|
+
total = total + (tonumber(points[i]) or 0)
|
|
103
|
+
end
|
|
104
|
+
covered = covered + n
|
|
105
|
+
if covered > w then covered = w end
|
|
106
|
+
return total, covered
|
|
107
|
+
end
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
local hasCompleted, completedIn = side(KEYS[1], KEYS[3])
|
|
111
|
+
local hasFailed, failedIn = side(KEYS[2], KEYS[4])
|
|
112
|
+
|
|
113
|
+
local rates = {}
|
|
114
|
+
for _, w in ipairs(rateWindows) do
|
|
115
|
+
local c, cc = completedIn(w)
|
|
116
|
+
local f, fc = failedIn(w)
|
|
117
|
+
local covered = cc
|
|
118
|
+
if fc < covered then covered = fc end
|
|
119
|
+
rates[#rates + 1] = c
|
|
120
|
+
rates[#rates + 1] = f
|
|
121
|
+
rates[#rates + 1] = covered
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
-- Processing time: newest completed jobs inside the widest duration window.
|
|
125
|
+
local durations = {}
|
|
126
|
+
if #durationWindows > 0 then
|
|
127
|
+
local widest = 0
|
|
128
|
+
for _, w in ipairs(durationWindows) do
|
|
129
|
+
if w > widest then widest = w end
|
|
130
|
+
end
|
|
131
|
+
local limit = tonumber(ARGV[4]) or 100
|
|
132
|
+
local ids = rcall("ZREVRANGEBYSCORE", KEYS[5], "+inf", now - widest * 60000, "LIMIT", 0, limit)
|
|
133
|
+
-- { finishedOn, duration } newest first
|
|
134
|
+
local sampled = {}
|
|
135
|
+
for _, id in ipairs(ids) do
|
|
136
|
+
local v = rcall("HMGET", ARGV[5] .. id, "processedOn", "finishedOn")
|
|
137
|
+
local p, f = tonumber(v[1]), tonumber(v[2])
|
|
138
|
+
if p and f and f >= p then sampled[#sampled + 1] = { f, f - p } end
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
local function pct(sorted, q)
|
|
142
|
+
local idx = math.ceil(q / 100 * #sorted)
|
|
143
|
+
if idx < 1 then idx = 1 end
|
|
144
|
+
return sorted[idx]
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
for _, w in ipairs(durationWindows) do
|
|
148
|
+
local since = now - w * 60000
|
|
149
|
+
local ds = {}
|
|
150
|
+
for _, s in ipairs(sampled) do
|
|
151
|
+
if s[1] >= since then ds[#ds + 1] = s[2] end
|
|
152
|
+
end
|
|
153
|
+
table.sort(ds)
|
|
154
|
+
durations[#durations + 1] = #ds
|
|
155
|
+
if #ds > 0 then
|
|
156
|
+
durations[#durations + 1] = pct(ds, 50)
|
|
157
|
+
durations[#durations + 1] = pct(ds, 95)
|
|
158
|
+
else
|
|
159
|
+
durations[#durations + 1] = -1
|
|
160
|
+
durations[#durations + 1] = -1
|
|
161
|
+
end
|
|
162
|
+
end
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
local has = 0
|
|
166
|
+
if hasCompleted or hasFailed then has = 1 end
|
|
167
|
+
return { has, rates, durations }
|