railwatch 0.7.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +35 -0
- data/app/controllers/railwatch/profiles_controller.rb +14 -1
- data/app/models/railwatch/filter_query.rb +5 -5
- data/app/models/railwatch/telemetry/execution.rb +47 -0
- data/app/models/railwatch/telemetry/rollup.rb +73 -9
- data/app/models/railwatch/telemetry/tenant.rb +29 -5
- data/db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb +34 -0
- data/db/railwatch_telemetry_migrate/20260923000000_add_people_index.rb +15 -0
- data/lib/railwatch/version.rb +1 -1
- metadata +3 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: c621aa5be8712487e490453e9b62ab5d3f75666b951270d50acf77b00ad03a29
|
|
4
|
+
data.tar.gz: 5464f54bb6968924b1155456dcd22c5dada112980760340beb74632a169d5791
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 035642f4a96ae1803f01079f618315a7c1ff54657f6111f52071ae32896782118a40e3f7075ddfb60910625899e589c5cea8f5b3b7c999af42211d097eb2e161
|
|
7
|
+
data.tar.gz: 4dc2f2217692fa974f06c7c2c7b487896206a0cefccd7301f9802d3eb192dc47900bdfdf96a9a07809d86b41c6a59292fc0df27382b7636167331fcb59c02978
|
data/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,41 @@
|
|
|
6
6
|
filled in by the release commit, which is also the only commit that
|
|
7
7
|
touches lib/railwatch/version.rb and Gemfile.lock. See CONTRIBUTING.md. -->
|
|
8
8
|
|
|
9
|
+
## 0.8.1 (2026-09-23)
|
|
10
|
+
|
|
11
|
+
- The Tenants page reads only through the tenant-led indexes. Sparklines
|
|
12
|
+
are drawn for the tenants already found, by name. The untagged share is
|
|
13
|
+
the window's requests less the tagged ones, and the tagged ones are
|
|
14
|
+
counted per tenant. Both used to read every request in the window: over
|
|
15
|
+
a week, 11 s on an app with no tenants and 24 s on one with them.
|
|
16
|
+
- A partial covering index serves the People page: 4 s over 30 days,
|
|
17
|
+
now 18 ms, from under a megabyte.
|
|
18
|
+
- The Profiles page takes the window's execution count from rollups
|
|
19
|
+
instead of counting every execution (1.2 s over a week).
|
|
20
|
+
|
|
21
|
+
## 0.8.0 (2026-09-23)
|
|
22
|
+
|
|
23
|
+
- Percentiles merged across hours are about ten times faster. Each
|
|
24
|
+
rollup's stored t-digest is read straight from its bytes and the
|
|
25
|
+
centroids are sorted once, instead of being pushed one at a time into
|
|
26
|
+
a new digest. A week of the platform's own query rollups (15,000 rows)
|
|
27
|
+
went from 1.6 s to 157 ms. Every page that shows a p50, p95 or p99
|
|
28
|
+
over a window reads this. The percentile is now nearest rank over the
|
|
29
|
+
centroids, the same rule the sub-hour charts use, so the two agree.
|
|
30
|
+
- A new migration adds four indexes. Covering indexes serve the Jobs
|
|
31
|
+
page's per-queue breakdown and the Processes page's health series
|
|
32
|
+
(38 s and 19 s over a week of a 30 GB telemetry file, now 335 ms and
|
|
33
|
+
370 ms). There is an `occurred_at` index on broadcasts, the one
|
|
34
|
+
windowed table without one. A partial index covers executions that
|
|
35
|
+
carry an exception preview. The migration takes about 15 s on a
|
|
36
|
+
30 GB file.
|
|
37
|
+
- Free-text search over jobs (and `Execution.named_like`, which MCP's
|
|
38
|
+
request search uses) matches names through the window's rollups and
|
|
39
|
+
fetches rows by `group_hash`, instead of reading every execution in
|
|
40
|
+
the window. A search matching nothing over a week took 657 ms and
|
|
41
|
+
takes 27 ms. A common text still takes the old path, which finds a
|
|
42
|
+
page within the first few thousand rows.
|
|
43
|
+
|
|
9
44
|
## 0.7.0 (2026-09-23)
|
|
10
45
|
|
|
11
46
|
- The gem ships the dashboard's frontend source (`app/frontend`, without
|
|
@@ -13,7 +13,7 @@ module Railwatch
|
|
|
13
13
|
rows, count, avg_samples, profiled, executions = telemetry do
|
|
14
14
|
scope = Telemetry::Profile.between(from, to)
|
|
15
15
|
[ scope.recent.limit(SCAN_LIMIT).select(:id, :group_hash, :execution_preview, :duration, :samples, :occurred_at).to_a,
|
|
16
|
-
scope.count, scope.average(:samples), scope.distinct.count(:execution_id),
|
|
16
|
+
scope.count, scope.average(:samples), scope.distinct.count(:execution_id), executions_in(from, to) ]
|
|
17
17
|
end
|
|
18
18
|
render inertia: { profiles: groups(rows),
|
|
19
19
|
summary: { profiles: count, executions: executions, profiled: profiled, avg_samples: avg_samples.to_f.round(0) } }
|
|
@@ -29,6 +29,19 @@ module Railwatch
|
|
|
29
29
|
|
|
30
30
|
private
|
|
31
31
|
|
|
32
|
+
# How many executions the window holds, as the profiled share's
|
|
33
|
+
# denominator. Every kind is rolled up, so the whole hours come from a
|
|
34
|
+
# few hundred rollup rows rather than counting every execution (1.2 s
|
|
35
|
+
# over a week of the platform's own); only the part of the first hour
|
|
36
|
+
# inside the window is counted from the rows.
|
|
37
|
+
def executions_in(from, to)
|
|
38
|
+
whole = from.beginning_of_hour == from ? from : from.beginning_of_hour + 1.hour
|
|
39
|
+
return Telemetry::Execution.between(from, to).count if whole >= to
|
|
40
|
+
|
|
41
|
+
Telemetry::Execution.where(occurred_at: from...whole).count +
|
|
42
|
+
Telemetry::Rollup.for_type(Telemetry::Execution::KINDS).between(whole, to).sum(:count)
|
|
43
|
+
end
|
|
44
|
+
|
|
32
45
|
# Rows arrive newest-first, so the head of each group is the profile the
|
|
33
46
|
# page links to.
|
|
34
47
|
def groups(rows)
|
|
@@ -74,11 +74,12 @@ module Railwatch
|
|
|
74
74
|
resource = resource.to_sym
|
|
75
75
|
parsed = parse(query)
|
|
76
76
|
fields = parsed[:fields].except(*Array(except).map(&:to_s))
|
|
77
|
-
|
|
77
|
+
range = bounded_time_range(fields, from, to)
|
|
78
|
+
scope = scope.where(occurred_at: range)
|
|
78
79
|
scope = apply_common(scope, fields)
|
|
79
80
|
|
|
80
81
|
case resource
|
|
81
|
-
when :jobs then apply_jobs(scope, fields, parsed[:text])
|
|
82
|
+
when :jobs then apply_jobs(scope, fields, parsed[:text], range)
|
|
82
83
|
when :exceptions then apply_exceptions(scope, fields, parsed[:text])
|
|
83
84
|
when :logs then apply_logs(scope, fields, parsed[:text])
|
|
84
85
|
when :queries then apply_queries(scope, fields, parsed[:text])
|
|
@@ -120,15 +121,14 @@ module Railwatch
|
|
|
120
121
|
end
|
|
121
122
|
private_class_method :apply_common
|
|
122
123
|
|
|
123
|
-
def self.apply_jobs(scope, fields, text)
|
|
124
|
+
def self.apply_jobs(scope, fields, text, range)
|
|
124
125
|
outcome = fields["outcome"].presence || fields["status"].presence
|
|
125
126
|
scope = scope.where(queue: fields["queue"]) if fields["queue"].present?
|
|
126
127
|
scope = scope.where(outcome: outcome) if outcome
|
|
127
128
|
scope = scope.where(name: fields["class"]) if fields["class"].present?
|
|
128
129
|
scope = scope.where(job_id: fields["job_id"]) if fields["job_id"].present?
|
|
129
130
|
return scope unless text.present?
|
|
130
|
-
|
|
131
|
-
scope.where("name LIKE :pattern OR exception_preview LIKE :pattern", pattern: pattern)
|
|
131
|
+
scope.merge(Telemetry::Execution.named_like("job_attempt", text, range, previews: true))
|
|
132
132
|
end
|
|
133
133
|
private_class_method :apply_jobs
|
|
134
134
|
|
|
@@ -17,6 +17,53 @@ module Railwatch
|
|
|
17
17
|
scope :between, ->(from, to) { where(occurred_at: from..to) }
|
|
18
18
|
scope :failed, -> { where("status >= 500 OR outcome = 'failed'") }
|
|
19
19
|
|
|
20
|
+
# Executions of `kind` in [from, to] whose name contains `text`, or
|
|
21
|
+
# (with previews: true) whose exception preview does. A LIKE on the rows
|
|
22
|
+
# themselves reads every execution in the window until it has a page:
|
|
23
|
+
# 32 s for a week of the platform's own requests when the text is rare.
|
|
24
|
+
# Names come from a few hundred groups, so the text is matched against
|
|
25
|
+
# the window's rollups and the rows are fetched by group_hash. Rollups
|
|
26
|
+
# trail ingest by up to a minute, so the last FRESH_NAMES are matched on
|
|
27
|
+
# the rows themselves. Only failures carry a preview, and a partial
|
|
28
|
+
# index holds just those rows.
|
|
29
|
+
#
|
|
30
|
+
# Each branch names its index and the outer lookup is by rowid, so the
|
|
31
|
+
# plan does not depend on planner statistics (an embedded install's are
|
|
32
|
+
# sampled, and sampled statistics make kind look selective). When the
|
|
33
|
+
# text is common the old walk is the fast one -- a page of matches
|
|
34
|
+
# turns up in the first few thousand rows -- so a text the rollups say
|
|
35
|
+
# matches DENSE_MATCHES rows or more keeps it. On a copy of the
|
|
36
|
+
# platform's own tenant, a week of job attempts: a text matching
|
|
37
|
+
# nothing took 657 ms walking and 27 ms here; one matching 32,766 took
|
|
38
|
+
# 11 ms walking and 215 ms here.
|
|
39
|
+
FRESH_NAMES = 5.minutes
|
|
40
|
+
DENSE_MATCHES = 1_000
|
|
41
|
+
|
|
42
|
+
# `range` is used as given, so an exclusive or empty one (FilterQuery's
|
|
43
|
+
# after:/before: can narrow a window to nothing) stays that way.
|
|
44
|
+
def self.named_like(kind, text, range, previews: false)
|
|
45
|
+
from, to = range.begin, range.end
|
|
46
|
+
pattern = "%#{sanitize_sql_like(text)}%"
|
|
47
|
+
window = where(kind: kind, occurred_at: range)
|
|
48
|
+
groups = Rollup.for_type(kind).between(from, to).where("name LIKE ? ESCAPE '\\'", pattern)
|
|
49
|
+
if !TelemetryRecord.sqlite? || groups.sum(:count) >= DENSE_MATCHES
|
|
50
|
+
like = previews ? "name LIKE :pattern ESCAPE '\\' OR exception_preview LIKE :pattern ESCAPE '\\'" : "name LIKE :pattern ESCAPE '\\'"
|
|
51
|
+
return window.where(like, pattern: pattern)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
by_group = indexed_by("index_executions_on_group_hash_and_occurred_at")
|
|
55
|
+
.where(group_hash: groups.distinct.select(:group_hash), occurred_at: range, kind: kind)
|
|
56
|
+
fresh = where(kind: kind, occurred_at: [ from, to - FRESH_NAMES ].max..to).where("name LIKE ? ESCAPE '\\'", pattern)
|
|
57
|
+
ids = [ by_group, fresh ]
|
|
58
|
+
ids << indexed_by("idx_executions_with_preview").where(kind: kind, occurred_at: range)
|
|
59
|
+
.where.not(exception_preview: nil).where("exception_preview LIKE ? ESCAPE '\\'", pattern) if previews
|
|
60
|
+
from("#{quoted_table_name} NOT INDEXED").where(kind: kind, occurred_at: range)
|
|
61
|
+
.where(ids.map { |branch| "#{quoted_table_name}.id IN (#{branch.select(:id).to_sql})" }.join(" OR "))
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def self.indexed_by(index) = unscoped.from("#{quoted_table_name} INDEXED BY #{index}")
|
|
65
|
+
private_class_method :indexed_by
|
|
66
|
+
|
|
20
67
|
# Servers that reported an execution since `time`. Asked once per kind:
|
|
21
68
|
# there is no index led by occurred_at alone, so the obvious
|
|
22
69
|
# `where(occurred_at: time..).distinct.pluck(:server)` scans the whole
|
|
@@ -55,21 +55,16 @@ module Railwatch
|
|
|
55
55
|
def self.summarize(relation)
|
|
56
56
|
rows = relation.to_a
|
|
57
57
|
return { count: 0, errors: 0, client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0, max: 0, extra: {} } if rows.empty?
|
|
58
|
-
|
|
59
|
-
# `+` built a brand-new digest from both operands' centroids on every
|
|
60
|
-
# row, so merging N rows re-pushed every earlier centroid N times:
|
|
61
|
-
# 311 rows took 1.4s where merge! takes 0.5s, for the same percentiles.
|
|
62
|
-
merged = TDigest::TDigest.new(0.01)
|
|
63
|
-
rows.each { |r| merged.merge!(TDigest::TDigest.from_bytes(r.digest)) if r.digest }
|
|
58
|
+
p50, p95, p99 = merged_percentiles(rows, [ 0.5, 0.95, 0.99 ])
|
|
64
59
|
count = rows.sum(&:count)
|
|
65
60
|
{
|
|
66
61
|
count: count,
|
|
67
62
|
errors: rows.sum(&:error_count),
|
|
68
63
|
client_errors: rows.sum(&:client_error_count),
|
|
69
64
|
avg: count.zero? ? 0 : (rows.sum(&:duration_sum) / count),
|
|
70
|
-
p50:
|
|
71
|
-
p95:
|
|
72
|
-
p99:
|
|
65
|
+
p50: p50,
|
|
66
|
+
p95: p95,
|
|
67
|
+
p99: p99,
|
|
73
68
|
max: rows.map(&:duration_max).max,
|
|
74
69
|
# Everything a type puts in `extra` -- llm_call's cost_nanos and
|
|
75
70
|
# token counts, cache_event's hits and misses -- merged the way
|
|
@@ -80,6 +75,75 @@ module Railwatch
|
|
|
80
75
|
}
|
|
81
76
|
end
|
|
82
77
|
|
|
78
|
+
# Percentiles of the union of every row's centroids, read straight out
|
|
79
|
+
# of the stored bytes. Merging through TDigest pushed each centroid into
|
|
80
|
+
# a red-black tree one at a time: 2.8 s for a week of the platform's
|
|
81
|
+
# own query rollups (18,674 rows) and 7 s for thirty days, on every
|
|
82
|
+
# cold page load. Sorting the centroids once and walking their
|
|
83
|
+
# cumulative counts answers the same question without re-clustering:
|
|
84
|
+
# 353 ms and 1 s for those two windows. The answer is the first
|
|
85
|
+
# centroid whose running total reaches n * p, which is nearest rank --
|
|
86
|
+
# the rule raw_series uses for sub-hour buckets, so both readings of a
|
|
87
|
+
# window agree. (TDigest#percentile compares each centroid's midpoint
|
|
88
|
+
# instead, which puts the median of 1,000 sevens and 300 nines at 9.)
|
|
89
|
+
def self.merged_percentiles(rows, ps)
|
|
90
|
+
means = []
|
|
91
|
+
counts = []
|
|
92
|
+
rows.each { |row| centroids(row.digest, means, counts) if row.digest }
|
|
93
|
+
return ps.map { 0 } if means.empty?
|
|
94
|
+
|
|
95
|
+
order = (0...means.size).sort_by { |i| means[i] }
|
|
96
|
+
total = counts.sum
|
|
97
|
+
targets = ps.map { |p| total * p }
|
|
98
|
+
found = []
|
|
99
|
+
seen = 0
|
|
100
|
+
order.each do |i|
|
|
101
|
+
seen += counts[i]
|
|
102
|
+
found << means[i] while found.size < targets.size && seen >= targets[found.size]
|
|
103
|
+
break if found.size == targets.size
|
|
104
|
+
end
|
|
105
|
+
found << means[order.last] while found.size < targets.size
|
|
106
|
+
found.map(&:to_i)
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# Appends one stored digest's centroid means and weights, in either of
|
|
110
|
+
# the tdigest gem's encodings (TDigest.from_bytes reads the same two).
|
|
111
|
+
def self.centroids(bytes, means, counts)
|
|
112
|
+
format, _compression, size = bytes.unpack("LdL")
|
|
113
|
+
case format
|
|
114
|
+
when TDigest::TDigest::VERBOSE_ENCODING
|
|
115
|
+
means.concat(bytes.unpack("@16d#{size}"))
|
|
116
|
+
counts.concat(bytes.unpack("@#{16 + 8 * size}L#{size}"))
|
|
117
|
+
when TDigest::TDigest::SMALL_ENCODING
|
|
118
|
+
# Means are delta-encoded 4-byte floats; weights are 7-bit varints,
|
|
119
|
+
# one byte each unless a centroid holds 128 or more samples.
|
|
120
|
+
mean = 0.0
|
|
121
|
+
bytes.unpack("@16f#{size}").each { |delta| means << (mean += delta) }
|
|
122
|
+
weights = bytes.byteslice(16 + 4 * size, bytes.bytesize).unpack("C*")
|
|
123
|
+
if weights.size == size
|
|
124
|
+
counts.concat(weights)
|
|
125
|
+
else
|
|
126
|
+
at = 0
|
|
127
|
+
size.times do
|
|
128
|
+
byte = weights[at]
|
|
129
|
+
at += 1
|
|
130
|
+
weight = byte & 0x7f
|
|
131
|
+
shift = 7
|
|
132
|
+
while byte & 0x80 != 0
|
|
133
|
+
byte = weights[at] || 0
|
|
134
|
+
at += 1
|
|
135
|
+
weight += (byte & 0x7f) << shift
|
|
136
|
+
shift += 7
|
|
137
|
+
end
|
|
138
|
+
counts << weight
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
else
|
|
142
|
+
raise ArgumentError, "unknown t-digest encoding #{format}"
|
|
143
|
+
end
|
|
144
|
+
end
|
|
145
|
+
private_class_method :centroids
|
|
146
|
+
|
|
83
147
|
def self.merge_extras(rows)
|
|
84
148
|
rows.each_with_object({}) do |row, out|
|
|
85
149
|
row.extra.each do |key, value|
|
|
@@ -18,6 +18,9 @@ module Railwatch
|
|
|
18
18
|
# aggregate), so only the busiest tenants get one; the rest report 0.
|
|
19
19
|
P95_TENANTS = 50
|
|
20
20
|
SPARKLINE_BUCKETS = 12
|
|
21
|
+
# Tenant names are bound one variable each, and SQLite allows 32,766
|
|
22
|
+
# per statement; an app with more tenants than that is asked in slices.
|
|
23
|
+
TENANT_SLICE = 1_000
|
|
21
24
|
SORTS = { "requests" => :requests, "errors" => :errors, "p95" => :p95, "users" => :users }.freeze
|
|
22
25
|
|
|
23
26
|
ERRORS_SQL = Arel.sql("SUM(CASE WHEN status >= 500 THEN 1 ELSE 0 END)")
|
|
@@ -35,7 +38,7 @@ module Railwatch
|
|
|
35
38
|
absorb_counts(rows, filtered(Telemetry::Exception.between(from, to), q), :exceptions)
|
|
36
39
|
absorb_counts(rows, filtered(Telemetry::Log.between(from, to), q), :logs)
|
|
37
40
|
absorb_last_seen(rows, from, to, q)
|
|
38
|
-
absorb_sparklines(rows, from, to
|
|
41
|
+
absorb_sparklines(rows, from, to)
|
|
39
42
|
absorb_p95s(rows.values.max_by(P95_TENANTS) { |r| r[:requests] }, from, to)
|
|
40
43
|
sorted(rows.values, sort, dir)
|
|
41
44
|
end
|
|
@@ -43,20 +46,31 @@ module Railwatch
|
|
|
43
46
|
# Headline numbers for the index page, computed from the rows it shows
|
|
44
47
|
# (so the top-tenant share is a share of the listed tenants' requests)
|
|
45
48
|
# plus the share of requests in the window that carry no tenant at all.
|
|
49
|
+
#
|
|
50
|
+
# The untagged count is the window's requests less the tagged ones,
|
|
51
|
+
# and the tagged ones are counted per tenant in app_tenant's index.
|
|
52
|
+
# Counting either side with app_tenant IS [NOT] NULL read every request
|
|
53
|
+
# in the window instead: 1 s on an untagged app, 24 s on a tagged one.
|
|
46
54
|
def self.overview(rows, from, to)
|
|
47
55
|
requests = Telemetry::Execution.requests.between(from, to)
|
|
48
56
|
total = requests.count
|
|
49
|
-
|
|
57
|
+
tagged_total = tagged_tenants(from, to).each_slice(TENANT_SLICE).sum { |slice| requests.where(app_tenant: slice).count }
|
|
50
58
|
tagged = rows.sum { |r| r[:requests] }
|
|
51
59
|
top = rows.max_by { |r| r[:requests] }
|
|
52
60
|
{
|
|
53
61
|
tenants: rows.size, top_tenant: top && top[:tenant],
|
|
54
62
|
top_share: tagged.zero? ? 0.0 : (top[:requests] * 100.0 / tagged).round(1),
|
|
55
63
|
with_errors: rows.count { |r| r[:errors].positive? },
|
|
56
|
-
untagged_share: total.zero? ? 0.0 : (
|
|
64
|
+
untagged_share: total.zero? ? 0.0 : ((total - tagged_total) * 100.0 / total).round(1)
|
|
57
65
|
}
|
|
58
66
|
end
|
|
59
67
|
|
|
68
|
+
# Every tenant that sent anything in the window, from the tenant-led
|
|
69
|
+
# index (a skip-scan over distinct values, not the window's rows).
|
|
70
|
+
def self.tagged_tenants(from, to)
|
|
71
|
+
Telemetry::Execution.between(from, to).where.not(app_tenant: nil).distinct.pluck(:app_tenant)
|
|
72
|
+
end
|
|
73
|
+
|
|
60
74
|
# Window totals for one tenant. Durations in milliseconds, like the rest
|
|
61
75
|
# of the tenant props.
|
|
62
76
|
def self.summary(tenant, from, to)
|
|
@@ -129,9 +143,19 @@ module Railwatch
|
|
|
129
143
|
.each { |tenant, at| rows[tenant][:last_seen_at] = at }
|
|
130
144
|
end
|
|
131
145
|
|
|
132
|
-
|
|
146
|
+
# Only for the tenants already found, by name: grouped by tenant and a
|
|
147
|
+
# time expression SQLite would not seek the app_tenant index and read
|
|
148
|
+
# every request in the window instead -- 11 s over a week on an app
|
|
149
|
+
# with no tenants at all, to draw nothing.
|
|
150
|
+
def self.absorb_sparklines(rows, from, to)
|
|
151
|
+
tenants = rows.keys.compact
|
|
152
|
+
return if tenants.empty?
|
|
153
|
+
|
|
133
154
|
width = bucket_width(from, to, SPARKLINE_BUCKETS)
|
|
134
|
-
counts =
|
|
155
|
+
counts = tenants.each_slice(TENANT_SLICE).flat_map do |slice|
|
|
156
|
+
Telemetry::Execution.requests.between(from, to).where(app_tenant: slice)
|
|
157
|
+
.group(:app_tenant, bucket_sql(from, width)).count.to_a
|
|
158
|
+
end
|
|
135
159
|
counts.each do |(tenant, index), count|
|
|
136
160
|
rows[tenant][:sparkline][[ index.to_i, SPARKLINE_BUCKETS - 1 ].min] += count
|
|
137
161
|
end
|
data/db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# Two dashboard aggregates read every row in the window and only a few
|
|
4
|
+
# narrow columns of each: the Jobs page's per-queue breakdown and the
|
|
5
|
+
# Processes page's health series. Through (kind, occurred_at) and
|
|
6
|
+
# (sampled_at) SQLite finds the rows in order but then fetches each one
|
|
7
|
+
# from the table, and on a telemetry file much larger than memory those
|
|
8
|
+
# rows are scattered pages on disk. On the platform's own 30 GB tenant,
|
|
9
|
+
# over a week: 38 s for the queue breakdown (491,000 job attempts) and
|
|
10
|
+
# 19 s for the health series (263,000 samples), with the GVL held the
|
|
11
|
+
# whole time, so every other request in the process waited too.
|
|
12
|
+
#
|
|
13
|
+
# Both indexes carry every column their query reads, so the whole answer
|
|
14
|
+
# comes from one contiguous index range: 335 ms and 370 ms on a copy of
|
|
15
|
+
# the same file. They cost 151 MB and 29 MB there, and took 13 s and 2 s
|
|
16
|
+
# to build.
|
|
17
|
+
#
|
|
18
|
+
# broadcasts is the one windowed table with no index led by occurred_at,
|
|
19
|
+
# so the Broadcasts page's "recent" list sorted the whole window to show
|
|
20
|
+
# 100 rows. Every other raw table has had this index since it was created.
|
|
21
|
+
#
|
|
22
|
+
# Free-text search over jobs and requests also matches the exception
|
|
23
|
+
# preview, which only a failed execution carries (15 of 491,000 job
|
|
24
|
+
# attempts in that same week). A partial index over just those rows lets
|
|
25
|
+
# Execution.named_like seek them instead of reading the window.
|
|
26
|
+
class AddCoveringIndexesForDashboardAggregates < ActiveRecord::Migration[8.1]
|
|
27
|
+
def change
|
|
28
|
+
add_index :executions, [ :kind, :occurred_at, :queue, :outcome, :queue_latency ], name: "idx_executions_queue_stats"
|
|
29
|
+
add_index :health_samples, [ :sampled_at, :threads_max, :threads_busy, :backlog, :queue_depth, :queue_latency ],
|
|
30
|
+
name: "idx_health_samples_series"
|
|
31
|
+
add_index :broadcasts, :occurred_at
|
|
32
|
+
add_index :executions, [ :kind, :occurred_at ], name: "idx_executions_with_preview", where: "exception_preview IS NOT NULL"
|
|
33
|
+
end
|
|
34
|
+
end
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# The People page groups the window's signed-in requests by user_ref.
|
|
4
|
+
# Through (user_ref, occurred_at) SQLite found those rows but fetched each
|
|
5
|
+
# one from the table for kind and status: 4 s over 30 days on the
|
|
6
|
+
# platform's own tenant, where 18,000 of 2.8 million executions carry a
|
|
7
|
+
# user. This index holds every column the query reads, and only for rows
|
|
8
|
+
# that have a user, so it answers in 18 ms from under a megabyte. It took
|
|
9
|
+
# 10 s to build on a copy of that 30 GB file.
|
|
10
|
+
class AddPeopleIndex < ActiveRecord::Migration[8.1]
|
|
11
|
+
def change
|
|
12
|
+
add_index :executions, [ :user_ref, :kind, :occurred_at, :status ], name: "idx_executions_people",
|
|
13
|
+
where: "user_ref IS NOT NULL"
|
|
14
|
+
end
|
|
15
|
+
end
|
data/lib/railwatch/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: railwatch
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.8.1
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Cole Robertson
|
|
@@ -403,6 +403,8 @@ files:
|
|
|
403
403
|
- db/railwatch_telemetry_migrate/20260915210000_add_slowest_index_to_queries.rb
|
|
404
404
|
- db/railwatch_telemetry_migrate/20260917010000_add_batch_ledger_to_ingest_batches.rb
|
|
405
405
|
- db/railwatch_telemetry_migrate/20260919000100_create_export_queue.rb
|
|
406
|
+
- db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb
|
|
407
|
+
- db/railwatch_telemetry_migrate/20260923000000_add_people_index.rb
|
|
406
408
|
- docs/ai-and-mcp.md
|
|
407
409
|
- docs/configuration.md
|
|
408
410
|
- docs/embedded.md
|