railwatch 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +23 -0
- data/app/models/railwatch/filter_query.rb +5 -5
- data/app/models/railwatch/telemetry/execution.rb +47 -0
- data/app/models/railwatch/telemetry/rollup.rb +73 -9
- data/db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb +34 -0
- data/lib/railwatch/version.rb +1 -1
- metadata +2 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 35e3f727f70f3f5c73c0cbec4dfd83a79dc58141fd4b341c653433e8ae5be5d2
|
|
4
|
+
data.tar.gz: ff3832906ce67466c738a1ea9ec2d8178006c60958f1b3491e4fc31bea84aa3d
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 674d67083f4ed5b3ca7153a18c0f6eaca9588763dc2d5ebf3fc7f40ce11ed54ed2626ea399186aa4d1acac4e22616775258ee3d76a6f01f3830aad11e5e422ad
|
|
7
|
+
data.tar.gz: 37db21eca95d40e29de0f47bf79b89e479acbd812d0381b808b606afe004efca328ac847b0a4db5a1bb78766df41fc44b9cf0e0738aa389e1aeead466ef441ef
|
data/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,29 @@
|
|
|
6
6
|
filled in by the release commit, which is also the only commit that
|
|
7
7
|
touches lib/railwatch/version.rb and Gemfile.lock. See CONTRIBUTING.md. -->
|
|
8
8
|
|
|
9
|
+
## 0.8.0 (2026-09-23)
|
|
10
|
+
|
|
11
|
+
- Percentiles merged across hours are about ten times faster. Each
|
|
12
|
+
rollup's stored t-digest is read straight from its bytes and the
|
|
13
|
+
centroids are sorted once, instead of being pushed one at a time into
|
|
14
|
+
a new digest. A week of the platform's own query rollups (15,000 rows)
|
|
15
|
+
went from 1.6 s to 157 ms. Every page that shows a p50, p95 or p99
|
|
16
|
+
over a window reads this. The percentile is now nearest rank over the
|
|
17
|
+
centroids, the same rule the sub-hour charts use, so the two agree.
|
|
18
|
+
- A new migration adds four indexes. Covering indexes serve the Jobs
|
|
19
|
+
page's per-queue breakdown and the Processes page's health series
|
|
20
|
+
(38 s and 19 s over a week of a 30 GB telemetry file, now 335 ms and
|
|
21
|
+
370 ms). There is an `occurred_at` index on broadcasts, the one
|
|
22
|
+
windowed table without one. A partial index covers executions that
|
|
23
|
+
carry an exception preview. The migration takes about 15 s on a
|
|
24
|
+
30 GB file.
|
|
25
|
+
- Free-text search over jobs (and `Execution.named_like`, which MCP's
|
|
26
|
+
request search uses) matches names through the window's rollups and
|
|
27
|
+
fetches rows by `group_hash`, instead of reading every execution in
|
|
28
|
+
the window. A search matching nothing over a week took 657 ms and
|
|
29
|
+
takes 27 ms. A common text still takes the old path, which finds a
|
|
30
|
+
page within the first few thousand rows.
|
|
31
|
+
|
|
9
32
|
## 0.7.0 (2026-09-23)
|
|
10
33
|
|
|
11
34
|
- The gem ships the dashboard's frontend source (`app/frontend`, without
|
|
@@ -74,11 +74,12 @@ module Railwatch
|
|
|
74
74
|
resource = resource.to_sym
|
|
75
75
|
parsed = parse(query)
|
|
76
76
|
fields = parsed[:fields].except(*Array(except).map(&:to_s))
|
|
77
|
-
|
|
77
|
+
range = bounded_time_range(fields, from, to)
|
|
78
|
+
scope = scope.where(occurred_at: range)
|
|
78
79
|
scope = apply_common(scope, fields)
|
|
79
80
|
|
|
80
81
|
case resource
|
|
81
|
-
when :jobs then apply_jobs(scope, fields, parsed[:text])
|
|
82
|
+
when :jobs then apply_jobs(scope, fields, parsed[:text], range)
|
|
82
83
|
when :exceptions then apply_exceptions(scope, fields, parsed[:text])
|
|
83
84
|
when :logs then apply_logs(scope, fields, parsed[:text])
|
|
84
85
|
when :queries then apply_queries(scope, fields, parsed[:text])
|
|
@@ -120,15 +121,14 @@ module Railwatch
|
|
|
120
121
|
end
|
|
121
122
|
private_class_method :apply_common
|
|
122
123
|
|
|
123
|
-
def self.apply_jobs(scope, fields, text)
|
|
124
|
+
def self.apply_jobs(scope, fields, text, range)
|
|
124
125
|
outcome = fields["outcome"].presence || fields["status"].presence
|
|
125
126
|
scope = scope.where(queue: fields["queue"]) if fields["queue"].present?
|
|
126
127
|
scope = scope.where(outcome: outcome) if outcome
|
|
127
128
|
scope = scope.where(name: fields["class"]) if fields["class"].present?
|
|
128
129
|
scope = scope.where(job_id: fields["job_id"]) if fields["job_id"].present?
|
|
129
130
|
return scope unless text.present?
|
|
130
|
-
|
|
131
|
-
scope.where("name LIKE :pattern OR exception_preview LIKE :pattern", pattern: pattern)
|
|
131
|
+
scope.merge(Telemetry::Execution.named_like("job_attempt", text, range, previews: true))
|
|
132
132
|
end
|
|
133
133
|
private_class_method :apply_jobs
|
|
134
134
|
|
|
@@ -17,6 +17,53 @@ module Railwatch
|
|
|
17
17
|
scope :between, ->(from, to) { where(occurred_at: from..to) }
|
|
18
18
|
scope :failed, -> { where("status >= 500 OR outcome = 'failed'") }
|
|
19
19
|
|
|
20
|
+
# Executions of `kind` in [from, to] whose name contains `text`, or
|
|
21
|
+
# (with previews: true) whose exception preview does. A LIKE on the rows
|
|
22
|
+
# themselves reads every execution in the window until it has a page:
|
|
23
|
+
# 32 s for a week of the platform's own requests when the text is rare.
|
|
24
|
+
# Names come from a few hundred groups, so the text is matched against
|
|
25
|
+
# the window's rollups and the rows are fetched by group_hash. Rollups
|
|
26
|
+
# trail ingest by up to a minute, so the last FRESH_NAMES are matched on
|
|
27
|
+
# the rows themselves. Only failures carry a preview, and a partial
|
|
28
|
+
# index holds just those rows.
|
|
29
|
+
#
|
|
30
|
+
# Each branch names its index and the outer lookup is by rowid, so the
|
|
31
|
+
# plan does not depend on planner statistics (an embedded install's are
|
|
32
|
+
# sampled, and sampled statistics make kind look selective). When the
|
|
33
|
+
# text is common the old walk is the fast one -- a page of matches
|
|
34
|
+
# turns up in the first few thousand rows -- so a text the rollups say
|
|
35
|
+
# matches DENSE_MATCHES rows or more keeps it. On a copy of the
|
|
36
|
+
# platform's own tenant, a week of job attempts: a text matching
|
|
37
|
+
# nothing took 657 ms walking and 27 ms here; one matching 32,766 took
|
|
38
|
+
# 11 ms walking and 215 ms here.
|
|
39
|
+
FRESH_NAMES = 5.minutes
|
|
40
|
+
DENSE_MATCHES = 1_000
|
|
41
|
+
|
|
42
|
+
# `range` is used as given, so an exclusive or empty one (FilterQuery's
|
|
43
|
+
# after:/before: can narrow a window to nothing) stays that way.
|
|
44
|
+
def self.named_like(kind, text, range, previews: false)
|
|
45
|
+
from, to = range.begin, range.end
|
|
46
|
+
pattern = "%#{sanitize_sql_like(text)}%"
|
|
47
|
+
window = where(kind: kind, occurred_at: range)
|
|
48
|
+
groups = Rollup.for_type(kind).between(from, to).where("name LIKE ? ESCAPE '\\'", pattern)
|
|
49
|
+
if !TelemetryRecord.sqlite? || groups.sum(:count) >= DENSE_MATCHES
|
|
50
|
+
like = previews ? "name LIKE :pattern ESCAPE '\\' OR exception_preview LIKE :pattern ESCAPE '\\'" : "name LIKE :pattern ESCAPE '\\'"
|
|
51
|
+
return window.where(like, pattern: pattern)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
by_group = indexed_by("index_executions_on_group_hash_and_occurred_at")
|
|
55
|
+
.where(group_hash: groups.distinct.select(:group_hash), occurred_at: range, kind: kind)
|
|
56
|
+
fresh = where(kind: kind, occurred_at: [ from, to - FRESH_NAMES ].max..to).where("name LIKE ? ESCAPE '\\'", pattern)
|
|
57
|
+
ids = [ by_group, fresh ]
|
|
58
|
+
ids << indexed_by("idx_executions_with_preview").where(kind: kind, occurred_at: range)
|
|
59
|
+
.where.not(exception_preview: nil).where("exception_preview LIKE ? ESCAPE '\\'", pattern) if previews
|
|
60
|
+
from("#{quoted_table_name} NOT INDEXED").where(kind: kind, occurred_at: range)
|
|
61
|
+
.where(ids.map { |branch| "#{quoted_table_name}.id IN (#{branch.select(:id).to_sql})" }.join(" OR "))
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def self.indexed_by(index) = unscoped.from("#{quoted_table_name} INDEXED BY #{index}")
|
|
65
|
+
private_class_method :indexed_by
|
|
66
|
+
|
|
20
67
|
# Servers that reported an execution since `time`. Asked once per kind:
|
|
21
68
|
# there is no index led by occurred_at alone, so the obvious
|
|
22
69
|
# `where(occurred_at: time..).distinct.pluck(:server)` scans the whole
|
|
@@ -55,21 +55,16 @@ module Railwatch
|
|
|
55
55
|
def self.summarize(relation)
|
|
56
56
|
rows = relation.to_a
|
|
57
57
|
return { count: 0, errors: 0, client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0, max: 0, extra: {} } if rows.empty?
|
|
58
|
-
|
|
59
|
-
# `+` built a brand-new digest from both operands' centroids on every
|
|
60
|
-
# row, so merging N rows re-pushed every earlier centroid N times:
|
|
61
|
-
# 311 rows took 1.4s where merge! takes 0.5s, for the same percentiles.
|
|
62
|
-
merged = TDigest::TDigest.new(0.01)
|
|
63
|
-
rows.each { |r| merged.merge!(TDigest::TDigest.from_bytes(r.digest)) if r.digest }
|
|
58
|
+
p50, p95, p99 = merged_percentiles(rows, [ 0.5, 0.95, 0.99 ])
|
|
64
59
|
count = rows.sum(&:count)
|
|
65
60
|
{
|
|
66
61
|
count: count,
|
|
67
62
|
errors: rows.sum(&:error_count),
|
|
68
63
|
client_errors: rows.sum(&:client_error_count),
|
|
69
64
|
avg: count.zero? ? 0 : (rows.sum(&:duration_sum) / count),
|
|
70
|
-
p50:
|
|
71
|
-
p95:
|
|
72
|
-
p99:
|
|
65
|
+
p50: p50,
|
|
66
|
+
p95: p95,
|
|
67
|
+
p99: p99,
|
|
73
68
|
max: rows.map(&:duration_max).max,
|
|
74
69
|
# Everything a type puts in `extra` -- llm_call's cost_nanos and
|
|
75
70
|
# token counts, cache_event's hits and misses -- merged the way
|
|
@@ -80,6 +75,75 @@ module Railwatch
|
|
|
80
75
|
}
|
|
81
76
|
end
|
|
82
77
|
|
|
78
|
+
# Percentiles of the union of every row's centroids, read straight out
|
|
79
|
+
# of the stored bytes. Merging through TDigest pushed each centroid into
|
|
80
|
+
# a red-black tree one at a time: 2.8 s for a week of the platform's
|
|
81
|
+
# own query rollups (18,674 rows) and 7 s for thirty days, on every
|
|
82
|
+
# cold page load. Sorting the centroids once and walking their
|
|
83
|
+
# cumulative counts answers the same question without re-clustering:
|
|
84
|
+
# 353 ms and 1 s for those two windows. The answer is the first
|
|
85
|
+
# centroid whose running total reaches n * p, which is nearest rank --
|
|
86
|
+
# the rule raw_series uses for sub-hour buckets, so both readings of a
|
|
87
|
+
# window agree. (TDigest#percentile compares each centroid's midpoint
|
|
88
|
+
# instead, which puts the median of 1,000 sevens and 300 nines at 9.)
|
|
89
|
+
def self.merged_percentiles(rows, ps)
|
|
90
|
+
means = []
|
|
91
|
+
counts = []
|
|
92
|
+
rows.each { |row| centroids(row.digest, means, counts) if row.digest }
|
|
93
|
+
return ps.map { 0 } if means.empty?
|
|
94
|
+
|
|
95
|
+
order = (0...means.size).sort_by { |i| means[i] }
|
|
96
|
+
total = counts.sum
|
|
97
|
+
targets = ps.map { |p| total * p }
|
|
98
|
+
found = []
|
|
99
|
+
seen = 0
|
|
100
|
+
order.each do |i|
|
|
101
|
+
seen += counts[i]
|
|
102
|
+
found << means[i] while found.size < targets.size && seen >= targets[found.size]
|
|
103
|
+
break if found.size == targets.size
|
|
104
|
+
end
|
|
105
|
+
found << means[order.last] while found.size < targets.size
|
|
106
|
+
found.map(&:to_i)
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# Appends one stored digest's centroid means and weights, in either of
|
|
110
|
+
# the tdigest gem's encodings (TDigest.from_bytes reads the same two).
|
|
111
|
+
def self.centroids(bytes, means, counts)
|
|
112
|
+
format, _compression, size = bytes.unpack("LdL")
|
|
113
|
+
case format
|
|
114
|
+
when TDigest::TDigest::VERBOSE_ENCODING
|
|
115
|
+
means.concat(bytes.unpack("@16d#{size}"))
|
|
116
|
+
counts.concat(bytes.unpack("@#{16 + 8 * size}L#{size}"))
|
|
117
|
+
when TDigest::TDigest::SMALL_ENCODING
|
|
118
|
+
# Means are delta-encoded 4-byte floats; weights are 7-bit varints,
|
|
119
|
+
# one byte each unless a centroid holds 128 or more samples.
|
|
120
|
+
mean = 0.0
|
|
121
|
+
bytes.unpack("@16f#{size}").each { |delta| means << (mean += delta) }
|
|
122
|
+
weights = bytes.byteslice(16 + 4 * size, bytes.bytesize).unpack("C*")
|
|
123
|
+
if weights.size == size
|
|
124
|
+
counts.concat(weights)
|
|
125
|
+
else
|
|
126
|
+
at = 0
|
|
127
|
+
size.times do
|
|
128
|
+
byte = weights[at]
|
|
129
|
+
at += 1
|
|
130
|
+
weight = byte & 0x7f
|
|
131
|
+
shift = 7
|
|
132
|
+
while byte & 0x80 != 0
|
|
133
|
+
byte = weights[at] || 0
|
|
134
|
+
at += 1
|
|
135
|
+
weight += (byte & 0x7f) << shift
|
|
136
|
+
shift += 7
|
|
137
|
+
end
|
|
138
|
+
counts << weight
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
else
|
|
142
|
+
raise ArgumentError, "unknown t-digest encoding #{format}"
|
|
143
|
+
end
|
|
144
|
+
end
|
|
145
|
+
private_class_method :centroids
|
|
146
|
+
|
|
83
147
|
def self.merge_extras(rows)
|
|
84
148
|
rows.each_with_object({}) do |row, out|
|
|
85
149
|
row.extra.each do |key, value|
|
data/db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# Two dashboard aggregates read every row in the window and only a few
|
|
4
|
+
# narrow columns of each: the Jobs page's per-queue breakdown and the
|
|
5
|
+
# Processes page's health series. Through (kind, occurred_at) and
|
|
6
|
+
# (sampled_at) SQLite finds the rows in order but then fetches each one
|
|
7
|
+
# from the table, and on a telemetry file much larger than memory those
|
|
8
|
+
# rows are scattered pages on disk. On the platform's own 30 GB tenant,
|
|
9
|
+
# over a week: 38 s for the queue breakdown (491,000 job attempts) and
|
|
10
|
+
# 19 s for the health series (263,000 samples), with the GVL held the
|
|
11
|
+
# whole time, so every other request in the process waited too.
|
|
12
|
+
#
|
|
13
|
+
# Both indexes carry every column their query reads, so the whole answer
|
|
14
|
+
# comes from one contiguous index range: 335 ms and 370 ms on a copy of
|
|
15
|
+
# the same file. They cost 151 MB and 29 MB there, and took 13 s and 2 s
|
|
16
|
+
# to build.
|
|
17
|
+
#
|
|
18
|
+
# broadcasts is the one windowed table with no index led by occurred_at,
|
|
19
|
+
# so the Broadcasts page's "recent" list sorted the whole window to show
|
|
20
|
+
# 100 rows. Every other raw table has had this index since it was created.
|
|
21
|
+
#
|
|
22
|
+
# Free-text search over jobs and requests also matches the exception
|
|
23
|
+
# preview, which only a failed execution carries (15 of 491,000 job
|
|
24
|
+
# attempts in that same week). A partial index over just those rows lets
|
|
25
|
+
# Execution.named_like seek them instead of reading the window.
|
|
26
|
+
class AddCoveringIndexesForDashboardAggregates < ActiveRecord::Migration[8.1]
|
|
27
|
+
def change
|
|
28
|
+
add_index :executions, [ :kind, :occurred_at, :queue, :outcome, :queue_latency ], name: "idx_executions_queue_stats"
|
|
29
|
+
add_index :health_samples, [ :sampled_at, :threads_max, :threads_busy, :backlog, :queue_depth, :queue_latency ],
|
|
30
|
+
name: "idx_health_samples_series"
|
|
31
|
+
add_index :broadcasts, :occurred_at
|
|
32
|
+
add_index :executions, [ :kind, :occurred_at ], name: "idx_executions_with_preview", where: "exception_preview IS NOT NULL"
|
|
33
|
+
end
|
|
34
|
+
end
|
data/lib/railwatch/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: railwatch
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.8.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Cole Robertson
|
|
@@ -403,6 +403,7 @@ files:
|
|
|
403
403
|
- db/railwatch_telemetry_migrate/20260915210000_add_slowest_index_to_queries.rb
|
|
404
404
|
- db/railwatch_telemetry_migrate/20260917010000_add_batch_ledger_to_ingest_batches.rb
|
|
405
405
|
- db/railwatch_telemetry_migrate/20260919000100_create_export_queue.rb
|
|
406
|
+
- db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb
|
|
406
407
|
- docs/ai-and-mcp.md
|
|
407
408
|
- docs/configuration.md
|
|
408
409
|
- docs/embedded.md
|