railwatch 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 2033d4d399f9d52afddba05178186c651c557d4a870decd78e0bf9cf90b6c9e7
4
- data.tar.gz: 19cbbe8a3963391f28742b889fd5be1b68c006a59e2cefbbd0bcf2d33da1d3ca
3
+ metadata.gz: 35e3f727f70f3f5c73c0cbec4dfd83a79dc58141fd4b341c653433e8ae5be5d2
4
+ data.tar.gz: ff3832906ce67466c738a1ea9ec2d8178006c60958f1b3491e4fc31bea84aa3d
5
5
  SHA512:
6
- metadata.gz: 43c847212862c33bf9bf9d6cb95354808ae3a033f79273351282c988a1334c83bb24f738789226f1feb9fc3798f60cba81bd460e2c783d8c1f8858ad1d4c1dc1
7
- data.tar.gz: f73a3052b983d2493ca2a3e26a264110dc79ef37398badd3359fac7fe9f254f6eac3b6e3f34fc1583fb31434872f3e27faf9b31286cd7cb43544e5a3ba4514e3
6
+ metadata.gz: 674d67083f4ed5b3ca7153a18c0f6eaca9588763dc2d5ebf3fc7f40ce11ed54ed2626ea399186aa4d1acac4e22616775258ee3d76a6f01f3830aad11e5e422ad
7
+ data.tar.gz: 37db21eca95d40e29de0f47bf79b89e479acbd812d0381b808b606afe004efca328ac847b0a4db5a1bb78766df41fc44b9cf0e0738aa389e1aeead466ef441ef
data/CHANGELOG.md CHANGED
@@ -6,6 +6,29 @@
6
6
  filled in by the release commit, which is also the only commit that
7
7
  touches lib/railwatch/version.rb and Gemfile.lock. See CONTRIBUTING.md. -->
8
8
 
9
+ ## 0.8.0 (2026-09-23)
10
+
11
+ - Percentiles merged across hours are about ten times faster. Each
12
+ rollup's stored t-digest is read straight from its bytes and the
13
+ centroids are sorted once, instead of being pushed one at a time into
14
+ a new digest. A week of the platform's own query rollups (15,000 rows)
15
+ went from 1.6 s to 157 ms. Every page that shows a p50, p95 or p99
16
+ over a window reads this. The percentile is now nearest rank over the
17
+ centroids, the same rule the sub-hour charts use, so the two agree.
18
+ - A new migration adds four indexes. Covering indexes serve the Jobs
19
+ page's per-queue breakdown and the Processes page's health series
20
+ (38 s and 19 s over a week of a 30 GB telemetry file, now 335 ms and
21
+ 370 ms). There is an `occurred_at` index on broadcasts, the one
22
+ windowed table without one. A partial index covers executions that
23
+ carry an exception preview. The migration takes about 15 s on a
24
+ 30 GB file.
25
+ - Free-text search over jobs (and `Execution.named_like`, which MCP's
26
+ request search uses) matches names through the window's rollups and
27
+ fetches rows by `group_hash`, instead of reading every execution in
28
+ the window. A search matching nothing over a week took 657 ms and
29
+ takes 27 ms. A common text still takes the old path, which finds a
30
+ page within the first few thousand rows.
31
+
9
32
  ## 0.7.0 (2026-09-23)
10
33
 
11
34
  - The gem ships the dashboard's frontend source (`app/frontend`, without
@@ -74,11 +74,12 @@ module Railwatch
74
74
  resource = resource.to_sym
75
75
  parsed = parse(query)
76
76
  fields = parsed[:fields].except(*Array(except).map(&:to_s))
77
- scope = scope.where(occurred_at: bounded_time_range(fields, from, to))
77
+ range = bounded_time_range(fields, from, to)
78
+ scope = scope.where(occurred_at: range)
78
79
  scope = apply_common(scope, fields)
79
80
 
80
81
  case resource
81
- when :jobs then apply_jobs(scope, fields, parsed[:text])
82
+ when :jobs then apply_jobs(scope, fields, parsed[:text], range)
82
83
  when :exceptions then apply_exceptions(scope, fields, parsed[:text])
83
84
  when :logs then apply_logs(scope, fields, parsed[:text])
84
85
  when :queries then apply_queries(scope, fields, parsed[:text])
@@ -120,15 +121,14 @@ module Railwatch
120
121
  end
121
122
  private_class_method :apply_common
122
123
 
123
- def self.apply_jobs(scope, fields, text)
124
+ def self.apply_jobs(scope, fields, text, range)
124
125
  outcome = fields["outcome"].presence || fields["status"].presence
125
126
  scope = scope.where(queue: fields["queue"]) if fields["queue"].present?
126
127
  scope = scope.where(outcome: outcome) if outcome
127
128
  scope = scope.where(name: fields["class"]) if fields["class"].present?
128
129
  scope = scope.where(job_id: fields["job_id"]) if fields["job_id"].present?
129
130
  return scope unless text.present?
130
- pattern = "%#{Telemetry::Execution.sanitize_sql_like(text)}%"
131
- scope.where("name LIKE :pattern OR exception_preview LIKE :pattern", pattern: pattern)
131
+ scope.merge(Telemetry::Execution.named_like("job_attempt", text, range, previews: true))
132
132
  end
133
133
  private_class_method :apply_jobs
134
134
 
@@ -17,6 +17,53 @@ module Railwatch
17
17
  scope :between, ->(from, to) { where(occurred_at: from..to) }
18
18
  scope :failed, -> { where("status >= 500 OR outcome = 'failed'") }
19
19
 
20
+ # Executions of `kind` in [from, to] whose name contains `text`, or
21
+ # (with previews: true) whose exception preview does. A LIKE on the rows
22
+ # themselves reads every execution in the window until it has a page:
23
+ # 32 s for a week of the platform's own requests when the text is rare.
24
+ # Names come from a few hundred groups, so the text is matched against
25
+ # the window's rollups and the rows are fetched by group_hash. Rollups
26
+ # trail ingest by up to a minute, so the last FRESH_NAMES are matched on
27
+ # the rows themselves. Only failures carry a preview, and a partial
28
+ # index holds just those rows.
29
+ #
30
+ # Each branch names its index and the outer lookup is by rowid, so the
31
+ # plan does not depend on planner statistics (an embedded install's are
32
+ # sampled, and sampled statistics make kind look selective). When the
33
+ # text is common the old walk is the fast one -- a page of matches
34
+ # turns up in the first few thousand rows -- so a text the rollups say
35
+ # matches DENSE_MATCHES rows or more keeps it. On a copy of the
36
+ # platform's own tenant, a week of job attempts: a text matching
37
+ # nothing took 657 ms walking and 27 ms here; one matching 32,766 took
38
+ # 11 ms walking and 215 ms here.
39
+ FRESH_NAMES = 5.minutes
40
+ DENSE_MATCHES = 1_000
41
+
42
+ # `range` is used as given, so an exclusive or empty one (FilterQuery's
43
+ # after:/before: can narrow a window to nothing) stays that way.
44
+ def self.named_like(kind, text, range, previews: false)
45
+ from, to = range.begin, range.end
46
+ pattern = "%#{sanitize_sql_like(text)}%"
47
+ window = where(kind: kind, occurred_at: range)
48
+ groups = Rollup.for_type(kind).between(from, to).where("name LIKE ? ESCAPE '\\'", pattern)
49
+ if !TelemetryRecord.sqlite? || groups.sum(:count) >= DENSE_MATCHES
50
+ like = previews ? "name LIKE :pattern ESCAPE '\\' OR exception_preview LIKE :pattern ESCAPE '\\'" : "name LIKE :pattern ESCAPE '\\'"
51
+ return window.where(like, pattern: pattern)
52
+ end
53
+
54
+ by_group = indexed_by("index_executions_on_group_hash_and_occurred_at")
55
+ .where(group_hash: groups.distinct.select(:group_hash), occurred_at: range, kind: kind)
56
+ fresh = where(kind: kind, occurred_at: [ from, to - FRESH_NAMES ].max..to).where("name LIKE ? ESCAPE '\\'", pattern)
57
+ ids = [ by_group, fresh ]
58
+ ids << indexed_by("idx_executions_with_preview").where(kind: kind, occurred_at: range)
59
+ .where.not(exception_preview: nil).where("exception_preview LIKE ? ESCAPE '\\'", pattern) if previews
60
+ from("#{quoted_table_name} NOT INDEXED").where(kind: kind, occurred_at: range)
61
+ .where(ids.map { |branch| "#{quoted_table_name}.id IN (#{branch.select(:id).to_sql})" }.join(" OR "))
62
+ end
63
+
64
+ def self.indexed_by(index) = unscoped.from("#{quoted_table_name} INDEXED BY #{index}")
65
+ private_class_method :indexed_by
66
+
20
67
  # Servers that reported an execution since `time`. Asked once per kind:
21
68
  # there is no index led by occurred_at alone, so the obvious
22
69
  # `where(occurred_at: time..).distinct.pluck(:server)` scans the whole
@@ -55,21 +55,16 @@ module Railwatch
55
55
  def self.summarize(relation)
56
56
  rows = relation.to_a
57
57
  return { count: 0, errors: 0, client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0, max: 0, extra: {} } if rows.empty?
58
- # merge! pushes the row's centroids into one accumulating digest.
59
- # `+` built a brand-new digest from both operands' centroids on every
60
- # row, so merging N rows re-pushed every earlier centroid N times:
61
- # 311 rows took 1.4s where merge! takes 0.5s, for the same percentiles.
62
- merged = TDigest::TDigest.new(0.01)
63
- rows.each { |r| merged.merge!(TDigest::TDigest.from_bytes(r.digest)) if r.digest }
58
+ p50, p95, p99 = merged_percentiles(rows, [ 0.5, 0.95, 0.99 ])
64
59
  count = rows.sum(&:count)
65
60
  {
66
61
  count: count,
67
62
  errors: rows.sum(&:error_count),
68
63
  client_errors: rows.sum(&:client_error_count),
69
64
  avg: count.zero? ? 0 : (rows.sum(&:duration_sum) / count),
70
- p50: merged.percentile(0.5).to_i,
71
- p95: merged.percentile(0.95).to_i,
72
- p99: merged.percentile(0.99).to_i,
65
+ p50: p50,
66
+ p95: p95,
67
+ p99: p99,
73
68
  max: rows.map(&:duration_max).max,
74
69
  # Everything a type puts in `extra` -- llm_call's cost_nanos and
75
70
  # token counts, cache_event's hits and misses -- merged the way
@@ -80,6 +75,75 @@ module Railwatch
80
75
  }
81
76
  end
82
77
 
78
+ # Percentiles of the union of every row's centroids, read straight out
79
+ # of the stored bytes. Merging through TDigest pushed each centroid into
80
+ # a red-black tree one at a time: 2.8 s for a week of the platform's
81
+ # own query rollups (18,674 rows) and 7 s for thirty days, on every
82
+ # cold page load. Sorting the centroids once and walking their
83
+ # cumulative counts answers the same question without re-clustering:
84
+ # 353 ms and 1 s for those two windows. The answer is the first
85
+ # centroid whose running total reaches n * p, which is nearest rank --
86
+ # the rule raw_series uses for sub-hour buckets, so both readings of a
87
+ # window agree. (TDigest#percentile compares each centroid's midpoint
88
+ # instead, which puts the median of 1,000 sevens and 300 nines at 9.)
89
+ def self.merged_percentiles(rows, ps)
90
+ means = []
91
+ counts = []
92
+ rows.each { |row| centroids(row.digest, means, counts) if row.digest }
93
+ return ps.map { 0 } if means.empty?
94
+
95
+ order = (0...means.size).sort_by { |i| means[i] }
96
+ total = counts.sum
97
+ targets = ps.map { |p| total * p }
98
+ found = []
99
+ seen = 0
100
+ order.each do |i|
101
+ seen += counts[i]
102
+ found << means[i] while found.size < targets.size && seen >= targets[found.size]
103
+ break if found.size == targets.size
104
+ end
105
+ found << means[order.last] while found.size < targets.size
106
+ found.map(&:to_i)
107
+ end
108
+
109
+ # Appends one stored digest's centroid means and weights, in either of
110
+ # the tdigest gem's encodings (TDigest.from_bytes reads the same two).
111
+ def self.centroids(bytes, means, counts)
112
+ format, _compression, size = bytes.unpack("LdL")
113
+ case format
114
+ when TDigest::TDigest::VERBOSE_ENCODING
115
+ means.concat(bytes.unpack("@16d#{size}"))
116
+ counts.concat(bytes.unpack("@#{16 + 8 * size}L#{size}"))
117
+ when TDigest::TDigest::SMALL_ENCODING
118
+ # Means are delta-encoded 4-byte floats; weights are 7-bit varints,
119
+ # one byte each unless a centroid holds 128 or more samples.
120
+ mean = 0.0
121
+ bytes.unpack("@16f#{size}").each { |delta| means << (mean += delta) }
122
+ weights = bytes.byteslice(16 + 4 * size, bytes.bytesize).unpack("C*")
123
+ if weights.size == size
124
+ counts.concat(weights)
125
+ else
126
+ at = 0
127
+ size.times do
128
+ byte = weights[at]
129
+ at += 1
130
+ weight = byte & 0x7f
131
+ shift = 7
132
+ while byte & 0x80 != 0
133
+ byte = weights[at] || 0
134
+ at += 1
135
+ weight += (byte & 0x7f) << shift
136
+ shift += 7
137
+ end
138
+ counts << weight
139
+ end
140
+ end
141
+ else
142
+ raise ArgumentError, "unknown t-digest encoding #{format}"
143
+ end
144
+ end
145
+ private_class_method :centroids
146
+
83
147
  def self.merge_extras(rows)
84
148
  rows.each_with_object({}) do |row, out|
85
149
  row.extra.each do |key, value|
@@ -0,0 +1,34 @@
1
+ # frozen_string_literal: true
2
+
3
+ # Two dashboard aggregates read every row in the window and only a few
4
+ # narrow columns of each: the Jobs page's per-queue breakdown and the
5
+ # Processes page's health series. Through (kind, occurred_at) and
6
+ # (sampled_at) SQLite finds the rows in order but then fetches each one
7
+ # from the table, and on a telemetry file much larger than memory those
8
+ # rows are scattered pages on disk. On the platform's own 30 GB tenant,
9
+ # over a week: 38 s for the queue breakdown (491,000 job attempts) and
10
+ # 19 s for the health series (263,000 samples), with the GVL held the
11
+ # whole time, so every other request in the process waited too.
12
+ #
13
+ # Both indexes carry every column their query reads, so the whole answer
14
+ # comes from one contiguous index range: 335 ms and 370 ms on a copy of
15
+ # the same file. They cost 151 MB and 29 MB there, and took 13 s and 2 s
16
+ # to build.
17
+ #
18
+ # broadcasts is the one windowed table with no index led by occurred_at,
19
+ # so the Broadcasts page's "recent" list sorted the whole window to show
20
+ # 100 rows. Every other raw table has had this index since it was created.
21
+ #
22
+ # Free-text search over jobs and requests also matches the exception
23
+ # preview, which only a failed execution carries (15 of 491,000 job
24
+ # attempts in that same week). A partial index over just those rows lets
25
+ # Execution.named_like seek them instead of reading the window.
26
+ class AddCoveringIndexesForDashboardAggregates < ActiveRecord::Migration[8.1]
27
+ def change
28
+ add_index :executions, [ :kind, :occurred_at, :queue, :outcome, :queue_latency ], name: "idx_executions_queue_stats"
29
+ add_index :health_samples, [ :sampled_at, :threads_max, :threads_busy, :backlog, :queue_depth, :queue_latency ],
30
+ name: "idx_health_samples_series"
31
+ add_index :broadcasts, :occurred_at
32
+ add_index :executions, [ :kind, :occurred_at ], name: "idx_executions_with_preview", where: "exception_preview IS NOT NULL"
33
+ end
34
+ end
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Railwatch
4
- VERSION = "0.7.0"
4
+ VERSION = "0.8.0"
5
5
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: railwatch
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.7.0
4
+ version: 0.8.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Cole Robertson
@@ -403,6 +403,7 @@ files:
403
403
  - db/railwatch_telemetry_migrate/20260915210000_add_slowest_index_to_queries.rb
404
404
  - db/railwatch_telemetry_migrate/20260917010000_add_batch_ledger_to_ingest_batches.rb
405
405
  - db/railwatch_telemetry_migrate/20260919000100_create_export_queue.rb
406
+ - db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb
406
407
  - docs/ai-and-mcp.md
407
408
  - docs/configuration.md
408
409
  - docs/embedded.md