railwatch 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 35e3f727f70f3f5c73c0cbec4dfd83a79dc58141fd4b341c653433e8ae5be5d2
4
- data.tar.gz: ff3832906ce67466c738a1ea9ec2d8178006c60958f1b3491e4fc31bea84aa3d
3
+ metadata.gz: d3e41b1890c573cc71fe35d7a9ac134dc3964e4250126bd7c898132800ba47d9
4
+ data.tar.gz: 84f59bb0bc68c014618c564533b0f3a5266062b1fac93a252a5a41c3632f0136
5
5
  SHA512:
6
- metadata.gz: 674d67083f4ed5b3ca7153a18c0f6eaca9588763dc2d5ebf3fc7f40ce11ed54ed2626ea399186aa4d1acac4e22616775258ee3d76a6f01f3830aad11e5e422ad
7
- data.tar.gz: 37db21eca95d40e29de0f47bf79b89e479acbd812d0381b808b606afe004efca328ac847b0a4db5a1bb78766df41fc44b9cf0e0738aa389e1aeead466ef441ef
6
+ metadata.gz: 4472da6ecdbb054cd17cae2a31cf3d1f0a5a6237e4964af2c9da04fa2b754a95a7aa91c268dd43ef9a29791762b82c2edffb446557b0f2b6b5a3a47f6b504a8c
7
+ data.tar.gz: 276639c1d1f4c9632892a419f83443407a41af5e33c062a426e50bf9a3f5fa8daf039b267157c14079e2c826dbd5b025e0a2e955d43450d28b59be31637fb6b4
data/CHANGELOG.md CHANGED
@@ -6,6 +6,25 @@
6
6
  filled in by the release commit, which is also the only commit that
7
7
  touches lib/railwatch/version.rb and Gemfile.lock. See CONTRIBUTING.md. -->
8
8
 
9
+ ## 0.8.2 (2026-09-23)
10
+
11
+ - The Profiles page's profiled share counts only the executions inside
12
+ its window. A window starting or ending mid-hour counted the whole of
13
+ its first and last hours, because a rollup holds its whole hour; the
14
+ partial hours are now counted from the rows.
15
+
16
+ ## 0.8.1 (2026-09-23)
17
+
18
+ - The Tenants page reads only through the tenant-led indexes. Sparklines
19
+ are drawn for the tenants already found, by name. The untagged share is
20
+ the window's requests less the tagged ones, and the tagged ones are
21
+ counted per tenant. Both used to read every request in the window: over
22
+ a week, 11 s on an app with no tenants and 24 s on one with them.
23
+ - A partial covering index serves the People page: 4 s over 30 days,
24
+ now 18 ms, from under a megabyte.
25
+ - The Profiles page takes the window's execution count from rollups
26
+ instead of counting every execution (1.2 s over a week).
27
+
9
28
  ## 0.8.0 (2026-09-23)
10
29
 
11
30
  - Percentiles merged across hours are about ten times faster. Each
@@ -13,7 +13,7 @@ module Railwatch
13
13
  rows, count, avg_samples, profiled, executions = telemetry do
14
14
  scope = Telemetry::Profile.between(from, to)
15
15
  [ scope.recent.limit(SCAN_LIMIT).select(:id, :group_hash, :execution_preview, :duration, :samples, :occurred_at).to_a,
16
- scope.count, scope.average(:samples), scope.distinct.count(:execution_id), Telemetry::Execution.between(from, to).count ]
16
+ scope.count, scope.average(:samples), scope.distinct.count(:execution_id), executions_in(from, to) ]
17
17
  end
18
18
  render inertia: { profiles: groups(rows),
19
19
  summary: { profiles: count, executions: executions, profiled: profiled, avg_samples: avg_samples.to_f.round(0) } }
@@ -29,6 +29,21 @@ module Railwatch
29
29
 
30
30
  private
31
31
 
32
+ # How many executions the window holds, as the profiled share's
33
+ # denominator. Every kind is rolled up, so whole hours come from a few
34
+ # hundred rollup rows rather than counting every execution (1.2 s over a
35
+ # week of the platform's own). A rollup holds its whole hour, so the
36
+ # partial hours at either end of the window are counted from the rows.
37
+ def executions_in(from, to)
38
+ first = from.beginning_of_hour == from ? from : from.beginning_of_hour + 1.hour
39
+ last = to.beginning_of_hour
40
+ return Telemetry::Execution.between(from, to).count if first >= last
41
+
42
+ Telemetry::Execution.where(occurred_at: from...first).count +
43
+ Telemetry::Rollup.for_type(Telemetry::Execution::KINDS).where(bucket: first...last).sum(:count) +
44
+ Telemetry::Execution.where(occurred_at: last..to).count
45
+ end
46
+
32
47
  # Rows arrive newest-first, so the head of each group is the profile the
33
48
  # page links to.
34
49
  def groups(rows)
@@ -18,6 +18,9 @@ module Railwatch
18
18
  # aggregate), so only the busiest tenants get one; the rest report 0.
19
19
  P95_TENANTS = 50
20
20
  SPARKLINE_BUCKETS = 12
21
+ # Tenant names are bound one variable each, and SQLite allows 32,766
22
+ # per statement; an app with more tenants than that is asked in slices.
23
+ TENANT_SLICE = 1_000
21
24
  SORTS = { "requests" => :requests, "errors" => :errors, "p95" => :p95, "users" => :users }.freeze
22
25
 
23
26
  ERRORS_SQL = Arel.sql("SUM(CASE WHEN status >= 500 THEN 1 ELSE 0 END)")
@@ -35,7 +38,7 @@ module Railwatch
35
38
  absorb_counts(rows, filtered(Telemetry::Exception.between(from, to), q), :exceptions)
36
39
  absorb_counts(rows, filtered(Telemetry::Log.between(from, to), q), :logs)
37
40
  absorb_last_seen(rows, from, to, q)
38
- absorb_sparklines(rows, from, to, q)
41
+ absorb_sparklines(rows, from, to)
39
42
  absorb_p95s(rows.values.max_by(P95_TENANTS) { |r| r[:requests] }, from, to)
40
43
  sorted(rows.values, sort, dir)
41
44
  end
@@ -43,20 +46,31 @@ module Railwatch
43
46
  # Headline numbers for the index page, computed from the rows it shows
44
47
  # (so the top-tenant share is a share of the listed tenants' requests)
45
48
  # plus the share of requests in the window that carry no tenant at all.
49
+ #
50
+ # The untagged count is the window's requests less the tagged ones,
51
+ # and the tagged ones are counted per tenant in app_tenant's index.
52
+ # Counting either side with app_tenant IS [NOT] NULL read every request
53
+ # in the window instead: 1 s on an untagged app, 24 s on a tagged one.
46
54
  def self.overview(rows, from, to)
47
55
  requests = Telemetry::Execution.requests.between(from, to)
48
56
  total = requests.count
49
- untagged = requests.where(app_tenant: nil).count
57
+ tagged_total = tagged_tenants(from, to).each_slice(TENANT_SLICE).sum { |slice| requests.where(app_tenant: slice).count }
50
58
  tagged = rows.sum { |r| r[:requests] }
51
59
  top = rows.max_by { |r| r[:requests] }
52
60
  {
53
61
  tenants: rows.size, top_tenant: top && top[:tenant],
54
62
  top_share: tagged.zero? ? 0.0 : (top[:requests] * 100.0 / tagged).round(1),
55
63
  with_errors: rows.count { |r| r[:errors].positive? },
56
- untagged_share: total.zero? ? 0.0 : (untagged * 100.0 / total).round(1)
64
+ untagged_share: total.zero? ? 0.0 : ((total - tagged_total) * 100.0 / total).round(1)
57
65
  }
58
66
  end
59
67
 
68
+ # Every tenant that sent anything in the window, from the tenant-led
69
+ # index (a skip-scan over distinct values, not the window's rows).
70
+ def self.tagged_tenants(from, to)
71
+ Telemetry::Execution.between(from, to).where.not(app_tenant: nil).distinct.pluck(:app_tenant)
72
+ end
73
+
60
74
  # Window totals for one tenant. Durations in milliseconds, like the rest
61
75
  # of the tenant props.
62
76
  def self.summary(tenant, from, to)
@@ -129,9 +143,19 @@ module Railwatch
129
143
  .each { |tenant, at| rows[tenant][:last_seen_at] = at }
130
144
  end
131
145
 
132
- def self.absorb_sparklines(rows, from, to, q)
146
+ # Only for the tenants already found, by name: grouped by tenant and a
147
+ # time expression SQLite would not seek the app_tenant index and read
148
+ # every request in the window instead -- 11 s over a week on an app
149
+ # with no tenants at all, to draw nothing.
150
+ def self.absorb_sparklines(rows, from, to)
151
+ tenants = rows.keys.compact
152
+ return if tenants.empty?
153
+
133
154
  width = bucket_width(from, to, SPARKLINE_BUCKETS)
134
- counts = filtered(Telemetry::Execution.requests.between(from, to), q).group(:app_tenant, bucket_sql(from, width)).count
155
+ counts = tenants.each_slice(TENANT_SLICE).flat_map do |slice|
156
+ Telemetry::Execution.requests.between(from, to).where(app_tenant: slice)
157
+ .group(:app_tenant, bucket_sql(from, width)).count.to_a
158
+ end
135
159
  counts.each do |(tenant, index), count|
136
160
  rows[tenant][:sparkline][[ index.to_i, SPARKLINE_BUCKETS - 1 ].min] += count
137
161
  end
@@ -0,0 +1,15 @@
1
+ # frozen_string_literal: true
2
+
3
+ # The People page groups the window's signed-in requests by user_ref.
4
+ # Through (user_ref, occurred_at) SQLite found those rows but fetched each
5
+ # one from the table for kind and status: 4 s over 30 days on the
6
+ # platform's own tenant, where 18,000 of 2.8 million executions carry a
7
+ # user. This index holds every column the query reads, and only for rows
8
+ # that have a user, so it answers in 18 ms from under a megabyte. It took
9
+ # 10 s to build on a copy of that 30 GB file.
10
+ class AddPeopleIndex < ActiveRecord::Migration[8.1]
11
+ def change
12
+ add_index :executions, [ :user_ref, :kind, :occurred_at, :status ], name: "idx_executions_people",
13
+ where: "user_ref IS NOT NULL"
14
+ end
15
+ end
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Railwatch
4
- VERSION = "0.8.0"
4
+ VERSION = "0.8.2"
5
5
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: railwatch
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.8.0
4
+ version: 0.8.2
5
5
  platform: ruby
6
6
  authors:
7
7
  - Cole Robertson
@@ -404,6 +404,7 @@ files:
404
404
  - db/railwatch_telemetry_migrate/20260917010000_add_batch_ledger_to_ingest_batches.rb
405
405
  - db/railwatch_telemetry_migrate/20260919000100_create_export_queue.rb
406
406
  - db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb
407
+ - db/railwatch_telemetry_migrate/20260923000000_add_people_index.rb
407
408
  - docs/ai-and-mcp.md
408
409
  - docs/configuration.md
409
410
  - docs/embedded.md