railwatch 0.8.4 → 0.8.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +33 -0
- data/app/models/railwatch/telemetry/tenant.rb +13 -3
- data/db/railwatch_telemetry_migrate/20260919000100_create_export_queue.rb +125 -87
- data/db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb +4 -4
- data/db/railwatch_telemetry_migrate/20260923000000_add_people_index.rb +1 -1
- data/db/railwatch_telemetry_migrate/20260925000000_add_tenant_summary_index.rb +17 -0
- data/lib/railwatch/version.rb +1 -1
- metadata +2 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: c483283faabf8cf2552cedbd8568e6d06f815c8e5bdf4f2cdffe1909a82fa80b
|
|
4
|
+
data.tar.gz: 2f56eb2aed77f09e55d869117c3f9bf1d6140f00fa93ebd2a5c2c89e4da62b7b
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: a80ac73b1c0e36d6fb681da86848ac37f3bde8d6b867fbd0a5134087da9c134344595fd1407f49f72f00d8573d2304b9944b2e667ae44fa63cd9c2337527da5a
|
|
7
|
+
data.tar.gz: a3e7028f181c255b69d0831f9e9a56c163b48bdacccca8464b3d05f6233976a2d412d06d11f3ad5320430186e2c067830e96ee048f415281933da5a6bc41a22f
|
data/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,39 @@
|
|
|
6
6
|
filled in by the release commit, which is also the only commit that
|
|
7
7
|
touches lib/railwatch/version.rb and Gemfile.lock. See CONTRIBUTING.md. -->
|
|
8
8
|
|
|
9
|
+
## 0.8.6 (2026-10-06)
|
|
10
|
+
|
|
11
|
+
- An install upgrading from 0.5.0 or older can migrate again. 0.5.1
|
|
12
|
+
renumbered the export queue migration (20260919000000 to
|
|
13
|
+
20260919000100), so a telemetry database that had already run it under
|
|
14
|
+
the old number saw it as pending, and its plain `create_table` raised
|
|
15
|
+
"table export_destinations already exists" -- in `db:prepare`, which a
|
|
16
|
+
deploy runs before the app boots. Every statement in it is now
|
|
17
|
+
`if_not_exists`, so it builds the queue on a new database, completes it
|
|
18
|
+
on one that ran the old number (keeping anything already queued), and
|
|
19
|
+
stays a no-op where the new number has run. Rolling it back on a
|
|
20
|
+
database that came through the old number leaves the queue in place,
|
|
21
|
+
since the old version still owns it; anywhere else it is reversed as
|
|
22
|
+
before.
|
|
23
|
+
- The 0.8.0, 0.8.1 and 0.8.5 index migrations accept an index that is
|
|
24
|
+
already there. Each one reads a whole table, and on a telemetry file
|
|
25
|
+
much larger than memory that is not seconds: about 40 s per table on
|
|
26
|
+
rebulk-system's 14.8 GB file, five builds in all, against a deploy
|
|
27
|
+
that health-checks within 90 s of the container starting -- and runs
|
|
28
|
+
`db:prepare` inside that window. An operator can now build them ahead
|
|
29
|
+
of the deploy, while the old release still serves, with the
|
|
30
|
+
migrations' own `CREATE INDEX` statements; `db:prepare` then records
|
|
31
|
+
them and moves on.
|
|
32
|
+
|
|
33
|
+
## 0.8.5 (2026-09-25)
|
|
34
|
+
|
|
35
|
+
- A new migration adds a partial covering index for the per-tenant sums
|
|
36
|
+
on the Tenants page and in MCP's `list_tenants`. Over 30 days of
|
|
37
|
+
rebulk-system's 358,000 tagged requests they took 8.1 s, reading every
|
|
38
|
+
row from the table; they now take 131 ms. It holds only tagged rows,
|
|
39
|
+
so an app without tenants gets an empty index. It took 1.6 s to build
|
|
40
|
+
on a 16 GB file.
|
|
41
|
+
|
|
9
42
|
## 0.8.4 (2026-09-23)
|
|
10
43
|
|
|
11
44
|
- A `class:` filter for a job class with many attempts in the window
|
|
@@ -120,8 +120,18 @@ module Railwatch
|
|
|
120
120
|
|
|
121
121
|
# -- Aggregation steps -----------------------------------------------------
|
|
122
122
|
|
|
123
|
+
# The per-tenant sums read through idx_executions_tenant_summary, which
|
|
124
|
+
# holds every column they need for tagged rows only. SQLite prefers the
|
|
125
|
+
# narrower (app_tenant, occurred_at) index on its own estimate and then
|
|
126
|
+
# fetches every row from the table (8 s against 131 ms over 30 days).
|
|
127
|
+
def self.summary_scope(kind)
|
|
128
|
+
scope = Telemetry::Execution
|
|
129
|
+
scope = scope.from("#{scope.quoted_table_name} INDEXED BY idx_executions_tenant_summary") if TelemetryRecord.sqlite?
|
|
130
|
+
scope.where(kind: kind).where.not(app_tenant: nil)
|
|
131
|
+
end
|
|
132
|
+
|
|
123
133
|
def self.absorb_requests(rows, from, to, q)
|
|
124
|
-
filtered(
|
|
134
|
+
filtered(summary_scope("request").between(from, to), q).group(:app_tenant)
|
|
125
135
|
.pluck(:app_tenant, COUNT_SQL, ERRORS_SQL, Arel.sql("AVG(duration)"), Arel.sql("MAX(duration)"), USERS_SQL)
|
|
126
136
|
.each do |tenant, count, errors, avg, max, users|
|
|
127
137
|
rows[tenant].merge!(requests: count, errors: errors, avg: ms(avg), max: ms(max), users: users)
|
|
@@ -129,7 +139,7 @@ module Railwatch
|
|
|
129
139
|
end
|
|
130
140
|
|
|
131
141
|
def self.absorb_jobs(rows, from, to, q)
|
|
132
|
-
filtered(
|
|
142
|
+
filtered(summary_scope("job_attempt").between(from, to), q).group(:app_tenant)
|
|
133
143
|
.pluck(:app_tenant, COUNT_SQL, FAILED_SQL)
|
|
134
144
|
.each { |tenant, count, failed| rows[tenant].merge!(jobs: count, failed_jobs: failed) }
|
|
135
145
|
end
|
|
@@ -139,7 +149,7 @@ module Railwatch
|
|
|
139
149
|
end
|
|
140
150
|
|
|
141
151
|
def self.absorb_last_seen(rows, from, to, q)
|
|
142
|
-
filtered(Telemetry::Execution.between(from, to), q).group(:app_tenant).maximum(:occurred_at)
|
|
152
|
+
filtered(summary_scope(Telemetry::Execution::KINDS).between(from, to), q).group(:app_tenant).maximum(:occurred_at)
|
|
143
153
|
.each { |tenant, at| rows[tenant][:last_seen_at] = at }
|
|
144
154
|
end
|
|
145
155
|
|
|
@@ -13,96 +13,134 @@
|
|
|
13
13
|
# is stored.
|
|
14
14
|
#
|
|
15
15
|
# Nothing here is written unless export is explicitly enabled.
|
|
16
|
+
#
|
|
17
|
+
# Every statement is if_not_exists, because this migration has run before
|
|
18
|
+
# under another number. Through 0.5.0 it was 20260919000000; 0.5.1 moved it
|
|
19
|
+
# here so it could not collide with a host's own telemetry migration at that
|
|
20
|
+
# number (Railwatch Cloud has one). An install upgrading from 0.5.0 or older
|
|
21
|
+
# therefore already holds every table, index and column below, recorded
|
|
22
|
+
# under the old version, and sees this one as pending: a plain create_table
|
|
23
|
+
# raised "table export_destinations already exists" in db:prepare, which a
|
|
24
|
+
# deploy runs before the app boots. The old run is detected by what it
|
|
25
|
+
# built, not by its version, because on a host like Railwatch Cloud the old
|
|
26
|
+
# number belongs to a different migration.
|
|
27
|
+
#
|
|
28
|
+
# Rolling back has the mirror problem: on a database that came through the
|
|
29
|
+
# old number, these tables and columns -- and anything queued in them -- are
|
|
30
|
+
# the old version's, which stays recorded as applied. So down leaves them
|
|
31
|
+
# alone there, and reverses the migration everywhere else.
|
|
16
32
|
class CreateExportQueue < ActiveRecord::Migration[8.1]
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
#
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
t.string :state, limit: 16, null: false, default: "ready"
|
|
30
|
-
t.datetime :retry_at, precision: 6
|
|
31
|
-
t.string :reason, limit: 64
|
|
32
|
-
|
|
33
|
-
t.string :lease_owner, limit: 36
|
|
34
|
-
t.bigint :lease_generation, null: false, default: 0
|
|
35
|
-
t.datetime :lease_expires_at, precision: 6
|
|
36
|
-
|
|
37
|
-
# Live totals, maintained in the same transaction as the rows they
|
|
38
|
-
# describe, so admission can be decided without counting the table.
|
|
39
|
-
t.bigint :queued_bytes, null: false, default: 0
|
|
40
|
-
t.bigint :queued_deliveries, null: false, default: 0
|
|
41
|
-
# Lifetime accounting: acked, rejected, expired, discarded, shed.
|
|
42
|
-
t.json :counters, null: false, default: {}
|
|
43
|
-
|
|
44
|
-
t.timestamps
|
|
33
|
+
OLD_VERSION = 20260919000000
|
|
34
|
+
|
|
35
|
+
def up
|
|
36
|
+
build_queue
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def down
|
|
40
|
+
if built_under_old_number?
|
|
41
|
+
say("left in place: #{OLD_VERSION} built this queue under its pre-0.5.1 number and is still recorded as applied")
|
|
42
|
+
else
|
|
43
|
+
revert { build_queue }
|
|
45
44
|
end
|
|
45
|
+
end
|
|
46
46
|
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
# delivery cannot be made young again by relabelling it.
|
|
54
|
-
t.string :delivery_id, limit: 36, null: false
|
|
55
|
-
# What this delivery is a delivery OF. One per source batch today; a
|
|
56
|
-
# future policy that selects across batches supplies its own key.
|
|
57
|
-
t.string :selection_key, limit: 160, null: false
|
|
58
|
-
|
|
59
|
-
t.binary :body
|
|
60
|
-
t.string :body_sha256, limit: 64, null: false
|
|
61
|
-
t.string :metadata_sha256, limit: 64, null: false
|
|
62
|
-
t.bigint :body_bytes, null: false
|
|
63
|
-
t.bigint :ndjson_bytes, null: false
|
|
64
|
-
t.integer :record_count, null: false
|
|
65
|
-
# Everything about the delivery that is not its body: version, drop
|
|
66
|
-
# counts, backpressure. Digested into metadata_sha256 so the same id
|
|
67
|
-
# arriving with different counts is a conflict, not an update.
|
|
68
|
-
t.json :wire_metadata, null: false, default: {}
|
|
69
|
-
|
|
70
|
-
t.string :state, limit: 8, null: false, default: "pending"
|
|
71
|
-
# Set only when done: acked, rejected, expired, discarded.
|
|
72
|
-
t.string :disposition, limit: 16
|
|
73
|
-
t.datetime :enqueued_at, null: false, precision: 6
|
|
74
|
-
t.datetime :expires_at, null: false, precision: 6
|
|
75
|
-
t.datetime :next_attempt_at, null: false, precision: 6
|
|
76
|
-
t.bigint :attempts, null: false, default: 0
|
|
77
|
-
|
|
78
|
-
t.string :claim_token, limit: 36
|
|
79
|
-
t.bigint :claim_generation
|
|
80
|
-
t.datetime :claim_expires_at, precision: 6
|
|
81
|
-
|
|
82
|
-
t.integer :last_status
|
|
83
|
-
t.string :last_reason, limit: 64
|
|
84
|
-
t.json :ack
|
|
85
|
-
t.datetime :finished_at, precision: 6
|
|
86
|
-
|
|
87
|
-
t.timestamps
|
|
47
|
+
private
|
|
48
|
+
# The old version is recorded, and no migration file this database knows
|
|
49
|
+
# of claims it: only the gem's own pre-0.5.1 CreateExportQueue did.
|
|
50
|
+
def built_under_old_number?
|
|
51
|
+
context = connection.pool.migration_context
|
|
52
|
+
context.get_all_versions.include?(OLD_VERSION) && context.migrations.none? { |m| m.version == OLD_VERSION }
|
|
88
53
|
end
|
|
89
54
|
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
55
|
+
def build_queue
|
|
56
|
+
# One row per destination this database has ever been pointed at.
|
|
57
|
+
create_table :export_destinations, if_not_exists: true do |t|
|
|
58
|
+
t.string :url, null: false
|
|
59
|
+
t.string :url_sha256, limit: 64, null: false
|
|
60
|
+
# Permanent for this database's lineage: it is how the receiver tells
|
|
61
|
+
# our deliveries from another installation's.
|
|
62
|
+
t.string :producer_id, limit: 36, null: false
|
|
63
|
+
# A fingerprint, never the token. If the token changes, queued bytes
|
|
64
|
+
# must not follow it to whatever tenant the new one belongs to.
|
|
65
|
+
t.string :credential_sha256, limit: 64, null: false
|
|
66
|
+
|
|
67
|
+
t.string :state, limit: 16, null: false, default: "ready"
|
|
68
|
+
t.datetime :retry_at, precision: 6
|
|
69
|
+
t.string :reason, limit: 64
|
|
70
|
+
|
|
71
|
+
t.string :lease_owner, limit: 36
|
|
72
|
+
t.bigint :lease_generation, null: false, default: 0
|
|
73
|
+
t.datetime :lease_expires_at, precision: 6
|
|
74
|
+
|
|
75
|
+
# Live totals, maintained in the same transaction as the rows they
|
|
76
|
+
# describe, so admission can be decided without counting the table.
|
|
77
|
+
t.bigint :queued_bytes, null: false, default: 0
|
|
78
|
+
t.bigint :queued_deliveries, null: false, default: 0
|
|
79
|
+
# Lifetime accounting: acked, rejected, expired, discarded, shed.
|
|
80
|
+
t.json :counters, null: false, default: {}
|
|
81
|
+
|
|
82
|
+
t.timestamps
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
add_index :export_destinations, :url_sha256, unique: true, if_not_exists: true
|
|
86
|
+
add_index :export_destinations, :producer_id, unique: true, if_not_exists: true
|
|
87
|
+
|
|
88
|
+
create_table :export_deliveries, if_not_exists: true do |t|
|
|
89
|
+
t.references :export_destination, null: false, foreign_key: true
|
|
90
|
+
# A UUIDv7: its embedded time is how the receiver ages it out, so a
|
|
91
|
+
# delivery cannot be made young again by relabelling it.
|
|
92
|
+
t.string :delivery_id, limit: 36, null: false
|
|
93
|
+
# What this delivery is a delivery OF. One per source batch today; a
|
|
94
|
+
# future policy that selects across batches supplies its own key.
|
|
95
|
+
t.string :selection_key, limit: 160, null: false
|
|
96
|
+
|
|
97
|
+
t.binary :body
|
|
98
|
+
t.string :body_sha256, limit: 64, null: false
|
|
99
|
+
t.string :metadata_sha256, limit: 64, null: false
|
|
100
|
+
t.bigint :body_bytes, null: false
|
|
101
|
+
t.bigint :ndjson_bytes, null: false
|
|
102
|
+
t.integer :record_count, null: false
|
|
103
|
+
# Everything about the delivery that is not its body: version, drop
|
|
104
|
+
# counts, backpressure. Digested into metadata_sha256 so the same id
|
|
105
|
+
# arriving with different counts is a conflict, not an update.
|
|
106
|
+
t.json :wire_metadata, null: false, default: {}
|
|
107
|
+
|
|
108
|
+
t.string :state, limit: 8, null: false, default: "pending"
|
|
109
|
+
# Set only when done: acked, rejected, expired, discarded.
|
|
110
|
+
t.string :disposition, limit: 16
|
|
111
|
+
t.datetime :enqueued_at, null: false, precision: 6
|
|
112
|
+
t.datetime :expires_at, null: false, precision: 6
|
|
113
|
+
t.datetime :next_attempt_at, null: false, precision: 6
|
|
114
|
+
t.bigint :attempts, null: false, default: 0
|
|
115
|
+
|
|
116
|
+
t.string :claim_token, limit: 36
|
|
117
|
+
t.bigint :claim_generation
|
|
118
|
+
t.datetime :claim_expires_at, precision: 6
|
|
119
|
+
|
|
120
|
+
t.integer :last_status
|
|
121
|
+
t.string :last_reason, limit: 64
|
|
122
|
+
t.json :ack
|
|
123
|
+
t.datetime :finished_at, precision: 6
|
|
124
|
+
|
|
125
|
+
t.timestamps
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
add_index :export_deliveries, %i[export_destination_id delivery_id], unique: true,
|
|
129
|
+
name: "index_export_deliveries_on_destination_and_delivery", if_not_exists: true
|
|
130
|
+
# One delivery per selection: a batch replayed into the same transaction
|
|
131
|
+
# cannot enqueue itself twice.
|
|
132
|
+
add_index :export_deliveries, %i[export_destination_id selection_key], unique: true,
|
|
133
|
+
name: "index_export_deliveries_on_destination_and_selection", if_not_exists: true
|
|
134
|
+
add_index :export_deliveries, %i[export_destination_id id], where: "state <> 'done'",
|
|
135
|
+
name: "index_export_deliveries_live", if_not_exists: true
|
|
136
|
+
add_index :export_deliveries, :expires_at, where: "body IS NOT NULL",
|
|
137
|
+
name: "index_export_deliveries_expiring", if_not_exists: true
|
|
138
|
+
add_index :export_deliveries, :finished_at, where: "state = 'done'",
|
|
139
|
+
name: "index_export_deliveries_finished", if_not_exists: true
|
|
140
|
+
|
|
141
|
+
# Why a batch did or did not enqueue anything. Null on every row an
|
|
142
|
+
# install without export ever writes.
|
|
143
|
+
add_column :ingest_batches, :export_disposition, :string, limit: 24, if_not_exists: true
|
|
144
|
+
add_column :ingest_batches, :export_record_count, :bigint, if_not_exists: true
|
|
145
|
+
end
|
|
108
146
|
end
|
data/db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb
CHANGED
|
@@ -25,10 +25,10 @@
|
|
|
25
25
|
# Execution.named_like seek them instead of reading the window.
|
|
26
26
|
class AddCoveringIndexesForDashboardAggregates < ActiveRecord::Migration[8.1]
|
|
27
27
|
def change
|
|
28
|
-
add_index :executions, [ :kind, :occurred_at, :queue, :outcome, :queue_latency ], name: "idx_executions_queue_stats"
|
|
28
|
+
add_index :executions, [ :kind, :occurred_at, :queue, :outcome, :queue_latency ], name: "idx_executions_queue_stats", if_not_exists: true
|
|
29
29
|
add_index :health_samples, [ :sampled_at, :threads_max, :threads_busy, :backlog, :queue_depth, :queue_latency ],
|
|
30
|
-
name: "idx_health_samples_series"
|
|
31
|
-
add_index :broadcasts, :occurred_at
|
|
32
|
-
add_index :executions, [ :kind, :occurred_at ], name: "idx_executions_with_preview", where: "exception_preview IS NOT NULL"
|
|
30
|
+
name: "idx_health_samples_series", if_not_exists: true
|
|
31
|
+
add_index :broadcasts, :occurred_at, if_not_exists: true
|
|
32
|
+
add_index :executions, [ :kind, :occurred_at ], name: "idx_executions_with_preview", where: "exception_preview IS NOT NULL", if_not_exists: true
|
|
33
33
|
end
|
|
34
34
|
end
|
|
@@ -10,6 +10,6 @@
|
|
|
10
10
|
class AddPeopleIndex < ActiveRecord::Migration[8.1]
|
|
11
11
|
def change
|
|
12
12
|
add_index :executions, [ :user_ref, :kind, :occurred_at, :status ], name: "idx_executions_people",
|
|
13
|
-
where: "user_ref IS NOT NULL"
|
|
13
|
+
where: "user_ref IS NOT NULL", if_not_exists: true
|
|
14
14
|
end
|
|
15
15
|
end
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# The Tenants page and MCP's list_tenants sum every tenant-tagged request
|
|
4
|
+
# and job in the window per tenant. Through (app_tenant, occurred_at)
|
|
5
|
+
# SQLite finds the rows but fetches each from the table for kind, status,
|
|
6
|
+
# outcome, duration and user_ref: 8.1 s over 30 days on rebulk-system
|
|
7
|
+
# (358,000 tagged requests), and 24 s once in production, which held the
|
|
8
|
+
# web process -- and every ingest request queued behind it -- the whole
|
|
9
|
+
# time. This index carries every column those aggregates read, only for
|
|
10
|
+
# tagged rows: 131 ms for the same query, 41 MB, 1.6 s to build on a copy
|
|
11
|
+
# of that 16 GB file. An untagged app gets an empty index.
|
|
12
|
+
class AddTenantSummaryIndex < ActiveRecord::Migration[8.1]
|
|
13
|
+
def change
|
|
14
|
+
add_index :executions, [ :kind, :app_tenant, :occurred_at, :status, :outcome, :duration, :user_ref ],
|
|
15
|
+
name: "idx_executions_tenant_summary", where: "app_tenant IS NOT NULL", if_not_exists: true
|
|
16
|
+
end
|
|
17
|
+
end
|
data/lib/railwatch/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: railwatch
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.8.
|
|
4
|
+
version: 0.8.6
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Cole Robertson
|
|
@@ -405,6 +405,7 @@ files:
|
|
|
405
405
|
- db/railwatch_telemetry_migrate/20260919000100_create_export_queue.rb
|
|
406
406
|
- db/railwatch_telemetry_migrate/20260922000000_add_covering_indexes_for_dashboard_aggregates.rb
|
|
407
407
|
- db/railwatch_telemetry_migrate/20260923000000_add_people_index.rb
|
|
408
|
+
- db/railwatch_telemetry_migrate/20260925000000_add_tenant_summary_index.rb
|
|
408
409
|
- docs/ai-and-mcp.md
|
|
409
410
|
- docs/configuration.md
|
|
410
411
|
- docs/embedded.md
|