railwatch 0.3.7 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +58 -0
- data/app/controllers/railwatch/commands_controller.rb +1 -1
- data/app/controllers/railwatch/environment_scoped.rb +18 -36
- data/app/controllers/railwatch/exceptions_controller.rb +8 -9
- data/app/controllers/railwatch/processes_controller.rb +1 -1
- data/app/controllers/railwatch/releases_controller.rb +2 -2
- data/app/controllers/railwatch/tenants_controller.rb +2 -2
- data/app/controllers/railwatch/thresholds_controller.rb +8 -1
- data/{lib/generators/railwatch/install/templates → app/frontend/lib}/railwatch.ts +162 -42
- data/app/jobs/railwatch/detect_performance_issues_job.rb +12 -2
- data/app/models/railwatch/anomaly_rule.rb +20 -2
- data/app/models/railwatch/ingest/batch.rb +1 -0
- data/app/models/railwatch/ingest/mapper.rb +12 -0
- data/app/models/railwatch/telemetry/aggregations.rb +154 -19
- data/app/models/railwatch/telemetry/exception.rb +7 -0
- data/app/models/railwatch/telemetry/execution.rb +18 -0
- data/app/models/railwatch/telemetry/log.rb +6 -0
- data/app/models/railwatch/telemetry/query.rb +7 -0
- data/app/models/railwatch/telemetry/release_health.rb +12 -6
- data/app/models/railwatch/telemetry/rollup.rb +17 -2
- data/app/models/railwatch/telemetry/tenant.rb +8 -23
- data/app/models/railwatch/telemetry_record.rb +8 -6
- data/app/models/railwatch/threshold.rb +41 -4
- data/app/models/railwatch/window.rb +66 -0
- data/docs/embedded.md +11 -0
- data/lib/generators/railwatch/install/install_generator.rb +5 -1
- data/lib/railwatch/transport/http.rb +15 -5
- data/lib/railwatch/version.rb +1 -1
- data/lib/railwatch/wire_fixtures.json +856 -0
- data/lib/railwatch.rb +22 -0
- data/lib/tasks/railwatch_tasks.rake +7 -0
- data/public/railwatch/assets/app-layout-DDyQa72H.js +1 -0
- data/public/railwatch/assets/{app-wordmark-1SRllgHv.js → app-wordmark-o9CODKP0.js} +1 -1
- data/public/railwatch/assets/{appearance-DbwahDrc.js → appearance-BwuCXabr.js} +1 -1
- data/public/railwatch/assets/application-B7h1MIhi.css +1 -0
- data/public/railwatch/assets/{arrow-up-DwKQcab4.js → arrow-up-C6PxDiY3.js} +1 -1
- data/public/railwatch/assets/{auth-layout-DXfxXAN_.js → auth-layout-BRt8MGFD.js} +1 -1
- data/public/railwatch/assets/{badge-CApjyWcg.js → badge-CAxXV8za.js} +1 -1
- data/public/railwatch/assets/{braces-kGSgxWnS.js → braces-DgomTCNf.js} +1 -1
- data/public/railwatch/assets/{card-BaJ8kNkp.js → card-cAtqCxWl.js} +1 -1
- data/public/railwatch/assets/{chart-Xnrz1Nqw.js → chart-BBeBkkNa.js} +1 -1
- data/public/railwatch/assets/{chart-hover-XYPNE2CX.js → chart-hover-B1M9jc0y.js} +1 -1
- data/public/railwatch/assets/chart-panel-DUQTz_C8.js +1 -0
- data/public/railwatch/assets/{checkbox-Bf41R8W6.js → checkbox-CmhMHWZO.js} +1 -1
- data/public/railwatch/assets/{code-a-9sh36A.js → code-DESvxyTj.js} +1 -1
- data/public/railwatch/assets/{copy-block-zykZWQym.js → copy-block-BkSU5832.js} +1 -1
- data/public/railwatch/assets/{copy-id-b6CW5ESL.js → copy-id-D03GhN9F.js} +1 -1
- data/public/railwatch/assets/{cursor-load-more-0cHzbTPT.js → cursor-load-more-CRyuMeQb.js} +1 -1
- data/public/railwatch/assets/{data-table-IvA7xvrK.js → data-table-BIlt7Rtm.js} +1 -1
- data/public/railwatch/assets/{edit-DTtRV1fO.js → edit-Bb6MKoe4.js} +1 -1
- data/public/railwatch/assets/{edit-7lmf3k80.js → edit-DJ0D0wHN.js} +1 -1
- data/public/railwatch/assets/{edit-ZKE15r-F.js → edit-O0NSBWxo.js} +1 -1
- data/public/railwatch/assets/{empty-state-BFTnQ2oT.js → empty-state-C38il627.js} +1 -1
- data/public/railwatch/assets/env-layout-REF7OM4q.js +1 -0
- data/public/railwatch/assets/{execution-path-Di5uUHi-.js → execution-path-FYLq1TwC.js} +1 -1
- data/public/railwatch/assets/{filter-bar-Btm__gQi.js → filter-bar-CYog9Alp.js} +1 -1
- data/public/railwatch/assets/{flamegraph-C-mZdqa7.js → flamegraph-DSs69foN.js} +1 -1
- data/public/railwatch/assets/{frames-KZBsjg6-.js → frames-BUi2J5Mk.js} +1 -1
- data/public/railwatch/assets/{google-sign-in-button-DJRoD72g.js → google-sign-in-button-BQiIKFdd.js} +1 -1
- data/public/railwatch/assets/{index-9LL_C3eG.js → index-8-hnAhOD.js} +1 -1
- data/public/railwatch/assets/{index-Bjw4sTXU.js → index-BaR1U9An.js} +1 -1
- data/public/railwatch/assets/{index-BtSlLmkg.js → index-BeOh2t_S.js} +1 -1
- data/public/railwatch/assets/{index-DXUwhSCQ.js → index-BiiyMcA0.js} +1 -1
- data/public/railwatch/assets/{index-P_e_cI0O.js → index-BoUBioBP.js} +1 -1
- data/public/railwatch/assets/{index-kNhZPLVj.js → index-C-PmdhXA.js} +1 -1
- data/public/railwatch/assets/{index-DCAho97K.js → index-C3A_9imx.js} +1 -1
- data/public/railwatch/assets/{index-BsrCnQqg.js → index-C7OtLq_3.js} +1 -1
- data/public/railwatch/assets/{index-Bot7kHMA.js → index-CFFpnzIS.js} +1 -1
- data/public/railwatch/assets/{index-Z1c4nFL0.js → index-CFRLPs4J.js} +1 -1
- data/public/railwatch/assets/{index-BAkZNNEh.js → index-CGs4m_fa.js} +1 -1
- data/public/railwatch/assets/{index-CIUl9POV.js → index-CICUIFHL.js} +1 -1
- data/public/railwatch/assets/{index-wOj5j0KC.js → index-C_upSl_k.js} +1 -1
- data/public/railwatch/assets/{index-Cj3_jD1E.js → index-CiPo4Gob.js} +1 -1
- data/public/railwatch/assets/{index-EGU__DOB.js → index-CpkI015n.js} +1 -1
- data/public/railwatch/assets/{index-DEfu_YvU.js → index-CrZ3vHDL.js} +1 -1
- data/public/railwatch/assets/{index-CAzQkw-h.js → index-CsoN51vW.js} +1 -1
- data/public/railwatch/assets/{index-CeqMDByC.js → index-DDI_Zx5V.js} +1 -1
- data/public/railwatch/assets/index-DSvlZVWG.js +1 -0
- data/public/railwatch/assets/{index-CU6yhEyt.js → index-DW2CBbxU.js} +1 -1
- data/public/railwatch/assets/{index-uHBVO8Lh.js → index-Dh4IRLFI.js} +1 -1
- data/public/railwatch/assets/{index-HCJmFcZu.js → index-DqTFTP8p.js} +1 -1
- data/public/railwatch/assets/index-DrcKVG2f.js +1 -0
- data/public/railwatch/assets/{index-QyHE9WsU.js → index-DtHmuB9Q.js} +1 -1
- data/public/railwatch/assets/{index-CBJZQgHA.js → index-DvjY3dPD.js} +1 -1
- data/public/railwatch/assets/{index-Bo8hrjbs.js → index-FhUaPPab.js} +1 -1
- data/public/railwatch/assets/{index-BWSyQRZI.js → index-JdCVBrw8.js} +1 -1
- data/public/railwatch/assets/{index-DAEzO_vq.js → index-QpTtwFwu.js} +1 -1
- data/public/railwatch/assets/{index-D4968-TC.js → index-ZOGOB8SA.js} +1 -1
- data/public/railwatch/assets/{index-CP6jrrPT.js → index-ZSZg9rtq.js} +1 -1
- data/public/railwatch/assets/{index-DcItJxa5.js → index-r0tSIplE.js} +1 -1
- data/public/railwatch/assets/{index-CYYmZpZu.js → index-sTYvcbkh.js} +1 -1
- data/public/railwatch/assets/{index-B0VXzXYG.js → index-so4lRrRq.js} +1 -1
- data/public/railwatch/assets/{index-B-bSm1jB.js → index-tpz-OGUP.js} +1 -1
- data/public/railwatch/assets/{index-CX8Br7m9.js → index-umIAl-pL.js} +1 -1
- data/public/railwatch/assets/{inertia-Degv_G9N.js → inertia-DLew8ZNx.js} +3 -3
- data/public/railwatch/assets/{input-error-C6SqO_iU.js → input-error-cvM6_Jht.js} +1 -1
- data/public/railwatch/assets/{json-viewer-DcQIxT86.js → json-viewer-D922McGi.js} +1 -1
- data/public/railwatch/assets/{klass-WgQEfWpo.js → klass-CrwICqN8.js} +1 -1
- data/public/railwatch/assets/{label-B_vK19tR.js → label-GWl7I6sf.js} +1 -1
- data/public/railwatch/assets/{layout-C8q7GUlZ.js → layout-0ZAnD3zl.js} +1 -1
- data/public/railwatch/assets/{live-dot-DaXEbkXo.js → live-dot-D1n_BreY.js} +1 -1
- data/public/railwatch/assets/{nav-BnD5XArZ.js → nav-DPxr1NNC.js} +1 -1
- data/public/railwatch/assets/{new-BL9OqXHe.js → new-Cdl6pqST.js} +1 -1
- data/public/railwatch/assets/{new-uxMnBtoY.js → new-D-ZzUK9a.js} +1 -1
- data/public/railwatch/assets/{new-BqP0NhOF.js → new-DEVkYv-z.js} +1 -1
- data/public/railwatch/assets/{new-CWP3wtwt.js → new-DHAHDrN7.js} +1 -1
- data/public/railwatch/assets/{new-DIjwu2jF.js → new-Dz4lZf1L.js} +1 -1
- data/public/railwatch/assets/{new-BhRnQn4w.js → new-GMrRFurX.js} +1 -1
- data/public/railwatch/assets/{onboarding-C_Y3xTsH.js → onboarding-D1vwaHYT.js} +1 -1
- data/public/railwatch/assets/{origin-identity-CNhOnfBa.js → origin-identity-Bk9yHWZ1.js} +1 -1
- data/public/railwatch/assets/{percentile-picker-BujVEhP6.js → percentile-picker-DfSx9yJO.js} +1 -1
- data/public/railwatch/assets/{relative-time-6CoKr1lN.js → relative-time-CjIjb8Lg.js} +1 -1
- data/public/railwatch/assets/{release-health-1Qzc_XhT.js → release-health-4b3tivEf.js} +1 -1
- data/public/railwatch/assets/{route-xqgYjOwJ.js → route-C_5BUtHK.js} +1 -1
- data/public/railwatch/assets/{segmented-CVKq2Xpa.js → segmented-BgbT3wZa.js} +1 -1
- data/public/railwatch/assets/{select-CsmSXsJ4.js → select-DmunxCKE.js} +1 -1
- data/public/railwatch/assets/{separator-CrM10Oot.js → separator-BXzEdZ_8.js} +1 -1
- data/public/railwatch/assets/series-chart-Xf49v9cv.js +1 -0
- data/public/railwatch/assets/{show-zxVxhUel.js → show-B2zLAW83.js} +1 -1
- data/public/railwatch/assets/{show-BM-E2iYc.js → show-BLpWUHWD.js} +1 -1
- data/public/railwatch/assets/{show-CvNV23gH.js → show-CAl7xcex.js} +1 -1
- data/public/railwatch/assets/{show-BlNR3ZOZ.js → show-DI8IhNUH.js} +1 -1
- data/public/railwatch/assets/{show-B0nHy31L.js → show-DSP9Cq_C.js} +2 -2
- data/public/railwatch/assets/{show-rysBbvt-.js → show-DXs4deaC.js} +1 -1
- data/public/railwatch/assets/{show-DUwQlohF.js → show-DYskfl3-.js} +1 -1
- data/public/railwatch/assets/{show-ixK2r2f1.js → show-DcpTFiLi.js} +1 -1
- data/public/railwatch/assets/{show-CQ4Ze-Un.js → show-Dily73Xk.js} +1 -1
- data/public/railwatch/assets/{show-B3izvRdU.js → show-DlRVS18-.js} +1 -1
- data/public/railwatch/assets/{show-8H9WAGav.js → show-Dn-GwFZL.js} +1 -1
- data/public/railwatch/assets/{show-DZwcZZrz.js → show-DnR1Dnjd.js} +1 -1
- data/public/railwatch/assets/{show-CMisHNAI.js → show-SHwZjXb7.js} +1 -1
- data/public/railwatch/assets/{show-B7NhcWgp.js → show-SvLOcPrx.js} +1 -1
- data/public/railwatch/assets/{show-CJMeX75E.js → show-Y74rM0VT.js} +1 -1
- data/public/railwatch/assets/{show-DMZlwzL_.js → show-mU38uGTg.js} +1 -1
- data/public/railwatch/assets/{show-9jDeBNpc.js → show-vQ4bndYD.js} +1 -1
- data/public/railwatch/assets/{sort-header-CKj_7G3U.js → sort-header-Dcq9bzmo.js} +1 -1
- data/public/railwatch/assets/{sparkline-cell-Ckg29yVq.js → sparkline-cell-BON3qQUB.js} +1 -1
- data/public/railwatch/assets/{stat-TxvoBiC8.js → stat-DFEyFxkO.js} +1 -1
- data/public/railwatch/assets/{status-badge-5NCVJHjo.js → status-badge-BaUKP7Yo.js} +1 -1
- data/public/railwatch/assets/{tenant-path-B5ZRK4x3.js → tenant-path-DPZPc985.js} +1 -1
- data/public/railwatch/assets/{text-link-BPllRj0F.js → text-link-BO77t9Xk.js} +1 -1
- data/public/railwatch/assets/{textarea-DtD5yZp_.js → textarea-DTqrCiV0.js} +1 -1
- data/public/railwatch/assets/{timeline-DZz5ZnHN.js → timeline-D5rJ0es2.js} +1 -1
- data/public/railwatch/assets/{transition-Ds6Cakuc.js → transition-DMIrZVth.js} +1 -1
- data/public/railwatch/assets/{use-clipboard-BLjjYH8V.js → use-clipboard-ByoUGQqA.js} +1 -1
- data/public/railwatch/assets/{use-live-GKze1mNC.js → use-live-D7xKz2ma.js} +1 -1
- data/public/railwatch/manifest.json +1284 -1284
- metadata +119 -117
- data/public/railwatch/assets/app-layout-B5Fcdwfz.js +0 -1
- data/public/railwatch/assets/application-CwshqwM6.css +0 -1
- data/public/railwatch/assets/chart-panel-BYk68mJJ.js +0 -1
- data/public/railwatch/assets/env-layout-erSDpIXg.js +0 -1
- data/public/railwatch/assets/index-BoTOVM0v.js +0 -1
- data/public/railwatch/assets/index-DBTxQw77.js +0 -1
- data/public/railwatch/assets/series-chart-D33UhdG2.js +0 -1
|
@@ -12,7 +12,8 @@ module Railwatch
|
|
|
12
12
|
MAX_ROLLUP_ROWS_PER_RULE = MAX_GROUPS_PER_RULE * 25
|
|
13
13
|
|
|
14
14
|
TYPE_FOR = { "requests" => "request", "jobs" => "job_attempt", "commands" => "command", "queries" => "query",
|
|
15
|
-
"scheduled_tasks" => "scheduled_task", "outgoing_requests" => "outgoing_request"
|
|
15
|
+
"scheduled_tasks" => "scheduled_task", "outgoing_requests" => "outgoing_request",
|
|
16
|
+
"llm_calls" => "llm_call", "llm_tools" => "llm_tool" }.freeze
|
|
16
17
|
|
|
17
18
|
def perform(environment)
|
|
18
19
|
now = Time.current
|
|
@@ -51,19 +52,28 @@ module Railwatch
|
|
|
51
52
|
.transform_values { |group_rows| [ group_rows.first.name, Telemetry::Rollup.summarize(group_rows) ] }
|
|
52
53
|
end
|
|
53
54
|
|
|
55
|
+
NANOS_PER_DOLLAR = 1_000_000_000.0
|
|
56
|
+
|
|
54
57
|
def value_for(metric, s)
|
|
58
|
+
extra = s[:extra] || {}
|
|
55
59
|
case metric
|
|
56
60
|
when "p95" then s[:p95] / 1000.0
|
|
57
61
|
when "max" then s[:max] / 1000.0
|
|
58
62
|
when "avg" then s[:avg] / 1000.0
|
|
59
63
|
when "error_rate", "failure_rate" then s[:count].zero? ? nil : (s[:errors] * 100.0 / s[:count])
|
|
64
|
+
# Money and tokens are sums over the window, not percentiles of it.
|
|
65
|
+
when "spend" then extra["cost_nanos"].to_i / NANOS_PER_DOLLAR
|
|
66
|
+
when "tokens" then extra["input_tokens"].to_i + extra["output_tokens"].to_i
|
|
67
|
+
# A cut-off answer is a successful call, so it is invisible to
|
|
68
|
+
# error_rate. This is the only way to be told about it.
|
|
69
|
+
when "truncation_rate" then s[:count].zero? ? nil : (extra["truncated"].to_i * 100.0 / s[:count])
|
|
60
70
|
end
|
|
61
71
|
end
|
|
62
72
|
|
|
63
73
|
def open_issue(environment, threshold, group_hash, name, value, from, now)
|
|
64
74
|
issue, outcome = Issue.record_occurrence!(
|
|
65
75
|
environment: environment, group_hash: "perf:#{threshold.id}:#{group_hash}", kind: "performance",
|
|
66
|
-
title: "#{name} exceeded #{threshold.metric} #{threshold.
|
|
76
|
+
title: "#{name} exceeded #{threshold.metric} #{threshold.format_value(threshold.limit)} (#{threshold.format_value(value.round(2))})",
|
|
67
77
|
culprit: name, occurred_at: now, deploy: nil, user_ref: nil,
|
|
68
78
|
sample: { threshold_id: threshold.id, value: value.round(2), metric: threshold.metric, limit: threshold.limit })
|
|
69
79
|
carry_detection(issue, outcome) do
|
|
@@ -10,13 +10,22 @@ module Railwatch
|
|
|
10
10
|
MAX_BASELINE_DAYS = 30
|
|
11
11
|
MAX_DEVIATION = 10
|
|
12
12
|
MAX_WINDOW_MINUTES = 1_440
|
|
13
|
-
TARGET_KINDS = %w[requests jobs queries outgoing_requests scheduled_tasks].freeze
|
|
14
|
-
|
|
13
|
+
TARGET_KINDS = %w[requests jobs queries outgoing_requests scheduled_tasks llm_calls].freeze
|
|
14
|
+
# spend included because "we are spending well above baseline for this
|
|
15
|
+
# hour of the week" is the LLM alert that needs no threshold picked.
|
|
16
|
+
METRICS = %w[p95 avg count error_rate spend].freeze
|
|
17
|
+
SHARED_METRICS = %w[p95 avg count error_rate].freeze
|
|
18
|
+
LLM_METRICS = %w[spend].freeze
|
|
19
|
+
|
|
20
|
+
def self.metrics_for(target_kind)
|
|
21
|
+
target_kind == "llm_calls" ? SHARED_METRICS + LLM_METRICS : SHARED_METRICS
|
|
22
|
+
end
|
|
15
23
|
|
|
16
24
|
def environment = Environment.current
|
|
17
25
|
|
|
18
26
|
validates :target_kind, inclusion: { in: TARGET_KINDS }
|
|
19
27
|
validates :metric, inclusion: { in: METRICS }
|
|
28
|
+
validate :metric_applies_to_target_kind
|
|
20
29
|
validates :target, presence: true, length: { maximum: 256 }
|
|
21
30
|
validates :deviation, numericality: { greater_than: 0, less_than_or_equal_to: MAX_DEVIATION }
|
|
22
31
|
validates :window_minutes, numericality: { only_integer: true, greater_than: 0,
|
|
@@ -29,5 +38,14 @@ module Railwatch
|
|
|
29
38
|
def description
|
|
30
39
|
"#{target_kind} #{target == '*' ? 'all' : target}: #{metric} > #{deviation}σ above #{baseline_days}-day baseline"
|
|
31
40
|
end
|
|
41
|
+
|
|
42
|
+
private
|
|
43
|
+
|
|
44
|
+
def metric_applies_to_target_kind
|
|
45
|
+
return if metric.blank? || target_kind.blank?
|
|
46
|
+
return if AnomalyRule.metrics_for(target_kind).include?(metric)
|
|
47
|
+
|
|
48
|
+
errors.add(:metric, "is not available for #{target_kind}")
|
|
49
|
+
end
|
|
32
50
|
end
|
|
33
51
|
end
|
|
@@ -182,6 +182,7 @@ module Railwatch
|
|
|
182
182
|
# batch does not need.
|
|
183
183
|
raise TypeError, "record must be an object" unless rec.is_a?(Hash)
|
|
184
184
|
raise TypeError, "t must be a string" unless rec["t"].is_a?(String)
|
|
185
|
+
Ingest::Mapper.validate_version!(rec)
|
|
185
186
|
else
|
|
186
187
|
Ingest::Mapper.validate_record!(rec)
|
|
187
188
|
end
|
|
@@ -111,6 +111,7 @@ module Railwatch
|
|
|
111
111
|
def validate_record!(rec, identifiers: true)
|
|
112
112
|
raise TypeError, "record must be an object" unless rec.is_a?(Hash)
|
|
113
113
|
raise TypeError, "t must be a string" unless rec["t"].is_a?(String)
|
|
114
|
+
validate_version!(rec)
|
|
114
115
|
|
|
115
116
|
if identifiers
|
|
116
117
|
validate_identifier!(rec, "_group", GROUP_HASH, "32 hexadecimal characters")
|
|
@@ -129,6 +130,17 @@ module Railwatch
|
|
|
129
130
|
validate_locals!(rec["locals"])
|
|
130
131
|
end
|
|
131
132
|
|
|
133
|
+
# A record shape is named by its type and version (Railwatch::Record::VERSIONS).
|
|
134
|
+
# This mapper only knows the versions the gem it ships with emits; a
|
|
135
|
+
# record from a newer or older gem is refused by name rather than read
|
|
136
|
+
# through columns that may no longer mean the same thing.
|
|
137
|
+
def validate_version!(rec)
|
|
138
|
+
expected = Record::VERSIONS[rec["t"].to_sym]
|
|
139
|
+
return if expected.nil? # unknown type: row_for answers nil and the batch rejects it as such
|
|
140
|
+
|
|
141
|
+
raise TypeError, "#{rec["t"]} v#{rec["v"].inspect} is not v#{expected}" unless rec["v"].is_a?(Integer) && rec["v"] == expected
|
|
142
|
+
end
|
|
143
|
+
|
|
132
144
|
def validate_identifier!(rec, field, pattern, description)
|
|
133
145
|
value = rec[field]
|
|
134
146
|
return if value.nil?
|
|
@@ -9,18 +9,154 @@ module Railwatch
|
|
|
9
9
|
class Aggregations
|
|
10
10
|
GROUPED_SORT_FIELDS = %i[count p50 p95 p99 errors avg max].freeze
|
|
11
11
|
|
|
12
|
-
#
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
12
|
+
# Bucket widths a chart can be drawn at. Anything under an hour is read
|
|
13
|
+
# from the raw tables (rollups are hourly); an hour and up groups rollups.
|
|
14
|
+
STEPS = { "1m" => 1.minute, "5m" => 5.minutes, "15m" => 15.minutes, "1h" => 1.hour, "6h" => 6.hours, "1d" => 1.day }.freeze
|
|
15
|
+
# A step is offered for a window when it draws at least two points and at
|
|
16
|
+
# most this many: more bars than pixels is mush, and a sub-hour step over
|
|
17
|
+
# a long window is also a long raw scan.
|
|
18
|
+
MAX_POINTS = 400
|
|
19
|
+
|
|
20
|
+
# The bucket a window is drawn at unless the page asks for another one.
|
|
21
|
+
# Sub-hour steps scan raw rows, so they are the default only where that
|
|
22
|
+
# scan is a few thousand rows; a day and beyond stays on rollups.
|
|
23
|
+
def self.default_step(from, to)
|
|
24
|
+
span = to - from
|
|
25
|
+
if span <= 2.hours then "1m"
|
|
26
|
+
elsif span <= 12.hours then "5m"
|
|
27
|
+
elsif span <= 7.days then "1h"
|
|
28
|
+
else "6h"
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# STEPS keys that make sense for the window, finest first.
|
|
33
|
+
def self.steps_for(from, to)
|
|
34
|
+
span = to - from
|
|
35
|
+
STEPS.select { |_key, step| (span / step).between?(2, MAX_POINTS) }.keys
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# Time-bucketed series for charts, one point per `step` from `from` to `to`
|
|
39
|
+
# with empty buckets filled in: [{t, count, errors, client_errors, avg,
|
|
40
|
+
# p50, p95, p99}]. Durations in milliseconds; an empty bucket carries nil
|
|
41
|
+
# for them so a latency line breaks instead of dropping to zero.
|
|
42
|
+
def self.series(environment, record_type, from:, to:, group_hash: nil, step: nil, tenant: nil)
|
|
43
|
+
environment.with_telemetry { points(record_type, from: from, to: to, group_hash: group_hash, step: step, tenant: tenant) }
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# `series` for a caller already inside environment.with_telemetry. A
|
|
47
|
+
# tenant narrows to one app_tenant, which rollups do not carry, so a
|
|
48
|
+
# tenant series reads raw rows at every step.
|
|
49
|
+
def self.points(record_type, from:, to:, group_hash: nil, step: nil, tenant: nil)
|
|
50
|
+
step = STEPS.fetch(step || default_step(from, to))
|
|
51
|
+
raw = step < 1.hour || tenant
|
|
52
|
+
points = raw ? raw_series(record_type, from, to, group_hash, step, tenant) : rollup_series(record_type, from, to, group_hash, step)
|
|
53
|
+
fill(points, from, to, step)
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Sums hourly rollups into `step`-wide buckets. Percentiles are the
|
|
57
|
+
# worst hour's, the same reading series always gave across groups.
|
|
58
|
+
def self.rollup_series(record_type, from, to, group_hash, step)
|
|
59
|
+
bucket = bucket_sql("bucket", step)
|
|
60
|
+
scope = Telemetry::Rollup.for_type(record_type).between(from, to)
|
|
61
|
+
scope = scope.where(group_hash: group_hash) if group_hash
|
|
62
|
+
scope.group(bucket).order(bucket)
|
|
63
|
+
.pluck(bucket, Arel.sql("SUM(count)"), Arel.sql("SUM(error_count)"), Arel.sql("SUM(client_error_count)"), Arel.sql("SUM(duration_sum)"), Arel.sql("MAX(p50)"), Arel.sql("MAX(p95)"), Arel.sql("MAX(p99)"))
|
|
64
|
+
.map { |b, c, e, ce, ds, p50, p95, p99| point(b, c, e, ce, ds, p50, p95, p99) }
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# What a raw row of each record type is, and when it counts as an error
|
|
68
|
+
# (or a client error), mirroring RollupJob#error?. Both readings of the
|
|
69
|
+
# same rows must agree, which spec/models/telemetry/aggregations_spec.rb
|
|
70
|
+
# checks for every type.
|
|
71
|
+
RAW = {
|
|
72
|
+
"request" => [ -> { Telemetry::Execution.requests }, "status >= 500", "status BETWEEN 400 AND 499" ],
|
|
73
|
+
"job_attempt" => [ -> { Telemetry::Execution.jobs }, "outcome = 'failed'" ],
|
|
74
|
+
"scheduled_task" => [ -> { Telemetry::Execution.scheduled }, "outcome = 'failed'" ],
|
|
75
|
+
"command" => [ -> { Telemetry::Execution.commands }, "status != 0" ],
|
|
76
|
+
"channel_action" => [ -> { Telemetry::Execution.channels }, "outcome = 'failed'" ],
|
|
77
|
+
"query" => [ -> { Telemetry::Query.all } ],
|
|
78
|
+
"outgoing_request" => [ -> { Telemetry::OutgoingRequest.all }, "status_code >= 500 OR status_code IS NULL OR status_code = 0" ],
|
|
79
|
+
"cache_event" => [ -> { Telemetry::CacheEvent.all } ],
|
|
80
|
+
"mail" => [ -> { Telemetry::Mail.all }, "failed" ],
|
|
81
|
+
"visit" => [ -> { Telemetry::Visit.all }, "status = 'error'" ],
|
|
82
|
+
"span" => [ -> { Telemetry::Span.all }, "status = 'failed'" ],
|
|
83
|
+
"notification" => [ -> { Telemetry::Notification.all }, "failed" ],
|
|
84
|
+
"view_render" => [ -> { Telemetry::ViewRender.all } ],
|
|
85
|
+
"transaction" => [ -> { Telemetry::Transaction.all }, "outcome = 'rollback'" ],
|
|
86
|
+
"llm_call" => [ -> { Telemetry::LlmCall.models }, "status = 'failed'" ],
|
|
87
|
+
"llm_tool" => [ -> { Telemetry::LlmCall.tools }, "status = 'failed'" ]
|
|
88
|
+
}.freeze
|
|
89
|
+
|
|
90
|
+
# Sub-hour buckets straight from the raw rows, in one statement:
|
|
91
|
+
# counts and sums per bucket, and exact nearest-rank percentiles from a
|
|
92
|
+
# ROW_NUMBER over each bucket's durations. Rollups are hourly, so this
|
|
93
|
+
# is the only reading finer than an hour; it is also the freshest one,
|
|
94
|
+
# since the current hour's rollup is up to a minute behind.
|
|
95
|
+
def self.raw_series(record_type, from, to, group_hash, step, tenant = nil)
|
|
96
|
+
scope, error_sql, client_error_sql = RAW.fetch(record_type)
|
|
97
|
+
rows = scope.call.where(occurred_at: from..to)
|
|
98
|
+
rows = rows.where(group_hash: group_hash) if group_hash
|
|
99
|
+
rows = rows.where(app_tenant: tenant) if tenant
|
|
100
|
+
model = rows.model
|
|
101
|
+
inner = rows.select(bucket_sql("occurred_at", step).to_s + " AS b", "duration", flag_sql(error_sql) + " AS err", flag_sql(client_error_sql) + " AS cerr")
|
|
102
|
+
ranked = model.unscoped.from(inner, "r").select("b", "duration", "err", "cerr",
|
|
103
|
+
"ROW_NUMBER() OVER (PARTITION BY b ORDER BY duration) AS rn", "COUNT(*) OVER (PARTITION BY b) AS n")
|
|
104
|
+
model.unscoped.from(ranked, "w").group("b").order("b")
|
|
105
|
+
.pluck(Arel.sql("b"), Arel.sql("COUNT(*)"), Arel.sql("SUM(err)"), Arel.sql("SUM(cerr)"), Arel.sql("SUM(duration)"),
|
|
106
|
+
rank_sql(50), rank_sql(95), rank_sql(99))
|
|
107
|
+
.map { |b, c, e, ce, ds, p50, p95, p99| point(b, c, e, ce, ds, p50, p95, p99) }
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
# The timestamp columns a series buckets on: raw rows by occurred_at,
|
|
111
|
+
# rollups and release health by their hourly bucket.
|
|
112
|
+
BUCKET_COLUMNS = %w[occurred_at bucket].freeze
|
|
113
|
+
|
|
114
|
+
# Epoch seconds of the start of the `step`-wide bucket holding `column`.
|
|
115
|
+
# `column` must be one of BUCKET_COLUMNS and `step` is a Duration, so the
|
|
116
|
+
# fragment can only ever hold a listed column name and an integer literal.
|
|
117
|
+
def self.bucket_sql(column, step)
|
|
118
|
+
raise ArgumentError, "unknown bucket column #{column.inspect}" unless BUCKET_COLUMNS.include?(column)
|
|
119
|
+
seconds = Integer(step.to_i)
|
|
120
|
+
Arel.sql("(strftime('%s', #{column}) / #{seconds}) * #{seconds}")
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
# One point per bucket from `from` to `to`, keeping the computed ones and
|
|
124
|
+
# zero-filling the rest, so a quiet minute is a gap of the right width.
|
|
125
|
+
# A series with its own point shape passes a block building its empty one.
|
|
126
|
+
def self.fill(points, from, to, step)
|
|
127
|
+
by_bucket = points.index_by { |p| p[:t] }
|
|
128
|
+
first = Time.at((from.to_i / step.to_i) * step.to_i).utc
|
|
129
|
+
last = Time.at((to.to_i / step.to_i) * step.to_i).utc
|
|
130
|
+
(first.to_i..last.to_i).step(step.to_i).map do |t|
|
|
131
|
+
at = Time.at(t).utc
|
|
132
|
+
by_bucket[at] || (block_given? ? yield(at) : empty_point(at))
|
|
21
133
|
end
|
|
22
134
|
end
|
|
23
135
|
|
|
136
|
+
def self.point(bucket, count, errors, client_errors, duration_sum, p50, p95, p99)
|
|
137
|
+
{ t: Time.at(bucket).utc, count: count, errors: errors, client_errors: client_errors,
|
|
138
|
+
avg: count.zero? ? 0 : duration_sum / count / 1000.0, p50: p50 / 1000.0, p95: p95 / 1000.0, p99: p99 / 1000.0 }
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
def self.empty_point(t)
|
|
142
|
+
{ t: t, count: 0, errors: 0, client_errors: 0, avg: nil, p50: nil, p95: nil, p99: nil }
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
def self.flag_sql(predicate)
|
|
146
|
+
predicate ? "CASE WHEN #{predicate} THEN 1 ELSE 0 END" : "0"
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
# The nearest-rank percentile: the duration whose rank is ceil(n * p/100).
|
|
150
|
+
RANK_SQL = { 50 => Arel.sql("MAX(CASE WHEN rn = (n * 50 + 99) / 100 THEN duration END)"),
|
|
151
|
+
95 => Arel.sql("MAX(CASE WHEN rn = (n * 95 + 99) / 100 THEN duration END)"),
|
|
152
|
+
99 => Arel.sql("MAX(CASE WHEN rn = (n * 99 + 99) / 100 THEN duration END)") }.freeze
|
|
153
|
+
|
|
154
|
+
def self.rank_sql(percent)
|
|
155
|
+
RANK_SQL.fetch(percent)
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
private_class_method :rollup_series, :raw_series, :point, :empty_point, :flag_sql, :rank_sql
|
|
159
|
+
|
|
24
160
|
# Per-group table rows for a record type in the window: count/error totals,
|
|
25
161
|
# merged percentiles (via Telemetry::Rollup.summarize), and a sparkline.
|
|
26
162
|
def self.grouped(environment, record_type, from:, to:, limit: 100, order: nil, dir: nil)
|
|
@@ -34,6 +170,15 @@ module Railwatch
|
|
|
34
170
|
cached(key) { grouped_uncached(environment, record_type, from: from, to: to, limit: limit, order: order, sign: sign) }
|
|
35
171
|
end
|
|
36
172
|
|
|
173
|
+
# The minute cache exists because the platform's rollups only change
|
|
174
|
+
# when RollupJob writes. Embedded, every batch updates them and one
|
|
175
|
+
# person is looking, so the cache would only make the page a minute
|
|
176
|
+
# stale for nothing.
|
|
177
|
+
def self.cached(key, &block)
|
|
178
|
+
return yield if Railwatch.config.local?
|
|
179
|
+
Rails.cache.fetch(key, expires_in: 1.minute, &block)
|
|
180
|
+
end
|
|
181
|
+
|
|
37
182
|
def self.grouped_uncached(environment, record_type, from:, to:, limit:, order:, sign:)
|
|
38
183
|
environment.with_telemetry do
|
|
39
184
|
rows = Telemetry::Rollup.for_type(record_type).between(from, to).to_a
|
|
@@ -60,16 +205,6 @@ module Railwatch
|
|
|
60
205
|
end
|
|
61
206
|
end
|
|
62
207
|
|
|
63
|
-
# The minute cache exists because the platform's rollups only change
|
|
64
|
-
# when RollupJob writes. Embedded, every batch updates them and one
|
|
65
|
-
# person is looking, so the cache would only make the page a minute
|
|
66
|
-
# stale for nothing.
|
|
67
|
-
def self.cached(key, &block)
|
|
68
|
-
return yield if Railwatch.config.local?
|
|
69
|
-
|
|
70
|
-
Rails.cache.fetch(key, expires_in: 1.minute, &block)
|
|
71
|
-
end
|
|
72
|
-
|
|
73
208
|
# The digest-free half of Rollup.summarize: everything the sort keys
|
|
74
209
|
# that are not percentiles need.
|
|
75
210
|
def self.cheap_summary(rows)
|
|
@@ -9,6 +9,13 @@ module Railwatch
|
|
|
9
9
|
|
|
10
10
|
scope :unhandled, -> { where(handled: false) }
|
|
11
11
|
|
|
12
|
+
def as_row
|
|
13
|
+
{ id: id, class_name: class_name, message: message.first(500), handled: handled, severity: severity, source: source,
|
|
14
|
+
file: file, line: line, occurred_at: occurred_at, deploy: deploy, execution_id: execution_id,
|
|
15
|
+
execution_source: execution_source, execution_preview: execution_preview, user_ref: user_ref, tenant: app_tenant,
|
|
16
|
+
group_hash: group_hash }
|
|
17
|
+
end
|
|
18
|
+
|
|
12
19
|
def timeline_label
|
|
13
20
|
"#{class_name}: #{message.to_s.first(100)}"
|
|
14
21
|
end
|
|
@@ -63,6 +63,24 @@ module Railwatch
|
|
|
63
63
|
duration / 1000.0
|
|
64
64
|
end
|
|
65
65
|
|
|
66
|
+
def queue_latency_ms
|
|
67
|
+
queue_latency && (queue_latency / 1000.0).round(1)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# The row every list surface shows for an execution: what the kind has
|
|
71
|
+
# in common, then what it adds.
|
|
72
|
+
def as_row
|
|
73
|
+
row = { execution_id: execution_id, kind: kind, name: name, status: status, outcome: outcome, duration: duration_ms.round(2),
|
|
74
|
+
occurred_at: occurred_at, deploy: deploy, server: server, user_ref: user_ref, tenant: app_tenant,
|
|
75
|
+
exception_preview: exception_preview }
|
|
76
|
+
case kind
|
|
77
|
+
when "request" then row.merge(method: self[:method], inertia_component: inertia_component, queries: counters["queries"])
|
|
78
|
+
when "job_attempt" then row.merge(queue: queue, attempt: attempt, job_id: job_id, queue_latency: queue_latency_ms)
|
|
79
|
+
when "scheduled_task" then row.merge(task_key: task_key)
|
|
80
|
+
else row
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
|
|
66
84
|
def children_count
|
|
67
85
|
counters.values.sum
|
|
68
86
|
end
|
|
@@ -77,6 +77,12 @@ module Railwatch
|
|
|
77
77
|
def quote(term) = %("#{term.gsub('"', '""')}")
|
|
78
78
|
end
|
|
79
79
|
|
|
80
|
+
def as_row
|
|
81
|
+
{ id: id, level: level, message: message.first(2000), tags: tags, occurred_at: occurred_at, deploy: deploy,
|
|
82
|
+
execution_id: execution_id, execution_source: execution_source, execution_preview: execution_preview,
|
|
83
|
+
source: source, tenant: app_tenant, user_ref: user_ref, context: context }
|
|
84
|
+
end
|
|
85
|
+
|
|
80
86
|
def timeline_label
|
|
81
87
|
message.to_s.first(120)
|
|
82
88
|
end
|
|
@@ -42,6 +42,13 @@ module Railwatch
|
|
|
42
42
|
text
|
|
43
43
|
end
|
|
44
44
|
|
|
45
|
+
def as_row
|
|
46
|
+
{ id: id, group_hash: group_hash, sql: sql.first(2_000), name: name, duration: duration_ms.round(3), occurred_at: occurred_at,
|
|
47
|
+
deploy: deploy, execution_id: execution_id, execution_preview: execution_preview, source: source,
|
|
48
|
+
connection: connection, role: role, adapter: adapter, row_count: row_count, tenant: app_tenant, user_ref: user_ref,
|
|
49
|
+
explain: explain.present? }
|
|
50
|
+
end
|
|
51
|
+
|
|
45
52
|
def timeline_label
|
|
46
53
|
sql.to_s.first(120)
|
|
47
54
|
end
|
|
@@ -36,13 +36,19 @@ module Railwatch
|
|
|
36
36
|
}
|
|
37
37
|
end
|
|
38
38
|
|
|
39
|
-
#
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
39
|
+
# Points for the stacked sessions-by-status chart, one per `step` across
|
|
40
|
+
# the window. The table is hourly and a session is counted once per hour
|
|
41
|
+
# it was seen, so a step under an hour is drawn at an hour: raw session
|
|
42
|
+
# rows would count a live session on every flush.
|
|
43
|
+
def self.series(deploy, from, to, step: nil)
|
|
44
|
+
step = [ Telemetry::Aggregations::STEPS.fetch(step || Telemetry::Aggregations.default_step(from, to)), 1.hour ].max
|
|
45
|
+
bucket = Telemetry::Aggregations.bucket_sql("bucket", step)
|
|
46
|
+
points = between(from, to).for_release(deploy).group(bucket).order(bucket)
|
|
47
|
+
.pluck(bucket, Arel.sql("SUM(sessions)"), Arel.sql("SUM(sessions_errored)"), Arel.sql("SUM(sessions_crashed)"))
|
|
48
|
+
.map { |b, sessions, errored, crashed|
|
|
49
|
+
{ t: Time.at(b).utc, sessions: sessions, ok: sessions - errored - crashed, errored: errored, crashed: crashed }
|
|
45
50
|
}
|
|
51
|
+
Telemetry::Aggregations.fill(points, from, to, step) { |t| { t: t, sessions: 0, ok: 0, errored: 0, crashed: 0 } }
|
|
46
52
|
end
|
|
47
53
|
|
|
48
54
|
# Each release's share of the window's sessions, as a percentage -- how
|
|
@@ -54,7 +54,7 @@ module Railwatch
|
|
|
54
54
|
# Aggregate a relation of rollups into one summary with merged percentiles.
|
|
55
55
|
def self.summarize(relation)
|
|
56
56
|
rows = relation.to_a
|
|
57
|
-
return { count: 0, errors: 0, client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0, max: 0 } if rows.empty?
|
|
57
|
+
return { count: 0, errors: 0, client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0, max: 0, extra: {} } if rows.empty?
|
|
58
58
|
# merge! pushes the row's centroids into one accumulating digest.
|
|
59
59
|
# `+` built a brand-new digest from both operands' centroids on every
|
|
60
60
|
# row, so merging N rows re-pushed every earlier centroid N times:
|
|
@@ -70,9 +70,24 @@ module Railwatch
|
|
|
70
70
|
p50: merged.percentile(0.5).to_i,
|
|
71
71
|
p95: merged.percentile(0.95).to_i,
|
|
72
72
|
p99: merged.percentile(0.99).to_i,
|
|
73
|
-
max: rows.map(&:duration_max).max
|
|
73
|
+
max: rows.map(&:duration_max).max,
|
|
74
|
+
# Everything a type puts in `extra` -- llm_call's cost_nanos and
|
|
75
|
+
# token counts, cache_event's hits and misses -- merged the way
|
|
76
|
+
# absorb! merges it: numbers add up, anything else is last-wins.
|
|
77
|
+
# Spend is a sum over a window rather than a percentile of
|
|
78
|
+
# durations, so a rule about money has nowhere else to read from.
|
|
79
|
+
extra: merge_extras(rows)
|
|
74
80
|
}
|
|
75
81
|
end
|
|
82
|
+
|
|
83
|
+
def self.merge_extras(rows)
|
|
84
|
+
rows.each_with_object({}) do |row, out|
|
|
85
|
+
row.extra.each do |key, value|
|
|
86
|
+
existing = out[key]
|
|
87
|
+
out[key] = existing.is_a?(Numeric) && value.is_a?(Numeric) ? existing + value : value
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
end
|
|
76
91
|
end
|
|
77
92
|
end
|
|
78
93
|
end
|
|
@@ -18,7 +18,6 @@ module Railwatch
|
|
|
18
18
|
# aggregate), so only the busiest tenants get one; the rest report 0.
|
|
19
19
|
P95_TENANTS = 50
|
|
20
20
|
SPARKLINE_BUCKETS = 12
|
|
21
|
-
SERIES_BUCKETS = 48
|
|
22
21
|
SORTS = { "requests" => :requests, "errors" => :errors, "p95" => :p95, "users" => :users }.freeze
|
|
23
22
|
|
|
24
23
|
ERRORS_SQL = Arel.sql("SUM(CASE WHEN status >= 500 THEN 1 ELSE 0 END)")
|
|
@@ -71,21 +70,12 @@ module Railwatch
|
|
|
71
70
|
}
|
|
72
71
|
end
|
|
73
72
|
|
|
74
|
-
# Request volume and duration bucketed over the window
|
|
75
|
-
# so the same charts render it.
|
|
76
|
-
#
|
|
77
|
-
#
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
width = bucket_width(from, to, SERIES_BUCKETS)
|
|
81
|
-
bucket = bucket_sql(from, width)
|
|
82
|
-
Telemetry::Execution.requests.between(from, to).where(app_tenant: tenant).group(bucket)
|
|
83
|
-
.pluck(bucket, COUNT_SQL, ERRORS_SQL, CLIENT_ERRORS_SQL, Arel.sql("SUM(duration)"), Arel.sql("MAX(duration)"))
|
|
84
|
-
.map { |index, count, errors, client_errors, sum, max|
|
|
85
|
-
avg = ms(sum / count)
|
|
86
|
-
{ t: bucket_at(from, width, index, SERIES_BUCKETS).iso8601, count: count, errors: errors, client_errors: client_errors,
|
|
87
|
-
avg: avg, p50: avg, p95: ms(max), p99: ms(max) }
|
|
88
|
-
}.sort_by { |point| point[:t] }
|
|
73
|
+
# Request volume and duration bucketed over the window at `step`,
|
|
74
|
+
# SeriesPoint-shaped so the same charts render it. Rollups carry no tenant,
|
|
75
|
+
# so every step reads raw rows (Telemetry::Aggregations.points does that
|
|
76
|
+
# for a tenant), with exact percentiles per bucket.
|
|
77
|
+
def self.series(tenant, from, to, step: nil)
|
|
78
|
+
Telemetry::Aggregations.points("request", from: from, to: to, step: step, tenant: tenant)
|
|
89
79
|
end
|
|
90
80
|
|
|
91
81
|
def self.routes(tenant, from, to, limit: 20)
|
|
@@ -109,14 +99,9 @@ module Railwatch
|
|
|
109
99
|
end
|
|
110
100
|
end
|
|
111
101
|
|
|
112
|
-
#
|
|
113
|
-
# page's "Recent requests" table.
|
|
102
|
+
# The tenant page's "Recent requests" table.
|
|
114
103
|
def self.recent_requests(tenant, from, to, limit: 50)
|
|
115
|
-
Telemetry::Execution.requests.between(from, to).where(app_tenant: tenant).recent.limit(limit).map
|
|
116
|
-
{ execution_id: r.execution_id, name: r.name, status: r.status, duration: r.duration_ms.round(2), occurred_at: r.occurred_at,
|
|
117
|
-
user_ref: r.user_ref, tenant: r.app_tenant, exception_preview: r.exception_preview, inertia_component: r.inertia_component,
|
|
118
|
-
queries: r.counters["queries"], deploy: r.deploy }
|
|
119
|
-
end
|
|
104
|
+
Telemetry::Execution.requests.between(from, to).where(app_tenant: tenant).recent.limit(limit).map(&:as_row)
|
|
120
105
|
end
|
|
121
106
|
|
|
122
107
|
# -- Aggregation steps -----------------------------------------------------
|
|
@@ -2,12 +2,14 @@
|
|
|
2
2
|
|
|
3
3
|
module Railwatch
|
|
4
4
|
# Base class for everything the engine stores about the host application.
|
|
5
|
-
#
|
|
6
|
-
#
|
|
7
|
-
#
|
|
8
|
-
#
|
|
9
|
-
#
|
|
10
|
-
#
|
|
5
|
+
# An embedded install monitors exactly one application, so this is a plain
|
|
6
|
+
# second database (`railwatch_telemetry` in the host's database.yml),
|
|
7
|
+
# separate from the host's own primary and from the engine's meta tables.
|
|
8
|
+
# The hosted platform keeps one SQLite file per monitored environment
|
|
9
|
+
# instead: it calls `Railwatch::TelemetryRecord.tenanted("telemetry")`
|
|
10
|
+
# (activerecord-tenanted) from an initializer, which replaces the
|
|
11
|
+
# connection below with a per-tenant one, and every telemetry query runs
|
|
12
|
+
# inside Environment#with_telemetry so the code path is the same in both.
|
|
11
13
|
class TelemetryRecord < ActiveRecord::Base
|
|
12
14
|
self.abstract_class = true
|
|
13
15
|
begin
|
|
@@ -7,8 +7,22 @@ module Railwatch
|
|
|
7
7
|
self.table_name = "railwatch_thresholds"
|
|
8
8
|
MAX_LIMIT = 1_000_000_000
|
|
9
9
|
MAX_WINDOW_MINUTES = 1_440
|
|
10
|
-
TARGET_KINDS = %w[requests jobs commands queries scheduled_tasks outgoing_requests
|
|
11
|
-
|
|
10
|
+
TARGET_KINDS = %w[requests jobs commands queries scheduled_tasks outgoing_requests
|
|
11
|
+
llm_calls llm_tools].freeze
|
|
12
|
+
METRICS = %w[p95 max avg error_rate failure_rate missed spend tokens truncation_rate].freeze
|
|
13
|
+
|
|
14
|
+
# Duration and rate metrics read the same on any timed record, so they are
|
|
15
|
+
# allowed everywhere. Money and tokens only mean something where a record
|
|
16
|
+
# carries them: a spend rule on tool calls could never fire, and a rule
|
|
17
|
+
# that cannot fire is worse than no rule -- it reads as coverage.
|
|
18
|
+
SHARED_METRICS = %w[p95 max avg error_rate failure_rate missed].freeze
|
|
19
|
+
LLM_METRICS = %w[spend tokens truncation_rate].freeze
|
|
20
|
+
|
|
21
|
+
# Money leads with its symbol; milliseconds and percent follow the number.
|
|
22
|
+
# The old code assumed two units and branched on the metric name ending in
|
|
23
|
+
# "rate", which cannot express a dollar sign in front.
|
|
24
|
+
UNITS = { "spend" => "$", "tokens" => "", "truncation_rate" => "%",
|
|
25
|
+
"error_rate" => "%", "failure_rate" => "%" }.freeze
|
|
12
26
|
|
|
13
27
|
def environment = Environment.current
|
|
14
28
|
|
|
@@ -20,10 +34,33 @@ module Railwatch
|
|
|
20
34
|
validates :target, presence: true, length: { maximum: 256 }
|
|
21
35
|
validates :limit, numericality: { less_than_or_equal_to: 100 }, if: -> { metric&.end_with?("rate") }
|
|
22
36
|
validates :target, uniqueness: { scope: %i[environment_id target_kind metric] }
|
|
37
|
+
validate :metric_applies_to_target_kind
|
|
38
|
+
|
|
39
|
+
def self.metrics_for(target_kind)
|
|
40
|
+
target_kind == "llm_calls" ? SHARED_METRICS + LLM_METRICS : SHARED_METRICS
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def unit = UNITS.fetch(metric, "ms")
|
|
44
|
+
|
|
45
|
+
# "$4.50", "2000ms", "5%", "150000" -- the symbol's position is part of
|
|
46
|
+
# the unit, not something a caller should have to know.
|
|
47
|
+
def format_value(value)
|
|
48
|
+
return "$#{value.is_a?(Float) ? format('%.2f', value) : value}" if unit == "$"
|
|
49
|
+
|
|
50
|
+
"#{value}#{unit}"
|
|
51
|
+
end
|
|
23
52
|
|
|
24
53
|
def description
|
|
25
|
-
|
|
26
|
-
|
|
54
|
+
"#{target_kind} #{target == '*' ? 'all' : target}: #{metric} over #{format_value(limit)} in #{window_minutes}m"
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
private
|
|
58
|
+
|
|
59
|
+
def metric_applies_to_target_kind
|
|
60
|
+
return if metric.blank? || target_kind.blank?
|
|
61
|
+
return if Threshold.metrics_for(target_kind).include?(metric)
|
|
62
|
+
|
|
63
|
+
errors.add(:metric, "is not available for #{target_kind}")
|
|
27
64
|
end
|
|
28
65
|
end
|
|
29
66
|
end
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Railwatch
|
|
4
|
+
# The time range a telemetry page, API call, or MCP tool looks at: one of the
|
|
5
|
+
# fixed presets ("1h" .. "30d") ending now, or a custom [from, to] pair. The
|
|
6
|
+
# dashboard, the JSON API, and MCP all resolve their ?window= / "window"
|
|
7
|
+
# argument through here so the presets and the default agree everywhere.
|
|
8
|
+
class Window
|
|
9
|
+
PRESETS = { "1h" => 1.hour, "6h" => 6.hours, "24h" => 24.hours, "7d" => 7.days, "30d" => 30.days }.freeze
|
|
10
|
+
DEFAULT = "24h"
|
|
11
|
+
MAX_CUSTOM_RANGE = 90.days
|
|
12
|
+
|
|
13
|
+
attr_reader :key, :from, :to
|
|
14
|
+
|
|
15
|
+
# A preset key, or anything else for the default.
|
|
16
|
+
def self.preset(key)
|
|
17
|
+
key = PRESETS.key?(key.to_s) ? key.to_s : DEFAULT
|
|
18
|
+
to = Time.current
|
|
19
|
+
new(key, to - PRESETS[key], to)
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# A custom range from ?from=&to= when both parse, are in order, and span
|
|
23
|
+
# no more than MAX_CUSTOM_RANGE; otherwise the preset (or default).
|
|
24
|
+
def self.parse(window: nil, from: nil, to: nil)
|
|
25
|
+
custom(from, to) || preset(window)
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def self.custom(from, to)
|
|
29
|
+
from = parse_time(from) or return
|
|
30
|
+
to = to.blank? ? Time.current : parse_time(to) or return
|
|
31
|
+
new("custom", from, to) if to > from && (to - from) <= MAX_CUSTOM_RANGE
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def self.parse_time(value)
|
|
35
|
+
return if value.blank?
|
|
36
|
+
Time.zone.parse(value.to_s)
|
|
37
|
+
rescue ArgumentError, TypeError
|
|
38
|
+
nil
|
|
39
|
+
end
|
|
40
|
+
private_class_method :parse_time
|
|
41
|
+
|
|
42
|
+
def initialize(key, from, to)
|
|
43
|
+
@key = key
|
|
44
|
+
@from = from
|
|
45
|
+
@to = to
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
def custom? = key == "custom"
|
|
49
|
+
def range = [ from, to ]
|
|
50
|
+
def span = to - from
|
|
51
|
+
|
|
52
|
+
# The same length immediately before this one, for "vs. previous period".
|
|
53
|
+
def previous
|
|
54
|
+
self.class.new(key, from - span, from)
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def to_h
|
|
58
|
+
{ from: from.iso8601(6), to: to.iso8601(6) }
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
# Query params that reproduce this window on another page.
|
|
62
|
+
def to_params
|
|
63
|
+
custom? ? to_h : { window: key }
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|