railwatch 0.4.0 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +40 -0
- data/app/controllers/railwatch/commands_controller.rb +1 -1
- data/app/controllers/railwatch/environment_scoped.rb +18 -36
- data/app/controllers/railwatch/exceptions_controller.rb +8 -9
- data/app/controllers/railwatch/processes_controller.rb +1 -1
- data/app/controllers/railwatch/releases_controller.rb +2 -2
- data/app/controllers/railwatch/requests_controller.rb +1 -19
- data/app/controllers/railwatch/tenants_controller.rb +2 -2
- data/app/controllers/railwatch/thresholds_controller.rb +8 -1
- data/app/jobs/railwatch/detect_performance_issues_job.rb +12 -2
- data/app/models/railwatch/anomaly_rule.rb +20 -2
- data/app/models/railwatch/execution_presenter.rb +1 -1
- data/app/models/railwatch/filter_query.rb +20 -0
- data/app/models/railwatch/saved_view.rb +1 -1
- data/app/models/railwatch/telemetry/aggregations.rb +154 -19
- data/app/models/railwatch/telemetry/exception.rb +7 -0
- data/app/models/railwatch/telemetry/execution.rb +18 -0
- data/app/models/railwatch/telemetry/log.rb +6 -0
- data/app/models/railwatch/telemetry/query.rb +7 -0
- data/app/models/railwatch/telemetry/release_health.rb +12 -6
- data/app/models/railwatch/telemetry/rollup.rb +17 -2
- data/app/models/railwatch/telemetry/tenant.rb +8 -23
- data/app/models/railwatch/telemetry_record.rb +8 -6
- data/app/models/railwatch/threshold.rb +41 -4
- data/app/models/railwatch/window.rb +66 -0
- data/docs/embedded.md +11 -0
- data/lib/railwatch/version.rb +1 -1
- data/lib/railwatch.rb +12 -0
- data/public/railwatch/assets/app-layout-DDyQa72H.js +1 -0
- data/public/railwatch/assets/{app-wordmark-DDNuuq5_.js → app-wordmark-o9CODKP0.js} +1 -1
- data/public/railwatch/assets/{appearance-C3LMdsDy.js → appearance-BwuCXabr.js} +1 -1
- data/public/railwatch/assets/application-B7h1MIhi.css +1 -0
- data/public/railwatch/assets/{arrow-up-DYTCBENO.js → arrow-up-C6PxDiY3.js} +1 -1
- data/public/railwatch/assets/{auth-layout-CzPRawuI.js → auth-layout-BRt8MGFD.js} +1 -1
- data/public/railwatch/assets/{badge-s7YDycYg.js → badge-CAxXV8za.js} +1 -1
- data/public/railwatch/assets/{braces-BFrg-bvp.js → braces-DgomTCNf.js} +1 -1
- data/public/railwatch/assets/{card-DiW-JhK3.js → card-cAtqCxWl.js} +1 -1
- data/public/railwatch/assets/{chart-nsdZV0Dv.js → chart-BBeBkkNa.js} +1 -1
- data/public/railwatch/assets/{chart-hover-DMWx5-Rs.js → chart-hover-B1M9jc0y.js} +1 -1
- data/public/railwatch/assets/chart-panel-DUQTz_C8.js +1 -0
- data/public/railwatch/assets/{checkbox-C2AqEZXX.js → checkbox-CmhMHWZO.js} +1 -1
- data/public/railwatch/assets/{code-DVJtYP7h.js → code-DESvxyTj.js} +1 -1
- data/public/railwatch/assets/{copy-block-D1myf7is.js → copy-block-BkSU5832.js} +1 -1
- data/public/railwatch/assets/{copy-id-CyRCMfdH.js → copy-id-D03GhN9F.js} +1 -1
- data/public/railwatch/assets/{cursor-load-more-JfSVBn9C.js → cursor-load-more-CRyuMeQb.js} +1 -1
- data/public/railwatch/assets/{data-table-CwtPkw_i.js → data-table-BIlt7Rtm.js} +1 -1
- data/public/railwatch/assets/{edit-fB83T9Yh.js → edit-Bb6MKoe4.js} +1 -1
- data/public/railwatch/assets/{edit-BvTugIpj.js → edit-DJ0D0wHN.js} +1 -1
- data/public/railwatch/assets/{edit-DvpDUM-t.js → edit-O0NSBWxo.js} +1 -1
- data/public/railwatch/assets/{empty-state-CshhuRUU.js → empty-state-C38il627.js} +1 -1
- data/public/railwatch/assets/env-layout-REF7OM4q.js +1 -0
- data/public/railwatch/assets/{execution-path-CdEerUGH.js → execution-path-FYLq1TwC.js} +1 -1
- data/public/railwatch/assets/{filter-bar-BMHf8xBb.js → filter-bar-CYog9Alp.js} +1 -1
- data/public/railwatch/assets/{flamegraph-uYWk1Yov.js → flamegraph-DSs69foN.js} +1 -1
- data/public/railwatch/assets/{frames-B5ZECkvt.js → frames-BUi2J5Mk.js} +1 -1
- data/public/railwatch/assets/{google-sign-in-button-CAvsXYb4.js → google-sign-in-button-BQiIKFdd.js} +1 -1
- data/public/railwatch/assets/{index-X13K9BbO.js → index-8-hnAhOD.js} +1 -1
- data/public/railwatch/assets/{index-Czq2bi1U.js → index-BaR1U9An.js} +1 -1
- data/public/railwatch/assets/{index-iVNm3uzI.js → index-BeOh2t_S.js} +1 -1
- data/public/railwatch/assets/{index-vC9KU2bt.js → index-BiiyMcA0.js} +1 -1
- data/public/railwatch/assets/{index-B-qRsj0W.js → index-BoUBioBP.js} +1 -1
- data/public/railwatch/assets/{index-B4IRsZXK.js → index-C-PmdhXA.js} +1 -1
- data/public/railwatch/assets/{index-fA7h22qS.js → index-C3A_9imx.js} +1 -1
- data/public/railwatch/assets/{index-DcuyMiuG.js → index-C7OtLq_3.js} +1 -1
- data/public/railwatch/assets/{index-BMgA9NNc.js → index-CFFpnzIS.js} +1 -1
- data/public/railwatch/assets/{index-CAkCcAty.js → index-CFRLPs4J.js} +1 -1
- data/public/railwatch/assets/{index-DMcsT2ES.js → index-CGs4m_fa.js} +1 -1
- data/public/railwatch/assets/{index-DuGGTdIJ.js → index-CICUIFHL.js} +1 -1
- data/public/railwatch/assets/{index-DVUwgfqP.js → index-C_upSl_k.js} +1 -1
- data/public/railwatch/assets/{index-CaJvo0Oc.js → index-CiPo4Gob.js} +1 -1
- data/public/railwatch/assets/{index-BIom4V2g.js → index-CpkI015n.js} +1 -1
- data/public/railwatch/assets/{index-Zf9xXwq7.js → index-CrZ3vHDL.js} +1 -1
- data/public/railwatch/assets/{index-ClJtZh5Q.js → index-CsoN51vW.js} +1 -1
- data/public/railwatch/assets/{index-1-78ZewP.js → index-DDI_Zx5V.js} +1 -1
- data/public/railwatch/assets/index-DSvlZVWG.js +1 -0
- data/public/railwatch/assets/{index-C7bU7oUF.js → index-DW2CBbxU.js} +1 -1
- data/public/railwatch/assets/{index-DzoUj38Q.js → index-Dh4IRLFI.js} +1 -1
- data/public/railwatch/assets/{index-CdM3F3jz.js → index-DqTFTP8p.js} +1 -1
- data/public/railwatch/assets/index-DrcKVG2f.js +1 -0
- data/public/railwatch/assets/{index-BIZsdfdM.js → index-DtHmuB9Q.js} +1 -1
- data/public/railwatch/assets/{index-QxAKxnHK.js → index-DvjY3dPD.js} +1 -1
- data/public/railwatch/assets/{index-dKn1tKSX.js → index-FhUaPPab.js} +1 -1
- data/public/railwatch/assets/{index-DYrF-A96.js → index-JdCVBrw8.js} +1 -1
- data/public/railwatch/assets/{index-DWMKU6mG.js → index-QpTtwFwu.js} +1 -1
- data/public/railwatch/assets/{index-Bw_ByIps.js → index-ZOGOB8SA.js} +1 -1
- data/public/railwatch/assets/{index-D-O1LUPq.js → index-ZSZg9rtq.js} +1 -1
- data/public/railwatch/assets/{index-ClG89uia.js → index-r0tSIplE.js} +1 -1
- data/public/railwatch/assets/{index-DJIXGkfB.js → index-sTYvcbkh.js} +1 -1
- data/public/railwatch/assets/{index-n16aviHx.js → index-so4lRrRq.js} +1 -1
- data/public/railwatch/assets/{index-CRShH-fR.js → index-tpz-OGUP.js} +1 -1
- data/public/railwatch/assets/{index-BpnfuXi-.js → index-umIAl-pL.js} +1 -1
- data/public/railwatch/assets/{inertia-CUTKXjxg.js → inertia-DLew8ZNx.js} +2 -2
- data/public/railwatch/assets/{input-error-C8ycyeD0.js → input-error-cvM6_Jht.js} +1 -1
- data/public/railwatch/assets/{json-viewer-CJECqZ0l.js → json-viewer-D922McGi.js} +1 -1
- data/public/railwatch/assets/{klass-Br9dng39.js → klass-CrwICqN8.js} +1 -1
- data/public/railwatch/assets/{label-B32Ks8n7.js → label-GWl7I6sf.js} +1 -1
- data/public/railwatch/assets/{layout-5ZhRtywp.js → layout-0ZAnD3zl.js} +1 -1
- data/public/railwatch/assets/{live-dot-Bb1UYT9F.js → live-dot-D1n_BreY.js} +1 -1
- data/public/railwatch/assets/{nav-XUcwnBso.js → nav-DPxr1NNC.js} +1 -1
- data/public/railwatch/assets/{new-0xyEF6NL.js → new-Cdl6pqST.js} +1 -1
- data/public/railwatch/assets/{new-GIkDYLt8.js → new-D-ZzUK9a.js} +1 -1
- data/public/railwatch/assets/{new-CdvlEBjJ.js → new-DEVkYv-z.js} +1 -1
- data/public/railwatch/assets/{new-YBWOo9_V.js → new-DHAHDrN7.js} +1 -1
- data/public/railwatch/assets/{new-CV011uLi.js → new-Dz4lZf1L.js} +1 -1
- data/public/railwatch/assets/{new-D9-ORiNW.js → new-GMrRFurX.js} +1 -1
- data/public/railwatch/assets/{onboarding-BqEuP-Ml.js → onboarding-D1vwaHYT.js} +1 -1
- data/public/railwatch/assets/{origin-identity-DrJ4D2TX.js → origin-identity-Bk9yHWZ1.js} +1 -1
- data/public/railwatch/assets/{percentile-picker-_wllqYoJ.js → percentile-picker-DfSx9yJO.js} +1 -1
- data/public/railwatch/assets/{relative-time-CV7V8aSP.js → relative-time-CjIjb8Lg.js} +1 -1
- data/public/railwatch/assets/{release-health-DRNLZtT7.js → release-health-4b3tivEf.js} +1 -1
- data/public/railwatch/assets/{route-D9slUz_a.js → route-C_5BUtHK.js} +1 -1
- data/public/railwatch/assets/{segmented-Bj70Htrs.js → segmented-BgbT3wZa.js} +1 -1
- data/public/railwatch/assets/{select-DJBZX2ky.js → select-DmunxCKE.js} +1 -1
- data/public/railwatch/assets/{separator-UUo0bkyc.js → separator-BXzEdZ_8.js} +1 -1
- data/public/railwatch/assets/series-chart-Xf49v9cv.js +1 -0
- data/public/railwatch/assets/{show-BQtEcTeD.js → show-B2zLAW83.js} +1 -1
- data/public/railwatch/assets/{show-DR6u4CR7.js → show-BLpWUHWD.js} +1 -1
- data/public/railwatch/assets/{show-Pu3Iehw5.js → show-CAl7xcex.js} +1 -1
- data/public/railwatch/assets/{show-DoCg0ifP.js → show-DI8IhNUH.js} +1 -1
- data/public/railwatch/assets/{show-BT3tKCgP.js → show-DSP9Cq_C.js} +2 -2
- data/public/railwatch/assets/{show-C0uC3gnA.js → show-DXs4deaC.js} +1 -1
- data/public/railwatch/assets/{show-D7tP881O.js → show-DYskfl3-.js} +1 -1
- data/public/railwatch/assets/{show-BpyCRL-I.js → show-DcpTFiLi.js} +1 -1
- data/public/railwatch/assets/{show-yQipmite.js → show-Dily73Xk.js} +1 -1
- data/public/railwatch/assets/{show-COfP8yWt.js → show-DlRVS18-.js} +1 -1
- data/public/railwatch/assets/{show-ChDJZjoj.js → show-Dn-GwFZL.js} +1 -1
- data/public/railwatch/assets/{show-BoG7jcwk.js → show-DnR1Dnjd.js} +1 -1
- data/public/railwatch/assets/{show-DnBCvlnd.js → show-SHwZjXb7.js} +1 -1
- data/public/railwatch/assets/{show-CefJwag7.js → show-SvLOcPrx.js} +1 -1
- data/public/railwatch/assets/{show-D7IL22Hy.js → show-Y74rM0VT.js} +1 -1
- data/public/railwatch/assets/{show-TTCk2Met.js → show-mU38uGTg.js} +1 -1
- data/public/railwatch/assets/{show-GruxlCov.js → show-vQ4bndYD.js} +1 -1
- data/public/railwatch/assets/{sort-header-CqQQhaxg.js → sort-header-Dcq9bzmo.js} +1 -1
- data/public/railwatch/assets/{sparkline-cell-BwdBz0yv.js → sparkline-cell-BON3qQUB.js} +1 -1
- data/public/railwatch/assets/{stat-IghK3WjD.js → stat-DFEyFxkO.js} +1 -1
- data/public/railwatch/assets/{status-badge-XYUN9gUA.js → status-badge-BaUKP7Yo.js} +1 -1
- data/public/railwatch/assets/{tenant-path-SxnZbfU1.js → tenant-path-DPZPc985.js} +1 -1
- data/public/railwatch/assets/{text-link-NCL9N5D5.js → text-link-BO77t9Xk.js} +1 -1
- data/public/railwatch/assets/{textarea-DhI1dmQa.js → textarea-DTqrCiV0.js} +1 -1
- data/public/railwatch/assets/{timeline-CLAwsAlo.js → timeline-D5rJ0es2.js} +1 -1
- data/public/railwatch/assets/{transition-DpTb9Yiv.js → transition-DMIrZVth.js} +1 -1
- data/public/railwatch/assets/{use-clipboard-D5i6e5kf.js → use-clipboard-ByoUGQqA.js} +1 -1
- data/public/railwatch/assets/{use-live-V-slSdFJ.js → use-live-D7xKz2ma.js} +1 -1
- data/public/railwatch/manifest.json +1288 -1288
- metadata +118 -117
- data/public/railwatch/assets/app-layout-DUk54qUM.js +0 -1
- data/public/railwatch/assets/application-CwshqwM6.css +0 -1
- data/public/railwatch/assets/chart-panel-BlirnMrQ.js +0 -1
- data/public/railwatch/assets/env-layout-Dkp_4NvM.js +0 -1
- data/public/railwatch/assets/index-CZWzIE8F.js +0 -1
- data/public/railwatch/assets/index-DpWQMikK.js +0 -1
- data/public/railwatch/assets/series-chart-C1YrNlID.js +0 -1
- /data/db/railwatch_telemetry_migrate/{20260919000000_create_export_queue.rb → 20260919000100_create_export_queue.rb} +0 -0
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 6f70fd1ff6ba5db7cfb11bff9fd076b480d6ac899c0a65694b9fbb7ce4753a10
|
|
4
|
+
data.tar.gz: da7db6bc23129be7120f38405d522b1be212b76dd7d30435088fbd462e795028
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 7c042d82399b79a4aeefaefd0b6a9d9af3d6868af5b4656d215115b46d84c233bbbe185c6e4692170c36c057431328f66d01ad2b4885d8b470e3cc534dbc17b1
|
|
7
|
+
data.tar.gz: 8eb054b07c1e559362647bd125e030add13ee7b4d84d03bdd72dd1bfe3b89cdb26d98f961d01c20c682cf320570700abd6d61baca1c7263c5b3b44004d71926b
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,45 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.5.1 (2026-09-21)
|
|
4
|
+
|
|
5
|
+
Three small seams for a host that runs these models on its own routes and
|
|
6
|
+
migration history, found while Railwatch Cloud deleted its fork of them.
|
|
7
|
+
|
|
8
|
+
- `Railwatch.url_helpers` is where the models resolve the links they build
|
|
9
|
+
(a trace from an execution page, a saved view's page). It is the engine's
|
|
10
|
+
routes unless the host sets it to its own.
|
|
11
|
+
- `FilterQuery.filter_routes` narrows grouped route rows by `method:`,
|
|
12
|
+
`route:`, and `status:`; it lived in the requests controller.
|
|
13
|
+
- The export queue migration is renumbered to 20260919000100. A host with
|
|
14
|
+
a telemetry migration of its own at the old number would otherwise fail
|
|
15
|
+
to run either.
|
|
16
|
+
|
|
17
|
+
## 0.5.0 (2026-09-21)
|
|
18
|
+
|
|
19
|
+
The telemetry layer the hosted platform runs is this gem's, not a copy. Everything
|
|
20
|
+
Railwatch Cloud had improved in its fork comes home, and the two seams the platform
|
|
21
|
+
needs are named.
|
|
22
|
+
|
|
23
|
+
- **Chart bucket widths.** `Telemetry::Aggregations` offers `STEPS` (1m to 1d),
|
|
24
|
+
`default_step`, and `steps_for` a window; `series` takes a `step:` and reads raw
|
|
25
|
+
rows with exact per-bucket percentiles under an hour, rollups at an hour and up;
|
|
26
|
+
`fill` zero-fills quiet buckets. The dashboard shows a step picker beside the
|
|
27
|
+
window picker, and every series page, release health, tenants, processes, and
|
|
28
|
+
exceptions draw at the chosen step. Latency lines break over an empty bucket
|
|
29
|
+
instead of dropping to zero.
|
|
30
|
+
- **One `Window`.** `Railwatch::Window` resolves `?window=` presets and custom
|
|
31
|
+
`?from=&to=` ranges, with the previous period for deltas; the controller concern
|
|
32
|
+
reads it instead of carrying the table.
|
|
33
|
+
- **Rollup extras.** `Telemetry::Rollup.summarize` merges each type's `extra`
|
|
34
|
+
(LLM spend and tokens, cache hits and misses) the way `absorb!` does.
|
|
35
|
+
- **LLM thresholds and anomaly rules.** `spend`, `tokens`, and `truncation_rate`
|
|
36
|
+
metrics on `llm_calls` and `llm_tools`; the form narrows metrics by kind and
|
|
37
|
+
shows the unit; `Threshold#format_value` prints dollars and percentages.
|
|
38
|
+
- **`as_row`** on Execution, Log, Exception, and Query: the one row shape every
|
|
39
|
+
list surface shows. The commands page reads the exit code from `status`.
|
|
40
|
+
- `Railwatch::TelemetryRecord` documents that the hosted platform tenants it, and
|
|
41
|
+
`docs/embedded.md` says what the platform is in terms of this engine.
|
|
42
|
+
|
|
3
43
|
## 0.4.0 (2026-09-21)
|
|
4
44
|
|
|
5
45
|
The gem is the one author of what goes over the wire and what runs in the
|
|
@@ -5,7 +5,7 @@ module Railwatch
|
|
|
5
5
|
def index
|
|
6
6
|
rows = telemetry { Telemetry::Execution.commands.between(*window_range).recent.limit(200).to_a }
|
|
7
7
|
render inertia: { commands: grouped("command", limit: 100),
|
|
8
|
-
runs: rows.map
|
|
8
|
+
runs: rows.map(&:as_row) }
|
|
9
9
|
end
|
|
10
10
|
|
|
11
11
|
def show
|
|
@@ -6,14 +6,13 @@ module Railwatch
|
|
|
6
6
|
module EnvironmentScoped
|
|
7
7
|
extend ActiveSupport::Concern
|
|
8
8
|
|
|
9
|
-
WINDOWS = { "1h" => 1.hour, "6h" => 6.hours, "24h" => 24.hours, "7d" => 7.days, "30d" => 30.days }.freeze
|
|
10
|
-
MAX_CUSTOM_RANGE = 90.days
|
|
11
|
-
|
|
12
9
|
included do
|
|
13
10
|
before_action :set_environment
|
|
14
11
|
inertia_share environment: -> { environment_props }
|
|
15
|
-
inertia_share window: -> {
|
|
16
|
-
inertia_share range: -> {
|
|
12
|
+
inertia_share window: -> { window.key }
|
|
13
|
+
inertia_share range: -> { window.to_h }
|
|
14
|
+
inertia_share step: -> { step_key }
|
|
15
|
+
inertia_share steps: -> { Telemetry::Aggregations.steps_for(*window_range) }
|
|
17
16
|
inertia_share saved_views: -> { SavedView.props_for(environment, Viewer.user) }
|
|
18
17
|
end
|
|
19
18
|
|
|
@@ -25,38 +24,19 @@ module Railwatch
|
|
|
25
24
|
@environment = Environment.find(params[:environment_id] || Environment::ID)
|
|
26
25
|
end
|
|
27
26
|
|
|
28
|
-
def
|
|
29
|
-
|
|
30
|
-
end
|
|
31
|
-
|
|
32
|
-
def window_range
|
|
33
|
-
return @window_range if defined?(@window_range)
|
|
34
|
-
@window_range = custom_range || begin
|
|
35
|
-
key = WINDOWS.key?(params[:window].to_s) ? params[:window].to_s : "24h"
|
|
36
|
-
to = Time.current
|
|
37
|
-
[ to - WINDOWS[key], to ]
|
|
38
|
-
end
|
|
27
|
+
def window
|
|
28
|
+
@window ||= Window.parse(window: params[:window], from: params[:from], to: params[:to])
|
|
39
29
|
end
|
|
40
30
|
|
|
41
|
-
def
|
|
42
|
-
return @custom_range if defined?(@custom_range)
|
|
43
|
-
from = parse_time(params[:from])
|
|
44
|
-
to = from && (parse_time(params[:to]) || Time.current)
|
|
45
|
-
valid = from && to && to > from && (to - from) <= MAX_CUSTOM_RANGE
|
|
46
|
-
@custom_range = valid ? [ from, to ] : nil
|
|
47
|
-
end
|
|
48
|
-
|
|
49
|
-
def parse_time(value)
|
|
50
|
-
return nil if value.blank?
|
|
51
|
-
Time.zone.parse(value.to_s)
|
|
52
|
-
rescue ArgumentError, TypeError
|
|
53
|
-
nil
|
|
54
|
-
end
|
|
31
|
+
def window_range = window.range
|
|
55
32
|
|
|
56
|
-
|
|
33
|
+
# The chart bucket width: ?step= when it is one the window offers, else the
|
|
34
|
+
# window's default (a minute for an hour, an hour for a day, and so on).
|
|
35
|
+
def step_key
|
|
36
|
+
return @step_key if defined?(@step_key)
|
|
57
37
|
from, to = window_range
|
|
58
|
-
|
|
59
|
-
[
|
|
38
|
+
offered = Telemetry::Aggregations.steps_for(from, to)
|
|
39
|
+
@step_key = offered.include?(params[:step].to_s) ? params[:step].to_s : Telemetry::Aggregations.default_step(from, to)
|
|
60
40
|
end
|
|
61
41
|
|
|
62
42
|
def telemetry(&block) = environment.with_telemetry(&block)
|
|
@@ -79,9 +59,11 @@ module Railwatch
|
|
|
79
59
|
environment.deploys.between(*window_range).recent.limit(50).map { |d| { deploy: d.deploy, ref: d.short_ref, at: d.deployed_at } }
|
|
80
60
|
end
|
|
81
61
|
|
|
82
|
-
|
|
62
|
+
# Time-bucketed series for charts, one point per step_key across the
|
|
63
|
+
# window: [{t, count, errors, client_errors, avg, p50, p95, p99}]
|
|
64
|
+
def series(record_type, group_hash: nil)
|
|
83
65
|
from, to = window_range
|
|
84
|
-
Telemetry::Aggregations.series(environment, record_type, from: from, to: to, group_hash: group_hash,
|
|
66
|
+
Telemetry::Aggregations.series(environment, record_type, from: from, to: to, group_hash: group_hash, step: step_key)
|
|
85
67
|
end
|
|
86
68
|
|
|
87
69
|
def grouped(record_type, limit: 100, order: nil, dir: nil)
|
|
@@ -91,7 +73,7 @@ module Railwatch
|
|
|
91
73
|
|
|
92
74
|
def summary_with_delta(record_type, group_hash: nil)
|
|
93
75
|
from, to = window_range
|
|
94
|
-
previous_from, previous_to =
|
|
76
|
+
previous_from, previous_to = window.previous.range
|
|
95
77
|
Telemetry::Aggregations.summary_with_delta(environment, record_type, from: from, to: to,
|
|
96
78
|
previous_from: previous_from, previous_to: previous_to, group_hash: group_hash)
|
|
97
79
|
end
|
|
@@ -38,16 +38,15 @@ module Railwatch
|
|
|
38
38
|
# `errors` carries the unhandled count so the chart legend (handled /
|
|
39
39
|
# unhandled) agrees with the Unhandled card above the table.
|
|
40
40
|
def exception_series(from, to)
|
|
41
|
+
step = Telemetry::Aggregations::STEPS.fetch(step_key)
|
|
42
|
+
bucket = Telemetry::Aggregations.bucket_sql("occurred_at", step)
|
|
41
43
|
buckets = Hash.new { |h, k| h[k] = { count: 0, errors: 0 } }
|
|
42
|
-
Telemetry::Exception.between(from, to)
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
end
|
|
49
|
-
buckets.map { |bucket, b| { t: bucket, count: b[:count], errors: b[:errors], client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0 } }
|
|
50
|
-
.sort_by { |r| r[:t] }
|
|
44
|
+
Telemetry::Exception.between(from, to).group(bucket, :handled).count.each do |(b, handled), c|
|
|
45
|
+
buckets[b][:count] += c
|
|
46
|
+
buckets[b][:errors] += c unless handled
|
|
47
|
+
end
|
|
48
|
+
points = buckets.map { |b, c| { t: Time.at(b).utc, count: c[:count], errors: c[:errors], client_errors: 0, avg: nil, p50: nil, p95: nil, p99: nil } }
|
|
49
|
+
Telemetry::Aggregations.fill(points, from, to, step)
|
|
51
50
|
end
|
|
52
51
|
end
|
|
53
52
|
end
|
|
@@ -6,7 +6,7 @@ module Railwatch
|
|
|
6
6
|
live_since = Telemetry::HealthSample::LIVE_WINDOW.ago
|
|
7
7
|
samples, series, queues, live_servers, processes = telemetry do
|
|
8
8
|
live = Telemetry::HealthSample.live(live_since).to_a
|
|
9
|
-
[ live, Telemetry::HealthSample.series(*window_range), Telemetry::HealthSample.queue_depths(live),
|
|
9
|
+
[ live, Telemetry::HealthSample.series(*window_range, bucket: Telemetry::Aggregations::STEPS.fetch(step_key)), Telemetry::HealthSample.queue_depths(live),
|
|
10
10
|
live.map(&:server) | Telemetry::Execution.servers_since(live_since),
|
|
11
11
|
Telemetry::Process.recent.limit(200).to_a ]
|
|
12
12
|
end
|
|
@@ -17,7 +17,7 @@ module Railwatch
|
|
|
17
17
|
new_issues = new_issue_counts(deploys, to)
|
|
18
18
|
summary, series, adoption, health = telemetry do
|
|
19
19
|
[ Telemetry::ReleaseHealth.for_deploy(nil, from, to),
|
|
20
|
-
Telemetry::ReleaseHealth.series(nil, from, to),
|
|
20
|
+
Telemetry::ReleaseHealth.series(nil, from, to, step: step_key),
|
|
21
21
|
Telemetry::ReleaseHealth.adoption(from, to),
|
|
22
22
|
deploys.to_h { |d| [ d.deploy, Telemetry::ReleaseHealth.for_deploy(d.deploy, from, to) ] } ]
|
|
23
23
|
end
|
|
@@ -37,7 +37,7 @@ module Railwatch
|
|
|
37
37
|
data = telemetry do
|
|
38
38
|
{ summary: Telemetry::ReleaseHealth.for_deploy(release, from, to),
|
|
39
39
|
previous_summary: previous && Telemetry::ReleaseHealth.for_deploy(previous.deploy, *previous.window),
|
|
40
|
-
series: Telemetry::ReleaseHealth.series(release, from, to),
|
|
40
|
+
series: Telemetry::ReleaseHealth.series(release, from, to, step: step_key),
|
|
41
41
|
sessions: Telemetry::Session.where(deploy: release).recent.limit(SESSIONS_SHOWN).map { |s| session_row(s) } }
|
|
42
42
|
end
|
|
43
43
|
render inertia: data.merge(
|
|
@@ -4,7 +4,7 @@ module Railwatch
|
|
|
4
4
|
class RequestsController < DashboardController
|
|
5
5
|
def index
|
|
6
6
|
routes = grouped("request", limit: 200, order: params[:sort], dir: params[:dir])
|
|
7
|
-
routes =
|
|
7
|
+
routes = FilterQuery.filter_routes(routes, params[:q])
|
|
8
8
|
render inertia: { routes: routes, series: series("request"), deploys: deploys_in_window,
|
|
9
9
|
sort: params[:sort] || "count", dir: params[:dir] || "desc", q: params[:q].to_s }
|
|
10
10
|
end
|
|
@@ -28,24 +28,6 @@ module Railwatch
|
|
|
28
28
|
|
|
29
29
|
private
|
|
30
30
|
|
|
31
|
-
# Rollup rows for "request" carry no raw status codes, only aggregated
|
|
32
|
-
# count/errors/client_errors, so status:5xx etc. is a family match against
|
|
33
|
-
# those buckets rather than an exact code lookup.
|
|
34
|
-
def apply_filters(routes, fields)
|
|
35
|
-
routes = routes.select { |r| r[:name].start_with?("#{fields['method'].upcase} ") } if fields["method"].present?
|
|
36
|
-
routes = routes.select { |r| r[:name].include?(fields["route"]) } if fields["route"].present?
|
|
37
|
-
if (range = FilterQuery.status_range(fields["status"]))
|
|
38
|
-
routes = routes.select do |r|
|
|
39
|
-
case range.begin
|
|
40
|
-
when 500..599 then r[:errors].positive?
|
|
41
|
-
when 400..499 then r[:client_errors].positive?
|
|
42
|
-
else (r[:count] - r[:errors] - r[:client_errors]).positive?
|
|
43
|
-
end
|
|
44
|
-
end
|
|
45
|
-
end
|
|
46
|
-
routes
|
|
47
|
-
end
|
|
48
|
-
|
|
49
31
|
def execution_row(r)
|
|
50
32
|
{ execution_id: r.execution_id, name: r.name, status: r.status, duration: r.duration_ms.round(2), occurred_at: r.occurred_at,
|
|
51
33
|
user_ref: r.user_ref, tenant: r.app_tenant, exception_preview: r.exception_preview, inertia_component: r.inertia_component,
|
|
@@ -20,8 +20,8 @@ module Railwatch
|
|
|
20
20
|
from, to = window_range
|
|
21
21
|
data = telemetry do
|
|
22
22
|
{ tenant: tenant, summary: { current: Telemetry::Tenant.summary(tenant, from, to),
|
|
23
|
-
previous: Telemetry::Tenant.summary(tenant, *
|
|
24
|
-
series: Telemetry::Tenant.series(tenant, from, to), routes: Telemetry::Tenant.routes(tenant, from, to),
|
|
23
|
+
previous: Telemetry::Tenant.summary(tenant, *window.previous.range) },
|
|
24
|
+
series: Telemetry::Tenant.series(tenant, from, to, step: step_key), routes: Telemetry::Tenant.routes(tenant, from, to),
|
|
25
25
|
jobs: Telemetry::Tenant.job_classes(tenant, from, to), exceptions: Telemetry::Tenant.exceptions(tenant, from, to),
|
|
26
26
|
people: Telemetry::Tenant.people(tenant), recent_requests: Telemetry::Tenant.recent_requests(tenant, from, to) }
|
|
27
27
|
end
|
|
@@ -8,7 +8,14 @@ module Railwatch
|
|
|
8
8
|
jobs: telemetry { Telemetry::Rollup.for_type("job_attempt").where("bucket > ?", 7.days.ago).distinct.pluck(:name).sort },
|
|
9
9
|
kinds: Threshold::TARGET_KINDS, metrics: Threshold::METRICS,
|
|
10
10
|
anomaly_rules: environment.anomaly_rules.order(:target_kind, :target).map { |r| r.slice(:id, :target_kind, :target, :metric, :deviation, :window_minutes, :baseline_days, :enabled, :last_fired_at).merge(description: r.description) },
|
|
11
|
-
|
|
11
|
+
# Spend and tokens only exist on llm_calls, so the form
|
|
12
|
+
# narrows with the kind rather than offering a rule that
|
|
13
|
+
# could never fire.
|
|
14
|
+
metrics_for_kind: Threshold::TARGET_KINDS.index_with { |k| Threshold.metrics_for(k) },
|
|
15
|
+
units: Threshold::UNITS,
|
|
16
|
+
llm_models: telemetry { Telemetry::Rollup.for_type("llm_call").where("bucket > ?", 7.days.ago).distinct.pluck(:name).compact.sort },
|
|
17
|
+
anomaly_target_kinds: AnomalyRule::TARGET_KINDS, anomaly_metrics: AnomalyRule::METRICS,
|
|
18
|
+
anomaly_metrics_for_kind: AnomalyRule::TARGET_KINDS.index_with { |k| AnomalyRule.metrics_for(k) } }
|
|
12
19
|
end
|
|
13
20
|
|
|
14
21
|
def create
|
|
@@ -12,7 +12,8 @@ module Railwatch
|
|
|
12
12
|
MAX_ROLLUP_ROWS_PER_RULE = MAX_GROUPS_PER_RULE * 25
|
|
13
13
|
|
|
14
14
|
TYPE_FOR = { "requests" => "request", "jobs" => "job_attempt", "commands" => "command", "queries" => "query",
|
|
15
|
-
"scheduled_tasks" => "scheduled_task", "outgoing_requests" => "outgoing_request"
|
|
15
|
+
"scheduled_tasks" => "scheduled_task", "outgoing_requests" => "outgoing_request",
|
|
16
|
+
"llm_calls" => "llm_call", "llm_tools" => "llm_tool" }.freeze
|
|
16
17
|
|
|
17
18
|
def perform(environment)
|
|
18
19
|
now = Time.current
|
|
@@ -51,19 +52,28 @@ module Railwatch
|
|
|
51
52
|
.transform_values { |group_rows| [ group_rows.first.name, Telemetry::Rollup.summarize(group_rows) ] }
|
|
52
53
|
end
|
|
53
54
|
|
|
55
|
+
NANOS_PER_DOLLAR = 1_000_000_000.0
|
|
56
|
+
|
|
54
57
|
def value_for(metric, s)
|
|
58
|
+
extra = s[:extra] || {}
|
|
55
59
|
case metric
|
|
56
60
|
when "p95" then s[:p95] / 1000.0
|
|
57
61
|
when "max" then s[:max] / 1000.0
|
|
58
62
|
when "avg" then s[:avg] / 1000.0
|
|
59
63
|
when "error_rate", "failure_rate" then s[:count].zero? ? nil : (s[:errors] * 100.0 / s[:count])
|
|
64
|
+
# Money and tokens are sums over the window, not percentiles of it.
|
|
65
|
+
when "spend" then extra["cost_nanos"].to_i / NANOS_PER_DOLLAR
|
|
66
|
+
when "tokens" then extra["input_tokens"].to_i + extra["output_tokens"].to_i
|
|
67
|
+
# A cut-off answer is a successful call, so it is invisible to
|
|
68
|
+
# error_rate. This is the only way to be told about it.
|
|
69
|
+
when "truncation_rate" then s[:count].zero? ? nil : (extra["truncated"].to_i * 100.0 / s[:count])
|
|
60
70
|
end
|
|
61
71
|
end
|
|
62
72
|
|
|
63
73
|
def open_issue(environment, threshold, group_hash, name, value, from, now)
|
|
64
74
|
issue, outcome = Issue.record_occurrence!(
|
|
65
75
|
environment: environment, group_hash: "perf:#{threshold.id}:#{group_hash}", kind: "performance",
|
|
66
|
-
title: "#{name} exceeded #{threshold.metric} #{threshold.
|
|
76
|
+
title: "#{name} exceeded #{threshold.metric} #{threshold.format_value(threshold.limit)} (#{threshold.format_value(value.round(2))})",
|
|
67
77
|
culprit: name, occurred_at: now, deploy: nil, user_ref: nil,
|
|
68
78
|
sample: { threshold_id: threshold.id, value: value.round(2), metric: threshold.metric, limit: threshold.limit })
|
|
69
79
|
carry_detection(issue, outcome) do
|
|
@@ -10,13 +10,22 @@ module Railwatch
|
|
|
10
10
|
MAX_BASELINE_DAYS = 30
|
|
11
11
|
MAX_DEVIATION = 10
|
|
12
12
|
MAX_WINDOW_MINUTES = 1_440
|
|
13
|
-
TARGET_KINDS = %w[requests jobs queries outgoing_requests scheduled_tasks].freeze
|
|
14
|
-
|
|
13
|
+
TARGET_KINDS = %w[requests jobs queries outgoing_requests scheduled_tasks llm_calls].freeze
|
|
14
|
+
# spend included because "we are spending well above baseline for this
|
|
15
|
+
# hour of the week" is the LLM alert that needs no threshold picked.
|
|
16
|
+
METRICS = %w[p95 avg count error_rate spend].freeze
|
|
17
|
+
SHARED_METRICS = %w[p95 avg count error_rate].freeze
|
|
18
|
+
LLM_METRICS = %w[spend].freeze
|
|
19
|
+
|
|
20
|
+
def self.metrics_for(target_kind)
|
|
21
|
+
target_kind == "llm_calls" ? SHARED_METRICS + LLM_METRICS : SHARED_METRICS
|
|
22
|
+
end
|
|
15
23
|
|
|
16
24
|
def environment = Environment.current
|
|
17
25
|
|
|
18
26
|
validates :target_kind, inclusion: { in: TARGET_KINDS }
|
|
19
27
|
validates :metric, inclusion: { in: METRICS }
|
|
28
|
+
validate :metric_applies_to_target_kind
|
|
20
29
|
validates :target, presence: true, length: { maximum: 256 }
|
|
21
30
|
validates :deviation, numericality: { greater_than: 0, less_than_or_equal_to: MAX_DEVIATION }
|
|
22
31
|
validates :window_minutes, numericality: { only_integer: true, greater_than: 0,
|
|
@@ -29,5 +38,14 @@ module Railwatch
|
|
|
29
38
|
def description
|
|
30
39
|
"#{target_kind} #{target == '*' ? 'all' : target}: #{metric} > #{deviation}σ above #{baseline_days}-day baseline"
|
|
31
40
|
end
|
|
41
|
+
|
|
42
|
+
private
|
|
43
|
+
|
|
44
|
+
def metric_applies_to_target_kind
|
|
45
|
+
return if metric.blank? || target_kind.blank?
|
|
46
|
+
return if AnomalyRule.metrics_for(target_kind).include?(metric)
|
|
47
|
+
|
|
48
|
+
errors.add(:metric, "is not available for #{target_kind}")
|
|
49
|
+
end
|
|
32
50
|
end
|
|
33
51
|
end
|
|
@@ -168,7 +168,7 @@ module Railwatch
|
|
|
168
168
|
# is not the whole trace.
|
|
169
169
|
def trace_url
|
|
170
170
|
return nil unless @exe.trace_id && trace.size > 1
|
|
171
|
-
Railwatch
|
|
171
|
+
Railwatch.url_helpers.application_environment_trace_path(@environment.application, @environment, @exe.trace_id)
|
|
172
172
|
end
|
|
173
173
|
|
|
174
174
|
def issues
|
|
@@ -38,6 +38,26 @@ module Railwatch
|
|
|
38
38
|
end
|
|
39
39
|
end
|
|
40
40
|
|
|
41
|
+
# Narrows grouped route rows (Telemetry::Aggregations.grouped for "request")
|
|
42
|
+
# by method:, route:, and status:. Rollup rows carry no raw status codes,
|
|
43
|
+
# only aggregated count/errors/client_errors, so status:5xx and friends are
|
|
44
|
+
# a family match against those buckets rather than an exact code lookup.
|
|
45
|
+
def self.filter_routes(routes, query)
|
|
46
|
+
fields = parse(query).fetch(:fields)
|
|
47
|
+
routes = routes.select { |r| r[:name].start_with?("#{fields['method'].upcase} ") } if fields["method"].present?
|
|
48
|
+
routes = routes.select { |r| r[:name].include?(fields["route"]) } if fields["route"].present?
|
|
49
|
+
if (range = status_range(fields["status"]))
|
|
50
|
+
routes = routes.select do |r|
|
|
51
|
+
case range.begin
|
|
52
|
+
when 500..599 then r[:errors].positive?
|
|
53
|
+
when 400..499 then r[:client_errors].positive?
|
|
54
|
+
else (r[:count] - r[:errors] - r[:client_errors]).positive?
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
routes
|
|
59
|
+
end
|
|
60
|
+
|
|
41
61
|
def self.fields_for(resource)
|
|
42
62
|
COMMON_FIELDS + RESOURCE_FIELDS.fetch(resource.to_sym)
|
|
43
63
|
end
|
|
@@ -56,7 +56,7 @@ module Railwatch
|
|
|
56
56
|
# This page's index path with the view's window, query, and extra params
|
|
57
57
|
# (sort/dir) applied -- the shareable URL.
|
|
58
58
|
def path_for(environment)
|
|
59
|
-
Railwatch
|
|
59
|
+
Railwatch.url_helpers.public_send(PAGE_PATHS.fetch(page), environment.application_id, environment.id,
|
|
60
60
|
{ window: window.presence, q: query.presence, **params.to_h.symbolize_keys }.compact)
|
|
61
61
|
end
|
|
62
62
|
end
|
|
@@ -9,18 +9,154 @@ module Railwatch
|
|
|
9
9
|
class Aggregations
|
|
10
10
|
GROUPED_SORT_FIELDS = %i[count p50 p95 p99 errors avg max].freeze
|
|
11
11
|
|
|
12
|
-
#
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
12
|
+
# Bucket widths a chart can be drawn at. Anything under an hour is read
|
|
13
|
+
# from the raw tables (rollups are hourly); an hour and up groups rollups.
|
|
14
|
+
STEPS = { "1m" => 1.minute, "5m" => 5.minutes, "15m" => 15.minutes, "1h" => 1.hour, "6h" => 6.hours, "1d" => 1.day }.freeze
|
|
15
|
+
# A step is offered for a window when it draws at least two points and at
|
|
16
|
+
# most this many: more bars than pixels is mush, and a sub-hour step over
|
|
17
|
+
# a long window is also a long raw scan.
|
|
18
|
+
MAX_POINTS = 400
|
|
19
|
+
|
|
20
|
+
# The bucket a window is drawn at unless the page asks for another one.
|
|
21
|
+
# Sub-hour steps scan raw rows, so they are the default only where that
|
|
22
|
+
# scan is a few thousand rows; a day and beyond stays on rollups.
|
|
23
|
+
def self.default_step(from, to)
|
|
24
|
+
span = to - from
|
|
25
|
+
if span <= 2.hours then "1m"
|
|
26
|
+
elsif span <= 12.hours then "5m"
|
|
27
|
+
elsif span <= 7.days then "1h"
|
|
28
|
+
else "6h"
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# STEPS keys that make sense for the window, finest first.
|
|
33
|
+
def self.steps_for(from, to)
|
|
34
|
+
span = to - from
|
|
35
|
+
STEPS.select { |_key, step| (span / step).between?(2, MAX_POINTS) }.keys
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
# Time-bucketed series for charts, one point per `step` from `from` to `to`
|
|
39
|
+
# with empty buckets filled in: [{t, count, errors, client_errors, avg,
|
|
40
|
+
# p50, p95, p99}]. Durations in milliseconds; an empty bucket carries nil
|
|
41
|
+
# for them so a latency line breaks instead of dropping to zero.
|
|
42
|
+
def self.series(environment, record_type, from:, to:, group_hash: nil, step: nil, tenant: nil)
|
|
43
|
+
environment.with_telemetry { points(record_type, from: from, to: to, group_hash: group_hash, step: step, tenant: tenant) }
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# `series` for a caller already inside environment.with_telemetry. A
|
|
47
|
+
# tenant narrows to one app_tenant, which rollups do not carry, so a
|
|
48
|
+
# tenant series reads raw rows at every step.
|
|
49
|
+
def self.points(record_type, from:, to:, group_hash: nil, step: nil, tenant: nil)
|
|
50
|
+
step = STEPS.fetch(step || default_step(from, to))
|
|
51
|
+
raw = step < 1.hour || tenant
|
|
52
|
+
points = raw ? raw_series(record_type, from, to, group_hash, step, tenant) : rollup_series(record_type, from, to, group_hash, step)
|
|
53
|
+
fill(points, from, to, step)
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Sums hourly rollups into `step`-wide buckets. Percentiles are the
|
|
57
|
+
# worst hour's, the same reading series always gave across groups.
|
|
58
|
+
def self.rollup_series(record_type, from, to, group_hash, step)
|
|
59
|
+
bucket = bucket_sql("bucket", step)
|
|
60
|
+
scope = Telemetry::Rollup.for_type(record_type).between(from, to)
|
|
61
|
+
scope = scope.where(group_hash: group_hash) if group_hash
|
|
62
|
+
scope.group(bucket).order(bucket)
|
|
63
|
+
.pluck(bucket, Arel.sql("SUM(count)"), Arel.sql("SUM(error_count)"), Arel.sql("SUM(client_error_count)"), Arel.sql("SUM(duration_sum)"), Arel.sql("MAX(p50)"), Arel.sql("MAX(p95)"), Arel.sql("MAX(p99)"))
|
|
64
|
+
.map { |b, c, e, ce, ds, p50, p95, p99| point(b, c, e, ce, ds, p50, p95, p99) }
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# What a raw row of each record type is, and when it counts as an error
|
|
68
|
+
# (or a client error), mirroring RollupJob#error?. Both readings of the
|
|
69
|
+
# same rows must agree, which spec/models/telemetry/aggregations_spec.rb
|
|
70
|
+
# checks for every type.
|
|
71
|
+
RAW = {
|
|
72
|
+
"request" => [ -> { Telemetry::Execution.requests }, "status >= 500", "status BETWEEN 400 AND 499" ],
|
|
73
|
+
"job_attempt" => [ -> { Telemetry::Execution.jobs }, "outcome = 'failed'" ],
|
|
74
|
+
"scheduled_task" => [ -> { Telemetry::Execution.scheduled }, "outcome = 'failed'" ],
|
|
75
|
+
"command" => [ -> { Telemetry::Execution.commands }, "status != 0" ],
|
|
76
|
+
"channel_action" => [ -> { Telemetry::Execution.channels }, "outcome = 'failed'" ],
|
|
77
|
+
"query" => [ -> { Telemetry::Query.all } ],
|
|
78
|
+
"outgoing_request" => [ -> { Telemetry::OutgoingRequest.all }, "status_code >= 500 OR status_code IS NULL OR status_code = 0" ],
|
|
79
|
+
"cache_event" => [ -> { Telemetry::CacheEvent.all } ],
|
|
80
|
+
"mail" => [ -> { Telemetry::Mail.all }, "failed" ],
|
|
81
|
+
"visit" => [ -> { Telemetry::Visit.all }, "status = 'error'" ],
|
|
82
|
+
"span" => [ -> { Telemetry::Span.all }, "status = 'failed'" ],
|
|
83
|
+
"notification" => [ -> { Telemetry::Notification.all }, "failed" ],
|
|
84
|
+
"view_render" => [ -> { Telemetry::ViewRender.all } ],
|
|
85
|
+
"transaction" => [ -> { Telemetry::Transaction.all }, "outcome = 'rollback'" ],
|
|
86
|
+
"llm_call" => [ -> { Telemetry::LlmCall.models }, "status = 'failed'" ],
|
|
87
|
+
"llm_tool" => [ -> { Telemetry::LlmCall.tools }, "status = 'failed'" ]
|
|
88
|
+
}.freeze
|
|
89
|
+
|
|
90
|
+
# Sub-hour buckets straight from the raw rows, in one statement:
|
|
91
|
+
# counts and sums per bucket, and exact nearest-rank percentiles from a
|
|
92
|
+
# ROW_NUMBER over each bucket's durations. Rollups are hourly, so this
|
|
93
|
+
# is the only reading finer than an hour; it is also the freshest one,
|
|
94
|
+
# since the current hour's rollup is up to a minute behind.
|
|
95
|
+
def self.raw_series(record_type, from, to, group_hash, step, tenant = nil)
|
|
96
|
+
scope, error_sql, client_error_sql = RAW.fetch(record_type)
|
|
97
|
+
rows = scope.call.where(occurred_at: from..to)
|
|
98
|
+
rows = rows.where(group_hash: group_hash) if group_hash
|
|
99
|
+
rows = rows.where(app_tenant: tenant) if tenant
|
|
100
|
+
model = rows.model
|
|
101
|
+
inner = rows.select(bucket_sql("occurred_at", step).to_s + " AS b", "duration", flag_sql(error_sql) + " AS err", flag_sql(client_error_sql) + " AS cerr")
|
|
102
|
+
ranked = model.unscoped.from(inner, "r").select("b", "duration", "err", "cerr",
|
|
103
|
+
"ROW_NUMBER() OVER (PARTITION BY b ORDER BY duration) AS rn", "COUNT(*) OVER (PARTITION BY b) AS n")
|
|
104
|
+
model.unscoped.from(ranked, "w").group("b").order("b")
|
|
105
|
+
.pluck(Arel.sql("b"), Arel.sql("COUNT(*)"), Arel.sql("SUM(err)"), Arel.sql("SUM(cerr)"), Arel.sql("SUM(duration)"),
|
|
106
|
+
rank_sql(50), rank_sql(95), rank_sql(99))
|
|
107
|
+
.map { |b, c, e, ce, ds, p50, p95, p99| point(b, c, e, ce, ds, p50, p95, p99) }
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
# The timestamp columns a series buckets on: raw rows by occurred_at,
|
|
111
|
+
# rollups and release health by their hourly bucket.
|
|
112
|
+
BUCKET_COLUMNS = %w[occurred_at bucket].freeze
|
|
113
|
+
|
|
114
|
+
# Epoch seconds of the start of the `step`-wide bucket holding `column`.
|
|
115
|
+
# `column` must be one of BUCKET_COLUMNS and `step` is a Duration, so the
|
|
116
|
+
# fragment can only ever hold a listed column name and an integer literal.
|
|
117
|
+
def self.bucket_sql(column, step)
|
|
118
|
+
raise ArgumentError, "unknown bucket column #{column.inspect}" unless BUCKET_COLUMNS.include?(column)
|
|
119
|
+
seconds = Integer(step.to_i)
|
|
120
|
+
Arel.sql("(strftime('%s', #{column}) / #{seconds}) * #{seconds}")
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
# One point per bucket from `from` to `to`, keeping the computed ones and
|
|
124
|
+
# zero-filling the rest, so a quiet minute is a gap of the right width.
|
|
125
|
+
# A series with its own point shape passes a block building its empty one.
|
|
126
|
+
def self.fill(points, from, to, step)
|
|
127
|
+
by_bucket = points.index_by { |p| p[:t] }
|
|
128
|
+
first = Time.at((from.to_i / step.to_i) * step.to_i).utc
|
|
129
|
+
last = Time.at((to.to_i / step.to_i) * step.to_i).utc
|
|
130
|
+
(first.to_i..last.to_i).step(step.to_i).map do |t|
|
|
131
|
+
at = Time.at(t).utc
|
|
132
|
+
by_bucket[at] || (block_given? ? yield(at) : empty_point(at))
|
|
21
133
|
end
|
|
22
134
|
end
|
|
23
135
|
|
|
136
|
+
def self.point(bucket, count, errors, client_errors, duration_sum, p50, p95, p99)
|
|
137
|
+
{ t: Time.at(bucket).utc, count: count, errors: errors, client_errors: client_errors,
|
|
138
|
+
avg: count.zero? ? 0 : duration_sum / count / 1000.0, p50: p50 / 1000.0, p95: p95 / 1000.0, p99: p99 / 1000.0 }
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
def self.empty_point(t)
|
|
142
|
+
{ t: t, count: 0, errors: 0, client_errors: 0, avg: nil, p50: nil, p95: nil, p99: nil }
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
def self.flag_sql(predicate)
|
|
146
|
+
predicate ? "CASE WHEN #{predicate} THEN 1 ELSE 0 END" : "0"
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
# The nearest-rank percentile: the duration whose rank is ceil(n * p/100).
|
|
150
|
+
RANK_SQL = { 50 => Arel.sql("MAX(CASE WHEN rn = (n * 50 + 99) / 100 THEN duration END)"),
|
|
151
|
+
95 => Arel.sql("MAX(CASE WHEN rn = (n * 95 + 99) / 100 THEN duration END)"),
|
|
152
|
+
99 => Arel.sql("MAX(CASE WHEN rn = (n * 99 + 99) / 100 THEN duration END)") }.freeze
|
|
153
|
+
|
|
154
|
+
def self.rank_sql(percent)
|
|
155
|
+
RANK_SQL.fetch(percent)
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
private_class_method :rollup_series, :raw_series, :point, :empty_point, :flag_sql, :rank_sql
|
|
159
|
+
|
|
24
160
|
# Per-group table rows for a record type in the window: count/error totals,
|
|
25
161
|
# merged percentiles (via Telemetry::Rollup.summarize), and a sparkline.
|
|
26
162
|
def self.grouped(environment, record_type, from:, to:, limit: 100, order: nil, dir: nil)
|
|
@@ -34,6 +170,15 @@ module Railwatch
|
|
|
34
170
|
cached(key) { grouped_uncached(environment, record_type, from: from, to: to, limit: limit, order: order, sign: sign) }
|
|
35
171
|
end
|
|
36
172
|
|
|
173
|
+
# The minute cache exists because the platform's rollups only change
|
|
174
|
+
# when RollupJob writes. Embedded, every batch updates them and one
|
|
175
|
+
# person is looking, so the cache would only make the page a minute
|
|
176
|
+
# stale for nothing.
|
|
177
|
+
def self.cached(key, &block)
|
|
178
|
+
return yield if Railwatch.config.local?
|
|
179
|
+
Rails.cache.fetch(key, expires_in: 1.minute, &block)
|
|
180
|
+
end
|
|
181
|
+
|
|
37
182
|
def self.grouped_uncached(environment, record_type, from:, to:, limit:, order:, sign:)
|
|
38
183
|
environment.with_telemetry do
|
|
39
184
|
rows = Telemetry::Rollup.for_type(record_type).between(from, to).to_a
|
|
@@ -60,16 +205,6 @@ module Railwatch
|
|
|
60
205
|
end
|
|
61
206
|
end
|
|
62
207
|
|
|
63
|
-
# The minute cache exists because the platform's rollups only change
|
|
64
|
-
# when RollupJob writes. Embedded, every batch updates them and one
|
|
65
|
-
# person is looking, so the cache would only make the page a minute
|
|
66
|
-
# stale for nothing.
|
|
67
|
-
def self.cached(key, &block)
|
|
68
|
-
return yield if Railwatch.config.local?
|
|
69
|
-
|
|
70
|
-
Rails.cache.fetch(key, expires_in: 1.minute, &block)
|
|
71
|
-
end
|
|
72
|
-
|
|
73
208
|
# The digest-free half of Rollup.summarize: everything the sort keys
|
|
74
209
|
# that are not percentiles need.
|
|
75
210
|
def self.cheap_summary(rows)
|
|
@@ -9,6 +9,13 @@ module Railwatch
|
|
|
9
9
|
|
|
10
10
|
scope :unhandled, -> { where(handled: false) }
|
|
11
11
|
|
|
12
|
+
def as_row
|
|
13
|
+
{ id: id, class_name: class_name, message: message.first(500), handled: handled, severity: severity, source: source,
|
|
14
|
+
file: file, line: line, occurred_at: occurred_at, deploy: deploy, execution_id: execution_id,
|
|
15
|
+
execution_source: execution_source, execution_preview: execution_preview, user_ref: user_ref, tenant: app_tenant,
|
|
16
|
+
group_hash: group_hash }
|
|
17
|
+
end
|
|
18
|
+
|
|
12
19
|
def timeline_label
|
|
13
20
|
"#{class_name}: #{message.to_s.first(100)}"
|
|
14
21
|
end
|
|
@@ -63,6 +63,24 @@ module Railwatch
|
|
|
63
63
|
duration / 1000.0
|
|
64
64
|
end
|
|
65
65
|
|
|
66
|
+
def queue_latency_ms
|
|
67
|
+
queue_latency && (queue_latency / 1000.0).round(1)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# The row every list surface shows for an execution: what the kind has
|
|
71
|
+
# in common, then what it adds.
|
|
72
|
+
def as_row
|
|
73
|
+
row = { execution_id: execution_id, kind: kind, name: name, status: status, outcome: outcome, duration: duration_ms.round(2),
|
|
74
|
+
occurred_at: occurred_at, deploy: deploy, server: server, user_ref: user_ref, tenant: app_tenant,
|
|
75
|
+
exception_preview: exception_preview }
|
|
76
|
+
case kind
|
|
77
|
+
when "request" then row.merge(method: self[:method], inertia_component: inertia_component, queries: counters["queries"])
|
|
78
|
+
when "job_attempt" then row.merge(queue: queue, attempt: attempt, job_id: job_id, queue_latency: queue_latency_ms)
|
|
79
|
+
when "scheduled_task" then row.merge(task_key: task_key)
|
|
80
|
+
else row
|
|
81
|
+
end
|
|
82
|
+
end
|
|
83
|
+
|
|
66
84
|
def children_count
|
|
67
85
|
counters.values.sum
|
|
68
86
|
end
|