railwatch 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (148) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +26 -0
  3. data/app/controllers/railwatch/commands_controller.rb +1 -1
  4. data/app/controllers/railwatch/environment_scoped.rb +18 -36
  5. data/app/controllers/railwatch/exceptions_controller.rb +8 -9
  6. data/app/controllers/railwatch/processes_controller.rb +1 -1
  7. data/app/controllers/railwatch/releases_controller.rb +2 -2
  8. data/app/controllers/railwatch/tenants_controller.rb +2 -2
  9. data/app/controllers/railwatch/thresholds_controller.rb +8 -1
  10. data/app/jobs/railwatch/detect_performance_issues_job.rb +12 -2
  11. data/app/models/railwatch/anomaly_rule.rb +20 -2
  12. data/app/models/railwatch/telemetry/aggregations.rb +154 -19
  13. data/app/models/railwatch/telemetry/exception.rb +7 -0
  14. data/app/models/railwatch/telemetry/execution.rb +18 -0
  15. data/app/models/railwatch/telemetry/log.rb +6 -0
  16. data/app/models/railwatch/telemetry/query.rb +7 -0
  17. data/app/models/railwatch/telemetry/release_health.rb +12 -6
  18. data/app/models/railwatch/telemetry/rollup.rb +17 -2
  19. data/app/models/railwatch/telemetry/tenant.rb +8 -23
  20. data/app/models/railwatch/telemetry_record.rb +8 -6
  21. data/app/models/railwatch/threshold.rb +41 -4
  22. data/app/models/railwatch/window.rb +66 -0
  23. data/docs/embedded.md +11 -0
  24. data/lib/railwatch/version.rb +1 -1
  25. data/public/railwatch/assets/app-layout-DDyQa72H.js +1 -0
  26. data/public/railwatch/assets/{app-wordmark-DDNuuq5_.js → app-wordmark-o9CODKP0.js} +1 -1
  27. data/public/railwatch/assets/{appearance-C3LMdsDy.js → appearance-BwuCXabr.js} +1 -1
  28. data/public/railwatch/assets/application-B7h1MIhi.css +1 -0
  29. data/public/railwatch/assets/{arrow-up-DYTCBENO.js → arrow-up-C6PxDiY3.js} +1 -1
  30. data/public/railwatch/assets/{auth-layout-CzPRawuI.js → auth-layout-BRt8MGFD.js} +1 -1
  31. data/public/railwatch/assets/{badge-s7YDycYg.js → badge-CAxXV8za.js} +1 -1
  32. data/public/railwatch/assets/{braces-BFrg-bvp.js → braces-DgomTCNf.js} +1 -1
  33. data/public/railwatch/assets/{card-DiW-JhK3.js → card-cAtqCxWl.js} +1 -1
  34. data/public/railwatch/assets/{chart-nsdZV0Dv.js → chart-BBeBkkNa.js} +1 -1
  35. data/public/railwatch/assets/{chart-hover-DMWx5-Rs.js → chart-hover-B1M9jc0y.js} +1 -1
  36. data/public/railwatch/assets/chart-panel-DUQTz_C8.js +1 -0
  37. data/public/railwatch/assets/{checkbox-C2AqEZXX.js → checkbox-CmhMHWZO.js} +1 -1
  38. data/public/railwatch/assets/{code-DVJtYP7h.js → code-DESvxyTj.js} +1 -1
  39. data/public/railwatch/assets/{copy-block-D1myf7is.js → copy-block-BkSU5832.js} +1 -1
  40. data/public/railwatch/assets/{copy-id-CyRCMfdH.js → copy-id-D03GhN9F.js} +1 -1
  41. data/public/railwatch/assets/{cursor-load-more-JfSVBn9C.js → cursor-load-more-CRyuMeQb.js} +1 -1
  42. data/public/railwatch/assets/{data-table-CwtPkw_i.js → data-table-BIlt7Rtm.js} +1 -1
  43. data/public/railwatch/assets/{edit-fB83T9Yh.js → edit-Bb6MKoe4.js} +1 -1
  44. data/public/railwatch/assets/{edit-BvTugIpj.js → edit-DJ0D0wHN.js} +1 -1
  45. data/public/railwatch/assets/{edit-DvpDUM-t.js → edit-O0NSBWxo.js} +1 -1
  46. data/public/railwatch/assets/{empty-state-CshhuRUU.js → empty-state-C38il627.js} +1 -1
  47. data/public/railwatch/assets/env-layout-REF7OM4q.js +1 -0
  48. data/public/railwatch/assets/{execution-path-CdEerUGH.js → execution-path-FYLq1TwC.js} +1 -1
  49. data/public/railwatch/assets/{filter-bar-BMHf8xBb.js → filter-bar-CYog9Alp.js} +1 -1
  50. data/public/railwatch/assets/{flamegraph-uYWk1Yov.js → flamegraph-DSs69foN.js} +1 -1
  51. data/public/railwatch/assets/{frames-B5ZECkvt.js → frames-BUi2J5Mk.js} +1 -1
  52. data/public/railwatch/assets/{google-sign-in-button-CAvsXYb4.js → google-sign-in-button-BQiIKFdd.js} +1 -1
  53. data/public/railwatch/assets/{index-X13K9BbO.js → index-8-hnAhOD.js} +1 -1
  54. data/public/railwatch/assets/{index-Czq2bi1U.js → index-BaR1U9An.js} +1 -1
  55. data/public/railwatch/assets/{index-iVNm3uzI.js → index-BeOh2t_S.js} +1 -1
  56. data/public/railwatch/assets/{index-vC9KU2bt.js → index-BiiyMcA0.js} +1 -1
  57. data/public/railwatch/assets/{index-B-qRsj0W.js → index-BoUBioBP.js} +1 -1
  58. data/public/railwatch/assets/{index-B4IRsZXK.js → index-C-PmdhXA.js} +1 -1
  59. data/public/railwatch/assets/{index-fA7h22qS.js → index-C3A_9imx.js} +1 -1
  60. data/public/railwatch/assets/{index-DcuyMiuG.js → index-C7OtLq_3.js} +1 -1
  61. data/public/railwatch/assets/{index-BMgA9NNc.js → index-CFFpnzIS.js} +1 -1
  62. data/public/railwatch/assets/{index-CAkCcAty.js → index-CFRLPs4J.js} +1 -1
  63. data/public/railwatch/assets/{index-DMcsT2ES.js → index-CGs4m_fa.js} +1 -1
  64. data/public/railwatch/assets/{index-DuGGTdIJ.js → index-CICUIFHL.js} +1 -1
  65. data/public/railwatch/assets/{index-DVUwgfqP.js → index-C_upSl_k.js} +1 -1
  66. data/public/railwatch/assets/{index-CaJvo0Oc.js → index-CiPo4Gob.js} +1 -1
  67. data/public/railwatch/assets/{index-BIom4V2g.js → index-CpkI015n.js} +1 -1
  68. data/public/railwatch/assets/{index-Zf9xXwq7.js → index-CrZ3vHDL.js} +1 -1
  69. data/public/railwatch/assets/{index-ClJtZh5Q.js → index-CsoN51vW.js} +1 -1
  70. data/public/railwatch/assets/{index-1-78ZewP.js → index-DDI_Zx5V.js} +1 -1
  71. data/public/railwatch/assets/index-DSvlZVWG.js +1 -0
  72. data/public/railwatch/assets/{index-C7bU7oUF.js → index-DW2CBbxU.js} +1 -1
  73. data/public/railwatch/assets/{index-DzoUj38Q.js → index-Dh4IRLFI.js} +1 -1
  74. data/public/railwatch/assets/{index-CdM3F3jz.js → index-DqTFTP8p.js} +1 -1
  75. data/public/railwatch/assets/index-DrcKVG2f.js +1 -0
  76. data/public/railwatch/assets/{index-BIZsdfdM.js → index-DtHmuB9Q.js} +1 -1
  77. data/public/railwatch/assets/{index-QxAKxnHK.js → index-DvjY3dPD.js} +1 -1
  78. data/public/railwatch/assets/{index-dKn1tKSX.js → index-FhUaPPab.js} +1 -1
  79. data/public/railwatch/assets/{index-DYrF-A96.js → index-JdCVBrw8.js} +1 -1
  80. data/public/railwatch/assets/{index-DWMKU6mG.js → index-QpTtwFwu.js} +1 -1
  81. data/public/railwatch/assets/{index-Bw_ByIps.js → index-ZOGOB8SA.js} +1 -1
  82. data/public/railwatch/assets/{index-D-O1LUPq.js → index-ZSZg9rtq.js} +1 -1
  83. data/public/railwatch/assets/{index-ClG89uia.js → index-r0tSIplE.js} +1 -1
  84. data/public/railwatch/assets/{index-DJIXGkfB.js → index-sTYvcbkh.js} +1 -1
  85. data/public/railwatch/assets/{index-n16aviHx.js → index-so4lRrRq.js} +1 -1
  86. data/public/railwatch/assets/{index-CRShH-fR.js → index-tpz-OGUP.js} +1 -1
  87. data/public/railwatch/assets/{index-BpnfuXi-.js → index-umIAl-pL.js} +1 -1
  88. data/public/railwatch/assets/{inertia-CUTKXjxg.js → inertia-DLew8ZNx.js} +2 -2
  89. data/public/railwatch/assets/{input-error-C8ycyeD0.js → input-error-cvM6_Jht.js} +1 -1
  90. data/public/railwatch/assets/{json-viewer-CJECqZ0l.js → json-viewer-D922McGi.js} +1 -1
  91. data/public/railwatch/assets/{klass-Br9dng39.js → klass-CrwICqN8.js} +1 -1
  92. data/public/railwatch/assets/{label-B32Ks8n7.js → label-GWl7I6sf.js} +1 -1
  93. data/public/railwatch/assets/{layout-5ZhRtywp.js → layout-0ZAnD3zl.js} +1 -1
  94. data/public/railwatch/assets/{live-dot-Bb1UYT9F.js → live-dot-D1n_BreY.js} +1 -1
  95. data/public/railwatch/assets/{nav-XUcwnBso.js → nav-DPxr1NNC.js} +1 -1
  96. data/public/railwatch/assets/{new-0xyEF6NL.js → new-Cdl6pqST.js} +1 -1
  97. data/public/railwatch/assets/{new-GIkDYLt8.js → new-D-ZzUK9a.js} +1 -1
  98. data/public/railwatch/assets/{new-CdvlEBjJ.js → new-DEVkYv-z.js} +1 -1
  99. data/public/railwatch/assets/{new-YBWOo9_V.js → new-DHAHDrN7.js} +1 -1
  100. data/public/railwatch/assets/{new-CV011uLi.js → new-Dz4lZf1L.js} +1 -1
  101. data/public/railwatch/assets/{new-D9-ORiNW.js → new-GMrRFurX.js} +1 -1
  102. data/public/railwatch/assets/{onboarding-BqEuP-Ml.js → onboarding-D1vwaHYT.js} +1 -1
  103. data/public/railwatch/assets/{origin-identity-DrJ4D2TX.js → origin-identity-Bk9yHWZ1.js} +1 -1
  104. data/public/railwatch/assets/{percentile-picker-_wllqYoJ.js → percentile-picker-DfSx9yJO.js} +1 -1
  105. data/public/railwatch/assets/{relative-time-CV7V8aSP.js → relative-time-CjIjb8Lg.js} +1 -1
  106. data/public/railwatch/assets/{release-health-DRNLZtT7.js → release-health-4b3tivEf.js} +1 -1
  107. data/public/railwatch/assets/{route-D9slUz_a.js → route-C_5BUtHK.js} +1 -1
  108. data/public/railwatch/assets/{segmented-Bj70Htrs.js → segmented-BgbT3wZa.js} +1 -1
  109. data/public/railwatch/assets/{select-DJBZX2ky.js → select-DmunxCKE.js} +1 -1
  110. data/public/railwatch/assets/{separator-UUo0bkyc.js → separator-BXzEdZ_8.js} +1 -1
  111. data/public/railwatch/assets/series-chart-Xf49v9cv.js +1 -0
  112. data/public/railwatch/assets/{show-BQtEcTeD.js → show-B2zLAW83.js} +1 -1
  113. data/public/railwatch/assets/{show-DR6u4CR7.js → show-BLpWUHWD.js} +1 -1
  114. data/public/railwatch/assets/{show-Pu3Iehw5.js → show-CAl7xcex.js} +1 -1
  115. data/public/railwatch/assets/{show-DoCg0ifP.js → show-DI8IhNUH.js} +1 -1
  116. data/public/railwatch/assets/{show-BT3tKCgP.js → show-DSP9Cq_C.js} +2 -2
  117. data/public/railwatch/assets/{show-C0uC3gnA.js → show-DXs4deaC.js} +1 -1
  118. data/public/railwatch/assets/{show-D7tP881O.js → show-DYskfl3-.js} +1 -1
  119. data/public/railwatch/assets/{show-BpyCRL-I.js → show-DcpTFiLi.js} +1 -1
  120. data/public/railwatch/assets/{show-yQipmite.js → show-Dily73Xk.js} +1 -1
  121. data/public/railwatch/assets/{show-COfP8yWt.js → show-DlRVS18-.js} +1 -1
  122. data/public/railwatch/assets/{show-ChDJZjoj.js → show-Dn-GwFZL.js} +1 -1
  123. data/public/railwatch/assets/{show-BoG7jcwk.js → show-DnR1Dnjd.js} +1 -1
  124. data/public/railwatch/assets/{show-DnBCvlnd.js → show-SHwZjXb7.js} +1 -1
  125. data/public/railwatch/assets/{show-CefJwag7.js → show-SvLOcPrx.js} +1 -1
  126. data/public/railwatch/assets/{show-D7IL22Hy.js → show-Y74rM0VT.js} +1 -1
  127. data/public/railwatch/assets/{show-TTCk2Met.js → show-mU38uGTg.js} +1 -1
  128. data/public/railwatch/assets/{show-GruxlCov.js → show-vQ4bndYD.js} +1 -1
  129. data/public/railwatch/assets/{sort-header-CqQQhaxg.js → sort-header-Dcq9bzmo.js} +1 -1
  130. data/public/railwatch/assets/{sparkline-cell-BwdBz0yv.js → sparkline-cell-BON3qQUB.js} +1 -1
  131. data/public/railwatch/assets/{stat-IghK3WjD.js → stat-DFEyFxkO.js} +1 -1
  132. data/public/railwatch/assets/{status-badge-XYUN9gUA.js → status-badge-BaUKP7Yo.js} +1 -1
  133. data/public/railwatch/assets/{tenant-path-SxnZbfU1.js → tenant-path-DPZPc985.js} +1 -1
  134. data/public/railwatch/assets/{text-link-NCL9N5D5.js → text-link-BO77t9Xk.js} +1 -1
  135. data/public/railwatch/assets/{textarea-DhI1dmQa.js → textarea-DTqrCiV0.js} +1 -1
  136. data/public/railwatch/assets/{timeline-CLAwsAlo.js → timeline-D5rJ0es2.js} +1 -1
  137. data/public/railwatch/assets/{transition-DpTb9Yiv.js → transition-DMIrZVth.js} +1 -1
  138. data/public/railwatch/assets/{use-clipboard-D5i6e5kf.js → use-clipboard-ByoUGQqA.js} +1 -1
  139. data/public/railwatch/assets/{use-live-V-slSdFJ.js → use-live-D7xKz2ma.js} +1 -1
  140. data/public/railwatch/manifest.json +1288 -1288
  141. metadata +117 -116
  142. data/public/railwatch/assets/app-layout-DUk54qUM.js +0 -1
  143. data/public/railwatch/assets/application-CwshqwM6.css +0 -1
  144. data/public/railwatch/assets/chart-panel-BlirnMrQ.js +0 -1
  145. data/public/railwatch/assets/env-layout-Dkp_4NvM.js +0 -1
  146. data/public/railwatch/assets/index-CZWzIE8F.js +0 -1
  147. data/public/railwatch/assets/index-DpWQMikK.js +0 -1
  148. data/public/railwatch/assets/series-chart-C1YrNlID.js +0 -1
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: e3cbe26df2ef93b1488b1cfb67d2cb15bb9c2930f63607651af4e5a9d4f526cb
4
- data.tar.gz: b62edbbb492f16ef449e4bc7ccc57ad90206f67c0e538c24232b30d931290339
3
+ metadata.gz: 538efe4f9ba268dd88e694a7052542638a86d64f34978bffd45e5edb9e71dd18
4
+ data.tar.gz: 3f5d5b22228381e4f9d6657544c44fb3e337c754c9b7a9d683dada967d1592af
5
5
  SHA512:
6
- metadata.gz: dc2578bf58e017c5f311d9642c9b51f52c7c203a84328d8bbe7845568c0836afe5bc7cbec5327aa286cd9e4bb378e930111cedce72b27337f2608e36a6305484
7
- data.tar.gz: 2e4f1467282564b4ef5d6d3f1446acb80f6cbfdf116588df899bb5a0cb3172330b817282d7d797d29d0cced2b7ab6aa1688d6fa266324d1634fcdafbb3cfe0df
6
+ metadata.gz: 923dc66f9d7e94815affc41d5d0df65db1891a6e46b9462ab6dd2931989f60786149fe70142de5f451a1b1d3ff13a8c551fb8b2487b8bd93aade6e95d0ce3600
7
+ data.tar.gz: 5f1fc6b7a9e2afade92562883b6467b1ace633c1dcd85ef2d6cde7c6b72f53328922f6e4c7e8c47eb596af2662dcb5b4d09103d061dd4a718484d512ff766913
data/CHANGELOG.md CHANGED
@@ -1,5 +1,31 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.5.0 (2026-09-21)
4
+
5
+ The telemetry layer the hosted platform runs is this gem's, not a copy. Everything
6
+ Railwatch Cloud had improved in its fork comes home, and the two seams the platform
7
+ needs are named.
8
+
9
+ - **Chart bucket widths.** `Telemetry::Aggregations` offers `STEPS` (1m to 1d),
10
+ `default_step`, and `steps_for` a window; `series` takes a `step:` and reads raw
11
+ rows with exact per-bucket percentiles under an hour, rollups at an hour and up;
12
+ `fill` zero-fills quiet buckets. The dashboard shows a step picker beside the
13
+ window picker, and every series page, release health, tenants, processes, and
14
+ exceptions draw at the chosen step. Latency lines break over an empty bucket
15
+ instead of dropping to zero.
16
+ - **One `Window`.** `Railwatch::Window` resolves `?window=` presets and custom
17
+ `?from=&to=` ranges, with the previous period for deltas; the controller concern
18
+ reads it instead of carrying the table.
19
+ - **Rollup extras.** `Telemetry::Rollup.summarize` merges each type's `extra`
20
+ (LLM spend and tokens, cache hits and misses) the way `absorb!` does.
21
+ - **LLM thresholds and anomaly rules.** `spend`, `tokens`, and `truncation_rate`
22
+ metrics on `llm_calls` and `llm_tools`; the form narrows metrics by kind and
23
+ shows the unit; `Threshold#format_value` prints dollars and percentages.
24
+ - **`as_row`** on Execution, Log, Exception, and Query: the one row shape every
25
+ list surface shows. The commands page reads the exit code from `status`.
26
+ - `Railwatch::TelemetryRecord` documents that the hosted platform tenants it, and
27
+ `docs/embedded.md` says what the platform is in terms of this engine.
28
+
3
29
  ## 0.4.0 (2026-09-21)
4
30
 
5
31
  The gem is the one author of what goes over the wire and what runs in the
@@ -5,7 +5,7 @@ module Railwatch
5
5
  def index
6
6
  rows = telemetry { Telemetry::Execution.commands.between(*window_range).recent.limit(200).to_a }
7
7
  render inertia: { commands: grouped("command", limit: 100),
8
- runs: rows.map { |r| { execution_id: r.execution_id, name: r.name, exit_code: r.status, duration: r.duration_ms.round(1), occurred_at: r.occurred_at, server: r.server, exception_preview: r.exception_preview } } }
8
+ runs: rows.map(&:as_row) }
9
9
  end
10
10
 
11
11
  def show
@@ -6,14 +6,13 @@ module Railwatch
6
6
  module EnvironmentScoped
7
7
  extend ActiveSupport::Concern
8
8
 
9
- WINDOWS = { "1h" => 1.hour, "6h" => 6.hours, "24h" => 24.hours, "7d" => 7.days, "30d" => 30.days }.freeze
10
- MAX_CUSTOM_RANGE = 90.days
11
-
12
9
  included do
13
10
  before_action :set_environment
14
11
  inertia_share environment: -> { environment_props }
15
- inertia_share window: -> { window_key }
16
- inertia_share range: -> { from, to = window_range; { from: from.iso8601(6), to: to.iso8601(6) } }
12
+ inertia_share window: -> { window.key }
13
+ inertia_share range: -> { window.to_h }
14
+ inertia_share step: -> { step_key }
15
+ inertia_share steps: -> { Telemetry::Aggregations.steps_for(*window_range) }
17
16
  inertia_share saved_views: -> { SavedView.props_for(environment, Viewer.user) }
18
17
  end
19
18
 
@@ -25,38 +24,19 @@ module Railwatch
25
24
  @environment = Environment.find(params[:environment_id] || Environment::ID)
26
25
  end
27
26
 
28
- def window_key
29
- custom_range ? "custom" : (WINDOWS.key?(params[:window].to_s) ? params[:window].to_s : "24h")
30
- end
31
-
32
- def window_range
33
- return @window_range if defined?(@window_range)
34
- @window_range = custom_range || begin
35
- key = WINDOWS.key?(params[:window].to_s) ? params[:window].to_s : "24h"
36
- to = Time.current
37
- [ to - WINDOWS[key], to ]
38
- end
27
+ def window
28
+ @window ||= Window.parse(window: params[:window], from: params[:from], to: params[:to])
39
29
  end
40
30
 
41
- def custom_range
42
- return @custom_range if defined?(@custom_range)
43
- from = parse_time(params[:from])
44
- to = from && (parse_time(params[:to]) || Time.current)
45
- valid = from && to && to > from && (to - from) <= MAX_CUSTOM_RANGE
46
- @custom_range = valid ? [ from, to ] : nil
47
- end
48
-
49
- def parse_time(value)
50
- return nil if value.blank?
51
- Time.zone.parse(value.to_s)
52
- rescue ArgumentError, TypeError
53
- nil
54
- end
31
+ def window_range = window.range
55
32
 
56
- def previous_window_range
33
+ # The chart bucket width: ?step= when it is one the window offers, else the
34
+ # window's default (a minute for an hour, an hour for a day, and so on).
35
+ def step_key
36
+ return @step_key if defined?(@step_key)
57
37
  from, to = window_range
58
- span = to - from
59
- [ from - span, from ]
38
+ offered = Telemetry::Aggregations.steps_for(from, to)
39
+ @step_key = offered.include?(params[:step].to_s) ? params[:step].to_s : Telemetry::Aggregations.default_step(from, to)
60
40
  end
61
41
 
62
42
  def telemetry(&block) = environment.with_telemetry(&block)
@@ -79,9 +59,11 @@ module Railwatch
79
59
  environment.deploys.between(*window_range).recent.limit(50).map { |d| { deploy: d.deploy, ref: d.short_ref, at: d.deployed_at } }
80
60
  end
81
61
 
82
- def series(record_type, group_hash: nil, name: nil)
62
+ # Time-bucketed series for charts, one point per step_key across the
63
+ # window: [{t, count, errors, client_errors, avg, p50, p95, p99}]
64
+ def series(record_type, group_hash: nil)
83
65
  from, to = window_range
84
- Telemetry::Aggregations.series(environment, record_type, from: from, to: to, group_hash: group_hash, name: name)
66
+ Telemetry::Aggregations.series(environment, record_type, from: from, to: to, group_hash: group_hash, step: step_key)
85
67
  end
86
68
 
87
69
  def grouped(record_type, limit: 100, order: nil, dir: nil)
@@ -91,7 +73,7 @@ module Railwatch
91
73
 
92
74
  def summary_with_delta(record_type, group_hash: nil)
93
75
  from, to = window_range
94
- previous_from, previous_to = previous_window_range
76
+ previous_from, previous_to = window.previous.range
95
77
  Telemetry::Aggregations.summary_with_delta(environment, record_type, from: from, to: to,
96
78
  previous_from: previous_from, previous_to: previous_to, group_hash: group_hash)
97
79
  end
@@ -38,16 +38,15 @@ module Railwatch
38
38
  # `errors` carries the unhandled count so the chart legend (handled /
39
39
  # unhandled) agrees with the Unhandled card above the table.
40
40
  def exception_series(from, to)
41
+ step = Telemetry::Aggregations::STEPS.fetch(step_key)
42
+ bucket = Telemetry::Aggregations.bucket_sql("occurred_at", step)
41
43
  buckets = Hash.new { |h, k| h[k] = { count: 0, errors: 0 } }
42
- Telemetry::Exception.between(from, to)
43
- .group(Arel.sql("strftime('%Y-%m-%dT%H:00:00Z', occurred_at)"), :handled)
44
- .count
45
- .each do |(bucket, handled), c|
46
- buckets[bucket][:count] += c
47
- buckets[bucket][:errors] += c unless handled
48
- end
49
- buckets.map { |bucket, b| { t: bucket, count: b[:count], errors: b[:errors], client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0 } }
50
- .sort_by { |r| r[:t] }
44
+ Telemetry::Exception.between(from, to).group(bucket, :handled).count.each do |(b, handled), c|
45
+ buckets[b][:count] += c
46
+ buckets[b][:errors] += c unless handled
47
+ end
48
+ points = buckets.map { |b, c| { t: Time.at(b).utc, count: c[:count], errors: c[:errors], client_errors: 0, avg: nil, p50: nil, p95: nil, p99: nil } }
49
+ Telemetry::Aggregations.fill(points, from, to, step)
51
50
  end
52
51
  end
53
52
  end
@@ -6,7 +6,7 @@ module Railwatch
6
6
  live_since = Telemetry::HealthSample::LIVE_WINDOW.ago
7
7
  samples, series, queues, live_servers, processes = telemetry do
8
8
  live = Telemetry::HealthSample.live(live_since).to_a
9
- [ live, Telemetry::HealthSample.series(*window_range), Telemetry::HealthSample.queue_depths(live),
9
+ [ live, Telemetry::HealthSample.series(*window_range, bucket: Telemetry::Aggregations::STEPS.fetch(step_key)), Telemetry::HealthSample.queue_depths(live),
10
10
  live.map(&:server) | Telemetry::Execution.servers_since(live_since),
11
11
  Telemetry::Process.recent.limit(200).to_a ]
12
12
  end
@@ -17,7 +17,7 @@ module Railwatch
17
17
  new_issues = new_issue_counts(deploys, to)
18
18
  summary, series, adoption, health = telemetry do
19
19
  [ Telemetry::ReleaseHealth.for_deploy(nil, from, to),
20
- Telemetry::ReleaseHealth.series(nil, from, to),
20
+ Telemetry::ReleaseHealth.series(nil, from, to, step: step_key),
21
21
  Telemetry::ReleaseHealth.adoption(from, to),
22
22
  deploys.to_h { |d| [ d.deploy, Telemetry::ReleaseHealth.for_deploy(d.deploy, from, to) ] } ]
23
23
  end
@@ -37,7 +37,7 @@ module Railwatch
37
37
  data = telemetry do
38
38
  { summary: Telemetry::ReleaseHealth.for_deploy(release, from, to),
39
39
  previous_summary: previous && Telemetry::ReleaseHealth.for_deploy(previous.deploy, *previous.window),
40
- series: Telemetry::ReleaseHealth.series(release, from, to),
40
+ series: Telemetry::ReleaseHealth.series(release, from, to, step: step_key),
41
41
  sessions: Telemetry::Session.where(deploy: release).recent.limit(SESSIONS_SHOWN).map { |s| session_row(s) } }
42
42
  end
43
43
  render inertia: data.merge(
@@ -20,8 +20,8 @@ module Railwatch
20
20
  from, to = window_range
21
21
  data = telemetry do
22
22
  { tenant: tenant, summary: { current: Telemetry::Tenant.summary(tenant, from, to),
23
- previous: Telemetry::Tenant.summary(tenant, *previous_window_range) },
24
- series: Telemetry::Tenant.series(tenant, from, to), routes: Telemetry::Tenant.routes(tenant, from, to),
23
+ previous: Telemetry::Tenant.summary(tenant, *window.previous.range) },
24
+ series: Telemetry::Tenant.series(tenant, from, to, step: step_key), routes: Telemetry::Tenant.routes(tenant, from, to),
25
25
  jobs: Telemetry::Tenant.job_classes(tenant, from, to), exceptions: Telemetry::Tenant.exceptions(tenant, from, to),
26
26
  people: Telemetry::Tenant.people(tenant), recent_requests: Telemetry::Tenant.recent_requests(tenant, from, to) }
27
27
  end
@@ -8,7 +8,14 @@ module Railwatch
8
8
  jobs: telemetry { Telemetry::Rollup.for_type("job_attempt").where("bucket > ?", 7.days.ago).distinct.pluck(:name).sort },
9
9
  kinds: Threshold::TARGET_KINDS, metrics: Threshold::METRICS,
10
10
  anomaly_rules: environment.anomaly_rules.order(:target_kind, :target).map { |r| r.slice(:id, :target_kind, :target, :metric, :deviation, :window_minutes, :baseline_days, :enabled, :last_fired_at).merge(description: r.description) },
11
- anomaly_target_kinds: AnomalyRule::TARGET_KINDS, anomaly_metrics: AnomalyRule::METRICS }
11
+ # Spend and tokens only exist on llm_calls, so the form
12
+ # narrows with the kind rather than offering a rule that
13
+ # could never fire.
14
+ metrics_for_kind: Threshold::TARGET_KINDS.index_with { |k| Threshold.metrics_for(k) },
15
+ units: Threshold::UNITS,
16
+ llm_models: telemetry { Telemetry::Rollup.for_type("llm_call").where("bucket > ?", 7.days.ago).distinct.pluck(:name).compact.sort },
17
+ anomaly_target_kinds: AnomalyRule::TARGET_KINDS, anomaly_metrics: AnomalyRule::METRICS,
18
+ anomaly_metrics_for_kind: AnomalyRule::TARGET_KINDS.index_with { |k| AnomalyRule.metrics_for(k) } }
12
19
  end
13
20
 
14
21
  def create
@@ -12,7 +12,8 @@ module Railwatch
12
12
  MAX_ROLLUP_ROWS_PER_RULE = MAX_GROUPS_PER_RULE * 25
13
13
 
14
14
  TYPE_FOR = { "requests" => "request", "jobs" => "job_attempt", "commands" => "command", "queries" => "query",
15
- "scheduled_tasks" => "scheduled_task", "outgoing_requests" => "outgoing_request" }.freeze
15
+ "scheduled_tasks" => "scheduled_task", "outgoing_requests" => "outgoing_request",
16
+ "llm_calls" => "llm_call", "llm_tools" => "llm_tool" }.freeze
16
17
 
17
18
  def perform(environment)
18
19
  now = Time.current
@@ -51,19 +52,28 @@ module Railwatch
51
52
  .transform_values { |group_rows| [ group_rows.first.name, Telemetry::Rollup.summarize(group_rows) ] }
52
53
  end
53
54
 
55
+ NANOS_PER_DOLLAR = 1_000_000_000.0
56
+
54
57
  def value_for(metric, s)
58
+ extra = s[:extra] || {}
55
59
  case metric
56
60
  when "p95" then s[:p95] / 1000.0
57
61
  when "max" then s[:max] / 1000.0
58
62
  when "avg" then s[:avg] / 1000.0
59
63
  when "error_rate", "failure_rate" then s[:count].zero? ? nil : (s[:errors] * 100.0 / s[:count])
64
+ # Money and tokens are sums over the window, not percentiles of it.
65
+ when "spend" then extra["cost_nanos"].to_i / NANOS_PER_DOLLAR
66
+ when "tokens" then extra["input_tokens"].to_i + extra["output_tokens"].to_i
67
+ # A cut-off answer is a successful call, so it is invisible to
68
+ # error_rate. This is the only way to be told about it.
69
+ when "truncation_rate" then s[:count].zero? ? nil : (extra["truncated"].to_i * 100.0 / s[:count])
60
70
  end
61
71
  end
62
72
 
63
73
  def open_issue(environment, threshold, group_hash, name, value, from, now)
64
74
  issue, outcome = Issue.record_occurrence!(
65
75
  environment: environment, group_hash: "perf:#{threshold.id}:#{group_hash}", kind: "performance",
66
- title: "#{name} exceeded #{threshold.metric} #{threshold.limit}#{threshold.metric.end_with?('rate') ? '%' : 'ms'} (#{value.round(1)})",
76
+ title: "#{name} exceeded #{threshold.metric} #{threshold.format_value(threshold.limit)} (#{threshold.format_value(value.round(2))})",
67
77
  culprit: name, occurred_at: now, deploy: nil, user_ref: nil,
68
78
  sample: { threshold_id: threshold.id, value: value.round(2), metric: threshold.metric, limit: threshold.limit })
69
79
  carry_detection(issue, outcome) do
@@ -10,13 +10,22 @@ module Railwatch
10
10
  MAX_BASELINE_DAYS = 30
11
11
  MAX_DEVIATION = 10
12
12
  MAX_WINDOW_MINUTES = 1_440
13
- TARGET_KINDS = %w[requests jobs queries outgoing_requests scheduled_tasks].freeze
14
- METRICS = %w[p95 avg count error_rate].freeze
13
+ TARGET_KINDS = %w[requests jobs queries outgoing_requests scheduled_tasks llm_calls].freeze
14
+ # spend included because "we are spending well above baseline for this
15
+ # hour of the week" is the LLM alert that needs no threshold picked.
16
+ METRICS = %w[p95 avg count error_rate spend].freeze
17
+ SHARED_METRICS = %w[p95 avg count error_rate].freeze
18
+ LLM_METRICS = %w[spend].freeze
19
+
20
+ def self.metrics_for(target_kind)
21
+ target_kind == "llm_calls" ? SHARED_METRICS + LLM_METRICS : SHARED_METRICS
22
+ end
15
23
 
16
24
  def environment = Environment.current
17
25
 
18
26
  validates :target_kind, inclusion: { in: TARGET_KINDS }
19
27
  validates :metric, inclusion: { in: METRICS }
28
+ validate :metric_applies_to_target_kind
20
29
  validates :target, presence: true, length: { maximum: 256 }
21
30
  validates :deviation, numericality: { greater_than: 0, less_than_or_equal_to: MAX_DEVIATION }
22
31
  validates :window_minutes, numericality: { only_integer: true, greater_than: 0,
@@ -29,5 +38,14 @@ module Railwatch
29
38
  def description
30
39
  "#{target_kind} #{target == '*' ? 'all' : target}: #{metric} > #{deviation}σ above #{baseline_days}-day baseline"
31
40
  end
41
+
42
+ private
43
+
44
+ def metric_applies_to_target_kind
45
+ return if metric.blank? || target_kind.blank?
46
+ return if AnomalyRule.metrics_for(target_kind).include?(metric)
47
+
48
+ errors.add(:metric, "is not available for #{target_kind}")
49
+ end
32
50
  end
33
51
  end
@@ -9,18 +9,154 @@ module Railwatch
9
9
  class Aggregations
10
10
  GROUPED_SORT_FIELDS = %i[count p50 p95 p99 errors avg max].freeze
11
11
 
12
- # Time-bucketed series from rollups for charts: [{t, count, errors, p50, p95, p99}]
13
- def self.series(environment, record_type, from:, to:, group_hash: nil, name: nil)
14
- environment.with_telemetry do
15
- scope = Telemetry::Rollup.for_type(record_type).between(from, to)
16
- scope = scope.where(group_hash: group_hash) if group_hash
17
- scope = scope.where(name: name) if name
18
- scope.group(:bucket).order(:bucket)
19
- .pluck(:bucket, Arel.sql("SUM(count)"), Arel.sql("SUM(error_count)"), Arel.sql("SUM(client_error_count)"), Arel.sql("SUM(duration_sum)"), Arel.sql("MAX(p50)"), Arel.sql("MAX(p95)"), Arel.sql("MAX(p99)"))
20
- .map { |b, c, e, ce, ds, p50, p95, p99| { t: b, count: c, errors: e, client_errors: ce, avg: c.zero? ? 0 : ds / c / 1000.0, p50: p50 / 1000.0, p95: p95 / 1000.0, p99: p99 / 1000.0 } }
12
+ # Bucket widths a chart can be drawn at. Anything under an hour is read
13
+ # from the raw tables (rollups are hourly); an hour and up groups rollups.
14
+ STEPS = { "1m" => 1.minute, "5m" => 5.minutes, "15m" => 15.minutes, "1h" => 1.hour, "6h" => 6.hours, "1d" => 1.day }.freeze
15
+ # A step is offered for a window when it draws at least two points and at
16
+ # most this many: more bars than pixels is mush, and a sub-hour step over
17
+ # a long window is also a long raw scan.
18
+ MAX_POINTS = 400
19
+
20
+ # The bucket a window is drawn at unless the page asks for another one.
21
+ # Sub-hour steps scan raw rows, so they are the default only where that
22
+ # scan is a few thousand rows; a day and beyond stays on rollups.
23
+ def self.default_step(from, to)
24
+ span = to - from
25
+ if span <= 2.hours then "1m"
26
+ elsif span <= 12.hours then "5m"
27
+ elsif span <= 7.days then "1h"
28
+ else "6h"
29
+ end
30
+ end
31
+
32
+ # STEPS keys that make sense for the window, finest first.
33
+ def self.steps_for(from, to)
34
+ span = to - from
35
+ STEPS.select { |_key, step| (span / step).between?(2, MAX_POINTS) }.keys
36
+ end
37
+
38
+ # Time-bucketed series for charts, one point per `step` from `from` to `to`
39
+ # with empty buckets filled in: [{t, count, errors, client_errors, avg,
40
+ # p50, p95, p99}]. Durations in milliseconds; an empty bucket carries nil
41
+ # for them so a latency line breaks instead of dropping to zero.
42
+ def self.series(environment, record_type, from:, to:, group_hash: nil, step: nil, tenant: nil)
43
+ environment.with_telemetry { points(record_type, from: from, to: to, group_hash: group_hash, step: step, tenant: tenant) }
44
+ end
45
+
46
+ # `series` for a caller already inside environment.with_telemetry. A
47
+ # tenant narrows to one app_tenant, which rollups do not carry, so a
48
+ # tenant series reads raw rows at every step.
49
+ def self.points(record_type, from:, to:, group_hash: nil, step: nil, tenant: nil)
50
+ step = STEPS.fetch(step || default_step(from, to))
51
+ raw = step < 1.hour || tenant
52
+ points = raw ? raw_series(record_type, from, to, group_hash, step, tenant) : rollup_series(record_type, from, to, group_hash, step)
53
+ fill(points, from, to, step)
54
+ end
55
+
56
+ # Sums hourly rollups into `step`-wide buckets. Percentiles are the
57
+ # worst hour's, the same reading series always gave across groups.
58
+ def self.rollup_series(record_type, from, to, group_hash, step)
59
+ bucket = bucket_sql("bucket", step)
60
+ scope = Telemetry::Rollup.for_type(record_type).between(from, to)
61
+ scope = scope.where(group_hash: group_hash) if group_hash
62
+ scope.group(bucket).order(bucket)
63
+ .pluck(bucket, Arel.sql("SUM(count)"), Arel.sql("SUM(error_count)"), Arel.sql("SUM(client_error_count)"), Arel.sql("SUM(duration_sum)"), Arel.sql("MAX(p50)"), Arel.sql("MAX(p95)"), Arel.sql("MAX(p99)"))
64
+ .map { |b, c, e, ce, ds, p50, p95, p99| point(b, c, e, ce, ds, p50, p95, p99) }
65
+ end
66
+
67
+ # What a raw row of each record type is, and when it counts as an error
68
+ # (or a client error), mirroring RollupJob#error?. Both readings of the
69
+ # same rows must agree, which spec/models/telemetry/aggregations_spec.rb
70
+ # checks for every type.
71
+ RAW = {
72
+ "request" => [ -> { Telemetry::Execution.requests }, "status >= 500", "status BETWEEN 400 AND 499" ],
73
+ "job_attempt" => [ -> { Telemetry::Execution.jobs }, "outcome = 'failed'" ],
74
+ "scheduled_task" => [ -> { Telemetry::Execution.scheduled }, "outcome = 'failed'" ],
75
+ "command" => [ -> { Telemetry::Execution.commands }, "status != 0" ],
76
+ "channel_action" => [ -> { Telemetry::Execution.channels }, "outcome = 'failed'" ],
77
+ "query" => [ -> { Telemetry::Query.all } ],
78
+ "outgoing_request" => [ -> { Telemetry::OutgoingRequest.all }, "status_code >= 500 OR status_code IS NULL OR status_code = 0" ],
79
+ "cache_event" => [ -> { Telemetry::CacheEvent.all } ],
80
+ "mail" => [ -> { Telemetry::Mail.all }, "failed" ],
81
+ "visit" => [ -> { Telemetry::Visit.all }, "status = 'error'" ],
82
+ "span" => [ -> { Telemetry::Span.all }, "status = 'failed'" ],
83
+ "notification" => [ -> { Telemetry::Notification.all }, "failed" ],
84
+ "view_render" => [ -> { Telemetry::ViewRender.all } ],
85
+ "transaction" => [ -> { Telemetry::Transaction.all }, "outcome = 'rollback'" ],
86
+ "llm_call" => [ -> { Telemetry::LlmCall.models }, "status = 'failed'" ],
87
+ "llm_tool" => [ -> { Telemetry::LlmCall.tools }, "status = 'failed'" ]
88
+ }.freeze
89
+
90
+ # Sub-hour buckets straight from the raw rows, in one statement:
91
+ # counts and sums per bucket, and exact nearest-rank percentiles from a
92
+ # ROW_NUMBER over each bucket's durations. Rollups are hourly, so this
93
+ # is the only reading finer than an hour; it is also the freshest one,
94
+ # since the current hour's rollup is up to a minute behind.
95
+ def self.raw_series(record_type, from, to, group_hash, step, tenant = nil)
96
+ scope, error_sql, client_error_sql = RAW.fetch(record_type)
97
+ rows = scope.call.where(occurred_at: from..to)
98
+ rows = rows.where(group_hash: group_hash) if group_hash
99
+ rows = rows.where(app_tenant: tenant) if tenant
100
+ model = rows.model
101
+ inner = rows.select(bucket_sql("occurred_at", step).to_s + " AS b", "duration", flag_sql(error_sql) + " AS err", flag_sql(client_error_sql) + " AS cerr")
102
+ ranked = model.unscoped.from(inner, "r").select("b", "duration", "err", "cerr",
103
+ "ROW_NUMBER() OVER (PARTITION BY b ORDER BY duration) AS rn", "COUNT(*) OVER (PARTITION BY b) AS n")
104
+ model.unscoped.from(ranked, "w").group("b").order("b")
105
+ .pluck(Arel.sql("b"), Arel.sql("COUNT(*)"), Arel.sql("SUM(err)"), Arel.sql("SUM(cerr)"), Arel.sql("SUM(duration)"),
106
+ rank_sql(50), rank_sql(95), rank_sql(99))
107
+ .map { |b, c, e, ce, ds, p50, p95, p99| point(b, c, e, ce, ds, p50, p95, p99) }
108
+ end
109
+
110
+ # The timestamp columns a series buckets on: raw rows by occurred_at,
111
+ # rollups and release health by their hourly bucket.
112
+ BUCKET_COLUMNS = %w[occurred_at bucket].freeze
113
+
114
+ # Epoch seconds of the start of the `step`-wide bucket holding `column`.
115
+ # `column` must be one of BUCKET_COLUMNS and `step` is a Duration, so the
116
+ # fragment can only ever hold a listed column name and an integer literal.
117
+ def self.bucket_sql(column, step)
118
+ raise ArgumentError, "unknown bucket column #{column.inspect}" unless BUCKET_COLUMNS.include?(column)
119
+ seconds = Integer(step.to_i)
120
+ Arel.sql("(strftime('%s', #{column}) / #{seconds}) * #{seconds}")
121
+ end
122
+
123
+ # One point per bucket from `from` to `to`, keeping the computed ones and
124
+ # zero-filling the rest, so a quiet minute is a gap of the right width.
125
+ # A series with its own point shape passes a block building its empty one.
126
+ def self.fill(points, from, to, step)
127
+ by_bucket = points.index_by { |p| p[:t] }
128
+ first = Time.at((from.to_i / step.to_i) * step.to_i).utc
129
+ last = Time.at((to.to_i / step.to_i) * step.to_i).utc
130
+ (first.to_i..last.to_i).step(step.to_i).map do |t|
131
+ at = Time.at(t).utc
132
+ by_bucket[at] || (block_given? ? yield(at) : empty_point(at))
21
133
  end
22
134
  end
23
135
 
136
+ def self.point(bucket, count, errors, client_errors, duration_sum, p50, p95, p99)
137
+ { t: Time.at(bucket).utc, count: count, errors: errors, client_errors: client_errors,
138
+ avg: count.zero? ? 0 : duration_sum / count / 1000.0, p50: p50 / 1000.0, p95: p95 / 1000.0, p99: p99 / 1000.0 }
139
+ end
140
+
141
+ def self.empty_point(t)
142
+ { t: t, count: 0, errors: 0, client_errors: 0, avg: nil, p50: nil, p95: nil, p99: nil }
143
+ end
144
+
145
+ def self.flag_sql(predicate)
146
+ predicate ? "CASE WHEN #{predicate} THEN 1 ELSE 0 END" : "0"
147
+ end
148
+
149
+ # The nearest-rank percentile: the duration whose rank is ceil(n * p/100).
150
+ RANK_SQL = { 50 => Arel.sql("MAX(CASE WHEN rn = (n * 50 + 99) / 100 THEN duration END)"),
151
+ 95 => Arel.sql("MAX(CASE WHEN rn = (n * 95 + 99) / 100 THEN duration END)"),
152
+ 99 => Arel.sql("MAX(CASE WHEN rn = (n * 99 + 99) / 100 THEN duration END)") }.freeze
153
+
154
+ def self.rank_sql(percent)
155
+ RANK_SQL.fetch(percent)
156
+ end
157
+
158
+ private_class_method :rollup_series, :raw_series, :point, :empty_point, :flag_sql, :rank_sql
159
+
24
160
  # Per-group table rows for a record type in the window: count/error totals,
25
161
  # merged percentiles (via Telemetry::Rollup.summarize), and a sparkline.
26
162
  def self.grouped(environment, record_type, from:, to:, limit: 100, order: nil, dir: nil)
@@ -34,6 +170,15 @@ module Railwatch
34
170
  cached(key) { grouped_uncached(environment, record_type, from: from, to: to, limit: limit, order: order, sign: sign) }
35
171
  end
36
172
 
173
+ # The minute cache exists because the platform's rollups only change
174
+ # when RollupJob writes. Embedded, every batch updates them and one
175
+ # person is looking, so the cache would only make the page a minute
176
+ # stale for nothing.
177
+ def self.cached(key, &block)
178
+ return yield if Railwatch.config.local?
179
+ Rails.cache.fetch(key, expires_in: 1.minute, &block)
180
+ end
181
+
37
182
  def self.grouped_uncached(environment, record_type, from:, to:, limit:, order:, sign:)
38
183
  environment.with_telemetry do
39
184
  rows = Telemetry::Rollup.for_type(record_type).between(from, to).to_a
@@ -60,16 +205,6 @@ module Railwatch
60
205
  end
61
206
  end
62
207
 
63
- # The minute cache exists because the platform's rollups only change
64
- # when RollupJob writes. Embedded, every batch updates them and one
65
- # person is looking, so the cache would only make the page a minute
66
- # stale for nothing.
67
- def self.cached(key, &block)
68
- return yield if Railwatch.config.local?
69
-
70
- Rails.cache.fetch(key, expires_in: 1.minute, &block)
71
- end
72
-
73
208
  # The digest-free half of Rollup.summarize: everything the sort keys
74
209
  # that are not percentiles need.
75
210
  def self.cheap_summary(rows)
@@ -9,6 +9,13 @@ module Railwatch
9
9
 
10
10
  scope :unhandled, -> { where(handled: false) }
11
11
 
12
+ def as_row
13
+ { id: id, class_name: class_name, message: message.first(500), handled: handled, severity: severity, source: source,
14
+ file: file, line: line, occurred_at: occurred_at, deploy: deploy, execution_id: execution_id,
15
+ execution_source: execution_source, execution_preview: execution_preview, user_ref: user_ref, tenant: app_tenant,
16
+ group_hash: group_hash }
17
+ end
18
+
12
19
  def timeline_label
13
20
  "#{class_name}: #{message.to_s.first(100)}"
14
21
  end
@@ -63,6 +63,24 @@ module Railwatch
63
63
  duration / 1000.0
64
64
  end
65
65
 
66
+ def queue_latency_ms
67
+ queue_latency && (queue_latency / 1000.0).round(1)
68
+ end
69
+
70
+ # The row every list surface shows for an execution: what the kind has
71
+ # in common, then what it adds.
72
+ def as_row
73
+ row = { execution_id: execution_id, kind: kind, name: name, status: status, outcome: outcome, duration: duration_ms.round(2),
74
+ occurred_at: occurred_at, deploy: deploy, server: server, user_ref: user_ref, tenant: app_tenant,
75
+ exception_preview: exception_preview }
76
+ case kind
77
+ when "request" then row.merge(method: self[:method], inertia_component: inertia_component, queries: counters["queries"])
78
+ when "job_attempt" then row.merge(queue: queue, attempt: attempt, job_id: job_id, queue_latency: queue_latency_ms)
79
+ when "scheduled_task" then row.merge(task_key: task_key)
80
+ else row
81
+ end
82
+ end
83
+
66
84
  def children_count
67
85
  counters.values.sum
68
86
  end
@@ -77,6 +77,12 @@ module Railwatch
77
77
  def quote(term) = %("#{term.gsub('"', '""')}")
78
78
  end
79
79
 
80
+ def as_row
81
+ { id: id, level: level, message: message.first(2000), tags: tags, occurred_at: occurred_at, deploy: deploy,
82
+ execution_id: execution_id, execution_source: execution_source, execution_preview: execution_preview,
83
+ source: source, tenant: app_tenant, user_ref: user_ref, context: context }
84
+ end
85
+
80
86
  def timeline_label
81
87
  message.to_s.first(120)
82
88
  end
@@ -42,6 +42,13 @@ module Railwatch
42
42
  text
43
43
  end
44
44
 
45
+ def as_row
46
+ { id: id, group_hash: group_hash, sql: sql.first(2_000), name: name, duration: duration_ms.round(3), occurred_at: occurred_at,
47
+ deploy: deploy, execution_id: execution_id, execution_preview: execution_preview, source: source,
48
+ connection: connection, role: role, adapter: adapter, row_count: row_count, tenant: app_tenant, user_ref: user_ref,
49
+ explain: explain.present? }
50
+ end
51
+
45
52
  def timeline_label
46
53
  sql.to_s.first(120)
47
54
  end
@@ -36,13 +36,19 @@ module Railwatch
36
36
  }
37
37
  end
38
38
 
39
- # Hourly points for the stacked sessions-by-status chart.
40
- def self.series(deploy, from, to)
41
- between(from, to).for_release(deploy).group(:bucket).order(:bucket)
42
- .pluck(:bucket, Arel.sql("SUM(sessions)"), Arel.sql("SUM(sessions_errored)"), Arel.sql("SUM(sessions_crashed)"))
43
- .map { |bucket, sessions, errored, crashed|
44
- { t: bucket, sessions: sessions, ok: sessions - errored - crashed, errored: errored, crashed: crashed }
39
+ # Points for the stacked sessions-by-status chart, one per `step` across
40
+ # the window. The table is hourly and a session is counted once per hour
41
+ # it was seen, so a step under an hour is drawn at an hour: raw session
42
+ # rows would count a live session on every flush.
43
+ def self.series(deploy, from, to, step: nil)
44
+ step = [ Telemetry::Aggregations::STEPS.fetch(step || Telemetry::Aggregations.default_step(from, to)), 1.hour ].max
45
+ bucket = Telemetry::Aggregations.bucket_sql("bucket", step)
46
+ points = between(from, to).for_release(deploy).group(bucket).order(bucket)
47
+ .pluck(bucket, Arel.sql("SUM(sessions)"), Arel.sql("SUM(sessions_errored)"), Arel.sql("SUM(sessions_crashed)"))
48
+ .map { |b, sessions, errored, crashed|
49
+ { t: Time.at(b).utc, sessions: sessions, ok: sessions - errored - crashed, errored: errored, crashed: crashed }
45
50
  }
51
+ Telemetry::Aggregations.fill(points, from, to, step) { |t| { t: t, sessions: 0, ok: 0, errored: 0, crashed: 0 } }
46
52
  end
47
53
 
48
54
  # Each release's share of the window's sessions, as a percentage -- how
@@ -54,7 +54,7 @@ module Railwatch
54
54
  # Aggregate a relation of rollups into one summary with merged percentiles.
55
55
  def self.summarize(relation)
56
56
  rows = relation.to_a
57
- return { count: 0, errors: 0, client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0, max: 0 } if rows.empty?
57
+ return { count: 0, errors: 0, client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0, max: 0, extra: {} } if rows.empty?
58
58
  # merge! pushes the row's centroids into one accumulating digest.
59
59
  # `+` built a brand-new digest from both operands' centroids on every
60
60
  # row, so merging N rows re-pushed every earlier centroid N times:
@@ -70,9 +70,24 @@ module Railwatch
70
70
  p50: merged.percentile(0.5).to_i,
71
71
  p95: merged.percentile(0.95).to_i,
72
72
  p99: merged.percentile(0.99).to_i,
73
- max: rows.map(&:duration_max).max
73
+ max: rows.map(&:duration_max).max,
74
+ # Everything a type puts in `extra` -- llm_call's cost_nanos and
75
+ # token counts, cache_event's hits and misses -- merged the way
76
+ # absorb! merges it: numbers add up, anything else is last-wins.
77
+ # Spend is a sum over a window rather than a percentile of
78
+ # durations, so a rule about money has nowhere else to read from.
79
+ extra: merge_extras(rows)
74
80
  }
75
81
  end
82
+
83
+ def self.merge_extras(rows)
84
+ rows.each_with_object({}) do |row, out|
85
+ row.extra.each do |key, value|
86
+ existing = out[key]
87
+ out[key] = existing.is_a?(Numeric) && value.is_a?(Numeric) ? existing + value : value
88
+ end
89
+ end
90
+ end
76
91
  end
77
92
  end
78
93
  end