railwatch 0.3.7 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (156) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +58 -0
  3. data/app/controllers/railwatch/commands_controller.rb +1 -1
  4. data/app/controllers/railwatch/environment_scoped.rb +18 -36
  5. data/app/controllers/railwatch/exceptions_controller.rb +8 -9
  6. data/app/controllers/railwatch/processes_controller.rb +1 -1
  7. data/app/controllers/railwatch/releases_controller.rb +2 -2
  8. data/app/controllers/railwatch/tenants_controller.rb +2 -2
  9. data/app/controllers/railwatch/thresholds_controller.rb +8 -1
  10. data/{lib/generators/railwatch/install/templates → app/frontend/lib}/railwatch.ts +162 -42
  11. data/app/jobs/railwatch/detect_performance_issues_job.rb +12 -2
  12. data/app/models/railwatch/anomaly_rule.rb +20 -2
  13. data/app/models/railwatch/ingest/batch.rb +1 -0
  14. data/app/models/railwatch/ingest/mapper.rb +12 -0
  15. data/app/models/railwatch/telemetry/aggregations.rb +154 -19
  16. data/app/models/railwatch/telemetry/exception.rb +7 -0
  17. data/app/models/railwatch/telemetry/execution.rb +18 -0
  18. data/app/models/railwatch/telemetry/log.rb +6 -0
  19. data/app/models/railwatch/telemetry/query.rb +7 -0
  20. data/app/models/railwatch/telemetry/release_health.rb +12 -6
  21. data/app/models/railwatch/telemetry/rollup.rb +17 -2
  22. data/app/models/railwatch/telemetry/tenant.rb +8 -23
  23. data/app/models/railwatch/telemetry_record.rb +8 -6
  24. data/app/models/railwatch/threshold.rb +41 -4
  25. data/app/models/railwatch/window.rb +66 -0
  26. data/docs/embedded.md +11 -0
  27. data/lib/generators/railwatch/install/install_generator.rb +5 -1
  28. data/lib/railwatch/transport/http.rb +15 -5
  29. data/lib/railwatch/version.rb +1 -1
  30. data/lib/railwatch/wire_fixtures.json +856 -0
  31. data/lib/railwatch.rb +22 -0
  32. data/lib/tasks/railwatch_tasks.rake +7 -0
  33. data/public/railwatch/assets/app-layout-DDyQa72H.js +1 -0
  34. data/public/railwatch/assets/{app-wordmark-1SRllgHv.js → app-wordmark-o9CODKP0.js} +1 -1
  35. data/public/railwatch/assets/{appearance-DbwahDrc.js → appearance-BwuCXabr.js} +1 -1
  36. data/public/railwatch/assets/application-B7h1MIhi.css +1 -0
  37. data/public/railwatch/assets/{arrow-up-DwKQcab4.js → arrow-up-C6PxDiY3.js} +1 -1
  38. data/public/railwatch/assets/{auth-layout-DXfxXAN_.js → auth-layout-BRt8MGFD.js} +1 -1
  39. data/public/railwatch/assets/{badge-CApjyWcg.js → badge-CAxXV8za.js} +1 -1
  40. data/public/railwatch/assets/{braces-kGSgxWnS.js → braces-DgomTCNf.js} +1 -1
  41. data/public/railwatch/assets/{card-BaJ8kNkp.js → card-cAtqCxWl.js} +1 -1
  42. data/public/railwatch/assets/{chart-Xnrz1Nqw.js → chart-BBeBkkNa.js} +1 -1
  43. data/public/railwatch/assets/{chart-hover-XYPNE2CX.js → chart-hover-B1M9jc0y.js} +1 -1
  44. data/public/railwatch/assets/chart-panel-DUQTz_C8.js +1 -0
  45. data/public/railwatch/assets/{checkbox-Bf41R8W6.js → checkbox-CmhMHWZO.js} +1 -1
  46. data/public/railwatch/assets/{code-a-9sh36A.js → code-DESvxyTj.js} +1 -1
  47. data/public/railwatch/assets/{copy-block-zykZWQym.js → copy-block-BkSU5832.js} +1 -1
  48. data/public/railwatch/assets/{copy-id-b6CW5ESL.js → copy-id-D03GhN9F.js} +1 -1
  49. data/public/railwatch/assets/{cursor-load-more-0cHzbTPT.js → cursor-load-more-CRyuMeQb.js} +1 -1
  50. data/public/railwatch/assets/{data-table-IvA7xvrK.js → data-table-BIlt7Rtm.js} +1 -1
  51. data/public/railwatch/assets/{edit-DTtRV1fO.js → edit-Bb6MKoe4.js} +1 -1
  52. data/public/railwatch/assets/{edit-7lmf3k80.js → edit-DJ0D0wHN.js} +1 -1
  53. data/public/railwatch/assets/{edit-ZKE15r-F.js → edit-O0NSBWxo.js} +1 -1
  54. data/public/railwatch/assets/{empty-state-BFTnQ2oT.js → empty-state-C38il627.js} +1 -1
  55. data/public/railwatch/assets/env-layout-REF7OM4q.js +1 -0
  56. data/public/railwatch/assets/{execution-path-Di5uUHi-.js → execution-path-FYLq1TwC.js} +1 -1
  57. data/public/railwatch/assets/{filter-bar-Btm__gQi.js → filter-bar-CYog9Alp.js} +1 -1
  58. data/public/railwatch/assets/{flamegraph-C-mZdqa7.js → flamegraph-DSs69foN.js} +1 -1
  59. data/public/railwatch/assets/{frames-KZBsjg6-.js → frames-BUi2J5Mk.js} +1 -1
  60. data/public/railwatch/assets/{google-sign-in-button-DJRoD72g.js → google-sign-in-button-BQiIKFdd.js} +1 -1
  61. data/public/railwatch/assets/{index-9LL_C3eG.js → index-8-hnAhOD.js} +1 -1
  62. data/public/railwatch/assets/{index-Bjw4sTXU.js → index-BaR1U9An.js} +1 -1
  63. data/public/railwatch/assets/{index-BtSlLmkg.js → index-BeOh2t_S.js} +1 -1
  64. data/public/railwatch/assets/{index-DXUwhSCQ.js → index-BiiyMcA0.js} +1 -1
  65. data/public/railwatch/assets/{index-P_e_cI0O.js → index-BoUBioBP.js} +1 -1
  66. data/public/railwatch/assets/{index-kNhZPLVj.js → index-C-PmdhXA.js} +1 -1
  67. data/public/railwatch/assets/{index-DCAho97K.js → index-C3A_9imx.js} +1 -1
  68. data/public/railwatch/assets/{index-BsrCnQqg.js → index-C7OtLq_3.js} +1 -1
  69. data/public/railwatch/assets/{index-Bot7kHMA.js → index-CFFpnzIS.js} +1 -1
  70. data/public/railwatch/assets/{index-Z1c4nFL0.js → index-CFRLPs4J.js} +1 -1
  71. data/public/railwatch/assets/{index-BAkZNNEh.js → index-CGs4m_fa.js} +1 -1
  72. data/public/railwatch/assets/{index-CIUl9POV.js → index-CICUIFHL.js} +1 -1
  73. data/public/railwatch/assets/{index-wOj5j0KC.js → index-C_upSl_k.js} +1 -1
  74. data/public/railwatch/assets/{index-Cj3_jD1E.js → index-CiPo4Gob.js} +1 -1
  75. data/public/railwatch/assets/{index-EGU__DOB.js → index-CpkI015n.js} +1 -1
  76. data/public/railwatch/assets/{index-DEfu_YvU.js → index-CrZ3vHDL.js} +1 -1
  77. data/public/railwatch/assets/{index-CAzQkw-h.js → index-CsoN51vW.js} +1 -1
  78. data/public/railwatch/assets/{index-CeqMDByC.js → index-DDI_Zx5V.js} +1 -1
  79. data/public/railwatch/assets/index-DSvlZVWG.js +1 -0
  80. data/public/railwatch/assets/{index-CU6yhEyt.js → index-DW2CBbxU.js} +1 -1
  81. data/public/railwatch/assets/{index-uHBVO8Lh.js → index-Dh4IRLFI.js} +1 -1
  82. data/public/railwatch/assets/{index-HCJmFcZu.js → index-DqTFTP8p.js} +1 -1
  83. data/public/railwatch/assets/index-DrcKVG2f.js +1 -0
  84. data/public/railwatch/assets/{index-QyHE9WsU.js → index-DtHmuB9Q.js} +1 -1
  85. data/public/railwatch/assets/{index-CBJZQgHA.js → index-DvjY3dPD.js} +1 -1
  86. data/public/railwatch/assets/{index-Bo8hrjbs.js → index-FhUaPPab.js} +1 -1
  87. data/public/railwatch/assets/{index-BWSyQRZI.js → index-JdCVBrw8.js} +1 -1
  88. data/public/railwatch/assets/{index-DAEzO_vq.js → index-QpTtwFwu.js} +1 -1
  89. data/public/railwatch/assets/{index-D4968-TC.js → index-ZOGOB8SA.js} +1 -1
  90. data/public/railwatch/assets/{index-CP6jrrPT.js → index-ZSZg9rtq.js} +1 -1
  91. data/public/railwatch/assets/{index-DcItJxa5.js → index-r0tSIplE.js} +1 -1
  92. data/public/railwatch/assets/{index-CYYmZpZu.js → index-sTYvcbkh.js} +1 -1
  93. data/public/railwatch/assets/{index-B0VXzXYG.js → index-so4lRrRq.js} +1 -1
  94. data/public/railwatch/assets/{index-B-bSm1jB.js → index-tpz-OGUP.js} +1 -1
  95. data/public/railwatch/assets/{index-CX8Br7m9.js → index-umIAl-pL.js} +1 -1
  96. data/public/railwatch/assets/{inertia-Degv_G9N.js → inertia-DLew8ZNx.js} +3 -3
  97. data/public/railwatch/assets/{input-error-C6SqO_iU.js → input-error-cvM6_Jht.js} +1 -1
  98. data/public/railwatch/assets/{json-viewer-DcQIxT86.js → json-viewer-D922McGi.js} +1 -1
  99. data/public/railwatch/assets/{klass-WgQEfWpo.js → klass-CrwICqN8.js} +1 -1
  100. data/public/railwatch/assets/{label-B_vK19tR.js → label-GWl7I6sf.js} +1 -1
  101. data/public/railwatch/assets/{layout-C8q7GUlZ.js → layout-0ZAnD3zl.js} +1 -1
  102. data/public/railwatch/assets/{live-dot-DaXEbkXo.js → live-dot-D1n_BreY.js} +1 -1
  103. data/public/railwatch/assets/{nav-BnD5XArZ.js → nav-DPxr1NNC.js} +1 -1
  104. data/public/railwatch/assets/{new-BL9OqXHe.js → new-Cdl6pqST.js} +1 -1
  105. data/public/railwatch/assets/{new-uxMnBtoY.js → new-D-ZzUK9a.js} +1 -1
  106. data/public/railwatch/assets/{new-BqP0NhOF.js → new-DEVkYv-z.js} +1 -1
  107. data/public/railwatch/assets/{new-CWP3wtwt.js → new-DHAHDrN7.js} +1 -1
  108. data/public/railwatch/assets/{new-DIjwu2jF.js → new-Dz4lZf1L.js} +1 -1
  109. data/public/railwatch/assets/{new-BhRnQn4w.js → new-GMrRFurX.js} +1 -1
  110. data/public/railwatch/assets/{onboarding-C_Y3xTsH.js → onboarding-D1vwaHYT.js} +1 -1
  111. data/public/railwatch/assets/{origin-identity-CNhOnfBa.js → origin-identity-Bk9yHWZ1.js} +1 -1
  112. data/public/railwatch/assets/{percentile-picker-BujVEhP6.js → percentile-picker-DfSx9yJO.js} +1 -1
  113. data/public/railwatch/assets/{relative-time-6CoKr1lN.js → relative-time-CjIjb8Lg.js} +1 -1
  114. data/public/railwatch/assets/{release-health-1Qzc_XhT.js → release-health-4b3tivEf.js} +1 -1
  115. data/public/railwatch/assets/{route-xqgYjOwJ.js → route-C_5BUtHK.js} +1 -1
  116. data/public/railwatch/assets/{segmented-CVKq2Xpa.js → segmented-BgbT3wZa.js} +1 -1
  117. data/public/railwatch/assets/{select-CsmSXsJ4.js → select-DmunxCKE.js} +1 -1
  118. data/public/railwatch/assets/{separator-CrM10Oot.js → separator-BXzEdZ_8.js} +1 -1
  119. data/public/railwatch/assets/series-chart-Xf49v9cv.js +1 -0
  120. data/public/railwatch/assets/{show-zxVxhUel.js → show-B2zLAW83.js} +1 -1
  121. data/public/railwatch/assets/{show-BM-E2iYc.js → show-BLpWUHWD.js} +1 -1
  122. data/public/railwatch/assets/{show-CvNV23gH.js → show-CAl7xcex.js} +1 -1
  123. data/public/railwatch/assets/{show-BlNR3ZOZ.js → show-DI8IhNUH.js} +1 -1
  124. data/public/railwatch/assets/{show-B0nHy31L.js → show-DSP9Cq_C.js} +2 -2
  125. data/public/railwatch/assets/{show-rysBbvt-.js → show-DXs4deaC.js} +1 -1
  126. data/public/railwatch/assets/{show-DUwQlohF.js → show-DYskfl3-.js} +1 -1
  127. data/public/railwatch/assets/{show-ixK2r2f1.js → show-DcpTFiLi.js} +1 -1
  128. data/public/railwatch/assets/{show-CQ4Ze-Un.js → show-Dily73Xk.js} +1 -1
  129. data/public/railwatch/assets/{show-B3izvRdU.js → show-DlRVS18-.js} +1 -1
  130. data/public/railwatch/assets/{show-8H9WAGav.js → show-Dn-GwFZL.js} +1 -1
  131. data/public/railwatch/assets/{show-DZwcZZrz.js → show-DnR1Dnjd.js} +1 -1
  132. data/public/railwatch/assets/{show-CMisHNAI.js → show-SHwZjXb7.js} +1 -1
  133. data/public/railwatch/assets/{show-B7NhcWgp.js → show-SvLOcPrx.js} +1 -1
  134. data/public/railwatch/assets/{show-CJMeX75E.js → show-Y74rM0VT.js} +1 -1
  135. data/public/railwatch/assets/{show-DMZlwzL_.js → show-mU38uGTg.js} +1 -1
  136. data/public/railwatch/assets/{show-9jDeBNpc.js → show-vQ4bndYD.js} +1 -1
  137. data/public/railwatch/assets/{sort-header-CKj_7G3U.js → sort-header-Dcq9bzmo.js} +1 -1
  138. data/public/railwatch/assets/{sparkline-cell-Ckg29yVq.js → sparkline-cell-BON3qQUB.js} +1 -1
  139. data/public/railwatch/assets/{stat-TxvoBiC8.js → stat-DFEyFxkO.js} +1 -1
  140. data/public/railwatch/assets/{status-badge-5NCVJHjo.js → status-badge-BaUKP7Yo.js} +1 -1
  141. data/public/railwatch/assets/{tenant-path-B5ZRK4x3.js → tenant-path-DPZPc985.js} +1 -1
  142. data/public/railwatch/assets/{text-link-BPllRj0F.js → text-link-BO77t9Xk.js} +1 -1
  143. data/public/railwatch/assets/{textarea-DtD5yZp_.js → textarea-DTqrCiV0.js} +1 -1
  144. data/public/railwatch/assets/{timeline-DZz5ZnHN.js → timeline-D5rJ0es2.js} +1 -1
  145. data/public/railwatch/assets/{transition-Ds6Cakuc.js → transition-DMIrZVth.js} +1 -1
  146. data/public/railwatch/assets/{use-clipboard-BLjjYH8V.js → use-clipboard-ByoUGQqA.js} +1 -1
  147. data/public/railwatch/assets/{use-live-GKze1mNC.js → use-live-D7xKz2ma.js} +1 -1
  148. data/public/railwatch/manifest.json +1284 -1284
  149. metadata +119 -117
  150. data/public/railwatch/assets/app-layout-B5Fcdwfz.js +0 -1
  151. data/public/railwatch/assets/application-CwshqwM6.css +0 -1
  152. data/public/railwatch/assets/chart-panel-BYk68mJJ.js +0 -1
  153. data/public/railwatch/assets/env-layout-erSDpIXg.js +0 -1
  154. data/public/railwatch/assets/index-BoTOVM0v.js +0 -1
  155. data/public/railwatch/assets/index-DBTxQw77.js +0 -1
  156. data/public/railwatch/assets/series-chart-D33UhdG2.js +0 -1
@@ -12,7 +12,8 @@ module Railwatch
12
12
  MAX_ROLLUP_ROWS_PER_RULE = MAX_GROUPS_PER_RULE * 25
13
13
 
14
14
  TYPE_FOR = { "requests" => "request", "jobs" => "job_attempt", "commands" => "command", "queries" => "query",
15
- "scheduled_tasks" => "scheduled_task", "outgoing_requests" => "outgoing_request" }.freeze
15
+ "scheduled_tasks" => "scheduled_task", "outgoing_requests" => "outgoing_request",
16
+ "llm_calls" => "llm_call", "llm_tools" => "llm_tool" }.freeze
16
17
 
17
18
  def perform(environment)
18
19
  now = Time.current
@@ -51,19 +52,28 @@ module Railwatch
51
52
  .transform_values { |group_rows| [ group_rows.first.name, Telemetry::Rollup.summarize(group_rows) ] }
52
53
  end
53
54
 
55
+ NANOS_PER_DOLLAR = 1_000_000_000.0
56
+
54
57
  def value_for(metric, s)
58
+ extra = s[:extra] || {}
55
59
  case metric
56
60
  when "p95" then s[:p95] / 1000.0
57
61
  when "max" then s[:max] / 1000.0
58
62
  when "avg" then s[:avg] / 1000.0
59
63
  when "error_rate", "failure_rate" then s[:count].zero? ? nil : (s[:errors] * 100.0 / s[:count])
64
+ # Money and tokens are sums over the window, not percentiles of it.
65
+ when "spend" then extra["cost_nanos"].to_i / NANOS_PER_DOLLAR
66
+ when "tokens" then extra["input_tokens"].to_i + extra["output_tokens"].to_i
67
+ # A cut-off answer is a successful call, so it is invisible to
68
+ # error_rate. This is the only way to be told about it.
69
+ when "truncation_rate" then s[:count].zero? ? nil : (extra["truncated"].to_i * 100.0 / s[:count])
60
70
  end
61
71
  end
62
72
 
63
73
  def open_issue(environment, threshold, group_hash, name, value, from, now)
64
74
  issue, outcome = Issue.record_occurrence!(
65
75
  environment: environment, group_hash: "perf:#{threshold.id}:#{group_hash}", kind: "performance",
66
- title: "#{name} exceeded #{threshold.metric} #{threshold.limit}#{threshold.metric.end_with?('rate') ? '%' : 'ms'} (#{value.round(1)})",
76
+ title: "#{name} exceeded #{threshold.metric} #{threshold.format_value(threshold.limit)} (#{threshold.format_value(value.round(2))})",
67
77
  culprit: name, occurred_at: now, deploy: nil, user_ref: nil,
68
78
  sample: { threshold_id: threshold.id, value: value.round(2), metric: threshold.metric, limit: threshold.limit })
69
79
  carry_detection(issue, outcome) do
@@ -10,13 +10,22 @@ module Railwatch
10
10
  MAX_BASELINE_DAYS = 30
11
11
  MAX_DEVIATION = 10
12
12
  MAX_WINDOW_MINUTES = 1_440
13
- TARGET_KINDS = %w[requests jobs queries outgoing_requests scheduled_tasks].freeze
14
- METRICS = %w[p95 avg count error_rate].freeze
13
+ TARGET_KINDS = %w[requests jobs queries outgoing_requests scheduled_tasks llm_calls].freeze
14
+ # spend included because "we are spending well above baseline for this
15
+ # hour of the week" is the LLM alert that needs no threshold picked.
16
+ METRICS = %w[p95 avg count error_rate spend].freeze
17
+ SHARED_METRICS = %w[p95 avg count error_rate].freeze
18
+ LLM_METRICS = %w[spend].freeze
19
+
20
+ def self.metrics_for(target_kind)
21
+ target_kind == "llm_calls" ? SHARED_METRICS + LLM_METRICS : SHARED_METRICS
22
+ end
15
23
 
16
24
  def environment = Environment.current
17
25
 
18
26
  validates :target_kind, inclusion: { in: TARGET_KINDS }
19
27
  validates :metric, inclusion: { in: METRICS }
28
+ validate :metric_applies_to_target_kind
20
29
  validates :target, presence: true, length: { maximum: 256 }
21
30
  validates :deviation, numericality: { greater_than: 0, less_than_or_equal_to: MAX_DEVIATION }
22
31
  validates :window_minutes, numericality: { only_integer: true, greater_than: 0,
@@ -29,5 +38,14 @@ module Railwatch
29
38
  def description
30
39
  "#{target_kind} #{target == '*' ? 'all' : target}: #{metric} > #{deviation}σ above #{baseline_days}-day baseline"
31
40
  end
41
+
42
+ private
43
+
44
+ def metric_applies_to_target_kind
45
+ return if metric.blank? || target_kind.blank?
46
+ return if AnomalyRule.metrics_for(target_kind).include?(metric)
47
+
48
+ errors.add(:metric, "is not available for #{target_kind}")
49
+ end
32
50
  end
33
51
  end
@@ -182,6 +182,7 @@ module Railwatch
182
182
  # batch does not need.
183
183
  raise TypeError, "record must be an object" unless rec.is_a?(Hash)
184
184
  raise TypeError, "t must be a string" unless rec["t"].is_a?(String)
185
+ Ingest::Mapper.validate_version!(rec)
185
186
  else
186
187
  Ingest::Mapper.validate_record!(rec)
187
188
  end
@@ -111,6 +111,7 @@ module Railwatch
111
111
  def validate_record!(rec, identifiers: true)
112
112
  raise TypeError, "record must be an object" unless rec.is_a?(Hash)
113
113
  raise TypeError, "t must be a string" unless rec["t"].is_a?(String)
114
+ validate_version!(rec)
114
115
 
115
116
  if identifiers
116
117
  validate_identifier!(rec, "_group", GROUP_HASH, "32 hexadecimal characters")
@@ -129,6 +130,17 @@ module Railwatch
129
130
  validate_locals!(rec["locals"])
130
131
  end
131
132
 
133
+ # A record shape is named by its type and version (Railwatch::Record::VERSIONS).
134
+ # This mapper only knows the versions the gem it ships with emits; a
135
+ # record from a newer or older gem is refused by name rather than read
136
+ # through columns that may no longer mean the same thing.
137
+ def validate_version!(rec)
138
+ expected = Record::VERSIONS[rec["t"].to_sym]
139
+ return if expected.nil? # unknown type: row_for answers nil and the batch rejects it as such
140
+
141
+ raise TypeError, "#{rec["t"]} v#{rec["v"].inspect} is not v#{expected}" unless rec["v"].is_a?(Integer) && rec["v"] == expected
142
+ end
143
+
132
144
  def validate_identifier!(rec, field, pattern, description)
133
145
  value = rec[field]
134
146
  return if value.nil?
@@ -9,18 +9,154 @@ module Railwatch
9
9
  class Aggregations
10
10
  GROUPED_SORT_FIELDS = %i[count p50 p95 p99 errors avg max].freeze
11
11
 
12
- # Time-bucketed series from rollups for charts: [{t, count, errors, p50, p95, p99}]
13
- def self.series(environment, record_type, from:, to:, group_hash: nil, name: nil)
14
- environment.with_telemetry do
15
- scope = Telemetry::Rollup.for_type(record_type).between(from, to)
16
- scope = scope.where(group_hash: group_hash) if group_hash
17
- scope = scope.where(name: name) if name
18
- scope.group(:bucket).order(:bucket)
19
- .pluck(:bucket, Arel.sql("SUM(count)"), Arel.sql("SUM(error_count)"), Arel.sql("SUM(client_error_count)"), Arel.sql("SUM(duration_sum)"), Arel.sql("MAX(p50)"), Arel.sql("MAX(p95)"), Arel.sql("MAX(p99)"))
20
- .map { |b, c, e, ce, ds, p50, p95, p99| { t: b, count: c, errors: e, client_errors: ce, avg: c.zero? ? 0 : ds / c / 1000.0, p50: p50 / 1000.0, p95: p95 / 1000.0, p99: p99 / 1000.0 } }
12
+ # Bucket widths a chart can be drawn at. Anything under an hour is read
13
+ # from the raw tables (rollups are hourly); an hour and up groups rollups.
14
+ STEPS = { "1m" => 1.minute, "5m" => 5.minutes, "15m" => 15.minutes, "1h" => 1.hour, "6h" => 6.hours, "1d" => 1.day }.freeze
15
+ # A step is offered for a window when it draws at least two points and at
16
+ # most this many: more bars than pixels is mush, and a sub-hour step over
17
+ # a long window is also a long raw scan.
18
+ MAX_POINTS = 400
19
+
20
+ # The bucket a window is drawn at unless the page asks for another one.
21
+ # Sub-hour steps scan raw rows, so they are the default only where that
22
+ # scan is a few thousand rows; a day and beyond stays on rollups.
23
+ def self.default_step(from, to)
24
+ span = to - from
25
+ if span <= 2.hours then "1m"
26
+ elsif span <= 12.hours then "5m"
27
+ elsif span <= 7.days then "1h"
28
+ else "6h"
29
+ end
30
+ end
31
+
32
+ # STEPS keys that make sense for the window, finest first.
33
+ def self.steps_for(from, to)
34
+ span = to - from
35
+ STEPS.select { |_key, step| (span / step).between?(2, MAX_POINTS) }.keys
36
+ end
37
+
38
+ # Time-bucketed series for charts, one point per `step` from `from` to `to`
39
+ # with empty buckets filled in: [{t, count, errors, client_errors, avg,
40
+ # p50, p95, p99}]. Durations in milliseconds; an empty bucket carries nil
41
+ # for them so a latency line breaks instead of dropping to zero.
42
+ def self.series(environment, record_type, from:, to:, group_hash: nil, step: nil, tenant: nil)
43
+ environment.with_telemetry { points(record_type, from: from, to: to, group_hash: group_hash, step: step, tenant: tenant) }
44
+ end
45
+
46
+ # `series` for a caller already inside environment.with_telemetry. A
47
+ # tenant narrows to one app_tenant, which rollups do not carry, so a
48
+ # tenant series reads raw rows at every step.
49
+ def self.points(record_type, from:, to:, group_hash: nil, step: nil, tenant: nil)
50
+ step = STEPS.fetch(step || default_step(from, to))
51
+ raw = step < 1.hour || tenant
52
+ points = raw ? raw_series(record_type, from, to, group_hash, step, tenant) : rollup_series(record_type, from, to, group_hash, step)
53
+ fill(points, from, to, step)
54
+ end
55
+
56
+ # Sums hourly rollups into `step`-wide buckets. Percentiles are the
57
+ # worst hour's, the same reading series always gave across groups.
58
+ def self.rollup_series(record_type, from, to, group_hash, step)
59
+ bucket = bucket_sql("bucket", step)
60
+ scope = Telemetry::Rollup.for_type(record_type).between(from, to)
61
+ scope = scope.where(group_hash: group_hash) if group_hash
62
+ scope.group(bucket).order(bucket)
63
+ .pluck(bucket, Arel.sql("SUM(count)"), Arel.sql("SUM(error_count)"), Arel.sql("SUM(client_error_count)"), Arel.sql("SUM(duration_sum)"), Arel.sql("MAX(p50)"), Arel.sql("MAX(p95)"), Arel.sql("MAX(p99)"))
64
+ .map { |b, c, e, ce, ds, p50, p95, p99| point(b, c, e, ce, ds, p50, p95, p99) }
65
+ end
66
+
67
+ # What a raw row of each record type is, and when it counts as an error
68
+ # (or a client error), mirroring RollupJob#error?. Both readings of the
69
+ # same rows must agree, which spec/models/telemetry/aggregations_spec.rb
70
+ # checks for every type.
71
+ RAW = {
72
+ "request" => [ -> { Telemetry::Execution.requests }, "status >= 500", "status BETWEEN 400 AND 499" ],
73
+ "job_attempt" => [ -> { Telemetry::Execution.jobs }, "outcome = 'failed'" ],
74
+ "scheduled_task" => [ -> { Telemetry::Execution.scheduled }, "outcome = 'failed'" ],
75
+ "command" => [ -> { Telemetry::Execution.commands }, "status != 0" ],
76
+ "channel_action" => [ -> { Telemetry::Execution.channels }, "outcome = 'failed'" ],
77
+ "query" => [ -> { Telemetry::Query.all } ],
78
+ "outgoing_request" => [ -> { Telemetry::OutgoingRequest.all }, "status_code >= 500 OR status_code IS NULL OR status_code = 0" ],
79
+ "cache_event" => [ -> { Telemetry::CacheEvent.all } ],
80
+ "mail" => [ -> { Telemetry::Mail.all }, "failed" ],
81
+ "visit" => [ -> { Telemetry::Visit.all }, "status = 'error'" ],
82
+ "span" => [ -> { Telemetry::Span.all }, "status = 'failed'" ],
83
+ "notification" => [ -> { Telemetry::Notification.all }, "failed" ],
84
+ "view_render" => [ -> { Telemetry::ViewRender.all } ],
85
+ "transaction" => [ -> { Telemetry::Transaction.all }, "outcome = 'rollback'" ],
86
+ "llm_call" => [ -> { Telemetry::LlmCall.models }, "status = 'failed'" ],
87
+ "llm_tool" => [ -> { Telemetry::LlmCall.tools }, "status = 'failed'" ]
88
+ }.freeze
89
+
90
+ # Sub-hour buckets straight from the raw rows, in one statement:
91
+ # counts and sums per bucket, and exact nearest-rank percentiles from a
92
+ # ROW_NUMBER over each bucket's durations. Rollups are hourly, so this
93
+ # is the only reading finer than an hour; it is also the freshest one,
94
+ # since the current hour's rollup is up to a minute behind.
95
+ def self.raw_series(record_type, from, to, group_hash, step, tenant = nil)
96
+ scope, error_sql, client_error_sql = RAW.fetch(record_type)
97
+ rows = scope.call.where(occurred_at: from..to)
98
+ rows = rows.where(group_hash: group_hash) if group_hash
99
+ rows = rows.where(app_tenant: tenant) if tenant
100
+ model = rows.model
101
+ inner = rows.select(bucket_sql("occurred_at", step).to_s + " AS b", "duration", flag_sql(error_sql) + " AS err", flag_sql(client_error_sql) + " AS cerr")
102
+ ranked = model.unscoped.from(inner, "r").select("b", "duration", "err", "cerr",
103
+ "ROW_NUMBER() OVER (PARTITION BY b ORDER BY duration) AS rn", "COUNT(*) OVER (PARTITION BY b) AS n")
104
+ model.unscoped.from(ranked, "w").group("b").order("b")
105
+ .pluck(Arel.sql("b"), Arel.sql("COUNT(*)"), Arel.sql("SUM(err)"), Arel.sql("SUM(cerr)"), Arel.sql("SUM(duration)"),
106
+ rank_sql(50), rank_sql(95), rank_sql(99))
107
+ .map { |b, c, e, ce, ds, p50, p95, p99| point(b, c, e, ce, ds, p50, p95, p99) }
108
+ end
109
+
110
+ # The timestamp columns a series buckets on: raw rows by occurred_at,
111
+ # rollups and release health by their hourly bucket.
112
+ BUCKET_COLUMNS = %w[occurred_at bucket].freeze
113
+
114
+ # Epoch seconds of the start of the `step`-wide bucket holding `column`.
115
+ # `column` must be one of BUCKET_COLUMNS and `step` is a Duration, so the
116
+ # fragment can only ever hold a listed column name and an integer literal.
117
+ def self.bucket_sql(column, step)
118
+ raise ArgumentError, "unknown bucket column #{column.inspect}" unless BUCKET_COLUMNS.include?(column)
119
+ seconds = Integer(step.to_i)
120
+ Arel.sql("(strftime('%s', #{column}) / #{seconds}) * #{seconds}")
121
+ end
122
+
123
+ # One point per bucket from `from` to `to`, keeping the computed ones and
124
+ # zero-filling the rest, so a quiet minute is a gap of the right width.
125
+ # A series with its own point shape passes a block building its empty one.
126
+ def self.fill(points, from, to, step)
127
+ by_bucket = points.index_by { |p| p[:t] }
128
+ first = Time.at((from.to_i / step.to_i) * step.to_i).utc
129
+ last = Time.at((to.to_i / step.to_i) * step.to_i).utc
130
+ (first.to_i..last.to_i).step(step.to_i).map do |t|
131
+ at = Time.at(t).utc
132
+ by_bucket[at] || (block_given? ? yield(at) : empty_point(at))
21
133
  end
22
134
  end
23
135
 
136
+ def self.point(bucket, count, errors, client_errors, duration_sum, p50, p95, p99)
137
+ { t: Time.at(bucket).utc, count: count, errors: errors, client_errors: client_errors,
138
+ avg: count.zero? ? 0 : duration_sum / count / 1000.0, p50: p50 / 1000.0, p95: p95 / 1000.0, p99: p99 / 1000.0 }
139
+ end
140
+
141
+ def self.empty_point(t)
142
+ { t: t, count: 0, errors: 0, client_errors: 0, avg: nil, p50: nil, p95: nil, p99: nil }
143
+ end
144
+
145
+ def self.flag_sql(predicate)
146
+ predicate ? "CASE WHEN #{predicate} THEN 1 ELSE 0 END" : "0"
147
+ end
148
+
149
+ # The nearest-rank percentile: the duration whose rank is ceil(n * p/100).
150
+ RANK_SQL = { 50 => Arel.sql("MAX(CASE WHEN rn = (n * 50 + 99) / 100 THEN duration END)"),
151
+ 95 => Arel.sql("MAX(CASE WHEN rn = (n * 95 + 99) / 100 THEN duration END)"),
152
+ 99 => Arel.sql("MAX(CASE WHEN rn = (n * 99 + 99) / 100 THEN duration END)") }.freeze
153
+
154
+ def self.rank_sql(percent)
155
+ RANK_SQL.fetch(percent)
156
+ end
157
+
158
+ private_class_method :rollup_series, :raw_series, :point, :empty_point, :flag_sql, :rank_sql
159
+
24
160
  # Per-group table rows for a record type in the window: count/error totals,
25
161
  # merged percentiles (via Telemetry::Rollup.summarize), and a sparkline.
26
162
  def self.grouped(environment, record_type, from:, to:, limit: 100, order: nil, dir: nil)
@@ -34,6 +170,15 @@ module Railwatch
34
170
  cached(key) { grouped_uncached(environment, record_type, from: from, to: to, limit: limit, order: order, sign: sign) }
35
171
  end
36
172
 
173
+ # The minute cache exists because the platform's rollups only change
174
+ # when RollupJob writes. Embedded, every batch updates them and one
175
+ # person is looking, so the cache would only make the page a minute
176
+ # stale for nothing.
177
+ def self.cached(key, &block)
178
+ return yield if Railwatch.config.local?
179
+ Rails.cache.fetch(key, expires_in: 1.minute, &block)
180
+ end
181
+
37
182
  def self.grouped_uncached(environment, record_type, from:, to:, limit:, order:, sign:)
38
183
  environment.with_telemetry do
39
184
  rows = Telemetry::Rollup.for_type(record_type).between(from, to).to_a
@@ -60,16 +205,6 @@ module Railwatch
60
205
  end
61
206
  end
62
207
 
63
- # The minute cache exists because the platform's rollups only change
64
- # when RollupJob writes. Embedded, every batch updates them and one
65
- # person is looking, so the cache would only make the page a minute
66
- # stale for nothing.
67
- def self.cached(key, &block)
68
- return yield if Railwatch.config.local?
69
-
70
- Rails.cache.fetch(key, expires_in: 1.minute, &block)
71
- end
72
-
73
208
  # The digest-free half of Rollup.summarize: everything the sort keys
74
209
  # that are not percentiles need.
75
210
  def self.cheap_summary(rows)
@@ -9,6 +9,13 @@ module Railwatch
9
9
 
10
10
  scope :unhandled, -> { where(handled: false) }
11
11
 
12
+ def as_row
13
+ { id: id, class_name: class_name, message: message.first(500), handled: handled, severity: severity, source: source,
14
+ file: file, line: line, occurred_at: occurred_at, deploy: deploy, execution_id: execution_id,
15
+ execution_source: execution_source, execution_preview: execution_preview, user_ref: user_ref, tenant: app_tenant,
16
+ group_hash: group_hash }
17
+ end
18
+
12
19
  def timeline_label
13
20
  "#{class_name}: #{message.to_s.first(100)}"
14
21
  end
@@ -63,6 +63,24 @@ module Railwatch
63
63
  duration / 1000.0
64
64
  end
65
65
 
66
+ def queue_latency_ms
67
+ queue_latency && (queue_latency / 1000.0).round(1)
68
+ end
69
+
70
+ # The row every list surface shows for an execution: what the kind has
71
+ # in common, then what it adds.
72
+ def as_row
73
+ row = { execution_id: execution_id, kind: kind, name: name, status: status, outcome: outcome, duration: duration_ms.round(2),
74
+ occurred_at: occurred_at, deploy: deploy, server: server, user_ref: user_ref, tenant: app_tenant,
75
+ exception_preview: exception_preview }
76
+ case kind
77
+ when "request" then row.merge(method: self[:method], inertia_component: inertia_component, queries: counters["queries"])
78
+ when "job_attempt" then row.merge(queue: queue, attempt: attempt, job_id: job_id, queue_latency: queue_latency_ms)
79
+ when "scheduled_task" then row.merge(task_key: task_key)
80
+ else row
81
+ end
82
+ end
83
+
66
84
  def children_count
67
85
  counters.values.sum
68
86
  end
@@ -77,6 +77,12 @@ module Railwatch
77
77
  def quote(term) = %("#{term.gsub('"', '""')}")
78
78
  end
79
79
 
80
+ def as_row
81
+ { id: id, level: level, message: message.first(2000), tags: tags, occurred_at: occurred_at, deploy: deploy,
82
+ execution_id: execution_id, execution_source: execution_source, execution_preview: execution_preview,
83
+ source: source, tenant: app_tenant, user_ref: user_ref, context: context }
84
+ end
85
+
80
86
  def timeline_label
81
87
  message.to_s.first(120)
82
88
  end
@@ -42,6 +42,13 @@ module Railwatch
42
42
  text
43
43
  end
44
44
 
45
+ def as_row
46
+ { id: id, group_hash: group_hash, sql: sql.first(2_000), name: name, duration: duration_ms.round(3), occurred_at: occurred_at,
47
+ deploy: deploy, execution_id: execution_id, execution_preview: execution_preview, source: source,
48
+ connection: connection, role: role, adapter: adapter, row_count: row_count, tenant: app_tenant, user_ref: user_ref,
49
+ explain: explain.present? }
50
+ end
51
+
45
52
  def timeline_label
46
53
  sql.to_s.first(120)
47
54
  end
@@ -36,13 +36,19 @@ module Railwatch
36
36
  }
37
37
  end
38
38
 
39
- # Hourly points for the stacked sessions-by-status chart.
40
- def self.series(deploy, from, to)
41
- between(from, to).for_release(deploy).group(:bucket).order(:bucket)
42
- .pluck(:bucket, Arel.sql("SUM(sessions)"), Arel.sql("SUM(sessions_errored)"), Arel.sql("SUM(sessions_crashed)"))
43
- .map { |bucket, sessions, errored, crashed|
44
- { t: bucket, sessions: sessions, ok: sessions - errored - crashed, errored: errored, crashed: crashed }
39
+ # Points for the stacked sessions-by-status chart, one per `step` across
40
+ # the window. The table is hourly and a session is counted once per hour
41
+ # it was seen, so a step under an hour is drawn at an hour: raw session
42
+ # rows would count a live session on every flush.
43
+ def self.series(deploy, from, to, step: nil)
44
+ step = [ Telemetry::Aggregations::STEPS.fetch(step || Telemetry::Aggregations.default_step(from, to)), 1.hour ].max
45
+ bucket = Telemetry::Aggregations.bucket_sql("bucket", step)
46
+ points = between(from, to).for_release(deploy).group(bucket).order(bucket)
47
+ .pluck(bucket, Arel.sql("SUM(sessions)"), Arel.sql("SUM(sessions_errored)"), Arel.sql("SUM(sessions_crashed)"))
48
+ .map { |b, sessions, errored, crashed|
49
+ { t: Time.at(b).utc, sessions: sessions, ok: sessions - errored - crashed, errored: errored, crashed: crashed }
45
50
  }
51
+ Telemetry::Aggregations.fill(points, from, to, step) { |t| { t: t, sessions: 0, ok: 0, errored: 0, crashed: 0 } }
46
52
  end
47
53
 
48
54
  # Each release's share of the window's sessions, as a percentage -- how
@@ -54,7 +54,7 @@ module Railwatch
54
54
  # Aggregate a relation of rollups into one summary with merged percentiles.
55
55
  def self.summarize(relation)
56
56
  rows = relation.to_a
57
- return { count: 0, errors: 0, client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0, max: 0 } if rows.empty?
57
+ return { count: 0, errors: 0, client_errors: 0, avg: 0, p50: 0, p95: 0, p99: 0, max: 0, extra: {} } if rows.empty?
58
58
  # merge! pushes the row's centroids into one accumulating digest.
59
59
  # `+` built a brand-new digest from both operands' centroids on every
60
60
  # row, so merging N rows re-pushed every earlier centroid N times:
@@ -70,9 +70,24 @@ module Railwatch
70
70
  p50: merged.percentile(0.5).to_i,
71
71
  p95: merged.percentile(0.95).to_i,
72
72
  p99: merged.percentile(0.99).to_i,
73
- max: rows.map(&:duration_max).max
73
+ max: rows.map(&:duration_max).max,
74
+ # Everything a type puts in `extra` -- llm_call's cost_nanos and
75
+ # token counts, cache_event's hits and misses -- merged the way
76
+ # absorb! merges it: numbers add up, anything else is last-wins.
77
+ # Spend is a sum over a window rather than a percentile of
78
+ # durations, so a rule about money has nowhere else to read from.
79
+ extra: merge_extras(rows)
74
80
  }
75
81
  end
82
+
83
+ def self.merge_extras(rows)
84
+ rows.each_with_object({}) do |row, out|
85
+ row.extra.each do |key, value|
86
+ existing = out[key]
87
+ out[key] = existing.is_a?(Numeric) && value.is_a?(Numeric) ? existing + value : value
88
+ end
89
+ end
90
+ end
76
91
  end
77
92
  end
78
93
  end
@@ -18,7 +18,6 @@ module Railwatch
18
18
  # aggregate), so only the busiest tenants get one; the rest report 0.
19
19
  P95_TENANTS = 50
20
20
  SPARKLINE_BUCKETS = 12
21
- SERIES_BUCKETS = 48
22
21
  SORTS = { "requests" => :requests, "errors" => :errors, "p95" => :p95, "users" => :users }.freeze
23
22
 
24
23
  ERRORS_SQL = Arel.sql("SUM(CASE WHEN status >= 500 THEN 1 ELSE 0 END)")
@@ -71,21 +70,12 @@ module Railwatch
71
70
  }
72
71
  end
73
72
 
74
- # Request volume and duration bucketed over the window, SeriesPoint-shaped
75
- # so the same charts render it. Percentiles are approximations: these
76
- # buckets come from raw rows, without the per-bucket t-digest
77
- # Telemetry::Rollup keeps, so p50 is the bucket's average and p95/p99 its
78
- # max.
79
- def self.series(tenant, from, to)
80
- width = bucket_width(from, to, SERIES_BUCKETS)
81
- bucket = bucket_sql(from, width)
82
- Telemetry::Execution.requests.between(from, to).where(app_tenant: tenant).group(bucket)
83
- .pluck(bucket, COUNT_SQL, ERRORS_SQL, CLIENT_ERRORS_SQL, Arel.sql("SUM(duration)"), Arel.sql("MAX(duration)"))
84
- .map { |index, count, errors, client_errors, sum, max|
85
- avg = ms(sum / count)
86
- { t: bucket_at(from, width, index, SERIES_BUCKETS).iso8601, count: count, errors: errors, client_errors: client_errors,
87
- avg: avg, p50: avg, p95: ms(max), p99: ms(max) }
88
- }.sort_by { |point| point[:t] }
73
+ # Request volume and duration bucketed over the window at `step`,
74
+ # SeriesPoint-shaped so the same charts render it. Rollups carry no tenant,
75
+ # so every step reads raw rows (Telemetry::Aggregations.points does that
76
+ # for a tenant), with exact percentiles per bucket.
77
+ def self.series(tenant, from, to, step: nil)
78
+ Telemetry::Aggregations.points("request", from: from, to: to, step: step, tenant: tenant)
89
79
  end
90
80
 
91
81
  def self.routes(tenant, from, to, limit: 20)
@@ -109,14 +99,9 @@ module Railwatch
109
99
  end
110
100
  end
111
101
 
112
- # Same field shape as RequestsController#execution_row, for the tenant
113
- # page's "Recent requests" table.
102
+ # The tenant page's "Recent requests" table.
114
103
  def self.recent_requests(tenant, from, to, limit: 50)
115
- Telemetry::Execution.requests.between(from, to).where(app_tenant: tenant).recent.limit(limit).map do |r|
116
- { execution_id: r.execution_id, name: r.name, status: r.status, duration: r.duration_ms.round(2), occurred_at: r.occurred_at,
117
- user_ref: r.user_ref, tenant: r.app_tenant, exception_preview: r.exception_preview, inertia_component: r.inertia_component,
118
- queries: r.counters["queries"], deploy: r.deploy }
119
- end
104
+ Telemetry::Execution.requests.between(from, to).where(app_tenant: tenant).recent.limit(limit).map(&:as_row)
120
105
  end
121
106
 
122
107
  # -- Aggregation steps -----------------------------------------------------
@@ -2,12 +2,14 @@
2
2
 
3
3
  module Railwatch
4
4
  # Base class for everything the engine stores about the host application.
5
- # The hosted platform keeps one SQLite file per monitored environment through
6
- # activerecord-tenanted; an embedded install monitors exactly one application,
7
- # so this is a plain second database (`railwatch_telemetry` in the host's
8
- # database.yml), separate from the host's own primary and from the engine's
9
- # meta tables. Every telemetry query still runs inside
10
- # Environment#with_telemetry so the code path matches the platform's.
5
+ # An embedded install monitors exactly one application, so this is a plain
6
+ # second database (`railwatch_telemetry` in the host's database.yml),
7
+ # separate from the host's own primary and from the engine's meta tables.
8
+ # The hosted platform keeps one SQLite file per monitored environment
9
+ # instead: it calls `Railwatch::TelemetryRecord.tenanted("telemetry")`
10
+ # (activerecord-tenanted) from an initializer, which replaces the
11
+ # connection below with a per-tenant one, and every telemetry query runs
12
+ # inside Environment#with_telemetry so the code path is the same in both.
11
13
  class TelemetryRecord < ActiveRecord::Base
12
14
  self.abstract_class = true
13
15
  begin
@@ -7,8 +7,22 @@ module Railwatch
7
7
  self.table_name = "railwatch_thresholds"
8
8
  MAX_LIMIT = 1_000_000_000
9
9
  MAX_WINDOW_MINUTES = 1_440
10
- TARGET_KINDS = %w[requests jobs commands queries scheduled_tasks outgoing_requests].freeze
11
- METRICS = %w[p95 max avg error_rate failure_rate missed].freeze
10
+ TARGET_KINDS = %w[requests jobs commands queries scheduled_tasks outgoing_requests
11
+ llm_calls llm_tools].freeze
12
+ METRICS = %w[p95 max avg error_rate failure_rate missed spend tokens truncation_rate].freeze
13
+
14
+ # Duration and rate metrics read the same on any timed record, so they are
15
+ # allowed everywhere. Money and tokens only mean something where a record
16
+ # carries them: a spend rule on tool calls could never fire, and a rule
17
+ # that cannot fire is worse than no rule -- it reads as coverage.
18
+ SHARED_METRICS = %w[p95 max avg error_rate failure_rate missed].freeze
19
+ LLM_METRICS = %w[spend tokens truncation_rate].freeze
20
+
21
+ # Money leads with its symbol; milliseconds and percent follow the number.
22
+ # The old code assumed two units and branched on the metric name ending in
23
+ # "rate", which cannot express a dollar sign in front.
24
+ UNITS = { "spend" => "$", "tokens" => "", "truncation_rate" => "%",
25
+ "error_rate" => "%", "failure_rate" => "%" }.freeze
12
26
 
13
27
  def environment = Environment.current
14
28
 
@@ -20,10 +34,33 @@ module Railwatch
20
34
  validates :target, presence: true, length: { maximum: 256 }
21
35
  validates :limit, numericality: { less_than_or_equal_to: 100 }, if: -> { metric&.end_with?("rate") }
22
36
  validates :target, uniqueness: { scope: %i[environment_id target_kind metric] }
37
+ validate :metric_applies_to_target_kind
38
+
39
+ def self.metrics_for(target_kind)
40
+ target_kind == "llm_calls" ? SHARED_METRICS + LLM_METRICS : SHARED_METRICS
41
+ end
42
+
43
+ def unit = UNITS.fetch(metric, "ms")
44
+
45
+ # "$4.50", "2000ms", "5%", "150000" -- the symbol's position is part of
46
+ # the unit, not something a caller should have to know.
47
+ def format_value(value)
48
+ return "$#{value.is_a?(Float) ? format('%.2f', value) : value}" if unit == "$"
49
+
50
+ "#{value}#{unit}"
51
+ end
23
52
 
24
53
  def description
25
- unit = metric.end_with?("rate") ? "%" : "ms"
26
- "#{target_kind} #{target == '*' ? 'all' : target}: #{metric} over #{limit}#{unit} in #{window_minutes}m"
54
+ "#{target_kind} #{target == '*' ? 'all' : target}: #{metric} over #{format_value(limit)} in #{window_minutes}m"
55
+ end
56
+
57
+ private
58
+
59
+ def metric_applies_to_target_kind
60
+ return if metric.blank? || target_kind.blank?
61
+ return if Threshold.metrics_for(target_kind).include?(metric)
62
+
63
+ errors.add(:metric, "is not available for #{target_kind}")
27
64
  end
28
65
  end
29
66
  end
@@ -0,0 +1,66 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Railwatch
4
+ # The time range a telemetry page, API call, or MCP tool looks at: one of the
5
+ # fixed presets ("1h" .. "30d") ending now, or a custom [from, to] pair. The
6
+ # dashboard, the JSON API, and MCP all resolve their ?window= / "window"
7
+ # argument through here so the presets and the default agree everywhere.
8
+ class Window
9
+ PRESETS = { "1h" => 1.hour, "6h" => 6.hours, "24h" => 24.hours, "7d" => 7.days, "30d" => 30.days }.freeze
10
+ DEFAULT = "24h"
11
+ MAX_CUSTOM_RANGE = 90.days
12
+
13
+ attr_reader :key, :from, :to
14
+
15
+ # A preset key, or anything else for the default.
16
+ def self.preset(key)
17
+ key = PRESETS.key?(key.to_s) ? key.to_s : DEFAULT
18
+ to = Time.current
19
+ new(key, to - PRESETS[key], to)
20
+ end
21
+
22
+ # A custom range from ?from=&to= when both parse, are in order, and span
23
+ # no more than MAX_CUSTOM_RANGE; otherwise the preset (or default).
24
+ def self.parse(window: nil, from: nil, to: nil)
25
+ custom(from, to) || preset(window)
26
+ end
27
+
28
+ def self.custom(from, to)
29
+ from = parse_time(from) or return
30
+ to = to.blank? ? Time.current : parse_time(to) or return
31
+ new("custom", from, to) if to > from && (to - from) <= MAX_CUSTOM_RANGE
32
+ end
33
+
34
+ def self.parse_time(value)
35
+ return if value.blank?
36
+ Time.zone.parse(value.to_s)
37
+ rescue ArgumentError, TypeError
38
+ nil
39
+ end
40
+ private_class_method :parse_time
41
+
42
+ def initialize(key, from, to)
43
+ @key = key
44
+ @from = from
45
+ @to = to
46
+ end
47
+
48
+ def custom? = key == "custom"
49
+ def range = [ from, to ]
50
+ def span = to - from
51
+
52
+ # The same length immediately before this one, for "vs. previous period".
53
+ def previous
54
+ self.class.new(key, from - span, from)
55
+ end
56
+
57
+ def to_h
58
+ { from: from.iso8601(6), to: to.iso8601(6) }
59
+ end
60
+
61
+ # Query params that reproduce this window on another page.
62
+ def to_params
63
+ custom? ? to_h : { window: key }
64
+ end
65
+ end
66
+ end