cronwatch 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. checksums.yaml +7 -0
  2. data/LICENSE +21 -0
  3. data/README.md +268 -0
  4. data/lib/cronwatch/abort_signal.rb +45 -0
  5. data/lib/cronwatch/active_record.rb +11 -0
  6. data/lib/cronwatch/alerts/console.rb +22 -0
  7. data/lib/cronwatch/alerts/custom.rb +26 -0
  8. data/lib/cronwatch/alerts/discord.rb +52 -0
  9. data/lib/cronwatch/alerts/slack.rb +58 -0
  10. data/lib/cronwatch/alerts/webhook.rb +46 -0
  11. data/lib/cronwatch/client.rb +925 -0
  12. data/lib/cronwatch/cron_pattern.rb +277 -0
  13. data/lib/cronwatch/duration.rb +77 -0
  14. data/lib/cronwatch/environment.rb +29 -0
  15. data/lib/cronwatch/evaluate.rb +345 -0
  16. data/lib/cronwatch/flight.rb +42 -0
  17. data/lib/cronwatch/format.rb +89 -0
  18. data/lib/cronwatch/http.rb +52 -0
  19. data/lib/cronwatch/job.rb +144 -0
  20. data/lib/cronwatch/js.rb +188 -0
  21. data/lib/cronwatch/monitored.rb +259 -0
  22. data/lib/cronwatch/output.rb +199 -0
  23. data/lib/cronwatch/rails/active_job.rb +55 -0
  24. data/lib/cronwatch/rails/check_job.rb +32 -0
  25. data/lib/cronwatch/rails/railtie.rb +37 -0
  26. data/lib/cronwatch/rails/tasks.rb +12 -0
  27. data/lib/cronwatch/rails.rb +35 -0
  28. data/lib/cronwatch/schedule.rb +191 -0
  29. data/lib/cronwatch/scheduler.rb +763 -0
  30. data/lib/cronwatch/serialize.rb +51 -0
  31. data/lib/cronwatch/sidekiq.rb +129 -0
  32. data/lib/cronwatch/stats.rb +23 -0
  33. data/lib/cronwatch/stores/active_record.rb +397 -0
  34. data/lib/cronwatch/stores/memory.rb +163 -0
  35. data/lib/cronwatch/ticker.rb +59 -0
  36. data/lib/cronwatch/triage/anthropic.rb +134 -0
  37. data/lib/cronwatch/types.rb +341 -0
  38. data/lib/cronwatch/version.rb +6 -0
  39. data/lib/cronwatch/walker.rb +137 -0
  40. data/lib/cronwatch/web/app.rb +484 -0
  41. data/lib/cronwatch/web/html.rb +314 -0
  42. data/lib/cronwatch/web.rb +17 -0
  43. data/lib/cronwatch/zone.rb +72 -0
  44. data/lib/cronwatch.rb +96 -0
  45. data/lib/generators/cronwatch/install/install_generator.rb +176 -0
  46. metadata +104 -0
@@ -0,0 +1,345 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Cronwatch
4
+ # Pure decisions about a job's health. Each function takes the current state
5
+ # and returns the new state plus the alerts that should go out. Nothing here
6
+ # touches a store or a network, which is what makes it testable.
7
+ module Evaluate
8
+ DEFAULT_GRACE_MS = 10 * 60_000
9
+ DEFAULT_TIMEOUT_MS = 60 * 60_000
10
+ # Runs faster than this are never called slow, whatever the baseline says.
11
+ SLOW_FLOOR_MS = 10_000
12
+ # How many earlier runs a baseline needs before it is trusted.
13
+ BASELINE_MIN_RUNS = 5
14
+ # How many successful runs a baseline looks at, and how many runs a summary covers.
15
+ BASELINE_WINDOW = 20
16
+
17
+ Evaluation = Struct.new(:state, :alerts, keyword_init: true)
18
+ CheckEvaluation = Struct.new(:state, :alerts, :next_expected_at, :due_at, keyword_init: true)
19
+ SlowThreshold = Struct.new(:threshold_ms, :basis, keyword_init: true)
20
+
21
+ module_function
22
+
23
+ def empty_state(job)
24
+ JobState.new(job: job, open: {}, consecutive_failures: 0, silenced_until: nil, last_alert_at: nil,
25
+ pending_recovery: [], undelivered: [])
26
+ end
27
+
28
+ # A stored state with every field present, or a fresh one. State written by an older version lacks the newer fields.
29
+ def normalize_state(state, job)
30
+ return empty_state(job) if state.nil?
31
+
32
+ state = JobState.from_h(state)
33
+ JobState.new(
34
+ job: state.job.nil? ? job : state.job,
35
+ open: (state.open || {}).dup,
36
+ consecutive_failures: state.consecutive_failures.nil? ? 0 : state.consecutive_failures,
37
+ silenced_until: state.silenced_until,
38
+ last_alert_at: state.last_alert_at,
39
+ pending_recovery: (state.pending_recovery || []).dup,
40
+ undelivered: (state.undelivered || []).dup,
41
+ version: state.version,
42
+ )
43
+ end
44
+
45
+ def clone_state(state)
46
+ normalize_state(state, state.job)
47
+ end
48
+
49
+ def open_condition(state, condition, now)
50
+ return false if state.open.key?(condition)
51
+
52
+ state.open[condition] = now
53
+ true
54
+ end
55
+
56
+ # Every open condition has alerted, so closing one owes a recovered message.
57
+ # It is remembered until a successful run leaves nothing open and sends it.
58
+ def close_condition(state, condition)
59
+ return false unless state.open.key?(condition)
60
+
61
+ state.open.delete(condition)
62
+ state.pending_recovery ||= []
63
+ state.pending_recovery << condition unless state.pending_recovery.include?(condition)
64
+ true
65
+ end
66
+
67
+ def open_conditions(state)
68
+ state.open.keys
69
+ end
70
+
71
+ def grace_ms(definition)
72
+ definition.grace.nil? ? DEFAULT_GRACE_MS : Duration.parse(definition.grace, "grace")
73
+ end
74
+
75
+ def timeout_ms(definition)
76
+ definition.timeout.nil? ? DEFAULT_TIMEOUT_MS : Duration.parse(definition.timeout, "timeout")
77
+ end
78
+
79
+ # Slow threshold for a successful run, or nil when there is nothing to compare against yet.
80
+ def slow_threshold(definition, history)
81
+ unless definition.max_duration.nil?
82
+ return SlowThreshold.new(threshold_ms: Duration.parse(definition.max_duration, "maxDuration"), basis: "maxDuration")
83
+ end
84
+
85
+ durations = history.select { |r| r.status == :ok && !r.duration_ms.nil? }.first(BASELINE_WINDOW).map(&:duration_ms)
86
+ return nil if durations.length < BASELINE_MIN_RUNS
87
+
88
+ p95 = Stats.percentile(durations, 95)
89
+ SlowThreshold.new(
90
+ threshold_ms: [2 * p95, SLOW_FLOOR_MS].max,
91
+ basis: "twice the p95 of the last #{durations.length} runs (#{Duration.format(p95)})",
92
+ )
93
+ end
94
+
95
+ def budget_breaches(definition, run, history)
96
+ breaches = []
97
+ budget = definition.budget
98
+ metrics = run.metrics || {}
99
+ JS.object_keys(metrics).each do |metric|
100
+ value = metrics[metric]
101
+ ceiling = budget&.[](metric.to_s)
102
+ unless ceiling.nil?
103
+ breaches << { metric: metric.to_s, value: value, limit: ceiling, basis: "budget" } if value > ceiling
104
+ next
105
+ end
106
+ past = history.select { |r| r.status == :ok && (r.metrics || {})[metric].is_a?(Numeric) }
107
+ .first(BASELINE_WINDOW).map { |r| r.metrics[metric] }
108
+ next if past.length < BASELINE_MIN_RUNS
109
+
110
+ usual = Stats.median(past)
111
+ if usual.positive? && value > 3 * usual
112
+ breaches << { metric: metric.to_s, value: value, limit: 3 * usual, basis: "three times the usual #{format_number(usual)}" }
113
+ end
114
+ end
115
+ breaches
116
+ end
117
+
118
+ # Whether `history` (newest first) holds a full baseline window of successful runs.
119
+ def full_baseline?(history)
120
+ history.count { |r| r.status == :ok } >= BASELINE_WINDOW
121
+ end
122
+
123
+ # toLocaleString("en-US"): digit groups, and a fraction rounded half up to
124
+ # at most four places. Worked on the shortest decimal digits, as ICU does.
125
+ def format_number(n)
126
+ return (n.negative? ? "-" : "") + (n.nan? ? "NaN" : "\u221e") if n.is_a?(Float) && !n.finite?
127
+
128
+ negative = n.negative? || (n.is_a?(Float) && n.zero? && (1.0 / n).negative?)
129
+ if n.is_a?(Integer)
130
+ whole = n.abs.to_s
131
+ fraction = ""
132
+ else
133
+ digits, point = JS.decimal(n.abs.to_f)
134
+ if point >= digits.length
135
+ whole = digits + ("0" * (point - digits.length))
136
+ fraction = ""
137
+ elsif point.positive?
138
+ whole = digits[0, point]
139
+ fraction = digits[point..]
140
+ else
141
+ whole = "0"
142
+ fraction = ("0" * -point) + digits
143
+ end
144
+ if fraction.length > 4
145
+ up = fraction[4].to_i >= 5
146
+ fraction = fraction[0, 4]
147
+ if up
148
+ rounded = ((whole + fraction).to_i + 1).to_s.rjust(whole.length + 4, "0")
149
+ whole = rounded[0...-4]
150
+ fraction = rounded[-4..]
151
+ end
152
+ end
153
+ fraction = fraction.sub(/0+\z/, "")
154
+ end
155
+ whole = whole.reverse.scan(/\d{1,3}/).join(",").reverse
156
+ text = fraction.empty? ? whole : "#{whole}.#{fraction}"
157
+ negative ? "-#{text}" : text
158
+ end
159
+
160
+ # Called when a run starts. Missed and stuck are about the absence of a run,
161
+ # so a run starting closes them without an alert; the recovered message
162
+ # waits for a successful finish.
163
+ def on_run_start(state)
164
+ next_state = clone_state(state)
165
+ close_condition(next_state, :missed)
166
+ close_condition(next_state, :stuck)
167
+ next_state
168
+ end
169
+
170
+ # Called when a run finishes with status ok, failed or timeout. `history` is
171
+ # the job's earlier runs, newest first, not including this one.
172
+ def on_run_finish(definition, run, state, history, now)
173
+ next_state = clone_state(state)
174
+ alerts = []
175
+
176
+ if run.status == :ok
177
+ next_state.consecutive_failures = 0
178
+ close_condition(next_state, :missed)
179
+ close_condition(next_state, :stuck)
180
+ close_condition(next_state, :failed)
181
+
182
+ slow = slow_threshold(definition, history)
183
+ if slow && !run.duration_ms.nil? && run.duration_ms > slow.threshold_ms
184
+ if open_condition(next_state, :slow, now)
185
+ alerts << AlertDraft.new(type: :slow, run: run,
186
+ details: { duration_ms: run.duration_ms, threshold_ms: slow.threshold_ms, basis: slow.basis })
187
+ end
188
+ else
189
+ close_condition(next_state, :slow)
190
+ end
191
+
192
+ breaches = budget_breaches(definition, run, history)
193
+ if breaches.any?
194
+ alerts << AlertDraft.new(type: :over_budget, run: run, details: { breaches: breaches }) if open_condition(next_state, :over_budget, now)
195
+ else
196
+ close_condition(next_state, :over_budget)
197
+ end
198
+
199
+ pending = next_state.pending_recovery || []
200
+ if pending.any? && open_conditions(next_state).empty?
201
+ alerts << AlertDraft.new(type: :recovered, run: run, details: { after: pending.dup })
202
+ next_state.pending_recovery = []
203
+ end
204
+ return Evaluation.new(state: next_state, alerts: alerts)
205
+ end
206
+
207
+ # failed or timeout
208
+ next_state.consecutive_failures += 1
209
+ close_condition(next_state, :missed)
210
+ threshold = [1, definition.failures_before_alert || 1].max
211
+ condition = run.status == :timeout ? :stuck : :failed
212
+ if next_state.consecutive_failures >= threshold && open_condition(next_state, condition, now)
213
+ alerts << AlertDraft.new(type: condition, run: run,
214
+ details: { consecutive_failures: next_state.consecutive_failures, threshold: threshold })
215
+ end
216
+ Evaluation.new(state: next_state, alerts: alerts)
217
+ end
218
+
219
+ # Called by check. Decides whether the schedule has been missed: the run
220
+ # the schedule wants next (see Schedule.expectation) has not started and its
221
+ # grace has run out. `last_run` is the most recent run of any status.
222
+ def on_check(definition, stored, last_run, state, now)
223
+ next_state = clone_state(state)
224
+ alerts = []
225
+ schedule = definition.schedule
226
+ if schedule.nil? || schedule == ""
227
+ return CheckEvaluation.new(state: next_state, alerts: alerts, next_expected_at: nil, due_at: nil)
228
+ end
229
+
230
+ parsed = Schedule.parse(schedule, definition.timezone)
231
+ grace = grace_ms(definition)
232
+ last_run_at = last_run&.started_at
233
+ exp = Schedule.expectation(parsed, last_run_at, stored.created_at, grace)
234
+ next_expected_at =
235
+ if parsed.interval? then Schedule.next_fire(parsed, stored.created_at, last_run_at)
236
+ else Schedule.next_fire(parsed, now, nil)
237
+ end
238
+ return CheckEvaluation.new(state: next_state, alerts: alerts, next_expected_at: next_expected_at, due_at: nil) unless exp
239
+
240
+ # An interval's next run is due a period after the last one started. If that
241
+ # run is still going, the job is busy, not late; stuck covers one that never ends.
242
+ if parsed.interval? && last_run&.status == :running
243
+ return CheckEvaluation.new(state: next_state, alerts: alerts, next_expected_at: next_expected_at, due_at: exp.due_at)
244
+ end
245
+
246
+ if now > exp.deadline
247
+ if open_condition(next_state, :missed, now)
248
+ alerts << AlertDraft.new(type: :missed, run: last_run,
249
+ details: { due_at: exp.due_at, deadline: exp.deadline, grace_ms: grace, last_run_at: last_run_at })
250
+ end
251
+ else
252
+ # A run has started since it opened, or the grace was widened.
253
+ close_condition(next_state, :missed)
254
+ end
255
+ CheckEvaluation.new(state: next_state, alerts: alerts, next_expected_at: next_expected_at, due_at: exp.due_at)
256
+ end
257
+
258
+ # Whether a running run has gone on longer than the job's timeout.
259
+ def stuck?(definition, run, now)
260
+ run.status == :running && now - run.started_at > timeout_ms(definition)
261
+ end
262
+
263
+ # While a job is silenced nothing new is recorded as an incident: conditions
264
+ # may close (so a job that recovered during the silence shows as healthy) but
265
+ # none may open, so the first problem after the silence ends alerts normally.
266
+ def mute_opens(previous, next_state)
267
+ muted = clone_state(next_state)
268
+ open_conditions(muted).each { |condition| muted.open.delete(condition) unless previous.open.key?(condition) }
269
+ muted
270
+ end
271
+
272
+ def silenced?(state, now)
273
+ !state.silenced_until.nil? && state.silenced_until > now
274
+ end
275
+
276
+ # An evaluation as it is saved and sent: while the job was silenced when it
277
+ # began, nothing opens and nothing is sent.
278
+ def apply_silence(previous, evaluation, now)
279
+ return evaluation unless silenced?(previous, now)
280
+
281
+ Evaluation.new(state: mute_opens(previous, evaluation.state), alerts: [])
282
+ end
283
+
284
+ # Whether an alert waiting to be retried no longer describes the job, so it
285
+ # is dropped rather than sent late. An alert for a condition is stale once
286
+ # that condition has closed, or has closed and opened again (it opened at a
287
+ # time other than the alert's). A recovery is stale when any condition it
288
+ # names is open again; while they all stay closed it is kept.
289
+ def stale_alert?(alert, state)
290
+ return Array(alert.details[:after]).any? { |condition| state.open.key?(condition.to_sym) } if alert.type == :recovered
291
+
292
+ state.open[alert.type] != alert.at
293
+ end
294
+
295
+ # How a job looks at a glance. Silence wins, then stuck, failing and late.
296
+ def job_health(definition, last_run, state, now)
297
+ open = open_conditions(state)
298
+ return :silenced if silenced?(state, now)
299
+ return :stuck if open.include?(:stuck) || (last_run && stuck?(definition, last_run, now))
300
+ return :failing if open.include?(:failed) || %i[failed timeout].include?(last_run&.status)
301
+ return :late if open.include?(:missed)
302
+ return :never_ran if last_run.nil?
303
+
304
+ :healthy
305
+ end
306
+
307
+ # A job's summary from its most recent runs (newest first; the first
308
+ # BASELINE_WINDOW are used) and its state. Stats cover runs of any status;
309
+ # the percentiles are over the successful ones among them.
310
+ def summarize(stored, recent, state, next_expected_at, now)
311
+ summary(stored, recent, state, next_expected_at) { |last_run| job_health(stored.definition, last_run, state, now) }
312
+ end
313
+
314
+ # The summary of a job that could not be evaluated, say because its stored
315
+ # schedule no longer parses. It reads nothing from the definition. The job
316
+ # shows as failing (or silenced, while it is), since it needs a look, and
317
+ # nothing is known about when it is next due.
318
+ def unevaluable_summary(stored, recent, state, now)
319
+ summary(stored, recent, state, nil) { silenced?(state, now) ? :silenced : :failing }
320
+ end
321
+
322
+ def summary(stored, recent, state, next_expected_at)
323
+ window = recent.first(BASELINE_WINDOW)
324
+ last_run = window.first
325
+ finished = window.reject { |r| r.status == :running }
326
+ ok_durations = window.select { |r| r.status == :ok && !r.duration_ms.nil? }.map(&:duration_ms)
327
+ JobSummary.new(
328
+ name: stored.name,
329
+ definition: stored.definition,
330
+ health: yield(last_run),
331
+ open: open_conditions(state),
332
+ last_run: last_run,
333
+ next_expected_at: next_expected_at,
334
+ consecutive_failures: state.consecutive_failures,
335
+ silenced_until: state.silenced_until,
336
+ stats: JobStats.new(
337
+ runs: finished.length,
338
+ ok_rate: finished.empty? ? 1 : finished.count { |r| r.status == :ok }.fdiv(finished.length),
339
+ p50_ms: Stats.percentile(ok_durations, 50),
340
+ p95_ms: Stats.percentile(ok_durations, 95),
341
+ ),
342
+ )
343
+ end
344
+ end
345
+ end
@@ -0,0 +1,42 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Cronwatch
4
+ class Client
5
+ # One shared check: the first caller runs it, the others wait for its result.
6
+ class Flight
7
+ def initialize
8
+ @lock = Mutex.new
9
+ @done = ConditionVariable.new
10
+ @finished = false
11
+ end
12
+
13
+ def resolve(value)
14
+ settle(value, nil)
15
+ end
16
+
17
+ def reject(error)
18
+ settle(nil, error)
19
+ end
20
+
21
+ def value
22
+ @lock.synchronize do
23
+ @done.wait(@lock) until @finished
24
+ raise @error if @error
25
+
26
+ @value
27
+ end
28
+ end
29
+
30
+ private
31
+
32
+ def settle(value, error)
33
+ @lock.synchronize do
34
+ @value = value
35
+ @error = error
36
+ @finished = true
37
+ @done.broadcast
38
+ end
39
+ end
40
+ end
41
+ end
42
+ end
@@ -0,0 +1,89 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Cronwatch
4
+ module Format
5
+ NAMED = /\A[A-Za-z_$][A-Za-z0-9_$]*: /
6
+
7
+ module_function
8
+
9
+ def at_time(at, now)
10
+ return "never" if at.nil?
11
+
12
+ "#{JS.iso(at).tr("T", " ")[0, 19]} UTC (#{Duration.relative(at, now)})"
13
+ end
14
+
15
+ def first_lines(text, n)
16
+ return "" if text.nil? || text.empty?
17
+
18
+ text.split("\n", -1).first(n).join("\n")
19
+ end
20
+
21
+ def tail(text, n)
22
+ return "" if text.nil? || text.empty?
23
+
24
+ lines = JS.trim_end(text).split("\n", -1)
25
+ lines.last(n).join("\n")
26
+ end
27
+
28
+ # "Error: x" for a bare message, but not "Error: TypeError: x" for one that already names itself.
29
+ def error_line(error)
30
+ text = first_lines(error, 4)
31
+ NAMED.match?(text) ? text : "Error: #{text}"
32
+ end
33
+
34
+ # Turns a draft into the title and message every channel shows.
35
+ def compose_alert(draft, definition, now)
36
+ name = definition.name
37
+ run = draft.run
38
+ details = draft.details
39
+ lines = []
40
+
41
+ title =
42
+ case draft.type
43
+ when :missed
44
+ lines << "Due #{at_time(details[:due_at], now)}, and no run had started by #{at_time(details[:deadline], now)} (grace #{Duration.format(details[:grace_ms])})."
45
+ lines << "Schedule: #{definition.schedule.nil? ? "undefined" : definition.schedule}#{definition.timezone.to_s.empty? ? "" : " (#{definition.timezone})"}."
46
+ lines << "Last run: #{run ? "#{run.status} #{at_time(run.started_at, now)}" : "never"}."
47
+ "#{name} missed its scheduled run"
48
+ when :failed
49
+ n = details[:consecutive_failures]
50
+ lines << "#{n} consecutive failures." if n > 1
51
+ if run
52
+ lines << "Started #{at_time(run.started_at, now)}#{run.duration_ms.nil? ? "" : ", ran #{Duration.format(run.duration_ms)}"}."
53
+ lines << error_line(run.error) if run.error && !run.error.empty?
54
+ out = tail(run.output, 8)
55
+ lines << "Output (tail):\n#{out}" unless out.empty?
56
+ end
57
+ "#{name} failed"
58
+ when :stuck
59
+ if run
60
+ lines << "Started #{at_time(run.started_at, now)} and never reported finishing. Marked as timed out after #{Duration.format(run.duration_ms.nil? ? now - run.started_at : run.duration_ms)}."
61
+ out = tail(run.output, 8)
62
+ lines << "Output so far (tail):\n#{out}" unless out.empty?
63
+ end
64
+ lines << "If the process was killed mid-run (a serverless timeout, a deploy), this is what that looks like."
65
+ "#{name} is stuck"
66
+ when :slow
67
+ lines << "Took #{Duration.format(details[:duration_ms])}; the limit is #{Duration.format(details[:threshold_ms])} (#{details[:basis]})."
68
+ lines << "Started #{at_time(run.started_at, now)}." if run
69
+ "#{name} was slow"
70
+ when :over_budget
71
+ details[:breaches].each do |b|
72
+ lines << "#{b[:metric]}: #{Evaluate.format_number(b[:value])}, limit #{Evaluate.format_number(b[:limit])} (#{b[:basis]})."
73
+ end
74
+ lines << "Started #{at_time(run.started_at, now)}." if run
75
+ "#{name} went over budget"
76
+ when :recovered
77
+ after = details[:after].map { |c| c.to_s.sub("_", " ") }.join(", ")
78
+ lines << "A run #{run ? at_time(run.started_at, now) : "just now"} succeeded#{after.empty? ? "" : " after: #{after}"}."
79
+ lines << "Ran #{Duration.format(run.duration_ms)}." if run && !run.duration_ms.nil?
80
+ "#{name} recovered"
81
+ else
82
+ raise ArgumentError, "unknown alert type #{draft.type.inspect}"
83
+ end
84
+
85
+ Alert.new(type: draft.type, run: run, details: details, job: name, definition: definition, title: title,
86
+ message: lines.join("\n"), at: now)
87
+ end
88
+ end
89
+ end
@@ -0,0 +1,52 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "net/http"
4
+ require "uri"
5
+
6
+ module Cronwatch
7
+ # POSTs for the alert channels. Anything with `post(url, body, headers)`
8
+ # returning a Response can stand in for it, which is how the tests run.
9
+ module HTTP
10
+ TIMEOUT = 10
11
+
12
+ Response = Struct.new(:status, :body, keyword_init: true) do
13
+ def ok? = status >= 200 && status < 300
14
+ end
15
+
16
+ class NetHTTP
17
+ def post(url, body, headers)
18
+ uri = URI(url)
19
+ request = Net::HTTP::Post.new(uri.request_uri)
20
+ headers.each { |k, v| request[k] = v }
21
+ request.body = body
22
+ response = connection(uri).request(request)
23
+ Response.new(status: response.code.to_i, body: response.body.to_s)
24
+ end
25
+
26
+ # A connection that gives up after ten seconds, as the SDK's requests do.
27
+ def connection(uri)
28
+ http = Net::HTTP.new(uri.host, uri.port)
29
+ http.use_ssl = uri.scheme == "https"
30
+ http.open_timeout = TIMEOUT
31
+ http.read_timeout = TIMEOUT
32
+ http.write_timeout = TIMEOUT
33
+ http
34
+ end
35
+ end
36
+
37
+ module_function
38
+
39
+ def default
40
+ @default ||= NetHTTP.new
41
+ end
42
+
43
+ # Compares two secrets without stopping at the first differing character.
44
+ def constant_time_equal?(a, b)
45
+ return false unless a.bytesize == b.bytesize
46
+
47
+ diff = 0
48
+ a.bytes.zip(b.bytes) { |x, y| diff |= x ^ y }
49
+ diff.zero?
50
+ end
51
+ end
52
+ end
@@ -0,0 +1,144 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "abort_signal"
4
+
5
+ module Cronwatch
6
+ # What a job's block receives.
7
+ class JobContext
8
+ attr_reader :name, :run_id, :started_at, :signal
9
+
10
+ def initialize(run, signal, recorder)
11
+ @name = run.job
12
+ @run_id = run.id
13
+ @started_at = run.started_at
14
+ @signal = signal
15
+ @recorder = recorder
16
+ end
17
+
18
+ # Append a line of output. Kept with the run, capped at 16 KB, shown in alerts and the dashboard.
19
+ def log(*parts)
20
+ @recorder.log(parts.map { |part| JobContext.stringify(part) }.join(" "))
21
+ nil
22
+ end
23
+
24
+ # Report a number for this run: tokens, cost, rows, anything. Watched against budgets and baselines.
25
+ def metric(name, value)
26
+ unless value.is_a?(Numeric) && value.real? && JS.finite?(value)
27
+ raise ArgumentError, "metric \"#{name}\" must be a finite number"
28
+ end
29
+
30
+ @recorder.metric(Output.utf8(name), value)
31
+ nil
32
+ end
33
+
34
+ def metrics(values = nil, **more)
35
+ (values || {}).merge(more).each { |k, v| metric(k, v) }
36
+ nil
37
+ end
38
+
39
+ # True once the job's timeout has passed. Honour it if the work can stop.
40
+ def aborted?
41
+ @signal.aborted?
42
+ end
43
+
44
+ # A logged value as text, the way the SDK writes it, always valid UTF-8.
45
+ def self.stringify(part)
46
+ text =
47
+ case part
48
+ when String then part
49
+ when Symbol then part.to_s
50
+ when Exception then "#{part.class.name || part.class}: #{Output.utf8(part.message)}"
51
+ when Hash, Array, Numeric, true, false, nil, Struct then JS.json(part)
52
+ else part.to_s
53
+ end
54
+ Output.utf8(text)
55
+ end
56
+ end
57
+
58
+ # A declared job. Keep it, and call #run with a block.
59
+ class JobHandle
60
+ attr_reader :name, :definition
61
+
62
+ def initialize(client, definition)
63
+ @client = client
64
+ @definition = definition
65
+ @name = definition.name
66
+ end
67
+
68
+ # Run the block now, recording the run. Returns what the block returns and
69
+ # re-raises what it raises, after the run is recorded.
70
+ def run(trigger: "run", &block)
71
+ raise ArgumentError, "run needs a block" unless block
72
+
73
+ outcome = @client.execute(@definition, trigger, &block)
74
+ raise outcome.error if outcome.threw
75
+
76
+ outcome.result
77
+ end
78
+ end
79
+
80
+ # Collects a run's output and metrics while its block runs.
81
+ class RunRecorder
82
+ # Lines are dropped from the front once the output is well past the cap;
83
+ # Output.cap trims it exactly at the end.
84
+ KEEP = 64 * 1024
85
+
86
+ attr_reader :context, :signal
87
+
88
+ def initialize(run, timeout_ms)
89
+ @lines = []
90
+ @size = 0
91
+ # The first lines logged, up to the cap, and whether any line has been
92
+ # dropped from @lines: what expect_text needs once the output runs long.
93
+ @head = []
94
+ @head_size = 0
95
+ @dropped = false
96
+ @metrics = {}
97
+ @lock = Mutex.new
98
+ @signal = AbortSignal.new(timeout_ms)
99
+ @context = JobContext.new(run, @signal, self)
100
+ end
101
+
102
+ def log(line)
103
+ length = JS.length16(line)
104
+ @lock.synchronize do
105
+ if @head_size < Output::CAP
106
+ @head << line
107
+ @head_size += length + 1
108
+ end
109
+ @lines << line
110
+ @size += length + 1
111
+ while @size > KEEP && @lines.length > 1
112
+ @size -= JS.length16(@lines.shift) + 1
113
+ @dropped = true
114
+ end
115
+ end
116
+ end
117
+
118
+ def metric(name, value)
119
+ @lock.synchronize { @metrics[name] = value }
120
+ end
121
+
122
+ def output
123
+ @lock.synchronize { @lines.empty? ? nil : Output.cap(@lines.join("\n")) }
124
+ end
125
+
126
+ # What an expect rule is checked against: everything logged, or when that
127
+ # ran long, the first 16 KB and the last 16 KB. The stored output keeps
128
+ # only the tail, so a "done" line printed early would otherwise be lost.
129
+ def expect_text
130
+ @lock.synchronize do
131
+ next nil if @lines.empty?
132
+
133
+ all = @lines.join("\n")
134
+ next all if !@dropped && JS.length16(all) <= 2 * Output::CAP
135
+
136
+ "#{JS.head16(@head.join("\n"), Output::CAP)}\n#{JS.tail16(all, Output::CAP)}"
137
+ end
138
+ end
139
+
140
+ def metrics
141
+ @lock.synchronize { @metrics.dup }
142
+ end
143
+ end
144
+ end