cronwatch 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. checksums.yaml +7 -0
  2. data/LICENSE +21 -0
  3. data/README.md +268 -0
  4. data/lib/cronwatch/abort_signal.rb +45 -0
  5. data/lib/cronwatch/active_record.rb +11 -0
  6. data/lib/cronwatch/alerts/console.rb +22 -0
  7. data/lib/cronwatch/alerts/custom.rb +26 -0
  8. data/lib/cronwatch/alerts/discord.rb +52 -0
  9. data/lib/cronwatch/alerts/slack.rb +58 -0
  10. data/lib/cronwatch/alerts/webhook.rb +46 -0
  11. data/lib/cronwatch/client.rb +925 -0
  12. data/lib/cronwatch/cron_pattern.rb +277 -0
  13. data/lib/cronwatch/duration.rb +77 -0
  14. data/lib/cronwatch/environment.rb +29 -0
  15. data/lib/cronwatch/evaluate.rb +345 -0
  16. data/lib/cronwatch/flight.rb +42 -0
  17. data/lib/cronwatch/format.rb +89 -0
  18. data/lib/cronwatch/http.rb +52 -0
  19. data/lib/cronwatch/job.rb +144 -0
  20. data/lib/cronwatch/js.rb +188 -0
  21. data/lib/cronwatch/monitored.rb +259 -0
  22. data/lib/cronwatch/output.rb +199 -0
  23. data/lib/cronwatch/rails/active_job.rb +55 -0
  24. data/lib/cronwatch/rails/check_job.rb +32 -0
  25. data/lib/cronwatch/rails/railtie.rb +37 -0
  26. data/lib/cronwatch/rails/tasks.rb +12 -0
  27. data/lib/cronwatch/rails.rb +35 -0
  28. data/lib/cronwatch/schedule.rb +191 -0
  29. data/lib/cronwatch/scheduler.rb +763 -0
  30. data/lib/cronwatch/serialize.rb +51 -0
  31. data/lib/cronwatch/sidekiq.rb +129 -0
  32. data/lib/cronwatch/stats.rb +23 -0
  33. data/lib/cronwatch/stores/active_record.rb +397 -0
  34. data/lib/cronwatch/stores/memory.rb +163 -0
  35. data/lib/cronwatch/ticker.rb +59 -0
  36. data/lib/cronwatch/triage/anthropic.rb +134 -0
  37. data/lib/cronwatch/types.rb +341 -0
  38. data/lib/cronwatch/version.rb +6 -0
  39. data/lib/cronwatch/walker.rb +137 -0
  40. data/lib/cronwatch/web/app.rb +484 -0
  41. data/lib/cronwatch/web/html.rb +314 -0
  42. data/lib/cronwatch/web.rb +17 -0
  43. data/lib/cronwatch/zone.rb +72 -0
  44. data/lib/cronwatch.rb +96 -0
  45. data/lib/generators/cronwatch/install/install_generator.rb +176 -0
  46. metadata +104 -0
@@ -0,0 +1,925 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "monitor"
4
+ require "securerandom"
5
+ require "set"
6
+ require_relative "abort_signal"
7
+ require_relative "environment"
8
+ require_relative "flight"
9
+ require_relative "ticker"
10
+
11
+ module Cronwatch
12
+ # Raised (and handed to on_error) when a channel or triage takes too long,
13
+ # or is skipped because its previous call has not finished.
14
+ class TimeoutError < StandardError; end
15
+
16
+ # What callers waiting on a shared check see when the check was stopped by
17
+ # an exception outside StandardError (Interrupt, Timeout, SystemExit). The
18
+ # caller that ran the check sees the original exception.
19
+ class InterruptedError < StandardError; end
20
+
21
+ class Client
22
+ NAME_RE = /\A[A-Za-z0-9][A-Za-z0-9._:-]{0,119}\z/
23
+ TRIAGE_TIMEOUT_MS = 25_000
24
+ # How long one channel may take to send one alert.
25
+ CHANNEL_TIMEOUT_MS = 15_000
26
+ PRUNE_INTERVAL_MS = 60 * 60_000
27
+ # Undelivered alerts kept per job for retry; the oldest go first.
28
+ MAX_UNDELIVERED = 20
29
+ # Wall-clock time one check spends retrying undelivered alerts, across
30
+ # every job. Once it is spent the rest wait for the next check.
31
+ RETRY_BUDGET_MS = 20_000
32
+ # Reads and writes of one job's state before an update gives up on a store that keeps changing under it.
33
+ STATE_ATTEMPTS = 10
34
+ # Runs read for a baseline, and the most read when failures crowd out the successes.
35
+ HISTORY_PAGE = Evaluate::BASELINE_WINDOW + 5
36
+ HISTORY_MAX = 200
37
+ DEFAULT_OPTIONS = %i[grace timeout timezone failures_before_alert].freeze
38
+ # Tells "cron_secret not given" (read CRON_SECRET) from "cron_secret: nil" (no secret on purpose).
39
+ UNSET = Object.new.freeze
40
+ # Guards the reset a forked child makes of the parent's locks and threads.
41
+ FORK_LOCK = Mutex.new
42
+ SILENCE_OPTIONS = %i[for].freeze
43
+
44
+ # What execute returns: the recorded run, and the block's own outcome.
45
+ ExecuteResult = Struct.new(:run, :result, :error, :threw, keyword_init: true)
46
+ # What a triage callable receives. Pass the signal to anything that can stop early.
47
+ TriageContext = Struct.new(:alert, :recent_runs, :signal, keyword_init: true)
48
+ # A job's summary and its newest runs, as the dashboard shows them.
49
+ JobWithRuns = Struct.new(:job, :runs, keyword_init: true) do
50
+ include Serializable
51
+
52
+ def to_h = { "job" => job.to_h, "runs" => runs.map(&:to_h) }
53
+ end
54
+
55
+ attr_reader :store, :alerts, :triage, :cron_secret, :retention_ms, :defaults
56
+
57
+ # store: where jobs, runs and state live. Defaults to an in-memory store that forgets on restart.
58
+ # alerts: where alerts go: objects with #call(alert) and #name. Defaults to the console.
59
+ # triage: a callable taking a TriageContext and returning a short diagnosis, added to every alert but recoveries.
60
+ # cron_secret: a second bearer Cronwatch::Web accepts for /api/check, for an outside cron. Defaults to
61
+ # ENV["CRON_SECRET"]; an empty string counts as unset. Pass nil for none.
62
+ # retention: how long finished runs are kept. Default "30d".
63
+ # defaults: grace, timeout, timezone and failures_before_alert applied to every job unless it sets its own.
64
+ # redact: applied to every run's output and error before it is stored, shown or sent to an alert
65
+ # channel or triage. The default (Output.redact_secrets) blanks values that look like secrets
66
+ # (password=..., Authorization headers, URL credentials, bearer tokens, JWTs, PEM private
67
+ # keys, webhook URLs, AWS, GitHub, Slack, Stripe, Google and API key formats). Pass your own
68
+ # callable, or false to keep output exactly as logged. A callable that raises or returns
69
+ # something other than a String is reported to on_error ("redact") and the default is used.
70
+ # now: the clock, a callable returning epoch milliseconds. Tests use this.
71
+ # on_error: called with (error, where) for anything that goes wrong outside a job: the store failing,
72
+ # an alert channel failing, a triage timeout.
73
+ def initialize(store: nil, alerts: nil, triage: nil, cron_secret: UNSET, retention: "30d", defaults: {}, redact: nil,
74
+ deliver: :now, now: nil, on_error: nil)
75
+ @using_default_store = store.nil?
76
+ @store = store || Stores::Memory.new
77
+ @alerts = alerts.nil? ? [Alerts::Console.new] : Array(alerts)
78
+ @triage = triage
79
+ secret = cron_secret.equal?(UNSET) ? ENV.fetch("CRON_SECRET", nil) : cron_secret
80
+ @cron_secret = secret.nil? || secret.to_s.empty? ? nil : secret.to_s
81
+ @retention_ms = Duration.parse(retention || "30d", "retention")
82
+ @defaults = (defaults || {}).transform_keys(&:to_sym)
83
+ unknown = @defaults.keys - DEFAULT_OPTIONS
84
+ raise ArgumentError, "defaults may set #{DEFAULT_OPTIONS.join(", ")}, not #{unknown.join(", ")}" if unknown.any?
85
+
86
+ if !(redact.nil? || redact == false || redact.respond_to?(:call))
87
+ raise ArgumentError, "redact must be a callable, or false to keep output as logged"
88
+ end
89
+
90
+ @redact =
91
+ if redact == false then ->(text) { text }
92
+ elsif redact.nil? then Output.method(:redact_secrets)
93
+ else guarded_redact(redact)
94
+ end
95
+ unless [:now, :check, "now", "check", nil].include?(deliver)
96
+ raise ArgumentError, "deliver must be \"now\" or \"check\", not #{deliver.inspect}"
97
+ end
98
+
99
+ # :check queues alerts in the store for another process's check to send
100
+ # (see "Delivery" in DESIGN.md, and deliver in the SDK).
101
+ @defer_delivery = deliver.to_s == "check"
102
+ @clock = now || -> { Process.clock_gettime(Process::CLOCK_REALTIME, :millisecond) }
103
+ @on_error = on_error || method(:default_on_error)
104
+ @definitions = {}
105
+ @synced = Set.new
106
+ @registry = Mutex.new
107
+ @ready = false
108
+ @last_prune_at = 0
109
+ # Seconds before start()'s first check, and how long a channel or triage may take. Tests shorten them.
110
+ @first_tick_s = 1.0
111
+ @channel_timeout_ms = CHANNEL_TIMEOUT_MS
112
+ @triage_timeout_ms = TRIAGE_TIMEOUT_MS
113
+ @retry_budget_ms = RETRY_BUDGET_MS
114
+ @warned_deferred_start = false
115
+ reset_process_state
116
+ end
117
+
118
+ # Epoch milliseconds, from the clock the client was given.
119
+ def now
120
+ @clock.call
121
+ end
122
+
123
+ # Declare a job. Call it once, when the app loads, and keep the handle.
124
+ def job(name, **options)
125
+ name = name.to_s if name.is_a?(Symbol)
126
+ unless name.is_a?(String) && NAME_RE.match?(name)
127
+ raise ArgumentError, "job name \"#{name}\" must be 1 to 120 characters of letters, digits, \".\", \"_\", \":\" or \"-\""
128
+ end
129
+
130
+ definition = build_definition(name, options)
131
+ validate_definition(definition)
132
+ @registry.synchronize do
133
+ @definitions[name] = definition
134
+ @synced.delete(name)
135
+ end
136
+ JobHandle.new(self, definition)
137
+ end
138
+
139
+ # With a block: run a job by name without keeping a handle, declaring it
140
+ # on first use (or again, when options are given). Without a block: the
141
+ # run with this id, as get_run.
142
+ def run(name_or_id, **options, &block)
143
+ unless block
144
+ raise ArgumentError, "run(#{name_or_id.inspect}, ...) needs a block; without one, run(id) reads a run" if options.any?
145
+
146
+ return get_run(name_or_id)
147
+ end
148
+
149
+ declared = @registry.synchronize { @definitions[name_or_id.to_s] }
150
+ handle = options.any? || declared.nil? ? job(name_or_id, **options) : JobHandle.new(self, declared)
151
+ handle.run(&block)
152
+ end
153
+
154
+ # The definitions declared in this process.
155
+ def defined_jobs
156
+ @registry.synchronize { @definitions.values }
157
+ end
158
+
159
+ # Runs a block as a recorded run. The block always runs, whatever the store
160
+ # is doing: store errors go to on_error, and the result is the block's own
161
+ # outcome. A StandardError from the block is not raised; see `threw`. An
162
+ # exception outside StandardError (Interrupt, SystemExit, Sidekiq::Shutdown,
163
+ # a Timeout, NotImplementedError) is recorded as a failed run and then
164
+ # raised again, so the run is never left running.
165
+ #
166
+ # `failure` is an optional callable that turns the block's result into an
167
+ # error message, or nil when the result is fine: an HTTP handler uses it to
168
+ # count a 500 response as a failed run.
169
+ #
170
+ # @api private For integrations (JobHandle#run, ActiveJob, Sidekiq), not apps.
171
+ def execute(definition, trigger, failure: nil)
172
+ after_fork_check
173
+ name = definition.name
174
+ started_at = now
175
+ run = Run.new(id: SecureRandom.uuid, job: name, status: :running, started_at: started_at, finished_at: nil,
176
+ duration_ms: nil, error: nil, output: nil, metrics: {}, trigger: trigger)
177
+ recorded = false
178
+ begin
179
+ sync(definition)
180
+ @store.insert_run(run.dup)
181
+ recorded = true
182
+ rescue StandardError => e
183
+ report(e, "recording #{name}")
184
+ end
185
+ # The SDK closes missed and stuck beside the running job; here it is done
186
+ # just before the block runs. The result is the same.
187
+ if recorded
188
+ begin
189
+ update_state(name) { |before| [Evaluate.on_run_start(before), nil] }
190
+ rescue StandardError => e
191
+ report(e, "starting #{name}")
192
+ end
193
+ end
194
+
195
+ recorder = RunRecorder.new(run, Evaluate.timeout_ms(definition))
196
+ result = nil
197
+ error = nil
198
+ threw = false
199
+ begin
200
+ result = yield(recorder.context)
201
+ rescue Exception => e # rubocop:disable Lint/RescueException -- recorded, then raised again below
202
+ error = e
203
+ threw = true
204
+ ensure
205
+ recorder.signal.settle!
206
+ end
207
+
208
+ finished_at = now
209
+ run.finished_at = finished_at
210
+ run.duration_ms = [0, finished_at - started_at].max
211
+ run.metrics = recorder.metrics
212
+ returned = result.is_a?(String) ? Output.utf8(result) : nil
213
+ run.output = recorder.output || (returned && Output.cap(returned))
214
+
215
+ if threw
216
+ run.status = :failed
217
+ run.error = Output.error_message(error)
218
+ elsif (problem = failure&.call(result))
219
+ run.status = :failed
220
+ run.error = Output.utf8(problem.to_s)
221
+ else
222
+ expect_text = recorder.expect_text || returned
223
+ unmet = Serialize.check_expectation(definition.expect, expect_text)
224
+ if unmet
225
+ run.status = :failed
226
+ run.error = unmet
227
+ else
228
+ run.status = :ok
229
+ end
230
+ end
231
+ # Redacted after the expect check, so a rule can still match what was
232
+ # logged. NULs go last, so not even a custom redact can store one.
233
+ run.output = Output.strip_nul(@redact.call(run.output)) unless run.output.nil?
234
+ run.error = Output.strip_nul(@redact.call(run.error)) unless run.error.nil?
235
+
236
+ stored = Serialize.to_stored(definition)
237
+ if recorded && marked_timed_out?(run)
238
+ # A check gave up on this run while it was going and already counted
239
+ # it as a stuck failure. A late failure must not count twice; a late
240
+ # success still closes stuck and recovers.
241
+ begin
242
+ @store.update_run(run)
243
+ rescue StandardError => e
244
+ report(e, "recording #{name}")
245
+ end
246
+ elsif recorded
247
+ finish_run(stored, run, finished_at, true)
248
+ else
249
+ # The start was never written; the store may be back by now.
250
+ begin
251
+ sync(definition)
252
+ @store.insert_run(run)
253
+ recorded = true
254
+ rescue StandardError => e
255
+ report(e, "recording #{name}")
256
+ end
257
+ finish_run(stored, run, finished_at, false) if recorded
258
+ end
259
+
260
+ raise error if threw && interrupted?(error)
261
+
262
+ ExecuteResult.new(run: run, result: result, error: error, threw: threw)
263
+ end
264
+
265
+ # Look for missed and stuck runs across every job, send alerts, retry
266
+ # alerts no channel accepted, and prune old runs. Call it from start(), a
267
+ # scheduled job (Cronwatch::CheckJob), or by hand. Concurrent calls share
268
+ # one check.
269
+ def check
270
+ after_fork_check
271
+ flight = nil
272
+ mine = false
273
+ @check_lock.synchronize do
274
+ flight = @checking ||= begin
275
+ mine = true
276
+ Flight.new
277
+ end
278
+ end
279
+ if mine
280
+ begin
281
+ flight.resolve(run_check)
282
+ rescue Exception => e # rubocop:disable Lint/RescueException -- the waiters must not hang
283
+ # An Interrupt or a Timeout is meant for this thread only, so the
284
+ # waiters get an error of their own and this thread the original.
285
+ flight.reject(interrupted?(e) ? InterruptedError.new("the check was interrupted by #{e.class}") : e)
286
+ raise if interrupted?(e)
287
+ ensure
288
+ @check_lock.synchronize { @checking = nil if @checking.equal?(flight) }
289
+ end
290
+ end
291
+ flight.value
292
+ end
293
+
294
+ # Every job the store knows about, with its health. Does not send alerts.
295
+ def jobs
296
+ jobs_with_runs(0).map(&:job)
297
+ end
298
+
299
+ # Every job's summary with its newest `limit` runs, read together. What the dashboard shows.
300
+ def jobs_with_runs(limit = 20)
301
+ after_fork_check
302
+ ensure_ready
303
+ defined_jobs.each { |definition| sync(definition) }
304
+ at = now
305
+ count = clamp_limit(limit, 20, 0)
306
+ @store.list_jobs.map { |stored| snapshot(stored, at, count) }
307
+ end
308
+
309
+ def job_summary(name)
310
+ ensure_ready
311
+ definition = @registry.synchronize { @definitions[name] }
312
+ sync(definition) if definition
313
+ stored = @store.get_job(name)
314
+ return nil unless stored
315
+
316
+ snapshot(stored, now, 0).job
317
+ end
318
+
319
+ # A job's runs, newest first. `limit` is a whole number from 1 to 500.
320
+ def runs(name, limit = 50)
321
+ ensure_ready
322
+ @store.list_runs(name, clamp_limit(limit, 50, 1))
323
+ end
324
+
325
+ def get_run(id)
326
+ ensure_ready
327
+ @store.get_run(id)
328
+ end
329
+
330
+ # Stop alerts for a job for a while. State keeps updating underneath.
331
+ # silence("nightly-report", for: "2h") # or silence("nightly-report", "2h")
332
+ def silence(name, duration = nil, **options)
333
+ unknown = options.keys - SILENCE_OPTIONS
334
+ raise ArgumentError, "silence takes for:, not #{unknown.map(&:inspect).join(", ")}" if unknown.any?
335
+ raise ArgumentError, "silence takes a duration or for:, not both" if !duration.nil? && options.key?(:for)
336
+
337
+ duration = options[:for] if duration.nil?
338
+ ms = Duration.parse(duration, "silence duration")
339
+ patch_state(name) { |state| state.silenced_until = now + ms }
340
+ end
341
+
342
+ def unsilence(name)
343
+ patch_state(name) { |state| state.silenced_until = nil }
344
+ end
345
+
346
+ # Remove a job and its runs from the store. A job still declared in code comes back on its next run.
347
+ def forget(name)
348
+ ensure_ready
349
+ @registry.synchronize do
350
+ @definitions.delete(name)
351
+ @synced.delete(name)
352
+ end
353
+ @store.delete_job(name)
354
+ nil
355
+ end
356
+
357
+ # Check on an interval, in a background thread, for long-running
358
+ # processes. Default every minute; the first check comes after a second.
359
+ # Calling it again while it runs does nothing, and a different interval
360
+ # is reported to on_error and ignored: stop first to change it. A forked
361
+ # child (Puma, Unicorn, Sidekiq) has no thread, so call start there.
362
+ def start(every = "1m")
363
+ after_fork_check
364
+ ms = [5_000, Duration.parse(every, "check interval")].max
365
+ @ticker_lock.synchronize do
366
+ if @ticker
367
+ unless @ticker_ms == ms
368
+ report(ArgumentError.new("start(#{every.inspect}) ignored: already checking every #{Duration.format(@ticker_ms)}; " \
369
+ "call stop first to change it"), "start")
370
+ end
371
+ return nil
372
+ end
373
+
374
+ @ticker_ms = ms
375
+ if @defer_delivery && !@warned_deferred_start
376
+ @warned_deferred_start = true
377
+ warn '[cronwatch] start() was called with deliver: "check", so these checks send no alerts. ' \
378
+ 'Another process must run checks with deliver: "now" (the default) to send them.'
379
+ end
380
+ @ticker = Ticker.new(ms / 1000.0, @first_tick_s) do
381
+ check
382
+ rescue StandardError => e
383
+ report(e, "check")
384
+ end
385
+ end
386
+ nil
387
+ end
388
+
389
+ def stop
390
+ after_fork_check
391
+ ticker = @ticker_lock.synchronize do
392
+ current = @ticker
393
+ @ticker = nil
394
+ current
395
+ end
396
+ ticker&.stop
397
+ nil
398
+ end
399
+
400
+ def close
401
+ stop
402
+ @store.close if @store.respond_to?(:close)
403
+ nil
404
+ end
405
+
406
+ # True in development or test. See Cronwatch::Environment.
407
+ def self.development?
408
+ Environment.development?
409
+ end
410
+
411
+ # Hands an error to on_error. An on_error that raises is not allowed to
412
+ # take the job down with it.
413
+ #
414
+ # @api private For integrations (Cronwatch::Web, ActiveJob, Sidekiq), not apps.
415
+ def report(error, where)
416
+ @on_error.call(error, where)
417
+ rescue StandardError => e
418
+ warn "[cronwatch] on_error raised #{e.class}: #{e.message} (reporting #{where}: #{error.message})"
419
+ end
420
+
421
+ private
422
+
423
+ # Locks, the check in flight, the interval thread and the channel and
424
+ # triage threads belong to one process. A forked child (Puma, Unicorn,
425
+ # Sidekiq) starts with fresh ones, so start and check work there.
426
+ def reset_process_state
427
+ @pid = Process.pid
428
+ @locks = {}
429
+ @check_lock = Mutex.new
430
+ @checking = nil
431
+ @ticker_lock = Mutex.new
432
+ @ticker = nil
433
+ @ticker_ms = nil
434
+ @ready_lock = Mutex.new
435
+ @sending_lock = Mutex.new
436
+ # Channel (by index) and triage threads that timed out and are still going.
437
+ @abandoned = {}
438
+ end
439
+
440
+ def after_fork_check
441
+ return if @pid == Process.pid
442
+
443
+ FORK_LOCK.synchronize { reset_process_state unless @pid == Process.pid }
444
+ end
445
+
446
+ # An exception outside StandardError: the thread is being stopped
447
+ # (Interrupt, SystemExit, Sidekiq::Shutdown, a Timeout) or the code is
448
+ # broken (NotImplementedError, LoadError).
449
+ def interrupted?(error)
450
+ !error.is_a?(StandardError)
451
+ end
452
+
453
+ def build_definition(name, options)
454
+ unknown = options.keys.map(&:to_sym) - JobDefinition::OPTIONS
455
+ raise ArgumentError, "job \"#{name}\": unknown option #{unknown.map(&:inspect).join(", ")}" if unknown.any?
456
+
457
+ fields = {}
458
+ @defaults.each { |k, v| fields[k] = v }
459
+ options.each do |k, v|
460
+ fields[k.to_sym] = v.is_a?(Hash) && k.to_sym == :budget ? v.transform_keys(&:to_s) : v
461
+ end
462
+ fields[:name] = name
463
+ JobDefinition.new(fields)
464
+ end
465
+
466
+ # Raises a clear error for options that would otherwise quietly turn a check off.
467
+ def validate_definition(definition)
468
+ name = definition.name
469
+ unless definition.schedule.nil?
470
+ unless definition.schedule.is_a?(String) && JS.trim(definition.schedule) != ""
471
+ raise ArgumentError, "job \"#{name}\": schedule must be a non-empty string"
472
+ end
473
+
474
+ Schedule.parse(definition.schedule, definition.timezone)
475
+ end
476
+ if !definition.timezone.nil? && !(definition.timezone.is_a?(String) && Zone.valid?(definition.timezone))
477
+ raise ArgumentError, "job \"#{name}\": timezone \"#{definition.timezone}\" is not an IANA timezone"
478
+ end
479
+
480
+ Duration.parse(definition.grace, "grace") unless definition.grace.nil?
481
+ if !definition.timeout.nil? && Duration.parse(definition.timeout, "timeout") <= 0
482
+ raise ArgumentError, "job \"#{name}\": timeout must be longer than zero"
483
+ end
484
+ if !definition.max_duration.nil? && Duration.parse(definition.max_duration, "maxDuration") <= 0
485
+ raise ArgumentError, "job \"#{name}\": maxDuration must be longer than zero"
486
+ end
487
+
488
+ failures = definition.failures_before_alert
489
+ if !failures.nil? && !(failures.is_a?(Numeric) && JS.integer?(failures) && failures >= 1)
490
+ raise ArgumentError, "job \"#{name}\": failuresBeforeAlert must be a whole number, 1 or more (got #{js_string(failures)})"
491
+ end
492
+ unless definition.budget.nil?
493
+ raise ArgumentError, "job \"#{name}\": budget must be an object of { metric: ceiling }" unless definition.budget.is_a?(Hash)
494
+
495
+ definition.budget.each do |metric, ceiling|
496
+ next if ceiling.is_a?(Numeric) && ceiling.real? && JS.finite?(ceiling) && ceiling >= 0
497
+
498
+ raise ArgumentError, "job \"#{name}\": budget.#{metric} must be a finite number, 0 or more (got #{js_string(ceiling)})"
499
+ end
500
+ end
501
+ expect = definition.expect
502
+ return if expect.nil? || expect.is_a?(String) || expect.is_a?(Regexp) || expect.respond_to?(:call)
503
+
504
+ raise ArgumentError, "job \"#{name}\": expect must be a string, a RegExp or a function"
505
+ end
506
+
507
+ # String(value) as JavaScript writes it, for the messages above.
508
+ def js_string(value)
509
+ case value
510
+ when nil then "null"
511
+ when Numeric then JS.number(value)
512
+ else value.to_s
513
+ end
514
+ end
515
+
516
+ def ensure_ready
517
+ return if @ready
518
+
519
+ @ready_lock.synchronize do
520
+ next if @ready
521
+
522
+ @store.init if @store.respond_to?(:init)
523
+ if @using_default_store && Environment.production?
524
+ warn "[cronwatch] using the in-memory store: runs and state are lost on restart. " \
525
+ "Pass a store such as Cronwatch::Stores::ActiveRecord."
526
+ end
527
+ # Only once init has gone through: a failure is tried again on the next call.
528
+ @ready = true
529
+ end
530
+ end
531
+
532
+ def sync(definition)
533
+ ensure_ready
534
+ return if @registry.synchronize { @synced.include?(definition.name) }
535
+
536
+ @store.upsert_job(Serialize.to_stored(definition), now)
537
+ @registry.synchronize { @synced << definition.name }
538
+ end
539
+
540
+ # Runs the block while holding the job's lock, so two runs (or a run and a
541
+ # check) in this process never read and write the job's state over each
542
+ # other. Other processes are coordinated by update_state instead. Only
543
+ # store reads and writes happen inside; alerts are sent outside it.
544
+ def serial(job, &block)
545
+ after_fork_check
546
+ lock = @registry.synchronize { @locks[job] ||= Monitor.new }
547
+ lock.synchronize(&block)
548
+ end
549
+
550
+ def read_state(job)
551
+ Evaluate.normalize_state(@store.get_state(job), job)
552
+ end
553
+
554
+ def same_state?(a, b)
555
+ JS.json(a.to_h) == JS.json(b.to_h)
556
+ end
557
+
558
+ # Every read-modify-write of a job's state goes through here. In turn with
559
+ # this process's other updates to the job (serial), it reads the state,
560
+ # yields it for the next one and a result (`[state, result]`), and writes
561
+ # that with the version one higher, only if the stored version is still
562
+ # the one read. When another process wrote in between, the write is
563
+ # refused and it starts again from a fresh read, up to STATE_ATTEMPTS
564
+ # times. So the block may run more than once and must only compute:
565
+ # whatever it returns from the attempt that was written is the result.
566
+ # Nothing is written when the state is unchanged. Returns
567
+ # `[state as stored, result]`.
568
+ def update_state(job)
569
+ serial(job) do
570
+ attempt = 0
571
+ loop do
572
+ attempt += 1
573
+ current = read_state(job)
574
+ state, result = yield(current)
575
+ break [current, result] if same_state?(state, current)
576
+
577
+ version = current.version || 0
578
+ following = state.dup
579
+ following.version = version + 1
580
+ break [following, result] if write_state(following, version)
581
+ if attempt >= STATE_ATTEMPTS
582
+ raise "the state of #{job} changed under #{STATE_ATTEMPTS} attempts in a row to update it; gave up"
583
+ end
584
+ end
585
+ end
586
+ end
587
+
588
+ # A conditional write, or for a store without compare_and_set_state, a
589
+ # plain one that always succeeds.
590
+ def write_state(state, expected_version)
591
+ return @store.compare_and_set_state(state, expected_version) if @store.respond_to?(:compare_and_set_state)
592
+
593
+ @store.set_state(state)
594
+ true
595
+ end
596
+
597
+ # A custom redact, made safe: one that raises or returns something other
598
+ # than a String is reported and the default is used instead, so a broken
599
+ # redact neither stops the run finishing nor leaks what it was given.
600
+ def guarded_redact(redact)
601
+ lambda do |text|
602
+ out = redact.call(text)
603
+ raise TypeError, "redact must return a string, not #{out.nil? ? "null" : out.class}" unless out.is_a?(String)
604
+
605
+ Output.utf8(out)
606
+ rescue StandardError => e
607
+ report(e, "redact")
608
+ Output.redact_secrets(text)
609
+ end
610
+ end
611
+
612
+ # Identifies an alert across retries.
613
+ def alert_key(alert)
614
+ "#{alert.type}|#{alert.at}|#{alert.run&.id}"
615
+ end
616
+
617
+ # Whether a check already marked this run as timed out, for a failure that finished late.
618
+ def marked_timed_out?(run)
619
+ return false if run.status == :ok
620
+
621
+ @store.get_run(run.id)&.status == :timeout
622
+ rescue StandardError
623
+ false
624
+ end
625
+
626
+ # Record a finished run (ok, failed, or timed out by a check), evaluate it
627
+ # against the job's state and send what that produces. Never raises.
628
+ def finish_run(definition, run, at, write)
629
+ drafts = nil
630
+ begin
631
+ @store.update_run(run) if write
632
+ past = nil
633
+ _, drafts = update_state(run.job) do |previous|
634
+ past ||= history(run)
635
+ settled = Evaluate.apply_silence(previous, Evaluate.on_run_finish(definition, run, previous, past, at), at)
636
+ [settled.state, settled.alerts]
637
+ end
638
+ rescue StandardError => e
639
+ report(e, "evaluating #{run.job}")
640
+ return []
641
+ end
642
+ dispatch(drafts, definition, at)
643
+ end
644
+
645
+ # The runs before `run`, newest first, with up to BASELINE_WINDOW
646
+ # successful ones when the store has them. One small read normally; a
647
+ # larger one only when failures crowd the successes out of it.
648
+ def history(run)
649
+ runs = @store.list_runs(run.job, HISTORY_PAGE)
650
+ if runs.length == HISTORY_PAGE && !Evaluate.full_baseline?(runs.reject { |r| r.id == run.id })
651
+ runs = @store.list_runs(run.job, HISTORY_MAX)
652
+ end
653
+ runs.reject { |r| r.id == run.id }
654
+ end
655
+
656
+ def run_check
657
+ ensure_ready
658
+ defined_jobs.each { |definition| sync(definition) }
659
+ at = now
660
+ alerts = []
661
+
662
+ # Runs that never reported back. One that cannot be judged (its job's
663
+ # stored timeout no longer parses, say) is reported and skipped.
664
+ @store.running_runs.each do |run|
665
+ declared = @registry.synchronize { @definitions[run.job] }
666
+ definition = declared ? Serialize.to_stored(declared) : @store.get_job(run.job)&.definition
667
+ next if definition.nil? || !Evaluate.stuck?(definition, run, at)
668
+
669
+ timeout = Evaluate.timeout_ms(definition)
670
+ run.status = :timeout
671
+ run.finished_at = at
672
+ run.duration_ms = at - run.started_at
673
+ run.error = "Still running after #{Duration.format(timeout)}; marked as timed out"
674
+ alerts.concat(finish_run(definition, run, at, true))
675
+ rescue StandardError => e
676
+ report(e, "checking #{run.job}")
677
+ end
678
+
679
+ # Each job on its own: one that cannot be evaluated (a stored schedule
680
+ # this process cannot read, one a Node process wrote, say) is reported,
681
+ # shown as failing (see Evaluate.unevaluable_summary) and does not stop
682
+ # the others.
683
+ jobs = []
684
+ retries = RetryBudget.new(0)
685
+ @store.list_jobs.each do |stored|
686
+ recent = @store.list_runs(stored.name, Evaluate::BASELINE_WINDOW)
687
+ next_expected_at = nil
688
+ state, drafts = update_state(stored.name) do |previous|
689
+ evaluation = Evaluate.on_check(stored.definition, stored, recent.first, previous, at)
690
+ next_expected_at = evaluation.next_expected_at
691
+ settled = Evaluate.apply_silence(previous, evaluation, at)
692
+ [settled.state, settled.alerts]
693
+ end
694
+ alerts.concat(retry_undelivered(stored.name, state, at, retries))
695
+ alerts.concat(dispatch(drafts, stored.definition, at))
696
+ jobs << Evaluate.summarize(stored, recent, state, next_expected_at, at)
697
+ rescue StandardError => e
698
+ report(e, "checking #{stored.name}")
699
+ jobs << unevaluable(stored, at)
700
+ end
701
+
702
+ pruned = 0
703
+ if at - @last_prune_at > PRUNE_INTERVAL_MS
704
+ @last_prune_at = at
705
+ begin
706
+ pruned = @store.prune(at - @retention_ms)
707
+ rescue StandardError => e
708
+ report(e, "pruning")
709
+ end
710
+ end
711
+
712
+ CheckResult.new(checked_at: at, jobs: jobs, alerts: alerts, pruned: pruned)
713
+ end
714
+
715
+ # A job's summary and its newest runs, without alerting. A job that cannot
716
+ # be evaluated (a stored schedule this process cannot read, say) is
717
+ # reported and shown as failing.
718
+ def snapshot(stored, at, count)
719
+ recent = []
720
+ begin
721
+ recent = @store.list_runs(stored.name, [count, Evaluate::BASELINE_WINDOW].max)
722
+ state = read_state(stored.name)
723
+ next_expected_at = Evaluate.on_check(stored.definition, stored, recent.first, state, at).next_expected_at
724
+ JobWithRuns.new(job: Evaluate.summarize(stored, recent, state, next_expected_at, at), runs: recent.first(count))
725
+ rescue StandardError => e
726
+ report(e, "reading #{stored.name}")
727
+ JobWithRuns.new(job: unevaluable(stored, at), runs: recent.first(count))
728
+ end
729
+ end
730
+
731
+ # The summary of a job whose evaluation failed, from whatever can still be read.
732
+ def unevaluable(stored, at)
733
+ recent = begin
734
+ @store.list_runs(stored.name, Evaluate::BASELINE_WINDOW)
735
+ rescue StandardError
736
+ []
737
+ end
738
+ state = begin
739
+ read_state(stored.name)
740
+ rescue StandardError
741
+ Evaluate.empty_state(stored.name)
742
+ end
743
+ Evaluate.unevaluable_summary(stored, recent, state, at)
744
+ end
745
+
746
+ # Read, change and write one job's state, in turn with every other update to it.
747
+ def patch_state(name)
748
+ ensure_ready
749
+ state, = update_state(name) do |current|
750
+ following = Evaluate.normalize_state(current, name)
751
+ yield following
752
+ [following, nil]
753
+ end
754
+ state
755
+ end
756
+
757
+ # Compose, triage and send each draft. The state was saved before this
758
+ # (update_state), so a slow channel holds up nothing else; afterwards only
759
+ # the delivery fields are written back, onto a fresh read of the state.
760
+ def dispatch(drafts, definition, at)
761
+ return [] if drafts.empty?
762
+
763
+ composed = []
764
+ delivered = []
765
+ failed = []
766
+ drafts.each do |draft|
767
+ alert = Format.compose_alert(draft, definition, at)
768
+ if @defer_delivery
769
+ failed << alert
770
+ else
771
+ add_triage(alert, @triage_timeout_ms) if @triage && alert.type != :recovered
772
+ (deliver(alert) ? delivered : failed) << alert
773
+ end
774
+ composed << alert
775
+ end
776
+ record_delivery(definition.name, delivered, failed, [], at)
777
+ composed
778
+ end
779
+
780
+ # The wall-clock milliseconds one check has spent retrying, across its jobs.
781
+ RetryBudget = Struct.new(:spent_ms)
782
+
783
+ # Send the alerts that no channel accepted last time, once each, oldest
784
+ # first. `state` is the job's state as this check left it: an alert that
785
+ # no longer describes it (Evaluate.stale_alert?) is dropped instead.
786
+ # Retries across a check share RETRY_BUDGET_MS of wall-clock time; once it
787
+ # is spent the rest stay queued for the next check.
788
+ def retry_undelivered(name, state, at, budget)
789
+ pending = state.undelivered || []
790
+ return [] if pending.empty? || Evaluate.silenced?(state, at) || @defer_delivery
791
+
792
+ delivered = []
793
+ failed = []
794
+ dropped = pending.select { |alert| Evaluate.stale_alert?(alert, state) }
795
+ pending.each do |alert|
796
+ next if dropped.any? { |d| d.equal?(alert) }
797
+
798
+ left = @retry_budget_ms - budget.spent_ms
799
+ break if left <= 0
800
+
801
+ started = AbortSignal.monotonic
802
+ # An alert queued by a process that delivers at check time was never
803
+ # triaged. One that was tried (triage: null) is not tried again.
804
+ add_triage(alert, [@triage_timeout_ms, left].min) if @triage && alert.type != :recovered && !alert.triage_tried?
805
+ (deliver(alert) ? delivered : failed) << alert
806
+ budget.spent_ms += [0, ((AbortSignal.monotonic - started) * 1000).round].max
807
+ end
808
+ record_delivery(name, delivered, failed, dropped, at)
809
+ delivered
810
+ end
811
+
812
+ # Mark delivered alerts done, drop stale ones, and keep failed ones for the
813
+ # next check. A failed alert replaces its stored copy, so a triage made on
814
+ # this attempt is kept. last_alert_at moves only on a delivery.
815
+ def record_delivery(name, delivered, failed, dropped, at)
816
+ _, trimmed = update_state(name) do |previous|
817
+ state = Evaluate.normalize_state(previous, name)
818
+ done = (delivered + dropped).map { |a| alert_key(a) }.to_set
819
+ retried = failed.to_h { |a| [alert_key(a), a] }
820
+ kept = state.undelivered.reject { |a| done.include?(alert_key(a)) }.map { |a| retried.fetch(alert_key(a), a) }
821
+ known = kept.map { |a| alert_key(a) }.to_set
822
+ kept.concat(failed.reject { |a| known.include?(alert_key(a)) })
823
+ state.undelivered = kept.last(MAX_UNDELIVERED)
824
+ state.last_alert_at = at if delivered.any?
825
+ [state, [0, kept.length - MAX_UNDELIVERED].max]
826
+ end
827
+ if trimmed.positive?
828
+ report(RuntimeError.new("#{trimmed} undelivered alert#{trimmed == 1 ? "" : "s"} for #{name} dropped: " \
829
+ "only the newest #{MAX_UNDELIVERED} are kept for retry"), "alert queue for #{name}")
830
+ end
831
+ rescue StandardError => e
832
+ report(e, "recording alert delivery for #{name}")
833
+ end
834
+
835
+ # Send to every channel at once, each in its own thread with its own
836
+ # timeout. True when at least one accepted it, or there are none. A
837
+ # channel that times out is left to finish on its own, and nothing more
838
+ # is sent to it until it has: meanwhile its alerts count as not
839
+ # delivered there, to be retried by a later check. So a hung channel
840
+ # holds one thread, not one per alert.
841
+ def deliver(alert)
842
+ return true if @alerts.empty?
843
+
844
+ outcomes = Array.new(@alerts.length)
845
+ threads = @alerts.each_with_index.map do |channel, i|
846
+ if @sending_lock.synchronize { @abandoned[i]&.alive? }
847
+ outcomes[i] = TimeoutError.new("skipped: an earlier alert timed out and is still being sent")
848
+ next nil
849
+ end
850
+
851
+ Thread.new do
852
+ Thread.current.report_on_exception = false
853
+ channel.call(alert)
854
+ outcomes[i] = true
855
+ rescue StandardError, ScriptError => e
856
+ outcomes[i] = e
857
+ end
858
+ end
859
+ deadline = AbortSignal.monotonic + (@channel_timeout_ms / 1000.0)
860
+ results = threads.each_with_index.map do |thread, i|
861
+ next outcomes[i] if thread.nil?
862
+ next outcomes[i] if thread.join([deadline - AbortSignal.monotonic, 0].max)
863
+
864
+ @sending_lock.synchronize { @abandoned[i] = thread }
865
+ TimeoutError.new("timed out after #{@channel_timeout_ms}ms")
866
+ end
867
+ results.each_with_index do |result, i|
868
+ report(result, "alert channel #{channel_name(@alerts[i])}") unless result == true
869
+ end
870
+ results.include?(true)
871
+ end
872
+
873
+ def channel_name(channel)
874
+ channel.respond_to?(:name) && channel.name ? channel.name : channel.class.name
875
+ end
876
+
877
+ # Sets the alert's triage to the diagnosis, or to nil (JSON null) when
878
+ # there is none (it raised, timed out or answered nil or ""), so it is
879
+ # tried once per alert. While a triage that timed out is still going,
880
+ # alerts go out without one rather than start another beside it.
881
+ def add_triage(alert, timeout_ms)
882
+ signal = AbortSignal.new
883
+ if @sending_lock.synchronize { @abandoned[:triage]&.alive? }
884
+ raise TimeoutError, "skipped: an earlier triage timed out and is still running"
885
+ end
886
+
887
+ recent = @store.list_runs(alert.job, 5)
888
+ context = TriageContext.new(alert: alert, recent_runs: recent, signal: signal)
889
+ outcome = nil
890
+ thread = Thread.new do
891
+ Thread.current.report_on_exception = false
892
+ outcome = [:ok, @triage.call(context)]
893
+ rescue StandardError, ScriptError => e
894
+ outcome = [:error, e]
895
+ end
896
+ unless thread.join(timeout_ms / 1000.0)
897
+ @sending_lock.synchronize { @abandoned[:triage] = thread }
898
+ raise TimeoutError, "timed out after #{timeout_ms}ms"
899
+ end
900
+ raise outcome[1] if outcome[0] == :error
901
+
902
+ diagnosis = outcome[1]
903
+ alert.triage_result = diagnosis.is_a?(String) && !diagnosis.empty? ? Output.utf8(diagnosis) : nil
904
+ rescue StandardError => e
905
+ signal&.abort!
906
+ alert.triage_result = nil
907
+ report(e, "triage for #{alert.job}")
908
+ end
909
+
910
+ # A whole number in range, or the fallback for anything that is not a number.
911
+ def clamp_limit(limit, fallback, min)
912
+ n = limit.is_a?(Numeric) && limit.real? && JS.finite?(limit) ? limit.truncate : fallback
913
+ [500, [min, n].max].min
914
+ end
915
+
916
+ def default_on_error(error, where)
917
+ message = "[cronwatch] #{where}: #{error.class}: #{error.message}"
918
+ if defined?(::Rails) && ::Rails.respond_to?(:logger) && ::Rails.logger
919
+ ::Rails.logger.error(message)
920
+ else
921
+ warn message
922
+ end
923
+ end
924
+ end
925
+ end