cronwatch 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/README.md +268 -0
- data/lib/cronwatch/abort_signal.rb +45 -0
- data/lib/cronwatch/active_record.rb +11 -0
- data/lib/cronwatch/alerts/console.rb +22 -0
- data/lib/cronwatch/alerts/custom.rb +26 -0
- data/lib/cronwatch/alerts/discord.rb +52 -0
- data/lib/cronwatch/alerts/slack.rb +58 -0
- data/lib/cronwatch/alerts/webhook.rb +46 -0
- data/lib/cronwatch/client.rb +925 -0
- data/lib/cronwatch/cron_pattern.rb +277 -0
- data/lib/cronwatch/duration.rb +77 -0
- data/lib/cronwatch/environment.rb +29 -0
- data/lib/cronwatch/evaluate.rb +345 -0
- data/lib/cronwatch/flight.rb +42 -0
- data/lib/cronwatch/format.rb +89 -0
- data/lib/cronwatch/http.rb +52 -0
- data/lib/cronwatch/job.rb +144 -0
- data/lib/cronwatch/js.rb +188 -0
- data/lib/cronwatch/monitored.rb +259 -0
- data/lib/cronwatch/output.rb +199 -0
- data/lib/cronwatch/rails/active_job.rb +55 -0
- data/lib/cronwatch/rails/check_job.rb +32 -0
- data/lib/cronwatch/rails/railtie.rb +37 -0
- data/lib/cronwatch/rails/tasks.rb +12 -0
- data/lib/cronwatch/rails.rb +35 -0
- data/lib/cronwatch/schedule.rb +191 -0
- data/lib/cronwatch/scheduler.rb +763 -0
- data/lib/cronwatch/serialize.rb +51 -0
- data/lib/cronwatch/sidekiq.rb +129 -0
- data/lib/cronwatch/stats.rb +23 -0
- data/lib/cronwatch/stores/active_record.rb +397 -0
- data/lib/cronwatch/stores/memory.rb +163 -0
- data/lib/cronwatch/ticker.rb +59 -0
- data/lib/cronwatch/triage/anthropic.rb +134 -0
- data/lib/cronwatch/types.rb +341 -0
- data/lib/cronwatch/version.rb +6 -0
- data/lib/cronwatch/walker.rb +137 -0
- data/lib/cronwatch/web/app.rb +484 -0
- data/lib/cronwatch/web/html.rb +314 -0
- data/lib/cronwatch/web.rb +17 -0
- data/lib/cronwatch/zone.rb +72 -0
- data/lib/cronwatch.rb +96 -0
- data/lib/generators/cronwatch/install/install_generator.rb +176 -0
- metadata +104 -0
|
@@ -0,0 +1,925 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "monitor"
|
|
4
|
+
require "securerandom"
|
|
5
|
+
require "set"
|
|
6
|
+
require_relative "abort_signal"
|
|
7
|
+
require_relative "environment"
|
|
8
|
+
require_relative "flight"
|
|
9
|
+
require_relative "ticker"
|
|
10
|
+
|
|
11
|
+
module Cronwatch
|
|
12
|
+
# Raised (and handed to on_error) when a channel or triage takes too long,
|
|
13
|
+
# or is skipped because its previous call has not finished.
|
|
14
|
+
class TimeoutError < StandardError; end
|
|
15
|
+
|
|
16
|
+
# What callers waiting on a shared check see when the check was stopped by
|
|
17
|
+
# an exception outside StandardError (Interrupt, Timeout, SystemExit). The
|
|
18
|
+
# caller that ran the check sees the original exception.
|
|
19
|
+
class InterruptedError < StandardError; end
|
|
20
|
+
|
|
21
|
+
class Client
|
|
22
|
+
NAME_RE = /\A[A-Za-z0-9][A-Za-z0-9._:-]{0,119}\z/
|
|
23
|
+
TRIAGE_TIMEOUT_MS = 25_000
|
|
24
|
+
# How long one channel may take to send one alert.
|
|
25
|
+
CHANNEL_TIMEOUT_MS = 15_000
|
|
26
|
+
PRUNE_INTERVAL_MS = 60 * 60_000
|
|
27
|
+
# Undelivered alerts kept per job for retry; the oldest go first.
|
|
28
|
+
MAX_UNDELIVERED = 20
|
|
29
|
+
# Wall-clock time one check spends retrying undelivered alerts, across
|
|
30
|
+
# every job. Once it is spent the rest wait for the next check.
|
|
31
|
+
RETRY_BUDGET_MS = 20_000
|
|
32
|
+
# Reads and writes of one job's state before an update gives up on a store that keeps changing under it.
|
|
33
|
+
STATE_ATTEMPTS = 10
|
|
34
|
+
# Runs read for a baseline, and the most read when failures crowd out the successes.
|
|
35
|
+
HISTORY_PAGE = Evaluate::BASELINE_WINDOW + 5
|
|
36
|
+
HISTORY_MAX = 200
|
|
37
|
+
DEFAULT_OPTIONS = %i[grace timeout timezone failures_before_alert].freeze
|
|
38
|
+
# Tells "cron_secret not given" (read CRON_SECRET) from "cron_secret: nil" (no secret on purpose).
|
|
39
|
+
UNSET = Object.new.freeze
|
|
40
|
+
# Guards the reset a forked child makes of the parent's locks and threads.
|
|
41
|
+
FORK_LOCK = Mutex.new
|
|
42
|
+
SILENCE_OPTIONS = %i[for].freeze
|
|
43
|
+
|
|
44
|
+
# What execute returns: the recorded run, and the block's own outcome.
|
|
45
|
+
ExecuteResult = Struct.new(:run, :result, :error, :threw, keyword_init: true)
|
|
46
|
+
# What a triage callable receives. Pass the signal to anything that can stop early.
|
|
47
|
+
TriageContext = Struct.new(:alert, :recent_runs, :signal, keyword_init: true)
|
|
48
|
+
# A job's summary and its newest runs, as the dashboard shows them.
|
|
49
|
+
JobWithRuns = Struct.new(:job, :runs, keyword_init: true) do
|
|
50
|
+
include Serializable
|
|
51
|
+
|
|
52
|
+
def to_h = { "job" => job.to_h, "runs" => runs.map(&:to_h) }
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
attr_reader :store, :alerts, :triage, :cron_secret, :retention_ms, :defaults
|
|
56
|
+
|
|
57
|
+
# store: where jobs, runs and state live. Defaults to an in-memory store that forgets on restart.
|
|
58
|
+
# alerts: where alerts go: objects with #call(alert) and #name. Defaults to the console.
|
|
59
|
+
# triage: a callable taking a TriageContext and returning a short diagnosis, added to every alert but recoveries.
|
|
60
|
+
# cron_secret: a second bearer Cronwatch::Web accepts for /api/check, for an outside cron. Defaults to
|
|
61
|
+
# ENV["CRON_SECRET"]; an empty string counts as unset. Pass nil for none.
|
|
62
|
+
# retention: how long finished runs are kept. Default "30d".
|
|
63
|
+
# defaults: grace, timeout, timezone and failures_before_alert applied to every job unless it sets its own.
|
|
64
|
+
# redact: applied to every run's output and error before it is stored, shown or sent to an alert
|
|
65
|
+
# channel or triage. The default (Output.redact_secrets) blanks values that look like secrets
|
|
66
|
+
# (password=..., Authorization headers, URL credentials, bearer tokens, JWTs, PEM private
|
|
67
|
+
# keys, webhook URLs, AWS, GitHub, Slack, Stripe, Google and API key formats). Pass your own
|
|
68
|
+
# callable, or false to keep output exactly as logged. A callable that raises or returns
|
|
69
|
+
# something other than a String is reported to on_error ("redact") and the default is used.
|
|
70
|
+
# now: the clock, a callable returning epoch milliseconds. Tests use this.
|
|
71
|
+
# on_error: called with (error, where) for anything that goes wrong outside a job: the store failing,
|
|
72
|
+
# an alert channel failing, a triage timeout.
|
|
73
|
+
def initialize(store: nil, alerts: nil, triage: nil, cron_secret: UNSET, retention: "30d", defaults: {}, redact: nil,
|
|
74
|
+
deliver: :now, now: nil, on_error: nil)
|
|
75
|
+
@using_default_store = store.nil?
|
|
76
|
+
@store = store || Stores::Memory.new
|
|
77
|
+
@alerts = alerts.nil? ? [Alerts::Console.new] : Array(alerts)
|
|
78
|
+
@triage = triage
|
|
79
|
+
secret = cron_secret.equal?(UNSET) ? ENV.fetch("CRON_SECRET", nil) : cron_secret
|
|
80
|
+
@cron_secret = secret.nil? || secret.to_s.empty? ? nil : secret.to_s
|
|
81
|
+
@retention_ms = Duration.parse(retention || "30d", "retention")
|
|
82
|
+
@defaults = (defaults || {}).transform_keys(&:to_sym)
|
|
83
|
+
unknown = @defaults.keys - DEFAULT_OPTIONS
|
|
84
|
+
raise ArgumentError, "defaults may set #{DEFAULT_OPTIONS.join(", ")}, not #{unknown.join(", ")}" if unknown.any?
|
|
85
|
+
|
|
86
|
+
if !(redact.nil? || redact == false || redact.respond_to?(:call))
|
|
87
|
+
raise ArgumentError, "redact must be a callable, or false to keep output as logged"
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
@redact =
|
|
91
|
+
if redact == false then ->(text) { text }
|
|
92
|
+
elsif redact.nil? then Output.method(:redact_secrets)
|
|
93
|
+
else guarded_redact(redact)
|
|
94
|
+
end
|
|
95
|
+
unless [:now, :check, "now", "check", nil].include?(deliver)
|
|
96
|
+
raise ArgumentError, "deliver must be \"now\" or \"check\", not #{deliver.inspect}"
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
# :check queues alerts in the store for another process's check to send
|
|
100
|
+
# (see "Delivery" in DESIGN.md, and deliver in the SDK).
|
|
101
|
+
@defer_delivery = deliver.to_s == "check"
|
|
102
|
+
@clock = now || -> { Process.clock_gettime(Process::CLOCK_REALTIME, :millisecond) }
|
|
103
|
+
@on_error = on_error || method(:default_on_error)
|
|
104
|
+
@definitions = {}
|
|
105
|
+
@synced = Set.new
|
|
106
|
+
@registry = Mutex.new
|
|
107
|
+
@ready = false
|
|
108
|
+
@last_prune_at = 0
|
|
109
|
+
# Seconds before start()'s first check, and how long a channel or triage may take. Tests shorten them.
|
|
110
|
+
@first_tick_s = 1.0
|
|
111
|
+
@channel_timeout_ms = CHANNEL_TIMEOUT_MS
|
|
112
|
+
@triage_timeout_ms = TRIAGE_TIMEOUT_MS
|
|
113
|
+
@retry_budget_ms = RETRY_BUDGET_MS
|
|
114
|
+
@warned_deferred_start = false
|
|
115
|
+
reset_process_state
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
# Epoch milliseconds, from the clock the client was given.
|
|
119
|
+
def now
|
|
120
|
+
@clock.call
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
# Declare a job. Call it once, when the app loads, and keep the handle.
|
|
124
|
+
def job(name, **options)
|
|
125
|
+
name = name.to_s if name.is_a?(Symbol)
|
|
126
|
+
unless name.is_a?(String) && NAME_RE.match?(name)
|
|
127
|
+
raise ArgumentError, "job name \"#{name}\" must be 1 to 120 characters of letters, digits, \".\", \"_\", \":\" or \"-\""
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
definition = build_definition(name, options)
|
|
131
|
+
validate_definition(definition)
|
|
132
|
+
@registry.synchronize do
|
|
133
|
+
@definitions[name] = definition
|
|
134
|
+
@synced.delete(name)
|
|
135
|
+
end
|
|
136
|
+
JobHandle.new(self, definition)
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
# With a block: run a job by name without keeping a handle, declaring it
|
|
140
|
+
# on first use (or again, when options are given). Without a block: the
|
|
141
|
+
# run with this id, as get_run.
|
|
142
|
+
def run(name_or_id, **options, &block)
|
|
143
|
+
unless block
|
|
144
|
+
raise ArgumentError, "run(#{name_or_id.inspect}, ...) needs a block; without one, run(id) reads a run" if options.any?
|
|
145
|
+
|
|
146
|
+
return get_run(name_or_id)
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
declared = @registry.synchronize { @definitions[name_or_id.to_s] }
|
|
150
|
+
handle = options.any? || declared.nil? ? job(name_or_id, **options) : JobHandle.new(self, declared)
|
|
151
|
+
handle.run(&block)
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
# The definitions declared in this process.
|
|
155
|
+
def defined_jobs
|
|
156
|
+
@registry.synchronize { @definitions.values }
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
# Runs a block as a recorded run. The block always runs, whatever the store
|
|
160
|
+
# is doing: store errors go to on_error, and the result is the block's own
|
|
161
|
+
# outcome. A StandardError from the block is not raised; see `threw`. An
|
|
162
|
+
# exception outside StandardError (Interrupt, SystemExit, Sidekiq::Shutdown,
|
|
163
|
+
# a Timeout, NotImplementedError) is recorded as a failed run and then
|
|
164
|
+
# raised again, so the run is never left running.
|
|
165
|
+
#
|
|
166
|
+
# `failure` is an optional callable that turns the block's result into an
|
|
167
|
+
# error message, or nil when the result is fine: an HTTP handler uses it to
|
|
168
|
+
# count a 500 response as a failed run.
|
|
169
|
+
#
|
|
170
|
+
# @api private For integrations (JobHandle#run, ActiveJob, Sidekiq), not apps.
|
|
171
|
+
def execute(definition, trigger, failure: nil)
|
|
172
|
+
after_fork_check
|
|
173
|
+
name = definition.name
|
|
174
|
+
started_at = now
|
|
175
|
+
run = Run.new(id: SecureRandom.uuid, job: name, status: :running, started_at: started_at, finished_at: nil,
|
|
176
|
+
duration_ms: nil, error: nil, output: nil, metrics: {}, trigger: trigger)
|
|
177
|
+
recorded = false
|
|
178
|
+
begin
|
|
179
|
+
sync(definition)
|
|
180
|
+
@store.insert_run(run.dup)
|
|
181
|
+
recorded = true
|
|
182
|
+
rescue StandardError => e
|
|
183
|
+
report(e, "recording #{name}")
|
|
184
|
+
end
|
|
185
|
+
# The SDK closes missed and stuck beside the running job; here it is done
|
|
186
|
+
# just before the block runs. The result is the same.
|
|
187
|
+
if recorded
|
|
188
|
+
begin
|
|
189
|
+
update_state(name) { |before| [Evaluate.on_run_start(before), nil] }
|
|
190
|
+
rescue StandardError => e
|
|
191
|
+
report(e, "starting #{name}")
|
|
192
|
+
end
|
|
193
|
+
end
|
|
194
|
+
|
|
195
|
+
recorder = RunRecorder.new(run, Evaluate.timeout_ms(definition))
|
|
196
|
+
result = nil
|
|
197
|
+
error = nil
|
|
198
|
+
threw = false
|
|
199
|
+
begin
|
|
200
|
+
result = yield(recorder.context)
|
|
201
|
+
rescue Exception => e # rubocop:disable Lint/RescueException -- recorded, then raised again below
|
|
202
|
+
error = e
|
|
203
|
+
threw = true
|
|
204
|
+
ensure
|
|
205
|
+
recorder.signal.settle!
|
|
206
|
+
end
|
|
207
|
+
|
|
208
|
+
finished_at = now
|
|
209
|
+
run.finished_at = finished_at
|
|
210
|
+
run.duration_ms = [0, finished_at - started_at].max
|
|
211
|
+
run.metrics = recorder.metrics
|
|
212
|
+
returned = result.is_a?(String) ? Output.utf8(result) : nil
|
|
213
|
+
run.output = recorder.output || (returned && Output.cap(returned))
|
|
214
|
+
|
|
215
|
+
if threw
|
|
216
|
+
run.status = :failed
|
|
217
|
+
run.error = Output.error_message(error)
|
|
218
|
+
elsif (problem = failure&.call(result))
|
|
219
|
+
run.status = :failed
|
|
220
|
+
run.error = Output.utf8(problem.to_s)
|
|
221
|
+
else
|
|
222
|
+
expect_text = recorder.expect_text || returned
|
|
223
|
+
unmet = Serialize.check_expectation(definition.expect, expect_text)
|
|
224
|
+
if unmet
|
|
225
|
+
run.status = :failed
|
|
226
|
+
run.error = unmet
|
|
227
|
+
else
|
|
228
|
+
run.status = :ok
|
|
229
|
+
end
|
|
230
|
+
end
|
|
231
|
+
# Redacted after the expect check, so a rule can still match what was
|
|
232
|
+
# logged. NULs go last, so not even a custom redact can store one.
|
|
233
|
+
run.output = Output.strip_nul(@redact.call(run.output)) unless run.output.nil?
|
|
234
|
+
run.error = Output.strip_nul(@redact.call(run.error)) unless run.error.nil?
|
|
235
|
+
|
|
236
|
+
stored = Serialize.to_stored(definition)
|
|
237
|
+
if recorded && marked_timed_out?(run)
|
|
238
|
+
# A check gave up on this run while it was going and already counted
|
|
239
|
+
# it as a stuck failure. A late failure must not count twice; a late
|
|
240
|
+
# success still closes stuck and recovers.
|
|
241
|
+
begin
|
|
242
|
+
@store.update_run(run)
|
|
243
|
+
rescue StandardError => e
|
|
244
|
+
report(e, "recording #{name}")
|
|
245
|
+
end
|
|
246
|
+
elsif recorded
|
|
247
|
+
finish_run(stored, run, finished_at, true)
|
|
248
|
+
else
|
|
249
|
+
# The start was never written; the store may be back by now.
|
|
250
|
+
begin
|
|
251
|
+
sync(definition)
|
|
252
|
+
@store.insert_run(run)
|
|
253
|
+
recorded = true
|
|
254
|
+
rescue StandardError => e
|
|
255
|
+
report(e, "recording #{name}")
|
|
256
|
+
end
|
|
257
|
+
finish_run(stored, run, finished_at, false) if recorded
|
|
258
|
+
end
|
|
259
|
+
|
|
260
|
+
raise error if threw && interrupted?(error)
|
|
261
|
+
|
|
262
|
+
ExecuteResult.new(run: run, result: result, error: error, threw: threw)
|
|
263
|
+
end
|
|
264
|
+
|
|
265
|
+
# Look for missed and stuck runs across every job, send alerts, retry
|
|
266
|
+
# alerts no channel accepted, and prune old runs. Call it from start(), a
|
|
267
|
+
# scheduled job (Cronwatch::CheckJob), or by hand. Concurrent calls share
|
|
268
|
+
# one check.
|
|
269
|
+
def check
|
|
270
|
+
after_fork_check
|
|
271
|
+
flight = nil
|
|
272
|
+
mine = false
|
|
273
|
+
@check_lock.synchronize do
|
|
274
|
+
flight = @checking ||= begin
|
|
275
|
+
mine = true
|
|
276
|
+
Flight.new
|
|
277
|
+
end
|
|
278
|
+
end
|
|
279
|
+
if mine
|
|
280
|
+
begin
|
|
281
|
+
flight.resolve(run_check)
|
|
282
|
+
rescue Exception => e # rubocop:disable Lint/RescueException -- the waiters must not hang
|
|
283
|
+
# An Interrupt or a Timeout is meant for this thread only, so the
|
|
284
|
+
# waiters get an error of their own and this thread the original.
|
|
285
|
+
flight.reject(interrupted?(e) ? InterruptedError.new("the check was interrupted by #{e.class}") : e)
|
|
286
|
+
raise if interrupted?(e)
|
|
287
|
+
ensure
|
|
288
|
+
@check_lock.synchronize { @checking = nil if @checking.equal?(flight) }
|
|
289
|
+
end
|
|
290
|
+
end
|
|
291
|
+
flight.value
|
|
292
|
+
end
|
|
293
|
+
|
|
294
|
+
# Every job the store knows about, with its health. Does not send alerts.
|
|
295
|
+
def jobs
|
|
296
|
+
jobs_with_runs(0).map(&:job)
|
|
297
|
+
end
|
|
298
|
+
|
|
299
|
+
# Every job's summary with its newest `limit` runs, read together. What the dashboard shows.
|
|
300
|
+
def jobs_with_runs(limit = 20)
|
|
301
|
+
after_fork_check
|
|
302
|
+
ensure_ready
|
|
303
|
+
defined_jobs.each { |definition| sync(definition) }
|
|
304
|
+
at = now
|
|
305
|
+
count = clamp_limit(limit, 20, 0)
|
|
306
|
+
@store.list_jobs.map { |stored| snapshot(stored, at, count) }
|
|
307
|
+
end
|
|
308
|
+
|
|
309
|
+
def job_summary(name)
|
|
310
|
+
ensure_ready
|
|
311
|
+
definition = @registry.synchronize { @definitions[name] }
|
|
312
|
+
sync(definition) if definition
|
|
313
|
+
stored = @store.get_job(name)
|
|
314
|
+
return nil unless stored
|
|
315
|
+
|
|
316
|
+
snapshot(stored, now, 0).job
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
# A job's runs, newest first. `limit` is a whole number from 1 to 500.
|
|
320
|
+
def runs(name, limit = 50)
|
|
321
|
+
ensure_ready
|
|
322
|
+
@store.list_runs(name, clamp_limit(limit, 50, 1))
|
|
323
|
+
end
|
|
324
|
+
|
|
325
|
+
def get_run(id)
|
|
326
|
+
ensure_ready
|
|
327
|
+
@store.get_run(id)
|
|
328
|
+
end
|
|
329
|
+
|
|
330
|
+
# Stop alerts for a job for a while. State keeps updating underneath.
|
|
331
|
+
# silence("nightly-report", for: "2h") # or silence("nightly-report", "2h")
|
|
332
|
+
def silence(name, duration = nil, **options)
|
|
333
|
+
unknown = options.keys - SILENCE_OPTIONS
|
|
334
|
+
raise ArgumentError, "silence takes for:, not #{unknown.map(&:inspect).join(", ")}" if unknown.any?
|
|
335
|
+
raise ArgumentError, "silence takes a duration or for:, not both" if !duration.nil? && options.key?(:for)
|
|
336
|
+
|
|
337
|
+
duration = options[:for] if duration.nil?
|
|
338
|
+
ms = Duration.parse(duration, "silence duration")
|
|
339
|
+
patch_state(name) { |state| state.silenced_until = now + ms }
|
|
340
|
+
end
|
|
341
|
+
|
|
342
|
+
def unsilence(name)
|
|
343
|
+
patch_state(name) { |state| state.silenced_until = nil }
|
|
344
|
+
end
|
|
345
|
+
|
|
346
|
+
# Remove a job and its runs from the store. A job still declared in code comes back on its next run.
|
|
347
|
+
def forget(name)
|
|
348
|
+
ensure_ready
|
|
349
|
+
@registry.synchronize do
|
|
350
|
+
@definitions.delete(name)
|
|
351
|
+
@synced.delete(name)
|
|
352
|
+
end
|
|
353
|
+
@store.delete_job(name)
|
|
354
|
+
nil
|
|
355
|
+
end
|
|
356
|
+
|
|
357
|
+
# Check on an interval, in a background thread, for long-running
|
|
358
|
+
# processes. Default every minute; the first check comes after a second.
|
|
359
|
+
# Calling it again while it runs does nothing, and a different interval
|
|
360
|
+
# is reported to on_error and ignored: stop first to change it. A forked
|
|
361
|
+
# child (Puma, Unicorn, Sidekiq) has no thread, so call start there.
|
|
362
|
+
def start(every = "1m")
|
|
363
|
+
after_fork_check
|
|
364
|
+
ms = [5_000, Duration.parse(every, "check interval")].max
|
|
365
|
+
@ticker_lock.synchronize do
|
|
366
|
+
if @ticker
|
|
367
|
+
unless @ticker_ms == ms
|
|
368
|
+
report(ArgumentError.new("start(#{every.inspect}) ignored: already checking every #{Duration.format(@ticker_ms)}; " \
|
|
369
|
+
"call stop first to change it"), "start")
|
|
370
|
+
end
|
|
371
|
+
return nil
|
|
372
|
+
end
|
|
373
|
+
|
|
374
|
+
@ticker_ms = ms
|
|
375
|
+
if @defer_delivery && !@warned_deferred_start
|
|
376
|
+
@warned_deferred_start = true
|
|
377
|
+
warn '[cronwatch] start() was called with deliver: "check", so these checks send no alerts. ' \
|
|
378
|
+
'Another process must run checks with deliver: "now" (the default) to send them.'
|
|
379
|
+
end
|
|
380
|
+
@ticker = Ticker.new(ms / 1000.0, @first_tick_s) do
|
|
381
|
+
check
|
|
382
|
+
rescue StandardError => e
|
|
383
|
+
report(e, "check")
|
|
384
|
+
end
|
|
385
|
+
end
|
|
386
|
+
nil
|
|
387
|
+
end
|
|
388
|
+
|
|
389
|
+
def stop
|
|
390
|
+
after_fork_check
|
|
391
|
+
ticker = @ticker_lock.synchronize do
|
|
392
|
+
current = @ticker
|
|
393
|
+
@ticker = nil
|
|
394
|
+
current
|
|
395
|
+
end
|
|
396
|
+
ticker&.stop
|
|
397
|
+
nil
|
|
398
|
+
end
|
|
399
|
+
|
|
400
|
+
def close
|
|
401
|
+
stop
|
|
402
|
+
@store.close if @store.respond_to?(:close)
|
|
403
|
+
nil
|
|
404
|
+
end
|
|
405
|
+
|
|
406
|
+
# True in development or test. See Cronwatch::Environment.
|
|
407
|
+
def self.development?
|
|
408
|
+
Environment.development?
|
|
409
|
+
end
|
|
410
|
+
|
|
411
|
+
# Hands an error to on_error. An on_error that raises is not allowed to
|
|
412
|
+
# take the job down with it.
|
|
413
|
+
#
|
|
414
|
+
# @api private For integrations (Cronwatch::Web, ActiveJob, Sidekiq), not apps.
|
|
415
|
+
def report(error, where)
|
|
416
|
+
@on_error.call(error, where)
|
|
417
|
+
rescue StandardError => e
|
|
418
|
+
warn "[cronwatch] on_error raised #{e.class}: #{e.message} (reporting #{where}: #{error.message})"
|
|
419
|
+
end
|
|
420
|
+
|
|
421
|
+
private
|
|
422
|
+
|
|
423
|
+
# Locks, the check in flight, the interval thread and the channel and
|
|
424
|
+
# triage threads belong to one process. A forked child (Puma, Unicorn,
|
|
425
|
+
# Sidekiq) starts with fresh ones, so start and check work there.
|
|
426
|
+
def reset_process_state
|
|
427
|
+
@pid = Process.pid
|
|
428
|
+
@locks = {}
|
|
429
|
+
@check_lock = Mutex.new
|
|
430
|
+
@checking = nil
|
|
431
|
+
@ticker_lock = Mutex.new
|
|
432
|
+
@ticker = nil
|
|
433
|
+
@ticker_ms = nil
|
|
434
|
+
@ready_lock = Mutex.new
|
|
435
|
+
@sending_lock = Mutex.new
|
|
436
|
+
# Channel (by index) and triage threads that timed out and are still going.
|
|
437
|
+
@abandoned = {}
|
|
438
|
+
end
|
|
439
|
+
|
|
440
|
+
def after_fork_check
|
|
441
|
+
return if @pid == Process.pid
|
|
442
|
+
|
|
443
|
+
FORK_LOCK.synchronize { reset_process_state unless @pid == Process.pid }
|
|
444
|
+
end
|
|
445
|
+
|
|
446
|
+
# An exception outside StandardError: the thread is being stopped
|
|
447
|
+
# (Interrupt, SystemExit, Sidekiq::Shutdown, a Timeout) or the code is
|
|
448
|
+
# broken (NotImplementedError, LoadError).
|
|
449
|
+
def interrupted?(error)
|
|
450
|
+
!error.is_a?(StandardError)
|
|
451
|
+
end
|
|
452
|
+
|
|
453
|
+
def build_definition(name, options)
|
|
454
|
+
unknown = options.keys.map(&:to_sym) - JobDefinition::OPTIONS
|
|
455
|
+
raise ArgumentError, "job \"#{name}\": unknown option #{unknown.map(&:inspect).join(", ")}" if unknown.any?
|
|
456
|
+
|
|
457
|
+
fields = {}
|
|
458
|
+
@defaults.each { |k, v| fields[k] = v }
|
|
459
|
+
options.each do |k, v|
|
|
460
|
+
fields[k.to_sym] = v.is_a?(Hash) && k.to_sym == :budget ? v.transform_keys(&:to_s) : v
|
|
461
|
+
end
|
|
462
|
+
fields[:name] = name
|
|
463
|
+
JobDefinition.new(fields)
|
|
464
|
+
end
|
|
465
|
+
|
|
466
|
+
# Raises a clear error for options that would otherwise quietly turn a check off.
|
|
467
|
+
def validate_definition(definition)
|
|
468
|
+
name = definition.name
|
|
469
|
+
unless definition.schedule.nil?
|
|
470
|
+
unless definition.schedule.is_a?(String) && JS.trim(definition.schedule) != ""
|
|
471
|
+
raise ArgumentError, "job \"#{name}\": schedule must be a non-empty string"
|
|
472
|
+
end
|
|
473
|
+
|
|
474
|
+
Schedule.parse(definition.schedule, definition.timezone)
|
|
475
|
+
end
|
|
476
|
+
if !definition.timezone.nil? && !(definition.timezone.is_a?(String) && Zone.valid?(definition.timezone))
|
|
477
|
+
raise ArgumentError, "job \"#{name}\": timezone \"#{definition.timezone}\" is not an IANA timezone"
|
|
478
|
+
end
|
|
479
|
+
|
|
480
|
+
Duration.parse(definition.grace, "grace") unless definition.grace.nil?
|
|
481
|
+
if !definition.timeout.nil? && Duration.parse(definition.timeout, "timeout") <= 0
|
|
482
|
+
raise ArgumentError, "job \"#{name}\": timeout must be longer than zero"
|
|
483
|
+
end
|
|
484
|
+
if !definition.max_duration.nil? && Duration.parse(definition.max_duration, "maxDuration") <= 0
|
|
485
|
+
raise ArgumentError, "job \"#{name}\": maxDuration must be longer than zero"
|
|
486
|
+
end
|
|
487
|
+
|
|
488
|
+
failures = definition.failures_before_alert
|
|
489
|
+
if !failures.nil? && !(failures.is_a?(Numeric) && JS.integer?(failures) && failures >= 1)
|
|
490
|
+
raise ArgumentError, "job \"#{name}\": failuresBeforeAlert must be a whole number, 1 or more (got #{js_string(failures)})"
|
|
491
|
+
end
|
|
492
|
+
unless definition.budget.nil?
|
|
493
|
+
raise ArgumentError, "job \"#{name}\": budget must be an object of { metric: ceiling }" unless definition.budget.is_a?(Hash)
|
|
494
|
+
|
|
495
|
+
definition.budget.each do |metric, ceiling|
|
|
496
|
+
next if ceiling.is_a?(Numeric) && ceiling.real? && JS.finite?(ceiling) && ceiling >= 0
|
|
497
|
+
|
|
498
|
+
raise ArgumentError, "job \"#{name}\": budget.#{metric} must be a finite number, 0 or more (got #{js_string(ceiling)})"
|
|
499
|
+
end
|
|
500
|
+
end
|
|
501
|
+
expect = definition.expect
|
|
502
|
+
return if expect.nil? || expect.is_a?(String) || expect.is_a?(Regexp) || expect.respond_to?(:call)
|
|
503
|
+
|
|
504
|
+
raise ArgumentError, "job \"#{name}\": expect must be a string, a RegExp or a function"
|
|
505
|
+
end
|
|
506
|
+
|
|
507
|
+
# String(value) as JavaScript writes it, for the messages above.
|
|
508
|
+
def js_string(value)
|
|
509
|
+
case value
|
|
510
|
+
when nil then "null"
|
|
511
|
+
when Numeric then JS.number(value)
|
|
512
|
+
else value.to_s
|
|
513
|
+
end
|
|
514
|
+
end
|
|
515
|
+
|
|
516
|
+
def ensure_ready
|
|
517
|
+
return if @ready
|
|
518
|
+
|
|
519
|
+
@ready_lock.synchronize do
|
|
520
|
+
next if @ready
|
|
521
|
+
|
|
522
|
+
@store.init if @store.respond_to?(:init)
|
|
523
|
+
if @using_default_store && Environment.production?
|
|
524
|
+
warn "[cronwatch] using the in-memory store: runs and state are lost on restart. " \
|
|
525
|
+
"Pass a store such as Cronwatch::Stores::ActiveRecord."
|
|
526
|
+
end
|
|
527
|
+
# Only once init has gone through: a failure is tried again on the next call.
|
|
528
|
+
@ready = true
|
|
529
|
+
end
|
|
530
|
+
end
|
|
531
|
+
|
|
532
|
+
def sync(definition)
|
|
533
|
+
ensure_ready
|
|
534
|
+
return if @registry.synchronize { @synced.include?(definition.name) }
|
|
535
|
+
|
|
536
|
+
@store.upsert_job(Serialize.to_stored(definition), now)
|
|
537
|
+
@registry.synchronize { @synced << definition.name }
|
|
538
|
+
end
|
|
539
|
+
|
|
540
|
+
# Runs the block while holding the job's lock, so two runs (or a run and a
|
|
541
|
+
# check) in this process never read and write the job's state over each
|
|
542
|
+
# other. Other processes are coordinated by update_state instead. Only
|
|
543
|
+
# store reads and writes happen inside; alerts are sent outside it.
|
|
544
|
+
def serial(job, &block)
|
|
545
|
+
after_fork_check
|
|
546
|
+
lock = @registry.synchronize { @locks[job] ||= Monitor.new }
|
|
547
|
+
lock.synchronize(&block)
|
|
548
|
+
end
|
|
549
|
+
|
|
550
|
+
def read_state(job)
|
|
551
|
+
Evaluate.normalize_state(@store.get_state(job), job)
|
|
552
|
+
end
|
|
553
|
+
|
|
554
|
+
def same_state?(a, b)
|
|
555
|
+
JS.json(a.to_h) == JS.json(b.to_h)
|
|
556
|
+
end
|
|
557
|
+
|
|
558
|
+
# Every read-modify-write of a job's state goes through here. In turn with
|
|
559
|
+
# this process's other updates to the job (serial), it reads the state,
|
|
560
|
+
# yields it for the next one and a result (`[state, result]`), and writes
|
|
561
|
+
# that with the version one higher, only if the stored version is still
|
|
562
|
+
# the one read. When another process wrote in between, the write is
|
|
563
|
+
# refused and it starts again from a fresh read, up to STATE_ATTEMPTS
|
|
564
|
+
# times. So the block may run more than once and must only compute:
|
|
565
|
+
# whatever it returns from the attempt that was written is the result.
|
|
566
|
+
# Nothing is written when the state is unchanged. Returns
|
|
567
|
+
# `[state as stored, result]`.
|
|
568
|
+
def update_state(job)
|
|
569
|
+
serial(job) do
|
|
570
|
+
attempt = 0
|
|
571
|
+
loop do
|
|
572
|
+
attempt += 1
|
|
573
|
+
current = read_state(job)
|
|
574
|
+
state, result = yield(current)
|
|
575
|
+
break [current, result] if same_state?(state, current)
|
|
576
|
+
|
|
577
|
+
version = current.version || 0
|
|
578
|
+
following = state.dup
|
|
579
|
+
following.version = version + 1
|
|
580
|
+
break [following, result] if write_state(following, version)
|
|
581
|
+
if attempt >= STATE_ATTEMPTS
|
|
582
|
+
raise "the state of #{job} changed under #{STATE_ATTEMPTS} attempts in a row to update it; gave up"
|
|
583
|
+
end
|
|
584
|
+
end
|
|
585
|
+
end
|
|
586
|
+
end
|
|
587
|
+
|
|
588
|
+
# A conditional write, or for a store without compare_and_set_state, a
|
|
589
|
+
# plain one that always succeeds.
|
|
590
|
+
def write_state(state, expected_version)
|
|
591
|
+
return @store.compare_and_set_state(state, expected_version) if @store.respond_to?(:compare_and_set_state)
|
|
592
|
+
|
|
593
|
+
@store.set_state(state)
|
|
594
|
+
true
|
|
595
|
+
end
|
|
596
|
+
|
|
597
|
+
# A custom redact, made safe: one that raises or returns something other
|
|
598
|
+
# than a String is reported and the default is used instead, so a broken
|
|
599
|
+
# redact neither stops the run finishing nor leaks what it was given.
|
|
600
|
+
def guarded_redact(redact)
|
|
601
|
+
lambda do |text|
|
|
602
|
+
out = redact.call(text)
|
|
603
|
+
raise TypeError, "redact must return a string, not #{out.nil? ? "null" : out.class}" unless out.is_a?(String)
|
|
604
|
+
|
|
605
|
+
Output.utf8(out)
|
|
606
|
+
rescue StandardError => e
|
|
607
|
+
report(e, "redact")
|
|
608
|
+
Output.redact_secrets(text)
|
|
609
|
+
end
|
|
610
|
+
end
|
|
611
|
+
|
|
612
|
+
# Identifies an alert across retries.
|
|
613
|
+
def alert_key(alert)
|
|
614
|
+
"#{alert.type}|#{alert.at}|#{alert.run&.id}"
|
|
615
|
+
end
|
|
616
|
+
|
|
617
|
+
# Whether a check already marked this run as timed out, for a failure that finished late.
|
|
618
|
+
def marked_timed_out?(run)
|
|
619
|
+
return false if run.status == :ok
|
|
620
|
+
|
|
621
|
+
@store.get_run(run.id)&.status == :timeout
|
|
622
|
+
rescue StandardError
|
|
623
|
+
false
|
|
624
|
+
end
|
|
625
|
+
|
|
626
|
+
# Record a finished run (ok, failed, or timed out by a check), evaluate it
|
|
627
|
+
# against the job's state and send what that produces. Never raises.
|
|
628
|
+
def finish_run(definition, run, at, write)
|
|
629
|
+
drafts = nil
|
|
630
|
+
begin
|
|
631
|
+
@store.update_run(run) if write
|
|
632
|
+
past = nil
|
|
633
|
+
_, drafts = update_state(run.job) do |previous|
|
|
634
|
+
past ||= history(run)
|
|
635
|
+
settled = Evaluate.apply_silence(previous, Evaluate.on_run_finish(definition, run, previous, past, at), at)
|
|
636
|
+
[settled.state, settled.alerts]
|
|
637
|
+
end
|
|
638
|
+
rescue StandardError => e
|
|
639
|
+
report(e, "evaluating #{run.job}")
|
|
640
|
+
return []
|
|
641
|
+
end
|
|
642
|
+
dispatch(drafts, definition, at)
|
|
643
|
+
end
|
|
644
|
+
|
|
645
|
+
# The runs before `run`, newest first, with up to BASELINE_WINDOW
|
|
646
|
+
# successful ones when the store has them. One small read normally; a
|
|
647
|
+
# larger one only when failures crowd the successes out of it.
|
|
648
|
+
def history(run)
|
|
649
|
+
runs = @store.list_runs(run.job, HISTORY_PAGE)
|
|
650
|
+
if runs.length == HISTORY_PAGE && !Evaluate.full_baseline?(runs.reject { |r| r.id == run.id })
|
|
651
|
+
runs = @store.list_runs(run.job, HISTORY_MAX)
|
|
652
|
+
end
|
|
653
|
+
runs.reject { |r| r.id == run.id }
|
|
654
|
+
end
|
|
655
|
+
|
|
656
|
+
def run_check
|
|
657
|
+
ensure_ready
|
|
658
|
+
defined_jobs.each { |definition| sync(definition) }
|
|
659
|
+
at = now
|
|
660
|
+
alerts = []
|
|
661
|
+
|
|
662
|
+
# Runs that never reported back. One that cannot be judged (its job's
|
|
663
|
+
# stored timeout no longer parses, say) is reported and skipped.
|
|
664
|
+
@store.running_runs.each do |run|
|
|
665
|
+
declared = @registry.synchronize { @definitions[run.job] }
|
|
666
|
+
definition = declared ? Serialize.to_stored(declared) : @store.get_job(run.job)&.definition
|
|
667
|
+
next if definition.nil? || !Evaluate.stuck?(definition, run, at)
|
|
668
|
+
|
|
669
|
+
timeout = Evaluate.timeout_ms(definition)
|
|
670
|
+
run.status = :timeout
|
|
671
|
+
run.finished_at = at
|
|
672
|
+
run.duration_ms = at - run.started_at
|
|
673
|
+
run.error = "Still running after #{Duration.format(timeout)}; marked as timed out"
|
|
674
|
+
alerts.concat(finish_run(definition, run, at, true))
|
|
675
|
+
rescue StandardError => e
|
|
676
|
+
report(e, "checking #{run.job}")
|
|
677
|
+
end
|
|
678
|
+
|
|
679
|
+
# Each job on its own: one that cannot be evaluated (a stored schedule
|
|
680
|
+
# this process cannot read, one a Node process wrote, say) is reported,
|
|
681
|
+
# shown as failing (see Evaluate.unevaluable_summary) and does not stop
|
|
682
|
+
# the others.
|
|
683
|
+
jobs = []
|
|
684
|
+
retries = RetryBudget.new(0)
|
|
685
|
+
@store.list_jobs.each do |stored|
|
|
686
|
+
recent = @store.list_runs(stored.name, Evaluate::BASELINE_WINDOW)
|
|
687
|
+
next_expected_at = nil
|
|
688
|
+
state, drafts = update_state(stored.name) do |previous|
|
|
689
|
+
evaluation = Evaluate.on_check(stored.definition, stored, recent.first, previous, at)
|
|
690
|
+
next_expected_at = evaluation.next_expected_at
|
|
691
|
+
settled = Evaluate.apply_silence(previous, evaluation, at)
|
|
692
|
+
[settled.state, settled.alerts]
|
|
693
|
+
end
|
|
694
|
+
alerts.concat(retry_undelivered(stored.name, state, at, retries))
|
|
695
|
+
alerts.concat(dispatch(drafts, stored.definition, at))
|
|
696
|
+
jobs << Evaluate.summarize(stored, recent, state, next_expected_at, at)
|
|
697
|
+
rescue StandardError => e
|
|
698
|
+
report(e, "checking #{stored.name}")
|
|
699
|
+
jobs << unevaluable(stored, at)
|
|
700
|
+
end
|
|
701
|
+
|
|
702
|
+
pruned = 0
|
|
703
|
+
if at - @last_prune_at > PRUNE_INTERVAL_MS
|
|
704
|
+
@last_prune_at = at
|
|
705
|
+
begin
|
|
706
|
+
pruned = @store.prune(at - @retention_ms)
|
|
707
|
+
rescue StandardError => e
|
|
708
|
+
report(e, "pruning")
|
|
709
|
+
end
|
|
710
|
+
end
|
|
711
|
+
|
|
712
|
+
CheckResult.new(checked_at: at, jobs: jobs, alerts: alerts, pruned: pruned)
|
|
713
|
+
end
|
|
714
|
+
|
|
715
|
+
# A job's summary and its newest runs, without alerting. A job that cannot
|
|
716
|
+
# be evaluated (a stored schedule this process cannot read, say) is
|
|
717
|
+
# reported and shown as failing.
|
|
718
|
+
def snapshot(stored, at, count)
|
|
719
|
+
recent = []
|
|
720
|
+
begin
|
|
721
|
+
recent = @store.list_runs(stored.name, [count, Evaluate::BASELINE_WINDOW].max)
|
|
722
|
+
state = read_state(stored.name)
|
|
723
|
+
next_expected_at = Evaluate.on_check(stored.definition, stored, recent.first, state, at).next_expected_at
|
|
724
|
+
JobWithRuns.new(job: Evaluate.summarize(stored, recent, state, next_expected_at, at), runs: recent.first(count))
|
|
725
|
+
rescue StandardError => e
|
|
726
|
+
report(e, "reading #{stored.name}")
|
|
727
|
+
JobWithRuns.new(job: unevaluable(stored, at), runs: recent.first(count))
|
|
728
|
+
end
|
|
729
|
+
end
|
|
730
|
+
|
|
731
|
+
# The summary of a job whose evaluation failed, from whatever can still be read.
|
|
732
|
+
def unevaluable(stored, at)
|
|
733
|
+
recent = begin
|
|
734
|
+
@store.list_runs(stored.name, Evaluate::BASELINE_WINDOW)
|
|
735
|
+
rescue StandardError
|
|
736
|
+
[]
|
|
737
|
+
end
|
|
738
|
+
state = begin
|
|
739
|
+
read_state(stored.name)
|
|
740
|
+
rescue StandardError
|
|
741
|
+
Evaluate.empty_state(stored.name)
|
|
742
|
+
end
|
|
743
|
+
Evaluate.unevaluable_summary(stored, recent, state, at)
|
|
744
|
+
end
|
|
745
|
+
|
|
746
|
+
# Read, change and write one job's state, in turn with every other update to it.
|
|
747
|
+
def patch_state(name)
|
|
748
|
+
ensure_ready
|
|
749
|
+
state, = update_state(name) do |current|
|
|
750
|
+
following = Evaluate.normalize_state(current, name)
|
|
751
|
+
yield following
|
|
752
|
+
[following, nil]
|
|
753
|
+
end
|
|
754
|
+
state
|
|
755
|
+
end
|
|
756
|
+
|
|
757
|
+
# Compose, triage and send each draft. The state was saved before this
|
|
758
|
+
# (update_state), so a slow channel holds up nothing else; afterwards only
|
|
759
|
+
# the delivery fields are written back, onto a fresh read of the state.
|
|
760
|
+
def dispatch(drafts, definition, at)
|
|
761
|
+
return [] if drafts.empty?
|
|
762
|
+
|
|
763
|
+
composed = []
|
|
764
|
+
delivered = []
|
|
765
|
+
failed = []
|
|
766
|
+
drafts.each do |draft|
|
|
767
|
+
alert = Format.compose_alert(draft, definition, at)
|
|
768
|
+
if @defer_delivery
|
|
769
|
+
failed << alert
|
|
770
|
+
else
|
|
771
|
+
add_triage(alert, @triage_timeout_ms) if @triage && alert.type != :recovered
|
|
772
|
+
(deliver(alert) ? delivered : failed) << alert
|
|
773
|
+
end
|
|
774
|
+
composed << alert
|
|
775
|
+
end
|
|
776
|
+
record_delivery(definition.name, delivered, failed, [], at)
|
|
777
|
+
composed
|
|
778
|
+
end
|
|
779
|
+
|
|
780
|
+
# The wall-clock milliseconds one check has spent retrying, across its jobs.
|
|
781
|
+
RetryBudget = Struct.new(:spent_ms)
|
|
782
|
+
|
|
783
|
+
# Send the alerts that no channel accepted last time, once each, oldest
|
|
784
|
+
# first. `state` is the job's state as this check left it: an alert that
|
|
785
|
+
# no longer describes it (Evaluate.stale_alert?) is dropped instead.
|
|
786
|
+
# Retries across a check share RETRY_BUDGET_MS of wall-clock time; once it
|
|
787
|
+
# is spent the rest stay queued for the next check.
|
|
788
|
+
def retry_undelivered(name, state, at, budget)
|
|
789
|
+
pending = state.undelivered || []
|
|
790
|
+
return [] if pending.empty? || Evaluate.silenced?(state, at) || @defer_delivery
|
|
791
|
+
|
|
792
|
+
delivered = []
|
|
793
|
+
failed = []
|
|
794
|
+
dropped = pending.select { |alert| Evaluate.stale_alert?(alert, state) }
|
|
795
|
+
pending.each do |alert|
|
|
796
|
+
next if dropped.any? { |d| d.equal?(alert) }
|
|
797
|
+
|
|
798
|
+
left = @retry_budget_ms - budget.spent_ms
|
|
799
|
+
break if left <= 0
|
|
800
|
+
|
|
801
|
+
started = AbortSignal.monotonic
|
|
802
|
+
# An alert queued by a process that delivers at check time was never
|
|
803
|
+
# triaged. One that was tried (triage: null) is not tried again.
|
|
804
|
+
add_triage(alert, [@triage_timeout_ms, left].min) if @triage && alert.type != :recovered && !alert.triage_tried?
|
|
805
|
+
(deliver(alert) ? delivered : failed) << alert
|
|
806
|
+
budget.spent_ms += [0, ((AbortSignal.monotonic - started) * 1000).round].max
|
|
807
|
+
end
|
|
808
|
+
record_delivery(name, delivered, failed, dropped, at)
|
|
809
|
+
delivered
|
|
810
|
+
end
|
|
811
|
+
|
|
812
|
+
# Mark delivered alerts done, drop stale ones, and keep failed ones for the
|
|
813
|
+
# next check. A failed alert replaces its stored copy, so a triage made on
|
|
814
|
+
# this attempt is kept. last_alert_at moves only on a delivery.
|
|
815
|
+
def record_delivery(name, delivered, failed, dropped, at)
|
|
816
|
+
_, trimmed = update_state(name) do |previous|
|
|
817
|
+
state = Evaluate.normalize_state(previous, name)
|
|
818
|
+
done = (delivered + dropped).map { |a| alert_key(a) }.to_set
|
|
819
|
+
retried = failed.to_h { |a| [alert_key(a), a] }
|
|
820
|
+
kept = state.undelivered.reject { |a| done.include?(alert_key(a)) }.map { |a| retried.fetch(alert_key(a), a) }
|
|
821
|
+
known = kept.map { |a| alert_key(a) }.to_set
|
|
822
|
+
kept.concat(failed.reject { |a| known.include?(alert_key(a)) })
|
|
823
|
+
state.undelivered = kept.last(MAX_UNDELIVERED)
|
|
824
|
+
state.last_alert_at = at if delivered.any?
|
|
825
|
+
[state, [0, kept.length - MAX_UNDELIVERED].max]
|
|
826
|
+
end
|
|
827
|
+
if trimmed.positive?
|
|
828
|
+
report(RuntimeError.new("#{trimmed} undelivered alert#{trimmed == 1 ? "" : "s"} for #{name} dropped: " \
|
|
829
|
+
"only the newest #{MAX_UNDELIVERED} are kept for retry"), "alert queue for #{name}")
|
|
830
|
+
end
|
|
831
|
+
rescue StandardError => e
|
|
832
|
+
report(e, "recording alert delivery for #{name}")
|
|
833
|
+
end
|
|
834
|
+
|
|
835
|
+
# Send to every channel at once, each in its own thread with its own
|
|
836
|
+
# timeout. True when at least one accepted it, or there are none. A
|
|
837
|
+
# channel that times out is left to finish on its own, and nothing more
|
|
838
|
+
# is sent to it until it has: meanwhile its alerts count as not
|
|
839
|
+
# delivered there, to be retried by a later check. So a hung channel
|
|
840
|
+
# holds one thread, not one per alert.
|
|
841
|
+
def deliver(alert)
|
|
842
|
+
return true if @alerts.empty?
|
|
843
|
+
|
|
844
|
+
outcomes = Array.new(@alerts.length)
|
|
845
|
+
threads = @alerts.each_with_index.map do |channel, i|
|
|
846
|
+
if @sending_lock.synchronize { @abandoned[i]&.alive? }
|
|
847
|
+
outcomes[i] = TimeoutError.new("skipped: an earlier alert timed out and is still being sent")
|
|
848
|
+
next nil
|
|
849
|
+
end
|
|
850
|
+
|
|
851
|
+
Thread.new do
|
|
852
|
+
Thread.current.report_on_exception = false
|
|
853
|
+
channel.call(alert)
|
|
854
|
+
outcomes[i] = true
|
|
855
|
+
rescue StandardError, ScriptError => e
|
|
856
|
+
outcomes[i] = e
|
|
857
|
+
end
|
|
858
|
+
end
|
|
859
|
+
deadline = AbortSignal.monotonic + (@channel_timeout_ms / 1000.0)
|
|
860
|
+
results = threads.each_with_index.map do |thread, i|
|
|
861
|
+
next outcomes[i] if thread.nil?
|
|
862
|
+
next outcomes[i] if thread.join([deadline - AbortSignal.monotonic, 0].max)
|
|
863
|
+
|
|
864
|
+
@sending_lock.synchronize { @abandoned[i] = thread }
|
|
865
|
+
TimeoutError.new("timed out after #{@channel_timeout_ms}ms")
|
|
866
|
+
end
|
|
867
|
+
results.each_with_index do |result, i|
|
|
868
|
+
report(result, "alert channel #{channel_name(@alerts[i])}") unless result == true
|
|
869
|
+
end
|
|
870
|
+
results.include?(true)
|
|
871
|
+
end
|
|
872
|
+
|
|
873
|
+
def channel_name(channel)
|
|
874
|
+
channel.respond_to?(:name) && channel.name ? channel.name : channel.class.name
|
|
875
|
+
end
|
|
876
|
+
|
|
877
|
+
# Sets the alert's triage to the diagnosis, or to nil (JSON null) when
|
|
878
|
+
# there is none (it raised, timed out or answered nil or ""), so it is
|
|
879
|
+
# tried once per alert. While a triage that timed out is still going,
|
|
880
|
+
# alerts go out without one rather than start another beside it.
|
|
881
|
+
def add_triage(alert, timeout_ms)
|
|
882
|
+
signal = AbortSignal.new
|
|
883
|
+
if @sending_lock.synchronize { @abandoned[:triage]&.alive? }
|
|
884
|
+
raise TimeoutError, "skipped: an earlier triage timed out and is still running"
|
|
885
|
+
end
|
|
886
|
+
|
|
887
|
+
recent = @store.list_runs(alert.job, 5)
|
|
888
|
+
context = TriageContext.new(alert: alert, recent_runs: recent, signal: signal)
|
|
889
|
+
outcome = nil
|
|
890
|
+
thread = Thread.new do
|
|
891
|
+
Thread.current.report_on_exception = false
|
|
892
|
+
outcome = [:ok, @triage.call(context)]
|
|
893
|
+
rescue StandardError, ScriptError => e
|
|
894
|
+
outcome = [:error, e]
|
|
895
|
+
end
|
|
896
|
+
unless thread.join(timeout_ms / 1000.0)
|
|
897
|
+
@sending_lock.synchronize { @abandoned[:triage] = thread }
|
|
898
|
+
raise TimeoutError, "timed out after #{timeout_ms}ms"
|
|
899
|
+
end
|
|
900
|
+
raise outcome[1] if outcome[0] == :error
|
|
901
|
+
|
|
902
|
+
diagnosis = outcome[1]
|
|
903
|
+
alert.triage_result = diagnosis.is_a?(String) && !diagnosis.empty? ? Output.utf8(diagnosis) : nil
|
|
904
|
+
rescue StandardError => e
|
|
905
|
+
signal&.abort!
|
|
906
|
+
alert.triage_result = nil
|
|
907
|
+
report(e, "triage for #{alert.job}")
|
|
908
|
+
end
|
|
909
|
+
|
|
910
|
+
# A whole number in range, or the fallback for anything that is not a number.
|
|
911
|
+
def clamp_limit(limit, fallback, min)
|
|
912
|
+
n = limit.is_a?(Numeric) && limit.real? && JS.finite?(limit) ? limit.truncate : fallback
|
|
913
|
+
[500, [min, n].max].min
|
|
914
|
+
end
|
|
915
|
+
|
|
916
|
+
def default_on_error(error, where)
|
|
917
|
+
message = "[cronwatch] #{where}: #{error.class}: #{error.message}"
|
|
918
|
+
if defined?(::Rails) && ::Rails.respond_to?(:logger) && ::Rails.logger
|
|
919
|
+
::Rails.logger.error(message)
|
|
920
|
+
else
|
|
921
|
+
warn message
|
|
922
|
+
end
|
|
923
|
+
end
|
|
924
|
+
end
|
|
925
|
+
end
|