cronwatch 0.3.1 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +19 -3
- data/lib/cronwatch/alerts/bugsnag.rb +69 -0
- data/lib/cronwatch/alerts/custom.rb +9 -2
- data/lib/cronwatch/alerts/datadog.rb +54 -0
- data/lib/cronwatch/alerts/email.rb +78 -0
- data/lib/cronwatch/alerts/honeybadger.rb +59 -0
- data/lib/cronwatch/alerts/mailgun.rb +36 -0
- data/lib/cronwatch/alerts/newrelic.rb +52 -0
- data/lib/cronwatch/alerts/postmark.rb +41 -0
- data/lib/cronwatch/alerts/provider.rb +159 -0
- data/lib/cronwatch/alerts/resend.rb +39 -0
- data/lib/cronwatch/alerts/rollbar.rb +53 -0
- data/lib/cronwatch/alerts/sendgrid.rb +41 -0
- data/lib/cronwatch/alerts/sentry.rb +94 -0
- data/lib/cronwatch/alerts/ses.rb +72 -0
- data/lib/cronwatch/alerts/sigv4.rb +84 -0
- data/lib/cronwatch/alerts/twilio.rb +206 -0
- data/lib/cronwatch/alerts/webhook.rb +7 -2
- data/lib/cronwatch/client.rb +473 -52
- data/lib/cronwatch/evaluate.rb +14 -1
- data/lib/cronwatch/format.rb +10 -4
- data/lib/cronwatch/http.rb +67 -6
- data/lib/cronwatch/job.rb +17 -0
- data/lib/cronwatch/pg_cron.rb +524 -0
- data/lib/cronwatch/run_handle.rb +203 -0
- data/lib/cronwatch/stores/active_record.rb +23 -0
- data/lib/cronwatch/stores/memory.rb +33 -10
- data/lib/cronwatch/types.rb +1 -0
- data/lib/cronwatch/version.rb +1 -1
- data/lib/cronwatch/web/app.rb +38 -11
- data/lib/cronwatch/web/origin.rb +133 -0
- data/lib/cronwatch/web.rb +1 -0
- data/lib/cronwatch.rb +17 -1
- metadata +19 -1
data/lib/cronwatch/client.rb
CHANGED
|
@@ -40,9 +40,24 @@ module Cronwatch
|
|
|
40
40
|
# Guards the reset a forked child makes of the parent's locks and threads.
|
|
41
41
|
FORK_LOCK = Mutex.new
|
|
42
42
|
SILENCE_OPTIONS = %i[for].freeze
|
|
43
|
+
# Run ids that start with this belong to the pg_cron source (Sources::PgCron).
|
|
44
|
+
RESERVED_RUN_ID_PREFIX = "pgcron:"
|
|
43
45
|
|
|
44
46
|
# What execute returns: the recorded run, and the block's own outcome.
|
|
45
47
|
ExecuteResult = Struct.new(:run, :result, :error, :threw, keyword_init: true)
|
|
48
|
+
# What a channel's call receives with each alert (the SDK's ChannelContext).
|
|
49
|
+
# on_error(error) reports a problem that did not stop the alert going
|
|
50
|
+
# out, such as one of several recipients refusing it, to the client's on_error.
|
|
51
|
+
class ChannelContext
|
|
52
|
+
def initialize(on_error)
|
|
53
|
+
@on_error = on_error
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
def on_error(error)
|
|
57
|
+
@on_error.call(error)
|
|
58
|
+
nil
|
|
59
|
+
end
|
|
60
|
+
end
|
|
46
61
|
# What a triage callable receives. Pass the signal to anything that can stop early.
|
|
47
62
|
TriageContext = Struct.new(:alert, :recent_runs, :signal, keyword_init: true)
|
|
48
63
|
# A job's summary and its newest runs, as the dashboard shows them.
|
|
@@ -52,7 +67,7 @@ module Cronwatch
|
|
|
52
67
|
def to_h = { "job" => job.to_h, "runs" => runs.map(&:to_h) }
|
|
53
68
|
end
|
|
54
69
|
|
|
55
|
-
attr_reader :store, :alerts, :triage, :cron_secret, :retention_ms, :defaults
|
|
70
|
+
attr_reader :store, :alerts, :triage, :cron_secret, :retention_ms, :defaults, :sources
|
|
56
71
|
|
|
57
72
|
# store: where jobs, runs and state live. Defaults to an in-memory store that forgets on restart.
|
|
58
73
|
# alerts: where alerts go: objects with #call(alert) and #name. Defaults to the console.
|
|
@@ -70,11 +85,18 @@ module Cronwatch
|
|
|
70
85
|
# now: the clock, a callable returning epoch milliseconds. Tests use this.
|
|
71
86
|
# on_error: called with (error, where) for anything that goes wrong outside a job: the store failing,
|
|
72
87
|
# an alert channel failing, a triage timeout.
|
|
88
|
+
# sources: where runs this process does not wrap come from, such as pg_cron jobs
|
|
89
|
+
# (Cronwatch::Sources::PgCron). Each is synced at the start of every check; one that
|
|
90
|
+
# raises is reported to on_error and the check carries on. See "Sources" in DESIGN.md.
|
|
73
91
|
def initialize(store: nil, alerts: nil, triage: nil, cron_secret: UNSET, retention: "30d", defaults: {}, redact: nil,
|
|
74
|
-
deliver: :now, now: nil, on_error: nil)
|
|
92
|
+
deliver: :now, now: nil, on_error: nil, sources: nil)
|
|
75
93
|
@using_default_store = store.nil?
|
|
76
94
|
@store = store || Stores::Memory.new
|
|
77
95
|
@alerts = alerts.nil? ? [Alerts::Console.new] : Array(alerts)
|
|
96
|
+
@sources = sources.nil? ? [] : Array(sources)
|
|
97
|
+
@sources.each do |source|
|
|
98
|
+
raise ArgumentError, "a source must respond to sync(host)" unless source.respond_to?(:sync)
|
|
99
|
+
end
|
|
78
100
|
@triage = triage
|
|
79
101
|
secret = cron_secret.equal?(UNSET) ? ENV.fetch("CRON_SECRET", nil) : cron_secret
|
|
80
102
|
@cron_secret = secret.nil? || secret.to_s.empty? ? nil : secret.to_s
|
|
@@ -212,54 +234,143 @@ module Cronwatch
|
|
|
212
234
|
returned = result.is_a?(String) ? Output.utf8(result) : nil
|
|
213
235
|
run.output = recorder.output || (returned && Output.cap(returned))
|
|
214
236
|
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
237
|
+
conclude(definition, run, result, error, threw, recorder.expect_text || returned, failure: failure)
|
|
238
|
+
begin
|
|
239
|
+
ignored = record_finish(definition, run, recorded, finished_at)
|
|
240
|
+
report(RuntimeError.new("run #{run.id} of #{name} #{ignored}; ignored"), "finishing #{name}") if ignored
|
|
241
|
+
rescue StandardError => e
|
|
242
|
+
report(e, "recording #{name}")
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
raise error if threw && interrupted?(error)
|
|
246
|
+
|
|
247
|
+
ExecuteResult.new(run: run, result: result, error: error, threw: threw)
|
|
248
|
+
end
|
|
249
|
+
|
|
250
|
+
# A handle on a run started elsewhere, as job(name).resume(run_id). The
|
|
251
|
+
# job must be declared in this process, or it raises ArgumentError.
|
|
252
|
+
def resume_run(name, run_id)
|
|
253
|
+
name = name.to_s if name.is_a?(Symbol)
|
|
254
|
+
definition = @registry.synchronize { @definitions[name] }
|
|
255
|
+
raise ArgumentError, "resume_run: job \"#{name}\" is not declared; call job first" unless definition
|
|
256
|
+
|
|
257
|
+
resume_handle(definition, run_id)
|
|
258
|
+
end
|
|
259
|
+
|
|
260
|
+
# JobHandle#start: records a running run and returns a RunHandle to
|
|
261
|
+
# finish it. Two starts with one id at once in this process record one
|
|
262
|
+
# run: the second waits for the first, then finds its run.
|
|
263
|
+
#
|
|
264
|
+
# @api private For JobHandle#start, not apps.
|
|
265
|
+
def start_run(definition, trigger: nil, id: nil)
|
|
266
|
+
after_fork_check
|
|
267
|
+
trigger = "start" if trigger.nil?
|
|
268
|
+
return record_start(definition, trigger, nil) if id.nil?
|
|
269
|
+
|
|
270
|
+
check_run_id(definition.name, id, "start")
|
|
271
|
+
# Keyed by job as well, so another job's start with the same id is not
|
|
272
|
+
# handed this job's run: it fails as it would one call later.
|
|
273
|
+
key = "#{definition.name}\n#{id}"
|
|
274
|
+
entry = @registry.synchronize do
|
|
275
|
+
slot = (@starting[key] ||= [Mutex.new, 0])
|
|
276
|
+
slot[1] += 1
|
|
277
|
+
slot
|
|
278
|
+
end
|
|
279
|
+
begin
|
|
280
|
+
entry[0].synchronize { record_start(definition, trigger, id) }
|
|
281
|
+
ensure
|
|
282
|
+
@registry.synchronize do
|
|
283
|
+
entry[1] -= 1
|
|
284
|
+
@starting.delete(key) if entry[1].zero? && @starting[key].equal?(entry)
|
|
285
|
+
end
|
|
286
|
+
end
|
|
287
|
+
end
|
|
288
|
+
|
|
289
|
+
# JobHandle#resume and resume_run. A store that cannot be read is
|
|
290
|
+
# reported, and the handle's finish reads it again.
|
|
291
|
+
#
|
|
292
|
+
# @api private For JobHandle#resume, not apps.
|
|
293
|
+
def resume_handle(definition, run_id)
|
|
294
|
+
after_fork_check
|
|
295
|
+
check_run_id(definition.name, run_id, "resume")
|
|
296
|
+
begin
|
|
297
|
+
ensure_ready
|
|
298
|
+
stored = @store.get_run(run_id)
|
|
299
|
+
rescue StandardError => e
|
|
300
|
+
report(e, "resuming #{definition.name}")
|
|
301
|
+
return run_handle(definition, run_id, nil, true, nil)
|
|
302
|
+
end
|
|
303
|
+
return run_handle(definition, run_id, nil, true, "was not found") unless stored
|
|
304
|
+
|
|
305
|
+
existing_handle(definition, stored)
|
|
306
|
+
end
|
|
307
|
+
|
|
308
|
+
# Record a run that happened outside this process, for a source. Its job
|
|
309
|
+
# must be declared with job first. Runs are keyed by id: a new one is
|
|
310
|
+
# inserted, a stored one still running (or marked timeout by a check) is
|
|
311
|
+
# finished when this one is not running, and anything else is left alone,
|
|
312
|
+
# so recording the same run twice changes nothing. Finishing is
|
|
313
|
+
# conditional (the store's update_run_if): when two processes record the
|
|
314
|
+
# same finish, only the one whose write lands evaluates it, and the other
|
|
315
|
+
# reports it as already finished. A stored run of another job is left
|
|
316
|
+
# alone and reported. A finished run is judged as if it had been wrapped
|
|
317
|
+
# here (expect, failures, duration, budgets) and its output and error are
|
|
318
|
+
# redacted the same way; one finishing after a check marked it timeout is
|
|
319
|
+
# judged only when it succeeded, as RunHandle#finish does. `evaluate:
|
|
320
|
+
# false` stores it without judging it, for history imported on first
|
|
321
|
+
# sight. Returns the alerts it sent.
|
|
322
|
+
#
|
|
323
|
+
# `run` is a Cronwatch::Run, or a hash of its fields (camelCase or snake_case keys).
|
|
324
|
+
def record_run(run, evaluate: true)
|
|
325
|
+
after_fork_check
|
|
326
|
+
input = run.is_a?(Run) ? run : Run.from_h(run.is_a?(Hash) ? run.to_h { |k, v| [Naming.camel(k), v] } : run)
|
|
327
|
+
declared = @registry.synchronize { @definitions[input.job] }
|
|
328
|
+
raise ArgumentError, "record_run: job \"#{input.job}\" is not declared; call job first" unless declared
|
|
329
|
+
|
|
330
|
+
sync(declared)
|
|
331
|
+
run = input.dup
|
|
332
|
+
run.status = run.status&.to_sym
|
|
333
|
+
run.metrics = (run.metrics || {}).transform_keys(&:to_s)
|
|
334
|
+
run.output = Output.utf8(run.output.to_s) unless run.output.nil?
|
|
335
|
+
run.error = Output.utf8(run.error.to_s) unless run.error.nil?
|
|
336
|
+
if run.status == :ok
|
|
337
|
+
unmet = Serialize.check_expectation(declared.expect, run.output)
|
|
224
338
|
if unmet
|
|
225
339
|
run.status = :failed
|
|
226
340
|
run.error = unmet
|
|
227
|
-
else
|
|
228
|
-
run.status = :ok
|
|
229
341
|
end
|
|
230
342
|
end
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
run.error = Output.strip_nul(@redact.call(run.error)) unless run.error.nil?
|
|
343
|
+
run.output = Output.strip_nul(@redact.call(Output.cap(run.output))) unless run.output.nil?
|
|
344
|
+
run.error = Output.strip_nul(@redact.call(Output.cap(run.error))) unless run.error.nil?
|
|
345
|
+
definition = Serialize.to_stored(declared)
|
|
235
346
|
|
|
236
|
-
stored =
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
finish_run(stored, run, finished_at, true)
|
|
248
|
-
else
|
|
249
|
-
# The start was never written; the store may be back by now.
|
|
250
|
-
begin
|
|
251
|
-
sync(definition)
|
|
252
|
-
@store.insert_run(run)
|
|
253
|
-
recorded = true
|
|
254
|
-
rescue StandardError => e
|
|
255
|
-
report(e, "recording #{name}")
|
|
347
|
+
stored = @store.get_run(run.id)
|
|
348
|
+
return record_over(definition, stored, run, evaluate) if stored
|
|
349
|
+
|
|
350
|
+
begin
|
|
351
|
+
@store.insert_run(run)
|
|
352
|
+
rescue StandardError
|
|
353
|
+
# Another process recorded it first.
|
|
354
|
+
again = begin
|
|
355
|
+
@store.get_run(run.id)
|
|
356
|
+
rescue StandardError
|
|
357
|
+
nil
|
|
256
358
|
end
|
|
257
|
-
|
|
359
|
+
return record_over(definition, again, run, evaluate) if again
|
|
360
|
+
|
|
361
|
+
raise
|
|
258
362
|
end
|
|
363
|
+
return [] unless evaluate
|
|
259
364
|
|
|
260
|
-
|
|
365
|
+
update_state(run.job) { |before| [Evaluate.on_run_start(before), nil] }
|
|
366
|
+
return [] if run.status == :running
|
|
261
367
|
|
|
262
|
-
|
|
368
|
+
finish_run(definition, run, now)
|
|
369
|
+
end
|
|
370
|
+
|
|
371
|
+
# Hands an error to on_error, as a source reports what went wrong. See #report.
|
|
372
|
+
def on_error(error, where)
|
|
373
|
+
report(error, where)
|
|
263
374
|
end
|
|
264
375
|
|
|
265
376
|
# Look for missed and stuck runs across every job, send alerts, retry
|
|
@@ -433,6 +544,8 @@ module Cronwatch
|
|
|
433
544
|
@ticker_ms = nil
|
|
434
545
|
@ready_lock = Mutex.new
|
|
435
546
|
@sending_lock = Mutex.new
|
|
547
|
+
# start_run calls with an id still in flight: id => [Mutex, callers].
|
|
548
|
+
@starting = {}
|
|
436
549
|
# Channel (by index) and triage threads that timed out and are still going.
|
|
437
550
|
@abandoned = {}
|
|
438
551
|
end
|
|
@@ -614,21 +727,295 @@ module Cronwatch
|
|
|
614
727
|
"#{alert.type}|#{alert.at}|#{alert.run&.id}"
|
|
615
728
|
end
|
|
616
729
|
|
|
617
|
-
#
|
|
618
|
-
def
|
|
619
|
-
|
|
730
|
+
# record_run for a run already stored.
|
|
731
|
+
def record_over(definition, stored, run, evaluate)
|
|
732
|
+
if stored.job != run.job
|
|
733
|
+
report(RuntimeError.new("run #{run.id} of #{run.job} belongs to job \"#{stored.job}\"; ignored"), "recording #{run.job}")
|
|
734
|
+
return []
|
|
735
|
+
end
|
|
736
|
+
return [] if !%i[running timeout].include?(stored.status) || run.status == :running
|
|
737
|
+
|
|
738
|
+
late, ignored = claim_finish(run)
|
|
739
|
+
if ignored
|
|
740
|
+
report(RuntimeError.new("run #{run.id} of #{run.job} #{ignored}; ignored"), "recording #{run.job}")
|
|
741
|
+
return []
|
|
742
|
+
end
|
|
743
|
+
return [] if !evaluate || (late && run.status != :ok)
|
|
744
|
+
|
|
745
|
+
finish_run(definition, run, now)
|
|
746
|
+
end
|
|
747
|
+
|
|
748
|
+
# A conditional write (the store's update_run_if), or for a store
|
|
749
|
+
# without one, a read then a plain write, which is safe only while one
|
|
750
|
+
# process at a time finishes a given run.
|
|
751
|
+
def write_run_if(run, from_statuses)
|
|
752
|
+
return @store.update_run_if(run, from_statuses) if @store.respond_to?(:update_run_if)
|
|
753
|
+
|
|
754
|
+
stored = @store.get_run(run.id)
|
|
755
|
+
return false if stored.nil? || !from_statuses.include?(stored.status)
|
|
756
|
+
|
|
757
|
+
@store.update_run(run)
|
|
758
|
+
true
|
|
759
|
+
end
|
|
760
|
+
|
|
761
|
+
# Writes a finished run over its stored row, only while that row is
|
|
762
|
+
# still running, or else still marked timeout by a check. Only the
|
|
763
|
+
# process whose write lands goes on to evaluate the run. Returns
|
|
764
|
+
# `[late_after_timeout, nil]` once written, or `[nil, why]` when nothing
|
|
765
|
+
# was: late_after_timeout means a check already counted the run as a
|
|
766
|
+
# stuck failure, so a late failure must not count twice while a late
|
|
767
|
+
# success still closes stuck and recovers. Raises when the store does.
|
|
768
|
+
def claim_finish(run)
|
|
769
|
+
return [false, nil] if write_run_if(run, [:running])
|
|
770
|
+
return [true, nil] if write_run_if(run, [:timeout])
|
|
771
|
+
|
|
772
|
+
stored = @store.get_run(run.id)
|
|
773
|
+
[nil, stored ? "was already finished as #{stored.status}" : "was not found"]
|
|
774
|
+
end
|
|
775
|
+
|
|
776
|
+
# Sets a finished run's status and error from how it ended, then redacts
|
|
777
|
+
# its output and error. Shared by execute and RunHandle#finish.
|
|
778
|
+
def conclude(definition, run, result, error, threw, expect_text, failure: nil)
|
|
779
|
+
if threw
|
|
780
|
+
run.status = :failed
|
|
781
|
+
run.error = Output.error_message(error)
|
|
782
|
+
elsif (problem = failure&.call(result))
|
|
783
|
+
run.status = :failed
|
|
784
|
+
run.error = Output.utf8(problem.to_s)
|
|
785
|
+
else
|
|
786
|
+
unmet = Serialize.check_expectation(definition.expect, expect_text)
|
|
787
|
+
if unmet
|
|
788
|
+
run.status = :failed
|
|
789
|
+
run.error = unmet
|
|
790
|
+
else
|
|
791
|
+
run.status = :ok
|
|
792
|
+
end
|
|
793
|
+
end
|
|
794
|
+
# Redacted after the expect check, so a rule can still match what was
|
|
795
|
+
# logged. NULs go last, so not even a custom redact can store one.
|
|
796
|
+
run.output = Output.strip_nul(@redact.call(run.output)) unless run.output.nil?
|
|
797
|
+
run.error = Output.strip_nul(@redact.call(run.error)) unless run.error.nil?
|
|
798
|
+
end
|
|
799
|
+
|
|
800
|
+
# Writes a finished run and evaluates it. `recorded` says whether its
|
|
801
|
+
# start was written; if not, it is inserted now. Returns why nothing was
|
|
802
|
+
# recorded (another process finished the run first, say), or nil. Raises
|
|
803
|
+
# when the store does, so a handle can be finished again. Shared by
|
|
804
|
+
# execute and RunHandle#finish.
|
|
805
|
+
def record_finish(definition, run, recorded, finished_at)
|
|
806
|
+
unless recorded
|
|
807
|
+
# The start was never written; the store may be back by now.
|
|
808
|
+
sync(definition)
|
|
809
|
+
begin
|
|
810
|
+
@store.insert_run(run)
|
|
811
|
+
finish_run(Serialize.to_stored(definition), run, finished_at)
|
|
812
|
+
return nil
|
|
813
|
+
rescue StandardError => e
|
|
814
|
+
# Another process may have recorded a run with this id meanwhile.
|
|
815
|
+
stored = begin
|
|
816
|
+
@store.get_run(run.id)
|
|
817
|
+
rescue StandardError
|
|
818
|
+
nil
|
|
819
|
+
end
|
|
820
|
+
raise e unless stored
|
|
821
|
+
return "belongs to job \"#{stored.job}\"" if stored.job != run.job
|
|
822
|
+
end
|
|
823
|
+
end
|
|
824
|
+
late, ignored = claim_finish(run)
|
|
825
|
+
return ignored if ignored
|
|
826
|
+
|
|
827
|
+
finish_run(Serialize.to_stored(definition), run, finished_at) if !late || run.status == :ok
|
|
828
|
+
nil
|
|
829
|
+
end
|
|
830
|
+
|
|
831
|
+
# The start of execute without the block: the run is inserted and missed
|
|
832
|
+
# and stuck close (on_run_start). A store that fails is reported and the
|
|
833
|
+
# handle inserts the finished run instead, as execute does.
|
|
834
|
+
def record_start(definition, trigger, id)
|
|
835
|
+
name = definition.name
|
|
836
|
+
unless id.nil?
|
|
837
|
+
stored = nil
|
|
838
|
+
begin
|
|
839
|
+
ensure_ready
|
|
840
|
+
stored = @store.get_run(id)
|
|
841
|
+
rescue StandardError => e
|
|
842
|
+
report(e, "recording #{name}")
|
|
843
|
+
end
|
|
844
|
+
return existing_handle(definition, stored) if stored
|
|
845
|
+
end
|
|
846
|
+
run = Run.new(id: id || SecureRandom.uuid, job: name, status: :running, started_at: now, finished_at: nil,
|
|
847
|
+
duration_ms: nil, error: nil, output: nil, metrics: {}, trigger: trigger)
|
|
848
|
+
recorded = false
|
|
849
|
+
begin
|
|
850
|
+
sync(definition)
|
|
851
|
+
@store.insert_run(run.dup)
|
|
852
|
+
recorded = true
|
|
853
|
+
rescue StandardError => e
|
|
854
|
+
# Another process may have started a run with this id first.
|
|
855
|
+
stored = begin
|
|
856
|
+
id.nil? ? nil : @store.get_run(id)
|
|
857
|
+
rescue StandardError
|
|
858
|
+
nil
|
|
859
|
+
end
|
|
860
|
+
return existing_handle(definition, stored) if stored
|
|
861
|
+
|
|
862
|
+
report(e, "recording #{name}")
|
|
863
|
+
end
|
|
864
|
+
if recorded
|
|
865
|
+
begin
|
|
866
|
+
update_state(name) { |before| [Evaluate.on_run_start(before), nil] }
|
|
867
|
+
rescue StandardError => e
|
|
868
|
+
report(e, "starting #{name}")
|
|
869
|
+
end
|
|
870
|
+
end
|
|
871
|
+
run_handle(definition, run.id, run, recorded, nil, started: true)
|
|
872
|
+
end
|
|
873
|
+
|
|
874
|
+
# A handle on a stored run. One still running, or marked timeout by a check, can be finished.
|
|
875
|
+
def existing_handle(definition, stored)
|
|
876
|
+
if stored.job != definition.name
|
|
877
|
+
raise ArgumentError, "run \"#{stored.id}\" belongs to job \"#{stored.job}\", not \"#{definition.name}\""
|
|
878
|
+
end
|
|
879
|
+
|
|
880
|
+
finished = %i[ok failed].include?(stored.status)
|
|
881
|
+
run_handle(definition, stored.id, stored, true, finished ? "already finished as #{stored.status}" : nil)
|
|
882
|
+
end
|
|
620
883
|
|
|
621
|
-
|
|
622
|
-
|
|
884
|
+
# The handle itself. `base` is the run as last known here, `recorded`
|
|
885
|
+
# whether its start is in the store, and `inactive` why finish has
|
|
886
|
+
# nothing to do, or nil. The handle keeps its lines and metrics until
|
|
887
|
+
# flush or finish merges them onto a fresh read of the stored run.
|
|
888
|
+
# `started` is true for a handle whose run this process inserted.
|
|
889
|
+
def run_handle(definition, id, base, recorded, inactive, started: false)
|
|
890
|
+
name = definition.name
|
|
891
|
+
RunHandle.new(
|
|
892
|
+
id: id, job: name, started_at: base&.started_at, inactive: inactive,
|
|
893
|
+
finish: ->(recorder, outcome, head) { finish_handle(definition, id, base, recorded, recorder, outcome, head, started) },
|
|
894
|
+
flush: recorded ? ->(lines, metrics) { flush_handle(name, id, lines, metrics) } : nil,
|
|
895
|
+
ignored: ->(why) { ignore_finish(id, name, why) },
|
|
896
|
+
)
|
|
897
|
+
end
|
|
898
|
+
|
|
899
|
+
# A finish that records nothing, reported rather than raised.
|
|
900
|
+
def ignore_finish(id, name, why)
|
|
901
|
+
report(RuntimeError.new("run #{id} of #{name} #{why}; ignored"), "finishing #{name}")
|
|
902
|
+
nil
|
|
903
|
+
end
|
|
904
|
+
|
|
905
|
+
# RunHandle#finish, in turn with the handle's flushes: the stored run,
|
|
906
|
+
# read again, with the handle's lines and metrics added, judged like any
|
|
907
|
+
# run. `head` is the start of what the handle flushed, for expect.
|
|
908
|
+
# Returns the run as recorded, or nil when nothing was. A store that
|
|
909
|
+
# fails is reported and raises RunHandle::Retry, which leaves the handle
|
|
910
|
+
# active to be finished again.
|
|
911
|
+
def finish_handle(definition, id, base, recorded, recorder, outcome, head = nil, started = false)
|
|
912
|
+
name = definition.name
|
|
913
|
+
from = base
|
|
914
|
+
if recorded
|
|
915
|
+
begin
|
|
916
|
+
stored = @store.get_run(id)
|
|
917
|
+
if stored
|
|
918
|
+
from = stored
|
|
919
|
+
elsif started
|
|
920
|
+
# Inserted by this process, yet gone: the start was written
|
|
921
|
+
# inside a transaction that rolled back (on SQLite the store
|
|
922
|
+
# joins the app's). Insert it now, as execute does for a start
|
|
923
|
+
# it could not record.
|
|
924
|
+
recorded = false
|
|
925
|
+
end
|
|
926
|
+
rescue StandardError => e
|
|
927
|
+
retry_finish(e, name)
|
|
928
|
+
end
|
|
929
|
+
end
|
|
930
|
+
return ignore_finish(id, name, "was not found") if from.nil?
|
|
931
|
+
return ignore_finish(id, name, "belongs to job \"#{from.job}\"") if from.job != name
|
|
932
|
+
return ignore_finish(id, name, "was already finished as #{from.status}") if %i[ok failed].include?(from.status)
|
|
933
|
+
|
|
934
|
+
failed, result, error = RunHandle.read_outcome(outcome)
|
|
935
|
+
finished_at = now
|
|
936
|
+
returned = result.is_a?(String) ? Output.utf8(result) : nil
|
|
937
|
+
added = recorder.output || (returned && Output.cap(returned))
|
|
938
|
+
run = from.dup
|
|
939
|
+
run.status = :running
|
|
940
|
+
run.finished_at = finished_at
|
|
941
|
+
run.duration_ms = [0, finished_at - from.started_at].max
|
|
942
|
+
run.error = nil
|
|
943
|
+
run.output = join_output(from.output, added)
|
|
944
|
+
run.metrics = (from.metrics || {}).merge(recorder.metrics)
|
|
945
|
+
expect_text = join_lines(head, join_lines(from.output, recorder.expect_text || returned))
|
|
946
|
+
conclude(definition, run, result, error, failed, expect_text)
|
|
947
|
+
why = begin
|
|
948
|
+
record_finish(definition, run, recorded, finished_at)
|
|
949
|
+
rescue StandardError => e
|
|
950
|
+
retry_finish(e, name)
|
|
951
|
+
end
|
|
952
|
+
return ignore_finish(id, name, why) if why
|
|
953
|
+
|
|
954
|
+
run
|
|
955
|
+
end
|
|
956
|
+
|
|
957
|
+
# The store failed part way through a finish and nothing was recorded:
|
|
958
|
+
# reported, and the handle left active so finish can be called again.
|
|
959
|
+
def retry_finish(error, name)
|
|
960
|
+
report(error, "finishing #{name}")
|
|
961
|
+
raise RunHandle::Retry
|
|
962
|
+
end
|
|
963
|
+
|
|
964
|
+
# RunHandle#flush: appends lines and metrics to the stored run while it
|
|
965
|
+
# is still running and belongs to this job, written only over a row still
|
|
966
|
+
# running, so a flush never undoes a finish. True once written; false
|
|
967
|
+
# when the handle should keep them for finish (the run is not running or
|
|
968
|
+
# is another job's, or the store failed).
|
|
969
|
+
def flush_handle(name, id, lines, metrics)
|
|
970
|
+
stored = @store.get_run(id)
|
|
971
|
+
# Not running: the lines stay in the handle for finish, which reports why it cannot record them.
|
|
972
|
+
return false if stored.nil? || stored.status != :running
|
|
973
|
+
|
|
974
|
+
if stored.job != name
|
|
975
|
+
report(RuntimeError.new("run #{id} of #{name} belongs to job \"#{stored.job}\"; ignored"), "flushing #{name}")
|
|
976
|
+
return false
|
|
977
|
+
end
|
|
978
|
+
|
|
979
|
+
updated = stored.dup
|
|
980
|
+
updated.output = join_output(stored.output, Output.strip_nul(@redact.call(lines))) unless lines.nil?
|
|
981
|
+
updated.metrics = (stored.metrics || {}).merge(metrics)
|
|
982
|
+
write_run_if(updated, [:running])
|
|
983
|
+
rescue StandardError => e
|
|
984
|
+
report(e, "flushing #{name}")
|
|
623
985
|
false
|
|
624
986
|
end
|
|
625
987
|
|
|
626
|
-
#
|
|
627
|
-
|
|
628
|
-
|
|
988
|
+
# Raises for a run id no store could hold, or one reserved for the pg_cron source.
|
|
989
|
+
def check_run_id(job, id, method)
|
|
990
|
+
unless id.is_a?(String) && !id.empty? && JS.length16(id) <= 200
|
|
991
|
+
got = id.is_a?(String) ? "#{JS.length16(id)} characters" : id.class.to_s
|
|
992
|
+
raise ArgumentError, "job \"#{job}\": #{method}() needs a run id of 1 to 200 characters (got #{got})"
|
|
993
|
+
end
|
|
994
|
+
return unless id.start_with?(RESERVED_RUN_ID_PREFIX)
|
|
995
|
+
|
|
996
|
+
raise ArgumentError, "job \"#{job}\": #{method}() cannot take a run id starting with \"#{RESERVED_RUN_ID_PREFIX}\", " \
|
|
997
|
+
"which the pg_cron source uses for its runs"
|
|
998
|
+
end
|
|
999
|
+
|
|
1000
|
+
# Two stretches of text as one, a line apart; either may be nil.
|
|
1001
|
+
def join_lines(before, after)
|
|
1002
|
+
return after if before.nil? || before.empty?
|
|
1003
|
+
return before if after.nil?
|
|
1004
|
+
|
|
1005
|
+
"#{before}\n#{after}"
|
|
1006
|
+
end
|
|
1007
|
+
|
|
1008
|
+
# Output appended to stored output, capped like any run's.
|
|
1009
|
+
def join_output(before, after)
|
|
1010
|
+
joined = join_lines(before, after)
|
|
1011
|
+
joined && Output.cap(joined)
|
|
1012
|
+
end
|
|
1013
|
+
|
|
1014
|
+
# Evaluate a finished run (ok, failed, or timed out by a check), already
|
|
1015
|
+
# written, against the job's state and send what that produces. Never raises.
|
|
1016
|
+
def finish_run(definition, run, at)
|
|
629
1017
|
drafts = nil
|
|
630
1018
|
begin
|
|
631
|
-
@store.update_run(run) if write
|
|
632
1019
|
past = nil
|
|
633
1020
|
_, drafts = update_state(run.job) do |previous|
|
|
634
1021
|
past ||= history(run)
|
|
@@ -655,9 +1042,16 @@ module Cronwatch
|
|
|
655
1042
|
|
|
656
1043
|
def run_check
|
|
657
1044
|
ensure_ready
|
|
1045
|
+
alerts = []
|
|
1046
|
+
# Sources first, so what they record is evaluated in this check.
|
|
1047
|
+
@sources.each do |source|
|
|
1048
|
+
found = source.sync(self)
|
|
1049
|
+
alerts.concat(Array(found)) if found.is_a?(Array)
|
|
1050
|
+
rescue StandardError => e
|
|
1051
|
+
report(e, "source #{channel_name(source)}")
|
|
1052
|
+
end
|
|
658
1053
|
defined_jobs.each { |definition| sync(definition) }
|
|
659
1054
|
at = now
|
|
660
|
-
alerts = []
|
|
661
1055
|
|
|
662
1056
|
# Runs that never reported back. One that cannot be judged (its job's
|
|
663
1057
|
# stored timeout no longer parses, say) is reported and skipped.
|
|
@@ -671,7 +1065,10 @@ module Cronwatch
|
|
|
671
1065
|
run.finished_at = at
|
|
672
1066
|
run.duration_ms = at - run.started_at
|
|
673
1067
|
run.error = "Still running after #{Duration.format(timeout)}; marked as timed out"
|
|
674
|
-
|
|
1068
|
+
# Only over a row still running: a finish that landed meanwhile wins.
|
|
1069
|
+
next unless write_run_if(run, [:running])
|
|
1070
|
+
|
|
1071
|
+
alerts.concat(finish_run(definition, run, at))
|
|
675
1072
|
rescue StandardError => e
|
|
676
1073
|
report(e, "checking #{run.job}")
|
|
677
1074
|
end
|
|
@@ -850,7 +1247,7 @@ module Cronwatch
|
|
|
850
1247
|
|
|
851
1248
|
Thread.new do
|
|
852
1249
|
Thread.current.report_on_exception = false
|
|
853
|
-
channel
|
|
1250
|
+
send_to(channel, alert)
|
|
854
1251
|
outcomes[i] = true
|
|
855
1252
|
rescue StandardError, ScriptError => e
|
|
856
1253
|
outcomes[i] = e
|
|
@@ -870,6 +1267,30 @@ module Cronwatch
|
|
|
870
1267
|
results.include?(true)
|
|
871
1268
|
end
|
|
872
1269
|
|
|
1270
|
+
# channel.call(alert, context), the context's on_error reporting a problem
|
|
1271
|
+
# that did not stop the alert going out (one of several recipients
|
|
1272
|
+
# refusing it, say). A channel whose call takes only the alert, as custom
|
|
1273
|
+
# channels written before the context did, is called with the alert alone.
|
|
1274
|
+
def send_to(channel, alert)
|
|
1275
|
+
name = channel_name(channel)
|
|
1276
|
+
context = ChannelContext.new(->(error) { report(error, "alert channel #{name}") })
|
|
1277
|
+
if Client.takes_context?(channel)
|
|
1278
|
+
channel.call(alert, context)
|
|
1279
|
+
else
|
|
1280
|
+
channel.call(alert)
|
|
1281
|
+
end
|
|
1282
|
+
end
|
|
1283
|
+
|
|
1284
|
+
# Whether a channel's call (or a Proc or Method itself) accepts a second argument.
|
|
1285
|
+
#
|
|
1286
|
+
# @api private
|
|
1287
|
+
def self.takes_context?(channel)
|
|
1288
|
+
params = channel.is_a?(Proc) || channel.is_a?(Method) ? channel.parameters : channel.method(:call).parameters
|
|
1289
|
+
params.any? { |kind, _| kind == :rest } || params.count { |kind, _| %i[req opt].include?(kind) } >= 2
|
|
1290
|
+
rescue NameError
|
|
1291
|
+
false
|
|
1292
|
+
end
|
|
1293
|
+
|
|
873
1294
|
def channel_name(channel)
|
|
874
1295
|
channel.respond_to?(:name) && channel.name ? channel.name : channel.class.name
|
|
875
1296
|
end
|
data/lib/cronwatch/evaluate.rb
CHANGED
|
@@ -218,12 +218,25 @@ module Cronwatch
|
|
|
218
218
|
|
|
219
219
|
# Called by check. Decides whether the schedule has been missed: the run
|
|
220
220
|
# the schedule wants next (see Schedule.expectation) has not started and its
|
|
221
|
-
# grace has run out. `last_run` is the most recent run of any status.
|
|
221
|
+
# grace has run out. `last_run` is the most recent run of any status. A job
|
|
222
|
+
# with no schedule is never missed, and one whose schedule was removed while
|
|
223
|
+
# missed was open gets a recovered alert (reason :unscheduled) for missed alone.
|
|
222
224
|
def on_check(definition, stored, last_run, state, now)
|
|
223
225
|
next_state = clone_state(state)
|
|
224
226
|
alerts = []
|
|
225
227
|
schedule = definition.schedule
|
|
226
228
|
if schedule.nil? || schedule == ""
|
|
229
|
+
since = next_state.open[:missed]
|
|
230
|
+
unless since.nil?
|
|
231
|
+
# The schedule went away while missed was open (the job was declared
|
|
232
|
+
# again without one, or a source retired it), so nothing is due any
|
|
233
|
+
# more. Missed closes now with a recovery of its own; other open
|
|
234
|
+
# conditions keep their own rules. Missed is taken out of the pending
|
|
235
|
+
# recovery too, so the next successful run does not name it again.
|
|
236
|
+
next_state.open.delete(:missed)
|
|
237
|
+
next_state.pending_recovery = (next_state.pending_recovery || []).reject { |c| c == :missed }
|
|
238
|
+
alerts << AlertDraft.new(type: :recovered, run: last_run, details: { after: [:missed], reason: :unscheduled, since: since })
|
|
239
|
+
end
|
|
227
240
|
return CheckEvaluation.new(state: next_state, alerts: alerts, next_expected_at: nil, due_at: nil)
|
|
228
241
|
end
|
|
229
242
|
|
data/lib/cronwatch/format.rb
CHANGED
|
@@ -74,10 +74,16 @@ module Cronwatch
|
|
|
74
74
|
lines << "Started #{at_time(run.started_at, now)}." if run
|
|
75
75
|
"#{name} went over budget"
|
|
76
76
|
when :recovered
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
77
|
+
if details[:reason]&.to_sym == :unscheduled
|
|
78
|
+
since = details[:since]
|
|
79
|
+
lines << "#{since.nil? ? "" : "Missed since #{at_time(since, now)}. "}It has no schedule now, so nothing is due; the missed alert is closed."
|
|
80
|
+
"#{name} is no longer scheduled"
|
|
81
|
+
else
|
|
82
|
+
after = details[:after].map { |c| c.to_s.sub("_", " ") }.join(", ")
|
|
83
|
+
lines << "A run #{run ? at_time(run.started_at, now) : "just now"} succeeded#{after.empty? ? "" : " after: #{after}"}."
|
|
84
|
+
lines << "Ran #{Duration.format(run.duration_ms)}." if run && !run.duration_ms.nil?
|
|
85
|
+
"#{name} recovered"
|
|
86
|
+
end
|
|
81
87
|
else
|
|
82
88
|
raise ArgumentError, "unknown alert type #{draft.type.inspect}"
|
|
83
89
|
end
|