cronwatch 0.3.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -40,9 +40,24 @@ module Cronwatch
40
40
  # Guards the reset a forked child makes of the parent's locks and threads.
41
41
  FORK_LOCK = Mutex.new
42
42
  SILENCE_OPTIONS = %i[for].freeze
43
+ # Run ids that start with this belong to the pg_cron source (Sources::PgCron).
44
+ RESERVED_RUN_ID_PREFIX = "pgcron:"
43
45
 
44
46
  # What execute returns: the recorded run, and the block's own outcome.
45
47
  ExecuteResult = Struct.new(:run, :result, :error, :threw, keyword_init: true)
48
+ # What a channel's call receives with each alert (the SDK's ChannelContext).
49
+ # on_error(error) reports a problem that did not stop the alert going
50
+ # out, such as one of several recipients refusing it, to the client's on_error.
51
+ class ChannelContext
52
+ def initialize(on_error)
53
+ @on_error = on_error
54
+ end
55
+
56
+ def on_error(error)
57
+ @on_error.call(error)
58
+ nil
59
+ end
60
+ end
46
61
  # What a triage callable receives. Pass the signal to anything that can stop early.
47
62
  TriageContext = Struct.new(:alert, :recent_runs, :signal, keyword_init: true)
48
63
  # A job's summary and its newest runs, as the dashboard shows them.
@@ -52,7 +67,7 @@ module Cronwatch
52
67
  def to_h = { "job" => job.to_h, "runs" => runs.map(&:to_h) }
53
68
  end
54
69
 
55
- attr_reader :store, :alerts, :triage, :cron_secret, :retention_ms, :defaults
70
+ attr_reader :store, :alerts, :triage, :cron_secret, :retention_ms, :defaults, :sources
56
71
 
57
72
  # store: where jobs, runs and state live. Defaults to an in-memory store that forgets on restart.
58
73
  # alerts: where alerts go: objects with #call(alert) and #name. Defaults to the console.
@@ -70,11 +85,18 @@ module Cronwatch
70
85
  # now: the clock, a callable returning epoch milliseconds. Tests use this.
71
86
  # on_error: called with (error, where) for anything that goes wrong outside a job: the store failing,
72
87
  # an alert channel failing, a triage timeout.
88
+ # sources: where runs this process does not wrap come from, such as pg_cron jobs
89
+ # (Cronwatch::Sources::PgCron). Each is synced at the start of every check; one that
90
+ # raises is reported to on_error and the check carries on. See "Sources" in DESIGN.md.
73
91
  def initialize(store: nil, alerts: nil, triage: nil, cron_secret: UNSET, retention: "30d", defaults: {}, redact: nil,
74
- deliver: :now, now: nil, on_error: nil)
92
+ deliver: :now, now: nil, on_error: nil, sources: nil)
75
93
  @using_default_store = store.nil?
76
94
  @store = store || Stores::Memory.new
77
95
  @alerts = alerts.nil? ? [Alerts::Console.new] : Array(alerts)
96
+ @sources = sources.nil? ? [] : Array(sources)
97
+ @sources.each do |source|
98
+ raise ArgumentError, "a source must respond to sync(host)" unless source.respond_to?(:sync)
99
+ end
78
100
  @triage = triage
79
101
  secret = cron_secret.equal?(UNSET) ? ENV.fetch("CRON_SECRET", nil) : cron_secret
80
102
  @cron_secret = secret.nil? || secret.to_s.empty? ? nil : secret.to_s
@@ -212,54 +234,143 @@ module Cronwatch
212
234
  returned = result.is_a?(String) ? Output.utf8(result) : nil
213
235
  run.output = recorder.output || (returned && Output.cap(returned))
214
236
 
215
- if threw
216
- run.status = :failed
217
- run.error = Output.error_message(error)
218
- elsif (problem = failure&.call(result))
219
- run.status = :failed
220
- run.error = Output.utf8(problem.to_s)
221
- else
222
- expect_text = recorder.expect_text || returned
223
- unmet = Serialize.check_expectation(definition.expect, expect_text)
237
+ conclude(definition, run, result, error, threw, recorder.expect_text || returned, failure: failure)
238
+ begin
239
+ ignored = record_finish(definition, run, recorded, finished_at)
240
+ report(RuntimeError.new("run #{run.id} of #{name} #{ignored}; ignored"), "finishing #{name}") if ignored
241
+ rescue StandardError => e
242
+ report(e, "recording #{name}")
243
+ end
244
+
245
+ raise error if threw && interrupted?(error)
246
+
247
+ ExecuteResult.new(run: run, result: result, error: error, threw: threw)
248
+ end
249
+
250
+ # A handle on a run started elsewhere, as job(name).resume(run_id). The
251
+ # job must be declared in this process, or it raises ArgumentError.
252
+ def resume_run(name, run_id)
253
+ name = name.to_s if name.is_a?(Symbol)
254
+ definition = @registry.synchronize { @definitions[name] }
255
+ raise ArgumentError, "resume_run: job \"#{name}\" is not declared; call job first" unless definition
256
+
257
+ resume_handle(definition, run_id)
258
+ end
259
+
260
+ # JobHandle#start: records a running run and returns a RunHandle to
261
+ # finish it. Two starts with one id at once in this process record one
262
+ # run: the second waits for the first, then finds its run.
263
+ #
264
+ # @api private For JobHandle#start, not apps.
265
+ def start_run(definition, trigger: nil, id: nil)
266
+ after_fork_check
267
+ trigger = "start" if trigger.nil?
268
+ return record_start(definition, trigger, nil) if id.nil?
269
+
270
+ check_run_id(definition.name, id, "start")
271
+ # Keyed by job as well, so another job's start with the same id is not
272
+ # handed this job's run: it fails as it would one call later.
273
+ key = "#{definition.name}\n#{id}"
274
+ entry = @registry.synchronize do
275
+ slot = (@starting[key] ||= [Mutex.new, 0])
276
+ slot[1] += 1
277
+ slot
278
+ end
279
+ begin
280
+ entry[0].synchronize { record_start(definition, trigger, id) }
281
+ ensure
282
+ @registry.synchronize do
283
+ entry[1] -= 1
284
+ @starting.delete(key) if entry[1].zero? && @starting[key].equal?(entry)
285
+ end
286
+ end
287
+ end
288
+
289
+ # JobHandle#resume and resume_run. A store that cannot be read is
290
+ # reported, and the handle's finish reads it again.
291
+ #
292
+ # @api private For JobHandle#resume, not apps.
293
+ def resume_handle(definition, run_id)
294
+ after_fork_check
295
+ check_run_id(definition.name, run_id, "resume")
296
+ begin
297
+ ensure_ready
298
+ stored = @store.get_run(run_id)
299
+ rescue StandardError => e
300
+ report(e, "resuming #{definition.name}")
301
+ return run_handle(definition, run_id, nil, true, nil)
302
+ end
303
+ return run_handle(definition, run_id, nil, true, "was not found") unless stored
304
+
305
+ existing_handle(definition, stored)
306
+ end
307
+
308
+ # Record a run that happened outside this process, for a source. Its job
309
+ # must be declared with job first. Runs are keyed by id: a new one is
310
+ # inserted, a stored one still running (or marked timeout by a check) is
311
+ # finished when this one is not running, and anything else is left alone,
312
+ # so recording the same run twice changes nothing. Finishing is
313
+ # conditional (the store's update_run_if): when two processes record the
314
+ # same finish, only the one whose write lands evaluates it, and the other
315
+ # reports it as already finished. A stored run of another job is left
316
+ # alone and reported. A finished run is judged as if it had been wrapped
317
+ # here (expect, failures, duration, budgets) and its output and error are
318
+ # redacted the same way; one finishing after a check marked it timeout is
319
+ # judged only when it succeeded, as RunHandle#finish does. `evaluate:
320
+ # false` stores it without judging it, for history imported on first
321
+ # sight. Returns the alerts it sent.
322
+ #
323
+ # `run` is a Cronwatch::Run, or a hash of its fields (camelCase or snake_case keys).
324
+ def record_run(run, evaluate: true)
325
+ after_fork_check
326
+ input = run.is_a?(Run) ? run : Run.from_h(run.is_a?(Hash) ? run.to_h { |k, v| [Naming.camel(k), v] } : run)
327
+ declared = @registry.synchronize { @definitions[input.job] }
328
+ raise ArgumentError, "record_run: job \"#{input.job}\" is not declared; call job first" unless declared
329
+
330
+ sync(declared)
331
+ run = input.dup
332
+ run.status = run.status&.to_sym
333
+ run.metrics = (run.metrics || {}).transform_keys(&:to_s)
334
+ run.output = Output.utf8(run.output.to_s) unless run.output.nil?
335
+ run.error = Output.utf8(run.error.to_s) unless run.error.nil?
336
+ if run.status == :ok
337
+ unmet = Serialize.check_expectation(declared.expect, run.output)
224
338
  if unmet
225
339
  run.status = :failed
226
340
  run.error = unmet
227
- else
228
- run.status = :ok
229
341
  end
230
342
  end
231
- # Redacted after the expect check, so a rule can still match what was
232
- # logged. NULs go last, so not even a custom redact can store one.
233
- run.output = Output.strip_nul(@redact.call(run.output)) unless run.output.nil?
234
- run.error = Output.strip_nul(@redact.call(run.error)) unless run.error.nil?
343
+ run.output = Output.strip_nul(@redact.call(Output.cap(run.output))) unless run.output.nil?
344
+ run.error = Output.strip_nul(@redact.call(Output.cap(run.error))) unless run.error.nil?
345
+ definition = Serialize.to_stored(declared)
235
346
 
236
- stored = Serialize.to_stored(definition)
237
- if recorded && marked_timed_out?(run)
238
- # A check gave up on this run while it was going and already counted
239
- # it as a stuck failure. A late failure must not count twice; a late
240
- # success still closes stuck and recovers.
241
- begin
242
- @store.update_run(run)
243
- rescue StandardError => e
244
- report(e, "recording #{name}")
245
- end
246
- elsif recorded
247
- finish_run(stored, run, finished_at, true)
248
- else
249
- # The start was never written; the store may be back by now.
250
- begin
251
- sync(definition)
252
- @store.insert_run(run)
253
- recorded = true
254
- rescue StandardError => e
255
- report(e, "recording #{name}")
347
+ stored = @store.get_run(run.id)
348
+ return record_over(definition, stored, run, evaluate) if stored
349
+
350
+ begin
351
+ @store.insert_run(run)
352
+ rescue StandardError
353
+ # Another process recorded it first.
354
+ again = begin
355
+ @store.get_run(run.id)
356
+ rescue StandardError
357
+ nil
256
358
  end
257
- finish_run(stored, run, finished_at, false) if recorded
359
+ return record_over(definition, again, run, evaluate) if again
360
+
361
+ raise
258
362
  end
363
+ return [] unless evaluate
259
364
 
260
- raise error if threw && interrupted?(error)
365
+ update_state(run.job) { |before| [Evaluate.on_run_start(before), nil] }
366
+ return [] if run.status == :running
261
367
 
262
- ExecuteResult.new(run: run, result: result, error: error, threw: threw)
368
+ finish_run(definition, run, now)
369
+ end
370
+
371
+ # Hands an error to on_error, as a source reports what went wrong. See #report.
372
+ def on_error(error, where)
373
+ report(error, where)
263
374
  end
264
375
 
265
376
  # Look for missed and stuck runs across every job, send alerts, retry
@@ -433,6 +544,8 @@ module Cronwatch
433
544
  @ticker_ms = nil
434
545
  @ready_lock = Mutex.new
435
546
  @sending_lock = Mutex.new
547
+ # start_run calls with an id still in flight: id => [Mutex, callers].
548
+ @starting = {}
436
549
  # Channel (by index) and triage threads that timed out and are still going.
437
550
  @abandoned = {}
438
551
  end
@@ -614,21 +727,295 @@ module Cronwatch
614
727
  "#{alert.type}|#{alert.at}|#{alert.run&.id}"
615
728
  end
616
729
 
617
- # Whether a check already marked this run as timed out, for a failure that finished late.
618
- def marked_timed_out?(run)
619
- return false if run.status == :ok
730
+ # record_run for a run already stored.
731
+ def record_over(definition, stored, run, evaluate)
732
+ if stored.job != run.job
733
+ report(RuntimeError.new("run #{run.id} of #{run.job} belongs to job \"#{stored.job}\"; ignored"), "recording #{run.job}")
734
+ return []
735
+ end
736
+ return [] if !%i[running timeout].include?(stored.status) || run.status == :running
737
+
738
+ late, ignored = claim_finish(run)
739
+ if ignored
740
+ report(RuntimeError.new("run #{run.id} of #{run.job} #{ignored}; ignored"), "recording #{run.job}")
741
+ return []
742
+ end
743
+ return [] if !evaluate || (late && run.status != :ok)
744
+
745
+ finish_run(definition, run, now)
746
+ end
747
+
748
+ # A conditional write (the store's update_run_if), or for a store
749
+ # without one, a read then a plain write, which is safe only while one
750
+ # process at a time finishes a given run.
751
+ def write_run_if(run, from_statuses)
752
+ return @store.update_run_if(run, from_statuses) if @store.respond_to?(:update_run_if)
753
+
754
+ stored = @store.get_run(run.id)
755
+ return false if stored.nil? || !from_statuses.include?(stored.status)
756
+
757
+ @store.update_run(run)
758
+ true
759
+ end
760
+
761
+ # Writes a finished run over its stored row, only while that row is
762
+ # still running, or else still marked timeout by a check. Only the
763
+ # process whose write lands goes on to evaluate the run. Returns
764
+ # `[late_after_timeout, nil]` once written, or `[nil, why]` when nothing
765
+ # was: late_after_timeout means a check already counted the run as a
766
+ # stuck failure, so a late failure must not count twice while a late
767
+ # success still closes stuck and recovers. Raises when the store does.
768
+ def claim_finish(run)
769
+ return [false, nil] if write_run_if(run, [:running])
770
+ return [true, nil] if write_run_if(run, [:timeout])
771
+
772
+ stored = @store.get_run(run.id)
773
+ [nil, stored ? "was already finished as #{stored.status}" : "was not found"]
774
+ end
775
+
776
+ # Sets a finished run's status and error from how it ended, then redacts
777
+ # its output and error. Shared by execute and RunHandle#finish.
778
+ def conclude(definition, run, result, error, threw, expect_text, failure: nil)
779
+ if threw
780
+ run.status = :failed
781
+ run.error = Output.error_message(error)
782
+ elsif (problem = failure&.call(result))
783
+ run.status = :failed
784
+ run.error = Output.utf8(problem.to_s)
785
+ else
786
+ unmet = Serialize.check_expectation(definition.expect, expect_text)
787
+ if unmet
788
+ run.status = :failed
789
+ run.error = unmet
790
+ else
791
+ run.status = :ok
792
+ end
793
+ end
794
+ # Redacted after the expect check, so a rule can still match what was
795
+ # logged. NULs go last, so not even a custom redact can store one.
796
+ run.output = Output.strip_nul(@redact.call(run.output)) unless run.output.nil?
797
+ run.error = Output.strip_nul(@redact.call(run.error)) unless run.error.nil?
798
+ end
799
+
800
+ # Writes a finished run and evaluates it. `recorded` says whether its
801
+ # start was written; if not, it is inserted now. Returns why nothing was
802
+ # recorded (another process finished the run first, say), or nil. Raises
803
+ # when the store does, so a handle can be finished again. Shared by
804
+ # execute and RunHandle#finish.
805
+ def record_finish(definition, run, recorded, finished_at)
806
+ unless recorded
807
+ # The start was never written; the store may be back by now.
808
+ sync(definition)
809
+ begin
810
+ @store.insert_run(run)
811
+ finish_run(Serialize.to_stored(definition), run, finished_at)
812
+ return nil
813
+ rescue StandardError => e
814
+ # Another process may have recorded a run with this id meanwhile.
815
+ stored = begin
816
+ @store.get_run(run.id)
817
+ rescue StandardError
818
+ nil
819
+ end
820
+ raise e unless stored
821
+ return "belongs to job \"#{stored.job}\"" if stored.job != run.job
822
+ end
823
+ end
824
+ late, ignored = claim_finish(run)
825
+ return ignored if ignored
826
+
827
+ finish_run(Serialize.to_stored(definition), run, finished_at) if !late || run.status == :ok
828
+ nil
829
+ end
830
+
831
+ # The start of execute without the block: the run is inserted and missed
832
+ # and stuck close (on_run_start). A store that fails is reported and the
833
+ # handle inserts the finished run instead, as execute does.
834
+ def record_start(definition, trigger, id)
835
+ name = definition.name
836
+ unless id.nil?
837
+ stored = nil
838
+ begin
839
+ ensure_ready
840
+ stored = @store.get_run(id)
841
+ rescue StandardError => e
842
+ report(e, "recording #{name}")
843
+ end
844
+ return existing_handle(definition, stored) if stored
845
+ end
846
+ run = Run.new(id: id || SecureRandom.uuid, job: name, status: :running, started_at: now, finished_at: nil,
847
+ duration_ms: nil, error: nil, output: nil, metrics: {}, trigger: trigger)
848
+ recorded = false
849
+ begin
850
+ sync(definition)
851
+ @store.insert_run(run.dup)
852
+ recorded = true
853
+ rescue StandardError => e
854
+ # Another process may have started a run with this id first.
855
+ stored = begin
856
+ id.nil? ? nil : @store.get_run(id)
857
+ rescue StandardError
858
+ nil
859
+ end
860
+ return existing_handle(definition, stored) if stored
861
+
862
+ report(e, "recording #{name}")
863
+ end
864
+ if recorded
865
+ begin
866
+ update_state(name) { |before| [Evaluate.on_run_start(before), nil] }
867
+ rescue StandardError => e
868
+ report(e, "starting #{name}")
869
+ end
870
+ end
871
+ run_handle(definition, run.id, run, recorded, nil, started: true)
872
+ end
873
+
874
+ # A handle on a stored run. One still running, or marked timeout by a check, can be finished.
875
+ def existing_handle(definition, stored)
876
+ if stored.job != definition.name
877
+ raise ArgumentError, "run \"#{stored.id}\" belongs to job \"#{stored.job}\", not \"#{definition.name}\""
878
+ end
879
+
880
+ finished = %i[ok failed].include?(stored.status)
881
+ run_handle(definition, stored.id, stored, true, finished ? "already finished as #{stored.status}" : nil)
882
+ end
620
883
 
621
- @store.get_run(run.id)&.status == :timeout
622
- rescue StandardError
884
+ # The handle itself. `base` is the run as last known here, `recorded`
885
+ # whether its start is in the store, and `inactive` why finish has
886
+ # nothing to do, or nil. The handle keeps its lines and metrics until
887
+ # flush or finish merges them onto a fresh read of the stored run.
888
+ # `started` is true for a handle whose run this process inserted.
889
+ def run_handle(definition, id, base, recorded, inactive, started: false)
890
+ name = definition.name
891
+ RunHandle.new(
892
+ id: id, job: name, started_at: base&.started_at, inactive: inactive,
893
+ finish: ->(recorder, outcome, head) { finish_handle(definition, id, base, recorded, recorder, outcome, head, started) },
894
+ flush: recorded ? ->(lines, metrics) { flush_handle(name, id, lines, metrics) } : nil,
895
+ ignored: ->(why) { ignore_finish(id, name, why) },
896
+ )
897
+ end
898
+
899
+ # A finish that records nothing, reported rather than raised.
900
+ def ignore_finish(id, name, why)
901
+ report(RuntimeError.new("run #{id} of #{name} #{why}; ignored"), "finishing #{name}")
902
+ nil
903
+ end
904
+
905
+ # RunHandle#finish, in turn with the handle's flushes: the stored run,
906
+ # read again, with the handle's lines and metrics added, judged like any
907
+ # run. `head` is the start of what the handle flushed, for expect.
908
+ # Returns the run as recorded, or nil when nothing was. A store that
909
+ # fails is reported and raises RunHandle::Retry, which leaves the handle
910
+ # active to be finished again.
911
+ def finish_handle(definition, id, base, recorded, recorder, outcome, head = nil, started = false)
912
+ name = definition.name
913
+ from = base
914
+ if recorded
915
+ begin
916
+ stored = @store.get_run(id)
917
+ if stored
918
+ from = stored
919
+ elsif started
920
+ # Inserted by this process, yet gone: the start was written
921
+ # inside a transaction that rolled back (on SQLite the store
922
+ # joins the app's). Insert it now, as execute does for a start
923
+ # it could not record.
924
+ recorded = false
925
+ end
926
+ rescue StandardError => e
927
+ retry_finish(e, name)
928
+ end
929
+ end
930
+ return ignore_finish(id, name, "was not found") if from.nil?
931
+ return ignore_finish(id, name, "belongs to job \"#{from.job}\"") if from.job != name
932
+ return ignore_finish(id, name, "was already finished as #{from.status}") if %i[ok failed].include?(from.status)
933
+
934
+ failed, result, error = RunHandle.read_outcome(outcome)
935
+ finished_at = now
936
+ returned = result.is_a?(String) ? Output.utf8(result) : nil
937
+ added = recorder.output || (returned && Output.cap(returned))
938
+ run = from.dup
939
+ run.status = :running
940
+ run.finished_at = finished_at
941
+ run.duration_ms = [0, finished_at - from.started_at].max
942
+ run.error = nil
943
+ run.output = join_output(from.output, added)
944
+ run.metrics = (from.metrics || {}).merge(recorder.metrics)
945
+ expect_text = join_lines(head, join_lines(from.output, recorder.expect_text || returned))
946
+ conclude(definition, run, result, error, failed, expect_text)
947
+ why = begin
948
+ record_finish(definition, run, recorded, finished_at)
949
+ rescue StandardError => e
950
+ retry_finish(e, name)
951
+ end
952
+ return ignore_finish(id, name, why) if why
953
+
954
+ run
955
+ end
956
+
957
+ # The store failed part way through a finish and nothing was recorded:
958
+ # reported, and the handle left active so finish can be called again.
959
+ def retry_finish(error, name)
960
+ report(error, "finishing #{name}")
961
+ raise RunHandle::Retry
962
+ end
963
+
964
+ # RunHandle#flush: appends lines and metrics to the stored run while it
965
+ # is still running and belongs to this job, written only over a row still
966
+ # running, so a flush never undoes a finish. True once written; false
967
+ # when the handle should keep them for finish (the run is not running or
968
+ # is another job's, or the store failed).
969
+ def flush_handle(name, id, lines, metrics)
970
+ stored = @store.get_run(id)
971
+ # Not running: the lines stay in the handle for finish, which reports why it cannot record them.
972
+ return false if stored.nil? || stored.status != :running
973
+
974
+ if stored.job != name
975
+ report(RuntimeError.new("run #{id} of #{name} belongs to job \"#{stored.job}\"; ignored"), "flushing #{name}")
976
+ return false
977
+ end
978
+
979
+ updated = stored.dup
980
+ updated.output = join_output(stored.output, Output.strip_nul(@redact.call(lines))) unless lines.nil?
981
+ updated.metrics = (stored.metrics || {}).merge(metrics)
982
+ write_run_if(updated, [:running])
983
+ rescue StandardError => e
984
+ report(e, "flushing #{name}")
623
985
  false
624
986
  end
625
987
 
626
- # Record a finished run (ok, failed, or timed out by a check), evaluate it
627
- # against the job's state and send what that produces. Never raises.
628
- def finish_run(definition, run, at, write)
988
+ # Raises for a run id no store could hold, or one reserved for the pg_cron source.
989
+ def check_run_id(job, id, method)
990
+ unless id.is_a?(String) && !id.empty? && JS.length16(id) <= 200
991
+ got = id.is_a?(String) ? "#{JS.length16(id)} characters" : id.class.to_s
992
+ raise ArgumentError, "job \"#{job}\": #{method}() needs a run id of 1 to 200 characters (got #{got})"
993
+ end
994
+ return unless id.start_with?(RESERVED_RUN_ID_PREFIX)
995
+
996
+ raise ArgumentError, "job \"#{job}\": #{method}() cannot take a run id starting with \"#{RESERVED_RUN_ID_PREFIX}\", " \
997
+ "which the pg_cron source uses for its runs"
998
+ end
999
+
1000
+ # Two stretches of text as one, a line apart; either may be nil.
1001
+ def join_lines(before, after)
1002
+ return after if before.nil? || before.empty?
1003
+ return before if after.nil?
1004
+
1005
+ "#{before}\n#{after}"
1006
+ end
1007
+
1008
+ # Output appended to stored output, capped like any run's.
1009
+ def join_output(before, after)
1010
+ joined = join_lines(before, after)
1011
+ joined && Output.cap(joined)
1012
+ end
1013
+
1014
+ # Evaluate a finished run (ok, failed, or timed out by a check), already
1015
+ # written, against the job's state and send what that produces. Never raises.
1016
+ def finish_run(definition, run, at)
629
1017
  drafts = nil
630
1018
  begin
631
- @store.update_run(run) if write
632
1019
  past = nil
633
1020
  _, drafts = update_state(run.job) do |previous|
634
1021
  past ||= history(run)
@@ -655,9 +1042,16 @@ module Cronwatch
655
1042
 
656
1043
  def run_check
657
1044
  ensure_ready
1045
+ alerts = []
1046
+ # Sources first, so what they record is evaluated in this check.
1047
+ @sources.each do |source|
1048
+ found = source.sync(self)
1049
+ alerts.concat(Array(found)) if found.is_a?(Array)
1050
+ rescue StandardError => e
1051
+ report(e, "source #{channel_name(source)}")
1052
+ end
658
1053
  defined_jobs.each { |definition| sync(definition) }
659
1054
  at = now
660
- alerts = []
661
1055
 
662
1056
  # Runs that never reported back. One that cannot be judged (its job's
663
1057
  # stored timeout no longer parses, say) is reported and skipped.
@@ -671,7 +1065,10 @@ module Cronwatch
671
1065
  run.finished_at = at
672
1066
  run.duration_ms = at - run.started_at
673
1067
  run.error = "Still running after #{Duration.format(timeout)}; marked as timed out"
674
- alerts.concat(finish_run(definition, run, at, true))
1068
+ # Only over a row still running: a finish that landed meanwhile wins.
1069
+ next unless write_run_if(run, [:running])
1070
+
1071
+ alerts.concat(finish_run(definition, run, at))
675
1072
  rescue StandardError => e
676
1073
  report(e, "checking #{run.job}")
677
1074
  end
@@ -850,7 +1247,7 @@ module Cronwatch
850
1247
 
851
1248
  Thread.new do
852
1249
  Thread.current.report_on_exception = false
853
- channel.call(alert)
1250
+ send_to(channel, alert)
854
1251
  outcomes[i] = true
855
1252
  rescue StandardError, ScriptError => e
856
1253
  outcomes[i] = e
@@ -870,6 +1267,30 @@ module Cronwatch
870
1267
  results.include?(true)
871
1268
  end
872
1269
 
1270
+ # channel.call(alert, context), the context's on_error reporting a problem
1271
+ # that did not stop the alert going out (one of several recipients
1272
+ # refusing it, say). A channel whose call takes only the alert, as custom
1273
+ # channels written before the context did, is called with the alert alone.
1274
+ def send_to(channel, alert)
1275
+ name = channel_name(channel)
1276
+ context = ChannelContext.new(->(error) { report(error, "alert channel #{name}") })
1277
+ if Client.takes_context?(channel)
1278
+ channel.call(alert, context)
1279
+ else
1280
+ channel.call(alert)
1281
+ end
1282
+ end
1283
+
1284
+ # Whether a channel's call (or a Proc or Method itself) accepts a second argument.
1285
+ #
1286
+ # @api private
1287
+ def self.takes_context?(channel)
1288
+ params = channel.is_a?(Proc) || channel.is_a?(Method) ? channel.parameters : channel.method(:call).parameters
1289
+ params.any? { |kind, _| kind == :rest } || params.count { |kind, _| %i[req opt].include?(kind) } >= 2
1290
+ rescue NameError
1291
+ false
1292
+ end
1293
+
873
1294
  def channel_name(channel)
874
1295
  channel.respond_to?(:name) && channel.name ? channel.name : channel.class.name
875
1296
  end
@@ -218,12 +218,25 @@ module Cronwatch
218
218
 
219
219
  # Called by check. Decides whether the schedule has been missed: the run
220
220
  # the schedule wants next (see Schedule.expectation) has not started and its
221
- # grace has run out. `last_run` is the most recent run of any status.
221
+ # grace has run out. `last_run` is the most recent run of any status. A job
222
+ # with no schedule is never missed, and one whose schedule was removed while
223
+ # missed was open gets a recovered alert (reason :unscheduled) for missed alone.
222
224
  def on_check(definition, stored, last_run, state, now)
223
225
  next_state = clone_state(state)
224
226
  alerts = []
225
227
  schedule = definition.schedule
226
228
  if schedule.nil? || schedule == ""
229
+ since = next_state.open[:missed]
230
+ unless since.nil?
231
+ # The schedule went away while missed was open (the job was declared
232
+ # again without one, or a source retired it), so nothing is due any
233
+ # more. Missed closes now with a recovery of its own; other open
234
+ # conditions keep their own rules. Missed is taken out of the pending
235
+ # recovery too, so the next successful run does not name it again.
236
+ next_state.open.delete(:missed)
237
+ next_state.pending_recovery = (next_state.pending_recovery || []).reject { |c| c == :missed }
238
+ alerts << AlertDraft.new(type: :recovered, run: last_run, details: { after: [:missed], reason: :unscheduled, since: since })
239
+ end
227
240
  return CheckEvaluation.new(state: next_state, alerts: alerts, next_expected_at: nil, due_at: nil)
228
241
  end
229
242
 
@@ -74,10 +74,16 @@ module Cronwatch
74
74
  lines << "Started #{at_time(run.started_at, now)}." if run
75
75
  "#{name} went over budget"
76
76
  when :recovered
77
- after = details[:after].map { |c| c.to_s.sub("_", " ") }.join(", ")
78
- lines << "A run #{run ? at_time(run.started_at, now) : "just now"} succeeded#{after.empty? ? "" : " after: #{after}"}."
79
- lines << "Ran #{Duration.format(run.duration_ms)}." if run && !run.duration_ms.nil?
80
- "#{name} recovered"
77
+ if details[:reason]&.to_sym == :unscheduled
78
+ since = details[:since]
79
+ lines << "#{since.nil? ? "" : "Missed since #{at_time(since, now)}. "}It has no schedule now, so nothing is due; the missed alert is closed."
80
+ "#{name} is no longer scheduled"
81
+ else
82
+ after = details[:after].map { |c| c.to_s.sub("_", " ") }.join(", ")
83
+ lines << "A run #{run ? at_time(run.started_at, now) : "just now"} succeeded#{after.empty? ? "" : " after: #{after}"}."
84
+ lines << "Ran #{Duration.format(run.duration_ms)}." if run && !run.duration_ms.nil?
85
+ "#{name} recovered"
86
+ end
81
87
  else
82
88
  raise ArgumentError, "unknown alert type #{draft.type.inspect}"
83
89
  end