cronwatch 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/README.md +268 -0
- data/lib/cronwatch/abort_signal.rb +45 -0
- data/lib/cronwatch/active_record.rb +11 -0
- data/lib/cronwatch/alerts/console.rb +22 -0
- data/lib/cronwatch/alerts/custom.rb +26 -0
- data/lib/cronwatch/alerts/discord.rb +52 -0
- data/lib/cronwatch/alerts/slack.rb +58 -0
- data/lib/cronwatch/alerts/webhook.rb +46 -0
- data/lib/cronwatch/client.rb +925 -0
- data/lib/cronwatch/cron_pattern.rb +277 -0
- data/lib/cronwatch/duration.rb +77 -0
- data/lib/cronwatch/environment.rb +29 -0
- data/lib/cronwatch/evaluate.rb +345 -0
- data/lib/cronwatch/flight.rb +42 -0
- data/lib/cronwatch/format.rb +89 -0
- data/lib/cronwatch/http.rb +52 -0
- data/lib/cronwatch/job.rb +144 -0
- data/lib/cronwatch/js.rb +188 -0
- data/lib/cronwatch/monitored.rb +259 -0
- data/lib/cronwatch/output.rb +199 -0
- data/lib/cronwatch/rails/active_job.rb +55 -0
- data/lib/cronwatch/rails/check_job.rb +32 -0
- data/lib/cronwatch/rails/railtie.rb +37 -0
- data/lib/cronwatch/rails/tasks.rb +12 -0
- data/lib/cronwatch/rails.rb +35 -0
- data/lib/cronwatch/schedule.rb +191 -0
- data/lib/cronwatch/scheduler.rb +763 -0
- data/lib/cronwatch/serialize.rb +51 -0
- data/lib/cronwatch/sidekiq.rb +129 -0
- data/lib/cronwatch/stats.rb +23 -0
- data/lib/cronwatch/stores/active_record.rb +397 -0
- data/lib/cronwatch/stores/memory.rb +163 -0
- data/lib/cronwatch/ticker.rb +59 -0
- data/lib/cronwatch/triage/anthropic.rb +134 -0
- data/lib/cronwatch/types.rb +341 -0
- data/lib/cronwatch/version.rb +6 -0
- data/lib/cronwatch/walker.rb +137 -0
- data/lib/cronwatch/web/app.rb +484 -0
- data/lib/cronwatch/web/html.rb +314 -0
- data/lib/cronwatch/web.rb +17 -0
- data/lib/cronwatch/zone.rb +72 -0
- data/lib/cronwatch.rb +96 -0
- data/lib/generators/cronwatch/install/install_generator.rb +176 -0
- metadata +104 -0
|
@@ -0,0 +1,345 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Cronwatch
|
|
4
|
+
# Pure decisions about a job's health. Each function takes the current state
|
|
5
|
+
# and returns the new state plus the alerts that should go out. Nothing here
|
|
6
|
+
# touches a store or a network, which is what makes it testable.
|
|
7
|
+
module Evaluate
|
|
8
|
+
DEFAULT_GRACE_MS = 10 * 60_000
|
|
9
|
+
DEFAULT_TIMEOUT_MS = 60 * 60_000
|
|
10
|
+
# Runs faster than this are never called slow, whatever the baseline says.
|
|
11
|
+
SLOW_FLOOR_MS = 10_000
|
|
12
|
+
# How many earlier runs a baseline needs before it is trusted.
|
|
13
|
+
BASELINE_MIN_RUNS = 5
|
|
14
|
+
# How many successful runs a baseline looks at, and how many runs a summary covers.
|
|
15
|
+
BASELINE_WINDOW = 20
|
|
16
|
+
|
|
17
|
+
Evaluation = Struct.new(:state, :alerts, keyword_init: true)
|
|
18
|
+
CheckEvaluation = Struct.new(:state, :alerts, :next_expected_at, :due_at, keyword_init: true)
|
|
19
|
+
SlowThreshold = Struct.new(:threshold_ms, :basis, keyword_init: true)
|
|
20
|
+
|
|
21
|
+
module_function
|
|
22
|
+
|
|
23
|
+
def empty_state(job)
|
|
24
|
+
JobState.new(job: job, open: {}, consecutive_failures: 0, silenced_until: nil, last_alert_at: nil,
|
|
25
|
+
pending_recovery: [], undelivered: [])
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
# A stored state with every field present, or a fresh one. State written by an older version lacks the newer fields.
|
|
29
|
+
def normalize_state(state, job)
|
|
30
|
+
return empty_state(job) if state.nil?
|
|
31
|
+
|
|
32
|
+
state = JobState.from_h(state)
|
|
33
|
+
JobState.new(
|
|
34
|
+
job: state.job.nil? ? job : state.job,
|
|
35
|
+
open: (state.open || {}).dup,
|
|
36
|
+
consecutive_failures: state.consecutive_failures.nil? ? 0 : state.consecutive_failures,
|
|
37
|
+
silenced_until: state.silenced_until,
|
|
38
|
+
last_alert_at: state.last_alert_at,
|
|
39
|
+
pending_recovery: (state.pending_recovery || []).dup,
|
|
40
|
+
undelivered: (state.undelivered || []).dup,
|
|
41
|
+
version: state.version,
|
|
42
|
+
)
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
def clone_state(state)
|
|
46
|
+
normalize_state(state, state.job)
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
def open_condition(state, condition, now)
|
|
50
|
+
return false if state.open.key?(condition)
|
|
51
|
+
|
|
52
|
+
state.open[condition] = now
|
|
53
|
+
true
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Every open condition has alerted, so closing one owes a recovered message.
|
|
57
|
+
# It is remembered until a successful run leaves nothing open and sends it.
|
|
58
|
+
def close_condition(state, condition)
|
|
59
|
+
return false unless state.open.key?(condition)
|
|
60
|
+
|
|
61
|
+
state.open.delete(condition)
|
|
62
|
+
state.pending_recovery ||= []
|
|
63
|
+
state.pending_recovery << condition unless state.pending_recovery.include?(condition)
|
|
64
|
+
true
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
def open_conditions(state)
|
|
68
|
+
state.open.keys
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
def grace_ms(definition)
|
|
72
|
+
definition.grace.nil? ? DEFAULT_GRACE_MS : Duration.parse(definition.grace, "grace")
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def timeout_ms(definition)
|
|
76
|
+
definition.timeout.nil? ? DEFAULT_TIMEOUT_MS : Duration.parse(definition.timeout, "timeout")
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
# Slow threshold for a successful run, or nil when there is nothing to compare against yet.
|
|
80
|
+
def slow_threshold(definition, history)
|
|
81
|
+
unless definition.max_duration.nil?
|
|
82
|
+
return SlowThreshold.new(threshold_ms: Duration.parse(definition.max_duration, "maxDuration"), basis: "maxDuration")
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
durations = history.select { |r| r.status == :ok && !r.duration_ms.nil? }.first(BASELINE_WINDOW).map(&:duration_ms)
|
|
86
|
+
return nil if durations.length < BASELINE_MIN_RUNS
|
|
87
|
+
|
|
88
|
+
p95 = Stats.percentile(durations, 95)
|
|
89
|
+
SlowThreshold.new(
|
|
90
|
+
threshold_ms: [2 * p95, SLOW_FLOOR_MS].max,
|
|
91
|
+
basis: "twice the p95 of the last #{durations.length} runs (#{Duration.format(p95)})",
|
|
92
|
+
)
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
def budget_breaches(definition, run, history)
|
|
96
|
+
breaches = []
|
|
97
|
+
budget = definition.budget
|
|
98
|
+
metrics = run.metrics || {}
|
|
99
|
+
JS.object_keys(metrics).each do |metric|
|
|
100
|
+
value = metrics[metric]
|
|
101
|
+
ceiling = budget&.[](metric.to_s)
|
|
102
|
+
unless ceiling.nil?
|
|
103
|
+
breaches << { metric: metric.to_s, value: value, limit: ceiling, basis: "budget" } if value > ceiling
|
|
104
|
+
next
|
|
105
|
+
end
|
|
106
|
+
past = history.select { |r| r.status == :ok && (r.metrics || {})[metric].is_a?(Numeric) }
|
|
107
|
+
.first(BASELINE_WINDOW).map { |r| r.metrics[metric] }
|
|
108
|
+
next if past.length < BASELINE_MIN_RUNS
|
|
109
|
+
|
|
110
|
+
usual = Stats.median(past)
|
|
111
|
+
if usual.positive? && value > 3 * usual
|
|
112
|
+
breaches << { metric: metric.to_s, value: value, limit: 3 * usual, basis: "three times the usual #{format_number(usual)}" }
|
|
113
|
+
end
|
|
114
|
+
end
|
|
115
|
+
breaches
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
# Whether `history` (newest first) holds a full baseline window of successful runs.
|
|
119
|
+
def full_baseline?(history)
|
|
120
|
+
history.count { |r| r.status == :ok } >= BASELINE_WINDOW
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
# toLocaleString("en-US"): digit groups, and a fraction rounded half up to
|
|
124
|
+
# at most four places. Worked on the shortest decimal digits, as ICU does.
|
|
125
|
+
def format_number(n)
|
|
126
|
+
return (n.negative? ? "-" : "") + (n.nan? ? "NaN" : "\u221e") if n.is_a?(Float) && !n.finite?
|
|
127
|
+
|
|
128
|
+
negative = n.negative? || (n.is_a?(Float) && n.zero? && (1.0 / n).negative?)
|
|
129
|
+
if n.is_a?(Integer)
|
|
130
|
+
whole = n.abs.to_s
|
|
131
|
+
fraction = ""
|
|
132
|
+
else
|
|
133
|
+
digits, point = JS.decimal(n.abs.to_f)
|
|
134
|
+
if point >= digits.length
|
|
135
|
+
whole = digits + ("0" * (point - digits.length))
|
|
136
|
+
fraction = ""
|
|
137
|
+
elsif point.positive?
|
|
138
|
+
whole = digits[0, point]
|
|
139
|
+
fraction = digits[point..]
|
|
140
|
+
else
|
|
141
|
+
whole = "0"
|
|
142
|
+
fraction = ("0" * -point) + digits
|
|
143
|
+
end
|
|
144
|
+
if fraction.length > 4
|
|
145
|
+
up = fraction[4].to_i >= 5
|
|
146
|
+
fraction = fraction[0, 4]
|
|
147
|
+
if up
|
|
148
|
+
rounded = ((whole + fraction).to_i + 1).to_s.rjust(whole.length + 4, "0")
|
|
149
|
+
whole = rounded[0...-4]
|
|
150
|
+
fraction = rounded[-4..]
|
|
151
|
+
end
|
|
152
|
+
end
|
|
153
|
+
fraction = fraction.sub(/0+\z/, "")
|
|
154
|
+
end
|
|
155
|
+
whole = whole.reverse.scan(/\d{1,3}/).join(",").reverse
|
|
156
|
+
text = fraction.empty? ? whole : "#{whole}.#{fraction}"
|
|
157
|
+
negative ? "-#{text}" : text
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
# Called when a run starts. Missed and stuck are about the absence of a run,
|
|
161
|
+
# so a run starting closes them without an alert; the recovered message
|
|
162
|
+
# waits for a successful finish.
|
|
163
|
+
def on_run_start(state)
|
|
164
|
+
next_state = clone_state(state)
|
|
165
|
+
close_condition(next_state, :missed)
|
|
166
|
+
close_condition(next_state, :stuck)
|
|
167
|
+
next_state
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
# Called when a run finishes with status ok, failed or timeout. `history` is
|
|
171
|
+
# the job's earlier runs, newest first, not including this one.
|
|
172
|
+
def on_run_finish(definition, run, state, history, now)
|
|
173
|
+
next_state = clone_state(state)
|
|
174
|
+
alerts = []
|
|
175
|
+
|
|
176
|
+
if run.status == :ok
|
|
177
|
+
next_state.consecutive_failures = 0
|
|
178
|
+
close_condition(next_state, :missed)
|
|
179
|
+
close_condition(next_state, :stuck)
|
|
180
|
+
close_condition(next_state, :failed)
|
|
181
|
+
|
|
182
|
+
slow = slow_threshold(definition, history)
|
|
183
|
+
if slow && !run.duration_ms.nil? && run.duration_ms > slow.threshold_ms
|
|
184
|
+
if open_condition(next_state, :slow, now)
|
|
185
|
+
alerts << AlertDraft.new(type: :slow, run: run,
|
|
186
|
+
details: { duration_ms: run.duration_ms, threshold_ms: slow.threshold_ms, basis: slow.basis })
|
|
187
|
+
end
|
|
188
|
+
else
|
|
189
|
+
close_condition(next_state, :slow)
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
breaches = budget_breaches(definition, run, history)
|
|
193
|
+
if breaches.any?
|
|
194
|
+
alerts << AlertDraft.new(type: :over_budget, run: run, details: { breaches: breaches }) if open_condition(next_state, :over_budget, now)
|
|
195
|
+
else
|
|
196
|
+
close_condition(next_state, :over_budget)
|
|
197
|
+
end
|
|
198
|
+
|
|
199
|
+
pending = next_state.pending_recovery || []
|
|
200
|
+
if pending.any? && open_conditions(next_state).empty?
|
|
201
|
+
alerts << AlertDraft.new(type: :recovered, run: run, details: { after: pending.dup })
|
|
202
|
+
next_state.pending_recovery = []
|
|
203
|
+
end
|
|
204
|
+
return Evaluation.new(state: next_state, alerts: alerts)
|
|
205
|
+
end
|
|
206
|
+
|
|
207
|
+
# failed or timeout
|
|
208
|
+
next_state.consecutive_failures += 1
|
|
209
|
+
close_condition(next_state, :missed)
|
|
210
|
+
threshold = [1, definition.failures_before_alert || 1].max
|
|
211
|
+
condition = run.status == :timeout ? :stuck : :failed
|
|
212
|
+
if next_state.consecutive_failures >= threshold && open_condition(next_state, condition, now)
|
|
213
|
+
alerts << AlertDraft.new(type: condition, run: run,
|
|
214
|
+
details: { consecutive_failures: next_state.consecutive_failures, threshold: threshold })
|
|
215
|
+
end
|
|
216
|
+
Evaluation.new(state: next_state, alerts: alerts)
|
|
217
|
+
end
|
|
218
|
+
|
|
219
|
+
# Called by check. Decides whether the schedule has been missed: the run
|
|
220
|
+
# the schedule wants next (see Schedule.expectation) has not started and its
|
|
221
|
+
# grace has run out. `last_run` is the most recent run of any status.
|
|
222
|
+
def on_check(definition, stored, last_run, state, now)
|
|
223
|
+
next_state = clone_state(state)
|
|
224
|
+
alerts = []
|
|
225
|
+
schedule = definition.schedule
|
|
226
|
+
if schedule.nil? || schedule == ""
|
|
227
|
+
return CheckEvaluation.new(state: next_state, alerts: alerts, next_expected_at: nil, due_at: nil)
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
parsed = Schedule.parse(schedule, definition.timezone)
|
|
231
|
+
grace = grace_ms(definition)
|
|
232
|
+
last_run_at = last_run&.started_at
|
|
233
|
+
exp = Schedule.expectation(parsed, last_run_at, stored.created_at, grace)
|
|
234
|
+
next_expected_at =
|
|
235
|
+
if parsed.interval? then Schedule.next_fire(parsed, stored.created_at, last_run_at)
|
|
236
|
+
else Schedule.next_fire(parsed, now, nil)
|
|
237
|
+
end
|
|
238
|
+
return CheckEvaluation.new(state: next_state, alerts: alerts, next_expected_at: next_expected_at, due_at: nil) unless exp
|
|
239
|
+
|
|
240
|
+
# An interval's next run is due a period after the last one started. If that
|
|
241
|
+
# run is still going, the job is busy, not late; stuck covers one that never ends.
|
|
242
|
+
if parsed.interval? && last_run&.status == :running
|
|
243
|
+
return CheckEvaluation.new(state: next_state, alerts: alerts, next_expected_at: next_expected_at, due_at: exp.due_at)
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
if now > exp.deadline
|
|
247
|
+
if open_condition(next_state, :missed, now)
|
|
248
|
+
alerts << AlertDraft.new(type: :missed, run: last_run,
|
|
249
|
+
details: { due_at: exp.due_at, deadline: exp.deadline, grace_ms: grace, last_run_at: last_run_at })
|
|
250
|
+
end
|
|
251
|
+
else
|
|
252
|
+
# A run has started since it opened, or the grace was widened.
|
|
253
|
+
close_condition(next_state, :missed)
|
|
254
|
+
end
|
|
255
|
+
CheckEvaluation.new(state: next_state, alerts: alerts, next_expected_at: next_expected_at, due_at: exp.due_at)
|
|
256
|
+
end
|
|
257
|
+
|
|
258
|
+
# Whether a running run has gone on longer than the job's timeout.
|
|
259
|
+
def stuck?(definition, run, now)
|
|
260
|
+
run.status == :running && now - run.started_at > timeout_ms(definition)
|
|
261
|
+
end
|
|
262
|
+
|
|
263
|
+
# While a job is silenced nothing new is recorded as an incident: conditions
|
|
264
|
+
# may close (so a job that recovered during the silence shows as healthy) but
|
|
265
|
+
# none may open, so the first problem after the silence ends alerts normally.
|
|
266
|
+
def mute_opens(previous, next_state)
|
|
267
|
+
muted = clone_state(next_state)
|
|
268
|
+
open_conditions(muted).each { |condition| muted.open.delete(condition) unless previous.open.key?(condition) }
|
|
269
|
+
muted
|
|
270
|
+
end
|
|
271
|
+
|
|
272
|
+
def silenced?(state, now)
|
|
273
|
+
!state.silenced_until.nil? && state.silenced_until > now
|
|
274
|
+
end
|
|
275
|
+
|
|
276
|
+
# An evaluation as it is saved and sent: while the job was silenced when it
|
|
277
|
+
# began, nothing opens and nothing is sent.
|
|
278
|
+
def apply_silence(previous, evaluation, now)
|
|
279
|
+
return evaluation unless silenced?(previous, now)
|
|
280
|
+
|
|
281
|
+
Evaluation.new(state: mute_opens(previous, evaluation.state), alerts: [])
|
|
282
|
+
end
|
|
283
|
+
|
|
284
|
+
# Whether an alert waiting to be retried no longer describes the job, so it
|
|
285
|
+
# is dropped rather than sent late. An alert for a condition is stale once
|
|
286
|
+
# that condition has closed, or has closed and opened again (it opened at a
|
|
287
|
+
# time other than the alert's). A recovery is stale when any condition it
|
|
288
|
+
# names is open again; while they all stay closed it is kept.
|
|
289
|
+
def stale_alert?(alert, state)
|
|
290
|
+
return Array(alert.details[:after]).any? { |condition| state.open.key?(condition.to_sym) } if alert.type == :recovered
|
|
291
|
+
|
|
292
|
+
state.open[alert.type] != alert.at
|
|
293
|
+
end
|
|
294
|
+
|
|
295
|
+
# How a job looks at a glance. Silence wins, then stuck, failing and late.
|
|
296
|
+
def job_health(definition, last_run, state, now)
|
|
297
|
+
open = open_conditions(state)
|
|
298
|
+
return :silenced if silenced?(state, now)
|
|
299
|
+
return :stuck if open.include?(:stuck) || (last_run && stuck?(definition, last_run, now))
|
|
300
|
+
return :failing if open.include?(:failed) || %i[failed timeout].include?(last_run&.status)
|
|
301
|
+
return :late if open.include?(:missed)
|
|
302
|
+
return :never_ran if last_run.nil?
|
|
303
|
+
|
|
304
|
+
:healthy
|
|
305
|
+
end
|
|
306
|
+
|
|
307
|
+
# A job's summary from its most recent runs (newest first; the first
|
|
308
|
+
# BASELINE_WINDOW are used) and its state. Stats cover runs of any status;
|
|
309
|
+
# the percentiles are over the successful ones among them.
|
|
310
|
+
def summarize(stored, recent, state, next_expected_at, now)
|
|
311
|
+
summary(stored, recent, state, next_expected_at) { |last_run| job_health(stored.definition, last_run, state, now) }
|
|
312
|
+
end
|
|
313
|
+
|
|
314
|
+
# The summary of a job that could not be evaluated, say because its stored
|
|
315
|
+
# schedule no longer parses. It reads nothing from the definition. The job
|
|
316
|
+
# shows as failing (or silenced, while it is), since it needs a look, and
|
|
317
|
+
# nothing is known about when it is next due.
|
|
318
|
+
def unevaluable_summary(stored, recent, state, now)
|
|
319
|
+
summary(stored, recent, state, nil) { silenced?(state, now) ? :silenced : :failing }
|
|
320
|
+
end
|
|
321
|
+
|
|
322
|
+
def summary(stored, recent, state, next_expected_at)
|
|
323
|
+
window = recent.first(BASELINE_WINDOW)
|
|
324
|
+
last_run = window.first
|
|
325
|
+
finished = window.reject { |r| r.status == :running }
|
|
326
|
+
ok_durations = window.select { |r| r.status == :ok && !r.duration_ms.nil? }.map(&:duration_ms)
|
|
327
|
+
JobSummary.new(
|
|
328
|
+
name: stored.name,
|
|
329
|
+
definition: stored.definition,
|
|
330
|
+
health: yield(last_run),
|
|
331
|
+
open: open_conditions(state),
|
|
332
|
+
last_run: last_run,
|
|
333
|
+
next_expected_at: next_expected_at,
|
|
334
|
+
consecutive_failures: state.consecutive_failures,
|
|
335
|
+
silenced_until: state.silenced_until,
|
|
336
|
+
stats: JobStats.new(
|
|
337
|
+
runs: finished.length,
|
|
338
|
+
ok_rate: finished.empty? ? 1 : finished.count { |r| r.status == :ok }.fdiv(finished.length),
|
|
339
|
+
p50_ms: Stats.percentile(ok_durations, 50),
|
|
340
|
+
p95_ms: Stats.percentile(ok_durations, 95),
|
|
341
|
+
),
|
|
342
|
+
)
|
|
343
|
+
end
|
|
344
|
+
end
|
|
345
|
+
end
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Cronwatch
|
|
4
|
+
class Client
|
|
5
|
+
# One shared check: the first caller runs it, the others wait for its result.
|
|
6
|
+
class Flight
|
|
7
|
+
def initialize
|
|
8
|
+
@lock = Mutex.new
|
|
9
|
+
@done = ConditionVariable.new
|
|
10
|
+
@finished = false
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
def resolve(value)
|
|
14
|
+
settle(value, nil)
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def reject(error)
|
|
18
|
+
settle(nil, error)
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def value
|
|
22
|
+
@lock.synchronize do
|
|
23
|
+
@done.wait(@lock) until @finished
|
|
24
|
+
raise @error if @error
|
|
25
|
+
|
|
26
|
+
@value
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
private
|
|
31
|
+
|
|
32
|
+
def settle(value, error)
|
|
33
|
+
@lock.synchronize do
|
|
34
|
+
@value = value
|
|
35
|
+
@error = error
|
|
36
|
+
@finished = true
|
|
37
|
+
@done.broadcast
|
|
38
|
+
end
|
|
39
|
+
end
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
end
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Cronwatch
|
|
4
|
+
module Format
|
|
5
|
+
NAMED = /\A[A-Za-z_$][A-Za-z0-9_$]*: /
|
|
6
|
+
|
|
7
|
+
module_function
|
|
8
|
+
|
|
9
|
+
def at_time(at, now)
|
|
10
|
+
return "never" if at.nil?
|
|
11
|
+
|
|
12
|
+
"#{JS.iso(at).tr("T", " ")[0, 19]} UTC (#{Duration.relative(at, now)})"
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
def first_lines(text, n)
|
|
16
|
+
return "" if text.nil? || text.empty?
|
|
17
|
+
|
|
18
|
+
text.split("\n", -1).first(n).join("\n")
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def tail(text, n)
|
|
22
|
+
return "" if text.nil? || text.empty?
|
|
23
|
+
|
|
24
|
+
lines = JS.trim_end(text).split("\n", -1)
|
|
25
|
+
lines.last(n).join("\n")
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
# "Error: x" for a bare message, but not "Error: TypeError: x" for one that already names itself.
|
|
29
|
+
def error_line(error)
|
|
30
|
+
text = first_lines(error, 4)
|
|
31
|
+
NAMED.match?(text) ? text : "Error: #{text}"
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# Turns a draft into the title and message every channel shows.
|
|
35
|
+
def compose_alert(draft, definition, now)
|
|
36
|
+
name = definition.name
|
|
37
|
+
run = draft.run
|
|
38
|
+
details = draft.details
|
|
39
|
+
lines = []
|
|
40
|
+
|
|
41
|
+
title =
|
|
42
|
+
case draft.type
|
|
43
|
+
when :missed
|
|
44
|
+
lines << "Due #{at_time(details[:due_at], now)}, and no run had started by #{at_time(details[:deadline], now)} (grace #{Duration.format(details[:grace_ms])})."
|
|
45
|
+
lines << "Schedule: #{definition.schedule.nil? ? "undefined" : definition.schedule}#{definition.timezone.to_s.empty? ? "" : " (#{definition.timezone})"}."
|
|
46
|
+
lines << "Last run: #{run ? "#{run.status} #{at_time(run.started_at, now)}" : "never"}."
|
|
47
|
+
"#{name} missed its scheduled run"
|
|
48
|
+
when :failed
|
|
49
|
+
n = details[:consecutive_failures]
|
|
50
|
+
lines << "#{n} consecutive failures." if n > 1
|
|
51
|
+
if run
|
|
52
|
+
lines << "Started #{at_time(run.started_at, now)}#{run.duration_ms.nil? ? "" : ", ran #{Duration.format(run.duration_ms)}"}."
|
|
53
|
+
lines << error_line(run.error) if run.error && !run.error.empty?
|
|
54
|
+
out = tail(run.output, 8)
|
|
55
|
+
lines << "Output (tail):\n#{out}" unless out.empty?
|
|
56
|
+
end
|
|
57
|
+
"#{name} failed"
|
|
58
|
+
when :stuck
|
|
59
|
+
if run
|
|
60
|
+
lines << "Started #{at_time(run.started_at, now)} and never reported finishing. Marked as timed out after #{Duration.format(run.duration_ms.nil? ? now - run.started_at : run.duration_ms)}."
|
|
61
|
+
out = tail(run.output, 8)
|
|
62
|
+
lines << "Output so far (tail):\n#{out}" unless out.empty?
|
|
63
|
+
end
|
|
64
|
+
lines << "If the process was killed mid-run (a serverless timeout, a deploy), this is what that looks like."
|
|
65
|
+
"#{name} is stuck"
|
|
66
|
+
when :slow
|
|
67
|
+
lines << "Took #{Duration.format(details[:duration_ms])}; the limit is #{Duration.format(details[:threshold_ms])} (#{details[:basis]})."
|
|
68
|
+
lines << "Started #{at_time(run.started_at, now)}." if run
|
|
69
|
+
"#{name} was slow"
|
|
70
|
+
when :over_budget
|
|
71
|
+
details[:breaches].each do |b|
|
|
72
|
+
lines << "#{b[:metric]}: #{Evaluate.format_number(b[:value])}, limit #{Evaluate.format_number(b[:limit])} (#{b[:basis]})."
|
|
73
|
+
end
|
|
74
|
+
lines << "Started #{at_time(run.started_at, now)}." if run
|
|
75
|
+
"#{name} went over budget"
|
|
76
|
+
when :recovered
|
|
77
|
+
after = details[:after].map { |c| c.to_s.sub("_", " ") }.join(", ")
|
|
78
|
+
lines << "A run #{run ? at_time(run.started_at, now) : "just now"} succeeded#{after.empty? ? "" : " after: #{after}"}."
|
|
79
|
+
lines << "Ran #{Duration.format(run.duration_ms)}." if run && !run.duration_ms.nil?
|
|
80
|
+
"#{name} recovered"
|
|
81
|
+
else
|
|
82
|
+
raise ArgumentError, "unknown alert type #{draft.type.inspect}"
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
Alert.new(type: draft.type, run: run, details: details, job: name, definition: definition, title: title,
|
|
86
|
+
message: lines.join("\n"), at: now)
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
end
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "net/http"
|
|
4
|
+
require "uri"
|
|
5
|
+
|
|
6
|
+
module Cronwatch
|
|
7
|
+
# POSTs for the alert channels. Anything with `post(url, body, headers)`
|
|
8
|
+
# returning a Response can stand in for it, which is how the tests run.
|
|
9
|
+
module HTTP
|
|
10
|
+
TIMEOUT = 10
|
|
11
|
+
|
|
12
|
+
Response = Struct.new(:status, :body, keyword_init: true) do
|
|
13
|
+
def ok? = status >= 200 && status < 300
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
class NetHTTP
|
|
17
|
+
def post(url, body, headers)
|
|
18
|
+
uri = URI(url)
|
|
19
|
+
request = Net::HTTP::Post.new(uri.request_uri)
|
|
20
|
+
headers.each { |k, v| request[k] = v }
|
|
21
|
+
request.body = body
|
|
22
|
+
response = connection(uri).request(request)
|
|
23
|
+
Response.new(status: response.code.to_i, body: response.body.to_s)
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# A connection that gives up after ten seconds, as the SDK's requests do.
|
|
27
|
+
def connection(uri)
|
|
28
|
+
http = Net::HTTP.new(uri.host, uri.port)
|
|
29
|
+
http.use_ssl = uri.scheme == "https"
|
|
30
|
+
http.open_timeout = TIMEOUT
|
|
31
|
+
http.read_timeout = TIMEOUT
|
|
32
|
+
http.write_timeout = TIMEOUT
|
|
33
|
+
http
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
module_function
|
|
38
|
+
|
|
39
|
+
def default
|
|
40
|
+
@default ||= NetHTTP.new
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# Compares two secrets without stopping at the first differing character.
|
|
44
|
+
def constant_time_equal?(a, b)
|
|
45
|
+
return false unless a.bytesize == b.bytesize
|
|
46
|
+
|
|
47
|
+
diff = 0
|
|
48
|
+
a.bytes.zip(b.bytes) { |x, y| diff |= x ^ y }
|
|
49
|
+
diff.zero?
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
end
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "abort_signal"
|
|
4
|
+
|
|
5
|
+
module Cronwatch
|
|
6
|
+
# What a job's block receives.
|
|
7
|
+
class JobContext
|
|
8
|
+
attr_reader :name, :run_id, :started_at, :signal
|
|
9
|
+
|
|
10
|
+
def initialize(run, signal, recorder)
|
|
11
|
+
@name = run.job
|
|
12
|
+
@run_id = run.id
|
|
13
|
+
@started_at = run.started_at
|
|
14
|
+
@signal = signal
|
|
15
|
+
@recorder = recorder
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
# Append a line of output. Kept with the run, capped at 16 KB, shown in alerts and the dashboard.
|
|
19
|
+
def log(*parts)
|
|
20
|
+
@recorder.log(parts.map { |part| JobContext.stringify(part) }.join(" "))
|
|
21
|
+
nil
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# Report a number for this run: tokens, cost, rows, anything. Watched against budgets and baselines.
|
|
25
|
+
def metric(name, value)
|
|
26
|
+
unless value.is_a?(Numeric) && value.real? && JS.finite?(value)
|
|
27
|
+
raise ArgumentError, "metric \"#{name}\" must be a finite number"
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
@recorder.metric(Output.utf8(name), value)
|
|
31
|
+
nil
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def metrics(values = nil, **more)
|
|
35
|
+
(values || {}).merge(more).each { |k, v| metric(k, v) }
|
|
36
|
+
nil
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# True once the job's timeout has passed. Honour it if the work can stop.
|
|
40
|
+
def aborted?
|
|
41
|
+
@signal.aborted?
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# A logged value as text, the way the SDK writes it, always valid UTF-8.
|
|
45
|
+
def self.stringify(part)
|
|
46
|
+
text =
|
|
47
|
+
case part
|
|
48
|
+
when String then part
|
|
49
|
+
when Symbol then part.to_s
|
|
50
|
+
when Exception then "#{part.class.name || part.class}: #{Output.utf8(part.message)}"
|
|
51
|
+
when Hash, Array, Numeric, true, false, nil, Struct then JS.json(part)
|
|
52
|
+
else part.to_s
|
|
53
|
+
end
|
|
54
|
+
Output.utf8(text)
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# A declared job. Keep it, and call #run with a block.
|
|
59
|
+
class JobHandle
|
|
60
|
+
attr_reader :name, :definition
|
|
61
|
+
|
|
62
|
+
def initialize(client, definition)
|
|
63
|
+
@client = client
|
|
64
|
+
@definition = definition
|
|
65
|
+
@name = definition.name
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
# Run the block now, recording the run. Returns what the block returns and
|
|
69
|
+
# re-raises what it raises, after the run is recorded.
|
|
70
|
+
def run(trigger: "run", &block)
|
|
71
|
+
raise ArgumentError, "run needs a block" unless block
|
|
72
|
+
|
|
73
|
+
outcome = @client.execute(@definition, trigger, &block)
|
|
74
|
+
raise outcome.error if outcome.threw
|
|
75
|
+
|
|
76
|
+
outcome.result
|
|
77
|
+
end
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# Collects a run's output and metrics while its block runs.
|
|
81
|
+
class RunRecorder
|
|
82
|
+
# Lines are dropped from the front once the output is well past the cap;
|
|
83
|
+
# Output.cap trims it exactly at the end.
|
|
84
|
+
KEEP = 64 * 1024
|
|
85
|
+
|
|
86
|
+
attr_reader :context, :signal
|
|
87
|
+
|
|
88
|
+
def initialize(run, timeout_ms)
|
|
89
|
+
@lines = []
|
|
90
|
+
@size = 0
|
|
91
|
+
# The first lines logged, up to the cap, and whether any line has been
|
|
92
|
+
# dropped from @lines: what expect_text needs once the output runs long.
|
|
93
|
+
@head = []
|
|
94
|
+
@head_size = 0
|
|
95
|
+
@dropped = false
|
|
96
|
+
@metrics = {}
|
|
97
|
+
@lock = Mutex.new
|
|
98
|
+
@signal = AbortSignal.new(timeout_ms)
|
|
99
|
+
@context = JobContext.new(run, @signal, self)
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
def log(line)
|
|
103
|
+
length = JS.length16(line)
|
|
104
|
+
@lock.synchronize do
|
|
105
|
+
if @head_size < Output::CAP
|
|
106
|
+
@head << line
|
|
107
|
+
@head_size += length + 1
|
|
108
|
+
end
|
|
109
|
+
@lines << line
|
|
110
|
+
@size += length + 1
|
|
111
|
+
while @size > KEEP && @lines.length > 1
|
|
112
|
+
@size -= JS.length16(@lines.shift) + 1
|
|
113
|
+
@dropped = true
|
|
114
|
+
end
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
def metric(name, value)
|
|
119
|
+
@lock.synchronize { @metrics[name] = value }
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
def output
|
|
123
|
+
@lock.synchronize { @lines.empty? ? nil : Output.cap(@lines.join("\n")) }
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
# What an expect rule is checked against: everything logged, or when that
|
|
127
|
+
# ran long, the first 16 KB and the last 16 KB. The stored output keeps
|
|
128
|
+
# only the tail, so a "done" line printed early would otherwise be lost.
|
|
129
|
+
def expect_text
|
|
130
|
+
@lock.synchronize do
|
|
131
|
+
next nil if @lines.empty?
|
|
132
|
+
|
|
133
|
+
all = @lines.join("\n")
|
|
134
|
+
next all if !@dropped && JS.length16(all) <= 2 * Output::CAP
|
|
135
|
+
|
|
136
|
+
"#{JS.head16(@head.join("\n"), Output::CAP)}\n#{JS.tail16(all, Output::CAP)}"
|
|
137
|
+
end
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
def metrics
|
|
141
|
+
@lock.synchronize { @metrics.dup }
|
|
142
|
+
end
|
|
143
|
+
end
|
|
144
|
+
end
|