cronwatch 0.3.1 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +19 -3
  3. data/lib/cronwatch/alerts/bugsnag.rb +69 -0
  4. data/lib/cronwatch/alerts/custom.rb +9 -2
  5. data/lib/cronwatch/alerts/datadog.rb +54 -0
  6. data/lib/cronwatch/alerts/email.rb +78 -0
  7. data/lib/cronwatch/alerts/honeybadger.rb +59 -0
  8. data/lib/cronwatch/alerts/mailgun.rb +36 -0
  9. data/lib/cronwatch/alerts/newrelic.rb +52 -0
  10. data/lib/cronwatch/alerts/postmark.rb +41 -0
  11. data/lib/cronwatch/alerts/provider.rb +159 -0
  12. data/lib/cronwatch/alerts/resend.rb +39 -0
  13. data/lib/cronwatch/alerts/rollbar.rb +53 -0
  14. data/lib/cronwatch/alerts/sendgrid.rb +41 -0
  15. data/lib/cronwatch/alerts/sentry.rb +94 -0
  16. data/lib/cronwatch/alerts/ses.rb +72 -0
  17. data/lib/cronwatch/alerts/sigv4.rb +84 -0
  18. data/lib/cronwatch/alerts/twilio.rb +206 -0
  19. data/lib/cronwatch/alerts/webhook.rb +7 -2
  20. data/lib/cronwatch/client.rb +473 -52
  21. data/lib/cronwatch/evaluate.rb +14 -1
  22. data/lib/cronwatch/format.rb +10 -4
  23. data/lib/cronwatch/http.rb +67 -6
  24. data/lib/cronwatch/job.rb +17 -0
  25. data/lib/cronwatch/pg_cron.rb +524 -0
  26. data/lib/cronwatch/run_handle.rb +203 -0
  27. data/lib/cronwatch/schedule.rb +28 -0
  28. data/lib/cronwatch/stores/active_record.rb +23 -0
  29. data/lib/cronwatch/stores/memory.rb +33 -10
  30. data/lib/cronwatch/types.rb +1 -0
  31. data/lib/cronwatch/version.rb +1 -1
  32. data/lib/cronwatch/web/app.rb +97 -20
  33. data/lib/cronwatch/web/html.rb +395 -140
  34. data/lib/cronwatch/web/icons.rb +19 -0
  35. data/lib/cronwatch/web/origin.rb +133 -0
  36. data/lib/cronwatch/web/pwa.rb +137 -0
  37. data/lib/cronwatch/web/timeline.rb +476 -0
  38. data/lib/cronwatch/web.rb +4 -0
  39. data/lib/cronwatch.rb +17 -1
  40. metadata +28 -3
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require "net/http"
4
+ require "timeout"
4
5
  require "uri"
5
6
 
6
7
  module Cronwatch
@@ -13,25 +14,71 @@ module Cronwatch
13
14
  def ok? = status >= 200 && status < 300
14
15
  end
15
16
 
17
+ # The request took longer than its deadline. fetch's message for it.
18
+ class TimeoutError < ::Timeout::Error
19
+ def initialize(message = "The operation was aborted due to timeout")
20
+ super
21
+ end
22
+ end
23
+
16
24
  class NetHTTP
25
+ # The deadline is the whole request's, as fetch's
26
+ # AbortSignal.timeout(10_000) is; `timeout` is for tests.
27
+ def initialize(timeout: TIMEOUT)
28
+ @timeout = timeout
29
+ end
30
+
31
+ # Header values are sent with the whitespace around them trimmed, as
32
+ # fetch trims them, so a credential read with a trailing newline still
33
+ # sends. Past the deadline before an answer, raises HTTP::TimeoutError;
34
+ # past it while the body is still arriving, returns the answer with an
35
+ # empty body, as the SDK's channels treat a body they could not read.
17
36
  def post(url, body, headers)
18
37
  uri = URI(url)
38
+ deadline = HTTP.monotonic + @timeout
19
39
  request = Net::HTTP::Post.new(uri.request_uri)
20
- headers.each { |k, v| request[k] = v }
40
+ headers.each { |k, v| request[k] = HTTP.trim_header(v) }
21
41
  request.body = body
22
- response = connection(uri).request(request)
23
- Response.new(status: response.code.to_i, body: response.body.to_s)
42
+ http = connection(uri)
43
+ http.open_timeout = HTTP.remaining(deadline)
44
+ http.start do |conn|
45
+ remaining = HTTP.remaining(deadline)
46
+ conn.read_timeout = remaining
47
+ conn.write_timeout = remaining
48
+ conn.request(request) do |response|
49
+ return Response.new(status: response.code.to_i, body: read_body(conn, response, deadline))
50
+ end
51
+ end
52
+ rescue Net::OpenTimeout, Net::ReadTimeout, Net::WriteTimeout
53
+ raise TimeoutError
24
54
  end
25
55
 
26
56
  # A connection that gives up after ten seconds, as the SDK's requests do.
27
57
  def connection(uri)
28
58
  http = Net::HTTP.new(uri.host, uri.port)
29
59
  http.use_ssl = uri.scheme == "https"
30
- http.open_timeout = TIMEOUT
31
- http.read_timeout = TIMEOUT
32
- http.write_timeout = TIMEOUT
60
+ http.open_timeout = @timeout
61
+ http.read_timeout = @timeout
62
+ http.write_timeout = @timeout
33
63
  http
34
64
  end
65
+
66
+ private
67
+
68
+ # The body, read in chunks until the deadline; "" when it passes first.
69
+ def read_body(conn, response, deadline)
70
+ chunks = []
71
+ conn.read_timeout = HTTP.remaining(deadline)
72
+ response.read_body do |chunk|
73
+ raise TimeoutError if HTTP.monotonic >= deadline
74
+
75
+ chunks << chunk
76
+ conn.read_timeout = HTTP.remaining(deadline)
77
+ end
78
+ chunks.join.b
79
+ rescue Net::ReadTimeout, TimeoutError
80
+ ""
81
+ end
35
82
  end
36
83
 
37
84
  module_function
@@ -40,6 +87,20 @@ module Cronwatch
40
87
  @default ||= NetHTTP.new
41
88
  end
42
89
 
90
+ def monotonic
91
+ Process.clock_gettime(Process::CLOCK_MONOTONIC)
92
+ end
93
+
94
+ # Seconds left before `deadline`, never quite zero (Net::HTTP reads 0 as no wait).
95
+ def remaining(deadline)
96
+ [deadline - monotonic, 0.001].max
97
+ end
98
+
99
+ # A header value without the spaces, tabs and line breaks around it, as fetch sends it.
100
+ def trim_header(value)
101
+ value.to_s.gsub(/\A[ \t\r\n]+|[ \t\r\n]+\z/, "")
102
+ end
103
+
43
104
  # Compares two secrets without stopping at the first differing character.
44
105
  def constant_time_equal?(a, b)
45
106
  return false unless a.bytesize == b.bytesize
data/lib/cronwatch/job.rb CHANGED
@@ -75,6 +75,23 @@ module Cronwatch
75
75
 
76
76
  outcome.result
77
77
  end
78
+
79
+ # Record a running run now and finish it later, perhaps from another
80
+ # process (see #resume). Returns a RunHandle. `trigger` defaults to
81
+ # "start". `id` is your own stable id for the run, 1 to 200 characters,
82
+ # such as a queue's message id: a start with an id already recorded for
83
+ # this job records nothing and returns a handle on that run instead.
84
+ # Store failures go to on_error; it never raises for them. A run that is
85
+ # never finished is marked stuck by the first check after the job's timeout.
86
+ def start(trigger: nil, id: nil)
87
+ @client.start_run(@definition, trigger: trigger, id: id)
88
+ end
89
+
90
+ # A RunHandle on a run this job started elsewhere, by its id, so this
91
+ # process can log to it and finish it. Raises only for a run of another job.
92
+ def resume(run_id)
93
+ @client.resume_handle(@definition, run_id)
94
+ end
78
95
  end
79
96
 
80
97
  # Collects a run's output and metrics while its block runs.
@@ -0,0 +1,524 @@
1
+ # frozen_string_literal: true
2
+
3
+ # The pg_cron reader. Needs no gem of its own: it queries through the
4
+ # connection it is given (ActiveRecord, the pg gem, or your own).
5
+ #
6
+ # require "cronwatch/pg_cron"
7
+ require "set"
8
+ require "time"
9
+ require "cronwatch" unless defined?(Cronwatch::Client)
10
+
11
+ module Cronwatch
12
+ # Where runs this process does not wrap come from. A source is anything
13
+ # with `name` and `sync(host)`: on every check the client calls sync with
14
+ # itself as the host, and the source declares jobs (`host.job`), reads the
15
+ # store (`host.store`), records the runs it found (`host.record_run`) and
16
+ # reports problems (`host.on_error(error, where)`). sync returns the alerts
17
+ # recording sent, or anything else for none.
18
+ module Sources
19
+ # Watches pg_cron jobs, which run inside Postgres where nothing can wrap
20
+ # them (the SDK's sources/pgcron.ts). As a source, on every check it
21
+ # reads cron.job and declares each job with its schedule, then copies new
22
+ # rows of cron.job_run_details in as runs (ids "pgcron:<runid>"), so the
23
+ # usual evaluation raises missed, failed, stuck and slow alerts.
24
+ #
25
+ # A job that is renamed, unscheduled or no longer picked keeps its old
26
+ # name's runs and history, and that name is declared again without a
27
+ # schedule, so it is never reported missed. Its description says why.
28
+ #
29
+ # require "cronwatch/pg_cron"
30
+ # Cronwatch.configure do |c|
31
+ # c.store = Cronwatch::Stores::ActiveRecord.new
32
+ # c.sources = [Cronwatch::Sources::PgCron.new(ActiveRecord::Base)]
33
+ # end
34
+ #
35
+ # `db` is an ActiveRecord class, connection pool or connection
36
+ # (queried through exec_query), a PG::Connection (exec_params), or
37
+ # anything with `query(sql, params)` returning rows as hashes with string
38
+ # keys.
39
+ #
40
+ # jobs: which jobs to watch: names or ids, or a callable that picks them (given a Job). Default every job the role can see.
41
+ # prefix: put before every job name, to keep them apart from your own ("db:"). Also keeps run ids apart.
42
+ # job_name: a callable giving the CronWatch name for a Job. Default its jobname with anything other than
43
+ # letters, digits, ".", "_", ":" and "-" turned into "-", or "pg_cron:<jobid>" when it has none.
44
+ # The prefix goes in front either way.
45
+ # options: grace, timeout, max_duration, expect and the rest, for every job (a hash) or per job (a callable
46
+ # given a Job). The schedule and timezone always come from pg_cron.
47
+ # timezone: the timezone pg_cron reads its cron expressions in. Default the server's cron.timezone, read from
48
+ # pg_settings, which shows it only to roles with pg_read_all_settings; UTC (pg_cron's default) is
49
+ # assumed when it cannot be read.
50
+ class PgCron
51
+ # A row of cron.job.
52
+ Job = Struct.new(:jobid, :jobname, :schedule, :database, :username, :active, keyword_init: true)
53
+
54
+ # How many of a job's newest runs are copied, without alerting, the first time it is seen.
55
+ BACKFILL = 20
56
+ # Run details read per query, and the most pages read in one sync.
57
+ PAGE = 500
58
+ MAX_PAGES = 10
59
+ # How long a run pg_cron has queued but not started (no start_time yet)
60
+ # is waited for. After that it is copied as running from when it was
61
+ # first seen, so a run that never starts is marked stuck like any other.
62
+ HOLD_MS = 10 * 60_000
63
+
64
+ JOBS_SQL = "SELECT jobid, jobname, schedule, database, username, active FROM cron.job ORDER BY jobid"
65
+ # pg_settings has no row for a setting the role may not read, where
66
+ # current_setting() raises an error that would abort the caller's transaction.
67
+ SETTING_SQL = "SELECT setting FROM pg_settings WHERE name = $1"
68
+ COLUMNS = "d.runid, d.jobid, d.status, d.return_message, d.start_time, d.end_time"
69
+ # Every tracked job's runs after its cursor, and any run still open here, whatever its job.
70
+ RUNS_SQL = <<~SQL.chomp
71
+ SELECT #{COLUMNS}
72
+ FROM cron.job_run_details d
73
+ LEFT JOIN unnest($1::bigint[], $2::bigint[]) AS c(jobid, after) ON d.jobid = c.jobid
74
+ WHERE d.runid > c.after OR d.runid = ANY($3::bigint[])
75
+ ORDER BY d.runid LIMIT #{PAGE}
76
+ SQL
77
+ NEWEST_SQL = "SELECT #{COLUMNS} FROM cron.job_run_details d WHERE d.jobid = $1 ORDER BY d.runid DESC LIMIT #{BACKFILL}"
78
+
79
+ SECONDS = Regexp.new("\\A(\\d+)[#{JS::WHITESPACE}]*seconds?\\z", Regexp::IGNORECASE)
80
+ REBOOT = /\A@reboot\z/i
81
+ UTC = /\A(gmt|utc|z)\z/i
82
+ SCHEDULE_ONLY = %i[schedule timezone].freeze
83
+ # The options of a definition that are declared again, without its schedule, for a name no longer in use.
84
+ UNSCHEDULED = %i[description tags grace timeout max_duration budget failures_before_alert].freeze
85
+ DESCRIBED = /\Apg_cron job (\d+) in /
86
+
87
+ attr_reader :name
88
+
89
+ def initialize(db, jobs: nil, prefix: "", job_name: nil, options: nil, timezone: nil)
90
+ @db = PgCron.adapter(db)
91
+ @jobs = jobs
92
+ @prefix = prefix.to_s
93
+ @id_prefix = "pgcron:#{@prefix}"
94
+ @job_name = job_name
95
+ @options = options
96
+ @timezone = timezone
97
+ # The newest runid read for each jobid, once known.
98
+ @cursors = {}
99
+ # The start of the newest run copied for each jobid: where a restart row with no times is put.
100
+ @last_at = {}
101
+ # Runs copied while still going, by runid, with their job: read again until they finish, even once a check marks them timeout.
102
+ @pending = {}
103
+ # Runs read before they started, by runid, with when they were first seen.
104
+ @held = {}
105
+ # Each job's name and definition as last declared, by jobid.
106
+ @known = {}
107
+ # The last definition declared for each name, so an unchanged job is not declared again.
108
+ @declared = {}
109
+ # Names declared again without a schedule by retire, whose open runs are still read.
110
+ @retired = Set.new
111
+ @scanned = false
112
+ @warned = Set.new
113
+ @name = "pg_cron"
114
+ end
115
+
116
+ # pg_cron takes a cron expression, with "$" for the last day of the
117
+ # month, or "N seconds" for 1 to 59 seconds. Returns the CronWatch
118
+ # schedule, or nil for one that has no cadence to watch. pg_cron reads
119
+ # only the first five fields of an expression and ignores the rest, so
120
+ # only those are kept (a sixth would otherwise be read as seconds).
121
+ def self.schedule(schedule)
122
+ text = JS.trim(schedule.to_s)
123
+ seconds = SECONDS.match(text)
124
+ return "every #{seconds[1].to_i}s" if seconds
125
+ return nil if REBOOT.match?(text)
126
+
127
+ fields = text.split(JS::SPACES)
128
+ fields = fields.first(5) if fields.length > 5 && !fields[0].start_with?("@")
129
+ fields[2] = fields[2].tr("$", "L") if fields.length == 5 && fields[2].include?("$")
130
+ fields.join(" ")
131
+ end
132
+
133
+ # The default CronWatch name for a pg_cron job, before the prefix.
134
+ def self.job_name(job)
135
+ cleaned = job.jobname.to_s.gsub(/[^A-Za-z0-9._:-]+/, "-").sub(/\A[^A-Za-z0-9]+/, "")[0, 100]
136
+ cleaned.empty? ? "pg_cron:#{job.jobid}" : cleaned
137
+ end
138
+
139
+ # Whether a row's status says the run is over.
140
+ def self.finished_status?(status)
141
+ %w[succeeded failed].include?(status.to_s)
142
+ end
143
+
144
+ # A row of cron.job_run_details as a CronWatch run, or nil for one that
145
+ # has not started (no start_time, not finished). A finished row with no
146
+ # start_time (pg_cron writes these for runs a server restart cut off,
147
+ # "server restarted") starts at its end_time, else at `fallback_at`
148
+ # (the reader passes the job's newest run's start, or now).
149
+ def self.run(row, job, id_prefix, fallback_at = nil)
150
+ finished_at = row["end_time"].nil? ? nil : epoch_ms(row["end_time"])
151
+ done = finished_status?(row["status"])
152
+ return nil if row["start_time"].nil? && !done
153
+
154
+ started_at =
155
+ if row["start_time"].nil? then finished_at || fallback_at || Process.clock_gettime(Process::CLOCK_REALTIME, :millisecond)
156
+ else epoch_ms(row["start_time"])
157
+ end
158
+ message = row["return_message"].nil? ? nil : JS.trim(Output.utf8(row["return_message"]))
159
+ message = nil if message == ""
160
+ status = { "succeeded" => :ok, "failed" => :failed }.fetch(row["status"].to_s, :running)
161
+ finish = done ? [started_at, finished_at || started_at].max : nil
162
+ Run.new(
163
+ id: "#{id_prefix}#{row["runid"]}", job: job, status: status, started_at: started_at, finished_at: finish,
164
+ duration_ms: finish.nil? ? nil : finish - started_at,
165
+ error: status == :failed ? (message || "pg_cron reported the run as failed") : nil,
166
+ output: status == :ok ? message : nil, metrics: {}, trigger: "pg_cron",
167
+ )
168
+ end
169
+
170
+ # A timestamp as epoch milliseconds: a Time (ActiveRecord decodes them), a string as Postgres writes one, or a number.
171
+ def self.epoch_ms(value)
172
+ case value
173
+ when Integer then value
174
+ when Time then (value.to_r * 1000).floor
175
+ when String then (Time.parse(value).to_r * 1000).floor
176
+ else value.respond_to?(:to_time) ? (value.to_time.to_r * 1000).floor : Integer(value)
177
+ end
178
+ end
179
+
180
+ # A value as a query parameter: arrays as Postgres array literals, the rest as given.
181
+ def self.encode(value)
182
+ value.is_a?(Array) ? "{#{value.map { |v| Integer(v) }.join(",")}}" : value
183
+ end
184
+
185
+ def self.boolean(value)
186
+ [true, "t", "true", 1, "1"].include?(value)
187
+ end
188
+
189
+ # The query adapter for `db`.
190
+ def self.adapter(db)
191
+ if db.respond_to?(:exec_params) then PGConnection.new(db)
192
+ elsif db.respond_to?(:connection_pool) || db.respond_to?(:with_connection) || db.respond_to?(:exec_query)
193
+ ActiveRecordConnection.new(db)
194
+ elsif db.respond_to?(:query) then db
195
+ else
196
+ raise ArgumentError, "Cronwatch::Sources::PgCron needs an ActiveRecord class or connection, a PG::Connection, " \
197
+ "or an object with query(sql, params)"
198
+ end
199
+ end
200
+
201
+ # Queries through the pg gem's PG::Connection#exec_params.
202
+ class PGConnection
203
+ def initialize(connection)
204
+ @connection = connection
205
+ @lock = Mutex.new
206
+ end
207
+
208
+ def query(sql, params = [])
209
+ @lock.synchronize { @connection.exec_params(sql, params.map { |v| PgCron.encode(v) }).to_a }
210
+ end
211
+ end
212
+
213
+ # Queries through an ActiveRecord connection's exec_query, checking one
214
+ # out of the pool for each query when given a class or a pool.
215
+ class ActiveRecordConnection
216
+ def initialize(source)
217
+ @source = source
218
+ end
219
+
220
+ def query(sql, params = [])
221
+ with_connection do |connection|
222
+ connection.exec_query(sql, "Cronwatch pg_cron", params.map { |v| PgCron.encode(v) }).to_a
223
+ end
224
+ end
225
+
226
+ private
227
+
228
+ # A class's pool is the writing role's, as the store's is, even
229
+ # inside the app's connected_to(role: :reading): cron.job_run_details
230
+ # is read where pg_cron writes it, not on a replica that may lag or
231
+ # not be configured at all.
232
+ def with_connection(&block)
233
+ if record_class?
234
+ ::ActiveRecord::Base.connected_to(role: ::ActiveRecord.writing_role, prevent_writes: false) do
235
+ @source.connection_pool.with_connection(&block)
236
+ end
237
+ elsif @source.respond_to?(:connection_pool) then @source.connection_pool.with_connection(&block)
238
+ elsif @source.respond_to?(:with_connection) then @source.with_connection(&block)
239
+ else yield @source
240
+ end
241
+ end
242
+
243
+ def record_class?
244
+ defined?(::ActiveRecord::Base) && @source.is_a?(Class) && @source <= ::ActiveRecord::Base
245
+ end
246
+ end
247
+
248
+ # Declares the jobs and records their new runs. Returns the alerts recording them sent.
249
+ def sync(host)
250
+ now = host.now
251
+ timezone = @timezone
252
+ if timezone.nil? || timezone.to_s.empty?
253
+ tz = setting("cron.timezone")
254
+ if tz.nil?
255
+ warn_once(host, "tz", "could not read cron.timezone; assuming UTC. Grant pg_read_all_settings or pass " \
256
+ "Cronwatch::Sources::PgCron.new(db, timezone: ...).")
257
+ end
258
+ timezone = tz.nil? || UTC.match?(tz) ? "UTC" : tz
259
+ end
260
+ recording = setting("cron.log_run") != "off"
261
+ unless recording
262
+ warn_once(host, "log_run", "cron.log_run is off, so pg_cron records no runs: jobs are watched without their " \
263
+ "schedules and no run can fail. Turn it on to watch them.")
264
+ end
265
+
266
+ rows = @db.query(JOBS_SQL, [])
267
+ if rows.empty?
268
+ warn_once(host, "empty", "cron.job shows no jobs. pg_cron's row level security shows a role only the jobs it " \
269
+ "scheduled: connect as that role, or give this one BYPASSRLS.")
270
+ end
271
+ all = rows.map { |r| job_from(r) }
272
+ jobs = all.select { |job| picks?(job) }
273
+ names, definitions = declare(host, jobs, timezone, recording)
274
+ retire_unused(host, names, definitions, all)
275
+ return [] if !recording || names.empty?
276
+
277
+ alerts = []
278
+ start_cursors(host, names, alerts, now)
279
+ read_new(host, names, alerts, now)
280
+ alerts
281
+ end
282
+
283
+ private
284
+
285
+ def job_from(row)
286
+ Job.new(jobid: Integer(row["jobid"]), jobname: row["jobname"], schedule: row["schedule"].to_s,
287
+ database: row["database"], username: row["username"], active: PgCron.boolean(row["active"]))
288
+ end
289
+
290
+ def warn_once(host, key, message)
291
+ return if @warned.include?(key)
292
+
293
+ @warned << key
294
+ host.on_error(RuntimeError.new(message), "source pg_cron")
295
+ end
296
+
297
+ def picks?(job)
298
+ return true if @jobs.nil?
299
+ return @jobs.call(job) ? true : false if @jobs.respond_to?(:call)
300
+
301
+ Array(@jobs).any? { |j| j.is_a?(Integer) ? j == job.jobid : j.to_s == job.jobname }
302
+ end
303
+
304
+ def setting(name)
305
+ rows = @db.query(SETTING_SQL, [name])
306
+ value = rows.first && rows.first["setting"]
307
+ value&.to_s
308
+ rescue StandardError
309
+ nil
310
+ end
311
+
312
+ def run_id_of(id)
313
+ return nil unless id.start_with?(@id_prefix)
314
+
315
+ rest = id[@id_prefix.length..]
316
+ /\A\d{1,15}\z/.match?(rest) ? rest.to_i : nil
317
+ end
318
+
319
+ # Declares each job. A paused one (active = false) keeps its failures but loses its schedule, so it is not missed.
320
+ # Returns [{ jobid => name }, { jobid => definition as declared }].
321
+ def declare(host, jobs, timezone, recording)
322
+ names = {}
323
+ definitions = {}
324
+ used = Set.new
325
+ jobs.each do |job|
326
+ name = @prefix + (@job_name ? @job_name.call(job).to_s : PgCron.job_name(job))
327
+ name = "#{name}:#{job.jobid}" if used.include?(name)
328
+ used << name
329
+ extra = (@options.respond_to?(:call) ? @options.call(job) : @options) || {}
330
+ extra = extra.to_h.transform_keys(&:to_sym).except(*SCHEDULE_ONLY)
331
+ schedule = job.active && recording ? PgCron.schedule(job.schedule) : nil
332
+ definition = {
333
+ description: "pg_cron job #{job.jobid} in #{job.database} as #{job.username}#{job.active ? "" : " (paused)"}",
334
+ tags: ["pg_cron"],
335
+ }.merge(extra)
336
+ definition.merge!(schedule: schedule, timezone: timezone) if schedule
337
+ begin
338
+ key = definition_key(definition)
339
+ if @declared[name] != key
340
+ begin
341
+ host.job(name, **definition)
342
+ rescue StandardError => e
343
+ raise unless schedule
344
+
345
+ # A schedule CronWatch cannot read: watch the runs, not the cadence.
346
+ host.on_error(RuntimeError.new("pg_cron job #{job.jobid}: #{e.message}; watching it without a schedule"), "source pg_cron")
347
+ definition = definition.except(*SCHEDULE_ONLY)
348
+ host.job(name, **definition)
349
+ end
350
+ @declared[name] = key
351
+ end
352
+ names[job.jobid] = name
353
+ definitions[job.jobid] = definition
354
+ rescue StandardError => e
355
+ host.on_error(e, "source pg_cron: job #{job.jobid}")
356
+ end
357
+ end
358
+ [names, definitions]
359
+ end
360
+
361
+ # A definition as text, to tell a changed one from the last declared.
362
+ def definition_key(definition)
363
+ JS.json(definition.transform_values { |v| v.is_a?(Regexp) ? v.inspect : v.respond_to?(:call) ? "function" : v })
364
+ end
365
+
366
+ # The options of a stored or declared definition that can be declared again, without its schedule.
367
+ def unscheduled(definition)
368
+ UNSCHEDULED.each_with_object({}) do |field, out|
369
+ value = definition[field]
370
+ out[field] = value unless value.nil?
371
+ end
372
+ end
373
+
374
+ # Declares a name this source no longer uses for any job again, without its schedule.
375
+ def retire(host, name, definition, why)
376
+ base = unscheduled(definition)
377
+ following = base.merge(description: "#{base[:description] || "pg_cron job"} (#{why})")
378
+ host.job(name, **following)
379
+ @declared[name] = definition_key(following)
380
+ @retired << name
381
+ rescue StandardError => e
382
+ host.on_error(e, "source pg_cron: job #{name}")
383
+ end
384
+
385
+ # A name this source used for a job that has since been renamed, unscheduled or dropped from `jobs`
386
+ # is declared again without its schedule. Once per process, the same for names left scheduled in the
387
+ # store while no process was watching.
388
+ def retire_unused(host, names, definitions, all)
389
+ in_use = names.values.to_set
390
+ in_use.each { |name| @retired.delete(name) }
391
+ @known.each do |jobid, previous|
392
+ next if in_use.include?(previous[:name])
393
+
394
+ renamed = names[jobid]
395
+ retire(host, previous[:name], previous[:definition], renamed ? "renamed to #{renamed}" : "no longer watched")
396
+ end
397
+ @known = names.to_h { |jobid, name| [jobid, { name: name, definition: definitions[jobid] }] }
398
+ return if @scanned || all.empty?
399
+
400
+ @scanned = true
401
+ begin
402
+ visible = all.map(&:jobid).to_set
403
+ host.store.list_jobs.each do |stored|
404
+ definition = stored.definition
405
+ next if !stored.name.start_with?(@prefix) || in_use.include?(stored.name)
406
+ next if definition.schedule.nil? || definition.schedule.to_s.empty? || !Array(definition.tags).include?("pg_cron")
407
+
408
+ match = DESCRIBED.match(definition.description.to_s)
409
+ next unless match
410
+
411
+ jobid = match[1].to_i
412
+ current = names[jobid]
413
+ if !visible.include?(jobid)
414
+ retire(host, stored.name, definition, "no longer in cron.job")
415
+ elsif current && !stored.name.end_with?(current[@prefix.length..])
416
+ # Another pg_cron source's name for the same job ends the same way: that one is left alone.
417
+ retire(host, stored.name, definition, "renamed to #{current}")
418
+ end
419
+ end
420
+ rescue StandardError => e
421
+ host.on_error(e, "source pg_cron")
422
+ end
423
+ end
424
+
425
+ # Copies one detail row in as a run. A row that cannot be recorded is
426
+ # reported and skipped; it never stops the others.
427
+ def record(host, names, row, evaluate, alerts, now)
428
+ runid = Integer(row["runid"])
429
+ jobid = Integer(row["jobid"])
430
+ name = @pending[runid] || names[jobid]
431
+ unless name
432
+ @held.delete(runid)
433
+ return
434
+ end
435
+ if row["start_time"].nil? && !PgCron.finished_status?(row["status"])
436
+ since = @held.fetch(runid, now)
437
+ if now - since < HOLD_MS
438
+ @held[runid] = since
439
+ return
440
+ end
441
+ run = PgCron.run(row.merge("start_time" => since), name, @id_prefix)
442
+ else
443
+ run = PgCron.run(row, name, @id_prefix, @last_at.fetch(jobid, now))
444
+ end
445
+ @held.delete(runid)
446
+ return unless run
447
+
448
+ begin
449
+ alerts.concat(host.record_run(run, evaluate: evaluate))
450
+ rescue StandardError => e
451
+ host.on_error(e, "source pg_cron: run #{runid}")
452
+ return
453
+ end
454
+ if run.status == :running
455
+ @pending[runid] = name
456
+ else
457
+ @pending.delete(runid)
458
+ end
459
+ @last_at[jobid] = run.started_at if !@last_at.key?(jobid) || run.started_at > @last_at[jobid]
460
+ end
461
+
462
+ # Where each job left off. Found from the store the first time, so a restart carries on.
463
+ def start_cursors(host, names, alerts, now)
464
+ names.each do |jobid, name|
465
+ next if @cursors.key?(jobid)
466
+
467
+ ours = host.store.list_runs(name, BACKFILL).select { |r| run_id_of(r.id) }
468
+ if ours.any?
469
+ @cursors[jobid] = ours.map { |r| run_id_of(r.id) }.max
470
+ @last_at[jobid] = ours.map(&:started_at).max
471
+ ours.each { |r| @pending[run_id_of(r.id)] = r.job if %i[running timeout].include?(r.status) }
472
+ next
473
+ end
474
+ # First sight: copy recent history quietly, and judge only from the newest finished run on.
475
+ # The cursor goes to the newest row read, whatever is held, so history is never judged later.
476
+ ordered = @db.query(NEWEST_SQL, [jobid]).reverse
477
+ last_finished = -1
478
+ ordered.each_with_index { |r, i| last_finished = i if PgCron.finished_status?(r["status"]) }
479
+ ordered.each_with_index do |row, i|
480
+ # Already copied under another name (the job was renamed while no process watched): left there.
481
+ next if host.store.get_run("#{@id_prefix}#{row["runid"]}")
482
+
483
+ record(host, names, row, i >= last_finished, alerts, now)
484
+ end
485
+ @cursors[jobid] = ordered.empty? ? 0 : Integer(ordered.last["runid"])
486
+ end
487
+ end
488
+
489
+ # New runs, runs copied while still going (or since marked timeout), and runs not yet started.
490
+ def read_new(host, names, alerts, now)
491
+ watched = names.values.to_set | @retired
492
+ host.store.running_runs.each do |run|
493
+ id = run_id_of(run.id)
494
+ @pending[id] = run.job if id && watched.include?(run.job)
495
+ end
496
+ open = (@pending.keys + @held.keys).to_set
497
+ complete = false
498
+ MAX_PAGES.times do
499
+ jobids = names.keys
500
+ details = @db.query(RUNS_SQL, [jobids, jobids.map { |j| @cursors.fetch(j, 0) }, open.to_a])
501
+ details.each do |row|
502
+ jobid = Integer(row["jobid"])
503
+ runid = Integer(row["runid"])
504
+ open.delete(runid)
505
+ record(host, names, row, true, alerts, now)
506
+ # Held or not, the cursor moves on: a held run is read again by its runid.
507
+ @cursors[jobid] = runid if names.key?(jobid) && runid > @cursors.fetch(jobid, 0)
508
+ end
509
+ if details.length < PAGE
510
+ complete = true
511
+ break
512
+ end
513
+ end
514
+ # Every row was read and these were not among them: pg_cron no longer has them.
515
+ return unless complete
516
+
517
+ open.each do |runid|
518
+ @pending.delete(runid)
519
+ @held.delete(runid)
520
+ end
521
+ end
522
+ end
523
+ end
524
+ end