cronwatch 0.3.1 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +19 -3
- data/lib/cronwatch/alerts/bugsnag.rb +69 -0
- data/lib/cronwatch/alerts/custom.rb +9 -2
- data/lib/cronwatch/alerts/datadog.rb +54 -0
- data/lib/cronwatch/alerts/email.rb +78 -0
- data/lib/cronwatch/alerts/honeybadger.rb +59 -0
- data/lib/cronwatch/alerts/mailgun.rb +36 -0
- data/lib/cronwatch/alerts/newrelic.rb +52 -0
- data/lib/cronwatch/alerts/postmark.rb +41 -0
- data/lib/cronwatch/alerts/provider.rb +159 -0
- data/lib/cronwatch/alerts/resend.rb +39 -0
- data/lib/cronwatch/alerts/rollbar.rb +53 -0
- data/lib/cronwatch/alerts/sendgrid.rb +41 -0
- data/lib/cronwatch/alerts/sentry.rb +94 -0
- data/lib/cronwatch/alerts/ses.rb +72 -0
- data/lib/cronwatch/alerts/sigv4.rb +84 -0
- data/lib/cronwatch/alerts/twilio.rb +206 -0
- data/lib/cronwatch/alerts/webhook.rb +7 -2
- data/lib/cronwatch/client.rb +473 -52
- data/lib/cronwatch/evaluate.rb +14 -1
- data/lib/cronwatch/format.rb +10 -4
- data/lib/cronwatch/http.rb +67 -6
- data/lib/cronwatch/job.rb +17 -0
- data/lib/cronwatch/pg_cron.rb +524 -0
- data/lib/cronwatch/run_handle.rb +203 -0
- data/lib/cronwatch/stores/active_record.rb +23 -0
- data/lib/cronwatch/stores/memory.rb +33 -10
- data/lib/cronwatch/types.rb +1 -0
- data/lib/cronwatch/version.rb +1 -1
- data/lib/cronwatch/web/app.rb +38 -11
- data/lib/cronwatch/web/origin.rb +133 -0
- data/lib/cronwatch/web.rb +1 -0
- data/lib/cronwatch.rb +17 -1
- metadata +19 -1
data/lib/cronwatch/http.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require "net/http"
|
|
4
|
+
require "timeout"
|
|
4
5
|
require "uri"
|
|
5
6
|
|
|
6
7
|
module Cronwatch
|
|
@@ -13,25 +14,71 @@ module Cronwatch
|
|
|
13
14
|
def ok? = status >= 200 && status < 300
|
|
14
15
|
end
|
|
15
16
|
|
|
17
|
+
# The request took longer than its deadline. fetch's message for it.
|
|
18
|
+
class TimeoutError < ::Timeout::Error
|
|
19
|
+
def initialize(message = "The operation was aborted due to timeout")
|
|
20
|
+
super
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
|
|
16
24
|
class NetHTTP
|
|
25
|
+
# The deadline is the whole request's, as fetch's
|
|
26
|
+
# AbortSignal.timeout(10_000) is; `timeout` is for tests.
|
|
27
|
+
def initialize(timeout: TIMEOUT)
|
|
28
|
+
@timeout = timeout
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# Header values are sent with the whitespace around them trimmed, as
|
|
32
|
+
# fetch trims them, so a credential read with a trailing newline still
|
|
33
|
+
# sends. Past the deadline before an answer, raises HTTP::TimeoutError;
|
|
34
|
+
# past it while the body is still arriving, returns the answer with an
|
|
35
|
+
# empty body, as the SDK's channels treat a body they could not read.
|
|
17
36
|
def post(url, body, headers)
|
|
18
37
|
uri = URI(url)
|
|
38
|
+
deadline = HTTP.monotonic + @timeout
|
|
19
39
|
request = Net::HTTP::Post.new(uri.request_uri)
|
|
20
|
-
headers.each { |k, v| request[k] = v }
|
|
40
|
+
headers.each { |k, v| request[k] = HTTP.trim_header(v) }
|
|
21
41
|
request.body = body
|
|
22
|
-
|
|
23
|
-
|
|
42
|
+
http = connection(uri)
|
|
43
|
+
http.open_timeout = HTTP.remaining(deadline)
|
|
44
|
+
http.start do |conn|
|
|
45
|
+
remaining = HTTP.remaining(deadline)
|
|
46
|
+
conn.read_timeout = remaining
|
|
47
|
+
conn.write_timeout = remaining
|
|
48
|
+
conn.request(request) do |response|
|
|
49
|
+
return Response.new(status: response.code.to_i, body: read_body(conn, response, deadline))
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
rescue Net::OpenTimeout, Net::ReadTimeout, Net::WriteTimeout
|
|
53
|
+
raise TimeoutError
|
|
24
54
|
end
|
|
25
55
|
|
|
26
56
|
# A connection that gives up after ten seconds, as the SDK's requests do.
|
|
27
57
|
def connection(uri)
|
|
28
58
|
http = Net::HTTP.new(uri.host, uri.port)
|
|
29
59
|
http.use_ssl = uri.scheme == "https"
|
|
30
|
-
http.open_timeout =
|
|
31
|
-
http.read_timeout =
|
|
32
|
-
http.write_timeout =
|
|
60
|
+
http.open_timeout = @timeout
|
|
61
|
+
http.read_timeout = @timeout
|
|
62
|
+
http.write_timeout = @timeout
|
|
33
63
|
http
|
|
34
64
|
end
|
|
65
|
+
|
|
66
|
+
private
|
|
67
|
+
|
|
68
|
+
# The body, read in chunks until the deadline; "" when it passes first.
|
|
69
|
+
def read_body(conn, response, deadline)
|
|
70
|
+
chunks = []
|
|
71
|
+
conn.read_timeout = HTTP.remaining(deadline)
|
|
72
|
+
response.read_body do |chunk|
|
|
73
|
+
raise TimeoutError if HTTP.monotonic >= deadline
|
|
74
|
+
|
|
75
|
+
chunks << chunk
|
|
76
|
+
conn.read_timeout = HTTP.remaining(deadline)
|
|
77
|
+
end
|
|
78
|
+
chunks.join.b
|
|
79
|
+
rescue Net::ReadTimeout, TimeoutError
|
|
80
|
+
""
|
|
81
|
+
end
|
|
35
82
|
end
|
|
36
83
|
|
|
37
84
|
module_function
|
|
@@ -40,6 +87,20 @@ module Cronwatch
|
|
|
40
87
|
@default ||= NetHTTP.new
|
|
41
88
|
end
|
|
42
89
|
|
|
90
|
+
def monotonic
|
|
91
|
+
Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
# Seconds left before `deadline`, never quite zero (Net::HTTP reads 0 as no wait).
|
|
95
|
+
def remaining(deadline)
|
|
96
|
+
[deadline - monotonic, 0.001].max
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
# A header value without the spaces, tabs and line breaks around it, as fetch sends it.
|
|
100
|
+
def trim_header(value)
|
|
101
|
+
value.to_s.gsub(/\A[ \t\r\n]+|[ \t\r\n]+\z/, "")
|
|
102
|
+
end
|
|
103
|
+
|
|
43
104
|
# Compares two secrets without stopping at the first differing character.
|
|
44
105
|
def constant_time_equal?(a, b)
|
|
45
106
|
return false unless a.bytesize == b.bytesize
|
data/lib/cronwatch/job.rb
CHANGED
|
@@ -75,6 +75,23 @@ module Cronwatch
|
|
|
75
75
|
|
|
76
76
|
outcome.result
|
|
77
77
|
end
|
|
78
|
+
|
|
79
|
+
# Record a running run now and finish it later, perhaps from another
|
|
80
|
+
# process (see #resume). Returns a RunHandle. `trigger` defaults to
|
|
81
|
+
# "start". `id` is your own stable id for the run, 1 to 200 characters,
|
|
82
|
+
# such as a queue's message id: a start with an id already recorded for
|
|
83
|
+
# this job records nothing and returns a handle on that run instead.
|
|
84
|
+
# Store failures go to on_error; it never raises for them. A run that is
|
|
85
|
+
# never finished is marked stuck by the first check after the job's timeout.
|
|
86
|
+
def start(trigger: nil, id: nil)
|
|
87
|
+
@client.start_run(@definition, trigger: trigger, id: id)
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
# A RunHandle on a run this job started elsewhere, by its id, so this
|
|
91
|
+
# process can log to it and finish it. Raises only for a run of another job.
|
|
92
|
+
def resume(run_id)
|
|
93
|
+
@client.resume_handle(@definition, run_id)
|
|
94
|
+
end
|
|
78
95
|
end
|
|
79
96
|
|
|
80
97
|
# Collects a run's output and metrics while its block runs.
|
|
@@ -0,0 +1,524 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# The pg_cron reader. Needs no gem of its own: it queries through the
|
|
4
|
+
# connection it is given (ActiveRecord, the pg gem, or your own).
|
|
5
|
+
#
|
|
6
|
+
# require "cronwatch/pg_cron"
|
|
7
|
+
require "set"
|
|
8
|
+
require "time"
|
|
9
|
+
require "cronwatch" unless defined?(Cronwatch::Client)
|
|
10
|
+
|
|
11
|
+
module Cronwatch
|
|
12
|
+
# Where runs this process does not wrap come from. A source is anything
|
|
13
|
+
# with `name` and `sync(host)`: on every check the client calls sync with
|
|
14
|
+
# itself as the host, and the source declares jobs (`host.job`), reads the
|
|
15
|
+
# store (`host.store`), records the runs it found (`host.record_run`) and
|
|
16
|
+
# reports problems (`host.on_error(error, where)`). sync returns the alerts
|
|
17
|
+
# recording sent, or anything else for none.
|
|
18
|
+
module Sources
|
|
19
|
+
# Watches pg_cron jobs, which run inside Postgres where nothing can wrap
|
|
20
|
+
# them (the SDK's sources/pgcron.ts). As a source, on every check it
|
|
21
|
+
# reads cron.job and declares each job with its schedule, then copies new
|
|
22
|
+
# rows of cron.job_run_details in as runs (ids "pgcron:<runid>"), so the
|
|
23
|
+
# usual evaluation raises missed, failed, stuck and slow alerts.
|
|
24
|
+
#
|
|
25
|
+
# A job that is renamed, unscheduled or no longer picked keeps its old
|
|
26
|
+
# name's runs and history, and that name is declared again without a
|
|
27
|
+
# schedule, so it is never reported missed. Its description says why.
|
|
28
|
+
#
|
|
29
|
+
# require "cronwatch/pg_cron"
|
|
30
|
+
# Cronwatch.configure do |c|
|
|
31
|
+
# c.store = Cronwatch::Stores::ActiveRecord.new
|
|
32
|
+
# c.sources = [Cronwatch::Sources::PgCron.new(ActiveRecord::Base)]
|
|
33
|
+
# end
|
|
34
|
+
#
|
|
35
|
+
# `db` is an ActiveRecord class, connection pool or connection
|
|
36
|
+
# (queried through exec_query), a PG::Connection (exec_params), or
|
|
37
|
+
# anything with `query(sql, params)` returning rows as hashes with string
|
|
38
|
+
# keys.
|
|
39
|
+
#
|
|
40
|
+
# jobs: which jobs to watch: names or ids, or a callable that picks them (given a Job). Default every job the role can see.
|
|
41
|
+
# prefix: put before every job name, to keep them apart from your own ("db:"). Also keeps run ids apart.
|
|
42
|
+
# job_name: a callable giving the CronWatch name for a Job. Default its jobname with anything other than
|
|
43
|
+
# letters, digits, ".", "_", ":" and "-" turned into "-", or "pg_cron:<jobid>" when it has none.
|
|
44
|
+
# The prefix goes in front either way.
|
|
45
|
+
# options: grace, timeout, max_duration, expect and the rest, for every job (a hash) or per job (a callable
|
|
46
|
+
# given a Job). The schedule and timezone always come from pg_cron.
|
|
47
|
+
# timezone: the timezone pg_cron reads its cron expressions in. Default the server's cron.timezone, read from
|
|
48
|
+
# pg_settings, which shows it only to roles with pg_read_all_settings; UTC (pg_cron's default) is
|
|
49
|
+
# assumed when it cannot be read.
|
|
50
|
+
class PgCron
|
|
51
|
+
# A row of cron.job.
|
|
52
|
+
Job = Struct.new(:jobid, :jobname, :schedule, :database, :username, :active, keyword_init: true)
|
|
53
|
+
|
|
54
|
+
# How many of a job's newest runs are copied, without alerting, the first time it is seen.
|
|
55
|
+
BACKFILL = 20
|
|
56
|
+
# Run details read per query, and the most pages read in one sync.
|
|
57
|
+
PAGE = 500
|
|
58
|
+
MAX_PAGES = 10
|
|
59
|
+
# How long a run pg_cron has queued but not started (no start_time yet)
|
|
60
|
+
# is waited for. After that it is copied as running from when it was
|
|
61
|
+
# first seen, so a run that never starts is marked stuck like any other.
|
|
62
|
+
HOLD_MS = 10 * 60_000
|
|
63
|
+
|
|
64
|
+
JOBS_SQL = "SELECT jobid, jobname, schedule, database, username, active FROM cron.job ORDER BY jobid"
|
|
65
|
+
# pg_settings has no row for a setting the role may not read, where
|
|
66
|
+
# current_setting() raises an error that would abort the caller's transaction.
|
|
67
|
+
SETTING_SQL = "SELECT setting FROM pg_settings WHERE name = $1"
|
|
68
|
+
COLUMNS = "d.runid, d.jobid, d.status, d.return_message, d.start_time, d.end_time"
|
|
69
|
+
# Every tracked job's runs after its cursor, and any run still open here, whatever its job.
|
|
70
|
+
RUNS_SQL = <<~SQL.chomp
|
|
71
|
+
SELECT #{COLUMNS}
|
|
72
|
+
FROM cron.job_run_details d
|
|
73
|
+
LEFT JOIN unnest($1::bigint[], $2::bigint[]) AS c(jobid, after) ON d.jobid = c.jobid
|
|
74
|
+
WHERE d.runid > c.after OR d.runid = ANY($3::bigint[])
|
|
75
|
+
ORDER BY d.runid LIMIT #{PAGE}
|
|
76
|
+
SQL
|
|
77
|
+
NEWEST_SQL = "SELECT #{COLUMNS} FROM cron.job_run_details d WHERE d.jobid = $1 ORDER BY d.runid DESC LIMIT #{BACKFILL}"
|
|
78
|
+
|
|
79
|
+
SECONDS = Regexp.new("\\A(\\d+)[#{JS::WHITESPACE}]*seconds?\\z", Regexp::IGNORECASE)
|
|
80
|
+
REBOOT = /\A@reboot\z/i
|
|
81
|
+
UTC = /\A(gmt|utc|z)\z/i
|
|
82
|
+
SCHEDULE_ONLY = %i[schedule timezone].freeze
|
|
83
|
+
# The options of a definition that are declared again, without its schedule, for a name no longer in use.
|
|
84
|
+
UNSCHEDULED = %i[description tags grace timeout max_duration budget failures_before_alert].freeze
|
|
85
|
+
DESCRIBED = /\Apg_cron job (\d+) in /
|
|
86
|
+
|
|
87
|
+
attr_reader :name
|
|
88
|
+
|
|
89
|
+
def initialize(db, jobs: nil, prefix: "", job_name: nil, options: nil, timezone: nil)
|
|
90
|
+
@db = PgCron.adapter(db)
|
|
91
|
+
@jobs = jobs
|
|
92
|
+
@prefix = prefix.to_s
|
|
93
|
+
@id_prefix = "pgcron:#{@prefix}"
|
|
94
|
+
@job_name = job_name
|
|
95
|
+
@options = options
|
|
96
|
+
@timezone = timezone
|
|
97
|
+
# The newest runid read for each jobid, once known.
|
|
98
|
+
@cursors = {}
|
|
99
|
+
# The start of the newest run copied for each jobid: where a restart row with no times is put.
|
|
100
|
+
@last_at = {}
|
|
101
|
+
# Runs copied while still going, by runid, with their job: read again until they finish, even once a check marks them timeout.
|
|
102
|
+
@pending = {}
|
|
103
|
+
# Runs read before they started, by runid, with when they were first seen.
|
|
104
|
+
@held = {}
|
|
105
|
+
# Each job's name and definition as last declared, by jobid.
|
|
106
|
+
@known = {}
|
|
107
|
+
# The last definition declared for each name, so an unchanged job is not declared again.
|
|
108
|
+
@declared = {}
|
|
109
|
+
# Names declared again without a schedule by retire, whose open runs are still read.
|
|
110
|
+
@retired = Set.new
|
|
111
|
+
@scanned = false
|
|
112
|
+
@warned = Set.new
|
|
113
|
+
@name = "pg_cron"
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# pg_cron takes a cron expression, with "$" for the last day of the
|
|
117
|
+
# month, or "N seconds" for 1 to 59 seconds. Returns the CronWatch
|
|
118
|
+
# schedule, or nil for one that has no cadence to watch. pg_cron reads
|
|
119
|
+
# only the first five fields of an expression and ignores the rest, so
|
|
120
|
+
# only those are kept (a sixth would otherwise be read as seconds).
|
|
121
|
+
def self.schedule(schedule)
|
|
122
|
+
text = JS.trim(schedule.to_s)
|
|
123
|
+
seconds = SECONDS.match(text)
|
|
124
|
+
return "every #{seconds[1].to_i}s" if seconds
|
|
125
|
+
return nil if REBOOT.match?(text)
|
|
126
|
+
|
|
127
|
+
fields = text.split(JS::SPACES)
|
|
128
|
+
fields = fields.first(5) if fields.length > 5 && !fields[0].start_with?("@")
|
|
129
|
+
fields[2] = fields[2].tr("$", "L") if fields.length == 5 && fields[2].include?("$")
|
|
130
|
+
fields.join(" ")
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
# The default CronWatch name for a pg_cron job, before the prefix.
|
|
134
|
+
def self.job_name(job)
|
|
135
|
+
cleaned = job.jobname.to_s.gsub(/[^A-Za-z0-9._:-]+/, "-").sub(/\A[^A-Za-z0-9]+/, "")[0, 100]
|
|
136
|
+
cleaned.empty? ? "pg_cron:#{job.jobid}" : cleaned
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
# Whether a row's status says the run is over.
|
|
140
|
+
def self.finished_status?(status)
|
|
141
|
+
%w[succeeded failed].include?(status.to_s)
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
# A row of cron.job_run_details as a CronWatch run, or nil for one that
|
|
145
|
+
# has not started (no start_time, not finished). A finished row with no
|
|
146
|
+
# start_time (pg_cron writes these for runs a server restart cut off,
|
|
147
|
+
# "server restarted") starts at its end_time, else at `fallback_at`
|
|
148
|
+
# (the reader passes the job's newest run's start, or now).
|
|
149
|
+
def self.run(row, job, id_prefix, fallback_at = nil)
|
|
150
|
+
finished_at = row["end_time"].nil? ? nil : epoch_ms(row["end_time"])
|
|
151
|
+
done = finished_status?(row["status"])
|
|
152
|
+
return nil if row["start_time"].nil? && !done
|
|
153
|
+
|
|
154
|
+
started_at =
|
|
155
|
+
if row["start_time"].nil? then finished_at || fallback_at || Process.clock_gettime(Process::CLOCK_REALTIME, :millisecond)
|
|
156
|
+
else epoch_ms(row["start_time"])
|
|
157
|
+
end
|
|
158
|
+
message = row["return_message"].nil? ? nil : JS.trim(Output.utf8(row["return_message"]))
|
|
159
|
+
message = nil if message == ""
|
|
160
|
+
status = { "succeeded" => :ok, "failed" => :failed }.fetch(row["status"].to_s, :running)
|
|
161
|
+
finish = done ? [started_at, finished_at || started_at].max : nil
|
|
162
|
+
Run.new(
|
|
163
|
+
id: "#{id_prefix}#{row["runid"]}", job: job, status: status, started_at: started_at, finished_at: finish,
|
|
164
|
+
duration_ms: finish.nil? ? nil : finish - started_at,
|
|
165
|
+
error: status == :failed ? (message || "pg_cron reported the run as failed") : nil,
|
|
166
|
+
output: status == :ok ? message : nil, metrics: {}, trigger: "pg_cron",
|
|
167
|
+
)
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
# A timestamp as epoch milliseconds: a Time (ActiveRecord decodes them), a string as Postgres writes one, or a number.
|
|
171
|
+
def self.epoch_ms(value)
|
|
172
|
+
case value
|
|
173
|
+
when Integer then value
|
|
174
|
+
when Time then (value.to_r * 1000).floor
|
|
175
|
+
when String then (Time.parse(value).to_r * 1000).floor
|
|
176
|
+
else value.respond_to?(:to_time) ? (value.to_time.to_r * 1000).floor : Integer(value)
|
|
177
|
+
end
|
|
178
|
+
end
|
|
179
|
+
|
|
180
|
+
# A value as a query parameter: arrays as Postgres array literals, the rest as given.
|
|
181
|
+
def self.encode(value)
|
|
182
|
+
value.is_a?(Array) ? "{#{value.map { |v| Integer(v) }.join(",")}}" : value
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
def self.boolean(value)
|
|
186
|
+
[true, "t", "true", 1, "1"].include?(value)
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
# The query adapter for `db`.
|
|
190
|
+
def self.adapter(db)
|
|
191
|
+
if db.respond_to?(:exec_params) then PGConnection.new(db)
|
|
192
|
+
elsif db.respond_to?(:connection_pool) || db.respond_to?(:with_connection) || db.respond_to?(:exec_query)
|
|
193
|
+
ActiveRecordConnection.new(db)
|
|
194
|
+
elsif db.respond_to?(:query) then db
|
|
195
|
+
else
|
|
196
|
+
raise ArgumentError, "Cronwatch::Sources::PgCron needs an ActiveRecord class or connection, a PG::Connection, " \
|
|
197
|
+
"or an object with query(sql, params)"
|
|
198
|
+
end
|
|
199
|
+
end
|
|
200
|
+
|
|
201
|
+
# Queries through the pg gem's PG::Connection#exec_params.
|
|
202
|
+
class PGConnection
|
|
203
|
+
def initialize(connection)
|
|
204
|
+
@connection = connection
|
|
205
|
+
@lock = Mutex.new
|
|
206
|
+
end
|
|
207
|
+
|
|
208
|
+
def query(sql, params = [])
|
|
209
|
+
@lock.synchronize { @connection.exec_params(sql, params.map { |v| PgCron.encode(v) }).to_a }
|
|
210
|
+
end
|
|
211
|
+
end
|
|
212
|
+
|
|
213
|
+
# Queries through an ActiveRecord connection's exec_query, checking one
|
|
214
|
+
# out of the pool for each query when given a class or a pool.
|
|
215
|
+
class ActiveRecordConnection
|
|
216
|
+
def initialize(source)
|
|
217
|
+
@source = source
|
|
218
|
+
end
|
|
219
|
+
|
|
220
|
+
def query(sql, params = [])
|
|
221
|
+
with_connection do |connection|
|
|
222
|
+
connection.exec_query(sql, "Cronwatch pg_cron", params.map { |v| PgCron.encode(v) }).to_a
|
|
223
|
+
end
|
|
224
|
+
end
|
|
225
|
+
|
|
226
|
+
private
|
|
227
|
+
|
|
228
|
+
# A class's pool is the writing role's, as the store's is, even
|
|
229
|
+
# inside the app's connected_to(role: :reading): cron.job_run_details
|
|
230
|
+
# is read where pg_cron writes it, not on a replica that may lag or
|
|
231
|
+
# not be configured at all.
|
|
232
|
+
def with_connection(&block)
|
|
233
|
+
if record_class?
|
|
234
|
+
::ActiveRecord::Base.connected_to(role: ::ActiveRecord.writing_role, prevent_writes: false) do
|
|
235
|
+
@source.connection_pool.with_connection(&block)
|
|
236
|
+
end
|
|
237
|
+
elsif @source.respond_to?(:connection_pool) then @source.connection_pool.with_connection(&block)
|
|
238
|
+
elsif @source.respond_to?(:with_connection) then @source.with_connection(&block)
|
|
239
|
+
else yield @source
|
|
240
|
+
end
|
|
241
|
+
end
|
|
242
|
+
|
|
243
|
+
def record_class?
|
|
244
|
+
defined?(::ActiveRecord::Base) && @source.is_a?(Class) && @source <= ::ActiveRecord::Base
|
|
245
|
+
end
|
|
246
|
+
end
|
|
247
|
+
|
|
248
|
+
# Declares the jobs and records their new runs. Returns the alerts recording them sent.
|
|
249
|
+
def sync(host)
|
|
250
|
+
now = host.now
|
|
251
|
+
timezone = @timezone
|
|
252
|
+
if timezone.nil? || timezone.to_s.empty?
|
|
253
|
+
tz = setting("cron.timezone")
|
|
254
|
+
if tz.nil?
|
|
255
|
+
warn_once(host, "tz", "could not read cron.timezone; assuming UTC. Grant pg_read_all_settings or pass " \
|
|
256
|
+
"Cronwatch::Sources::PgCron.new(db, timezone: ...).")
|
|
257
|
+
end
|
|
258
|
+
timezone = tz.nil? || UTC.match?(tz) ? "UTC" : tz
|
|
259
|
+
end
|
|
260
|
+
recording = setting("cron.log_run") != "off"
|
|
261
|
+
unless recording
|
|
262
|
+
warn_once(host, "log_run", "cron.log_run is off, so pg_cron records no runs: jobs are watched without their " \
|
|
263
|
+
"schedules and no run can fail. Turn it on to watch them.")
|
|
264
|
+
end
|
|
265
|
+
|
|
266
|
+
rows = @db.query(JOBS_SQL, [])
|
|
267
|
+
if rows.empty?
|
|
268
|
+
warn_once(host, "empty", "cron.job shows no jobs. pg_cron's row level security shows a role only the jobs it " \
|
|
269
|
+
"scheduled: connect as that role, or give this one BYPASSRLS.")
|
|
270
|
+
end
|
|
271
|
+
all = rows.map { |r| job_from(r) }
|
|
272
|
+
jobs = all.select { |job| picks?(job) }
|
|
273
|
+
names, definitions = declare(host, jobs, timezone, recording)
|
|
274
|
+
retire_unused(host, names, definitions, all)
|
|
275
|
+
return [] if !recording || names.empty?
|
|
276
|
+
|
|
277
|
+
alerts = []
|
|
278
|
+
start_cursors(host, names, alerts, now)
|
|
279
|
+
read_new(host, names, alerts, now)
|
|
280
|
+
alerts
|
|
281
|
+
end
|
|
282
|
+
|
|
283
|
+
private
|
|
284
|
+
|
|
285
|
+
def job_from(row)
|
|
286
|
+
Job.new(jobid: Integer(row["jobid"]), jobname: row["jobname"], schedule: row["schedule"].to_s,
|
|
287
|
+
database: row["database"], username: row["username"], active: PgCron.boolean(row["active"]))
|
|
288
|
+
end
|
|
289
|
+
|
|
290
|
+
def warn_once(host, key, message)
|
|
291
|
+
return if @warned.include?(key)
|
|
292
|
+
|
|
293
|
+
@warned << key
|
|
294
|
+
host.on_error(RuntimeError.new(message), "source pg_cron")
|
|
295
|
+
end
|
|
296
|
+
|
|
297
|
+
def picks?(job)
|
|
298
|
+
return true if @jobs.nil?
|
|
299
|
+
return @jobs.call(job) ? true : false if @jobs.respond_to?(:call)
|
|
300
|
+
|
|
301
|
+
Array(@jobs).any? { |j| j.is_a?(Integer) ? j == job.jobid : j.to_s == job.jobname }
|
|
302
|
+
end
|
|
303
|
+
|
|
304
|
+
def setting(name)
|
|
305
|
+
rows = @db.query(SETTING_SQL, [name])
|
|
306
|
+
value = rows.first && rows.first["setting"]
|
|
307
|
+
value&.to_s
|
|
308
|
+
rescue StandardError
|
|
309
|
+
nil
|
|
310
|
+
end
|
|
311
|
+
|
|
312
|
+
def run_id_of(id)
|
|
313
|
+
return nil unless id.start_with?(@id_prefix)
|
|
314
|
+
|
|
315
|
+
rest = id[@id_prefix.length..]
|
|
316
|
+
/\A\d{1,15}\z/.match?(rest) ? rest.to_i : nil
|
|
317
|
+
end
|
|
318
|
+
|
|
319
|
+
# Declares each job. A paused one (active = false) keeps its failures but loses its schedule, so it is not missed.
|
|
320
|
+
# Returns [{ jobid => name }, { jobid => definition as declared }].
|
|
321
|
+
def declare(host, jobs, timezone, recording)
|
|
322
|
+
names = {}
|
|
323
|
+
definitions = {}
|
|
324
|
+
used = Set.new
|
|
325
|
+
jobs.each do |job|
|
|
326
|
+
name = @prefix + (@job_name ? @job_name.call(job).to_s : PgCron.job_name(job))
|
|
327
|
+
name = "#{name}:#{job.jobid}" if used.include?(name)
|
|
328
|
+
used << name
|
|
329
|
+
extra = (@options.respond_to?(:call) ? @options.call(job) : @options) || {}
|
|
330
|
+
extra = extra.to_h.transform_keys(&:to_sym).except(*SCHEDULE_ONLY)
|
|
331
|
+
schedule = job.active && recording ? PgCron.schedule(job.schedule) : nil
|
|
332
|
+
definition = {
|
|
333
|
+
description: "pg_cron job #{job.jobid} in #{job.database} as #{job.username}#{job.active ? "" : " (paused)"}",
|
|
334
|
+
tags: ["pg_cron"],
|
|
335
|
+
}.merge(extra)
|
|
336
|
+
definition.merge!(schedule: schedule, timezone: timezone) if schedule
|
|
337
|
+
begin
|
|
338
|
+
key = definition_key(definition)
|
|
339
|
+
if @declared[name] != key
|
|
340
|
+
begin
|
|
341
|
+
host.job(name, **definition)
|
|
342
|
+
rescue StandardError => e
|
|
343
|
+
raise unless schedule
|
|
344
|
+
|
|
345
|
+
# A schedule CronWatch cannot read: watch the runs, not the cadence.
|
|
346
|
+
host.on_error(RuntimeError.new("pg_cron job #{job.jobid}: #{e.message}; watching it without a schedule"), "source pg_cron")
|
|
347
|
+
definition = definition.except(*SCHEDULE_ONLY)
|
|
348
|
+
host.job(name, **definition)
|
|
349
|
+
end
|
|
350
|
+
@declared[name] = key
|
|
351
|
+
end
|
|
352
|
+
names[job.jobid] = name
|
|
353
|
+
definitions[job.jobid] = definition
|
|
354
|
+
rescue StandardError => e
|
|
355
|
+
host.on_error(e, "source pg_cron: job #{job.jobid}")
|
|
356
|
+
end
|
|
357
|
+
end
|
|
358
|
+
[names, definitions]
|
|
359
|
+
end
|
|
360
|
+
|
|
361
|
+
# A definition as text, to tell a changed one from the last declared.
|
|
362
|
+
def definition_key(definition)
|
|
363
|
+
JS.json(definition.transform_values { |v| v.is_a?(Regexp) ? v.inspect : v.respond_to?(:call) ? "function" : v })
|
|
364
|
+
end
|
|
365
|
+
|
|
366
|
+
# The options of a stored or declared definition that can be declared again, without its schedule.
|
|
367
|
+
def unscheduled(definition)
|
|
368
|
+
UNSCHEDULED.each_with_object({}) do |field, out|
|
|
369
|
+
value = definition[field]
|
|
370
|
+
out[field] = value unless value.nil?
|
|
371
|
+
end
|
|
372
|
+
end
|
|
373
|
+
|
|
374
|
+
# Declares a name this source no longer uses for any job again, without its schedule.
|
|
375
|
+
def retire(host, name, definition, why)
|
|
376
|
+
base = unscheduled(definition)
|
|
377
|
+
following = base.merge(description: "#{base[:description] || "pg_cron job"} (#{why})")
|
|
378
|
+
host.job(name, **following)
|
|
379
|
+
@declared[name] = definition_key(following)
|
|
380
|
+
@retired << name
|
|
381
|
+
rescue StandardError => e
|
|
382
|
+
host.on_error(e, "source pg_cron: job #{name}")
|
|
383
|
+
end
|
|
384
|
+
|
|
385
|
+
# A name this source used for a job that has since been renamed, unscheduled or dropped from `jobs`
|
|
386
|
+
# is declared again without its schedule. Once per process, the same for names left scheduled in the
|
|
387
|
+
# store while no process was watching.
|
|
388
|
+
def retire_unused(host, names, definitions, all)
|
|
389
|
+
in_use = names.values.to_set
|
|
390
|
+
in_use.each { |name| @retired.delete(name) }
|
|
391
|
+
@known.each do |jobid, previous|
|
|
392
|
+
next if in_use.include?(previous[:name])
|
|
393
|
+
|
|
394
|
+
renamed = names[jobid]
|
|
395
|
+
retire(host, previous[:name], previous[:definition], renamed ? "renamed to #{renamed}" : "no longer watched")
|
|
396
|
+
end
|
|
397
|
+
@known = names.to_h { |jobid, name| [jobid, { name: name, definition: definitions[jobid] }] }
|
|
398
|
+
return if @scanned || all.empty?
|
|
399
|
+
|
|
400
|
+
@scanned = true
|
|
401
|
+
begin
|
|
402
|
+
visible = all.map(&:jobid).to_set
|
|
403
|
+
host.store.list_jobs.each do |stored|
|
|
404
|
+
definition = stored.definition
|
|
405
|
+
next if !stored.name.start_with?(@prefix) || in_use.include?(stored.name)
|
|
406
|
+
next if definition.schedule.nil? || definition.schedule.to_s.empty? || !Array(definition.tags).include?("pg_cron")
|
|
407
|
+
|
|
408
|
+
match = DESCRIBED.match(definition.description.to_s)
|
|
409
|
+
next unless match
|
|
410
|
+
|
|
411
|
+
jobid = match[1].to_i
|
|
412
|
+
current = names[jobid]
|
|
413
|
+
if !visible.include?(jobid)
|
|
414
|
+
retire(host, stored.name, definition, "no longer in cron.job")
|
|
415
|
+
elsif current && !stored.name.end_with?(current[@prefix.length..])
|
|
416
|
+
# Another pg_cron source's name for the same job ends the same way: that one is left alone.
|
|
417
|
+
retire(host, stored.name, definition, "renamed to #{current}")
|
|
418
|
+
end
|
|
419
|
+
end
|
|
420
|
+
rescue StandardError => e
|
|
421
|
+
host.on_error(e, "source pg_cron")
|
|
422
|
+
end
|
|
423
|
+
end
|
|
424
|
+
|
|
425
|
+
# Copies one detail row in as a run. A row that cannot be recorded is
|
|
426
|
+
# reported and skipped; it never stops the others.
|
|
427
|
+
def record(host, names, row, evaluate, alerts, now)
|
|
428
|
+
runid = Integer(row["runid"])
|
|
429
|
+
jobid = Integer(row["jobid"])
|
|
430
|
+
name = @pending[runid] || names[jobid]
|
|
431
|
+
unless name
|
|
432
|
+
@held.delete(runid)
|
|
433
|
+
return
|
|
434
|
+
end
|
|
435
|
+
if row["start_time"].nil? && !PgCron.finished_status?(row["status"])
|
|
436
|
+
since = @held.fetch(runid, now)
|
|
437
|
+
if now - since < HOLD_MS
|
|
438
|
+
@held[runid] = since
|
|
439
|
+
return
|
|
440
|
+
end
|
|
441
|
+
run = PgCron.run(row.merge("start_time" => since), name, @id_prefix)
|
|
442
|
+
else
|
|
443
|
+
run = PgCron.run(row, name, @id_prefix, @last_at.fetch(jobid, now))
|
|
444
|
+
end
|
|
445
|
+
@held.delete(runid)
|
|
446
|
+
return unless run
|
|
447
|
+
|
|
448
|
+
begin
|
|
449
|
+
alerts.concat(host.record_run(run, evaluate: evaluate))
|
|
450
|
+
rescue StandardError => e
|
|
451
|
+
host.on_error(e, "source pg_cron: run #{runid}")
|
|
452
|
+
return
|
|
453
|
+
end
|
|
454
|
+
if run.status == :running
|
|
455
|
+
@pending[runid] = name
|
|
456
|
+
else
|
|
457
|
+
@pending.delete(runid)
|
|
458
|
+
end
|
|
459
|
+
@last_at[jobid] = run.started_at if !@last_at.key?(jobid) || run.started_at > @last_at[jobid]
|
|
460
|
+
end
|
|
461
|
+
|
|
462
|
+
# Where each job left off. Found from the store the first time, so a restart carries on.
|
|
463
|
+
def start_cursors(host, names, alerts, now)
|
|
464
|
+
names.each do |jobid, name|
|
|
465
|
+
next if @cursors.key?(jobid)
|
|
466
|
+
|
|
467
|
+
ours = host.store.list_runs(name, BACKFILL).select { |r| run_id_of(r.id) }
|
|
468
|
+
if ours.any?
|
|
469
|
+
@cursors[jobid] = ours.map { |r| run_id_of(r.id) }.max
|
|
470
|
+
@last_at[jobid] = ours.map(&:started_at).max
|
|
471
|
+
ours.each { |r| @pending[run_id_of(r.id)] = r.job if %i[running timeout].include?(r.status) }
|
|
472
|
+
next
|
|
473
|
+
end
|
|
474
|
+
# First sight: copy recent history quietly, and judge only from the newest finished run on.
|
|
475
|
+
# The cursor goes to the newest row read, whatever is held, so history is never judged later.
|
|
476
|
+
ordered = @db.query(NEWEST_SQL, [jobid]).reverse
|
|
477
|
+
last_finished = -1
|
|
478
|
+
ordered.each_with_index { |r, i| last_finished = i if PgCron.finished_status?(r["status"]) }
|
|
479
|
+
ordered.each_with_index do |row, i|
|
|
480
|
+
# Already copied under another name (the job was renamed while no process watched): left there.
|
|
481
|
+
next if host.store.get_run("#{@id_prefix}#{row["runid"]}")
|
|
482
|
+
|
|
483
|
+
record(host, names, row, i >= last_finished, alerts, now)
|
|
484
|
+
end
|
|
485
|
+
@cursors[jobid] = ordered.empty? ? 0 : Integer(ordered.last["runid"])
|
|
486
|
+
end
|
|
487
|
+
end
|
|
488
|
+
|
|
489
|
+
# New runs, runs copied while still going (or since marked timeout), and runs not yet started.
|
|
490
|
+
def read_new(host, names, alerts, now)
|
|
491
|
+
watched = names.values.to_set | @retired
|
|
492
|
+
host.store.running_runs.each do |run|
|
|
493
|
+
id = run_id_of(run.id)
|
|
494
|
+
@pending[id] = run.job if id && watched.include?(run.job)
|
|
495
|
+
end
|
|
496
|
+
open = (@pending.keys + @held.keys).to_set
|
|
497
|
+
complete = false
|
|
498
|
+
MAX_PAGES.times do
|
|
499
|
+
jobids = names.keys
|
|
500
|
+
details = @db.query(RUNS_SQL, [jobids, jobids.map { |j| @cursors.fetch(j, 0) }, open.to_a])
|
|
501
|
+
details.each do |row|
|
|
502
|
+
jobid = Integer(row["jobid"])
|
|
503
|
+
runid = Integer(row["runid"])
|
|
504
|
+
open.delete(runid)
|
|
505
|
+
record(host, names, row, true, alerts, now)
|
|
506
|
+
# Held or not, the cursor moves on: a held run is read again by its runid.
|
|
507
|
+
@cursors[jobid] = runid if names.key?(jobid) && runid > @cursors.fetch(jobid, 0)
|
|
508
|
+
end
|
|
509
|
+
if details.length < PAGE
|
|
510
|
+
complete = true
|
|
511
|
+
break
|
|
512
|
+
end
|
|
513
|
+
end
|
|
514
|
+
# Every row was read and these were not among them: pg_cron no longer has them.
|
|
515
|
+
return unless complete
|
|
516
|
+
|
|
517
|
+
open.each do |runid|
|
|
518
|
+
@pending.delete(runid)
|
|
519
|
+
@held.delete(runid)
|
|
520
|
+
end
|
|
521
|
+
end
|
|
522
|
+
end
|
|
523
|
+
end
|
|
524
|
+
end
|