cronwatch 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/README.md +268 -0
- data/lib/cronwatch/abort_signal.rb +45 -0
- data/lib/cronwatch/active_record.rb +11 -0
- data/lib/cronwatch/alerts/console.rb +22 -0
- data/lib/cronwatch/alerts/custom.rb +26 -0
- data/lib/cronwatch/alerts/discord.rb +52 -0
- data/lib/cronwatch/alerts/slack.rb +58 -0
- data/lib/cronwatch/alerts/webhook.rb +46 -0
- data/lib/cronwatch/client.rb +925 -0
- data/lib/cronwatch/cron_pattern.rb +277 -0
- data/lib/cronwatch/duration.rb +77 -0
- data/lib/cronwatch/environment.rb +29 -0
- data/lib/cronwatch/evaluate.rb +345 -0
- data/lib/cronwatch/flight.rb +42 -0
- data/lib/cronwatch/format.rb +89 -0
- data/lib/cronwatch/http.rb +52 -0
- data/lib/cronwatch/job.rb +144 -0
- data/lib/cronwatch/js.rb +188 -0
- data/lib/cronwatch/monitored.rb +259 -0
- data/lib/cronwatch/output.rb +199 -0
- data/lib/cronwatch/rails/active_job.rb +55 -0
- data/lib/cronwatch/rails/check_job.rb +32 -0
- data/lib/cronwatch/rails/railtie.rb +37 -0
- data/lib/cronwatch/rails/tasks.rb +12 -0
- data/lib/cronwatch/rails.rb +35 -0
- data/lib/cronwatch/schedule.rb +191 -0
- data/lib/cronwatch/scheduler.rb +763 -0
- data/lib/cronwatch/serialize.rb +51 -0
- data/lib/cronwatch/sidekiq.rb +129 -0
- data/lib/cronwatch/stats.rb +23 -0
- data/lib/cronwatch/stores/active_record.rb +397 -0
- data/lib/cronwatch/stores/memory.rb +163 -0
- data/lib/cronwatch/ticker.rb +59 -0
- data/lib/cronwatch/triage/anthropic.rb +134 -0
- data/lib/cronwatch/types.rb +341 -0
- data/lib/cronwatch/version.rb +6 -0
- data/lib/cronwatch/walker.rb +137 -0
- data/lib/cronwatch/web/app.rb +484 -0
- data/lib/cronwatch/web/html.rb +314 -0
- data/lib/cronwatch/web.rb +17 -0
- data/lib/cronwatch/zone.rb +72 -0
- data/lib/cronwatch.rb +96 -0
- data/lib/generators/cronwatch/install/install_generator.rb +176 -0
- metadata +104 -0
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Cronwatch
|
|
4
|
+
module Serialize
|
|
5
|
+
module_function
|
|
6
|
+
|
|
7
|
+
# A definition as a store can hold it: `expect` becomes a description, and
|
|
8
|
+
# moves to the end, as it does in the SDK.
|
|
9
|
+
def to_stored(definition)
|
|
10
|
+
fields = definition.fields
|
|
11
|
+
# 15.minutes is stored as the milliseconds it means, the unit every
|
|
12
|
+
# reader (this gem, the SDK) takes a plain number in.
|
|
13
|
+
fields.each { |key, value| fields[key] = Duration.parse(value, key.to_s) if Duration.active_support?(value) }
|
|
14
|
+
expect = fields.delete(:expect)
|
|
15
|
+
unless expect.nil?
|
|
16
|
+
fields[:expect] =
|
|
17
|
+
case expect
|
|
18
|
+
when String then "contains #{JS.quote(expect)}"
|
|
19
|
+
when Regexp then "matches #{expect.inspect}"
|
|
20
|
+
else "custom function"
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
JobDefinition.new(fields)
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# nil when the output satisfies `expect`, or why it does not. A callable's
|
|
27
|
+
# answer is read with Ruby's truthiness.
|
|
28
|
+
def check_expectation(expect, output)
|
|
29
|
+
return nil if expect.nil?
|
|
30
|
+
|
|
31
|
+
text = output || ""
|
|
32
|
+
case expect
|
|
33
|
+
when String
|
|
34
|
+
text.include?(expect) ? nil : "Output did not contain #{JS.quote(expect)}"
|
|
35
|
+
when Regexp
|
|
36
|
+
# match? keeps no position between calls (JavaScript's /g and /y do,
|
|
37
|
+
# which is why the SDK resets lastIndex) and does not touch $~, so
|
|
38
|
+
# every run is checked from the start whatever the flags.
|
|
39
|
+
expect.match?(text) ? nil : "Output did not match #{expect.inspect}"
|
|
40
|
+
else
|
|
41
|
+
ok = false
|
|
42
|
+
begin
|
|
43
|
+
ok = expect.call(text)
|
|
44
|
+
rescue StandardError => e
|
|
45
|
+
return "Output check threw: #{e.message}"
|
|
46
|
+
end
|
|
47
|
+
ok ? nil : "Output did not pass the expect() check"
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
end
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# Sidekiq jobs that include Sidekiq::Job (or Sidekiq::Worker) directly,
|
|
4
|
+
# without ActiveJob: the Cronwatch::Sidekiq module, the server middleware
|
|
5
|
+
# that records their runs, and Cronwatch::Sidekiq::CheckWorker. Needs the
|
|
6
|
+
# sidekiq gem (7 or newer), which stays optional.
|
|
7
|
+
#
|
|
8
|
+
# In a Rails app with Sidekiq in the Gemfile this loads with the Rails
|
|
9
|
+
# integration, and the Railtie adds the middleware to Sidekiq's server.
|
|
10
|
+
# Elsewhere, require it and add the middleware yourself:
|
|
11
|
+
#
|
|
12
|
+
# require "cronwatch/sidekiq"
|
|
13
|
+
#
|
|
14
|
+
# Sidekiq.configure_server do |config|
|
|
15
|
+
# config.server_middleware { |chain| chain.add Cronwatch::Sidekiq::ServerMiddleware }
|
|
16
|
+
# config.on(:startup) { Cronwatch::Sidekiq.ready! } # after Cronwatch.configure, with the jobs loaded
|
|
17
|
+
# end
|
|
18
|
+
begin
|
|
19
|
+
require "sidekiq"
|
|
20
|
+
rescue LoadError => e
|
|
21
|
+
raise LoadError, "cronwatch/sidekiq needs the sidekiq gem (7 or newer): add `gem \"sidekiq\"` to your Gemfile (#{e.message})"
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
require "cronwatch" unless defined?(Cronwatch::Client)
|
|
25
|
+
require_relative "scheduler"
|
|
26
|
+
|
|
27
|
+
module Cronwatch
|
|
28
|
+
# Monitors a Sidekiq job class. Each perform run by a Sidekiq server with
|
|
29
|
+
# Cronwatch::Sidekiq::ServerMiddleware is a recorded run with the trigger
|
|
30
|
+
# "sidekiq"; `cronwatch` in the job is the run's context, for log and
|
|
31
|
+
# metric. A perform that raises is recorded as failed and then raises as
|
|
32
|
+
# before, so Sidekiq's retries, death handlers and error handlers see it
|
|
33
|
+
# unchanged.
|
|
34
|
+
#
|
|
35
|
+
# class NightlyReportJob
|
|
36
|
+
# include Sidekiq::Job
|
|
37
|
+
# include Cronwatch::Sidekiq
|
|
38
|
+
# cronwatch schedule: "0 2 * * *", grace: "15m" # name: "nightly-report"
|
|
39
|
+
#
|
|
40
|
+
# def perform
|
|
41
|
+
# cronwatch.log("Report written")
|
|
42
|
+
# end
|
|
43
|
+
# end
|
|
44
|
+
#
|
|
45
|
+
# The name, the options and when the job is declared are as for
|
|
46
|
+
# Cronwatch::ActiveJob, including schedule: :from_scheduler.
|
|
47
|
+
module Sidekiq
|
|
48
|
+
include Cronwatch::Monitored
|
|
49
|
+
|
|
50
|
+
TRIGGER = "sidekiq"
|
|
51
|
+
|
|
52
|
+
def self.included(base)
|
|
53
|
+
if defined?(::ActiveJob::Base) && base.is_a?(Class) && base < ::ActiveJob::Base
|
|
54
|
+
raise ArgumentError, "cronwatch: #{base.name || "this class"} is an ActiveJob class; include Cronwatch::ActiveJob instead"
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
super
|
|
58
|
+
base.extend(Cronwatch::Monitored::ClassMethods)
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
class << self
|
|
62
|
+
# Marks the app as booted and declares every monitored class's job on
|
|
63
|
+
# Cronwatch.client, so a check knows about jobs that have not run yet.
|
|
64
|
+
# The Railtie does this in Rails; call it outside Rails once
|
|
65
|
+
# Cronwatch.configure has run and the job classes are loaded.
|
|
66
|
+
def ready!
|
|
67
|
+
Cronwatch::Monitored.boot!
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# Adds ServerMiddleware to the server's chain, when this process is a
|
|
71
|
+
# Sidekiq server. The Railtie calls it; adding it twice is harmless.
|
|
72
|
+
def install
|
|
73
|
+
::Sidekiq.configure_server do |config|
|
|
74
|
+
config.server_middleware do |chain|
|
|
75
|
+
chain.add ServerMiddleware unless chain.exists?(ServerMiddleware)
|
|
76
|
+
end
|
|
77
|
+
end
|
|
78
|
+
nil
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
# The declaration whose job a Sidekiq job instance runs as: its class's
|
|
82
|
+
# own `cronwatch`, or its entry in the scheduler's config when
|
|
83
|
+
# Cronwatch.declare_from_scheduler! declared it. Nil for anything else,
|
|
84
|
+
# including ActiveJob's wrapper, which Cronwatch::ActiveJob records.
|
|
85
|
+
def declaration_for(job, payload)
|
|
86
|
+
return nil if payload.is_a?(Hash) && payload["wrapped"]
|
|
87
|
+
|
|
88
|
+
klass = job.class
|
|
89
|
+
return klass.cronwatch_declaration if klass.respond_to?(:cronwatch_declaration) && klass.cronwatch_declaration
|
|
90
|
+
|
|
91
|
+
Cronwatch::Scheduler.declaration_for_class(klass.name)
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
# Records the runs of monitored Sidekiq jobs. Anything else passes
|
|
96
|
+
# through untouched.
|
|
97
|
+
class ServerMiddleware
|
|
98
|
+
include ::Sidekiq::ServerMiddleware if defined?(::Sidekiq::ServerMiddleware)
|
|
99
|
+
|
|
100
|
+
def call(job, payload, _queue, &block)
|
|
101
|
+
declaration = Cronwatch::Sidekiq.declaration_for(job, payload)
|
|
102
|
+
return yield unless declaration
|
|
103
|
+
|
|
104
|
+
Cronwatch::Monitored.record(declaration, TRIGGER, job, &block)
|
|
105
|
+
end
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
# Cronwatch::CheckJob for apps whose Sidekiq jobs do not go through
|
|
109
|
+
# ActiveJob: declares every monitored job and runs a check. Schedule it
|
|
110
|
+
# every five minutes with sidekiq-cron:
|
|
111
|
+
#
|
|
112
|
+
# cronwatch_check:
|
|
113
|
+
# cron: "*/5 * * * *"
|
|
114
|
+
# class: "Cronwatch::Sidekiq::CheckWorker"
|
|
115
|
+
class CheckWorker
|
|
116
|
+
include ::Sidekiq::Job
|
|
117
|
+
|
|
118
|
+
# A check that fails is repeated by the next one five minutes later.
|
|
119
|
+
sidekiq_options retry: false
|
|
120
|
+
|
|
121
|
+
def perform
|
|
122
|
+
Cronwatch::Monitored.load_app_jobs
|
|
123
|
+
Cronwatch::Monitored.register_all(strict: false)
|
|
124
|
+
Cronwatch.client.check
|
|
125
|
+
nil
|
|
126
|
+
end
|
|
127
|
+
end
|
|
128
|
+
end
|
|
129
|
+
end
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Cronwatch
|
|
4
|
+
module Stats
|
|
5
|
+
module_function
|
|
6
|
+
|
|
7
|
+
def percentile(values, p)
|
|
8
|
+
return nil if values.empty?
|
|
9
|
+
|
|
10
|
+
sorted = values.sort
|
|
11
|
+
index = [sorted.length - 1, [0, ((p / 100.0) * sorted.length).ceil - 1].max].min
|
|
12
|
+
sorted[index]
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
def median(values)
|
|
16
|
+
return nil if values.empty?
|
|
17
|
+
|
|
18
|
+
sorted = values.sort
|
|
19
|
+
mid = sorted.length / 2
|
|
20
|
+
sorted.length.even? ? (sorted[mid - 1] + sorted[mid]) / 2.0 : sorted[mid]
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
end
|
|
@@ -0,0 +1,397 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
begin
|
|
4
|
+
require "active_record"
|
|
5
|
+
rescue LoadError => e
|
|
6
|
+
raise LoadError, "cronwatch/active_record needs the activerecord gem: add `gem \"activerecord\"` to your Gemfile " \
|
|
7
|
+
"(Rails apps already have it) (#{e.message})"
|
|
8
|
+
end
|
|
9
|
+
|
|
10
|
+
module Cronwatch
|
|
11
|
+
module Stores
|
|
12
|
+
# Keeps jobs, runs and state in the app's database through ActiveRecord.
|
|
13
|
+
# The same three tables, columns, indexes and JSON as the SDK's SQLite and
|
|
14
|
+
# Postgres stores (packages/sdk/src/stores/sql.ts), so a Node process and a
|
|
15
|
+
# Ruby process can share one database. Postgres and SQLite are supported
|
|
16
|
+
# and tested; any other adapter (MySQL among them) is refused.
|
|
17
|
+
#
|
|
18
|
+
# Cronwatch.configure do |c|
|
|
19
|
+
# c.store = Cronwatch::Stores::ActiveRecord.new
|
|
20
|
+
# end
|
|
21
|
+
#
|
|
22
|
+
# It never creates tables on its own: in Rails the install generator's
|
|
23
|
+
# migration does, and elsewhere call ActiveRecord.create_tables!. Every
|
|
24
|
+
# call checks a connection out for just that call, so checks and runs in
|
|
25
|
+
# other threads are fine, and always on the writing role, so it works
|
|
26
|
+
# inside connected_to(role: :reading).
|
|
27
|
+
#
|
|
28
|
+
# On Postgres the store connects through a pool of its own (an abstract
|
|
29
|
+
# class under this one, with the connection_class's writing database
|
|
30
|
+
# config), so its writes never join a transaction the app has open: a run
|
|
31
|
+
# recorded inside one is kept when it rolls back, and a check waiting on a
|
|
32
|
+
# job's row cannot deadlock with it. That pool opens up to the config's
|
|
33
|
+
# `pool` connections per process, beside the app's. SQLite allows one
|
|
34
|
+
# writer at a time, so there the store uses the app's pool, and inside an
|
|
35
|
+
# open transaction each call runs in a savepoint of it.
|
|
36
|
+
class ActiveRecord
|
|
37
|
+
DEFAULT_PREFIX = "cronwatch_"
|
|
38
|
+
# Postgres truncates identifiers past 63 bytes; the longest name built is the prefix plus "runs_job_started".
|
|
39
|
+
MAX_PREFIX = 63 - "runs_job_started".length
|
|
40
|
+
NAME = "Cronwatch"
|
|
41
|
+
|
|
42
|
+
# Raised by init when the tables are not there.
|
|
43
|
+
class MissingTables < StandardError; end
|
|
44
|
+
|
|
45
|
+
# Raised for a database the store does not write: anything but Postgres and SQLite.
|
|
46
|
+
class UnsupportedAdapter < ArgumentError; end
|
|
47
|
+
|
|
48
|
+
# The abstract classes whose pools give the store its own Postgres
|
|
49
|
+
# connections, one per database config.
|
|
50
|
+
@pools = {}
|
|
51
|
+
@pools_lock = Mutex.new
|
|
52
|
+
|
|
53
|
+
attr_reader :prefix
|
|
54
|
+
|
|
55
|
+
# prefix: table name prefix, a plain lowercase identifier. Default "cronwatch_".
|
|
56
|
+
# connection_class: the ActiveRecord class whose connection pool to use (or its name, looked
|
|
57
|
+
# up on first use). Default ActiveRecord::Base.
|
|
58
|
+
def initialize(prefix: DEFAULT_PREFIX, connection_class: nil)
|
|
59
|
+
@prefix = self.class.table_prefix(prefix)
|
|
60
|
+
@connection_class = connection_class
|
|
61
|
+
@statements = {}
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# Table names are built from the prefix, so it must be a plain lowercase
|
|
65
|
+
# identifier. Uppercase is refused rather than folded: Postgres lowercases
|
|
66
|
+
# unquoted names, so "Monitoring_" would quietly become "monitoring_".
|
|
67
|
+
def self.table_prefix(prefix = DEFAULT_PREFIX)
|
|
68
|
+
unless prefix.is_a?(String) && /\A[a-z_][a-z0-9_]*\z/.match?(prefix) && prefix.length <= MAX_PREFIX
|
|
69
|
+
raise ArgumentError,
|
|
70
|
+
"cronwatch: invalid table prefix #{JS.json(prefix.to_s)}. Use lowercase letters, digits and underscores, " \
|
|
71
|
+
"not starting with a digit, at most #{MAX_PREFIX} characters."
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
prefix
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# :postgres or :sqlite, from a connection or an adapter's name. Raises
|
|
78
|
+
# UnsupportedAdapter for anything else.
|
|
79
|
+
def self.dialect(connection)
|
|
80
|
+
adapter = connection.respond_to?(:adapter_name) ? connection.adapter_name : connection.to_s
|
|
81
|
+
return :postgres if /postg/i.match?(adapter)
|
|
82
|
+
return :sqlite if /sqlite/i.match?(adapter)
|
|
83
|
+
|
|
84
|
+
raise UnsupportedAdapter,
|
|
85
|
+
"cronwatch: the ActiveRecord store supports PostgreSQL and SQLite, not #{adapter}. MySQL is not " \
|
|
86
|
+
"supported yet: the SDK's statements (ON CONFLICT, TEXT primary keys) do not run on it."
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# The abstract class, named under this one, whose pool is the store's
|
|
90
|
+
# own for `db_config`: created once per config, on first use.
|
|
91
|
+
def self.own_connection_class(db_config)
|
|
92
|
+
@pools_lock.synchronize do
|
|
93
|
+
@pools[[db_config.env_name, db_config.name, db_config.configuration_hash]] ||= begin
|
|
94
|
+
klass = Class.new(::ActiveRecord::Base) { self.abstract_class = true }
|
|
95
|
+
# Named before it connects: ActiveRecord keys the pool by the class's name.
|
|
96
|
+
const_set("Connection#{@pools.length + 1}", klass)
|
|
97
|
+
config = ::ActiveRecord::DatabaseConfigurations::HashConfig.new(
|
|
98
|
+
db_config.env_name, "#{db_config.name}_cronwatch", db_config.configuration_hash,
|
|
99
|
+
)
|
|
100
|
+
::ActiveRecord::Base.connected_to(role: ::ActiveRecord.writing_role) { klass.establish_connection(config) }
|
|
101
|
+
klass
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# The SDK's schema, character for character (sql.ts `schema`).
|
|
107
|
+
def self.schema(dialect, prefix = DEFAULT_PREFIX)
|
|
108
|
+
p = table_prefix(prefix)
|
|
109
|
+
pg = dialect.to_sym == :postgres
|
|
110
|
+
int = pg ? "BIGINT" : "INTEGER"
|
|
111
|
+
json = pg ? "JSONB" : "TEXT"
|
|
112
|
+
text = <<-SQL
|
|
113
|
+
CREATE TABLE IF NOT EXISTS #{p}jobs (
|
|
114
|
+
name TEXT PRIMARY KEY,
|
|
115
|
+
definition #{json} NOT NULL,
|
|
116
|
+
created_at #{int} NOT NULL,
|
|
117
|
+
updated_at #{int} NOT NULL
|
|
118
|
+
);
|
|
119
|
+
CREATE TABLE IF NOT EXISTS #{p}runs (#{pg ? "\n seq BIGSERIAL," : ""}
|
|
120
|
+
id TEXT PRIMARY KEY,
|
|
121
|
+
job TEXT NOT NULL,
|
|
122
|
+
status TEXT NOT NULL,
|
|
123
|
+
started_at #{int} NOT NULL,
|
|
124
|
+
finished_at #{int},
|
|
125
|
+
duration_ms #{int},
|
|
126
|
+
error TEXT,
|
|
127
|
+
output TEXT,
|
|
128
|
+
metrics #{json} NOT NULL DEFAULT '{}',
|
|
129
|
+
trigger TEXT NOT NULL DEFAULT 'run'
|
|
130
|
+
);
|
|
131
|
+
CREATE INDEX IF NOT EXISTS #{p}runs_job_started ON #{p}runs (job, started_at DESC);
|
|
132
|
+
CREATE INDEX IF NOT EXISTS #{p}runs_running ON #{p}runs (status) WHERE status = 'running';
|
|
133
|
+
CREATE TABLE IF NOT EXISTS #{p}state (
|
|
134
|
+
job TEXT PRIMARY KEY,
|
|
135
|
+
state #{json} NOT NULL
|
|
136
|
+
);
|
|
137
|
+
SQL
|
|
138
|
+
"\n#{text} "
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
# Creates the three tables if they are missing, with the SDK's DDL. For
|
|
142
|
+
# the install generator's migration and for apps outside Rails. On
|
|
143
|
+
# Postgres it holds the SDK's advisory lock for the prefix, so it takes
|
|
144
|
+
# turns with Node processes creating the same tables.
|
|
145
|
+
def self.create_tables!(connection = ::ActiveRecord::Base.connection, prefix: DEFAULT_PREFIX)
|
|
146
|
+
p = table_prefix(prefix)
|
|
147
|
+
if dialect(connection) == :postgres
|
|
148
|
+
connection.transaction(requires_new: true) do
|
|
149
|
+
connection.execute("SELECT pg_advisory_xact_lock(hashtext(#{connection.quote("cronwatch:#{p}")}))")
|
|
150
|
+
connection.execute(schema(:postgres, p))
|
|
151
|
+
end
|
|
152
|
+
else
|
|
153
|
+
# One statement at a time, each the same text the SDK hands SQLite.
|
|
154
|
+
schema(:sqlite, p).split(";").each do |statement|
|
|
155
|
+
connection.execute(statement) unless statement.strip.empty?
|
|
156
|
+
end
|
|
157
|
+
end
|
|
158
|
+
nil
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
def self.drop_tables!(connection = ::ActiveRecord::Base.connection, prefix: DEFAULT_PREFIX)
|
|
162
|
+
p = table_prefix(prefix)
|
|
163
|
+
%w[state runs jobs].each { |table| connection.execute("DROP TABLE IF EXISTS #{p}#{table}") }
|
|
164
|
+
nil
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
# Checks that the tables are there, with a clear error when they are not.
|
|
168
|
+
def init
|
|
169
|
+
with_connection do |conn|
|
|
170
|
+
missing = %w[jobs runs state].map { |t| "#{@prefix}#{t}" }.reject { |t| conn.table_exists?(t) }
|
|
171
|
+
unless missing.empty?
|
|
172
|
+
raise MissingTables,
|
|
173
|
+
"cronwatch: missing table#{missing.length == 1 ? "" : "s"} #{missing.join(", ")}. In Rails run " \
|
|
174
|
+
"`bin/rails generate cronwatch:install` and migrate; elsewhere call " \
|
|
175
|
+
"Cronwatch::Stores::ActiveRecord.create_tables!(connection#{@prefix == DEFAULT_PREFIX ? "" : ", prefix: #{@prefix.inspect}"})."
|
|
176
|
+
end
|
|
177
|
+
end
|
|
178
|
+
nil
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
def upsert_job(definition, now)
|
|
182
|
+
definition = JobDefinition.from_h(definition)
|
|
183
|
+
write(:upsert_job, [definition.name, JS.json(definition.to_h), now, now])
|
|
184
|
+
nil
|
|
185
|
+
end
|
|
186
|
+
|
|
187
|
+
def get_job(name)
|
|
188
|
+
row = read(:get_job, [name]).first
|
|
189
|
+
row && row_to_job(row)
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
# Byte order on both databases, as the SDK's SQL stores sort.
|
|
193
|
+
def list_jobs
|
|
194
|
+
read(:list_jobs, []).map { |row| row_to_job(row) }
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
def delete_job(name)
|
|
198
|
+
with_connection do |conn|
|
|
199
|
+
conn.transaction(requires_new: true) do
|
|
200
|
+
%i[delete_runs delete_state delete_job].each { |key| conn.exec_update(sql(conn, key), NAME, [name]) }
|
|
201
|
+
end
|
|
202
|
+
end
|
|
203
|
+
nil
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
def insert_run(run)
|
|
207
|
+
write(:insert_run, [run.id, run.job, run.status.to_s, run.started_at, run.finished_at, run.duration_ms,
|
|
208
|
+
run.error, run.output, JS.json(run.metrics || {}), run.trigger])
|
|
209
|
+
nil
|
|
210
|
+
end
|
|
211
|
+
|
|
212
|
+
# A run that is gone (its job was forgotten) stays gone.
|
|
213
|
+
def update_run(run)
|
|
214
|
+
write(:update_run, [run.status.to_s, run.finished_at, run.duration_ms, run.error, run.output,
|
|
215
|
+
JS.json(run.metrics || {}), run.id])
|
|
216
|
+
nil
|
|
217
|
+
end
|
|
218
|
+
|
|
219
|
+
def get_run(id)
|
|
220
|
+
row = read(:get_run, [id]).first
|
|
221
|
+
row && row_to_run(row)
|
|
222
|
+
end
|
|
223
|
+
|
|
224
|
+
# Newest first; runs that started in the same millisecond, last written first.
|
|
225
|
+
def list_runs(job, limit)
|
|
226
|
+
read(:list_runs, [job, [limit.to_i, 0].max]).map { |row| row_to_run(row) }
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
def last_run(job)
|
|
230
|
+
list_runs(job, 1).first
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
# Oldest first, then in the order they were written.
|
|
234
|
+
def running_runs
|
|
235
|
+
read(:running_runs, []).map { |row| row_to_run(row) }
|
|
236
|
+
end
|
|
237
|
+
|
|
238
|
+
def get_state(job)
|
|
239
|
+
row = read(:get_state, [job]).first
|
|
240
|
+
row && JobState.from_h(json(row["state"]))
|
|
241
|
+
end
|
|
242
|
+
|
|
243
|
+
def set_state(state)
|
|
244
|
+
write(:set_state, [state.job, JS.json(state.to_h)])
|
|
245
|
+
nil
|
|
246
|
+
end
|
|
247
|
+
|
|
248
|
+
# Writes `state` only when the stored state's version (absent, or no
|
|
249
|
+
# row at all, counts as 0) is `expected_version`, in one statement.
|
|
250
|
+
# Returns whether it wrote. This is what keeps two processes sharing the
|
|
251
|
+
# database (Ruby or Node) from overwriting each other's updates.
|
|
252
|
+
def compare_and_set_state(state, expected_version)
|
|
253
|
+
text = JS.json(state.to_h)
|
|
254
|
+
changed =
|
|
255
|
+
if expected_version.zero? then write(:cas_insert, [state.job, text])
|
|
256
|
+
else write(:cas_update, [text, state.job, expected_version])
|
|
257
|
+
end
|
|
258
|
+
changed.to_i.positive?
|
|
259
|
+
end
|
|
260
|
+
|
|
261
|
+
# Deletes finished runs that started before this time, except each job's
|
|
262
|
+
# newest run. Returns how many.
|
|
263
|
+
def prune(before)
|
|
264
|
+
write(:prune, [before])
|
|
265
|
+
end
|
|
266
|
+
|
|
267
|
+
# The pool belongs to the app; there is nothing to close.
|
|
268
|
+
def close
|
|
269
|
+
nil
|
|
270
|
+
end
|
|
271
|
+
|
|
272
|
+
private
|
|
273
|
+
|
|
274
|
+
def connection_class
|
|
275
|
+
klass = @connection_class || ::ActiveRecord::Base
|
|
276
|
+
klass.is_a?(String) ? Object.const_get(klass) : klass
|
|
277
|
+
end
|
|
278
|
+
|
|
279
|
+
# The pool a call checks a connection out of: on Postgres the store's
|
|
280
|
+
# own, on SQLite the connection_class's writing pool.
|
|
281
|
+
def pool
|
|
282
|
+
source = connection_class
|
|
283
|
+
found = source.connection_handler.retrieve_connection_pool(
|
|
284
|
+
source.connection_specification_name, role: ::ActiveRecord.writing_role, shard: source.default_shard,
|
|
285
|
+
) || source.connection_pool
|
|
286
|
+
return found if self.class.dialect(found.db_config.adapter) == :sqlite
|
|
287
|
+
|
|
288
|
+
self.class.own_connection_class(found.db_config).connection_pool
|
|
289
|
+
end
|
|
290
|
+
|
|
291
|
+
# Every call runs on the writing role, even inside the app's
|
|
292
|
+
# connected_to(role: :reading, prevent_writes: true).
|
|
293
|
+
def with_connection(&block)
|
|
294
|
+
::ActiveRecord::Base.connected_to(role: ::ActiveRecord.writing_role, prevent_writes: false) do
|
|
295
|
+
pool.with_connection do |conn|
|
|
296
|
+
# A failed statement aborts a Postgres transaction; a savepoint keeps that from reaching an open one.
|
|
297
|
+
conn.transaction_open? ? conn.transaction(requires_new: true) { block.call(conn) } : block.call(conn)
|
|
298
|
+
end
|
|
299
|
+
end
|
|
300
|
+
end
|
|
301
|
+
|
|
302
|
+
# Reads skip the query cache: a job's own run is read back right after it is written.
|
|
303
|
+
def read(key, binds)
|
|
304
|
+
with_connection { |conn| conn.uncached { conn.select_all(sql(conn, key), NAME, binds).to_a } }
|
|
305
|
+
end
|
|
306
|
+
|
|
307
|
+
def write(key, binds)
|
|
308
|
+
with_connection { |conn| conn.exec_update(sql(conn, key), NAME, binds) }
|
|
309
|
+
end
|
|
310
|
+
|
|
311
|
+
def sql(conn, key)
|
|
312
|
+
dialect = self.class.dialect(conn)
|
|
313
|
+
(@statements[dialect] ||= statements(dialect)).fetch(key)
|
|
314
|
+
end
|
|
315
|
+
|
|
316
|
+
# sql.ts `statements`: the same text, with `?` numbered for Postgres.
|
|
317
|
+
def statements(dialect)
|
|
318
|
+
p = @prefix
|
|
319
|
+
pg = dialect == :postgres
|
|
320
|
+
# Insertion order, to break ties between runs that started in the same millisecond.
|
|
321
|
+
seq = pg ? "seq" : "rowid"
|
|
322
|
+
# Byte order on both, so names sort the same whatever the database's collation.
|
|
323
|
+
by_name = pg ? 'name COLLATE "C"' : "name"
|
|
324
|
+
# The version inside a state's JSON, 0 when it has none.
|
|
325
|
+
version = ->(column) { pg ? "COALESCE((#{column}->>'version')::bigint, 0)" : "COALESCE(json_extract(#{column}, '$.version'), 0)" }
|
|
326
|
+
sql = {
|
|
327
|
+
upsert_job: "INSERT INTO #{p}jobs (name, definition, created_at, updated_at) VALUES (?, ?, ?, ?)\n " \
|
|
328
|
+
"ON CONFLICT (name) DO UPDATE SET definition = excluded.definition, updated_at = excluded.updated_at",
|
|
329
|
+
get_job: "SELECT * FROM #{p}jobs WHERE name = ?",
|
|
330
|
+
list_jobs: "SELECT * FROM #{p}jobs ORDER BY #{by_name}",
|
|
331
|
+
delete_runs: "DELETE FROM #{p}runs WHERE job = ?",
|
|
332
|
+
delete_state: "DELETE FROM #{p}state WHERE job = ?",
|
|
333
|
+
delete_job: "DELETE FROM #{p}jobs WHERE name = ?",
|
|
334
|
+
insert_run: "INSERT INTO #{p}runs (id, job, status, started_at, finished_at, duration_ms, error, output, metrics, trigger)\n " \
|
|
335
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
336
|
+
update_run: "UPDATE #{p}runs SET status = ?, finished_at = ?, duration_ms = ?, error = ?, output = ?, metrics = ? WHERE id = ?",
|
|
337
|
+
get_run: "SELECT * FROM #{p}runs WHERE id = ?",
|
|
338
|
+
list_runs: "SELECT * FROM #{p}runs WHERE job = ? ORDER BY started_at DESC, #{seq} DESC LIMIT ?",
|
|
339
|
+
running_runs: "SELECT * FROM #{p}runs WHERE status = 'running' ORDER BY started_at, #{seq}",
|
|
340
|
+
get_state: "SELECT state FROM #{p}state WHERE job = ?",
|
|
341
|
+
set_state: "INSERT INTO #{p}state (job, state) VALUES (?, ?) ON CONFLICT (job) DO UPDATE SET state = excluded.state",
|
|
342
|
+
# compare_and_set_state. Expecting version 0 also matches a missing
|
|
343
|
+
# row, so that case inserts; any other version must find its row.
|
|
344
|
+
cas_insert: "INSERT INTO #{p}state (job, state) VALUES (?, ?)\n " \
|
|
345
|
+
"ON CONFLICT (job) DO UPDATE SET state = excluded.state WHERE #{version.call("#{p}state.state")} = 0",
|
|
346
|
+
cas_update: "UPDATE #{p}state SET state = ? WHERE job = ? AND #{version.call("state")} = ?",
|
|
347
|
+
# Each job's newest run is kept whatever its age: without it, a job that
|
|
348
|
+
# runs less often than the retention looks like it never ran.
|
|
349
|
+
prune: "DELETE FROM #{p}runs WHERE status <> 'running' AND started_at < ?\n " \
|
|
350
|
+
"AND started_at < (SELECT MAX(r.started_at) FROM #{p}runs r WHERE r.job = #{p}runs.job)",
|
|
351
|
+
}
|
|
352
|
+
if pg
|
|
353
|
+
sql.transform_values! do |text|
|
|
354
|
+
n = 0
|
|
355
|
+
text.gsub("?") { "$#{n += 1}" }
|
|
356
|
+
end
|
|
357
|
+
end
|
|
358
|
+
sql.freeze
|
|
359
|
+
end
|
|
360
|
+
|
|
361
|
+
# SQLite hands back JSON as TEXT; Postgres drivers may hand back JSONB
|
|
362
|
+
# parsed or as text, and BIGINT as a string.
|
|
363
|
+
def json(value)
|
|
364
|
+
js_numbers(value.is_a?(String) ? JS.parse(value) : value)
|
|
365
|
+
end
|
|
366
|
+
|
|
367
|
+
# JSONB writes 1e21 back as 1000000000000000000000, which JavaScript
|
|
368
|
+
# reads as a double and Ruby as an Integer. Past 2**53 an Integer is
|
|
369
|
+
# turned into the Float JavaScript would hold, so Node and Ruby read the
|
|
370
|
+
# same number and write it back the same.
|
|
371
|
+
def js_numbers(value)
|
|
372
|
+
case value
|
|
373
|
+
when Integer then value.abs > JS::MAX_SAFE_INTEGER ? value.to_f : value
|
|
374
|
+
when Hash then value.transform_values! { |v| js_numbers(v) }
|
|
375
|
+
when Array then value.map! { |v| js_numbers(v) }
|
|
376
|
+
else value
|
|
377
|
+
end
|
|
378
|
+
end
|
|
379
|
+
|
|
380
|
+
def int(value)
|
|
381
|
+
value.nil? ? nil : Integer(value)
|
|
382
|
+
end
|
|
383
|
+
|
|
384
|
+
def row_to_job(row)
|
|
385
|
+
StoredJob.new(name: row["name"], definition: JobDefinition.from_h(json(row["definition"])),
|
|
386
|
+
created_at: int(row["created_at"]), updated_at: int(row["updated_at"]))
|
|
387
|
+
end
|
|
388
|
+
|
|
389
|
+
def row_to_run(row)
|
|
390
|
+
metrics = row["metrics"].nil? ? {} : json(row["metrics"])
|
|
391
|
+
Run.new(id: row["id"], job: row["job"], status: row["status"].to_sym, started_at: int(row["started_at"]),
|
|
392
|
+
finished_at: int(row["finished_at"]), duration_ms: int(row["duration_ms"]), error: row["error"],
|
|
393
|
+
output: row["output"], metrics: metrics, trigger: row["trigger"])
|
|
394
|
+
end
|
|
395
|
+
end
|
|
396
|
+
end
|
|
397
|
+
end
|