deployangel 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +33 -0
- data/LICENSE.txt +21 -0
- data/README.md +293 -0
- data/exe/deployangel +6 -0
- data/lib/deployangel/agent.rb +242 -0
- data/lib/deployangel/capistrano/steps.rb +61 -0
- data/lib/deployangel/capistrano/tasks.rake +54 -0
- data/lib/deployangel/capistrano.rb +13 -0
- data/lib/deployangel/ci_environment.rb +46 -0
- data/lib/deployangel/cli/formatter.rb +96 -0
- data/lib/deployangel/cli.rb +257 -0
- data/lib/deployangel/client.rb +92 -0
- data/lib/deployangel/configuration.rb +49 -0
- data/lib/deployangel/core/aggregator.rb +181 -0
- data/lib/deployangel/core/buffer.rb +43 -0
- data/lib/deployangel/core/fingerprint.rb +91 -0
- data/lib/deployangel/core/histogram.rb +42 -0
- data/lib/deployangel/core/instance.rb +30 -0
- data/lib/deployangel/core/protocol.rb +60 -0
- data/lib/deployangel/core/release.rb +88 -0
- data/lib/deployangel/core/transport.rb +66 -0
- data/lib/deployangel/fork_hook.rb +15 -0
- data/lib/deployangel/mcp.rb +205 -0
- data/lib/deployangel/rails/active_job.rb +70 -0
- data/lib/deployangel/rails/error_subscriber.rb +19 -0
- data/lib/deployangel/rails/http.rb +65 -0
- data/lib/deployangel/rails/metadata.rb +96 -0
- data/lib/deployangel/rails/railtie.rb +35 -0
- data/lib/deployangel/sidekiq.rb +71 -0
- data/lib/deployangel/verification_waiter.rb +97 -0
- data/lib/deployangel/version.rb +5 -0
- data/lib/deployangel.rb +118 -0
- metadata +81 -0
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "set"
|
|
4
|
+
|
|
5
|
+
module DeployAngel
|
|
6
|
+
# Accumulates request behavior into one-minute periods inside the process.
|
|
7
|
+
# Recording is a few hash updates under a mutex; nothing here touches the
|
|
8
|
+
# network.
|
|
9
|
+
class Aggregator
|
|
10
|
+
OTHER_ROUTE = "__other__"
|
|
11
|
+
PERIOD_SECONDS = 60
|
|
12
|
+
MAX_EXCEPTIONS = 20
|
|
13
|
+
MAX_BACKTRACES = 5
|
|
14
|
+
MAX_CHECKPOINTS = 100
|
|
15
|
+
|
|
16
|
+
RouteStats = Struct.new(:requests, :status_counts, :histogram) do
|
|
17
|
+
def self.empty
|
|
18
|
+
new(0, Hash.new(0), Histogram.new)
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
JobStats = Struct.new(:processed, :failed, :discarded, :duration, :queue_latency) do
|
|
23
|
+
def self.empty
|
|
24
|
+
new(0, 0, 0, Histogram.new, Histogram.new)
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def record(failed, discarded, duration_ms, queue_latency_ms)
|
|
28
|
+
self.processed += 1
|
|
29
|
+
self.failed += 1 if failed
|
|
30
|
+
self.discarded += 1 if discarded
|
|
31
|
+
duration.record(duration_ms)
|
|
32
|
+
queue_latency.record(queue_latency_ms) if queue_latency_ms
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
def record_discard
|
|
36
|
+
self.discarded += 1
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
class Period
|
|
41
|
+
attr_reader :started_at, :requests, :status_counts, :unhandled_exceptions, :histogram, :routes,
|
|
42
|
+
:jobs, :job_classes, :exceptions, :exceptions_truncated, :checkpoints
|
|
43
|
+
|
|
44
|
+
def initialize(started_at)
|
|
45
|
+
@started_at = started_at
|
|
46
|
+
@requests = 0
|
|
47
|
+
@status_counts = Hash.new(0)
|
|
48
|
+
@unhandled_exceptions = 0
|
|
49
|
+
@histogram = Histogram.new
|
|
50
|
+
@routes = {}
|
|
51
|
+
@jobs = JobStats.empty
|
|
52
|
+
@job_classes = {}
|
|
53
|
+
@exceptions = {}
|
|
54
|
+
@exceptions_truncated = 0
|
|
55
|
+
@checkpoints = Hash.new(0)
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# Up to 100 names per period; the rest are counted together.
|
|
59
|
+
def record_checkpoint(name, count)
|
|
60
|
+
key = @checkpoints.key?(name) || @checkpoints.size < MAX_CHECKPOINTS - 1 ? name : OTHER_ROUTE
|
|
61
|
+
@checkpoints[key] += count
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# Up to 20 fingerprints per period; the rest are only counted.
|
|
65
|
+
def record_exception(details, source, handled, backtrace)
|
|
66
|
+
entry = @exceptions[details["fingerprint"]]
|
|
67
|
+
unless entry
|
|
68
|
+
if @exceptions.size >= MAX_EXCEPTIONS
|
|
69
|
+
@exceptions_truncated += 1
|
|
70
|
+
return
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
entry = @exceptions[details["fingerprint"]] = details.merge("count" => 0, "handled_count" => 0, "sources" => Hash.new(0))
|
|
74
|
+
end
|
|
75
|
+
handled ? entry["handled_count"] += 1 : entry["count"] += 1
|
|
76
|
+
entry["sources"][source] += 1 if source
|
|
77
|
+
entry["backtrace"] ||= backtrace if backtrace
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def record_job(job_class, failed, discarded, duration_ms, queue_latency_ms, max_classes)
|
|
81
|
+
@jobs.record(failed, discarded, duration_ms, queue_latency_ms)
|
|
82
|
+
key = @job_classes.key?(job_class) || @job_classes.size < max_classes - 1 ? job_class : OTHER_ROUTE
|
|
83
|
+
(@job_classes[key] ||= JobStats.empty).record(failed, discarded, duration_ms, queue_latency_ms)
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
def record_discard(job_class, max_classes)
|
|
87
|
+
@jobs.record_discard
|
|
88
|
+
key = @job_classes.key?(job_class) || @job_classes.size < max_classes - 1 ? job_class : OTHER_ROUTE
|
|
89
|
+
(@job_classes[key] ||= JobStats.empty).record_discard
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
def record(route_key, status, duration_ms, unhandled, max_routes)
|
|
93
|
+
@requests += 1
|
|
94
|
+
@status_counts[status.to_s] += 1 if status >= 400
|
|
95
|
+
@unhandled_exceptions += 1 if unhandled
|
|
96
|
+
@histogram.record(duration_ms)
|
|
97
|
+
|
|
98
|
+
key = @routes.key?(route_key) || @routes.size < max_routes - 1 ? route_key : OTHER_ROUTE
|
|
99
|
+
route = (@routes[key] ||= RouteStats.empty)
|
|
100
|
+
route.requests += 1
|
|
101
|
+
route.status_counts[status.to_s] += 1 if status >= 400
|
|
102
|
+
route.histogram.record(duration_ms)
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
def initialize(max_routes: 100, clock: -> { Time.now.utc })
|
|
107
|
+
@max_routes = max_routes
|
|
108
|
+
@clock = clock
|
|
109
|
+
@periods = {}
|
|
110
|
+
@seen_fingerprints = Set.new
|
|
111
|
+
@last_drained_at = period_start(@clock.call) - PERIOD_SECONDS
|
|
112
|
+
@mutex = Mutex.new
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
def record(route_key:, status:, duration_ms:, unhandled: false)
|
|
116
|
+
started_at = period_start(@clock.call)
|
|
117
|
+
@mutex.synchronize do
|
|
118
|
+
(@periods[started_at] ||= Period.new(started_at))
|
|
119
|
+
.record(route_key, status.to_i, duration_ms, unhandled, @max_routes)
|
|
120
|
+
end
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
# failed: the attempt raised (or was retried or discarded by the job);
|
|
124
|
+
# discarded: the job will not run again.
|
|
125
|
+
def record_job(job_class:, duration_ms:, failed: false, discarded: false, queue_latency_ms: nil)
|
|
126
|
+
started_at = period_start(@clock.call)
|
|
127
|
+
@mutex.synchronize do
|
|
128
|
+
(@periods[started_at] ||= Period.new(started_at))
|
|
129
|
+
.record_job(job_class.to_s, failed, discarded, duration_ms, queue_latency_ms, @max_routes)
|
|
130
|
+
end
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
# A job that will not run again, recorded without another attempt (for
|
|
134
|
+
# example a Sidekiq job that exhausted its retries).
|
|
135
|
+
def record_discard(job_class:)
|
|
136
|
+
started_at = period_start(@clock.call)
|
|
137
|
+
@mutex.synchronize { (@periods[started_at] ||= Period.new(started_at)).record_discard(job_class.to_s, @max_routes) }
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
def record_checkpoint(name:, count: 1)
|
|
141
|
+
started_at = period_start(@clock.call)
|
|
142
|
+
@mutex.synchronize { (@periods[started_at] ||= Period.new(started_at)).record_checkpoint(name, count) }
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
# A representative backtrace is kept only the first time this process
|
|
146
|
+
# sees a fingerprint, and at most 5 per period.
|
|
147
|
+
def record_exception(details, source: nil, handled: false, backtrace: nil)
|
|
148
|
+
started_at = period_start(@clock.call)
|
|
149
|
+
@mutex.synchronize do
|
|
150
|
+
period = (@periods[started_at] ||= Period.new(started_at))
|
|
151
|
+
new_here = @seen_fingerprints.add?(details["fingerprint"])
|
|
152
|
+
keep_trace = new_here && period.exceptions.count { |_, e| e["backtrace"] } < MAX_BACKTRACES
|
|
153
|
+
period.record_exception(details, source, handled, keep_trace ? backtrace : nil)
|
|
154
|
+
end
|
|
155
|
+
end
|
|
156
|
+
|
|
157
|
+
# Completed periods, oldest first. Minutes with no requests are returned
|
|
158
|
+
# as empty periods, so idle processes still report their release.
|
|
159
|
+
# include_current also closes the in-progress minute (used at shutdown).
|
|
160
|
+
def drain(include_current: false, max_periods: 10)
|
|
161
|
+
current = period_start(@clock.call)
|
|
162
|
+
last = include_current ? current : current - PERIOD_SECONDS
|
|
163
|
+
|
|
164
|
+
@mutex.synchronize do
|
|
165
|
+
first = [ @last_drained_at + PERIOD_SECONDS, last - (max_periods - 1) * PERIOD_SECONDS ].max
|
|
166
|
+
drained = (first.to_i..last.to_i).step(PERIOD_SECONDS).map do |seconds|
|
|
167
|
+
started_at = Time.at(seconds).utc
|
|
168
|
+
@periods.delete(started_at) || Period.new(started_at)
|
|
169
|
+
end
|
|
170
|
+
@periods.delete_if { |started_at, _| started_at <= last }
|
|
171
|
+
@last_drained_at = last if drained.any?
|
|
172
|
+
drained
|
|
173
|
+
end
|
|
174
|
+
end
|
|
175
|
+
|
|
176
|
+
private
|
|
177
|
+
def period_start(time)
|
|
178
|
+
Time.at(time.to_i - (time.to_i % PERIOD_SECONDS)).utc
|
|
179
|
+
end
|
|
180
|
+
end
|
|
181
|
+
end
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module DeployAngel
|
|
4
|
+
# Bounded FIFO of payloads waiting to be sent. When full, the oldest
|
|
5
|
+
# payload is dropped: losing telemetry is always preferable to growing
|
|
6
|
+
# memory in the customer's process.
|
|
7
|
+
class Buffer
|
|
8
|
+
attr_reader :dropped
|
|
9
|
+
|
|
10
|
+
def initialize(limit)
|
|
11
|
+
@limit = limit
|
|
12
|
+
@items = []
|
|
13
|
+
@dropped = 0
|
|
14
|
+
@mutex = Mutex.new
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def push(item)
|
|
18
|
+
@mutex.synchronize do
|
|
19
|
+
@items << item
|
|
20
|
+
while @items.size > @limit
|
|
21
|
+
@items.shift
|
|
22
|
+
@dropped += 1
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def shift
|
|
28
|
+
@mutex.synchronize { @items.shift }
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def unshift(item)
|
|
32
|
+
@mutex.synchronize { @items.unshift(item) if @items.size < @limit }
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
def size
|
|
36
|
+
@mutex.synchronize { @items.size }
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def clear
|
|
40
|
+
@mutex.synchronize { @items.clear }
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
end
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "digest"
|
|
4
|
+
|
|
5
|
+
module DeployAngel
|
|
6
|
+
# Exception fingerprint algorithm v1. Stable across deployments:
|
|
7
|
+
# no line numbers, no messages, no gem versions, no absolute paths, and no
|
|
8
|
+
# Ruby-version-specific label formatting.
|
|
9
|
+
module Fingerprint
|
|
10
|
+
VERSION = 1
|
|
11
|
+
GEM_PATH = %r{/gems/([^/]+?)-\d[^/]*/(.+)\z}
|
|
12
|
+
MESSAGE_PLACEHOLDERS = [
|
|
13
|
+
[ /\b[\w.+-]+@[\w-]+\.[\w.-]+\b/, "<email>" ],
|
|
14
|
+
[ /\b[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}\b/i, "<uuid>" ],
|
|
15
|
+
[ /\b0x[0-9a-f]+\b|\b[0-9a-f]{16,}\b/i, "<hex>" ],
|
|
16
|
+
[ /(["'`]).*?\1/, "<string>" ],
|
|
17
|
+
[ /\b\d+(\.\d+)?\b/, "<n>" ]
|
|
18
|
+
].freeze
|
|
19
|
+
MAX_MESSAGE = 200
|
|
20
|
+
MAX_BACKTRACE = 20
|
|
21
|
+
|
|
22
|
+
module_function
|
|
23
|
+
|
|
24
|
+
def for(exception, root:)
|
|
25
|
+
frame = top_frame(exception, root: root)
|
|
26
|
+
{
|
|
27
|
+
"fingerprint" => Digest::SHA256.hexdigest([ "v#{VERSION}", exception.class.name, frame ].join("|"))[0, 32],
|
|
28
|
+
"fingerprint_version" => VERSION,
|
|
29
|
+
"exception_class" => exception.class.name,
|
|
30
|
+
"message" => normalize_message(exception.message),
|
|
31
|
+
"top_frame" => frame,
|
|
32
|
+
"app_frame" => app_frame?(frame)
|
|
33
|
+
}
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# First application frame as "relative/path.rb#method", else the first
|
|
37
|
+
# frame normalized as "gem/path/in/gem.rb#method".
|
|
38
|
+
def top_frame(exception, root:)
|
|
39
|
+
frames = locations(exception)
|
|
40
|
+
app = frames.find { |path, _| app_path?(path, root) }
|
|
41
|
+
path, label = app || frames.first
|
|
42
|
+
return "unknown" unless path
|
|
43
|
+
|
|
44
|
+
"#{normalize_path(path, root)}##{label}"
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# Application frames only, relative to the root.
|
|
48
|
+
def backtrace(exception, root:)
|
|
49
|
+
locations(exception).select { |path, _| app_path?(path, root) }.first(MAX_BACKTRACE)
|
|
50
|
+
.map { |path, label| "#{normalize_path(path, root)}##{label}" }
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def normalize_message(message)
|
|
54
|
+
text = message.to_s.lines.first.to_s.strip
|
|
55
|
+
MESSAGE_PLACEHOLDERS.each { |pattern, placeholder| text = text.gsub(pattern, placeholder) }
|
|
56
|
+
text[0, MAX_MESSAGE]
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def app_frame?(frame)
|
|
60
|
+
frame.start_with?("app/", "lib/", "config/")
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def locations(exception)
|
|
64
|
+
if exception.backtrace_locations
|
|
65
|
+
exception.backtrace_locations.map { |location| [ location.absolute_path || location.path, location.base_label ] }
|
|
66
|
+
else
|
|
67
|
+
Array(exception.backtrace).map do |line|
|
|
68
|
+
path, _, label = line.partition(":in ")
|
|
69
|
+
[ path.sub(/:\d+\z/, ""), label.delete("`'").split(/[#.]/).last.to_s.sub(/\A(block|rescue|ensure) (\(\d+ levels\) )?in /, "") ]
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
def app_path?(path, root)
|
|
75
|
+
return false unless root && path&.start_with?(root)
|
|
76
|
+
|
|
77
|
+
relative = path.delete_prefix(root).delete_prefix("/")
|
|
78
|
+
!relative.start_with?("vendor/", "node_modules/", "tmp/")
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def normalize_path(path, root)
|
|
82
|
+
if root && path.start_with?(root)
|
|
83
|
+
path.delete_prefix(root).delete_prefix("/")
|
|
84
|
+
elsif (match = GEM_PATH.match(path))
|
|
85
|
+
"#{match[1]}/#{match[2]}"
|
|
86
|
+
else
|
|
87
|
+
File.basename(path.to_s)
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
end
|
|
91
|
+
end
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module DeployAngel
|
|
4
|
+
# Sparse log-scale latency histogram, scheme log1.1_ms_v1 (Agent Protocol
|
|
5
|
+
# v1). The agent only counts; percentiles are computed in the cloud after
|
|
6
|
+
# merging every process, because percentiles cannot be averaged.
|
|
7
|
+
class Histogram
|
|
8
|
+
SCHEME = "log1.1_ms_v1"
|
|
9
|
+
LOG_GAMMA = Math.log(1.1)
|
|
10
|
+
MAX_BUCKET = 400
|
|
11
|
+
|
|
12
|
+
def self.bucket_for(milliseconds)
|
|
13
|
+
return 0 if milliseconds < 1
|
|
14
|
+
|
|
15
|
+
(Math.log(milliseconds) / LOG_GAMMA).ceil.clamp(0, MAX_BUCKET)
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
def initialize
|
|
19
|
+
@counts = Hash.new(0)
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def record(milliseconds)
|
|
23
|
+
@counts[self.class.bucket_for(milliseconds)] += 1
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def merge!(other)
|
|
27
|
+
other.counts.each { |bucket, count| @counts[bucket] += count }
|
|
28
|
+
self
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def count
|
|
32
|
+
@counts.values.sum
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
def to_protocol
|
|
36
|
+
{ "scheme" => SCHEME, "counts" => @counts.sort.to_h.transform_keys(&:to_s) }
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
protected
|
|
40
|
+
attr_reader :counts
|
|
41
|
+
end
|
|
42
|
+
end
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "securerandom"
|
|
4
|
+
require "socket"
|
|
5
|
+
|
|
6
|
+
module DeployAngel
|
|
7
|
+
# One OS process. The random suffix is regenerated after fork, so a forked
|
|
8
|
+
# child never reuses its parent's identity.
|
|
9
|
+
class Instance
|
|
10
|
+
attr_reader :id, :host, :pid, :process_type, :started_at
|
|
11
|
+
|
|
12
|
+
def initialize(env: ENV, now: Time.now.utc)
|
|
13
|
+
@host = env["DYNO"] || Socket.gethostname
|
|
14
|
+
@pid = Process.pid
|
|
15
|
+
@process_type = env["DYNO"]&.split(".")&.first
|
|
16
|
+
@id = "#{@host}:#{@pid}:#{SecureRandom.hex(3)}"
|
|
17
|
+
@started_at = now
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
def to_protocol
|
|
21
|
+
{
|
|
22
|
+
"id" => id,
|
|
23
|
+
"host" => host,
|
|
24
|
+
"pid" => pid,
|
|
25
|
+
"process_type" => process_type,
|
|
26
|
+
"started_at" => started_at.iso8601
|
|
27
|
+
}.compact
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
end
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module DeployAngel
|
|
4
|
+
# Builds Agent Protocol v1 telemetry payloads. Only mergeable values are
|
|
5
|
+
# sent: counts and histograms, never percentiles or rates.
|
|
6
|
+
module Protocol
|
|
7
|
+
VERSION = 1
|
|
8
|
+
module_function
|
|
9
|
+
|
|
10
|
+
def telemetry(period, instance:, release:, runtime:, capabilities: %w[http])
|
|
11
|
+
{
|
|
12
|
+
"protocol_version" => VERSION,
|
|
13
|
+
"agent" => { "name" => "deployangel-ruby", "version" => DeployAngel::VERSION },
|
|
14
|
+
"runtime" => runtime,
|
|
15
|
+
"instance" => instance.to_protocol,
|
|
16
|
+
"release" => release.to_protocol,
|
|
17
|
+
"capabilities" => capabilities,
|
|
18
|
+
"period" => { "started_at" => period.started_at.iso8601, "duration_seconds" => Aggregator::PERIOD_SECONDS },
|
|
19
|
+
"http" => {
|
|
20
|
+
"requests" => period.requests,
|
|
21
|
+
"status_counts" => period.status_counts.to_h,
|
|
22
|
+
"unhandled_exceptions" => period.unhandled_exceptions,
|
|
23
|
+
"latency_histogram" => period.histogram.to_protocol
|
|
24
|
+
},
|
|
25
|
+
"routes" => period.routes.map do |key, route|
|
|
26
|
+
{
|
|
27
|
+
"key" => key,
|
|
28
|
+
"requests" => route.requests,
|
|
29
|
+
"status_counts" => route.status_counts.to_h,
|
|
30
|
+
"latency_histogram" => route.histogram.to_protocol
|
|
31
|
+
}
|
|
32
|
+
end,
|
|
33
|
+
"exceptions" => period.exceptions.values.map { |e| e.merge("sources" => e["sources"].to_h).compact },
|
|
34
|
+
"exceptions_truncated" => period.exceptions_truncated,
|
|
35
|
+
"jobs" => job_stats(period.jobs),
|
|
36
|
+
"job_classes" => period.job_classes.map { |key, stats| { "key" => key }.merge(job_stats(stats)) },
|
|
37
|
+
"checkpoints" => period.checkpoints.map { |key, count| { "key" => key, "count" => count } }
|
|
38
|
+
}
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
def job_stats(stats)
|
|
42
|
+
{
|
|
43
|
+
"processed" => stats.processed,
|
|
44
|
+
"failed" => stats.failed,
|
|
45
|
+
"discarded" => stats.discarded,
|
|
46
|
+
"duration_histogram" => stats.duration.to_protocol,
|
|
47
|
+
"queue_latency_histogram" => stats.queue_latency.to_protocol
|
|
48
|
+
}
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def runtime(framework: nil, framework_version: nil)
|
|
52
|
+
{
|
|
53
|
+
"language" => "ruby",
|
|
54
|
+
"language_version" => RUBY_VERSION,
|
|
55
|
+
"framework" => framework,
|
|
56
|
+
"framework_version" => framework_version
|
|
57
|
+
}.compact
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module DeployAngel
|
|
4
|
+
# Which release this process is running, resolved once at boot so that
|
|
5
|
+
# telemetry can be attributed to a deployment.
|
|
6
|
+
class Release < Struct.new(:version, :commit, :source)
|
|
7
|
+
COMMIT_FORMAT = /\A[0-9a-f]{7,40}\z/
|
|
8
|
+
|
|
9
|
+
# Order: explicit configuration, Heroku dyno metadata, the hosting
|
|
10
|
+
# platform's own variables (Kamal, Render, Fly.io, Railway, Coolify, and
|
|
11
|
+
# Dokku's GIT_REV), then a REVISION file.
|
|
12
|
+
def self.resolve(config:, env: ENV, root: nil)
|
|
13
|
+
if present?(config.release_version) || present?(config.revision)
|
|
14
|
+
build(config.release_version, config.revision, "config")
|
|
15
|
+
elsif present?(env["HEROKU_RELEASE_VERSION"]) || present?(env["HEROKU_SLUG_COMMIT"])
|
|
16
|
+
build(env["HEROKU_RELEASE_VERSION"], env["HEROKU_SLUG_COMMIT"], "heroku_dyno_metadata")
|
|
17
|
+
elsif present?(env["KAMAL_VERSION"])
|
|
18
|
+
kamal(env["KAMAL_VERSION"])
|
|
19
|
+
elsif present?(env["RENDER_GIT_COMMIT"])
|
|
20
|
+
build(nil, env["RENDER_GIT_COMMIT"], "render")
|
|
21
|
+
elsif (tag = fly_tag(env["FLY_IMAGE_REF"]))
|
|
22
|
+
build(tag, nil, "fly")
|
|
23
|
+
elsif present?(env["RAILWAY_GIT_COMMIT_SHA"]) || present?(env["RAILWAY_DEPLOYMENT_ID"])
|
|
24
|
+
railway(env)
|
|
25
|
+
elsif present?(env["SOURCE_COMMIT"]) && present?(env["COOLIFY_CONTAINER_NAME"] || env["COOLIFY_RESOURCE_UUID"])
|
|
26
|
+
build(nil, env["SOURCE_COMMIT"], "coolify")
|
|
27
|
+
# Dokku sets GIT_REV, but the name is generic, so the source says so.
|
|
28
|
+
elsif present?(env["GIT_REV"])
|
|
29
|
+
build(nil, env["GIT_REV"], "git_rev")
|
|
30
|
+
elsif root && File.file?(revision_path = File.join(root, "REVISION"))
|
|
31
|
+
build(nil, File.read(revision_path, 100), "revision_file")
|
|
32
|
+
else
|
|
33
|
+
new(nil, nil, "unknown")
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# Kamal sets KAMAL_VERSION in every app container: the git commit by
|
|
38
|
+
# default, with "_uncommitted_<random>" added for a dirty working tree,
|
|
39
|
+
# or a version you declared. A plain commit is reported as the commit,
|
|
40
|
+
# so it matches deploys registered by commit; anything else is the
|
|
41
|
+
# version, with the commit it starts with.
|
|
42
|
+
def self.kamal(value)
|
|
43
|
+
value = value.to_s.strip
|
|
44
|
+
plain_commit = value.downcase.match?(COMMIT_FORMAT)
|
|
45
|
+
build(plain_commit ? nil : value, value.split("_", 2).first, "kamal")
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
# Railway sets the commit for deploys from GitHub. Others, such as
|
|
49
|
+
# `railway up`, only have a deployment ID, which identifies the release.
|
|
50
|
+
def self.railway(env)
|
|
51
|
+
if present?(env["RAILWAY_GIT_COMMIT_SHA"])
|
|
52
|
+
build(nil, env["RAILWAY_GIT_COMMIT_SHA"], "railway")
|
|
53
|
+
else
|
|
54
|
+
build(env["RAILWAY_DEPLOYMENT_ID"], nil, "railway")
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# Fly.io tags each deploy's image ("registry.fly.io/shop:deployment-01H9…"),
|
|
59
|
+
# and the tag identifies the release. Fly.io sets no commit; pass one in
|
|
60
|
+
# with DEPLOYANGEL_REVISION for commit-level change tracking.
|
|
61
|
+
def self.fly_tag(image_ref)
|
|
62
|
+
name = image_ref.to_s.strip.split("@", 2).first.to_s.split("/").last.to_s
|
|
63
|
+
tag = name.split(":", 2)[1].to_s.delete_prefix("deployment-")
|
|
64
|
+
tag unless tag.empty?
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# The cloud rejects malformed commits, which would drop every payload,
|
|
68
|
+
# so anything that is not a hex SHA is left out.
|
|
69
|
+
def self.build(version, commit, source)
|
|
70
|
+
commit = commit.to_s.strip.downcase
|
|
71
|
+
new(present?(version) ? version.to_s.strip[0, 100] : nil,
|
|
72
|
+
commit.match?(COMMIT_FORMAT) ? commit : nil,
|
|
73
|
+
source)
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
def self.present?(value)
|
|
77
|
+
!value.to_s.strip.empty?
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def unknown?
|
|
81
|
+
version.nil? && commit.nil?
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def to_protocol
|
|
85
|
+
{ "version" => version, "commit" => commit, "source" => source }
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
end
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
require "net/http"
|
|
5
|
+
require "uri"
|
|
6
|
+
require "zlib"
|
|
7
|
+
require "stringio"
|
|
8
|
+
|
|
9
|
+
module DeployAngel
|
|
10
|
+
# Sends gzipped JSON to DeployAngel. Only ever called from the background
|
|
11
|
+
# thread, never from a request.
|
|
12
|
+
class Transport
|
|
13
|
+
Result = Struct.new(:outcome, :status, :retry_after, :body) do
|
|
14
|
+
def ok?
|
|
15
|
+
outcome == :ok
|
|
16
|
+
end
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def initialize(config)
|
|
20
|
+
@config = config
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
# :ok (2xx), :retry (network error or 5xx), or :drop (any other status,
|
|
24
|
+
# including 429 rate limiting, which the agent honors by pausing).
|
|
25
|
+
def post(path, body)
|
|
26
|
+
uri = URI.join(@config.endpoint.end_with?("/") ? @config.endpoint : "#{@config.endpoint}/", path.delete_prefix("/"))
|
|
27
|
+
request = Net::HTTP::Post.new(uri)
|
|
28
|
+
request["Authorization"] = "Bearer #{@config.token}"
|
|
29
|
+
request["Content-Type"] = "application/json"
|
|
30
|
+
request["Content-Encoding"] = "gzip"
|
|
31
|
+
request["User-Agent"] = "deployangel-ruby/#{DeployAngel::VERSION} ruby/#{RUBY_VERSION}"
|
|
32
|
+
request.body = gzip(JSON.generate(body))
|
|
33
|
+
|
|
34
|
+
response = Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == "https",
|
|
35
|
+
open_timeout: @config.open_timeout, read_timeout: @config.read_timeout,
|
|
36
|
+
write_timeout: @config.read_timeout) { |http| http.request(request) }
|
|
37
|
+
classify(response)
|
|
38
|
+
rescue StandardError => e
|
|
39
|
+
Result.new(:retry, nil, nil).tap { @config.logger&.debug("DeployAngel transport error: #{e.class}: #{e.message}") }
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
private
|
|
43
|
+
def classify(response)
|
|
44
|
+
status = response.code.to_i
|
|
45
|
+
case status
|
|
46
|
+
when 200..299 then Result.new(:ok, status, nil, parse_body(response.body))
|
|
47
|
+
when 500..599 then Result.new(:retry, status, nil)
|
|
48
|
+
else Result.new(:drop, status, (response["Retry-After"].to_i if status == 429))
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def parse_body(body)
|
|
53
|
+
body.to_s.empty? ? {} : JSON.parse(body)
|
|
54
|
+
rescue JSON::ParserError
|
|
55
|
+
{}
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def gzip(string)
|
|
59
|
+
io = StringIO.new
|
|
60
|
+
writer = Zlib::GzipWriter.new(io)
|
|
61
|
+
writer.write(string)
|
|
62
|
+
writer.close
|
|
63
|
+
io.string
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module DeployAngel
|
|
4
|
+
# Resets the agent in forked children (Puma cluster mode, Sidekiq swarm,
|
|
5
|
+
# etc.) using Process._fork, available since Ruby 3.1.
|
|
6
|
+
module ForkHook
|
|
7
|
+
def _fork
|
|
8
|
+
pid = super
|
|
9
|
+
DeployAngel.after_fork if pid.zero?
|
|
10
|
+
pid
|
|
11
|
+
end
|
|
12
|
+
end
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
Process.singleton_class.prepend(DeployAngel::ForkHook)
|