oppex_sdk 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +201 -0
- data/README.md +166 -0
- data/lib/oppex_sdk/endpoint.rb +6 -0
- data/lib/oppex_sdk/errors.rb +44 -0
- data/lib/oppex_sdk/incident_client.rb +187 -0
- data/lib/oppex_sdk/incident_request.rb +82 -0
- data/lib/oppex_sdk/incident_response.rb +14 -0
- data/lib/oppex_sdk/internal/async_dispatcher.rb +135 -0
- data/lib/oppex_sdk/internal/http_executor.rb +106 -0
- data/lib/oppex_sdk/internal/interrupt.rb +56 -0
- data/lib/oppex_sdk/internal/rate_limited_drop_logger.rb +40 -0
- data/lib/oppex_sdk/internal/retry_executor.rb +56 -0
- data/lib/oppex_sdk/internal/wire_codec.rb +89 -0
- data/lib/oppex_sdk/logging.rb +24 -0
- data/lib/oppex_sdk/severity.rb +49 -0
- data/lib/oppex_sdk/version.rb +6 -0
- data/lib/oppex_sdk.rb +32 -0
- metadata +60 -0
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "errors"
|
|
4
|
+
require_relative "severity"
|
|
5
|
+
|
|
6
|
+
module Oppex
|
|
7
|
+
# A validated, immutable incident submission.
|
|
8
|
+
#
|
|
9
|
+
# Validation happens in the constructor, so a malformed incident is rejected at
|
|
10
|
+
# the call site rather than inside a worker thread where only a log line would
|
|
11
|
+
# remain.
|
|
12
|
+
#
|
|
13
|
+
# request = Oppex::IncidentRequest.new(
|
|
14
|
+
# title: "Checkout latency breached the SLO",
|
|
15
|
+
# source: "checkout-api",
|
|
16
|
+
# severity: Oppex::Severity::HIGH
|
|
17
|
+
# )
|
|
18
|
+
#
|
|
19
|
+
# A request may override the service key configured on the client.
|
|
20
|
+
class IncidentRequest
|
|
21
|
+
# The maximum length of +source+, counted in characters.
|
|
22
|
+
MAX_SOURCE_LENGTH = 255
|
|
23
|
+
|
|
24
|
+
attr_reader :title, :source, :severity, :priority, :src_timestamp,
|
|
25
|
+
:service_key, :component, :group, :type, :details
|
|
26
|
+
|
|
27
|
+
def initialize(title:, source:, severity:, priority: 1, src_timestamp: nil,
|
|
28
|
+
service_key: nil, component: nil, group: nil, type: nil, details: nil)
|
|
29
|
+
@title = require_non_blank(title, "title")
|
|
30
|
+
@source = require_source(source)
|
|
31
|
+
@severity = Severity.validate!(severity)
|
|
32
|
+
@priority = validate_priority(priority)
|
|
33
|
+
@src_timestamp = validate_timestamp(src_timestamp)
|
|
34
|
+
@service_key = reject_blank(service_key, "service_key")
|
|
35
|
+
@component = reject_blank(component, "component")
|
|
36
|
+
@group = reject_blank(group, "group")
|
|
37
|
+
@type = reject_blank(type, "type")
|
|
38
|
+
@details = reject_blank(details, "details")
|
|
39
|
+
freeze
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
private
|
|
43
|
+
|
|
44
|
+
def require_non_blank(value, field)
|
|
45
|
+
raise TypeError, "#{field} must be a String" unless value.is_a?(String)
|
|
46
|
+
raise ArgumentError, "#{field} must not be blank" if value.strip.empty?
|
|
47
|
+
|
|
48
|
+
value
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def require_source(source)
|
|
52
|
+
source = require_non_blank(source, "source")
|
|
53
|
+
raise ArgumentError, "source must not exceed #{MAX_SOURCE_LENGTH} characters" if source.length > MAX_SOURCE_LENGTH
|
|
54
|
+
|
|
55
|
+
source
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# An optional field may be absent, but a present-yet-blank value is nearly
|
|
59
|
+
# always a bug at the call site, so it is rejected rather than silently sent.
|
|
60
|
+
def reject_blank(value, field)
|
|
61
|
+
return nil if value.nil?
|
|
62
|
+
|
|
63
|
+
require_non_blank(value, field)
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def validate_priority(priority)
|
|
67
|
+
raise TypeError, "priority must be an Integer" unless priority.is_a?(Integer)
|
|
68
|
+
raise ArgumentError, "priority must be between 1 and 5" unless (1..5).cover?(priority)
|
|
69
|
+
|
|
70
|
+
priority
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
def validate_timestamp(src_timestamp)
|
|
74
|
+
return (Time.now.to_f * 1000).to_i if src_timestamp.nil?
|
|
75
|
+
|
|
76
|
+
raise TypeError, "src_timestamp must be an Integer" unless src_timestamp.is_a?(Integer)
|
|
77
|
+
raise ArgumentError, "src_timestamp must be greater than zero" unless src_timestamp.positive?
|
|
78
|
+
|
|
79
|
+
src_timestamp
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
end
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Oppex
|
|
4
|
+
# The immutable result of a delivered incident.
|
|
5
|
+
#
|
|
6
|
+
# +successful+ and +code+ fall back to what the HTTP status already said when
|
|
7
|
+
# the response body does not carry its own.
|
|
8
|
+
IncidentResponse = Data.define(:successful, :code, :message, :incident_id) do
|
|
9
|
+
# @return [Boolean]
|
|
10
|
+
def successful?
|
|
11
|
+
successful
|
|
12
|
+
end
|
|
13
|
+
end
|
|
14
|
+
end
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "rate_limited_drop_logger"
|
|
4
|
+
|
|
5
|
+
module Oppex
|
|
6
|
+
module Internal
|
|
7
|
+
# Delivers asynchronous incidents on a fixed number of worker threads,
|
|
8
|
+
# queueing the rest up to +capacity+ and dropping the oldest entry once full.
|
|
9
|
+
#
|
|
10
|
+
# Dropping the oldest keeps the newest incident, which is the one most likely
|
|
11
|
+
# to still matter, and submission never blocks the application.
|
|
12
|
+
#
|
|
13
|
+
# Not public API.
|
|
14
|
+
class AsyncDispatcher
|
|
15
|
+
QUEUE_CAPACITY = 5000
|
|
16
|
+
WORKER_COUNT = 2
|
|
17
|
+
CLOSE_DRAIN_TIMEOUT_SECONDS = 10.0
|
|
18
|
+
|
|
19
|
+
def initialize(logger, worker_count: WORKER_COUNT, capacity: QUEUE_CAPACITY)
|
|
20
|
+
@logger = logger
|
|
21
|
+
@capacity = capacity
|
|
22
|
+
# One mutex guards the queue and every lifecycle flag together. Splitting
|
|
23
|
+
# them would reintroduce the missed-wakeup race: a worker reads
|
|
24
|
+
# "draining", close clears it and broadcasts, and only then does the
|
|
25
|
+
# worker start waiting, on a broadcast that already happened.
|
|
26
|
+
@mutex = Mutex.new
|
|
27
|
+
@changed = ConditionVariable.new
|
|
28
|
+
@tasks = []
|
|
29
|
+
@accepting = true
|
|
30
|
+
@draining = true
|
|
31
|
+
@active = 0
|
|
32
|
+
@drop_logger = RateLimitedDropLogger.new
|
|
33
|
+
@workers = Array.new(worker_count) { Thread.new { work } }
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# Queues a task, evicting the oldest queued task when the queue is full.
|
|
37
|
+
# Returns whether the task was accepted; only a closed dispatcher refuses.
|
|
38
|
+
#
|
|
39
|
+
# rubocop:disable Naming/PredicateMethod -- this is a command that reports
|
|
40
|
+
# whether it was accepted, not a question; `submit?` would read as one.
|
|
41
|
+
def submit(&task)
|
|
42
|
+
dropped = @mutex.synchronize do
|
|
43
|
+
next :refused unless @accepting
|
|
44
|
+
|
|
45
|
+
evicted = if @tasks.length >= @capacity
|
|
46
|
+
@tasks.shift
|
|
47
|
+
@drop_logger.record_drop
|
|
48
|
+
end
|
|
49
|
+
@tasks.push(task)
|
|
50
|
+
@changed.broadcast
|
|
51
|
+
evicted
|
|
52
|
+
end
|
|
53
|
+
return false if dropped == :refused
|
|
54
|
+
|
|
55
|
+
@logger.warn("oppex: dropped #{dropped} incidents in the last minute") if dropped
|
|
56
|
+
true
|
|
57
|
+
end
|
|
58
|
+
# rubocop:enable Naming/PredicateMethod
|
|
59
|
+
|
|
60
|
+
# Stops admitting work and drains what is already queued and in flight for
|
|
61
|
+
# up to +timeout_seconds+, then abandons the rest. Idempotent, and safe to
|
|
62
|
+
# call from any thread.
|
|
63
|
+
def close(timeout_seconds = CLOSE_DRAIN_TIMEOUT_SECONDS)
|
|
64
|
+
was_accepting = @mutex.synchronize do
|
|
65
|
+
accepting = @accepting
|
|
66
|
+
@accepting = false
|
|
67
|
+
accepting
|
|
68
|
+
end
|
|
69
|
+
return unless was_accepting
|
|
70
|
+
|
|
71
|
+
abandoned = drain_until(monotonic_now + timeout_seconds)
|
|
72
|
+
|
|
73
|
+
# The workers are deliberately not joined without a bound. A Ruby thread
|
|
74
|
+
# cannot be interrupted safely, so waiting on one that is blocked in an
|
|
75
|
+
# HTTP attempt would push close past its own timeout. Java makes the same
|
|
76
|
+
# trade: it waits for the drain, calls shutdownNow, and returns without
|
|
77
|
+
# waiting for whatever that failed to interrupt.
|
|
78
|
+
@workers.each { |worker| worker.join(0) }
|
|
79
|
+
|
|
80
|
+
# Reported once and directly, rather than through the rate-limited
|
|
81
|
+
# overflow counter: a shutdown loss and an overload loss are different
|
|
82
|
+
# events, and a rate-limited counter's last batch can go unreported.
|
|
83
|
+
@logger.warn("oppex: force-dropped #{abandoned} pending incidents during close") if abandoned.positive?
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
private
|
|
87
|
+
|
|
88
|
+
def work
|
|
89
|
+
loop do
|
|
90
|
+
task = @mutex.synchronize do
|
|
91
|
+
while @tasks.empty?
|
|
92
|
+
return unless @draining
|
|
93
|
+
|
|
94
|
+
@changed.wait(@mutex)
|
|
95
|
+
end
|
|
96
|
+
@active += 1
|
|
97
|
+
@tasks.shift
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
begin
|
|
101
|
+
task.call
|
|
102
|
+
ensure
|
|
103
|
+
@mutex.synchronize do
|
|
104
|
+
@active -= 1
|
|
105
|
+
@changed.broadcast
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
# Waits until nothing is queued or in flight, or the deadline passes. Then
|
|
112
|
+
# stops the workers and returns how many tasks were abandoned.
|
|
113
|
+
def drain_until(deadline)
|
|
114
|
+
@mutex.synchronize do
|
|
115
|
+
until @tasks.empty? && @active.zero?
|
|
116
|
+
remaining = deadline - monotonic_now
|
|
117
|
+
break if remaining <= 0
|
|
118
|
+
|
|
119
|
+
@changed.wait(@mutex, remaining)
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
@draining = false
|
|
123
|
+
@changed.broadcast
|
|
124
|
+
abandoned = @tasks.length
|
|
125
|
+
@tasks.clear
|
|
126
|
+
abandoned
|
|
127
|
+
end
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
def monotonic_now
|
|
131
|
+
Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
end
|
|
135
|
+
end
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "net/http"
|
|
4
|
+
require "uri"
|
|
5
|
+
|
|
6
|
+
require_relative "../endpoint"
|
|
7
|
+
require_relative "../errors"
|
|
8
|
+
require_relative "retry_executor"
|
|
9
|
+
require_relative "wire_codec"
|
|
10
|
+
|
|
11
|
+
module Oppex
|
|
12
|
+
module Internal
|
|
13
|
+
# The HTTP adapter. Not public API.
|
|
14
|
+
class HttpExecutor
|
|
15
|
+
OPEN_TIMEOUT_SECONDS = 3
|
|
16
|
+
READ_TIMEOUT_SECONDS = 5
|
|
17
|
+
WRITE_TIMEOUT_SECONDS = 5
|
|
18
|
+
# The API returns a small envelope. Anything larger is a misbehaving proxy,
|
|
19
|
+
# and reading it in full would let that proxy dictate this client's memory
|
|
20
|
+
# use.
|
|
21
|
+
MAX_RESPONSE_BYTES = 1024 * 1024
|
|
22
|
+
|
|
23
|
+
# Resolves the target URL.
|
|
24
|
+
#
|
|
25
|
+
# The override exists so tests and the external smoke consumer can point the
|
|
26
|
+
# whole delivery path at a loopback server. It is deliberately not client
|
|
27
|
+
# configuration, which would add a public knob that exists only to ease
|
|
28
|
+
# testing.
|
|
29
|
+
def self.resolve_endpoint
|
|
30
|
+
override = ENV.fetch("OPPEX_TEST_ENDPOINT_URL", nil)
|
|
31
|
+
override.nil? || override.empty? ? Oppex::DEFAULT_ENDPOINT : override
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def initialize(api_key, endpoint: nil)
|
|
35
|
+
@api_key = api_key
|
|
36
|
+
@uri = URI.parse(endpoint || self.class.resolve_endpoint)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# Performs one attempt. Every failure it raises is an IncidentError whose
|
|
40
|
+
# +retryable?+ already carries the retry decision, so the retry policy never
|
|
41
|
+
# has to know which library raised what.
|
|
42
|
+
def execute(payload)
|
|
43
|
+
response = send_request(payload)
|
|
44
|
+
status = response.code.to_i
|
|
45
|
+
body = truncate(response.body)
|
|
46
|
+
|
|
47
|
+
return WireCodec.parse_response(status, body) if (200...300).cover?(status)
|
|
48
|
+
|
|
49
|
+
raise status_failure(status, body)
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def close
|
|
53
|
+
# Net::HTTP instances are not thread safe, so this executor opens a
|
|
54
|
+
# connection per attempt rather than holding a pool. There is nothing to
|
|
55
|
+
# release here; the method exists so the client's lifecycle stays the same
|
|
56
|
+
# shape as every other Oppex SDK's.
|
|
57
|
+
nil
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
private
|
|
61
|
+
|
|
62
|
+
def send_request(payload)
|
|
63
|
+
request = Net::HTTP::Post.new(@uri)
|
|
64
|
+
request["Content-Type"] = "application/json"
|
|
65
|
+
request["Accept"] = "application/json"
|
|
66
|
+
request["X-API-KEY"] = @api_key
|
|
67
|
+
request.body = payload
|
|
68
|
+
|
|
69
|
+
Net::HTTP.start(@uri.hostname, @uri.port, use_ssl: @uri.scheme == "https",
|
|
70
|
+
open_timeout: OPEN_TIMEOUT_SECONDS,
|
|
71
|
+
read_timeout: READ_TIMEOUT_SECONDS,
|
|
72
|
+
write_timeout: WRITE_TIMEOUT_SECONDS) do |http|
|
|
73
|
+
http.request(request)
|
|
74
|
+
end
|
|
75
|
+
rescue StandardError => e
|
|
76
|
+
# No status line was received, so the failure is transport-level and
|
|
77
|
+
# retryable regardless of which library raised it. Catching StandardError
|
|
78
|
+
# rather than an enumerated list is deliberate: Net::HTTP, OpenSSL,
|
|
79
|
+
# resolv and socket code between them raise well over a dozen classes, and
|
|
80
|
+
# an omission would escape as a raw exception from a public method.
|
|
81
|
+
raise IncidentError.new("incident delivery failed before a response was received",
|
|
82
|
+
retryable: true, cause_error: e)
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def truncate(body)
|
|
86
|
+
return body if body.nil? || body.bytesize <= MAX_RESPONSE_BYTES
|
|
87
|
+
|
|
88
|
+
body.byteslice(0, MAX_RESPONSE_BYTES)
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# Turns a non-2xx response into a delivery failure, adding the API's own
|
|
92
|
+
# message when the body carries one.
|
|
93
|
+
def status_failure(status, body)
|
|
94
|
+
message = "Oppex returned HTTP #{status}"
|
|
95
|
+
detail = begin
|
|
96
|
+
WireCodec.parse_response(status, body).message
|
|
97
|
+
rescue IncidentError
|
|
98
|
+
nil # The status code remains sufficient when the body is not JSON.
|
|
99
|
+
end
|
|
100
|
+
message = "#{message}: #{detail}" if detail && !detail.strip.empty?
|
|
101
|
+
|
|
102
|
+
IncidentError.new(message, status_code: status, retryable: RetryExecutor.retryable_status?(status))
|
|
103
|
+
end
|
|
104
|
+
end
|
|
105
|
+
end
|
|
106
|
+
end
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Oppex
|
|
4
|
+
module Internal
|
|
5
|
+
# A one-way "stop waiting" signal.
|
|
6
|
+
#
|
|
7
|
+
# A sleeping Ruby thread cannot be interrupted safely the way a Java thread
|
|
8
|
+
# can, so a backoff delay waits on this condition variable instead of calling
|
|
9
|
+
# +sleep+. Closing the client signals it once, which wakes every waiting
|
|
10
|
+
# retry immediately rather than after up to eight more seconds.
|
|
11
|
+
#
|
|
12
|
+
# Not public API.
|
|
13
|
+
class Interrupt
|
|
14
|
+
def initialize
|
|
15
|
+
@mutex = Mutex.new
|
|
16
|
+
@changed = ConditionVariable.new
|
|
17
|
+
@signalled = false
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
# Signals every current and future waiter. Idempotent.
|
|
21
|
+
def signal
|
|
22
|
+
@mutex.synchronize do
|
|
23
|
+
@signalled = true
|
|
24
|
+
@changed.broadcast
|
|
25
|
+
end
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def signalled?
|
|
29
|
+
@mutex.synchronize { @signalled }
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# Waits up to +seconds+. Returns true when the wait was cut short by a
|
|
33
|
+
# signal, false when the full duration elapsed.
|
|
34
|
+
def wait(seconds)
|
|
35
|
+
deadline = monotonic_now + seconds
|
|
36
|
+
@mutex.synchronize do
|
|
37
|
+
until @signalled
|
|
38
|
+
remaining = deadline - monotonic_now
|
|
39
|
+
return false if remaining <= 0
|
|
40
|
+
|
|
41
|
+
@changed.wait(@mutex, remaining)
|
|
42
|
+
end
|
|
43
|
+
true
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
private
|
|
48
|
+
|
|
49
|
+
# A monotonic clock, so a system clock adjustment mid-backoff cannot turn a
|
|
50
|
+
# half-second wait into a very long one.
|
|
51
|
+
def monotonic_now
|
|
52
|
+
Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Oppex
|
|
4
|
+
module Internal
|
|
5
|
+
# Counts every dropped incident but reports at most one summary per interval.
|
|
6
|
+
#
|
|
7
|
+
# A saturated queue drops continuously, so logging each drop would replace
|
|
8
|
+
# one overload with another. Not public API.
|
|
9
|
+
class RateLimitedDropLogger
|
|
10
|
+
INTERVAL_SECONDS = 60.0
|
|
11
|
+
|
|
12
|
+
def initialize(interval_seconds: INTERVAL_SECONDS, clock: -> { Process.clock_gettime(Process::CLOCK_MONOTONIC) })
|
|
13
|
+
@interval_seconds = interval_seconds
|
|
14
|
+
@clock = clock
|
|
15
|
+
@mutex = Mutex.new
|
|
16
|
+
@dropped = 0
|
|
17
|
+
@started_at = clock.call
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
# Records one drop.
|
|
21
|
+
#
|
|
22
|
+
# Returns the accumulated count when the interval has elapsed and the
|
|
23
|
+
# caller should emit a summary, or nil otherwise. Returning the count
|
|
24
|
+
# instead of logging keeps this class free of any logging decision, which
|
|
25
|
+
# is also what makes it testable without capturing output.
|
|
26
|
+
def record_drop
|
|
27
|
+
@mutex.synchronize do
|
|
28
|
+
@dropped += 1
|
|
29
|
+
now = @clock.call
|
|
30
|
+
next nil if now - @started_at < @interval_seconds
|
|
31
|
+
|
|
32
|
+
dropped = @dropped
|
|
33
|
+
@dropped = 0
|
|
34
|
+
@started_at = now
|
|
35
|
+
dropped
|
|
36
|
+
end
|
|
37
|
+
end
|
|
38
|
+
end
|
|
39
|
+
end
|
|
40
|
+
end
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "../errors"
|
|
4
|
+
|
|
5
|
+
module Oppex
|
|
6
|
+
module Internal
|
|
7
|
+
# The retry policy. Not public API.
|
|
8
|
+
module RetryExecutor
|
|
9
|
+
# Five retries after the first attempt, doubling, without jitter.
|
|
10
|
+
# Deliberately not configurable.
|
|
11
|
+
DEFAULT_DELAYS = [0.5, 1.0, 2.0, 4.0, 8.0].freeze
|
|
12
|
+
|
|
13
|
+
# An explicit list rather than "any 5xx". Widening it is a deliberate
|
|
14
|
+
# policy change, not an incidental one.
|
|
15
|
+
RETRYABLE_STATUSES = [429, 500, 502, 503, 504].freeze
|
|
16
|
+
|
|
17
|
+
INTERRUPTED_MESSAGE = "incident delivery was interrupted during retry"
|
|
18
|
+
|
|
19
|
+
module_function
|
|
20
|
+
|
|
21
|
+
def retryable_status?(status)
|
|
22
|
+
RETRYABLE_STATUSES.include?(status)
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
# Runs the block until it succeeds, fails with a non-retryable error, or
|
|
26
|
+
# exhausts +delays+.
|
|
27
|
+
#
|
|
28
|
+
# Only the final failure is raised. Individual attempts are never logged,
|
|
29
|
+
# so a saturated Oppex API cannot flood the host's logs.
|
|
30
|
+
def execute(interrupt, delays: DEFAULT_DELAYS)
|
|
31
|
+
attempt = 0
|
|
32
|
+
loop do
|
|
33
|
+
begin
|
|
34
|
+
return yield
|
|
35
|
+
rescue IncidentError => e
|
|
36
|
+
raise annotate(e, attempt + 1) if !e.retryable? || attempt >= delays.length
|
|
37
|
+
|
|
38
|
+
raise IncidentError, INTERRUPTED_MESSAGE if interrupt.wait(delays[attempt])
|
|
39
|
+
end
|
|
40
|
+
attempt += 1
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# Annotates only a failure that never reached a status line. An HTTP
|
|
45
|
+
# failure keeps the message its status already explains.
|
|
46
|
+
def annotate(error, attempts)
|
|
47
|
+
return error if error.http_status? || attempts < 2
|
|
48
|
+
|
|
49
|
+
IncidentError.new("#{error.message} (after #{attempts} attempts)",
|
|
50
|
+
status_code: error.status_code,
|
|
51
|
+
retryable: error.retryable?,
|
|
52
|
+
cause_error: error.cause_error)
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
|
|
5
|
+
require_relative "../errors"
|
|
6
|
+
require_relative "../incident_response"
|
|
7
|
+
|
|
8
|
+
module Oppex
|
|
9
|
+
module Internal
|
|
10
|
+
# Renders the wire payload and decodes the response envelope.
|
|
11
|
+
#
|
|
12
|
+
# Not public API; every method here may change in any release.
|
|
13
|
+
module WireCodec
|
|
14
|
+
# Wire field order. Ruby hashes preserve insertion order, and +to_json+
|
|
15
|
+
# follows it, so the payload matches the other SDKs byte for byte.
|
|
16
|
+
|
|
17
|
+
module_function
|
|
18
|
+
|
|
19
|
+
# A nil resolved service key is omitted entirely so the API routes the
|
|
20
|
+
# incident by its own rules. Every other absent optional field is omitted
|
|
21
|
+
# rather than sent as null.
|
|
22
|
+
def serialize_request(request, resolved_service_key)
|
|
23
|
+
fields = {
|
|
24
|
+
serviceKey: resolved_service_key,
|
|
25
|
+
title: request.title,
|
|
26
|
+
source: request.source,
|
|
27
|
+
severity: request.severity,
|
|
28
|
+
priority: request.priority,
|
|
29
|
+
srcTimestamp: request.src_timestamp,
|
|
30
|
+
component: request.component,
|
|
31
|
+
group: request.group,
|
|
32
|
+
type: request.type,
|
|
33
|
+
detailsJSON: request.details
|
|
34
|
+
}
|
|
35
|
+
# One rejection covers every optional field, including the service key.
|
|
36
|
+
# The required fields cannot be nil: IncidentRequest validated them.
|
|
37
|
+
JSON.generate(fields.compact)
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
# Decodes a response body, falling back to the HTTP status for any field
|
|
41
|
+
# the body does not carry.
|
|
42
|
+
#
|
|
43
|
+
# A body that is not JSON is never echoed into the raised error: a proxy or
|
|
44
|
+
# WAF can return an error page that repeats request headers, including
|
|
45
|
+
# X-API-KEY, and that text would then leak into the host's logs.
|
|
46
|
+
def parse_response(http_status, body)
|
|
47
|
+
successful = (200...300).cover?(http_status)
|
|
48
|
+
return IncidentResponse.new(successful:, code: http_status, message: nil, incident_id: nil) if blank?(body)
|
|
49
|
+
|
|
50
|
+
envelope = parse_json(body, http_status)
|
|
51
|
+
IncidentResponse.new(
|
|
52
|
+
successful: boolean_or(envelope["success"], successful),
|
|
53
|
+
code: integer_or(envelope["code"], http_status),
|
|
54
|
+
message: string_or_nil(envelope["message"]),
|
|
55
|
+
incident_id: string_or_nil(envelope["data"])
|
|
56
|
+
)
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def blank?(body)
|
|
60
|
+
body.nil? || body.strip.empty?
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def parse_json(body, http_status)
|
|
64
|
+
envelope = JSON.parse(body)
|
|
65
|
+
unless envelope.is_a?(Hash)
|
|
66
|
+
raise IncidentError.new("Oppex returned a non-object JSON response (status #{http_status})",
|
|
67
|
+
status_code: http_status)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
envelope
|
|
71
|
+
rescue JSON::ParserError => e
|
|
72
|
+
raise IncidentError.new("Oppex returned a non-JSON response (status #{http_status})",
|
|
73
|
+
status_code: http_status, cause_error: e)
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
def boolean_or(value, fallback)
|
|
77
|
+
[true, false].include?(value) ? value : fallback
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def integer_or(value, fallback)
|
|
81
|
+
value.is_a?(Integer) ? value : fallback
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def string_or_nil(value)
|
|
85
|
+
value.is_a?(String) ? value : nil
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
end
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Oppex
|
|
4
|
+
# The default logging sink: writes one line per message to standard error.
|
|
5
|
+
#
|
|
6
|
+
# This SDK deliberately does not require the +logger+ standard library. It
|
|
7
|
+
# stops being a default gem in Ruby 4.0, so requiring it would turn a
|
|
8
|
+
# dependency-free gem into one with a runtime dependency, for four method
|
|
9
|
+
# calls. Any object that responds to +debug+, +info+, +warn+ and +error+ works
|
|
10
|
+
# instead, which means a host's +Logger+, +Rails.logger+, SemanticLogger or
|
|
11
|
+
# anything else drops in with no adapter.
|
|
12
|
+
class StderrLogger
|
|
13
|
+
LEVELS = %i[debug info warn error].freeze
|
|
14
|
+
|
|
15
|
+
LEVELS.each do |level|
|
|
16
|
+
define_method(level) do |message|
|
|
17
|
+
# A broken logging destination must never crash incident delivery.
|
|
18
|
+
warn("[oppex-sdk] #{level.to_s.upcase} #{message}")
|
|
19
|
+
rescue StandardError
|
|
20
|
+
nil
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
end
|
|
24
|
+
end
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Oppex
|
|
4
|
+
# Oppex incident severity, on a scale from 1 (lowest) to 5 (highest).
|
|
5
|
+
#
|
|
6
|
+
# The constants are the wire values, so an integer from another system can be
|
|
7
|
+
# passed straight through after {Severity.validate!}.
|
|
8
|
+
module Severity
|
|
9
|
+
LOWEST = 1
|
|
10
|
+
LOW = 2
|
|
11
|
+
MEDIUM = 3
|
|
12
|
+
HIGH = 4
|
|
13
|
+
CRITICAL = 5
|
|
14
|
+
|
|
15
|
+
ALL = [LOWEST, LOW, MEDIUM, HIGH, CRITICAL].freeze
|
|
16
|
+
|
|
17
|
+
NAMES = {
|
|
18
|
+
LOWEST => "LOWEST",
|
|
19
|
+
LOW => "LOW",
|
|
20
|
+
MEDIUM => "MEDIUM",
|
|
21
|
+
HIGH => "HIGH",
|
|
22
|
+
CRITICAL => "CRITICAL"
|
|
23
|
+
}.freeze
|
|
24
|
+
|
|
25
|
+
class << self
|
|
26
|
+
# @return [Boolean] whether +value+ is within the Oppex scale.
|
|
27
|
+
def valid?(value)
|
|
28
|
+
ALL.include?(value)
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# @return [Integer] +value+ unchanged.
|
|
32
|
+
# @raise [ArgumentError] when +value+ is outside 1 to 5.
|
|
33
|
+
# @raise [TypeError] when +value+ is not an Integer.
|
|
34
|
+
def validate!(value)
|
|
35
|
+
# true and false are not Integers in Ruby, so no bool special case is
|
|
36
|
+
# needed the way the Python SDK needs one.
|
|
37
|
+
raise TypeError, "severity must be an Integer" unless value.is_a?(Integer)
|
|
38
|
+
raise ArgumentError, "severity must be between 1 and 5" unless valid?(value)
|
|
39
|
+
|
|
40
|
+
value
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# @return [String, nil] the constant name for a wire value.
|
|
44
|
+
def name_for(value)
|
|
45
|
+
NAMES[value]
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
end
|
|
49
|
+
end
|