wide_events 0.1.4 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +46 -0
- data/README.md +149 -8
- data/lib/generators/wide_events/install/install_generator.rb +16 -2
- data/lib/generators/wide_events/skill_installer.rb +24 -0
- data/lib/generators/wide_events/skills/skills_generator.rb +9 -7
- data/lib/generators/wide_events/store/store_generator.rb +145 -0
- data/lib/generators/wide_events/store/templates/accessory.yml.erb +33 -0
- data/lib/wide_event/configuration.rb +64 -3
- data/lib/wide_event/job_instrumentation.rb +2 -0
- data/lib/wide_event/kamal/deploy_editor.rb +499 -0
- data/lib/wide_event/kamal/secrets_editor.rb +103 -0
- data/lib/wide_event/middleware.rb +2 -0
- data/lib/wide_event/railtie.rb +8 -0
- data/lib/wide_event/registry/defaults.yml +5 -0
- data/lib/wide_event/setup/checker.rb +402 -0
- data/lib/wide_event/setup/command_runner.rb +25 -0
- data/lib/wide_event/sinks/store.rb +33 -0
- data/lib/wide_event/store/client.rb +258 -0
- data/lib/wide_event/store/envelope.rb +60 -0
- data/lib/wide_event/store/formatter.rb +60 -0
- data/lib/wide_event/store/query_result.rb +19 -0
- data/lib/wide_event/store/sender.rb +448 -0
- data/lib/wide_event/tasks/setup.rake +55 -0
- data/lib/wide_event/tasks/store.rake +122 -0
- data/lib/wide_event/version.rb +1 -1
- data/lib/wide_event.rb +21 -0
- data/skills/debugging-with-wide-events/SKILL.md +92 -7
- data/skills/setting-up-wide-events-store/SKILL.md +120 -0
- metadata +36 -6
|
@@ -0,0 +1,448 @@
|
|
|
1
|
+
require "json"
|
|
2
|
+
require "zlib"
|
|
3
|
+
require "securerandom"
|
|
4
|
+
|
|
5
|
+
module WideEvent
|
|
6
|
+
module Store
|
|
7
|
+
# Owns the per-process bounded queue, batching, retry loop, and shutdown
|
|
8
|
+
# flush for delivering canonical event JSON to the store. Application
|
|
9
|
+
# threads only ever call #enqueue, which appends to an in-memory array
|
|
10
|
+
# under a mutex and never performs network I/O; a single worker thread
|
|
11
|
+
# seals batches, gzips them, and calls Client#ingest.
|
|
12
|
+
#
|
|
13
|
+
# PID-aware: every enqueue compares Process.pid (or the injected
|
|
14
|
+
# process_id) against the PID recorded at construction/last reset. A
|
|
15
|
+
# mismatch (post-fork) discards the inherited queue and worker thread and
|
|
16
|
+
# starts fresh in the child before admitting the new event.
|
|
17
|
+
class Sender
|
|
18
|
+
MAX_QUEUE_EVENTS = 1000
|
|
19
|
+
MAX_QUEUE_BYTES = 8 * 1024 * 1024
|
|
20
|
+
MAX_BATCH_EVENTS = 100
|
|
21
|
+
MAX_BATCH_BYTES = 1 * 1024 * 1024
|
|
22
|
+
MAX_COMPRESSED_BYTES = 512 * 1024
|
|
23
|
+
FLUSH_INTERVAL = 1.0
|
|
24
|
+
BASE_BACKOFF = 1.0
|
|
25
|
+
MAX_BACKOFF = 30.0
|
|
26
|
+
MAX_RETRY_AGE = 24 * 60 * 60
|
|
27
|
+
WARNING_INTERVAL = 60.0
|
|
28
|
+
SHUTDOWN_TIMEOUT = 2.0
|
|
29
|
+
|
|
30
|
+
WireBatch = Struct.new(:batch_id, :json, :compressed, :drop_count, :loss_report_id,
|
|
31
|
+
:sealed_at, :attempts, keyword_init: true)
|
|
32
|
+
|
|
33
|
+
def initialize(client:, service:, environment:,
|
|
34
|
+
max_queue_events: MAX_QUEUE_EVENTS, max_queue_bytes: MAX_QUEUE_BYTES,
|
|
35
|
+
max_batch_events: MAX_BATCH_EVENTS, max_batch_bytes: MAX_BATCH_BYTES,
|
|
36
|
+
max_compressed_bytes: MAX_COMPRESSED_BYTES,
|
|
37
|
+
flush_interval: FLUSH_INTERVAL, base_backoff: BASE_BACKOFF,
|
|
38
|
+
max_backoff: MAX_BACKOFF, max_retry_age: MAX_RETRY_AGE,
|
|
39
|
+
warning_interval: WARNING_INTERVAL,
|
|
40
|
+
clock: -> { Time.now }, uuid: -> { SecureRandom.uuid },
|
|
41
|
+
process_id: -> { Process.pid }, pid: nil, autostart: true)
|
|
42
|
+
@client = client
|
|
43
|
+
@service = service
|
|
44
|
+
@environment = environment
|
|
45
|
+
@max_queue_events = max_queue_events
|
|
46
|
+
@max_queue_bytes = max_queue_bytes
|
|
47
|
+
@max_batch_events = max_batch_events
|
|
48
|
+
@max_batch_bytes = max_batch_bytes
|
|
49
|
+
@max_compressed_bytes = max_compressed_bytes
|
|
50
|
+
@flush_interval = flush_interval
|
|
51
|
+
@base_backoff = base_backoff
|
|
52
|
+
@max_backoff = max_backoff
|
|
53
|
+
@max_retry_age = max_retry_age
|
|
54
|
+
@warning_interval = warning_interval
|
|
55
|
+
@clock = clock
|
|
56
|
+
@uuid = uuid
|
|
57
|
+
@process_id = process_id
|
|
58
|
+
@autostart = autostart
|
|
59
|
+
@max_event_bytes_per_batch = [ max_batch_bytes - envelope_overhead_bytes, 1 ].max
|
|
60
|
+
|
|
61
|
+
@mutex = Mutex.new
|
|
62
|
+
@cv = ConditionVariable.new
|
|
63
|
+
reset_state_locked
|
|
64
|
+
@pid = pid || @process_id.call
|
|
65
|
+
@thread = nil
|
|
66
|
+
@shutdown = false
|
|
67
|
+
@shutdown_deadline = nil
|
|
68
|
+
|
|
69
|
+
start_worker if @autostart
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
# Appends one canonical event JSON string to the bounded queue. Never
|
|
73
|
+
# performs network I/O and never raises into the caller.
|
|
74
|
+
def enqueue(event_json)
|
|
75
|
+
@mutex.synchronize do
|
|
76
|
+
reset_for_pid_change_locked
|
|
77
|
+
admit_locked(event_json)
|
|
78
|
+
end
|
|
79
|
+
@cv.broadcast
|
|
80
|
+
true
|
|
81
|
+
rescue StandardError => e
|
|
82
|
+
WideEvent.handle_error(e, "store_sender_enqueue")
|
|
83
|
+
false
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
def pending_drops
|
|
87
|
+
@mutex.synchronize { @pending_drops }
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
# Performs exactly one send attempt of the current (or newly sealed)
|
|
91
|
+
# batch, ignoring the retry backoff schedule. Test-only synchronous
|
|
92
|
+
# hook; the worker thread drives normal delivery.
|
|
93
|
+
def flush_once
|
|
94
|
+
tick(force: true)
|
|
95
|
+
nil
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
# Best-effort flush: keeps sending until nothing remains or the
|
|
99
|
+
# timeout elapses, then stops the worker thread.
|
|
100
|
+
def shutdown(timeout: SHUTDOWN_TIMEOUT)
|
|
101
|
+
@mutex.synchronize do
|
|
102
|
+
@shutdown = true
|
|
103
|
+
@shutdown_deadline = monotonic_now + timeout
|
|
104
|
+
end
|
|
105
|
+
@cv.broadcast
|
|
106
|
+
if @thread
|
|
107
|
+
@thread.join(timeout + 0.5)
|
|
108
|
+
@thread.kill if @thread.alive?
|
|
109
|
+
end
|
|
110
|
+
@thread = nil
|
|
111
|
+
nil
|
|
112
|
+
rescue StandardError => e
|
|
113
|
+
WideEvent.handle_error(e, "store_sender_shutdown")
|
|
114
|
+
nil
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# Test-only introspection of the raw pending queue.
|
|
118
|
+
def __queue_for_test
|
|
119
|
+
@mutex.synchronize { @queue.dup }
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
private
|
|
123
|
+
|
|
124
|
+
def reset_state_locked
|
|
125
|
+
@queue = []
|
|
126
|
+
@queue_bytes = 0
|
|
127
|
+
@queue_started_at = nil
|
|
128
|
+
@pending_drops = 0
|
|
129
|
+
@loss_report_id = nil
|
|
130
|
+
@last_drop_warning_at = nil
|
|
131
|
+
@last_drift_warning_at = nil
|
|
132
|
+
@in_flight = nil
|
|
133
|
+
@overflow = []
|
|
134
|
+
@next_attempt_at = nil
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
def reset_for_pid_change_locked
|
|
138
|
+
current = @process_id.call
|
|
139
|
+
return if current == @pid
|
|
140
|
+
|
|
141
|
+
@pid = current
|
|
142
|
+
reset_state_locked
|
|
143
|
+
@thread = nil # inherited thread reference is not ours to manage post-fork
|
|
144
|
+
start_worker if @autostart
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
def admit_locked(event_json)
|
|
148
|
+
@queue_started_at = @clock.call if @queue.empty?
|
|
149
|
+
bytes = event_json.bytesize
|
|
150
|
+
@queue << event_json
|
|
151
|
+
@queue_bytes += bytes
|
|
152
|
+
enforce_bounds_locked
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
def enforce_bounds_locked
|
|
156
|
+
while @queue.length > @max_queue_events || @queue_bytes > @max_queue_bytes
|
|
157
|
+
dropped = @queue.shift
|
|
158
|
+
break if dropped.nil?
|
|
159
|
+
@queue_bytes -= dropped.bytesize
|
|
160
|
+
record_drop_locked
|
|
161
|
+
end
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
def record_drop_locked
|
|
165
|
+
@pending_drops += 1
|
|
166
|
+
@loss_report_id ||= @uuid.call
|
|
167
|
+
warn_drop_locked
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
def warn_drop_locked
|
|
171
|
+
now = @clock.call
|
|
172
|
+
return if @last_drop_warning_at && (now - @last_drop_warning_at) < @warning_interval
|
|
173
|
+
@last_drop_warning_at = now
|
|
174
|
+
WideEvent.handle_error(
|
|
175
|
+
StandardError.new("wide_events store sender queue is full: #{@pending_drops} event(s) dropped so far"),
|
|
176
|
+
"store_sender_queue_overflow"
|
|
177
|
+
)
|
|
178
|
+
end
|
|
179
|
+
|
|
180
|
+
# --- worker loop -------------------------------------------------
|
|
181
|
+
|
|
182
|
+
def start_worker
|
|
183
|
+
@thread = Thread.new { worker_loop }
|
|
184
|
+
@thread.name = "wide_event.store.sender" if @thread.respond_to?(:name=)
|
|
185
|
+
@thread.abort_on_exception = false
|
|
186
|
+
end
|
|
187
|
+
|
|
188
|
+
def worker_loop
|
|
189
|
+
loop do
|
|
190
|
+
shutting_down, deadline = @mutex.synchronize { [ @shutdown, @shutdown_deadline ] }
|
|
191
|
+
if shutting_down
|
|
192
|
+
break unless work_remaining?
|
|
193
|
+
break if deadline && monotonic_now >= deadline
|
|
194
|
+
# Still honor any pending retry backoff (clamped to the shutdown
|
|
195
|
+
# deadline) instead of spinning: without this, a store that
|
|
196
|
+
# answers instantly with a failure (e.g. a stopped local
|
|
197
|
+
# sidecar giving immediate ECONNREFUSED) turns the drain into an
|
|
198
|
+
# unthrottled busy loop that hammers the store and floods the
|
|
199
|
+
# error handler for the whole shutdown window.
|
|
200
|
+
wait_for_retry_during_shutdown(deadline)
|
|
201
|
+
break if deadline && monotonic_now >= deadline
|
|
202
|
+
tick(force: true)
|
|
203
|
+
next
|
|
204
|
+
end
|
|
205
|
+
tick
|
|
206
|
+
wait_for_more_work
|
|
207
|
+
end
|
|
208
|
+
rescue StandardError => e
|
|
209
|
+
WideEvent.handle_error(e, "store_sender_worker")
|
|
210
|
+
end
|
|
211
|
+
|
|
212
|
+
def wait_for_retry_during_shutdown(deadline)
|
|
213
|
+
@mutex.synchronize do
|
|
214
|
+
return unless @in_flight && @next_attempt_at
|
|
215
|
+
wait_until = deadline ? [ @next_attempt_at, deadline ].min : @next_attempt_at
|
|
216
|
+
remaining = wait_until - monotonic_now
|
|
217
|
+
@cv.wait(@mutex, remaining) if remaining.positive?
|
|
218
|
+
end
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
def work_remaining?
|
|
222
|
+
@mutex.synchronize { !(@queue.empty? && @in_flight.nil? && @overflow.empty?) }
|
|
223
|
+
end
|
|
224
|
+
|
|
225
|
+
def wait_for_more_work
|
|
226
|
+
@mutex.synchronize do
|
|
227
|
+
return if @shutdown
|
|
228
|
+
timeout = next_wait_locked
|
|
229
|
+
@cv.wait(@mutex, timeout)
|
|
230
|
+
end
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
def next_wait_locked
|
|
234
|
+
if @in_flight && @next_attempt_at
|
|
235
|
+
remaining = @next_attempt_at - monotonic_now
|
|
236
|
+
return remaining.clamp(0.01, @flush_interval) if remaining.positive?
|
|
237
|
+
end
|
|
238
|
+
if @queue_started_at && !@queue.empty?
|
|
239
|
+
remaining = @flush_interval - (@clock.call - @queue_started_at)
|
|
240
|
+
return remaining.clamp(0.01, @flush_interval) if remaining.positive?
|
|
241
|
+
end
|
|
242
|
+
@flush_interval
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
# --- send/seal cycle ----------------------------------------------
|
|
246
|
+
|
|
247
|
+
def tick(force: false)
|
|
248
|
+
batch = nil
|
|
249
|
+
expired_batch = nil
|
|
250
|
+
@mutex.synchronize do
|
|
251
|
+
reset_for_pid_change_locked
|
|
252
|
+
if @in_flight && batch_expired_locked?(@in_flight)
|
|
253
|
+
expired_batch = @in_flight
|
|
254
|
+
@in_flight = nil
|
|
255
|
+
elsif !force && @in_flight && @next_attempt_at && monotonic_now < @next_attempt_at
|
|
256
|
+
return nil
|
|
257
|
+
else
|
|
258
|
+
batch = next_batch_locked(force: force)
|
|
259
|
+
end
|
|
260
|
+
end
|
|
261
|
+
if expired_batch
|
|
262
|
+
WideEvent.handle_error(
|
|
263
|
+
StandardError.new("batch exceeded the #{@max_retry_age}s retry horizon"),
|
|
264
|
+
"store_sender_batch_expired"
|
|
265
|
+
)
|
|
266
|
+
return nil
|
|
267
|
+
end
|
|
268
|
+
return nil if batch.nil?
|
|
269
|
+
respond_to_batch(batch)
|
|
270
|
+
nil
|
|
271
|
+
end
|
|
272
|
+
|
|
273
|
+
def batch_expired_locked?(batch)
|
|
274
|
+
(@clock.call - batch.sealed_at) > @max_retry_age
|
|
275
|
+
end
|
|
276
|
+
|
|
277
|
+
def next_batch_locked(force:)
|
|
278
|
+
return @in_flight if @in_flight
|
|
279
|
+
|
|
280
|
+
batch = @overflow.shift
|
|
281
|
+
unless batch
|
|
282
|
+
return nil unless force || seal_ready_locked?
|
|
283
|
+
sealed = seal_locked
|
|
284
|
+
return nil if sealed.empty?
|
|
285
|
+
batch = sealed.shift
|
|
286
|
+
@overflow.concat(sealed)
|
|
287
|
+
end
|
|
288
|
+
@in_flight = batch
|
|
289
|
+
end
|
|
290
|
+
|
|
291
|
+
# Only actually seal once the queue has reached the batch's size caps
|
|
292
|
+
# or the oldest queued event has been waiting flush_interval seconds —
|
|
293
|
+
# otherwise every enqueue's wakeup would immediately seal-and-send
|
|
294
|
+
# that single event as its own request, defeating the batch protocol
|
|
295
|
+
# under normal (sub-flush_interval) traffic. `force` (flush_once, and
|
|
296
|
+
# the shutdown drain) bypasses this and seals whatever is queued.
|
|
297
|
+
def seal_ready_locked?
|
|
298
|
+
return false if @queue.empty?
|
|
299
|
+
return true if @queue.length >= @max_batch_events
|
|
300
|
+
return true if @queue_bytes >= @max_event_bytes_per_batch
|
|
301
|
+
@queue_started_at && (@clock.call - @queue_started_at) >= @flush_interval
|
|
302
|
+
end
|
|
303
|
+
|
|
304
|
+
def seal_locked
|
|
305
|
+
return [] if @queue.empty?
|
|
306
|
+
|
|
307
|
+
events = []
|
|
308
|
+
bytes = 0
|
|
309
|
+
while !@queue.empty? && events.length < @max_batch_events
|
|
310
|
+
candidate_bytes = @queue.first.bytesize
|
|
311
|
+
break if events.any? && bytes + candidate_bytes > @max_event_bytes_per_batch
|
|
312
|
+
events << @queue.shift
|
|
313
|
+
bytes += candidate_bytes
|
|
314
|
+
@queue_bytes -= candidate_bytes
|
|
315
|
+
end
|
|
316
|
+
@queue_started_at = @queue.empty? ? nil : @clock.call
|
|
317
|
+
|
|
318
|
+
drop_count = @pending_drops
|
|
319
|
+
loss_report_id = drop_count.positive? ? (@loss_report_id ||= @uuid.call) : nil
|
|
320
|
+
split_into_wire_batches(events, drop_count, loss_report_id)
|
|
321
|
+
end
|
|
322
|
+
|
|
323
|
+
def split_into_wire_batches(events, drop_count, loss_report_id)
|
|
324
|
+
batch = build_wire_batch(events, drop_count, loss_report_id)
|
|
325
|
+
return [ batch ] if events.length <= 1 || batch.compressed.bytesize <= @max_compressed_bytes
|
|
326
|
+
|
|
327
|
+
mid = events.length / 2
|
|
328
|
+
split_into_wire_batches(events[0...mid], drop_count, loss_report_id) +
|
|
329
|
+
split_into_wire_batches(events[mid..], 0, nil)
|
|
330
|
+
end
|
|
331
|
+
|
|
332
|
+
# The store measures the whole canonical batch document against the same
|
|
333
|
+
# 1 MiB cap that the sealing loop applies to the sum of event bytes
|
|
334
|
+
# alone, so everything wrapped around those events has to come out of
|
|
335
|
+
# the same budget. Without that reserve a batch this sender considers
|
|
336
|
+
# legal can come back 400 - a permanent error, which drops the entire
|
|
337
|
+
# batch and reports nothing, since the loss counters only cover events
|
|
338
|
+
# the queue itself shed. Sized for the envelope at its largest: every
|
|
339
|
+
# fixed key present, 36-character batch and loss-report ids, a
|
|
340
|
+
# saturated drop count, and one separating comma per event.
|
|
341
|
+
def envelope_overhead_bytes
|
|
342
|
+
JSON.generate(
|
|
343
|
+
"protocol_version" => 1,
|
|
344
|
+
"batch_id" => "0" * 36,
|
|
345
|
+
"service" => @service,
|
|
346
|
+
"environment" => @environment,
|
|
347
|
+
"sent_at" => Time.now.utc.iso8601(3),
|
|
348
|
+
"dropped_events_since_last_batch" => (2**64) - 1,
|
|
349
|
+
"loss_report_id" => "0" * 36,
|
|
350
|
+
"events" => []
|
|
351
|
+
).bytesize + @max_batch_events
|
|
352
|
+
end
|
|
353
|
+
|
|
354
|
+
def build_wire_batch(events, drop_count, loss_report_id)
|
|
355
|
+
envelope = {
|
|
356
|
+
"protocol_version" => 1,
|
|
357
|
+
"batch_id" => @uuid.call,
|
|
358
|
+
"service" => @service,
|
|
359
|
+
"environment" => @environment,
|
|
360
|
+
"sent_at" => @clock.call.utc.iso8601(3),
|
|
361
|
+
"dropped_events_since_last_batch" => drop_count,
|
|
362
|
+
"loss_report_id" => loss_report_id,
|
|
363
|
+
"events" => events.map { |event_json| JSON.parse(event_json) }
|
|
364
|
+
}.compact
|
|
365
|
+
json = JSON.generate(envelope)
|
|
366
|
+
|
|
367
|
+
WireBatch.new(
|
|
368
|
+
batch_id: envelope["batch_id"],
|
|
369
|
+
json: json,
|
|
370
|
+
compressed: Zlib.gzip(json),
|
|
371
|
+
drop_count: drop_count,
|
|
372
|
+
loss_report_id: loss_report_id,
|
|
373
|
+
sealed_at: @clock.call,
|
|
374
|
+
attempts: 0
|
|
375
|
+
)
|
|
376
|
+
end
|
|
377
|
+
|
|
378
|
+
def respond_to_batch(batch)
|
|
379
|
+
batch.attempts += 1
|
|
380
|
+
ack = @client.ingest(batch.compressed)
|
|
381
|
+
on_ack(batch, ack)
|
|
382
|
+
rescue Client::Error => e
|
|
383
|
+
e.retryable? ? on_retryable_error(batch, e) : on_permanent_error(batch, e)
|
|
384
|
+
rescue StandardError => e
|
|
385
|
+
on_retryable_error(batch, e)
|
|
386
|
+
end
|
|
387
|
+
|
|
388
|
+
def on_ack(batch, ack)
|
|
389
|
+
warn_drift(ack.warnings) if ack.respond_to?(:warnings) && ack.warnings && !ack.warnings.empty?
|
|
390
|
+
@mutex.synchronize do
|
|
391
|
+
@in_flight = nil
|
|
392
|
+
if batch.loss_report_id
|
|
393
|
+
@pending_drops = [ @pending_drops - batch.drop_count, 0 ].max
|
|
394
|
+
@loss_report_id = nil
|
|
395
|
+
end
|
|
396
|
+
end
|
|
397
|
+
@cv.broadcast
|
|
398
|
+
end
|
|
399
|
+
|
|
400
|
+
def on_permanent_error(batch, error)
|
|
401
|
+
WideEvent.handle_error(error, "store_sender_batch_rejected")
|
|
402
|
+
@mutex.synchronize { @in_flight = nil }
|
|
403
|
+
@cv.broadcast
|
|
404
|
+
end
|
|
405
|
+
|
|
406
|
+
# The 24-hour retry-age expiry itself is checked proactively at the
|
|
407
|
+
# top of #tick (before a retry attempt is even made); this handler
|
|
408
|
+
# only schedules the next backoff for a batch that is still young
|
|
409
|
+
# enough to retry.
|
|
410
|
+
def on_retryable_error(batch, error)
|
|
411
|
+
WideEvent.handle_error(error, "store_sender_batch_retry")
|
|
412
|
+
delay = backoff_delay(batch.attempts)
|
|
413
|
+
@mutex.synchronize do
|
|
414
|
+
@in_flight = batch
|
|
415
|
+
@next_attempt_at = monotonic_now + delay
|
|
416
|
+
end
|
|
417
|
+
end
|
|
418
|
+
|
|
419
|
+
def backoff_delay(attempts)
|
|
420
|
+
# Clamp the exponent: a batch retried hundreds of times (e.g. during
|
|
421
|
+
# a hard shutdown drain, which retries with no delay) would
|
|
422
|
+
# otherwise compute 2**huge_number, a wasted bignum that still gets
|
|
423
|
+
# thrown away by the [.., @max_backoff].min below.
|
|
424
|
+
exponent = [ attempts - 1, 10 ].min
|
|
425
|
+
capped = [ @base_backoff * (2**exponent), @max_backoff ].min
|
|
426
|
+
capped * (0.5 + (rand * 0.5))
|
|
427
|
+
end
|
|
428
|
+
|
|
429
|
+
def warn_drift(warnings)
|
|
430
|
+
report = @mutex.synchronize do
|
|
431
|
+
now = @clock.call
|
|
432
|
+
next false if @last_drift_warning_at && (now - @last_drift_warning_at) < @warning_interval
|
|
433
|
+
@last_drift_warning_at = now
|
|
434
|
+
true
|
|
435
|
+
end
|
|
436
|
+
return unless report
|
|
437
|
+
WideEvent.handle_error(
|
|
438
|
+
StandardError.new(warnings.first.to_s[0, 200]),
|
|
439
|
+
"store_sender_ack_warning"
|
|
440
|
+
)
|
|
441
|
+
end
|
|
442
|
+
|
|
443
|
+
def monotonic_now
|
|
444
|
+
Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
445
|
+
end
|
|
446
|
+
end
|
|
447
|
+
end
|
|
448
|
+
end
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
require "wide_event/setup/checker"
|
|
2
|
+
|
|
3
|
+
# `bin/rails wide_events:setup:check` — the idempotent resume command for
|
|
4
|
+
# the setting-up-wide-events-store skill and for any human re-running setup
|
|
5
|
+
# by hand. It is correct with no flags whether it runs before the first
|
|
6
|
+
# `kamal setup`, while waiting on DNS to propagate, or against an
|
|
7
|
+
# already-deployed store: WideEvent::Setup::Checker re-derives host,
|
|
8
|
+
# hostname, service, and tokens from the app's own generated files.
|
|
9
|
+
#
|
|
10
|
+
# Guarded so re-`load`ing this file (the railtie does so once per process,
|
|
11
|
+
# but tests `load` it once per example) doesn't warn about reassigning the
|
|
12
|
+
# constants below.
|
|
13
|
+
unless defined?(WideEvent::Tasks::Setup)
|
|
14
|
+
module WideEvent
|
|
15
|
+
module Tasks
|
|
16
|
+
module Setup
|
|
17
|
+
# "failed" (a config/Kamal/verification error) and "degraded" (the
|
|
18
|
+
# store is reachable but actively refusing writes, e.g. low disk
|
|
19
|
+
# space) are both active faults, so both exit non-zero - a script,
|
|
20
|
+
# cron, or CI job relying on this exit status must see failure for
|
|
21
|
+
# either, not just for "failed". "needs_dns" (resumable, expected
|
|
22
|
+
# while DNS propagates), "ready_to_deploy", and "healthy" are all
|
|
23
|
+
# informational and exit zero.
|
|
24
|
+
FAULT_STATES = %w[failed degraded].freeze
|
|
25
|
+
|
|
26
|
+
# `checker:` is test-only: it lets the exit-status contract for
|
|
27
|
+
# each of the five states be pinned down with a fake Checker
|
|
28
|
+
# (a canned Result), instead of re-deriving all five through a
|
|
29
|
+
# fully fixtured Checker - that derivation is already covered by
|
|
30
|
+
# WideEvent::Setup::Checker's own test suite.
|
|
31
|
+
def self.run(stdout: $stdout, root: default_root, checker: nil)
|
|
32
|
+
checker ||= WideEvent::Setup::Checker.new(root: root)
|
|
33
|
+
result = checker.run
|
|
34
|
+
stdout.puts "wide_events setup: #{result.state}"
|
|
35
|
+
stdout.puts result.message if result.message && !result.message.empty?
|
|
36
|
+
abort("wide_events setup check reported #{result.state} - see the message above") if FAULT_STATES.include?(result.state)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def self.default_root
|
|
40
|
+
defined?(Rails) && Rails.respond_to?(:root) ? Rails.root.to_s : Dir.pwd
|
|
41
|
+
end
|
|
42
|
+
private_class_method :default_root
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
namespace :wide_events do
|
|
49
|
+
namespace :setup do
|
|
50
|
+
desc "Check (and resume) the Wide Events store setup: Kamal version/config, the host directory, DNS, and the deployed endpoint"
|
|
51
|
+
task check: :environment do
|
|
52
|
+
WideEvent::Tasks::Setup.run
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
require "wide_event/store/client"
|
|
2
|
+
require "wide_event/store/formatter"
|
|
3
|
+
|
|
4
|
+
# `bin/rails wide_events:sql` — an interactive prompt or a piped SQL runner
|
|
5
|
+
# against the store's read-only POST /v1/query. Split out as a plain module
|
|
6
|
+
# (rather than inline task code, per registry.rake's style) purely because
|
|
7
|
+
# the piped/interactive/format branching needs to be independently
|
|
8
|
+
# testable without going through Rake::Task#invoke.
|
|
9
|
+
#
|
|
10
|
+
# Guarded so re-`load`ing this file (the railtie does so once per process,
|
|
11
|
+
# but tests `load` it once per example via `with_rake`) doesn't warn about
|
|
12
|
+
# reassigning the constants below.
|
|
13
|
+
unless defined?(WideEvent::Tasks::Store)
|
|
14
|
+
module WideEvent
|
|
15
|
+
module Tasks
|
|
16
|
+
module Store
|
|
17
|
+
FORMATS = %w[table json csv].freeze
|
|
18
|
+
DEFAULT_FORMAT = "table"
|
|
19
|
+
EXIT_WORDS = %w[quit exit].freeze
|
|
20
|
+
PROMPT = "wide_events> "
|
|
21
|
+
|
|
22
|
+
# `interactive:` defaults to `stdin.tty?`; `readline:` defaults to
|
|
23
|
+
# the standard-library Readline. Both are overridable so tests can
|
|
24
|
+
# drive the interactive branch without a real terminal.
|
|
25
|
+
def self.run(stdin: $stdin, stdout: $stdout, readline: nil, interactive: nil)
|
|
26
|
+
config = WideEvent.config
|
|
27
|
+
abort(setup_guidance) if config.store_url.to_s.strip.empty? || config.store_query_token.to_s.strip.empty?
|
|
28
|
+
|
|
29
|
+
client = WideEvent::Store::Client.new(
|
|
30
|
+
url: config.store_url,
|
|
31
|
+
ingest_token: config.store_ingest_token,
|
|
32
|
+
query_token: config.store_query_token
|
|
33
|
+
)
|
|
34
|
+
format = resolve_format
|
|
35
|
+
run_interactively = interactive.nil? ? stdin.tty? : interactive
|
|
36
|
+
|
|
37
|
+
if run_interactively
|
|
38
|
+
interactive_loop(client, format, stdout, readline || default_readline)
|
|
39
|
+
else
|
|
40
|
+
run_piped(client, format, stdin, stdout)
|
|
41
|
+
end
|
|
42
|
+
rescue WideEvent::Store::Client::Error, ArgumentError => e
|
|
43
|
+
abort("wide_events:sql: #{e.message}")
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def self.setup_guidance
|
|
47
|
+
"wide_events:sql requires WIDE_EVENTS_URL and WIDE_EVENTS_QUERY_TOKEN to be set " \
|
|
48
|
+
"(see config/initializers/wide_events.rb, or run `bin/rails generate wide_events:install`)."
|
|
49
|
+
end
|
|
50
|
+
private_class_method :setup_guidance
|
|
51
|
+
|
|
52
|
+
def self.resolve_format
|
|
53
|
+
format = ENV["WIDE_EVENTS_FORMAT"].to_s.downcase
|
|
54
|
+
FORMATS.include?(format) ? format : DEFAULT_FORMAT
|
|
55
|
+
end
|
|
56
|
+
private_class_method :resolve_format
|
|
57
|
+
|
|
58
|
+
def self.default_readline
|
|
59
|
+
require "readline"
|
|
60
|
+
Readline
|
|
61
|
+
end
|
|
62
|
+
private_class_method :default_readline
|
|
63
|
+
|
|
64
|
+
def self.run_piped(client, format, stdin, stdout)
|
|
65
|
+
sql = stdin.read.to_s.strip
|
|
66
|
+
return if sql.empty?
|
|
67
|
+
execute_and_render(client, sql, format, stdout)
|
|
68
|
+
end
|
|
69
|
+
private_class_method :run_piped
|
|
70
|
+
|
|
71
|
+
# Accumulates lines until one ends with `;`, then runs that buffer
|
|
72
|
+
# as a single statement — this is what makes multiline SQL work.
|
|
73
|
+
# `quit` or `exit` typed alone at a fresh prompt (and EOF, i.e.
|
|
74
|
+
# `readline` returning nil) end the session instead. A query error
|
|
75
|
+
# prints and continues the loop rather than ending the session, so
|
|
76
|
+
# one typo doesn't lose the rest of an interactive investigation.
|
|
77
|
+
def self.interactive_loop(client, format, stdout, readline)
|
|
78
|
+
buffer = +""
|
|
79
|
+
loop do
|
|
80
|
+
prompt = buffer.empty? ? PROMPT : ""
|
|
81
|
+
line = readline.readline(prompt, true)
|
|
82
|
+
break if line.nil?
|
|
83
|
+
|
|
84
|
+
stripped = line.strip
|
|
85
|
+
break if buffer.empty? && EXIT_WORDS.include?(stripped.downcase)
|
|
86
|
+
|
|
87
|
+
buffer << line << "\n"
|
|
88
|
+
next unless stripped.end_with?(";")
|
|
89
|
+
|
|
90
|
+
sql = buffer.strip.sub(/;\z/, "").strip
|
|
91
|
+
buffer = +""
|
|
92
|
+
next if sql.empty?
|
|
93
|
+
|
|
94
|
+
begin
|
|
95
|
+
execute_and_render(client, sql, format, stdout)
|
|
96
|
+
rescue WideEvent::Store::Client::Error, ArgumentError => e
|
|
97
|
+
stdout.puts "error: #{e.message}"
|
|
98
|
+
end
|
|
99
|
+
end
|
|
100
|
+
end
|
|
101
|
+
private_class_method :interactive_loop
|
|
102
|
+
|
|
103
|
+
def self.execute_and_render(client, sql, format, stdout)
|
|
104
|
+
result = client.query(sql, format: :json)
|
|
105
|
+
case format
|
|
106
|
+
when "json" then WideEvent::Store::Formatter.json(result, io: stdout)
|
|
107
|
+
when "csv" then WideEvent::Store::Formatter.csv(result, io: stdout)
|
|
108
|
+
else WideEvent::Store::Formatter.table(result, io: stdout)
|
|
109
|
+
end
|
|
110
|
+
end
|
|
111
|
+
private_class_method :execute_and_render
|
|
112
|
+
end
|
|
113
|
+
end
|
|
114
|
+
end
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
namespace :wide_events do
|
|
118
|
+
desc "Run SQL against the wide-events store (interactive prompt, or piped stdin)"
|
|
119
|
+
task sql: :environment do
|
|
120
|
+
WideEvent::Tasks::Store.run
|
|
121
|
+
end
|
|
122
|
+
end
|
data/lib/wide_event/version.rb
CHANGED
data/lib/wide_event.rb
CHANGED
|
@@ -4,11 +4,17 @@ require "active_support/isolated_execution_state"
|
|
|
4
4
|
require "active_support/core_ext/object/blank"
|
|
5
5
|
|
|
6
6
|
require "wide_event/version"
|
|
7
|
+
require "wide_event/store/envelope"
|
|
8
|
+
require "wide_event/store/query_result"
|
|
9
|
+
require "wide_event/store/client"
|
|
10
|
+
require "wide_event/store/sender"
|
|
11
|
+
require "wide_event/store/formatter"
|
|
7
12
|
require "wide_event/configuration"
|
|
8
13
|
require "wide_event/registry"
|
|
9
14
|
require "wide_event/sinks/otel_span"
|
|
10
15
|
require "wide_event/sinks/log_line"
|
|
11
16
|
require "wide_event/sinks/memory"
|
|
17
|
+
require "wide_event/sinks/store"
|
|
12
18
|
require "wide_event/middleware"
|
|
13
19
|
require "wide_event/span_counter_processor"
|
|
14
20
|
require "wide_event/subscribers"
|
|
@@ -33,9 +39,24 @@ module WideEvent
|
|
|
33
39
|
end
|
|
34
40
|
|
|
35
41
|
def reset_configuration!
|
|
42
|
+
@config&.shutdown_store_sender!
|
|
36
43
|
@config = Configuration.new
|
|
37
44
|
end
|
|
38
45
|
|
|
46
|
+
# Best-effort flush of any in-flight store delivery: gives the sender's
|
|
47
|
+
# worker thread up to `timeout` seconds to drain its queue before the
|
|
48
|
+
# process exits. A no-op when no sink has been resolved yet (nothing to
|
|
49
|
+
# flush) or when the resolved sink has no shutdown hook. Never raises.
|
|
50
|
+
def shutdown!(timeout: 2.0)
|
|
51
|
+
return nil unless config.resolved?
|
|
52
|
+
sink = config.resolved_sink
|
|
53
|
+
sink.shutdown(timeout: timeout) if sink.respond_to?(:shutdown)
|
|
54
|
+
nil
|
|
55
|
+
rescue StandardError => e
|
|
56
|
+
handle_error(e, "shutdown")
|
|
57
|
+
nil
|
|
58
|
+
end
|
|
59
|
+
|
|
39
60
|
def enabled?
|
|
40
61
|
config.enabled
|
|
41
62
|
end
|