logtail 0.1.19 → 0.1.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: cd895e734e1d553b0dfdf1011f0a46b6f043a50c391d27822ee015a517759650
4
- data.tar.gz: 964ab22275d822db5df10f3a4616e25d500b0d9292057ea26f4467be40786e63
3
+ metadata.gz: b6611815d9d52cb947f7f547e988273c8d51a9663a906e6cebbb182a7df4029b
4
+ data.tar.gz: 99a3d5b9d3d6ced5187948f4bdb8f21435d0a7ee7c323431b49eb4944b6e5583
5
5
  SHA512:
6
- metadata.gz: 6556a1a6135a0c12070fab2080f43800e48fc0c7acfebab062d70ccc99999dadcbcb913145312d3fd61a0a5ad9f3ab411e6017a2cfe5cb55988867f882a748ec
7
- data.tar.gz: 10e8559ff2347820d3e7d0af52c019fbdd84c82507653f232b33e1bfdd9cd35382ba04a109537e6c537b429df857e0383bd2b3930cd60f576d121fb1e1f0f5ab
6
+ metadata.gz: 537aeaf25913ebd410554a3f1b9d8c6e49736d93a6ab180f45b84b88ee64656d4a34ce2193cdf52b3b439da5a02a4a5cad1404cb16c5cdd45156e5e6629071db
7
+ data.tar.gz: 83f2a2915f94e1864d078fe10c10f65d2b420ee3610d27f085c3dd7de45d5de6f63db859180b8922563b6d48c1f4be6a2e130689fcea9a0d419adc5352640705
data/Gemfile CHANGED
@@ -9,3 +9,6 @@ gem "rake", "~> 13.0"
9
9
 
10
10
  gem "rspec", "~> 3.0"
11
11
  gem "base64" if RUBY_VERSION >= "3.4.0"
12
+
13
+ # spec/logtail/json_encoding_spec.rb checks the JSON with ActiveSupport loaded
14
+ gem "activesupport"
@@ -32,6 +32,10 @@ module Logtail
32
32
 
33
33
  attr_writer :http_body_limit
34
34
 
35
+ def initialize
36
+ @debug_logger = nil
37
+ end
38
+
35
39
  # Whether a particular {Logtail::LogEntry} should be sent to Better Stack
36
40
  def send_to_better_stack?(log_entry)
37
41
  !@better_stack_filters&.any? { |blocker| blocker.call(log_entry) }
data/lib/logtail/event.rb CHANGED
@@ -17,7 +17,7 @@ module Logtail
17
17
  end
18
18
 
19
19
  def to_json(options = {})
20
- metadata.to_json(options)
20
+ Util.generate_json(metadata)
21
21
  end
22
22
 
23
23
  def to_hash
@@ -13,7 +13,7 @@ module Logtail
13
13
  @params = attributes[:params]
14
14
 
15
15
  if @params
16
- @params_json = @params.to_json
16
+ @params_json = Util.generate_json(@params)
17
17
  end
18
18
 
19
19
  @format = attributes[:format]
@@ -12,7 +12,7 @@ module Logtail
12
12
  @error_message = attributes[:error_message]
13
13
 
14
14
  if attributes[:backtrace]
15
- @backtrace_json = attributes[:backtrace].to_json
15
+ @backtrace_json = Util.generate_json(attributes[:backtrace])
16
16
  end
17
17
  end
18
18
 
@@ -4,11 +4,12 @@ module Logtail
4
4
  # Represents an attempt to deliver a request. Requests can be retried, hence
5
5
  # why we keep track of the number of attempts.
6
6
  class RequestAttempt
7
- attr_reader :attempts, :request
7
+ attr_reader :attempts, :request, :line_count
8
8
 
9
- def initialize(req)
9
+ def initialize(req, line_count = nil)
10
10
  @attempts = 0
11
11
  @request = req
12
+ @line_count = line_count
12
13
  end
13
14
 
14
15
  def attempted!
@@ -1,5 +1,7 @@
1
1
  require "msgpack"
2
2
  require "net/https"
3
+ require "set"
4
+ require "time"
3
5
  require "zlib"
4
6
 
5
7
  require "logtail/config"
@@ -22,6 +24,15 @@ module Logtail
22
24
  DEFAULT_INGESTING_SCHEME = "https".freeze
23
25
  CONTENT_TYPE = "application/msgpack".freeze
24
26
  USER_AGENT = "Logtail Ruby/#{Logtail::VERSION} (HTTP)".freeze
27
+ ENCODABLE_INTEGERS = (-2**63...2**64).freeze # the integers msgpack can encode
28
+ MAX_UNTRACKED_DEPTH = 100 # nested hashes and arrays, see #encodable_value
29
+ INITIAL_RECONNECT_WAIT = 1 # second
30
+ MAX_RECONNECT_WAIT = 30 # seconds
31
+ MAX_RETRY_AFTER = 60 # seconds
32
+ # The HTTP statuses of rejected batches this process has warned about, see {#report_rejected_batch}.
33
+ REPORTED_REJECTIONS = []
34
+ REPORTED_REJECTIONS_LOCK = Mutex.new
35
+ SYNCHRONOUS_DELIVERY_TIMEOUT = 5 # seconds, to connect and to read the response
25
36
 
26
37
  # Instantiates a new HTTP log device that can be passed to {Logtail::Logger#initialize}.
27
38
  #
@@ -79,10 +90,21 @@ module Logtail
79
90
  @flush_continuously = options[:flush_continuously] != false
80
91
  @flush_interval = options[:flush_interval] || 2 # 2 seconds
81
92
  @requests_per_conn = options[:requests_per_conn] || 2_500
93
+ # The process that owns the queues and threads, see {#reset_if_forked}
94
+ @pid = Process.pid
95
+ @fork_lock = Mutex.new
82
96
  @msg_queue = FlushableDroppingSizedQueue.new(@batch_size)
83
97
  @request_queue = options[:request_queue] || FlushableDroppingSizedQueue.new(25)
84
98
  @successive_error_count = 0
85
99
  @requests_in_flight = 0
100
+ @last_resp = nil
101
+ @reconnect_wait = INITIAL_RECONNECT_WAIT
102
+ @closed = false
103
+ @late_delivery_failed = false
104
+
105
+ # Delivers what is still buffered when the process exits. One hook per device, however
106
+ # many loggers write to it.
107
+ at_exit { close }
86
108
  end
87
109
 
88
110
  # Write a new log line message to the buffer, and flush asynchronously if the
@@ -90,15 +112,27 @@ module Logtail
90
112
  # size is constricted by the Logtail API. The actual application limit is a multiple
91
113
  # of this. Hence the `@request_queue`.
92
114
  def write(msg)
115
+ # Strings, e.g. from a plain ::Logger writing to this device, are sent as info lines.
116
+ msg = LogEntry.new(:info, Time.now, nil, msg.to_s.chomp, nil, nil) unless msg.is_a?(LogEntry)
93
117
  return unless Logtail.config.send_to_better_stack?(msg)
118
+ reset_if_forked
94
119
 
95
120
  @msg_queue.enq(msg)
121
+ # No thread delivers what is written after #close, e.g. by an at_exit hook that runs
122
+ # after the device's own.
123
+ return deliver_late_lines if @closed
96
124
 
97
125
  # Lazily start flush threads to ensure threads are alive after forking processes.
98
126
  # If the threads are started during instantiation they will not be copied when
99
127
  # the current process is forked. This is the case with various web servers,
100
128
  # such as phusion passenger.
101
- ensure_flush_threads_are_started
129
+ begin
130
+ ensure_flush_threads_are_started
131
+ rescue ThreadError
132
+ # Ruby refuses new threads while it shuts down, e.g. to a thread that logs in an
133
+ # `ensure` block as it is killed at exit.
134
+ return deliver_late_lines
135
+ end
102
136
 
103
137
  if @msg_queue.full?
104
138
  Logtail::Config.instance.debug { "Flushing HTTP buffer via write" }
@@ -108,16 +142,28 @@ module Logtail
108
142
  end
109
143
 
110
144
  # Flush all log messages in the buffer synchronously. This method will not return
111
- # until delivery of the messages has been successful. If you want to flush
145
+ # until delivery of the messages has been successful, or about 5 seconds have passed.
146
+ # When no outlet thread runs (`flush_continuously: false`, or a forked child that hasn't
147
+ # logged yet), the messages are delivered in the calling thread. If you want to flush
112
148
  # asynchronously see {#flush_async}.
113
149
  def flush
150
+ reset_if_forked
114
151
  flush_async
115
- wait_on_request_queue
152
+ if @request_outlet_thread && @request_outlet_thread.alive?
153
+ wait_on_request_queue
154
+ else
155
+ deliver_synchronously(dequeue_requests)
156
+ end
116
157
  true
117
158
  end
118
159
 
119
- # Closes the log device, cleans up, and attempts one last delivery.
160
+ # Closes the log device, cleans up, and attempts one last delivery. Closing it again does
161
+ # nothing; lines written after it are delivered right away (see {#write}).
120
162
  def close
163
+ reset_if_forked
164
+ return if @closed
165
+ @closed = true
166
+
121
167
  # Kill the flush thread immediately since we are about to flush again.
122
168
  @flush_thread.kill.join if @flush_thread
123
169
 
@@ -197,6 +243,38 @@ Logtail::Config.instance.debug_logger = ::Logger.new(STDOUT)
197
243
  end
198
244
  end
199
245
 
246
+ # The queues and threads belong to the process that created them. After a fork, the
247
+ # parent still delivers the lines it buffered, so a child that kept them would send them
248
+ # again, and the parent's threads don't run in the child. The child starts over with
249
+ # empty queues and starts its own threads once it logs, also when the parent closed the
250
+ # device before forking.
251
+ def reset_if_forked
252
+ return if @pid == Process.pid
253
+
254
+ @fork_lock.synchronize do
255
+ return if @pid == Process.pid
256
+
257
+ @msg_queue = FlushableDroppingSizedQueue.new(@batch_size)
258
+ # The request queue can be a SizedQueue passed as the :request_queue option
259
+ @request_queue.respond_to?(:flush) ? @request_queue.flush : @request_queue.clear
260
+ @flush_thread = @request_outlet_thread = nil
261
+ @requests_in_flight = 0
262
+ @reconnect_wait = INITIAL_RECONNECT_WAIT
263
+ @closed = @late_delivery_failed = false
264
+ @pid = Process.pid
265
+ end
266
+ end
267
+
268
+ # Takes the queued requests off the request queue, for {#flush} when no outlet thread
269
+ # runs. It checks the size first because a SizedQueue (see :request_queue) blocks when empty.
270
+ def dequeue_requests
271
+ requests = []
272
+ while @request_queue.size > 0 && (request_attempt = @request_queue.deq)
273
+ requests << request_attempt
274
+ end
275
+ requests
276
+ end
277
+
200
278
  # Builds an HTTP request based on the current messages queued.
201
279
  def build_request(msgs)
202
280
  path = '/'
@@ -205,16 +283,164 @@ Logtail::Config.instance.debug_logger = ::Logger.new(STDOUT)
205
283
  req['Content-Type'] = CONTENT_TYPE
206
284
  req['Content-Encoding'] = 'deflate'
207
285
  req['User-Agent'] = USER_AGENT
208
- uncompressed = msgs.map { |msg| force_utf8_encoding(msg.to_hash) }.to_msgpack
286
+ # Entries are encoded one at a time, so one that can't be encoded doesn't lose the batch.
287
+ packer = MessagePack::DefaultFactory.packer
288
+ uncompressed = packer.write_array_header(msgs.size).to_s
289
+ packer.reset
290
+ msgs.each { |msg| uncompressed << encode_log_entry(msg, packer) }
209
291
  req.body = Zlib::Deflate.deflate(uncompressed, Zlib::BEST_SPEED)
210
292
  req
211
293
  end
212
294
 
295
+ # Encodes a single log entry with msgpack, with the packer if given, which it leaves empty.
296
+ # An entry that still can't be encoded is replaced by one that says why, with the same
297
+ # level and time.
298
+ def encode_log_entry(msg, packer = MessagePack::DefaultFactory.packer)
299
+ packer.write(encodable_value(msg.to_hash)).to_s
300
+ rescue StandardError, SystemStackError => e
301
+ Logtail::Config.instance.debug { "Could not encode log entry: #{e.inspect}" }
302
+ error = force_utf8_encoding("#{e.class}: #{e.message}")
303
+ message = "Logtail could not encode this log line (#{error}): #{force_utf8_encoding(msg.message)}"
304
+ {
305
+ level: msg.level,
306
+ dt: msg.time.iso8601(LogEntry::DT_PRECISION),
307
+ message: message.byteslice(0, LogEntry::MESSAGE_MAX_BYTES).scrub(""),
308
+ }.to_msgpack
309
+ ensure
310
+ packer.reset
311
+ end
312
+
313
+ # Converts what msgpack can't encode, recursively, mostly into strings, and passes strings
314
+ # that aren't valid UTF-8 to {#force_utf8_encoding}. Returns the value itself when nothing
315
+ # needs to change, as for most log lines, and otherwise copies only the hashes and arrays
316
+ # that change. A hash or array that contains itself is cut off with "[circular]".
317
+ def encodable_value(value)
318
+ # The first pass doesn't keep track of the hashes and arrays it is in, and gives up when
319
+ # they nest too deep, as in a cycle. The second pass keeps track of them to find cycles.
320
+ catch(:too_deep) { return replacement_for(value, nil, 0) || value }
321
+ replacement_for(value, {}.compare_by_identity, 0) || value
322
+ end
323
+
324
+ # Returns what to send instead of the value, or nil to send the value as it is.
325
+ def replacement_for(value, parents, depth)
326
+ case value
327
+ when Hash
328
+ hash_replacement(value, parents, depth)
329
+ when String
330
+ force_utf8_encoding(value) unless value.valid_encoding? && (value.encoding == Encoding::UTF_8 || value.encoding == Encoding::US_ASCII)
331
+ when Integer
332
+ value.to_s unless value.bit_length < 64 || ENCODABLE_INTEGERS.cover?(value)
333
+ when nil, true, false, Symbol, Float
334
+ nil
335
+ when Array, Set, Struct
336
+ if parents
337
+ return "[circular]" if parents.key?(value)
338
+
339
+ parents[value] = true
340
+ elsif depth == MAX_UNTRACKED_DEPTH
341
+ throw :too_deep
342
+ end
343
+ replacement =
344
+ if value.is_a?(Array)
345
+ array_replacement(value, parents, depth + 1)
346
+ elsif value.is_a?(Set)
347
+ array_replacement(items = value.to_a, parents, depth + 1) || items
348
+ else
349
+ hash_replacement(members = value.to_h, parents, depth + 1) || members
350
+ end
351
+ parents.delete(value) if parents
352
+ replacement
353
+ else
354
+ force_utf8_encoding(converted_value(value))
355
+ end
356
+ end
357
+
358
+ # Returns a copy of the hash with the replacements for its keys and values, or nil if none
359
+ # needs one. The most common keys and values are checked right here, which is faster.
360
+ def hash_replacement(hash, parents, depth)
361
+ if parents
362
+ return "[circular]" if parents.key?(hash)
363
+
364
+ parents[hash] = true
365
+ elsif depth == MAX_UNTRACKED_DEPTH
366
+ throw :too_deep
367
+ end
368
+ copy = nil
369
+ key_changes = false
370
+ hash.each_pair do |key, item|
371
+ new_key = replacement_for(key, parents, depth + 1) unless key.is_a?(Symbol)
372
+ new_item =
373
+ if item.is_a?(String)
374
+ force_utf8_encoding(item) unless item.valid_encoding? && (item.encoding == Encoding::UTF_8 || item.encoding == Encoding::US_ASCII)
375
+ elsif item.is_a?(Hash)
376
+ hash_replacement(item, parents, depth + 1)
377
+ elsif !(item.nil? || item.is_a?(Integer) && item.bit_length < 64 || item.is_a?(Symbol) || item.is_a?(Float))
378
+ replacement_for(item, parents, depth + 1)
379
+ end
380
+ if new_key
381
+ key_changes = true
382
+ break
383
+ elsif new_item
384
+ (copy ||= Hash[hash])[key] = new_item
385
+ end
386
+ end
387
+ # A key that changes is rare, the copy is then built from scratch to keep the order of the keys
388
+ if key_changes
389
+ copy = {}
390
+ hash.each_pair { |key, item| copy[replacement_for(key, parents, depth + 1) || key] = replacement_for(item, parents, depth + 1) || item }
391
+ end
392
+ parents.delete(hash) if parents
393
+ copy
394
+ end
395
+
396
+ # Returns a copy of the array with the replacements for its items, or nil if none needs one.
397
+ def array_replacement(array, parents, depth)
398
+ copy = nil
399
+ array.each_with_index do |item, index|
400
+ new_item = replacement_for(item, parents, depth)
401
+ (copy ||= Array.new(array))[index] = new_item if new_item
402
+ end
403
+ copy
404
+ end
405
+
406
+ # Converts a value msgpack can't encode that isn't a hash, array, set or struct.
407
+ def converted_value(value)
408
+ case value
409
+ when Time, DateTime # Rails makes ActiveSupport::TimeWithZone match Time too
410
+ value.to_time.getutc.iso8601(LogEntry::DT_PRECISION)
411
+ when Date
412
+ value.iso8601
413
+ when Exception
414
+ { class: value.class.name, message: value.message }
415
+ when Numeric
416
+ # BigDecimal#to_s would use an exponent, "0.1999e2"
417
+ defined?(::BigDecimal) && value.is_a?(::BigDecimal) ? value.to_s("F") : value.to_s
418
+ else
419
+ # The public id of a Rack::Session::SessionId is the cookie of a server-side session
420
+ value.respond_to?(:private_id) ? value.private_id : value.to_s
421
+ end
422
+ end
423
+
213
424
  def force_utf8_encoding(data)
214
425
  if data.respond_to?(:force_encoding)
215
- data.dup.force_encoding('UTF-8')
216
- elsif data.respond_to?(:transform_values)
217
- data.transform_values { |val| force_utf8_encoding(val) }
426
+ # Only valid UTF-8 may leave: Better Stack stores anything else as invalid JSON. A string
427
+ # that is valid UTF-8 already, as nearly all are, is sent as it is.
428
+ return data if data.valid_encoding? && (data.encoding == Encoding::UTF_8 || data.encoding == Encoding::US_ASCII)
429
+
430
+ case data.encoding
431
+ when Encoding::UTF_8, Encoding::BINARY, Encoding::US_ASCII
432
+ data.dup.force_encoding('UTF-8').scrub
433
+ else
434
+ begin
435
+ data.encode('UTF-8', invalid: :replace, undef: :replace)
436
+ rescue Encoding::ConverterNotFoundError
437
+ data.dup.force_encoding('UTF-8').scrub
438
+ end
439
+ end
440
+ elsif data.is_a?(Hash)
441
+ data.each_with_object({}) { |(key, val), hash| hash[force_utf8_encoding(key)] = force_utf8_encoding(val) }
442
+ elsif data.is_a?(Array)
443
+ data.map { |val| force_utf8_encoding(val) }
218
444
  else
219
445
  data
220
446
  end
@@ -233,20 +459,63 @@ Logtail::Config.instance.debug_logger = ::Logger.new(STDOUT)
233
459
  req = build_request(msgs)
234
460
  if !req.nil?
235
461
  Logtail::Config.instance.debug { "New request placed on queue" }
236
- request_attempt = RequestAttempt.new(req)
462
+ request_attempt = RequestAttempt.new(req, msgs.size)
237
463
  @request_queue.enq(request_attempt)
238
464
  end
239
465
  end
240
466
 
467
+ # Sends the requests in the calling thread, for when no outlet thread delivers them.
468
+ # Returns whether all of them were delivered. A request answered with 408, 429 or 5xx isn't
469
+ # retried, nothing would deliver the retry; one rejected with any other status that isn't
470
+ # 2xx is reported like in {#deliver_requests}. Errors only go to the debug log, also those
471
+ # that aren't StandardErrors (WebMock refuses to connect with one); signals are raised as usual.
472
+ def deliver_synchronously(request_attempts)
473
+ return true if request_attempts.empty?
474
+
475
+ http = build_http
476
+ http.open_timeout = http.read_timeout = SYNCHRONOUS_DELIVERY_TIMEOUT
477
+ begin
478
+ http.start
479
+ rescue ThreadError
480
+ # While Ruby shuts down it refuses new threads, and Net::HTTP (before Ruby 4.0) needs
481
+ # one to time out connecting. Then it connects without a timeout, but only to a host
482
+ # that has answered before.
483
+ raise if @last_resp.nil?
484
+ http.open_timeout = nil
485
+ http.start
486
+ end
487
+ delivered = true
488
+ request_attempts.each do |request_attempt|
489
+ resp = @last_resp = http.request(request_attempt.request)
490
+ next if resp.code.start_with?("2")
491
+
492
+ delivered = false
493
+ Logtail::Config.instance.debug { "Log delivery failed! status: #{resp.code}, body: #{resp.body}" }
494
+ report_rejected_batch(request_attempt, resp) unless resp.code == "408" || resp.code == "429" || resp.code.start_with?("5")
495
+ end
496
+ delivered
497
+ rescue SignalException
498
+ raise
499
+ rescue Exception => e
500
+ Logtail::Config.instance.debug { "Synchronous delivery failed: #{e.message}" }
501
+ false
502
+ ensure
503
+ http.finish if http && http.started?
504
+ end
505
+
241
506
  # Waits on the request queue. This is used in {#flush} to ensure
242
507
  # the log data has been delivered before returning.
243
508
  def wait_on_request_queue
244
- # Wait 20 seconds
245
- 40.times do |i|
509
+ # Wait 5 seconds
510
+ 10.times do |i|
246
511
  if @request_queue.size == 0 && @requests_in_flight == 0
247
512
  Logtail::Config.instance.debug { "Request queue is empty and no requests are in flight, finish waiting" }
248
513
  return true
249
514
  end
515
+ if outlet_stalled?
516
+ Logtail::Config.instance.debug { "The HTTP outlet can't deliver the requests, finish waiting" }
517
+ return false
518
+ end
250
519
  Logtail::Config.instance.debug do
251
520
  "Request size #{@request_queue.size}, reqs in-flight #{@requests_in_flight}, " \
252
521
  "continue waiting (iteration #{i + 1})"
@@ -255,6 +524,30 @@ Logtail::Config.instance.debug_logger = ::Logger.new(STDOUT)
255
524
  end
256
525
  end
257
526
 
527
+ # Whether the outlet thread can't deliver anything while {#close} waits for it: the thread
528
+ # is dead, or the host has never answered and the outlet has already waited to reconnect
529
+ # after a failed connection (@reconnect_wait grows after each wait until a response).
530
+ def outlet_stalled?
531
+ return true unless @request_outlet_thread && @request_outlet_thread.alive?
532
+
533
+ @last_resp.nil? && @reconnect_wait > INITIAL_RECONNECT_WAIT
534
+ end
535
+
536
+ # Delivers the buffered lines in the calling thread, for {#write} when no thread can. After
537
+ # a delivery fails, e.g. to an unreachable host, later lines are dropped so they can't
538
+ # hold up the exit one by one.
539
+ def deliver_late_lines
540
+ msgs = @msg_queue.flush
541
+ return true if msgs.empty?
542
+
543
+ if @late_delivery_failed
544
+ Logtail::Config.instance.debug { "Dropping #{msgs.size} log lines, an earlier synchronous delivery failed" }
545
+ else
546
+ @late_delivery_failed = !deliver_synchronously([RequestAttempt.new(build_request(msgs), msgs.size)])
547
+ end
548
+ true
549
+ end
550
+
258
551
  # Flushes the message queue on an interval. You will notice that {#write} also
259
552
  # flushes the buffer if it is full. This method takes note of this via the
260
553
  # `@last_async_flush` variable as to not flush immediately after a write flush.
@@ -298,15 +591,20 @@ Logtail::Config.instance.debug_logger = ::Logger.new(STDOUT)
298
591
  http
299
592
  end
300
593
 
301
- # Creates a loop that processes the `@request_queue` on an interval.
594
+ # Creates a loop that processes the `@request_queue` on an interval. After a failed
595
+ # connection, or a 429 or 5xx response, it waits before reconnecting, twice as long after
596
+ # every consecutive failure up to {MAX_RECONNECT_WAIT}, so an unreachable host is not
597
+ # retried in a busy loop. A Retry-After header can make the wait longer. A delivered
598
+ # request starts the wait over (see {#deliver_requests}).
302
599
  def request_outlet
303
600
  loop do
304
601
  http = build_http
602
+ connection_healthy = false
305
603
 
306
604
  begin
307
605
  Logtail::Config.instance.debug { "Starting HTTP connection" }
308
606
 
309
- http.start do |conn|
607
+ connection_healthy = http.start do |conn|
310
608
  deliver_requests(conn)
311
609
  end
312
610
  rescue => e
@@ -315,13 +613,20 @@ Logtail::Config.instance.debug_logger = ::Logger.new(STDOUT)
315
613
  Logtail::Config.instance.debug { "Finishing HTTP connection" }
316
614
  http.finish if http.started?
317
615
  end
616
+
617
+ next if connection_healthy
618
+
619
+ Logtail::Config.instance.debug { "Reconnecting in #{@reconnect_wait} seconds" }
620
+ sleep(@reconnect_wait)
621
+ @reconnect_wait = [@reconnect_wait * 2, MAX_RECONNECT_WAIT].min
318
622
  end
319
623
  end
320
624
 
321
625
  # Creates a loop that delivers requests over an open (kept alive) HTTP connection.
322
626
  # If the connection dies, the request is thrown back onto the queue and
323
627
  # the method returns. It is the responsibility of the caller to implement retries
324
- # and establish a new connection.
628
+ # and establish a new connection. A 429 or 5xx response is handled the same way, and
629
+ # a request rejected with any other status is dropped (see {#report_rejected_batch}).
325
630
  def deliver_requests(conn)
326
631
  num_reqs = 0
327
632
 
@@ -330,28 +635,21 @@ Logtail::Config.instance.debug_logger = ::Logger.new(STDOUT)
330
635
  Logtail::Config.instance.debug { "Waiting on next request, threads waiting: #{@request_queue.size}" }
331
636
  end
332
637
 
638
+ # Counted as in flight before it leaves the queue, so close never sees neither
639
+ @requests_in_flight += 1
333
640
  request_attempt = @request_queue.deq
334
641
 
335
642
  if request_attempt.nil?
643
+ @requests_in_flight -= 1
336
644
  sleep(1)
337
645
  else
338
646
  request_attempt.attempted!
339
- @requests_in_flight += 1
340
647
 
341
648
  begin
342
649
  resp = conn.request(request_attempt.request)
343
650
  rescue => e
344
651
  Logtail::Config.instance.debug { "#deliver_requests error: #{e.message}" }
345
-
346
- # Throw the request back on the queue for a retry if it has been attempted less
347
- # than 3 times
348
- if request_attempt.attempts < 3
349
- Logtail::Config.instance.debug { "Request is being retried, #{request_attempt.attempts} previous attempts" }
350
- @request_queue.enq(request_attempt)
351
- else
352
- Logtail::Config.instance.debug { "Request is being dropped, #{request_attempt.attempts} previous attempts" }
353
- end
354
-
652
+ retry_or_drop(request_attempt)
355
653
  return false
356
654
  ensure
357
655
  @requests_in_flight -= 1
@@ -360,20 +658,70 @@ Logtail::Config.instance.debug_logger = ::Logger.new(STDOUT)
360
658
  num_reqs += 1
361
659
 
362
660
  @last_resp = resp
661
+ delivered = resp.code.start_with?("2")
363
662
 
364
663
  Logtail::Config.instance.debug do
365
- if resp.code == "202"
664
+ if delivered
366
665
  "Logs successfully sent! View your logs at https://telemetry.betterstack.com"
367
666
  else
368
667
  "Log delivery failed! status: #{resp.code}, body: #{resp.body}"
369
668
  end
370
669
  end
670
+
671
+ # A request the server didn't read (408, which Better Stack also sends for a new
672
+ # connection that stayed unused too long), too many requests or a server error: retry
673
+ # the request like after a failed connection, without starting the wait over, and
674
+ # wait as long as Retry-After asks.
675
+ if resp.code == "408" || resp.code == "429" || resp.code.start_with?("5")
676
+ retry_or_drop(request_attempt)
677
+ @reconnect_wait = [@reconnect_wait, retry_after(resp)].max
678
+ return false
679
+ end
680
+
681
+ report_rejected_batch(request_attempt, resp) unless delivered
682
+ @reconnect_wait = INITIAL_RECONNECT_WAIT
371
683
  end
372
684
  end
373
685
 
374
686
  true
375
687
  end
376
688
 
689
+ # Throws the request back on the queue for a retry if it has been attempted less
690
+ # than 3 times
691
+ def retry_or_drop(request_attempt)
692
+ if request_attempt.attempts < 3
693
+ Logtail::Config.instance.debug { "Request is being retried, #{request_attempt.attempts} previous attempts" }
694
+ @request_queue.enq(request_attempt)
695
+ else
696
+ Logtail::Config.instance.debug { "Request is being dropped, #{request_attempt.attempts} previous attempts" }
697
+ end
698
+ end
699
+
700
+ # The seconds to wait before a retry that the Retry-After header asks for, given in
701
+ # seconds or as an HTTP date, at most {MAX_RETRY_AFTER}. 0 without a valid header.
702
+ def retry_after(resp)
703
+ value = resp["Retry-After"].to_s.strip
704
+ seconds = value.match?(/\A\d+\z/) ? value.to_i : (Time.httpdate(value) - Time.now).ceil
705
+ seconds.clamp(0, MAX_RETRY_AFTER)
706
+ rescue ArgumentError
707
+ 0
708
+ end
709
+
710
+ # Warns about a batch Better Stack rejected, once per HTTP status in this process. It
711
+ # goes to stderr, never to a Logtail logger, whose lines would be rejected the same way.
712
+ def report_rejected_batch(request_attempt, resp)
713
+ first_rejection = REPORTED_REJECTIONS_LOCK.synchronize do
714
+ !REPORTED_REJECTIONS.include?(resp.code) && REPORTED_REJECTIONS.push(resp.code)
715
+ end
716
+ return unless first_rejection
717
+
718
+ lines = request_attempt.line_count
719
+ status = "HTTP #{resp.code} #{resp.message}".strip
720
+ hint = " - check your source token" if resp.code == "401" || resp.code == "403"
721
+ warn("Logtail: Better Stack rejected #{lines || "some"} log #{lines == 1 ? "line" : "lines"} " \
722
+ "with #{status}#{hint}. Further rejections with this status won't be reported.")
723
+ end
724
+
377
725
  # Builds the `Authorization` header value for HTTP delivery to the Logtail API.
378
726
  def authorization_payload
379
727
  "Bearer #{@source_token}"