log-foundry 0.10.2.dev103__tar.gz → 0.10.2.dev105__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/PKG-INFO +1 -1
  2. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/pyproject.toml +1 -1
  3. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/file.py +78 -2
  4. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/firehose.py +34 -4
  5. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/http.py +83 -4
  6. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/kinesis.py +75 -6
  7. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/pubsub.py +159 -50
  8. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/sentry.py +19 -1
  9. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/sns.py +33 -4
  10. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/sqs.py +36 -2
  11. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/LICENSE +0 -0
  12. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/README.md +0 -0
  13. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/__init__.py +0 -0
  14. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/_diag.py +0 -0
  15. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/_fork.py +0 -0
  16. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/_lifecycle.py +0 -0
  17. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/api.py +0 -0
  18. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/config.py +0 -0
  19. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/console.py +0 -0
  20. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/context.py +0 -0
  21. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/decorator.py +0 -0
  22. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/ids.py +0 -0
  23. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/model.py +0 -0
  24. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/py.typed +0 -0
  25. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/results.py +0 -0
  26. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sanitize.py +0 -0
  27. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/__init__.py +0 -0
  28. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/_batch.py +0 -0
  29. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/_chunk.py +0 -0
  30. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/_retry.py +0 -0
  31. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/_socket.py +0 -0
  32. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/_time.py +0 -0
  33. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/base.py +0 -0
  34. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/callback.py +0 -0
  35. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/clickhouse.py +0 -0
  36. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/datadog.py +0 -0
  37. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/elasticsearch.py +0 -0
  38. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/eventhubs.py +0 -0
  39. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/filtering.py +0 -0
  40. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/honeycomb.py +0 -0
  41. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/kafka.py +0 -0
  42. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/logging_sink.py +0 -0
  43. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/logstash.py +0 -0
  44. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/loki.py +0 -0
  45. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/memory.py +0 -0
  46. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/mongodb.py +0 -0
  47. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/multi.py +0 -0
  48. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/nats.py +0 -0
  49. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/newrelic.py +0 -0
  50. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/null.py +0 -0
  51. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/postgres.py +0 -0
  52. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/rabbitmq.py +0 -0
  53. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/redis.py +0 -0
  54. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/splunk.py +0 -0
  55. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/sqlite.py +0 -0
  56. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/stdout.py +0 -0
  57. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/syslog.py +0 -0
  58. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/sinks/transform.py +0 -0
  59. {log_foundry-0.10.2.dev103 → log_foundry-0.10.2.dev105}/src/log_foundry/worker.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: log-foundry
3
- Version: 0.10.2.dev103
3
+ Version: 0.10.2.dev105
4
4
  Summary: Generate logs for your console and JSON events for downstream consumption.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -74,7 +74,7 @@ keywords = [
74
74
  # vulnerability-reporting channel. The repository is still named `log-forge` — the ORIGINAL name,
75
75
  # which PyPI rejected for the distribution — so these URLs deliberately do not match the package
76
76
  # name. See the note on `name` above before "correcting" them.
77
- version = "0.10.2.dev103"
77
+ version = "0.10.2.dev105"
78
78
 
79
79
  [project.urls]
80
80
  Homepage = "https://github.com/agriffi10/log-forge"
@@ -8,6 +8,7 @@ import threading
8
8
  import time
9
9
  from typing import TextIO
10
10
 
11
+ from log_foundry import _diag
11
12
  from log_foundry.sinks.base import SinkDeliveryError
12
13
 
13
14
  __all__ = ["FileSink", "RotatingFileSink"]
@@ -267,6 +268,7 @@ class RotatingFileSink:
267
268
  self._stream: TextIO = open(path, "a", encoding=self._encoding)
268
269
  self._size = os.path.getsize(path) if os.path.exists(path) else 0
269
270
  self._next_rollover = self._schedule_next()
271
+ self._rotation_failing = False
270
272
  self._closed = False
271
273
  self._lock = threading.Lock()
272
274
 
@@ -278,6 +280,15 @@ class RotatingFileSink:
278
280
  ``_should_rotate`` check and the write that follows it must see the same stream, or a
279
281
  rotation between them sends the line to a closed handle.
280
282
 
283
+ **The batch is flushed before a rotation is attempted** (SPEC-048 FR-006). ``_rotate``
284
+ begins by closing the stream, which flushes it, while this loop otherwise flushes once at
285
+ the end — so under the canonical rotation failure, a full or read-only filesystem, it is
286
+ that flush that raises and the batch's buffered lines are gone before any rename is tried.
287
+ Flushing here means every event of the batch is on disk before the rotation can fail, at
288
+ every one of ``_rotate``'s raise sites rather than only at the renames. A flush that
289
+ raises *here* is deliberately not absorbed: nothing was written, so it is the
290
+ genuinely-total failure the worker's retry exists for.
291
+
281
292
  Args:
282
293
  batch: The events to write.
283
294
 
@@ -285,7 +296,9 @@ class RotatingFileSink:
285
296
  None.
286
297
 
287
298
  Raises:
288
- OSError: If a write, flush or rotation fails.
299
+ OSError: If a write or a flush fails. A failed *rotation* no longer raises; see
300
+ :meth:`_rotate_or_continue`.
301
+ SinkDeliveryError: If the sink is closed.
289
302
  """
290
303
  if not batch:
291
304
  return
@@ -298,7 +311,8 @@ class RotatingFileSink:
298
311
  line = json.dumps(event) + "\n"
299
312
  data = len(line.encode(self._encoding))
300
313
  if self._should_rotate(data):
301
- self._rotate()
314
+ self._stream.flush()
315
+ self._rotate_or_continue()
302
316
  self._stream.write(line)
303
317
  self._size += data
304
318
  self._stream.flush()
@@ -425,6 +439,68 @@ class RotatingFileSink:
425
439
  return True
426
440
  return self._next_rollover is not None and time.monotonic() >= self._next_rollover
427
441
 
442
+ def _rotate_or_continue(self) -> None:
443
+ """Rotates, or absorbs the failure and carries on writing to the un-rotated file.
444
+
445
+ A rotation that raised used to cost the batch twice. ``_rotate`` closes the active stream
446
+ first, so the events already written in this batch were on disk and the ``OSError``
447
+ propagated out of ``emit`` — the worker then re-sent the whole batch and wrote them again.
448
+ Measured: an 8-event batch failing after 3 were written put 11 lines on disk, 3 of them
449
+ duplicates. A persistent failure was worse: the sink kept a **closed** stream and every
450
+ later batch raised a raw ``PermissionError``, which is not a ``SinkDeliveryError`` and has
451
+ no ``losses()`` behind it.
452
+
453
+ Absorbing it costs nothing and duplicates nothing. The active file simply exceeds
454
+ ``max_bytes`` until a rotation succeeds, which is what happens anyway when rotation is
455
+ impossible — the trade SPEC-027 FR-004 already took, that a leaked resource beats a
456
+ corrupt write.
457
+
458
+ ``_next_rollover`` is re-armed as well as ``_size``: ``_rotate`` sets it on its last line,
459
+ so an absorbed failure would otherwise leave a **time** trigger permanently in the past.
460
+
461
+ **The re-arm does not damp the size trigger, and the diagnostic is what carries that.**
462
+ ``_size`` is re-seeded from a file that is now over ``max_bytes``, so ``_should_rotate``'s
463
+ size branch stays true and every subsequent event attempts a rotation again — measured at
464
+ 598 attempts over 600 events. The attempts are cheap and lose nothing, but an unthrottled
465
+ stderr write per event is not: ``PostgresSink._reconnect_if_broken`` records the same rule
466
+ for the same reason, that a diagnostic which floods is one an operator stops reading. So
467
+ the failure is announced **once per outage** and the flag clears on the next successful
468
+ rotation. The remaining per-event attempt is recorded in ``architecture.md`` §12 rather
469
+ than fixed here, because damping it means deferring a rotation the caller asked for.
470
+
471
+ **The reopen can itself raise, and that is not absorbed.** It is the same
472
+ ``open(self._path, "a")`` call ``_rotate`` ends with, so whatever defeats it there —
473
+ a read-only mount, ``EMFILE``, a directory that lost write permission — defeats it here.
474
+ At that point this sink has no stream and cannot continue, so there is nothing to absorb
475
+ *into*; the ``OSError`` reaches ``emit`` and the worker retries the batch, duplicating the
476
+ prefix the pre-rotation flush had already written. That residue is recorded rather than
477
+ fixed: it is unchanged from before SPEC-048, the surviving events are on disk rather than
478
+ lost, and inventing a half-open state to avoid a duplicate would trade a visible
479
+ duplication for a silent loss.
480
+
481
+ Args:
482
+ None.
483
+
484
+ Returns:
485
+ None.
486
+
487
+ Raises:
488
+ OSError: If the *reopen* fails, per the paragraph above. A failed **rotation** is
489
+ absorbed and announced through ``_diag``; the batch continues, nothing is dropped, no
490
+ counter moves, and this class still has no ``losses()``.
491
+ """
492
+ try:
493
+ self._rotate()
494
+ except OSError as err:
495
+ self._stream = open(self._path, "a", encoding=self._encoding)
496
+ self._size = os.path.getsize(self._path) if os.path.exists(self._path) else 0
497
+ self._next_rollover = self._schedule_next()
498
+ if not self._rotation_failing:
499
+ self._rotation_failing = True
500
+ _diag.absorbed("rotating RotatingFileSink", err)
501
+ return
502
+ self._rotation_failing = False
503
+
428
504
  def _rotate(self) -> None:
429
505
  """Closes the active file, shifts and prunes backups, then opens a fresh active file.
430
506
 
@@ -203,6 +203,22 @@ class FirehoseSink:
203
203
  this loop re-sends the records the destination flagged, and the canonical reason it flags
204
204
  them is throttling, which an immediate re-send makes worse.
205
205
 
206
+ **A client exception costs this chunk, never the batch** (SPEC-048 FR-002). It used to
207
+ propagate out of :meth:`emit`, so a failure on chunk N after chunks 1..N-1 had landed made
208
+ the worker re-send the whole batch and duplicate everything already delivered — the exit
209
+ drain, one large batch by construction, is exactly that shape. The guard sits around the
210
+ client call *inside* this loop rather than around ``_send`` in ``emit``, so only the
211
+ records still outstanding at the failing attempt are charged and the ones already accepted
212
+ still count toward the return.
213
+
214
+ A client exception is treated as **provable non-delivery** for the chunk, so a wholly
215
+ failed batch still raises and ``unknown`` is untouched: SPEC-018's "unadjudicable" is a
216
+ property of a *response*, and an exception is not a response. The cost is written down —
217
+ a read timeout means the request went out and the reply was lost, so a re-send may
218
+ duplicate — and taken because an unreachable endpoint is the common case and is exactly
219
+ what the worker's retry exists for, while suppressing the raise would lose every event of
220
+ every batch for a whole outage, silently.
221
+
206
222
  Args:
207
223
  records: One chunk's request entries.
208
224
 
@@ -214,13 +230,27 @@ class FirehoseSink:
214
230
  SPEC-018 settled must never be re-sent.
215
231
 
216
232
  Raises:
217
- Exception: Whatever the client raises.
233
+ None. ``Exception`` is caught rather than ``BaseException``, so a ``KeyboardInterrupt``
234
+ or ``SystemExit`` still reaches the caller (SPEC-025 FR-004).
218
235
  """
219
236
  sent = len(records)
220
237
  for attempt in range(self.max_retries + 1):
221
- response = self.client.put_record_batch(
222
- DeliveryStreamName=self.delivery_stream, Records=records
223
- )
238
+ try:
239
+ response = self.client.put_record_batch(
240
+ DeliveryStreamName=self.delivery_stream, Records=records
241
+ )
242
+ except Exception as err:
243
+ if attempt < self.max_retries:
244
+ wait(_BACKOFF_BASE * (2**attempt), self.log_foundry_stop_signal)
245
+ continue
246
+ with self._counter_lock:
247
+ self.failed += len(records)
248
+ _diag.lost(
249
+ "record",
250
+ len(records),
251
+ f"FirehoseSink, {self.max_retries + 1} attempt(s), {type(err).__name__}",
252
+ )
253
+ return sent - len(records)
224
254
  if not response.get("FailedPutCount"):
225
255
  return sent
226
256
  results = usable_results(response.get("RequestResponses"))
@@ -125,6 +125,77 @@ inside a typical serverless timeout while leaving room for the request itself.
125
125
  """
126
126
 
127
127
 
128
+ class _NoRedirect(urllib.request.HTTPRedirectHandler):
129
+ """Declines every redirect, so a ``3xx`` reaches the sink as an ``HTTPError`` (SPEC-048).
130
+
131
+ ``urlopen``'s default opener follows ``301``, ``302`` and ``303`` on a ``POST`` by rewriting
132
+ the method to ``GET`` and dropping the body, while keeping every header — so a collector
133
+ behind a load balancer that redirects to ``https://`` silently discarded every batch and
134
+ forwarded the ``Authorization`` header to whatever host the redirect named, and the sink read
135
+ the redirect target's ``200`` as delivery. ``307`` and ``308`` were already refused for a
136
+ ``POST`` by the base class.
137
+ """
138
+
139
+ def redirect_request(
140
+ self,
141
+ req: urllib.request.Request,
142
+ fp: Any,
143
+ code: int,
144
+ msg: str,
145
+ headers: Any,
146
+ newurl: str,
147
+ ) -> None:
148
+ """Declines the redirect, leaving the default error handler to raise ``HTTPError``.
149
+
150
+ Returning ``None`` is ``urllib``'s documented "I cannot handle this, let another handler
151
+ try", and no other handler does, so ``HTTPDefaultErrorHandler`` raises — which
152
+ :meth:`HTTPSink._attempt` already unifies into a status, putting a ``3xx`` on the same
153
+ counted, announced path as a ``4xx``.
154
+
155
+ Args:
156
+ req: The request that was redirected.
157
+ fp: The response file object.
158
+ code: The ``3xx`` status.
159
+ msg: The status message.
160
+ headers: The response headers.
161
+ newurl: The redirect target, deliberately unused and never contacted.
162
+
163
+ Returns:
164
+ None, always — which is what declines the redirect. A bare ``return`` rather than
165
+ ``return None`` only because ``ruff``'s ``RET501`` refuses the explicit form; the value
166
+ urllib receives is the same.
167
+
168
+ Raises:
169
+ None.
170
+ """
171
+ return
172
+
173
+
174
+ def _no_redirect_opener() -> Callable[..., Any]:
175
+ """Builds an opener that refuses redirects, for a sink the caller gave no ``opener=``.
176
+
177
+ Built per sink at construction rather than once at import, for two reasons. ``python.md`` §15
178
+ forbids a module doing real work at import time, which is why the ``StdoutSink`` default and
179
+ the worker are lazy too. And ``build_opener`` snapshots ``ProxyHandler``'s environment, so an
180
+ import-time opener would pin whatever ``http_proxy`` said when ``log_foundry`` was first
181
+ imported — a sink built after the application sets its proxy would silently ignore it, where
182
+ ``urlopen``'s lazily-built global opener would not. Construction time is when the caller is
183
+ standing there, so it is the honest moment to read the environment.
184
+
185
+ A sink is constructed once, so the cost is one object per sink and never per request.
186
+
187
+ Args:
188
+ None.
189
+
190
+ Returns:
191
+ A ``urlopen``-shaped callable that raises ``HTTPError`` on any ``3xx``.
192
+
193
+ Raises:
194
+ None.
195
+ """
196
+ return urllib.request.build_opener(_NoRedirect()).open
197
+
198
+
128
199
  def merge_headers(base: dict[str, str], http_kwargs: dict[str, object]) -> dict[str, str]:
129
200
  """Merges a platform sink's own headers with any caller-supplied ones, caller winning.
130
201
 
@@ -250,7 +321,9 @@ class HTTPSink:
250
321
  max_batch_bytes: Bytes one request's body may reach, or ``None`` for this class's
251
322
  :attr:`MAX_BATCH_BYTES`. Floored at one for the same reason.
252
323
  opener: A ``urlopen``-shaped callable, which a test can inject to assert on the
253
- request without any network access.
324
+ request without any network access. Defaults to :func:`_no_redirect_opener` rather
325
+ than ``urlopen`` (SPEC-048 FR-001); an injected one is used exactly as given, since
326
+ it is the caller's object and the library does not reshape it.
254
327
 
255
328
  Returns:
256
329
  None.
@@ -274,7 +347,7 @@ class HTTPSink:
274
347
  max_batch_bytes if max_batch_bytes is not None else self.MAX_BATCH_BYTES, 1
275
348
  )
276
349
  self.log_foundry_stop_signal: threading.Event | None = None
277
- self._opener = opener if opener is not None else urllib.request.urlopen
350
+ self._opener = opener if opener is not None else _no_redirect_opener()
278
351
  self.failed = 0
279
352
  self.dropped_oversized = 0
280
353
  self._counter_lock = threading.Lock()
@@ -796,7 +869,13 @@ class HTTPSink:
796
869
  ) -> tuple[dict[str, str], bytes]:
797
870
  """Builds the final header map and the optionally gzipped body.
798
871
 
799
- Caller-provided headers override the sink's defaults (FR-002).
872
+ Caller-provided headers override the sink's defaults (FR-002), and that now includes
873
+ ``Content-Encoding``: ``gzip=True`` used to overwrite one the caller had set, which made
874
+ it the single header of theirs that did not win. A caller's own value both survives *and*
875
+ suppresses the compression, because a header saying ``identity`` over a gzipped body
876
+ makes the destination decode garbage. The test is against ``self._headers`` rather than
877
+ the merged map, so a per-request ``extra_headers`` value — which sits *beneath* the
878
+ caller's own — does not switch compression off (SPEC-048 FR-007).
800
879
 
801
880
  Args:
802
881
  body: The serialized batch.
@@ -814,7 +893,7 @@ class HTTPSink:
814
893
  headers.update(extra_headers)
815
894
  headers.update(self._headers)
816
895
  data = body
817
- if self.gzip:
896
+ if self.gzip and "Content-Encoding" not in self._headers:
818
897
  data = _gzip.compress(body)
819
898
  headers["Content-Encoding"] = "gzip"
820
899
  self._apply_auth(headers)
@@ -180,17 +180,18 @@ class KinesisSink:
180
180
  records: list[dict[str, Any]] = []
181
181
  for event in batch:
182
182
  data = json.dumps(event).encode("utf-8")
183
- if len(data) > self.MAX_RECORD_BYTES:
183
+ key = _partition_key(str(event.get(self.partition_key_field) or "log-foundry"))
184
+ charged = len(data) + len(key.encode("utf-8"))
185
+ if charged > self.MAX_RECORD_BYTES:
184
186
  with self._counter_lock:
185
187
  self.dropped_oversized += 1
186
188
  _diag.lost(
187
189
  "event",
188
190
  1,
189
- f"KinesisSink, {len(data)} bytes exceeds the "
191
+ f"KinesisSink, {charged} bytes with its partition key exceeds the "
190
192
  f"{self.MAX_RECORD_BYTES}-byte per-record limit",
191
193
  )
192
194
  continue
193
- key = str(event.get(self.partition_key_field) or "log-foundry")[:256]
194
195
  records.append({"Data": data, "PartitionKey": key})
195
196
  return records
196
197
 
@@ -201,6 +202,22 @@ class KinesisSink:
201
202
  this loop re-sends the records the destination flagged, and the canonical reason it flags
202
203
  them is throttling, which an immediate re-send makes worse.
203
204
 
205
+ **A client exception costs this chunk, never the batch** (SPEC-048 FR-002). It used to
206
+ propagate out of :meth:`emit`, so a failure on chunk N after chunks 1..N-1 had landed made
207
+ the worker re-send the whole batch and duplicate everything already delivered — the exit
208
+ drain, one large batch by construction, is exactly that shape. The guard sits around the
209
+ client call *inside* this loop rather than around ``_send`` in ``emit``, so only the
210
+ records still outstanding at the failing attempt are charged and the ones already accepted
211
+ still count toward the return.
212
+
213
+ A client exception is treated as **provable non-delivery** for the chunk, so a wholly
214
+ failed batch still raises and ``unknown`` is untouched: SPEC-018's "unadjudicable" is a
215
+ property of a *response*, and an exception is not a response. The cost is written down —
216
+ a read timeout means the request went out and the reply was lost, so a re-send may
217
+ duplicate — and taken because an unreachable endpoint is the common case and is exactly
218
+ what the worker's retry exists for, while suppressing the raise would lose every event of
219
+ every batch for a whole outage, silently.
220
+
204
221
  Args:
205
222
  records: One chunk's request entries.
206
223
 
@@ -212,11 +229,27 @@ class KinesisSink:
212
229
  SPEC-018 settled must never be re-sent.
213
230
 
214
231
  Raises:
215
- Exception: Whatever the client raises.
232
+ None. ``Exception`` is caught rather than ``BaseException``, so a ``KeyboardInterrupt``
233
+ or ``SystemExit`` still reaches the caller (SPEC-025 FR-004).
216
234
  """
217
235
  sent = len(records)
218
236
  for attempt in range(self.max_retries + 1):
219
- response = self.client.put_records(StreamName=self.stream_name, Records=records)
237
+ try:
238
+ response = self.client.put_records(
239
+ StreamName=self.stream_name, Records=records
240
+ )
241
+ except Exception as err:
242
+ if attempt < self.max_retries:
243
+ wait(_BACKOFF_BASE * (2**attempt), self.log_foundry_stop_signal)
244
+ continue
245
+ with self._counter_lock:
246
+ self.failed += len(records)
247
+ _diag.lost(
248
+ "record",
249
+ len(records),
250
+ f"KinesisSink, {self.max_retries + 1} attempt(s), {type(err).__name__}",
251
+ )
252
+ return sent - len(records)
220
253
  if not response.get("FailedRecordCount"):
221
254
  return sent
222
255
  results = usable_results(response.get("Records"))
@@ -248,6 +281,38 @@ class KinesisSink:
248
281
  return 0
249
282
 
250
283
 
284
+ MAX_PARTITION_KEY_BYTES = 256
285
+ """The service's ceiling on a ``PutRecords`` partition key, in UTF-8 bytes."""
286
+
287
+
288
+ def _partition_key(raw: str) -> str:
289
+ """Bounds a partition key to the service's 256 **bytes**, never 256 characters.
290
+
291
+ Both encodes carry an ``errors=`` and both are load-bearing. ``sanitize.coerce`` passes a lone
292
+ surrogate through unchanged, so a bare ``encode("utf-8")`` on a caller's ``trace_id`` raises
293
+ ``UnicodeEncodeError`` — a raw exception out of ``emit``, which is the failure SPEC-048 exists
294
+ to remove rather than introduce. ``errors="ignore"`` on the decode drops a character the byte
295
+ cut landed inside, so the result always decodes cleanly.
296
+
297
+ ``sanitize.truncate_str`` is the inventoried byte-bounded clipper and is deliberately **not**
298
+ reused: it appends a truncation marker, and a marker inside a partition key changes the shard
299
+ the record lands on.
300
+
301
+ Args:
302
+ raw: The derived key.
303
+
304
+ Returns:
305
+ The key, at most :data:`MAX_PARTITION_KEY_BYTES` UTF-8 bytes and always valid UTF-8.
306
+
307
+ Raises:
308
+ None.
309
+ """
310
+ encoded = raw.encode("utf-8", errors="replace")
311
+ if len(encoded) <= MAX_PARTITION_KEY_BYTES:
312
+ return encoded.decode("utf-8")
313
+ return encoded[:MAX_PARTITION_KEY_BYTES].decode("utf-8", errors="ignore")
314
+
315
+
251
316
  def _record_size(record: dict[str, Any]) -> int:
252
317
  """Measures one request entry, partition key included (SPEC-038 FR-009).
253
318
 
@@ -256,6 +321,10 @@ def _record_size(record: dict[str, Any]) -> int:
256
321
  service reject a chunk this sink believed was inside the budget. ``SQSSink`` charges its FIFO
257
322
  ids for the same reason and records the same rationale.
258
323
 
324
+ The charge is the key's **UTF-8 byte length**, not its character count (SPEC-048 FR-003): the
325
+ two differ for any non-ASCII key, and the service bills bytes. No ``errors=`` is needed here,
326
+ because :func:`_partition_key` already built this value and its output always encodes cleanly.
327
+
259
328
  This applies to Kinesis alone: a Firehose record is ``{"Data": data}`` and that API has no
260
329
  partition key, so there is nothing there to charge.
261
330
 
@@ -268,4 +337,4 @@ def _record_size(record: dict[str, Any]) -> int:
268
337
  Raises:
269
338
  None.
270
339
  """
271
- return len(record["Data"]) + len(record["PartitionKey"])
340
+ return len(record["Data"]) + len(record["PartitionKey"].encode("utf-8"))
@@ -315,62 +315,172 @@ class GooglePubSubSink:
315
315
  """
316
316
  if not overflow:
317
317
  return
318
- deadline = time.monotonic() + self.overflow_timeout
319
- unresolved: list[Any] = []
320
- for index, future in enumerate(overflow):
318
+ expired, unboundable = self._resolve_within(
319
+ overflow, time.monotonic() + self.overflow_timeout, heed_stop=True
320
+ )
321
+ unresolved = expired + unboundable
322
+ if not unresolved:
323
+ return
324
+ with self._futures_lock:
325
+ if not self._closed:
326
+ self._futures[:0] = unresolved
327
+ return
328
+ self._drain_pending(unresolved)
329
+
330
+ def _resolve_within(
331
+ self, pending: list[Any], deadline: float, *, heed_stop: bool
332
+ ) -> tuple[list[Any], list[Any]]:
333
+ """Waits on each future until the deadline, splitting what did not settle in two.
334
+
335
+ One deadline covers the whole list rather than a timeout per future: at the shipped
336
+ ``max_pending`` a per-future wait is not a bound at all (SPEC-038's rule that a bound
337
+ applied per item is ``n x timeout``). ``_futures_lock`` is never held across a
338
+ ``result()``, because ``emit`` takes it per event and an application thread on the orphan
339
+ path would block behind it.
340
+
341
+ Args:
342
+ pending: The futures to resolve, already removed from the pending list.
343
+ deadline: A ``time.monotonic()`` reading the pass must not run past.
344
+ heed_stop: Whether a set stop signal also ends the pass. **True only on the flush
345
+ path.** ``Worker.shutdown`` sets that event *before* closing the sink inline, so a
346
+ close that heeded it would abandon everything on every ordinary shutdown -- SPEC-038's
347
+ rule that a shutdown shortens a *wait* and must never skip *work*, and the exit drain
348
+ is the one path a serverless process has.
349
+
350
+ Returns:
351
+ The futures whose wait expired, and separately the ones that cannot be waited on within
352
+ a timeout at all. They are split because they mean opposite things: an expired future is
353
+ unconfirmed, while an *unboundable* one is a healthy publish this pass simply cannot
354
+ poll, and counting the second as loss invents it (SPEC-036 measured three of four
355
+ healthy publishes reported ``failed`` that way).
356
+
357
+ Raises:
358
+ None.
359
+ """
360
+ expired: list[Any] = []
361
+ unboundable: list[Any] = []
362
+ for index, future in enumerate(pending):
321
363
  settled = False
322
- while not self._out_of_time(deadline):
364
+ unpollable = False
365
+ while not self._past(deadline, heed_stop=heed_stop):
323
366
  began = time.monotonic()
324
367
  slice_ = min(deadline - began, _POLL_INTERVAL)
325
368
  try:
326
369
  settled = self._resolve(future, slice_)
327
370
  except _Unboundable:
371
+ unpollable = True
328
372
  break
329
373
  if settled:
330
374
  break
331
375
  wait(slice_ - (time.monotonic() - began), self.log_foundry_stop_signal)
332
- if settled:
333
- continue
334
- unresolved.append(future)
335
- if self._out_of_time(deadline):
336
- unresolved.extend(overflow[index + 1 :])
376
+ else:
377
+ self._classify_remainder(pending[index:], expired, unboundable)
337
378
  break
338
- if not unresolved:
339
- return
340
- with self._futures_lock:
341
- if not self._closed:
342
- self._futures[:0] = unresolved
343
- return
344
- for future in unresolved:
345
- self._resolve(future)
379
+ if unpollable:
380
+ unboundable.append(future)
381
+ return expired, unboundable
382
+
383
+ def _classify_remainder(
384
+ self, remaining: list[Any], expired: list[Any], unboundable: list[Any]
385
+ ) -> None:
386
+ """Sorts the futures a deadline never reached into expired and unboundable.
346
387
 
347
- def _out_of_time(self, deadline: float) -> bool:
348
- """Reports whether the overflow pass must stop, on the deadline or on a shutdown.
388
+ The pass gives up on the whole remainder when its one deadline expires, and the two
389
+ outcomes must still be told apart: an expired future is unconfirmed and is counted, while
390
+ an **unboundable** one is a healthy publish that simply cannot be polled within a timeout,
391
+ and counting it invents loss. A blanket ``expired.extend(...)`` here charged three healthy
392
+ publishes as lost behind one stalled future, which is SPEC-036's measured defect
393
+ reintroduced by the fix that cites it.
349
394
 
350
- The stop signal is read here rather than once before the loop: read once, a shutdown
351
- arriving mid-pass was not noticed until every future had been waited on.
395
+ The probe is a zero-second wait, so it costs nothing against a deadline that has already
396
+ gone: a future whose ``result()`` takes no ``timeout`` raises ``_Unboundable`` on the way
397
+ in, and everything else either settles immediately or reports itself still in flight.
352
398
 
353
399
  Args:
354
- deadline: The monotonic time the pass may not run past.
400
+ remaining: The futures the pass did not reach, the current one first.
401
+ expired: The unconfirmed list, appended to in place.
402
+ unboundable: The cannot-be-polled list, appended to in place.
355
403
 
356
404
  Returns:
357
- True when no further waiting may happen.
405
+ None.
406
+
407
+ Raises:
408
+ None.
409
+ """
410
+ for future in remaining:
411
+ try:
412
+ if not self._resolve(future, 0):
413
+ expired.append(future)
414
+ except _Unboundable:
415
+ unboundable.append(future)
416
+
417
+ def _past(self, deadline: float, *, heed_stop: bool) -> bool:
418
+ """Reports whether a resolution pass must stop.
419
+
420
+ Args:
421
+ deadline: A ``time.monotonic()`` reading.
422
+ heed_stop: Whether a set stop signal also ends the pass.
423
+
424
+ Returns:
425
+ True on the deadline, or on the stop signal when this caller heeds it.
358
426
 
359
427
  Raises:
360
428
  None.
361
429
  """
362
430
  if time.monotonic() >= deadline:
363
431
  return True
432
+ if not heed_stop:
433
+ return False
364
434
  stop = self.log_foundry_stop_signal
365
435
  return stop is not None and stop.is_set()
366
436
 
437
+ def _drain_pending(self, pending: list[Any]) -> None:
438
+ """Resolves a swapped-out list within the close bound, counting what it abandons.
439
+
440
+ The tail every close-race site shares: :meth:`close`, and the two branches where a close
441
+ lands while another pass is mid-flight. Each used to resolve its leftovers with
442
+ ``timeout=None``, which is the unbounded wait ``Worker.shutdown`` performs inline against
443
+ a client publish deadline of 600 s. ``_await_overflow``'s copy is the one that mattered
444
+ most: it runs on whichever thread called ``emit``, which on the orphan path is an
445
+ application thread.
446
+
447
+ Unboundable futures still get ``timeout=None``, because that is the only wait they accept
448
+ and :meth:`_resolve` counts whatever they resolve to.
449
+
450
+ Args:
451
+ pending: The futures this caller owns and nothing else will resolve.
452
+
453
+ Returns:
454
+ None.
455
+
456
+ Raises:
457
+ None.
458
+ """
459
+ expired, unboundable = self._resolve_within(
460
+ pending, time.monotonic() + self.overflow_timeout, heed_stop=False
461
+ )
462
+ for future in unboundable:
463
+ self._resolve(future)
464
+ if not expired:
465
+ return
466
+ with self._counter_lock:
467
+ self.failed += len(expired)
468
+ _diag.lost(
469
+ "event",
470
+ len(expired),
471
+ f"GooglePubSubSink, {len(expired)} publish(es) still in flight when the "
472
+ f"{self.overflow_timeout}s close bound expired",
473
+ )
474
+
367
475
  def _resolve(self, future: Any, timeout: float | None = None) -> bool:
368
476
  """Waits for one publish to settle, counting and announcing a failure.
369
477
 
370
478
  Args:
371
479
  future: The publish future to resolve.
372
- timeout: Seconds to wait, or ``None`` to wait indefinitely — which is what ``close``
373
- does, and what a shutdown defers this to.
480
+ timeout: Seconds to wait, or ``None`` to wait indefinitely. Since SPEC-048 FR-004 that
481
+ is what an *unboundable* future gets, not what ``close`` does: ``close`` bounds itself
482
+ on ``overflow_timeout`` through :meth:`_drain_pending`, because it runs inline inside
483
+ ``Worker.shutdown`` against a client publish deadline of 600 s.
374
484
 
375
485
  Returns:
376
486
  True when the future settled, False when a *bounded* wait expired with it still in
@@ -450,26 +560,10 @@ class GooglePubSubSink:
450
560
  if not pending:
451
561
  return
452
562
 
453
- deadline = time.monotonic() + self.overflow_timeout
454
- unresolved: list[Any] = []
455
- for index, future in enumerate(pending):
456
- settled = False
457
- while not self._out_of_time(deadline):
458
- began = time.monotonic()
459
- slice_ = min(deadline - began, _POLL_INTERVAL)
460
- try:
461
- settled = self._resolve(future, slice_)
462
- except _Unboundable:
463
- break
464
- if settled:
465
- break
466
- wait(slice_ - (time.monotonic() - began), self.log_foundry_stop_signal)
467
- if settled:
468
- continue
469
- unresolved.append(future)
470
- if self._out_of_time(deadline):
471
- unresolved.extend(pending[index + 1 :])
472
- break
563
+ expired, unboundable = self._resolve_within(
564
+ pending, time.monotonic() + self.overflow_timeout, heed_stop=True
565
+ )
566
+ unresolved = expired + unboundable
473
567
 
474
568
  if not unresolved:
475
569
  return
@@ -478,14 +572,30 @@ class GooglePubSubSink:
478
572
  if not closed:
479
573
  self._futures[:0] = unresolved
480
574
  if closed:
481
- for future in unresolved:
482
- self._resolve(future)
575
+ self._drain_pending(unresolved)
483
576
  raise SinkDeliveryError(
484
577
  f"GooglePubSubSink flushed with {len(unresolved)} publish(es) still in flight"
485
578
  )
486
579
 
487
580
  def close(self) -> None:
488
- """Resolves all pending publish futures, counting and logging errors (FR-008).
581
+ """Resolves the pending publish futures within a bound, counting the rest (FR-008).
582
+
583
+ **Bounded since SPEC-048 FR-004.** It used to wait ``timeout=None`` per future, and
584
+ ``Worker.shutdown`` closes the live sink inline, so one unreachable destination held
585
+ process exit for the client's 600 s publish deadline per future. The bound is
586
+ ``overflow_timeout`` on the monotonic clock and deliberately **not** the stop signal;
587
+ :meth:`_resolve_within` records why.
588
+
589
+ **The bound does not cover a future that cannot be waited on within a timeout.** One whose
590
+ ``result()`` takes no ``timeout`` argument is resolved unbounded, because that is the only
591
+ wait it accepts and SPEC-036 measured that counting it instead invents loss on publishes
592
+ that were going to succeed. So a client handing out unboundable futures that also do not
593
+ settle still holds the close for its own deadline, once per future: measured at 27.0 s for
594
+ nine futures against a 3 s stand-in, where the bounded path took 2.0 s for sixty.
595
+ ``google-cloud-pubsub``'s own future accepts a timeout, so this is reachable only through
596
+ an injected ``client=`` that does not — but ``client=`` is a frozen public parameter, which
597
+ is why it is written down here rather than assumed away. Recorded in ``architecture.md``
598
+ §12; it is strictly better than before, where **every** future took this path.
489
599
 
490
600
  Idempotent. The pending list is swapped out under a lock rather than iterated and then
491
601
  cleared (SPEC-028 FR-002): ``emit`` appends to it from any thread, so the old
@@ -513,8 +623,7 @@ class GooglePubSubSink:
513
623
  with self._futures_lock:
514
624
  self._closed = True
515
625
  pending, self._futures = self._futures, []
516
- for future in pending:
517
- self._resolve(future)
626
+ self._drain_pending(pending)
518
627
 
519
628
 
520
629
  def _has_settled(future: Any) -> bool:
@@ -252,7 +252,21 @@ class SentrySink:
252
252
  flush()
253
253
 
254
254
  def close(self) -> None:
255
- """Forwards to the HTTP fallback, whose own close releases nothing (FR-012).
255
+ """Pushes the SDK transport, then forwards to the HTTP fallback (FR-012, SPEC-048 FR-005).
256
+
257
+ **The flush comes first, and without it a terminal ``shutdown()`` stranded everything.**
258
+ ``capture_event`` hands to the SDK's background transport and returns, so a process that
259
+ calls ``shutdown()`` and freezes — the whole of the serverless path, where the SDK's own
260
+ timer never fires again — left every captured event in the SDK's worker. :meth:`flush`
261
+ already pushes that queue; this method reached only the ``urllib`` fallback, which holds
262
+ nothing. Measured before the fix: 25 events captured, zero ``flush`` calls seen by the
263
+ client during ``close()``.
264
+
265
+ It is absorbed, because this is an isolation boundary and a failing flush must not stop
266
+ the release below it. And it is **not** suppressed on a repeat close: this sink adds no
267
+ post-close guard (SPEC-032 FR-003), so events can legitimately be captured between two
268
+ closes, and a flag suppressing the second flush would strand exactly what this exists to
269
+ un-strand. A repeat flush of a drained queue is a no-op.
256
270
 
257
271
  Idempotent, and it releases nothing — which is why the class docstring's post-close claim
258
272
  holds despite this method calling a ``close()``. ``HTTPSink.close`` is a documented no-op,
@@ -270,6 +284,10 @@ class SentrySink:
270
284
  Raises:
271
285
  None.
272
286
  """
287
+ try:
288
+ self.flush()
289
+ except Exception as err:
290
+ _diag.absorbed("closing SentrySink", err)
273
291
  if self._http is not None:
274
292
  _lifecycle.release(self._http, owner=self)
275
293
 
@@ -153,6 +153,21 @@ class SNSSink:
153
153
  this loop re-sends the entries the destination flagged, and the canonical reason it flags
154
154
  them is throttling, which an immediate re-send makes worse.
155
155
 
156
+ **A client exception costs this chunk, never the batch** (SPEC-048 FR-002). It used to
157
+ propagate out of :meth:`emit`, so a failure on chunk N after chunks 1..N-1 had landed made
158
+ the worker re-send the whole batch and duplicate everything already delivered — the exit
159
+ drain, one large batch by construction, is exactly that shape. The guard sits around the
160
+ client call *inside* this loop rather than around ``_send`` in ``emit``, so only the
161
+ messages still outstanding at the failing attempt are charged and the ones already accepted
162
+ still count toward the return.
163
+
164
+ A client exception is treated as **provable non-delivery** for the chunk, so a wholly
165
+ failed batch still raises. The cost is written down —
166
+ a read timeout means the request went out and the reply was lost, so a re-send may
167
+ duplicate — and taken because an unreachable endpoint is the common case and is exactly
168
+ what the worker's retry exists for, while suppressing the raise would lose every event of
169
+ every batch for a whole outage, silently.
170
+
156
171
  Args:
157
172
  bodies: One chunk's serialized events.
158
173
 
@@ -161,14 +176,28 @@ class SNSSink:
161
176
  success. A chunk whose entries all failed contributes zero.
162
177
 
163
178
  Raises:
164
- Exception: Whatever the client raises.
179
+ None. ``Exception`` is caught rather than ``BaseException``, so a ``KeyboardInterrupt``
180
+ or ``SystemExit`` still reaches the caller (SPEC-025 FR-004).
165
181
  """
166
182
  sent = len(bodies)
167
183
  entries = [{"Id": str(i), "Message": body} for i, body in enumerate(bodies)]
168
184
  for attempt in range(self.max_retries + 1):
169
- response = self.client.publish_batch(
170
- TopicArn=self.topic_arn, PublishBatchRequestEntries=entries
171
- )
185
+ try:
186
+ response = self.client.publish_batch(
187
+ TopicArn=self.topic_arn, PublishBatchRequestEntries=entries
188
+ )
189
+ except Exception as err:
190
+ if attempt < self.max_retries:
191
+ wait(_BACKOFF_BASE * (2**attempt), self.log_foundry_stop_signal)
192
+ continue
193
+ with self._counter_lock:
194
+ self.failed += len(entries)
195
+ _diag.lost(
196
+ "message",
197
+ len(entries),
198
+ f"SNSSink, {self.max_retries + 1} attempt(s), {type(err).__name__}",
199
+ )
200
+ return sent - len(entries)
172
201
  failed = response.get("Failed", [])
173
202
  if not failed:
174
203
  return sent
@@ -337,6 +337,24 @@ class SQSSink:
337
337
  alone in re-sending immediately, while its own docstring named throttling as the
338
338
  retryable case, which is exactly what an instant retry makes worse.
339
339
 
340
+ **A client exception costs this chunk, never the batch** (SPEC-048 FR-002). It used to
341
+ propagate out of :meth:`emit`, so a failure on chunk N after chunks 1..N-1 had landed made
342
+ the worker re-send the whole batch and duplicate everything already in the queue — the
343
+ exit drain, one large batch by construction, is exactly that shape. The guard sits around
344
+ the client call *inside* this loop rather than around ``_send`` in ``emit``, because by
345
+ the time an exception is raised ``accepted`` may already be non-zero and ``entries``
346
+ narrowed to the retryable subset: charging the whole chunk outside would report a partial
347
+ success as "nothing delivered" and provoke the very retry this removes.
348
+
349
+ A client exception is treated as **provable non-delivery** for the chunk, so it feeds the
350
+ recoverable term and a wholly-failed batch still raises. That is not free — a read timeout
351
+ means the request went out and the *response* was lost, so entries may have landed and a
352
+ re-send duplicates them, which is SPEC-018's "cannot prove nothing landed". It is taken on
353
+ expected cost: an unreachable or refused endpoint is the common client exception and is
354
+ exactly what the worker's retry exists for, and suppressing the raise for it would lose
355
+ every event of every batch for the whole outage, silently. ``boto3``'s own ``max_attempts``
356
+ retry already carries the duplication property, so this does not introduce it.
357
+
340
358
  Args:
341
359
  prepared: One chunk's serialized, costed entries.
342
360
 
@@ -346,7 +364,8 @@ class SQSSink:
346
364
  recover. A chunk SQS rejected wholesale as invalid reports False there.
347
365
 
348
366
  Raises:
349
- Exception: Whatever the client raises.
367
+ None. ``Exception`` is caught rather than ``BaseException``, so a ``KeyboardInterrupt``
368
+ or ``SystemExit`` still reaches the caller (SPEC-025 FR-004).
350
369
  """
351
370
  entries: list[dict[str, str]] = []
352
371
  for i, item in enumerate(prepared):
@@ -358,7 +377,22 @@ class SQSSink:
358
377
  entries.append(entry)
359
378
  accepted = 0
360
379
  for attempt in range(self.max_retries + 1):
361
- response = self.client.send_message_batch(QueueUrl=self.queue_url, Entries=entries)
380
+ try:
381
+ response = self.client.send_message_batch(
382
+ QueueUrl=self.queue_url, Entries=entries
383
+ )
384
+ except Exception as err:
385
+ if attempt < self.max_retries:
386
+ wait(_BACKOFF_BASE * (2**attempt), self.log_foundry_stop_signal)
387
+ continue
388
+ with self._counter_lock:
389
+ self.failed += len(entries)
390
+ _diag.lost(
391
+ "message",
392
+ len(entries),
393
+ f"SQSSink, {self.max_retries + 1} attempt(s), {type(err).__name__}",
394
+ )
395
+ return accepted, True
362
396
  failed = response.get("Failed", [])
363
397
  failed_ids = {item.get("Id") for item in failed}
364
398
  accepted += sum(1 for entry in entries if entry["Id"] not in failed_ids)