log-foundry 0.10.2.dev102__tar.gz → 0.10.2.dev106__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/PKG-INFO +1 -1
  2. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/pyproject.toml +1 -1
  3. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/_lifecycle.py +180 -14
  4. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/decorator.py +30 -7
  5. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/file.py +78 -2
  6. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/firehose.py +34 -4
  7. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/http.py +83 -4
  8. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/kinesis.py +75 -6
  9. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/pubsub.py +159 -50
  10. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/sentry.py +19 -1
  11. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/sns.py +33 -4
  12. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/sqs.py +36 -2
  13. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/worker.py +223 -24
  14. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/LICENSE +0 -0
  15. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/README.md +0 -0
  16. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/__init__.py +0 -0
  17. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/_diag.py +0 -0
  18. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/_fork.py +0 -0
  19. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/api.py +0 -0
  20. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/config.py +0 -0
  21. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/console.py +0 -0
  22. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/context.py +0 -0
  23. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/ids.py +0 -0
  24. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/model.py +0 -0
  25. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/py.typed +0 -0
  26. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/results.py +0 -0
  27. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sanitize.py +0 -0
  28. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/__init__.py +0 -0
  29. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/_batch.py +0 -0
  30. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/_chunk.py +0 -0
  31. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/_retry.py +0 -0
  32. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/_socket.py +0 -0
  33. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/_time.py +0 -0
  34. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/base.py +0 -0
  35. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/callback.py +0 -0
  36. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/clickhouse.py +0 -0
  37. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/datadog.py +0 -0
  38. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/elasticsearch.py +0 -0
  39. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/eventhubs.py +0 -0
  40. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/filtering.py +0 -0
  41. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/honeycomb.py +0 -0
  42. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/kafka.py +0 -0
  43. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/logging_sink.py +0 -0
  44. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/logstash.py +0 -0
  45. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/loki.py +0 -0
  46. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/memory.py +0 -0
  47. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/mongodb.py +0 -0
  48. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/multi.py +0 -0
  49. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/nats.py +0 -0
  50. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/newrelic.py +0 -0
  51. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/null.py +0 -0
  52. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/postgres.py +0 -0
  53. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/rabbitmq.py +0 -0
  54. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/redis.py +0 -0
  55. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/splunk.py +0 -0
  56. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/sqlite.py +0 -0
  57. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/stdout.py +0 -0
  58. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/syslog.py +0 -0
  59. {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/transform.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: log-foundry
3
- Version: 0.10.2.dev102
3
+ Version: 0.10.2.dev106
4
4
  Summary: Generate logs for your console and JSON events for downstream consumption.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -74,7 +74,7 @@ keywords = [
74
74
  # vulnerability-reporting channel. The repository is still named `log-forge` — the ORIGINAL name,
75
75
  # which PyPI rejected for the distribution — so these URLs deliberately do not match the package
76
76
  # name. See the note on `name` above before "correcting" them.
77
- version = "0.10.2.dev102"
77
+ version = "0.10.2.dev106"
78
78
 
79
79
  [project.urls]
80
80
  Homepage = "https://github.com/agriffi10/log-forge"
@@ -368,6 +368,56 @@ nested with either of the other two, so there is no cycle. It is held only acros
368
368
  membership test or a single mutation, never across a ``close()``.
369
369
  """
370
370
 
371
+ _orphan_closing = 0
372
+ """How many orphan closes are running right now, written under ``_state._lock`` (SPEC-050 FR-002).
373
+
374
+ The orphan path's half of the C3 residue :meth:`~log_foundry.worker.Worker._close_if_owed` fixes
375
+ for the worker: :func:`_close_orphan_sink` empties ``_orphan_owed`` under that lock and *then*
376
+ closes, so a second caller found nothing owed and returned instantly — and an ``atexit`` call that
377
+ returns while a background thread is still inside a close-is-delivery sink's ``close()`` exits
378
+ through it and kills it.
379
+
380
+ **A count and a gate, not one close's event.** A single slot was tried and is wrong, because the
381
+ orphan close is not once-only: ``_orphan_owed`` is repopulated by :func:`_note_orphan_emit` and
382
+ :func:`_adopt_declined_swap`, so a second close overwrote the first's event and its own completion
383
+ then cleared the slot — a bystander arriving after that read nothing, waited for nothing, and the
384
+ interpreter exited through the *first* close, still running. Measured against a one-second close:
385
+ the bystander waited 1.005 s alone and 0.000 s with a second close completing in between, losing
386
+ the first sink's whole buffer. It is the correction SPEC-045 made to the owed-close *record*,
387
+ arriving here for the same reason.
388
+
389
+ The predicate a bystander needs is "no orphan close is in flight", so that is what is published:
390
+ :data:`_orphan_idle` is cleared while the count is non-zero and set when it returns to zero. One
391
+ ``Event``, built once at module scope, which is also what keeps it where ``_fork``'s repair walk
392
+ can replace it.
393
+
394
+ **A leaked count is permanent, so the increment is inside the ``try`` and the decrement is
395
+ clamped.** Two ways it leaked, both reproduced: a ``KeyboardInterrupt`` delivered at a bytecode
396
+ boundary between the increment and the ``try`` — measured leaking once in a few hundred iterations
397
+ under a real ``SIGINT`` storm — and a ``fork()`` from *inside* the inline close, where the child's
398
+ handler zeroes the count and the forking thread's own ``finally`` then takes it to ``-1``, which
399
+ ``if not _orphan_closing`` never satisfies again. Either leaves the gate clear forever and every
400
+ later caller paying the whole grace, in a process that survives rather than exits. The count is
401
+ therefore taken and released under one ``try``, guarded by a flag set in the same critical section
402
+ as the increment, and floored at zero.
403
+
404
+ **The gate is process-wide, not per-sink**, so a worker-path ``shutdown()`` can pay the grace for
405
+ an orphan close it has nothing to do with — measured at 2.007 s wall, 0.000 s CPU, with the live
406
+ sink still drained and closed. That is the price of not exiting through a running close, and it is
407
+ bounded by the grace either way.
408
+ """
409
+
410
+ _orphan_idle = threading.Event()
411
+ """Set exactly when :data:`_orphan_closing` is zero — what a bystander waits on (SPEC-050 FR-002).
412
+
413
+ Starts set: a process with no orphan close in flight must not make the first bystander wait.
414
+ Not in :data:`_FORK_SKIP` — an ``Event`` is what ``_fork``'s repair walk exists to replace. What
415
+ the walk cannot know is that a child is idle whatever the parent was doing, since the threads that
416
+ would have finished those closes did not survive the fork; :func:`_clear_closing_after_fork` says
417
+ so.
418
+ """
419
+ _orphan_idle.set()
420
+
371
421
 
372
422
  def _closing(sink: Sink) -> bool:
373
423
  """Whether a release of this sink is in flight on some thread right now (FR-003).
@@ -402,6 +452,12 @@ def _clear_closing_after_fork() -> None:
402
452
  the placement is free rather than load-bearing: the registry holds ``int`` ids and no handler
403
453
  reads it.
404
454
 
455
+ :data:`_orphan_closing` is zeroed here for the same reason and by the same argument
456
+ (SPEC-050 FR-002). It counts closes running on threads that did not survive the fork, so a
457
+ child inheriting a non-zero count would make its next bystander wait out the whole closer
458
+ grace for closes that can never finish. The fork walk replaces the ``Event`` but not the count
459
+ that keeps it clear, which is the distinction this handler exists for.
460
+
405
461
  Args:
406
462
  None.
407
463
 
@@ -411,8 +467,12 @@ def _clear_closing_after_fork() -> None:
411
467
  Raises:
412
468
  None.
413
469
  """
470
+ global _orphan_closing
414
471
  with _closing_now_lock:
415
472
  _closing_now.clear()
473
+ with _state._lock:
474
+ _orphan_closing = 0
475
+ _orphan_idle.set()
416
476
 
417
477
 
418
478
  _FORK_SKIP = ("_owned",)
@@ -1451,7 +1511,62 @@ def _inline_close_choice(owed: list[Sink]) -> Sink:
1451
1511
  return owed[-1]
1452
1512
 
1453
1513
 
1454
- def _close_orphan_sink() -> None:
1514
+ def _bystander_grace(deadline: float | None) -> float:
1515
+ """Returns how long a caller may wait on a close another caller is performing (SPEC-050 FR-002).
1516
+
1517
+ ``join_closers``'s arithmetic, and :func:`~log_foundry.worker._closer_grace` is its twin on the
1518
+ worker path: capped at :data:`DEFAULT_CLOSER_GRACE` because this is an exit waiting on a close
1519
+ it does not own, and carved from the caller's own budget so a ``shutdown(timeout=0)`` does not
1520
+ inherit somebody else's. The flat cap this replaced made the two paths disagree — the worker
1521
+ half returned in under half a second on a ``timeout=0`` call while this one took the whole two
1522
+ seconds.
1523
+
1524
+ Args:
1525
+ deadline: The calling ``shutdown``'s monotonic deadline, or ``None`` for an unbounded caller,
1526
+ which takes the cap rather than waiting indefinitely.
1527
+
1528
+ Returns:
1529
+ Seconds to wait, never negative and never above the cap.
1530
+
1531
+ Raises:
1532
+ None.
1533
+ """
1534
+ if deadline is None:
1535
+ return DEFAULT_CLOSER_GRACE
1536
+ return max(0.0, min(DEFAULT_CLOSER_GRACE, deadline - monotonic()))
1537
+
1538
+
1539
+ def discharge_owed(sink: Sink) -> None:
1540
+ """Records that a sink's owed close is being performed elsewhere (SPEC-050 FR-004).
1541
+
1542
+ The orphan record and :attr:`~log_foundry.worker.Worker._unclosed_swaps` can name the **same**
1543
+ sink: an unconfirmed swap strands it in the worker's record, and an orphan emit that resolved
1544
+ it before the swap and resumed after the re-arm slot had moved on puts it back in
1545
+ ``_orphan_owed``. Both then close it — measured ``A.closes == 2`` with a preemption point at
1546
+ ``_ensure_sink``, which is SPEC-044 FR-004's shape at a record that did not exist then.
1547
+
1548
+ Called with ``_state._lock`` held, by the worker taking its own record under it, so the take
1549
+ and the discharge are one critical section: split, a concurrent ``_close_orphan_sink`` can read
1550
+ the sink out of ``_orphan_owed`` in the gap and close it alongside.
1551
+
1552
+ ``_orphan_closed_sink`` is latched as well as the entry removed, for the reason
1553
+ :func:`_get_worker` latches it: removal stops *this* close being performed twice, and the latch
1554
+ stops a later orphan emit re-arming a sink whose close is already under way.
1555
+
1556
+ Args:
1557
+ sink: The sink whose close the caller is about to perform.
1558
+
1559
+ Returns:
1560
+ None.
1561
+
1562
+ Raises:
1563
+ None.
1564
+ """
1565
+ _state._orphan_owed.pop(id(sink), None)
1566
+ _state._orphan_closed_sink = sink
1567
+
1568
+
1569
+ def _close_orphan_sink(deadline: float | None = None) -> None:
1455
1570
  """Closes a sink only the orphan path ever wrote to, once (SPEC-031 FR-006).
1456
1571
 
1457
1572
  A process that never opens a span builds no worker, so nothing owned the sink's close and
@@ -1459,6 +1574,33 @@ def _close_orphan_sink() -> None:
1459
1574
  on a synchronous one the flush and the resource were lost, and ``health()`` read all-clear
1460
1575
  because every field it carries describes a worker that does not exist.
1461
1576
 
1577
+ **A caller that finds nothing owed waits for the one that is closing** (SPEC-050 FR-002).
1578
+ The record is emptied under ``_state._lock`` before any close begins, so a second caller took
1579
+ the early return below while the first was still inside an unbounded ``close()`` — and where
1580
+ that first caller is a background thread and the second is ``atexit``, the interpreter exits
1581
+ through a running close and kills it. For a sink whose ``close()`` *is* the delivery that is
1582
+ total loss of its buffer.
1583
+
1584
+ **A separate caller cannot wait on itself, and the guard for that is the `if owed:` on the
1585
+ write.** (A sink whose own ``close()`` calls ``shutdown()`` re-enters on the closing thread and
1586
+ does wait out its grace — bounded, no deadlock, and pathological usage rather than a case this
1587
+ guards.)
1588
+ :data:`_orphan_closing` is installed only where ``owed`` is non-empty, so a caller that takes
1589
+ the work never reaches the read below and a bystander never installs anything. Capturing
1590
+ ``waiting`` before the write is *not* what makes this safe — both happen under
1591
+ ``_state._lock``, so reading the global there and reading the captured value are the same
1592
+ read, and mutating one into the other is an equivalent mutant. The conditional is the live
1593
+ guard: made unconditional, a caller with nothing owed leaves a permanently unset event behind
1594
+ it, and every call after the next one pays the whole grace on an event nothing will set —
1595
+ measured at the *third* successive orphan ``shutdown()``, which is why a test doing two of
1596
+ them cannot see it. The local exists so a reader can tell which of the two a caller is
1597
+ without re-deriving it.
1598
+
1599
+ The wait is :func:`_bystander_grace` — capped at :data:`DEFAULT_CLOSER_GRACE` and carved from
1600
+ the caller's own deadline, the same arithmetic the worker path uses, so a
1601
+ ``shutdown(timeout=0)`` does not inherit another caller's budget on one path and not the other.
1602
+ It took a flat cap first, which made the two paths disagree by two seconds on that call.
1603
+
1462
1604
  A worker that owns *this* sink closes it instead, and this returns — that is what makes a
1463
1605
  mixed process exactly one ``close()`` in either order. It also inherits that worker's reasons
1464
1606
  for *not* closing: an expired :meth:`Worker.shutdown` leaves the sink open because the drain
@@ -1516,7 +1658,8 @@ def _close_orphan_sink() -> None:
1516
1658
  choice is between closing inline and never closing at all.
1517
1659
 
1518
1660
  Args:
1519
- None.
1661
+ deadline: The calling ``shutdown``'s monotonic deadline, or ``None``. It bounds only
1662
+ the wait for another caller's close, never a close performed here.
1520
1663
 
1521
1664
  Returns:
1522
1665
  None.
@@ -1526,18 +1669,25 @@ def _close_orphan_sink() -> None:
1526
1669
  traceback carrying the message arch §6 keeps out of anything the library says about
1527
1670
  itself. ``Exception``, never ``BaseException`` (SPEC-025 FR-004).
1528
1671
  """
1529
- with _state._lock:
1530
- owed: list[Sink] = [
1531
- sink for sink in _state._orphan_owed.values() if not _state.worker_owns(sink)
1532
- ]
1533
- for sink in owed:
1534
- del _state._orphan_owed[id(sink)]
1535
- _state._orphan_closed_sink = sink
1536
- if not owed:
1537
- return
1538
- inline = _inline_close_choice(owed)
1672
+ global _orphan_closing
1673
+ took = False
1539
1674
  started: list[threading.Thread] = []
1540
1675
  try:
1676
+ with _state._lock:
1677
+ owed: list[Sink] = [
1678
+ sink for sink in _state._orphan_owed.values() if not _state.worker_owns(sink)
1679
+ ]
1680
+ for sink in owed:
1681
+ del _state._orphan_owed[id(sink)]
1682
+ _state._orphan_closed_sink = sink
1683
+ if owed:
1684
+ _orphan_closing += 1
1685
+ _orphan_idle.clear()
1686
+ took = True
1687
+ if not owed:
1688
+ _orphan_idle.wait(_bystander_grace(deadline))
1689
+ return
1690
+ inline = _inline_close_choice(owed)
1541
1691
  for sink in owed:
1542
1692
  if sink is inline:
1543
1693
  continue
@@ -1559,6 +1709,11 @@ def _close_orphan_sink() -> None:
1559
1709
  finally:
1560
1710
  for closer in started:
1561
1711
  closer.join()
1712
+ if took:
1713
+ with _state._lock:
1714
+ _orphan_closing = max(0, _orphan_closing - 1)
1715
+ if not _orphan_closing:
1716
+ _orphan_idle.set()
1562
1717
  def _shutdown_worker(timeout: float | None = DEFAULT_SHUTDOWN_TIMEOUT) -> None:
1563
1718
  """Drains and closes the process worker, or closes an orphan-only sink, backing ``shutdown()``.
1564
1719
 
@@ -1625,10 +1780,10 @@ def _shutdown_worker(timeout: float | None = DEFAULT_SHUTDOWN_TIMEOUT) -> None:
1625
1780
  _state._shutdown_running += 1
1626
1781
  if worker is not None:
1627
1782
  worker.shutdown(timeout)
1628
- _close_orphan_sink()
1783
+ _close_orphan_sink(deadline)
1629
1784
  return
1630
1785
  try:
1631
- _close_orphan_sink()
1786
+ _close_orphan_sink(deadline)
1632
1787
  finally:
1633
1788
  with _state._lock:
1634
1789
  late_worker = _state._late_worker
@@ -1699,6 +1854,15 @@ def _swap_sink(new_sink: Sink, timeout: float | None = DEFAULT_SWAP_TIMEOUT) ->
1699
1854
  but the library does not *rely* on that — it cannot enforce what a third-party sink does — so
1700
1855
  it performs one close (SPEC-032).
1701
1856
 
1857
+ **A sink this branch closes is dropped from the worker's owed-swap record** (SPEC-050
1858
+ FR-004). That record is what makes :meth:`~log_foundry.worker.Worker._close_if_owed`'s close
1859
+ of a stranded sink once-only, and this is the one route by which a sink in it can acquire a
1860
+ different closer: the re-arm guard next to this line is a single slot, so a second
1861
+ unconfirmed swap overwrites the first sink's protection and an orphan emit could put it back
1862
+ in the owed record. Defensive rather than reproduced — no reachable sequence for it was found
1863
+ — and one call, taken under the worker's own lock beneath this one, the same nesting
1864
+ :meth:`~log_foundry.worker.Worker.swap_sink` already performs.
1865
+
1702
1866
  The latch is keyed on ``worker.sink``, **not** on the orphan record. The sink
1703
1867
  ``Worker.swap_sink`` is about to close is the one the worker holds, and in the reproduced case
1704
1868
  the record is ``None`` — the worker cleared it when it was built — so keying on the record
@@ -1738,6 +1902,8 @@ def _swap_sink(new_sink: Sink, timeout: float | None = DEFAULT_SWAP_TIMEOUT) ->
1738
1902
  for stale in _state.take_orphan_owed():
1739
1903
  if stale is not new_sink and stale is not worker.sink:
1740
1904
  _state._orphan_closed_sink = stale
1905
+ with worker._lock:
1906
+ worker._discard_owed_swap(stale)
1741
1907
  closer = release(stale, detached=True)
1742
1908
  if closer is not None:
1743
1909
  closers.append(closer)
@@ -393,15 +393,19 @@ def _note_orphan_loss() -> None:
393
393
  _orphan_lost += 1
394
394
 
395
395
 
396
- def _note_in_span_loss() -> None:
397
- """Counts one event lost while being built inside a span (SPEC-036 FR-003).
396
+ def _note_in_span_loss(count: int = 1) -> None:
397
+ """Counts events lost while being built or handed over inside a span (SPEC-036 FR-003).
398
398
 
399
399
  Separate from :func:`_note_orphan_loss` because the two aggregate different failure
400
- populations: this path cannot fail at ``emit``, so a non-zero count here always means the
401
- data, never the destination.
400
+ populations: this path cannot fail at a sink's ``emit``, which is ``failed_batches``. It has
401
+ **two** causes, not one (SPEC-050 FR-003): a value that could not be built into an event, and
402
+ a process that could not give the library a thread to deliver through at all — where
403
+ ``Worker.__init__`` cannot start the drain thread, so the span's whole buffer is lost with no
404
+ worker in existence to record anything. The count is what the span was holding, so the second
405
+ cause moves it by more than one.
402
406
 
403
407
  Args:
404
- None.
408
+ count: How many events were lost, defaulting to the single event the build path loses.
405
409
 
406
410
  Returns:
407
411
  None.
@@ -411,7 +415,7 @@ def _note_in_span_loss() -> None:
411
415
  """
412
416
  global _in_span_lost
413
417
  with _loss_lock:
414
- _in_span_lost += 1
418
+ _in_span_lost += count
415
419
 
416
420
 
417
421
  def _read_losses() -> tuple[int, int]:
@@ -459,6 +463,15 @@ def _flush(span: Span) -> None:
459
463
  eight threads, 20,000 spans each). The **append** window this does not close is a different
460
464
  one, needs a per-span lock, and is declined in ``architecture.md`` §13.
461
465
 
466
+ **The worker is resolved before the buffer is detached** (SPEC-050 FR-003), which is what
467
+ lets :func:`_end` count what was lost. Resolving second, the events were already in a local
468
+ when :func:`~log_foundry._lifecycle._get_worker` raised, so they died with the exception and
469
+ ``len(span.events)`` read zero on exactly the path the count is for. Nothing between the
470
+ detach and the hand-off can fail: :meth:`~log_foundry.worker.Worker.submit` is total on every
471
+ path, so the events cannot be stranded in the local by this order. It also **narrows** the
472
+ SPEC-036 FR-004 window rather than widening it — the gap between detaching and submitting
473
+ falls from the whole of ``_get_worker()`` to a lock release.
474
+
462
475
  Args:
463
476
  span: The finished span whose buffered events are submitted.
464
477
 
@@ -468,9 +481,10 @@ def _flush(span: Span) -> None:
468
481
  Raises:
469
482
  Exception: Whatever creating the worker or submitting raises; :func:`_end` is the guard.
470
483
  """
484
+ worker = _lifecycle._get_worker()
471
485
  with _sweep_lock:
472
486
  events, span.events = span.events, []
473
- _lifecycle._get_worker().submit(events)
487
+ worker.submit(events)
474
488
 
475
489
 
476
490
  type _SpanScope = tuple[
@@ -546,6 +560,14 @@ def _end(
546
560
  would fall through that function's own guard to ``set(())`` and wipe the whole stack,
547
561
  detaching the parent of an untraced nested call and splitting its trace.
548
562
 
563
+ **The absorbed close is counted, not merely announced** (SPEC-050 FR-003). It used to write
564
+ one stderr line and move no field, so a process that could not start a drain thread lost every
565
+ span's events under ``Health(stopped_reason=None, in_span_lost=0)`` — all zeros over total
566
+ loss, which is the reading SPEC-036 FR-003 added the counter to remove. The count is
567
+ ``len(span.events)``, which :func:`_flush` leaves intact until there is a worker to take them,
568
+ and it is recorded **before** the announcement for the reason every other site here is:
569
+ stderr may be wedged, and the counter must not be lost to it.
570
+
549
571
  Args:
550
572
  span: The span to close, or ``None`` if none was opened.
551
573
  token: The span-stack token to release, or ``None``.
@@ -565,6 +587,7 @@ def _end(
565
587
  try:
566
588
  _close_span(span, status, error)
567
589
  except Exception as exc:
590
+ _note_in_span_loss(len(span.events))
568
591
  _diag.absorbed("closing a span", exc, "the span's events were lost")
569
592
  if token is not None:
570
593
  context.pop_span(token)
@@ -8,6 +8,7 @@ import threading
8
8
  import time
9
9
  from typing import TextIO
10
10
 
11
+ from log_foundry import _diag
11
12
  from log_foundry.sinks.base import SinkDeliveryError
12
13
 
13
14
  __all__ = ["FileSink", "RotatingFileSink"]
@@ -267,6 +268,7 @@ class RotatingFileSink:
267
268
  self._stream: TextIO = open(path, "a", encoding=self._encoding)
268
269
  self._size = os.path.getsize(path) if os.path.exists(path) else 0
269
270
  self._next_rollover = self._schedule_next()
271
+ self._rotation_failing = False
270
272
  self._closed = False
271
273
  self._lock = threading.Lock()
272
274
 
@@ -278,6 +280,15 @@ class RotatingFileSink:
278
280
  ``_should_rotate`` check and the write that follows it must see the same stream, or a
279
281
  rotation between them sends the line to a closed handle.
280
282
 
283
+ **The batch is flushed before a rotation is attempted** (SPEC-048 FR-006). ``_rotate``
284
+ begins by closing the stream, which flushes it, while this loop otherwise flushes once at
285
+ the end — so under the canonical rotation failure, a full or read-only filesystem, it is
286
+ that flush that raises and the batch's buffered lines are gone before any rename is tried.
287
+ Flushing here means every event of the batch is on disk before the rotation can fail, at
288
+ every one of ``_rotate``'s raise sites rather than only at the renames. A flush that
289
+ raises *here* is deliberately not absorbed: nothing was written, so it is the
290
+ genuinely-total failure the worker's retry exists for.
291
+
281
292
  Args:
282
293
  batch: The events to write.
283
294
 
@@ -285,7 +296,9 @@ class RotatingFileSink:
285
296
  None.
286
297
 
287
298
  Raises:
288
- OSError: If a write, flush or rotation fails.
299
+ OSError: If a write or a flush fails. A failed *rotation* no longer raises; see
300
+ :meth:`_rotate_or_continue`.
301
+ SinkDeliveryError: If the sink is closed.
289
302
  """
290
303
  if not batch:
291
304
  return
@@ -298,7 +311,8 @@ class RotatingFileSink:
298
311
  line = json.dumps(event) + "\n"
299
312
  data = len(line.encode(self._encoding))
300
313
  if self._should_rotate(data):
301
- self._rotate()
314
+ self._stream.flush()
315
+ self._rotate_or_continue()
302
316
  self._stream.write(line)
303
317
  self._size += data
304
318
  self._stream.flush()
@@ -425,6 +439,68 @@ class RotatingFileSink:
425
439
  return True
426
440
  return self._next_rollover is not None and time.monotonic() >= self._next_rollover
427
441
 
442
+ def _rotate_or_continue(self) -> None:
443
+ """Rotates, or absorbs the failure and carries on writing to the un-rotated file.
444
+
445
+ A rotation that raised used to cost the batch twice. ``_rotate`` closes the active stream
446
+ first, so the events already written in this batch were on disk and the ``OSError``
447
+ propagated out of ``emit`` — the worker then re-sent the whole batch and wrote them again.
448
+ Measured: an 8-event batch failing after 3 were written put 11 lines on disk, 3 of them
449
+ duplicates. A persistent failure was worse: the sink kept a **closed** stream and every
450
+ later batch raised a raw ``PermissionError``, which is not a ``SinkDeliveryError`` and has
451
+ no ``losses()`` behind it.
452
+
453
+ Absorbing it costs nothing and duplicates nothing. The active file simply exceeds
454
+ ``max_bytes`` until a rotation succeeds, which is what happens anyway when rotation is
455
+ impossible — the trade SPEC-027 FR-004 already took, that a leaked resource beats a
456
+ corrupt write.
457
+
458
+ ``_next_rollover`` is re-armed as well as ``_size``: ``_rotate`` sets it on its last line,
459
+ so an absorbed failure would otherwise leave a **time** trigger permanently in the past.
460
+
461
+ **The re-arm does not damp the size trigger, and the diagnostic is what carries that.**
462
+ ``_size`` is re-seeded from a file that is now over ``max_bytes``, so ``_should_rotate``'s
463
+ size branch stays true and every subsequent event attempts a rotation again — measured at
464
+ 598 attempts over 600 events. The attempts are cheap and lose nothing, but an unthrottled
465
+ stderr write per event is not: ``PostgresSink._reconnect_if_broken`` records the same rule
466
+ for the same reason, that a diagnostic which floods is one an operator stops reading. So
467
+ the failure is announced **once per outage** and the flag clears on the next successful
468
+ rotation. The remaining per-event attempt is recorded in ``architecture.md`` §12 rather
469
+ than fixed here, because damping it means deferring a rotation the caller asked for.
470
+
471
+ **The reopen can itself raise, and that is not absorbed.** It is the same
472
+ ``open(self._path, "a")`` call ``_rotate`` ends with, so whatever defeats it there —
473
+ a read-only mount, ``EMFILE``, a directory that lost write permission — defeats it here.
474
+ At that point this sink has no stream and cannot continue, so there is nothing to absorb
475
+ *into*; the ``OSError`` reaches ``emit`` and the worker retries the batch, duplicating the
476
+ prefix the pre-rotation flush had already written. That residue is recorded rather than
477
+ fixed: it is unchanged from before SPEC-048, the surviving events are on disk rather than
478
+ lost, and inventing a half-open state to avoid a duplicate would trade a visible
479
+ duplication for a silent loss.
480
+
481
+ Args:
482
+ None.
483
+
484
+ Returns:
485
+ None.
486
+
487
+ Raises:
488
+ OSError: If the *reopen* fails, per the paragraph above. A failed **rotation** is
489
+ absorbed and announced through ``_diag``; the batch continues, nothing is dropped, no
490
+ counter moves, and this class still has no ``losses()``.
491
+ """
492
+ try:
493
+ self._rotate()
494
+ except OSError as err:
495
+ self._stream = open(self._path, "a", encoding=self._encoding)
496
+ self._size = os.path.getsize(self._path) if os.path.exists(self._path) else 0
497
+ self._next_rollover = self._schedule_next()
498
+ if not self._rotation_failing:
499
+ self._rotation_failing = True
500
+ _diag.absorbed("rotating RotatingFileSink", err)
501
+ return
502
+ self._rotation_failing = False
503
+
428
504
  def _rotate(self) -> None:
429
505
  """Closes the active file, shifts and prunes backups, then opens a fresh active file.
430
506
 
@@ -203,6 +203,22 @@ class FirehoseSink:
203
203
  this loop re-sends the records the destination flagged, and the canonical reason it flags
204
204
  them is throttling, which an immediate re-send makes worse.
205
205
 
206
+ **A client exception costs this chunk, never the batch** (SPEC-048 FR-002). It used to
207
+ propagate out of :meth:`emit`, so a failure on chunk N after chunks 1..N-1 had landed made
208
+ the worker re-send the whole batch and duplicate everything already delivered — the exit
209
+ drain, one large batch by construction, is exactly that shape. The guard sits around the
210
+ client call *inside* this loop rather than around ``_send`` in ``emit``, so only the
211
+ records still outstanding at the failing attempt are charged and the ones already accepted
212
+ still count toward the return.
213
+
214
+ A client exception is treated as **provable non-delivery** for the chunk, so a wholly
215
+ failed batch still raises and ``unknown`` is untouched: SPEC-018's "unadjudicable" is a
216
+ property of a *response*, and an exception is not a response. The cost is written down —
217
+ a read timeout means the request went out and the reply was lost, so a re-send may
218
+ duplicate — and taken because an unreachable endpoint is the common case and is exactly
219
+ what the worker's retry exists for, while suppressing the raise would lose every event of
220
+ every batch for a whole outage, silently.
221
+
206
222
  Args:
207
223
  records: One chunk's request entries.
208
224
 
@@ -214,13 +230,27 @@ class FirehoseSink:
214
230
  SPEC-018 settled must never be re-sent.
215
231
 
216
232
  Raises:
217
- Exception: Whatever the client raises.
233
+ None. ``Exception`` is caught rather than ``BaseException``, so a ``KeyboardInterrupt``
234
+ or ``SystemExit`` still reaches the caller (SPEC-025 FR-004).
218
235
  """
219
236
  sent = len(records)
220
237
  for attempt in range(self.max_retries + 1):
221
- response = self.client.put_record_batch(
222
- DeliveryStreamName=self.delivery_stream, Records=records
223
- )
238
+ try:
239
+ response = self.client.put_record_batch(
240
+ DeliveryStreamName=self.delivery_stream, Records=records
241
+ )
242
+ except Exception as err:
243
+ if attempt < self.max_retries:
244
+ wait(_BACKOFF_BASE * (2**attempt), self.log_foundry_stop_signal)
245
+ continue
246
+ with self._counter_lock:
247
+ self.failed += len(records)
248
+ _diag.lost(
249
+ "record",
250
+ len(records),
251
+ f"FirehoseSink, {self.max_retries + 1} attempt(s), {type(err).__name__}",
252
+ )
253
+ return sent - len(records)
224
254
  if not response.get("FailedPutCount"):
225
255
  return sent
226
256
  results = usable_results(response.get("RequestResponses"))