log-foundry 0.10.2.dev102__tar.gz → 0.10.2.dev106__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/PKG-INFO +1 -1
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/pyproject.toml +1 -1
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/_lifecycle.py +180 -14
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/decorator.py +30 -7
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/file.py +78 -2
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/firehose.py +34 -4
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/http.py +83 -4
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/kinesis.py +75 -6
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/pubsub.py +159 -50
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/sentry.py +19 -1
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/sns.py +33 -4
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/sqs.py +36 -2
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/worker.py +223 -24
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/LICENSE +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/README.md +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/__init__.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/_diag.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/_fork.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/api.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/config.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/console.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/context.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/ids.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/model.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/py.typed +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/results.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sanitize.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/__init__.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/_batch.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/_chunk.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/_retry.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/_socket.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/_time.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/base.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/callback.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/clickhouse.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/datadog.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/elasticsearch.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/eventhubs.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/filtering.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/honeycomb.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/kafka.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/logging_sink.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/logstash.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/loki.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/memory.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/mongodb.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/multi.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/nats.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/newrelic.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/null.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/postgres.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/rabbitmq.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/redis.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/splunk.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/sqlite.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/stdout.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/syslog.py +0 -0
- {log_foundry-0.10.2.dev102 → log_foundry-0.10.2.dev106}/src/log_foundry/sinks/transform.py +0 -0
|
@@ -74,7 +74,7 @@ keywords = [
|
|
|
74
74
|
# vulnerability-reporting channel. The repository is still named `log-forge` — the ORIGINAL name,
|
|
75
75
|
# which PyPI rejected for the distribution — so these URLs deliberately do not match the package
|
|
76
76
|
# name. See the note on `name` above before "correcting" them.
|
|
77
|
-
version = "0.10.2.
|
|
77
|
+
version = "0.10.2.dev106"
|
|
78
78
|
|
|
79
79
|
[project.urls]
|
|
80
80
|
Homepage = "https://github.com/agriffi10/log-forge"
|
|
@@ -368,6 +368,56 @@ nested with either of the other two, so there is no cycle. It is held only acros
|
|
|
368
368
|
membership test or a single mutation, never across a ``close()``.
|
|
369
369
|
"""
|
|
370
370
|
|
|
371
|
+
_orphan_closing = 0
|
|
372
|
+
"""How many orphan closes are running right now, written under ``_state._lock`` (SPEC-050 FR-002).
|
|
373
|
+
|
|
374
|
+
The orphan path's half of the C3 residue :meth:`~log_foundry.worker.Worker._close_if_owed` fixes
|
|
375
|
+
for the worker: :func:`_close_orphan_sink` empties ``_orphan_owed`` under that lock and *then*
|
|
376
|
+
closes, so a second caller found nothing owed and returned instantly — and an ``atexit`` call that
|
|
377
|
+
returns while a background thread is still inside a close-is-delivery sink's ``close()`` exits
|
|
378
|
+
through it and kills it.
|
|
379
|
+
|
|
380
|
+
**A count and a gate, not one close's event.** A single slot was tried and is wrong, because the
|
|
381
|
+
orphan close is not once-only: ``_orphan_owed`` is repopulated by :func:`_note_orphan_emit` and
|
|
382
|
+
:func:`_adopt_declined_swap`, so a second close overwrote the first's event and its own completion
|
|
383
|
+
then cleared the slot — a bystander arriving after that read nothing, waited for nothing, and the
|
|
384
|
+
interpreter exited through the *first* close, still running. Measured against a one-second close:
|
|
385
|
+
the bystander waited 1.005 s alone and 0.000 s with a second close completing in between, losing
|
|
386
|
+
the first sink's whole buffer. It is the correction SPEC-045 made to the owed-close *record*,
|
|
387
|
+
arriving here for the same reason.
|
|
388
|
+
|
|
389
|
+
The predicate a bystander needs is "no orphan close is in flight", so that is what is published:
|
|
390
|
+
:data:`_orphan_idle` is cleared while the count is non-zero and set when it returns to zero. One
|
|
391
|
+
``Event``, built once at module scope, which is also what keeps it where ``_fork``'s repair walk
|
|
392
|
+
can replace it.
|
|
393
|
+
|
|
394
|
+
**A leaked count is permanent, so the increment is inside the ``try`` and the decrement is
|
|
395
|
+
clamped.** Two ways it leaked, both reproduced: a ``KeyboardInterrupt`` delivered at a bytecode
|
|
396
|
+
boundary between the increment and the ``try`` — measured leaking once in a few hundred iterations
|
|
397
|
+
under a real ``SIGINT`` storm — and a ``fork()`` from *inside* the inline close, where the child's
|
|
398
|
+
handler zeroes the count and the forking thread's own ``finally`` then takes it to ``-1``, which
|
|
399
|
+
``if not _orphan_closing`` never satisfies again. Either leaves the gate clear forever and every
|
|
400
|
+
later caller paying the whole grace, in a process that survives rather than exits. The count is
|
|
401
|
+
therefore taken and released under one ``try``, guarded by a flag set in the same critical section
|
|
402
|
+
as the increment, and floored at zero.
|
|
403
|
+
|
|
404
|
+
**The gate is process-wide, not per-sink**, so a worker-path ``shutdown()`` can pay the grace for
|
|
405
|
+
an orphan close it has nothing to do with — measured at 2.007 s wall, 0.000 s CPU, with the live
|
|
406
|
+
sink still drained and closed. That is the price of not exiting through a running close, and it is
|
|
407
|
+
bounded by the grace either way.
|
|
408
|
+
"""
|
|
409
|
+
|
|
410
|
+
_orphan_idle = threading.Event()
|
|
411
|
+
"""Set exactly when :data:`_orphan_closing` is zero — what a bystander waits on (SPEC-050 FR-002).
|
|
412
|
+
|
|
413
|
+
Starts set: a process with no orphan close in flight must not make the first bystander wait.
|
|
414
|
+
Not in :data:`_FORK_SKIP` — an ``Event`` is what ``_fork``'s repair walk exists to replace. What
|
|
415
|
+
the walk cannot know is that a child is idle whatever the parent was doing, since the threads that
|
|
416
|
+
would have finished those closes did not survive the fork; :func:`_clear_closing_after_fork` says
|
|
417
|
+
so.
|
|
418
|
+
"""
|
|
419
|
+
_orphan_idle.set()
|
|
420
|
+
|
|
371
421
|
|
|
372
422
|
def _closing(sink: Sink) -> bool:
|
|
373
423
|
"""Whether a release of this sink is in flight on some thread right now (FR-003).
|
|
@@ -402,6 +452,12 @@ def _clear_closing_after_fork() -> None:
|
|
|
402
452
|
the placement is free rather than load-bearing: the registry holds ``int`` ids and no handler
|
|
403
453
|
reads it.
|
|
404
454
|
|
|
455
|
+
:data:`_orphan_closing` is zeroed here for the same reason and by the same argument
|
|
456
|
+
(SPEC-050 FR-002). It counts closes running on threads that did not survive the fork, so a
|
|
457
|
+
child inheriting a non-zero count would make its next bystander wait out the whole closer
|
|
458
|
+
grace for closes that can never finish. The fork walk replaces the ``Event`` but not the count
|
|
459
|
+
that keeps it clear, which is the distinction this handler exists for.
|
|
460
|
+
|
|
405
461
|
Args:
|
|
406
462
|
None.
|
|
407
463
|
|
|
@@ -411,8 +467,12 @@ def _clear_closing_after_fork() -> None:
|
|
|
411
467
|
Raises:
|
|
412
468
|
None.
|
|
413
469
|
"""
|
|
470
|
+
global _orphan_closing
|
|
414
471
|
with _closing_now_lock:
|
|
415
472
|
_closing_now.clear()
|
|
473
|
+
with _state._lock:
|
|
474
|
+
_orphan_closing = 0
|
|
475
|
+
_orphan_idle.set()
|
|
416
476
|
|
|
417
477
|
|
|
418
478
|
_FORK_SKIP = ("_owned",)
|
|
@@ -1451,7 +1511,62 @@ def _inline_close_choice(owed: list[Sink]) -> Sink:
|
|
|
1451
1511
|
return owed[-1]
|
|
1452
1512
|
|
|
1453
1513
|
|
|
1454
|
-
def
|
|
1514
|
+
def _bystander_grace(deadline: float | None) -> float:
|
|
1515
|
+
"""Returns how long a caller may wait on a close another caller is performing (SPEC-050 FR-002).
|
|
1516
|
+
|
|
1517
|
+
``join_closers``'s arithmetic, and :func:`~log_foundry.worker._closer_grace` is its twin on the
|
|
1518
|
+
worker path: capped at :data:`DEFAULT_CLOSER_GRACE` because this is an exit waiting on a close
|
|
1519
|
+
it does not own, and carved from the caller's own budget so a ``shutdown(timeout=0)`` does not
|
|
1520
|
+
inherit somebody else's. The flat cap this replaced made the two paths disagree — the worker
|
|
1521
|
+
half returned in under half a second on a ``timeout=0`` call while this one took the whole two
|
|
1522
|
+
seconds.
|
|
1523
|
+
|
|
1524
|
+
Args:
|
|
1525
|
+
deadline: The calling ``shutdown``'s monotonic deadline, or ``None`` for an unbounded caller,
|
|
1526
|
+
which takes the cap rather than waiting indefinitely.
|
|
1527
|
+
|
|
1528
|
+
Returns:
|
|
1529
|
+
Seconds to wait, never negative and never above the cap.
|
|
1530
|
+
|
|
1531
|
+
Raises:
|
|
1532
|
+
None.
|
|
1533
|
+
"""
|
|
1534
|
+
if deadline is None:
|
|
1535
|
+
return DEFAULT_CLOSER_GRACE
|
|
1536
|
+
return max(0.0, min(DEFAULT_CLOSER_GRACE, deadline - monotonic()))
|
|
1537
|
+
|
|
1538
|
+
|
|
1539
|
+
def discharge_owed(sink: Sink) -> None:
|
|
1540
|
+
"""Records that a sink's owed close is being performed elsewhere (SPEC-050 FR-004).
|
|
1541
|
+
|
|
1542
|
+
The orphan record and :attr:`~log_foundry.worker.Worker._unclosed_swaps` can name the **same**
|
|
1543
|
+
sink: an unconfirmed swap strands it in the worker's record, and an orphan emit that resolved
|
|
1544
|
+
it before the swap and resumed after the re-arm slot had moved on puts it back in
|
|
1545
|
+
``_orphan_owed``. Both then close it — measured ``A.closes == 2`` with a preemption point at
|
|
1546
|
+
``_ensure_sink``, which is SPEC-044 FR-004's shape at a record that did not exist then.
|
|
1547
|
+
|
|
1548
|
+
Called with ``_state._lock`` held, by the worker taking its own record under it, so the take
|
|
1549
|
+
and the discharge are one critical section: split, a concurrent ``_close_orphan_sink`` can read
|
|
1550
|
+
the sink out of ``_orphan_owed`` in the gap and close it alongside.
|
|
1551
|
+
|
|
1552
|
+
``_orphan_closed_sink`` is latched as well as the entry removed, for the reason
|
|
1553
|
+
:func:`_get_worker` latches it: removal stops *this* close being performed twice, and the latch
|
|
1554
|
+
stops a later orphan emit re-arming a sink whose close is already under way.
|
|
1555
|
+
|
|
1556
|
+
Args:
|
|
1557
|
+
sink: The sink whose close the caller is about to perform.
|
|
1558
|
+
|
|
1559
|
+
Returns:
|
|
1560
|
+
None.
|
|
1561
|
+
|
|
1562
|
+
Raises:
|
|
1563
|
+
None.
|
|
1564
|
+
"""
|
|
1565
|
+
_state._orphan_owed.pop(id(sink), None)
|
|
1566
|
+
_state._orphan_closed_sink = sink
|
|
1567
|
+
|
|
1568
|
+
|
|
1569
|
+
def _close_orphan_sink(deadline: float | None = None) -> None:
|
|
1455
1570
|
"""Closes a sink only the orphan path ever wrote to, once (SPEC-031 FR-006).
|
|
1456
1571
|
|
|
1457
1572
|
A process that never opens a span builds no worker, so nothing owned the sink's close and
|
|
@@ -1459,6 +1574,33 @@ def _close_orphan_sink() -> None:
|
|
|
1459
1574
|
on a synchronous one the flush and the resource were lost, and ``health()`` read all-clear
|
|
1460
1575
|
because every field it carries describes a worker that does not exist.
|
|
1461
1576
|
|
|
1577
|
+
**A caller that finds nothing owed waits for the one that is closing** (SPEC-050 FR-002).
|
|
1578
|
+
The record is emptied under ``_state._lock`` before any close begins, so a second caller took
|
|
1579
|
+
the early return below while the first was still inside an unbounded ``close()`` — and where
|
|
1580
|
+
that first caller is a background thread and the second is ``atexit``, the interpreter exits
|
|
1581
|
+
through a running close and kills it. For a sink whose ``close()`` *is* the delivery that is
|
|
1582
|
+
total loss of its buffer.
|
|
1583
|
+
|
|
1584
|
+
**A separate caller cannot wait on itself, and the guard for that is the `if owed:` on the
|
|
1585
|
+
write.** (A sink whose own ``close()`` calls ``shutdown()`` re-enters on the closing thread and
|
|
1586
|
+
does wait out its grace — bounded, no deadlock, and pathological usage rather than a case this
|
|
1587
|
+
guards.)
|
|
1588
|
+
:data:`_orphan_closing` is installed only where ``owed`` is non-empty, so a caller that takes
|
|
1589
|
+
the work never reaches the read below and a bystander never installs anything. Capturing
|
|
1590
|
+
``waiting`` before the write is *not* what makes this safe — both happen under
|
|
1591
|
+
``_state._lock``, so reading the global there and reading the captured value are the same
|
|
1592
|
+
read, and mutating one into the other is an equivalent mutant. The conditional is the live
|
|
1593
|
+
guard: made unconditional, a caller with nothing owed leaves a permanently unset event behind
|
|
1594
|
+
it, and every call after the next one pays the whole grace on an event nothing will set —
|
|
1595
|
+
measured at the *third* successive orphan ``shutdown()``, which is why a test doing two of
|
|
1596
|
+
them cannot see it. The local exists so a reader can tell which of the two a caller is
|
|
1597
|
+
without re-deriving it.
|
|
1598
|
+
|
|
1599
|
+
The wait is :func:`_bystander_grace` — capped at :data:`DEFAULT_CLOSER_GRACE` and carved from
|
|
1600
|
+
the caller's own deadline, the same arithmetic the worker path uses, so a
|
|
1601
|
+
``shutdown(timeout=0)`` does not inherit another caller's budget on one path and not the other.
|
|
1602
|
+
It took a flat cap first, which made the two paths disagree by two seconds on that call.
|
|
1603
|
+
|
|
1462
1604
|
A worker that owns *this* sink closes it instead, and this returns — that is what makes a
|
|
1463
1605
|
mixed process exactly one ``close()`` in either order. It also inherits that worker's reasons
|
|
1464
1606
|
for *not* closing: an expired :meth:`Worker.shutdown` leaves the sink open because the drain
|
|
@@ -1516,7 +1658,8 @@ def _close_orphan_sink() -> None:
|
|
|
1516
1658
|
choice is between closing inline and never closing at all.
|
|
1517
1659
|
|
|
1518
1660
|
Args:
|
|
1519
|
-
None
|
|
1661
|
+
deadline: The calling ``shutdown``'s monotonic deadline, or ``None``. It bounds only
|
|
1662
|
+
the wait for another caller's close, never a close performed here.
|
|
1520
1663
|
|
|
1521
1664
|
Returns:
|
|
1522
1665
|
None.
|
|
@@ -1526,18 +1669,25 @@ def _close_orphan_sink() -> None:
|
|
|
1526
1669
|
traceback carrying the message arch §6 keeps out of anything the library says about
|
|
1527
1670
|
itself. ``Exception``, never ``BaseException`` (SPEC-025 FR-004).
|
|
1528
1671
|
"""
|
|
1529
|
-
|
|
1530
|
-
|
|
1531
|
-
sink for sink in _state._orphan_owed.values() if not _state.worker_owns(sink)
|
|
1532
|
-
]
|
|
1533
|
-
for sink in owed:
|
|
1534
|
-
del _state._orphan_owed[id(sink)]
|
|
1535
|
-
_state._orphan_closed_sink = sink
|
|
1536
|
-
if not owed:
|
|
1537
|
-
return
|
|
1538
|
-
inline = _inline_close_choice(owed)
|
|
1672
|
+
global _orphan_closing
|
|
1673
|
+
took = False
|
|
1539
1674
|
started: list[threading.Thread] = []
|
|
1540
1675
|
try:
|
|
1676
|
+
with _state._lock:
|
|
1677
|
+
owed: list[Sink] = [
|
|
1678
|
+
sink for sink in _state._orphan_owed.values() if not _state.worker_owns(sink)
|
|
1679
|
+
]
|
|
1680
|
+
for sink in owed:
|
|
1681
|
+
del _state._orphan_owed[id(sink)]
|
|
1682
|
+
_state._orphan_closed_sink = sink
|
|
1683
|
+
if owed:
|
|
1684
|
+
_orphan_closing += 1
|
|
1685
|
+
_orphan_idle.clear()
|
|
1686
|
+
took = True
|
|
1687
|
+
if not owed:
|
|
1688
|
+
_orphan_idle.wait(_bystander_grace(deadline))
|
|
1689
|
+
return
|
|
1690
|
+
inline = _inline_close_choice(owed)
|
|
1541
1691
|
for sink in owed:
|
|
1542
1692
|
if sink is inline:
|
|
1543
1693
|
continue
|
|
@@ -1559,6 +1709,11 @@ def _close_orphan_sink() -> None:
|
|
|
1559
1709
|
finally:
|
|
1560
1710
|
for closer in started:
|
|
1561
1711
|
closer.join()
|
|
1712
|
+
if took:
|
|
1713
|
+
with _state._lock:
|
|
1714
|
+
_orphan_closing = max(0, _orphan_closing - 1)
|
|
1715
|
+
if not _orphan_closing:
|
|
1716
|
+
_orphan_idle.set()
|
|
1562
1717
|
def _shutdown_worker(timeout: float | None = DEFAULT_SHUTDOWN_TIMEOUT) -> None:
|
|
1563
1718
|
"""Drains and closes the process worker, or closes an orphan-only sink, backing ``shutdown()``.
|
|
1564
1719
|
|
|
@@ -1625,10 +1780,10 @@ def _shutdown_worker(timeout: float | None = DEFAULT_SHUTDOWN_TIMEOUT) -> None:
|
|
|
1625
1780
|
_state._shutdown_running += 1
|
|
1626
1781
|
if worker is not None:
|
|
1627
1782
|
worker.shutdown(timeout)
|
|
1628
|
-
_close_orphan_sink()
|
|
1783
|
+
_close_orphan_sink(deadline)
|
|
1629
1784
|
return
|
|
1630
1785
|
try:
|
|
1631
|
-
_close_orphan_sink()
|
|
1786
|
+
_close_orphan_sink(deadline)
|
|
1632
1787
|
finally:
|
|
1633
1788
|
with _state._lock:
|
|
1634
1789
|
late_worker = _state._late_worker
|
|
@@ -1699,6 +1854,15 @@ def _swap_sink(new_sink: Sink, timeout: float | None = DEFAULT_SWAP_TIMEOUT) ->
|
|
|
1699
1854
|
but the library does not *rely* on that — it cannot enforce what a third-party sink does — so
|
|
1700
1855
|
it performs one close (SPEC-032).
|
|
1701
1856
|
|
|
1857
|
+
**A sink this branch closes is dropped from the worker's owed-swap record** (SPEC-050
|
|
1858
|
+
FR-004). That record is what makes :meth:`~log_foundry.worker.Worker._close_if_owed`'s close
|
|
1859
|
+
of a stranded sink once-only, and this is the one route by which a sink in it can acquire a
|
|
1860
|
+
different closer: the re-arm guard next to this line is a single slot, so a second
|
|
1861
|
+
unconfirmed swap overwrites the first sink's protection and an orphan emit could put it back
|
|
1862
|
+
in the owed record. Defensive rather than reproduced — no reachable sequence for it was found
|
|
1863
|
+
— and one call, taken under the worker's own lock beneath this one, the same nesting
|
|
1864
|
+
:meth:`~log_foundry.worker.Worker.swap_sink` already performs.
|
|
1865
|
+
|
|
1702
1866
|
The latch is keyed on ``worker.sink``, **not** on the orphan record. The sink
|
|
1703
1867
|
``Worker.swap_sink`` is about to close is the one the worker holds, and in the reproduced case
|
|
1704
1868
|
the record is ``None`` — the worker cleared it when it was built — so keying on the record
|
|
@@ -1738,6 +1902,8 @@ def _swap_sink(new_sink: Sink, timeout: float | None = DEFAULT_SWAP_TIMEOUT) ->
|
|
|
1738
1902
|
for stale in _state.take_orphan_owed():
|
|
1739
1903
|
if stale is not new_sink and stale is not worker.sink:
|
|
1740
1904
|
_state._orphan_closed_sink = stale
|
|
1905
|
+
with worker._lock:
|
|
1906
|
+
worker._discard_owed_swap(stale)
|
|
1741
1907
|
closer = release(stale, detached=True)
|
|
1742
1908
|
if closer is not None:
|
|
1743
1909
|
closers.append(closer)
|
|
@@ -393,15 +393,19 @@ def _note_orphan_loss() -> None:
|
|
|
393
393
|
_orphan_lost += 1
|
|
394
394
|
|
|
395
395
|
|
|
396
|
-
def _note_in_span_loss() -> None:
|
|
397
|
-
"""Counts
|
|
396
|
+
def _note_in_span_loss(count: int = 1) -> None:
|
|
397
|
+
"""Counts events lost while being built or handed over inside a span (SPEC-036 FR-003).
|
|
398
398
|
|
|
399
399
|
Separate from :func:`_note_orphan_loss` because the two aggregate different failure
|
|
400
|
-
populations: this path cannot fail at ``emit``,
|
|
401
|
-
|
|
400
|
+
populations: this path cannot fail at a sink's ``emit``, which is ``failed_batches``. It has
|
|
401
|
+
**two** causes, not one (SPEC-050 FR-003): a value that could not be built into an event, and
|
|
402
|
+
a process that could not give the library a thread to deliver through at all — where
|
|
403
|
+
``Worker.__init__`` cannot start the drain thread, so the span's whole buffer is lost with no
|
|
404
|
+
worker in existence to record anything. The count is what the span was holding, so the second
|
|
405
|
+
cause moves it by more than one.
|
|
402
406
|
|
|
403
407
|
Args:
|
|
404
|
-
|
|
408
|
+
count: How many events were lost, defaulting to the single event the build path loses.
|
|
405
409
|
|
|
406
410
|
Returns:
|
|
407
411
|
None.
|
|
@@ -411,7 +415,7 @@ def _note_in_span_loss() -> None:
|
|
|
411
415
|
"""
|
|
412
416
|
global _in_span_lost
|
|
413
417
|
with _loss_lock:
|
|
414
|
-
_in_span_lost +=
|
|
418
|
+
_in_span_lost += count
|
|
415
419
|
|
|
416
420
|
|
|
417
421
|
def _read_losses() -> tuple[int, int]:
|
|
@@ -459,6 +463,15 @@ def _flush(span: Span) -> None:
|
|
|
459
463
|
eight threads, 20,000 spans each). The **append** window this does not close is a different
|
|
460
464
|
one, needs a per-span lock, and is declined in ``architecture.md`` §13.
|
|
461
465
|
|
|
466
|
+
**The worker is resolved before the buffer is detached** (SPEC-050 FR-003), which is what
|
|
467
|
+
lets :func:`_end` count what was lost. Resolving second, the events were already in a local
|
|
468
|
+
when :func:`~log_foundry._lifecycle._get_worker` raised, so they died with the exception and
|
|
469
|
+
``len(span.events)`` read zero on exactly the path the count is for. Nothing between the
|
|
470
|
+
detach and the hand-off can fail: :meth:`~log_foundry.worker.Worker.submit` is total on every
|
|
471
|
+
path, so the events cannot be stranded in the local by this order. It also **narrows** the
|
|
472
|
+
SPEC-036 FR-004 window rather than widening it — the gap between detaching and submitting
|
|
473
|
+
falls from the whole of ``_get_worker()`` to a lock release.
|
|
474
|
+
|
|
462
475
|
Args:
|
|
463
476
|
span: The finished span whose buffered events are submitted.
|
|
464
477
|
|
|
@@ -468,9 +481,10 @@ def _flush(span: Span) -> None:
|
|
|
468
481
|
Raises:
|
|
469
482
|
Exception: Whatever creating the worker or submitting raises; :func:`_end` is the guard.
|
|
470
483
|
"""
|
|
484
|
+
worker = _lifecycle._get_worker()
|
|
471
485
|
with _sweep_lock:
|
|
472
486
|
events, span.events = span.events, []
|
|
473
|
-
|
|
487
|
+
worker.submit(events)
|
|
474
488
|
|
|
475
489
|
|
|
476
490
|
type _SpanScope = tuple[
|
|
@@ -546,6 +560,14 @@ def _end(
|
|
|
546
560
|
would fall through that function's own guard to ``set(())`` and wipe the whole stack,
|
|
547
561
|
detaching the parent of an untraced nested call and splitting its trace.
|
|
548
562
|
|
|
563
|
+
**The absorbed close is counted, not merely announced** (SPEC-050 FR-003). It used to write
|
|
564
|
+
one stderr line and move no field, so a process that could not start a drain thread lost every
|
|
565
|
+
span's events under ``Health(stopped_reason=None, in_span_lost=0)`` — all zeros over total
|
|
566
|
+
loss, which is the reading SPEC-036 FR-003 added the counter to remove. The count is
|
|
567
|
+
``len(span.events)``, which :func:`_flush` leaves intact until there is a worker to take them,
|
|
568
|
+
and it is recorded **before** the announcement for the reason every other site here is:
|
|
569
|
+
stderr may be wedged, and the counter must not be lost to it.
|
|
570
|
+
|
|
549
571
|
Args:
|
|
550
572
|
span: The span to close, or ``None`` if none was opened.
|
|
551
573
|
token: The span-stack token to release, or ``None``.
|
|
@@ -565,6 +587,7 @@ def _end(
|
|
|
565
587
|
try:
|
|
566
588
|
_close_span(span, status, error)
|
|
567
589
|
except Exception as exc:
|
|
590
|
+
_note_in_span_loss(len(span.events))
|
|
568
591
|
_diag.absorbed("closing a span", exc, "the span's events were lost")
|
|
569
592
|
if token is not None:
|
|
570
593
|
context.pop_span(token)
|
|
@@ -8,6 +8,7 @@ import threading
|
|
|
8
8
|
import time
|
|
9
9
|
from typing import TextIO
|
|
10
10
|
|
|
11
|
+
from log_foundry import _diag
|
|
11
12
|
from log_foundry.sinks.base import SinkDeliveryError
|
|
12
13
|
|
|
13
14
|
__all__ = ["FileSink", "RotatingFileSink"]
|
|
@@ -267,6 +268,7 @@ class RotatingFileSink:
|
|
|
267
268
|
self._stream: TextIO = open(path, "a", encoding=self._encoding)
|
|
268
269
|
self._size = os.path.getsize(path) if os.path.exists(path) else 0
|
|
269
270
|
self._next_rollover = self._schedule_next()
|
|
271
|
+
self._rotation_failing = False
|
|
270
272
|
self._closed = False
|
|
271
273
|
self._lock = threading.Lock()
|
|
272
274
|
|
|
@@ -278,6 +280,15 @@ class RotatingFileSink:
|
|
|
278
280
|
``_should_rotate`` check and the write that follows it must see the same stream, or a
|
|
279
281
|
rotation between them sends the line to a closed handle.
|
|
280
282
|
|
|
283
|
+
**The batch is flushed before a rotation is attempted** (SPEC-048 FR-006). ``_rotate``
|
|
284
|
+
begins by closing the stream, which flushes it, while this loop otherwise flushes once at
|
|
285
|
+
the end — so under the canonical rotation failure, a full or read-only filesystem, it is
|
|
286
|
+
that flush that raises and the batch's buffered lines are gone before any rename is tried.
|
|
287
|
+
Flushing here means every event of the batch is on disk before the rotation can fail, at
|
|
288
|
+
every one of ``_rotate``'s raise sites rather than only at the renames. A flush that
|
|
289
|
+
raises *here* is deliberately not absorbed: nothing was written, so it is the
|
|
290
|
+
genuinely-total failure the worker's retry exists for.
|
|
291
|
+
|
|
281
292
|
Args:
|
|
282
293
|
batch: The events to write.
|
|
283
294
|
|
|
@@ -285,7 +296,9 @@ class RotatingFileSink:
|
|
|
285
296
|
None.
|
|
286
297
|
|
|
287
298
|
Raises:
|
|
288
|
-
OSError: If a write
|
|
299
|
+
OSError: If a write or a flush fails. A failed *rotation* no longer raises; see
|
|
300
|
+
:meth:`_rotate_or_continue`.
|
|
301
|
+
SinkDeliveryError: If the sink is closed.
|
|
289
302
|
"""
|
|
290
303
|
if not batch:
|
|
291
304
|
return
|
|
@@ -298,7 +311,8 @@ class RotatingFileSink:
|
|
|
298
311
|
line = json.dumps(event) + "\n"
|
|
299
312
|
data = len(line.encode(self._encoding))
|
|
300
313
|
if self._should_rotate(data):
|
|
301
|
-
self.
|
|
314
|
+
self._stream.flush()
|
|
315
|
+
self._rotate_or_continue()
|
|
302
316
|
self._stream.write(line)
|
|
303
317
|
self._size += data
|
|
304
318
|
self._stream.flush()
|
|
@@ -425,6 +439,68 @@ class RotatingFileSink:
|
|
|
425
439
|
return True
|
|
426
440
|
return self._next_rollover is not None and time.monotonic() >= self._next_rollover
|
|
427
441
|
|
|
442
|
+
def _rotate_or_continue(self) -> None:
|
|
443
|
+
"""Rotates, or absorbs the failure and carries on writing to the un-rotated file.
|
|
444
|
+
|
|
445
|
+
A rotation that raised used to cost the batch twice. ``_rotate`` closes the active stream
|
|
446
|
+
first, so the events already written in this batch were on disk and the ``OSError``
|
|
447
|
+
propagated out of ``emit`` — the worker then re-sent the whole batch and wrote them again.
|
|
448
|
+
Measured: an 8-event batch failing after 3 were written put 11 lines on disk, 3 of them
|
|
449
|
+
duplicates. A persistent failure was worse: the sink kept a **closed** stream and every
|
|
450
|
+
later batch raised a raw ``PermissionError``, which is not a ``SinkDeliveryError`` and has
|
|
451
|
+
no ``losses()`` behind it.
|
|
452
|
+
|
|
453
|
+
Absorbing it costs nothing and duplicates nothing. The active file simply exceeds
|
|
454
|
+
``max_bytes`` until a rotation succeeds, which is what happens anyway when rotation is
|
|
455
|
+
impossible — the trade SPEC-027 FR-004 already took, that a leaked resource beats a
|
|
456
|
+
corrupt write.
|
|
457
|
+
|
|
458
|
+
``_next_rollover`` is re-armed as well as ``_size``: ``_rotate`` sets it on its last line,
|
|
459
|
+
so an absorbed failure would otherwise leave a **time** trigger permanently in the past.
|
|
460
|
+
|
|
461
|
+
**The re-arm does not damp the size trigger, and the diagnostic is what carries that.**
|
|
462
|
+
``_size`` is re-seeded from a file that is now over ``max_bytes``, so ``_should_rotate``'s
|
|
463
|
+
size branch stays true and every subsequent event attempts a rotation again — measured at
|
|
464
|
+
598 attempts over 600 events. The attempts are cheap and lose nothing, but an unthrottled
|
|
465
|
+
stderr write per event is not: ``PostgresSink._reconnect_if_broken`` records the same rule
|
|
466
|
+
for the same reason, that a diagnostic which floods is one an operator stops reading. So
|
|
467
|
+
the failure is announced **once per outage** and the flag clears on the next successful
|
|
468
|
+
rotation. The remaining per-event attempt is recorded in ``architecture.md`` §12 rather
|
|
469
|
+
than fixed here, because damping it means deferring a rotation the caller asked for.
|
|
470
|
+
|
|
471
|
+
**The reopen can itself raise, and that is not absorbed.** It is the same
|
|
472
|
+
``open(self._path, "a")`` call ``_rotate`` ends with, so whatever defeats it there —
|
|
473
|
+
a read-only mount, ``EMFILE``, a directory that lost write permission — defeats it here.
|
|
474
|
+
At that point this sink has no stream and cannot continue, so there is nothing to absorb
|
|
475
|
+
*into*; the ``OSError`` reaches ``emit`` and the worker retries the batch, duplicating the
|
|
476
|
+
prefix the pre-rotation flush had already written. That residue is recorded rather than
|
|
477
|
+
fixed: it is unchanged from before SPEC-048, the surviving events are on disk rather than
|
|
478
|
+
lost, and inventing a half-open state to avoid a duplicate would trade a visible
|
|
479
|
+
duplication for a silent loss.
|
|
480
|
+
|
|
481
|
+
Args:
|
|
482
|
+
None.
|
|
483
|
+
|
|
484
|
+
Returns:
|
|
485
|
+
None.
|
|
486
|
+
|
|
487
|
+
Raises:
|
|
488
|
+
OSError: If the *reopen* fails, per the paragraph above. A failed **rotation** is
|
|
489
|
+
absorbed and announced through ``_diag``; the batch continues, nothing is dropped, no
|
|
490
|
+
counter moves, and this class still has no ``losses()``.
|
|
491
|
+
"""
|
|
492
|
+
try:
|
|
493
|
+
self._rotate()
|
|
494
|
+
except OSError as err:
|
|
495
|
+
self._stream = open(self._path, "a", encoding=self._encoding)
|
|
496
|
+
self._size = os.path.getsize(self._path) if os.path.exists(self._path) else 0
|
|
497
|
+
self._next_rollover = self._schedule_next()
|
|
498
|
+
if not self._rotation_failing:
|
|
499
|
+
self._rotation_failing = True
|
|
500
|
+
_diag.absorbed("rotating RotatingFileSink", err)
|
|
501
|
+
return
|
|
502
|
+
self._rotation_failing = False
|
|
503
|
+
|
|
428
504
|
def _rotate(self) -> None:
|
|
429
505
|
"""Closes the active file, shifts and prunes backups, then opens a fresh active file.
|
|
430
506
|
|
|
@@ -203,6 +203,22 @@ class FirehoseSink:
|
|
|
203
203
|
this loop re-sends the records the destination flagged, and the canonical reason it flags
|
|
204
204
|
them is throttling, which an immediate re-send makes worse.
|
|
205
205
|
|
|
206
|
+
**A client exception costs this chunk, never the batch** (SPEC-048 FR-002). It used to
|
|
207
|
+
propagate out of :meth:`emit`, so a failure on chunk N after chunks 1..N-1 had landed made
|
|
208
|
+
the worker re-send the whole batch and duplicate everything already delivered — the exit
|
|
209
|
+
drain, one large batch by construction, is exactly that shape. The guard sits around the
|
|
210
|
+
client call *inside* this loop rather than around ``_send`` in ``emit``, so only the
|
|
211
|
+
records still outstanding at the failing attempt are charged and the ones already accepted
|
|
212
|
+
still count toward the return.
|
|
213
|
+
|
|
214
|
+
A client exception is treated as **provable non-delivery** for the chunk, so a wholly
|
|
215
|
+
failed batch still raises and ``unknown`` is untouched: SPEC-018's "unadjudicable" is a
|
|
216
|
+
property of a *response*, and an exception is not a response. The cost is written down —
|
|
217
|
+
a read timeout means the request went out and the reply was lost, so a re-send may
|
|
218
|
+
duplicate — and taken because an unreachable endpoint is the common case and is exactly
|
|
219
|
+
what the worker's retry exists for, while suppressing the raise would lose every event of
|
|
220
|
+
every batch for a whole outage, silently.
|
|
221
|
+
|
|
206
222
|
Args:
|
|
207
223
|
records: One chunk's request entries.
|
|
208
224
|
|
|
@@ -214,13 +230,27 @@ class FirehoseSink:
|
|
|
214
230
|
SPEC-018 settled must never be re-sent.
|
|
215
231
|
|
|
216
232
|
Raises:
|
|
217
|
-
Exception
|
|
233
|
+
None. ``Exception`` is caught rather than ``BaseException``, so a ``KeyboardInterrupt``
|
|
234
|
+
or ``SystemExit`` still reaches the caller (SPEC-025 FR-004).
|
|
218
235
|
"""
|
|
219
236
|
sent = len(records)
|
|
220
237
|
for attempt in range(self.max_retries + 1):
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
238
|
+
try:
|
|
239
|
+
response = self.client.put_record_batch(
|
|
240
|
+
DeliveryStreamName=self.delivery_stream, Records=records
|
|
241
|
+
)
|
|
242
|
+
except Exception as err:
|
|
243
|
+
if attempt < self.max_retries:
|
|
244
|
+
wait(_BACKOFF_BASE * (2**attempt), self.log_foundry_stop_signal)
|
|
245
|
+
continue
|
|
246
|
+
with self._counter_lock:
|
|
247
|
+
self.failed += len(records)
|
|
248
|
+
_diag.lost(
|
|
249
|
+
"record",
|
|
250
|
+
len(records),
|
|
251
|
+
f"FirehoseSink, {self.max_retries + 1} attempt(s), {type(err).__name__}",
|
|
252
|
+
)
|
|
253
|
+
return sent - len(records)
|
|
224
254
|
if not response.get("FailedPutCount"):
|
|
225
255
|
return sent
|
|
226
256
|
results = usable_results(response.get("RequestResponses"))
|