log-foundry 0.10.2.dev77__tar.gz → 0.10.2.dev79__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/PKG-INFO +15 -4
  2. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/README.md +14 -3
  3. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/pyproject.toml +1 -1
  4. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/__init__.py +10 -6
  5. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/_lifecycle.py +235 -12
  6. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/worker.py +28 -1
  7. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/LICENSE +0 -0
  8. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/_diag.py +0 -0
  9. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/_fork.py +0 -0
  10. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/api.py +0 -0
  11. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/config.py +0 -0
  12. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/console.py +0 -0
  13. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/context.py +0 -0
  14. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/decorator.py +0 -0
  15. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/ids.py +0 -0
  16. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/model.py +0 -0
  17. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/py.typed +0 -0
  18. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/results.py +0 -0
  19. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sanitize.py +0 -0
  20. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/__init__.py +0 -0
  21. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/_batch.py +0 -0
  22. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/_chunk.py +0 -0
  23. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/_retry.py +0 -0
  24. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/_socket.py +0 -0
  25. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/_time.py +0 -0
  26. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/base.py +0 -0
  27. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/callback.py +0 -0
  28. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/clickhouse.py +0 -0
  29. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/datadog.py +0 -0
  30. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/elasticsearch.py +0 -0
  31. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/eventhubs.py +0 -0
  32. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/file.py +0 -0
  33. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/filtering.py +0 -0
  34. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/firehose.py +0 -0
  35. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/honeycomb.py +0 -0
  36. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/http.py +0 -0
  37. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/kafka.py +0 -0
  38. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/kinesis.py +0 -0
  39. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/logging_sink.py +0 -0
  40. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/logstash.py +0 -0
  41. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/loki.py +0 -0
  42. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/memory.py +0 -0
  43. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/mongodb.py +0 -0
  44. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/multi.py +0 -0
  45. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/nats.py +0 -0
  46. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/newrelic.py +0 -0
  47. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/null.py +0 -0
  48. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/postgres.py +0 -0
  49. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/pubsub.py +0 -0
  50. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/rabbitmq.py +0 -0
  51. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/redis.py +0 -0
  52. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/sentry.py +0 -0
  53. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/sns.py +0 -0
  54. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/splunk.py +0 -0
  55. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/sqlite.py +0 -0
  56. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/sqs.py +0 -0
  57. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/stdout.py +0 -0
  58. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/syslog.py +0 -0
  59. {log_foundry-0.10.2.dev77 → log_foundry-0.10.2.dev79}/src/log_foundry/sinks/transform.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: log-foundry
3
- Version: 0.10.2.dev77
3
+ Version: 0.10.2.dev79
4
4
  Summary: Generate logs for your console and JSON events for downstream consumption.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -1096,21 +1096,32 @@ which one you want depends on whether the process is about to end:
1096
1096
  import log_foundry as lf
1097
1097
 
1098
1098
  lf.flush() # drain to the sink and keep going; truthy when everything landed
1099
- lf.shutdown() # drain, close the sink, and stop for good; blocks until drained (30s cap)
1099
+ lf.shutdown() # drain, close the sink, and stop for good; blocks until drained (30s cap on
1100
+ # the drain — the sink's own close() is not bounded by it, see below)
1100
1101
  ```
1101
1102
 
1102
- Both are bounded, because both can be called somewhere with a deadline. `flush(timeout=5.0)`
1103
+ Both take a deadline, because both can be called somewhere with one. `flush(timeout=5.0)`
1103
1104
  is falsy if the drain did not complete; `shutdown(timeout=30.0)` returns having stopped
1104
1105
  what it could, and reports `health().stopped_reason == "ShutdownTimeout"`. Passing `None` to
1105
1106
  either waits indefinitely, which is unsafe in any environment with an execution deadline.
1106
1107
 
1108
+ **What `shutdown()`'s timeout does not bound is the live sink's own `close()`**, which runs
1109
+ inline on either delivery path. A sink that blocks for a minute inside `close()` holds
1110
+ `shutdown()` — and the process — for that minute, whatever you passed. Measured 6.01 s against
1111
+ `shutdown(timeout=2.0)` with a 6-second close. This is deliberate rather than an oversight: both
1112
+ ways of bounding it were built and reverted (a daemon closer is killed at interpreter exit
1113
+ wherever it has reached, which for `SQLiteSink` can be inside `commit()`), and bounding it
1114
+ properly needs `close()` itself to be interruptible, which the sink contract does not require.
1115
+ If your sink's close can block, give it its own internal timeout.
1116
+
1107
1117
  **What a broken destination can cost you.** There is one drain thread, so a sink's backoff pauses
1108
1118
  *all* log delivery, and it spans `shutdown()`. At the defaults (`max_retries=3`) that is 0.7 s of
1109
1119
  backoff per batch for most sinks (per *message* for the socket-backed ones — ~70 s for a
1110
1120
  100-message batch against a dead syslog host), and up to 90 s for an HTTP sink whose destination
1111
1121
  is sending
1112
1122
  `Retry-After` — clamped to `max_retry_after=30.0` per wait, which you can lower. Every wait is cut
1113
- short by a shutdown, and `shutdown()`'s own timeout bounds the total either way. Each sink's class
1123
+ short by a shutdown, and `shutdown()`'s own timeout bounds the total either way — for the
1124
+ *drain*; a wait taken inside the sink's `close()` is not bounded by it (above). Each sink's class
1114
1125
  docstring states its own worst case.
1115
1126
 
1116
1127
  | | `flush()` | `shutdown()` |
@@ -1060,21 +1060,32 @@ which one you want depends on whether the process is about to end:
1060
1060
  import log_foundry as lf
1061
1061
 
1062
1062
  lf.flush() # drain to the sink and keep going; truthy when everything landed
1063
- lf.shutdown() # drain, close the sink, and stop for good; blocks until drained (30s cap)
1063
+ lf.shutdown() # drain, close the sink, and stop for good; blocks until drained (30s cap on
1064
+ # the drain — the sink's own close() is not bounded by it, see below)
1064
1065
  ```
1065
1066
 
1066
- Both are bounded, because both can be called somewhere with a deadline. `flush(timeout=5.0)`
1067
+ Both take a deadline, because both can be called somewhere with one. `flush(timeout=5.0)`
1067
1068
  is falsy if the drain did not complete; `shutdown(timeout=30.0)` returns having stopped
1068
1069
  what it could, and reports `health().stopped_reason == "ShutdownTimeout"`. Passing `None` to
1069
1070
  either waits indefinitely, which is unsafe in any environment with an execution deadline.
1070
1071
 
1072
+ **What `shutdown()`'s timeout does not bound is the live sink's own `close()`**, which runs
1073
+ inline on either delivery path. A sink that blocks for a minute inside `close()` holds
1074
+ `shutdown()` — and the process — for that minute, whatever you passed. Measured 6.01 s against
1075
+ `shutdown(timeout=2.0)` with a 6-second close. This is deliberate rather than an oversight: both
1076
+ ways of bounding it were built and reverted (a daemon closer is killed at interpreter exit
1077
+ wherever it has reached, which for `SQLiteSink` can be inside `commit()`), and bounding it
1078
+ properly needs `close()` itself to be interruptible, which the sink contract does not require.
1079
+ If your sink's close can block, give it its own internal timeout.
1080
+
1071
1081
  **What a broken destination can cost you.** There is one drain thread, so a sink's backoff pauses
1072
1082
  *all* log delivery, and it spans `shutdown()`. At the defaults (`max_retries=3`) that is 0.7 s of
1073
1083
  backoff per batch for most sinks (per *message* for the socket-backed ones — ~70 s for a
1074
1084
  100-message batch against a dead syslog host), and up to 90 s for an HTTP sink whose destination
1075
1085
  is sending
1076
1086
  `Retry-After` — clamped to `max_retry_after=30.0` per wait, which you can lower. Every wait is cut
1077
- short by a shutdown, and `shutdown()`'s own timeout bounds the total either way. Each sink's class
1087
+ short by a shutdown, and `shutdown()`'s own timeout bounds the total either way — for the
1088
+ *drain*; a wait taken inside the sink's `close()` is not bounded by it (above). Each sink's class
1078
1089
  docstring states its own worst case.
1079
1090
 
1080
1091
  | | `flush()` | `shutdown()` |
@@ -20,7 +20,7 @@ dependencies = [
20
20
  ]
21
21
 
22
22
  # Optional features. Install with: pip install log-foundry[aws]
23
- version = "0.10.2.dev77"
23
+ version = "0.10.2.dev79"
24
24
 
25
25
  [project.optional-dependencies]
26
26
  aws = ["boto3>=1.43.61"] # SQSSink, SNSSink, KinesisSink, FirehoseSink
@@ -152,12 +152,16 @@ def shutdown(timeout: float | None = DEFAULT_SHUTDOWN_TIMEOUT) -> None:
152
152
  Args:
153
153
  timeout: Seconds bounding the wait for the background thread and, carved from the same
154
154
  budget, a short grace for any sink still closing after a late ``configure(sink=...)``
155
- (SPEC-027 FR-004, SPEC-030 FR-003). ``None``
156
- waits indefinitely, which is what this did unconditionally before and is still
157
- available on request, but is unsafe anywhere with an execution deadline — ``atexit`` is
158
- one such place, where a sink blocked in a network call would hold the process open. An
159
- expired shutdown reports a ``stopped_reason`` of ``"ShutdownTimeout"`` and leaves the
160
- sink open, since the drain thread may still be inside ``emit``.
155
+ (SPEC-027 FR-004, SPEC-030 FR-003). It does **not** bound the live sink's own
156
+ ``close()``, which runs inline on either delivery path and is unbounded — so a sink
157
+ blocked in a network call inside ``close()`` holds this call, and the process, open for
158
+ as long as it blocks, whatever ``timeout`` says (measured 6.01 s against
159
+ ``timeout=2.0``; ``architecture.md`` §13). Bounding it needs ``Sink.close`` to be
160
+ interruptible, which the sink contract does not require. ``None`` waits indefinitely
161
+ for the parts that are bounded, which is what this did unconditionally before and is
162
+ still available on request. An expired shutdown reports a ``stopped_reason`` of
163
+ ``"ShutdownTimeout"`` and leaves the sink open, since the drain thread may still be
164
+ inside ``emit``.
161
165
 
162
166
  Returns:
163
167
  None.
@@ -109,6 +109,26 @@ class _Lifecycle:
109
109
  None.
110
110
  """
111
111
 
112
+ _FORK_SKIP = ("_orphan_closed_sink",)
113
+ """Keeps the superseded sink out of ``_fork``'s repair walk (SPEC-044 FR-005).
114
+
115
+ The same hazard the module-level :data:`_FORK_SKIP` declares for ``_owned``, at a slot that
116
+ declaration cannot reach. ``_fork._skipped_names`` reads the opt-out off **the holder of the
117
+ attribute** — a plain ``getattr`` — so a module global is consulted only for the module's own
118
+ namespace, while ``_orphan_closed_sink`` lives on this instance. Measured before the fix: a
119
+ child of ``configure(A)`` → ``info()`` → ``configure(B)`` ran ``reacquire_after_fork()`` on
120
+ both, and a ``FileSink`` in A's place would have its file re-opened on every fork forever.
121
+
122
+ A **class** attribute deliberately. ``_fork._namespace_items`` reads ``vars(holder)``, the
123
+ instance ``__dict__`` only, so a plain ``getattr`` finds this while the walk over
124
+ :data:`_state` does not see it. The walk does reach it once, through the class itself, and
125
+ harmlessly: it is a tuple of strings, which holds no primitive to replace and no sink to
126
+ hook. The module-level tuple is unchanged and still needed; neither is the whole rule.
127
+
128
+ Marking is not narrowed: :func:`_inheritance_roots` reads the slot directly, so a child still
129
+ marks an inherited superseded sink foreign and still refuses to close it (SPEC-042 FR-001).
130
+ """
131
+
112
132
  def __init__(self) -> None:
113
133
  """Builds the process's one lifecycle owner, in the cold state.
114
134
 
@@ -133,6 +153,8 @@ class _Lifecycle:
133
153
  self._orphan_closed_sink: Sink | None = None
134
154
  self._orphan_stop = threading.Event()
135
155
  self._orphan_retired = False
156
+ self._shutdown_running = 0
157
+ self._late_worker: Worker | None = None
136
158
 
137
159
  def worker_exists(self) -> Worker | None:
138
160
  """Existence — is there a worker at all, and therefore anything to do (arch §9.2).
@@ -291,6 +313,86 @@ all.
291
313
  _closers: list[threading.Thread] = []
292
314
  _closers_lock = threading.Lock()
293
315
 
316
+ _closing_now: set[int] = set()
317
+ """Ids of the sinks a :func:`release` is running against **right now** (SPEC-044 FR-003).
318
+
319
+ The *moment* term of an ownership-and-moment question (arch §9.2), applied to the close rather
320
+ than to the worker. :func:`_offer_orphan_signal` replaces a stop event that is already set, so a
321
+ sink is never left holding one that collapses every later backoff to zero — right, and pinned by
322
+ SPEC-033 FR-004 for a sink adopted **after** ``shutdown()``. What it must not do is cancel the
323
+ signal a close is *currently waiting on*: measured, an ``info()`` landing inside the close made
324
+ ``shutdown()`` serve an 8 s backoff in full, against 0.00 s with no racing log, on both delivery
325
+ paths.
326
+
327
+ Retirement is the wrong discriminator — it would break SPEC-033 FR-004's two tests — and this is
328
+ the right one, because :func:`release` is the single path by which the library ever closes a sink
329
+ (SPEC-042 FR-002), so one registration there covers the orphan close, ``Worker._close_if_owed``,
330
+ the swap's detached closer and every wrapper.
331
+
332
+ It holds ``int`` ids, not sinks, so it pins nothing against garbage collection and needs no
333
+ :data:`_FORK_SKIP` entry; the registration brackets ``close()`` inside :func:`release`, so an id
334
+ cannot be reused while it is registered. **A fork breaks that bracket** — the child inherits a
335
+ registration whose ``finally`` no thread will ever run — so :func:`_clear_closing_after_fork`
336
+ empties it in the child. Measured before that handler existed: the id survived, and once the
337
+ child set its own ``_orphan_stop`` that sink was handed the **set** event and backed off not at
338
+ all, permanently.
339
+ """
340
+ _closing_now_lock = threading.Lock()
341
+ """Guards :data:`_closing_now`, and sits **last** in the lock order.
342
+
343
+ ``_state._lock`` -> ``_config_lock`` -> ``_owned_lock`` is the order :data:`_owned_lock` states;
344
+ this one is taken under ``_state._lock`` (through :func:`_offer_orphan_signal`) and is never
345
+ nested with either of the other two, so there is no cycle. It is held only across a set
346
+ membership test or a single mutation, never across a ``close()``.
347
+ """
348
+
349
+
350
+ def _closing(sink: Sink) -> bool:
351
+ """Whether a release of this sink is in flight on some thread right now (FR-003).
352
+
353
+ Args:
354
+ sink: The sink about to be offered a stop signal.
355
+
356
+ Returns:
357
+ Whether :func:`release` is inside that sink's ``close()``.
358
+
359
+ Raises:
360
+ None.
361
+ """
362
+ with _closing_now_lock:
363
+ return id(sink) in _closing_now
364
+
365
+
366
+ def _clear_closing_after_fork() -> None:
367
+ """Drops the in-flight close registrations a child inherited (FR-003).
368
+
369
+ The registration is removed by a ``finally`` in :func:`release`, on the thread performing the
370
+ close — and a forked child has only the thread that called ``fork()``, so every inherited
371
+ entry is one nothing will ever clear. Left in place it is not a missed refresh but a
372
+ permanent one: once the child sets its own ``_orphan_stop``, that sink is handed the set
373
+ event and every backoff collapses to zero, which is SPEC-033 FR-004's tight retry loop.
374
+
375
+ Registered with ``_fork`` rather than reached for by it, the inversion SPEC-039 FR-006
376
+ requires so that ``_fork`` imports nothing but ``_diag``. It takes the registry's own lock,
377
+ which the repair walk re-initialised moments earlier.
378
+
379
+ It runs **after** :func:`_mark_inherited` and before :func:`_rebuild_worker_after_fork`, and
380
+ the placement is free rather than load-bearing: the registry holds ``int`` ids and no handler
381
+ reads it.
382
+
383
+ Args:
384
+ None.
385
+
386
+ Returns:
387
+ None.
388
+
389
+ Raises:
390
+ None.
391
+ """
392
+ with _closing_now_lock:
393
+ _closing_now.clear()
394
+
395
+
294
396
  _FORK_SKIP = ("_owned",)
295
397
  """Keeps the ownership record out of ``_fork``'s repair walk (``_fork._SKIP_ATTRIBUTE``).
296
398
 
@@ -299,6 +401,11 @@ otherwise reach ones the process abandoned several ``configure()`` calls ago and
299
401
  hooks — measured, a child announced a buffer discard for a superseded sink, and a ``FileSink``
300
402
  there would be reopened on every fork forever. A sink that is still live is reached through the
301
403
  config and the worker, so nothing the repair needs is lost.
404
+
405
+ This covers only **this module's own namespace**. ``_orphan_closed_sink`` pins a superseded sink
406
+ for the same reason and is an attribute of :data:`_state`, which ``_fork._skipped_names`` asks
407
+ separately — so :attr:`_Lifecycle._FORK_SKIP` declares it there, and the two together are the
408
+ rule (SPEC-044 FR-005).
302
409
  """
303
410
 
304
411
  _FOREIGN = -1
@@ -842,7 +949,13 @@ def release(sink: Sink, *, detached: bool = False, owner: object = None) -> thre
842
949
  return None
843
950
  if detached:
844
951
  return _start_closer(sink)
845
- sink.close()
952
+ with _closing_now_lock:
953
+ _closing_now.add(id(sink))
954
+ try:
955
+ sink.close()
956
+ finally:
957
+ with _closing_now_lock:
958
+ _closing_now.discard(id(sink))
846
959
  return None
847
960
 
848
961
 
@@ -1017,6 +1130,29 @@ def _get_worker() -> Worker:
1017
1130
  asking :meth:`_Lifecycle.worker_exists` rather than reading a module attribute. Against an
1018
1131
  ~18.5 µs traced call that is 0.06%, and end-to-end timing is identical either side.
1019
1132
 
1133
+ **The orphan record is not discarded without deciding who closes it** (SPEC-044 FR-002).
1134
+ ``configure()`` writes ``_config.sink`` *before* it takes this lock, so a ``configure(sink=B)``
1135
+ blocked here leaves ``_ensure_sink()`` returning B while the record still names A — which has
1136
+ events and has never been closed. Clearing unconditionally lost A's close entirely, with
1137
+ ``incomplete_swaps`` at zero because every field of ``Health`` describes a worker and the
1138
+ worker was fine (natural rate 6/400). Where the new worker did not adopt the recorded sink,
1139
+ this transition owns that close: latch it and release it detached, the shape
1140
+ :func:`_swap_sink`'s no-worker branch already uses. The closer is **not** joined — this branch
1141
+ runs at most once per process, ``join_closers`` grants it the exit grace and
1142
+ ``health().closing_sinks`` reports it live, and joining would put a sink's whole close on the
1143
+ first traced call in the process.
1144
+
1145
+ **A worker built while a ``shutdown()`` is running is registered for it** (SPEC-044 FR-001).
1146
+ ``_shutdown_worker`` raises a depth counter and reads the worker in one critical section, so a
1147
+ worker built after that read is registered here and drained by that same call rather than
1148
+ left running with nothing to stop it. ``sink_released`` covers the ordering where the orphan
1149
+ branch already closed this sink: the worker inherits a discharged close instead of performing
1150
+ a second one. In the other ordering the worker is registered first and
1151
+ :func:`_close_orphan_sink`'s ownership guard declines, so the worker performs the only close.
1152
+ Both are gated on the counter, which is what keeps this off the **sequential**
1153
+ ``shutdown()`` → ``@trace`` path, where a fresh worker still delivers by the decision
1154
+ :func:`_worker_health` records.
1155
+
1020
1156
  Args:
1021
1157
  None.
1022
1158
 
@@ -1035,9 +1171,20 @@ def _get_worker() -> Worker:
1035
1171
  worker = _state.worker_exists()
1036
1172
  if worker is None:
1037
1173
  _register_exit_handler()
1038
- worker = Worker(_ensure_sink())
1174
+ sink = _ensure_sink()
1175
+ worker = Worker(
1176
+ sink,
1177
+ sink_released=_state._shutdown_running > 0
1178
+ and _state._orphan_closed_sink is sink,
1179
+ )
1039
1180
  _state._worker = worker
1181
+ if _state._shutdown_running > 0:
1182
+ _state._late_worker = worker
1183
+ owed = _state._orphan_sink
1040
1184
  _state._orphan_sink = None
1185
+ if owed is not None and owed is not worker.sink:
1186
+ _state._orphan_closed_sink = owed
1187
+ release(owed, detached=True)
1041
1188
  return worker
1042
1189
  def _register_exit_handler() -> None:
1043
1190
  """Registers the one ``atexit`` handler that covers both delivery paths (SPEC-031 FR-006).
@@ -1170,6 +1317,14 @@ def _offer_orphan_signal(sink: Sink) -> None:
1170
1317
  destination is a tight retry loop. SPEC-027's contract is "cut short by a shutdown", not
1171
1318
  "never wait again".
1172
1319
 
1320
+ **Except while a release of this sink is in flight** (SPEC-044 FR-003), where the replacement
1321
+ cancels the signal the close is waiting on: measured, an ``info()`` landing inside the close
1322
+ made ``shutdown()`` serve an 8 s backoff in full, against 0.00 s with no racing log, on both
1323
+ delivery paths. The discriminator is the **moment**, not retirement — retirement would
1324
+ un-refresh a sink adopted after ``shutdown()``, which SPEC-033 FR-004 measured and pinned.
1325
+ Returning rather than offering the set event is the direct expression of "the signal it holds
1326
+ is not replaced", and depends on nothing about which event ``_orphan_stop`` currently is.
1327
+
1173
1328
  Callers hold ``_state._lock``.
1174
1329
 
1175
1330
  Args:
@@ -1183,6 +1338,8 @@ def _offer_orphan_signal(sink: Sink) -> None:
1183
1338
  """
1184
1339
  if _state.worker_owns_now(sink):
1185
1340
  return
1341
+ if _closing(sink):
1342
+ return
1186
1343
  offer_stop_signal(sink, _state.refresh_stop_signal())
1187
1344
  def _close_orphan_sink() -> None:
1188
1345
  """Closes a sink only the orphan path ever wrote to, once (SPEC-031 FR-006).
@@ -1257,6 +1414,26 @@ def _shutdown_worker(timeout: float | None = DEFAULT_SHUTDOWN_TIMEOUT) -> None:
1257
1414
  ``_state._orphan_stop`` is set **before** delegating, so a sink parked in a backoff is released
1258
1415
  while :meth:`Worker.shutdown` is still draining rather than after it has given up waiting.
1259
1416
 
1417
+ **The retirement latch and the worker read are one critical section** (SPEC-044 FR-001).
1418
+ Unlocked, a worker built between them sent this down the no-worker branch: the drain thread
1419
+ was never stopped and its sink never closed, while ``health()`` reported ``retired=True`` and
1420
+ later logs were delivered by a live worker with ``submitted_after_shutdown`` at zero.
1421
+ ``atexit`` recovered it in a process that exits; a frozen serverless container never does,
1422
+ which is the deployment this call exists for. Closing the read window alone only narrows it,
1423
+ so ``_shutdown_running`` stays raised for the whole of the no-worker branch and
1424
+ :func:`_get_worker` registers what it builds under the same lock — read and lowered in the
1425
+ last critical section, so there is no gap between the two.
1426
+
1427
+ It is a **depth counter, not a flag**. Two concurrent ``shutdown()`` calls are documented as
1428
+ normal (``Worker._close_if_owed``), and with a boolean the first to finish lowers it while the
1429
+ second is still running — measured, a worker built at that instant was registered nowhere and
1430
+ the second call returned having stopped nothing, which is the original defect verbatim.
1431
+
1432
+ The late worker's drain is charged against **this call's** deadline, and :func:`join_closers`
1433
+ is not also called on that path: :meth:`Worker.shutdown` grants the closer grace itself, and
1434
+ granting it twice measured 4.01 s against a 2 s grace — the same double charge this function's
1435
+ branches were already arranged to avoid.
1436
+
1260
1437
  The closer grace is granted **once**, by whichever path owns this call.
1261
1438
  :meth:`Worker.shutdown` already grants it — on its successful path and on its idempotent one,
1262
1439
  which is what covers a first shutdown that expired before reaching it — so joining again here
@@ -1273,15 +1450,27 @@ def _shutdown_worker(timeout: float | None = DEFAULT_SHUTDOWN_TIMEOUT) -> None:
1273
1450
  Raises:
1274
1451
  None.
1275
1452
  """
1276
- _state._orphan_retired = True
1277
- _state._orphan_stop.set()
1278
1453
  deadline = None if timeout is None else monotonic() + timeout
1279
- worker = _state.worker_exists()
1454
+ with _state._lock:
1455
+ _state._orphan_retired = True
1456
+ _state._orphan_stop.set()
1457
+ worker = _state.worker_exists()
1458
+ if worker is None:
1459
+ _state._shutdown_running += 1
1280
1460
  if worker is not None:
1281
1461
  worker.shutdown(timeout)
1282
1462
  _close_orphan_sink()
1283
1463
  return
1284
- _close_orphan_sink()
1464
+ try:
1465
+ _close_orphan_sink()
1466
+ finally:
1467
+ with _state._lock:
1468
+ late_worker = _state._late_worker
1469
+ _state._late_worker = None
1470
+ _state._shutdown_running -= 1
1471
+ if late_worker is not None:
1472
+ late_worker.shutdown(None if deadline is None else max(0.0, deadline - monotonic()))
1473
+ return
1285
1474
  join_closers(None if deadline is None else max(0.0, deadline - monotonic()))
1286
1475
  def _swap_sink(new_sink: Sink, timeout: float | None = DEFAULT_SWAP_TIMEOUT) -> None:
1287
1476
  """Retargets delivery at a new sink, backing a late ``configure(sink=...)``.
@@ -1335,6 +1524,32 @@ def _swap_sink(new_sink: Sink, timeout: float | None = DEFAULT_SWAP_TIMEOUT) ->
1335
1524
  that could not be confirmed (SPEC-030), and there is no drain here; an expired close join
1336
1525
  reports nothing at all, by the decision that made the bounded close available.
1337
1526
 
1527
+ **The worker branch latches what it hands over, and decides the close before clearing**
1528
+ (SPEC-044 FR-004, FR-002). It used to clear ``_state._orphan_sink`` and record nothing: an
1529
+ orphan emit that resolved the old sink before the swap and resumed after it then re-armed a
1530
+ sink :meth:`Worker.swap_sink` had already closed, and the exit close performed a second
1531
+ ``close()`` on it — measured ``A.closed == 2`` with a preemption point injected at
1532
+ ``_ensure_sink``. ``sinks/base.py`` asks an implementation to make its release idempotent,
1533
+ but the library does not *rely* on that — it cannot enforce what a third-party sink does — so
1534
+ it performs one close (SPEC-032).
1535
+
1536
+ The latch is keyed on ``worker.sink``, **not** on the orphan record. The sink
1537
+ ``Worker.swap_sink`` is about to close is the one the worker holds, and in the reproduced case
1538
+ the record is ``None`` — the worker cleared it when it was built — so keying on the record
1539
+ latched nothing and left the defect exactly where it was. It is set even where the drain
1540
+ cannot be confirmed and the swap therefore leaves that sink **open**: the reason there is not
1541
+ "it was closed" but that a racing orphan emit must not re-arm a sink whose drain thread may
1542
+ still be inside ``emit``.
1543
+
1544
+ Where the record names a **third** sink — neither the worker's nor the new one, reachable
1545
+ only by a preempted emit re-arming across an earlier swap — nothing else would close it, so
1546
+ this branch owns it and releases it detached, exactly as the no-worker branch does. That
1547
+ closer is not joined: the worker branch returns through :meth:`Worker.swap_sink`'s own
1548
+ deadline, ``join_closers`` grants the exit grace, and ``health().closing_sinks`` reports it
1549
+ live. The slot is single, so the worker's sink is latched **last** and wins: it is the case
1550
+ that reproduces without a compound race, and the bound is the one ``architecture.md`` §13
1551
+ already records for a latch that moves.
1552
+
1338
1553
  Args:
1339
1554
  new_sink: The sink already written to the config, to be made the live delivery target.
1340
1555
  timeout: Seconds bounding the whole swap — the drains, where there are any, and the close
@@ -1349,10 +1564,17 @@ def _swap_sink(new_sink: Sink, timeout: float | None = DEFAULT_SWAP_TIMEOUT) ->
1349
1564
  cannot start.
1350
1565
  """
1351
1566
  closer = None
1567
+ deadline = None if timeout is None else monotonic() + timeout
1352
1568
  with _state._lock:
1353
1569
  worker = _state.live_worker()
1354
1570
  if worker is not None:
1571
+ owed = _state._orphan_sink
1355
1572
  _state._orphan_sink = None
1573
+ if owed is not None and owed is not new_sink and owed is not worker.sink:
1574
+ _state._orphan_closed_sink = owed
1575
+ closer = release(owed, detached=True)
1576
+ if worker.sink is not new_sink:
1577
+ _state._orphan_closed_sink = worker.sink
1356
1578
  else:
1357
1579
  old = _state._orphan_sink
1358
1580
  if old is None or old is new_sink:
@@ -1369,12 +1591,11 @@ def _swap_sink(new_sink: Sink, timeout: float | None = DEFAULT_SWAP_TIMEOUT) ->
1369
1591
  _diag.absorbed(
1370
1592
  "swapping the log sink", exc, "events may still be delivered to the previous sink"
1371
1593
  )
1372
- return
1373
- if not worker_holds_sink:
1374
- _adopt_declined_swap(new_sink)
1375
- return
1594
+ else:
1595
+ if not worker_holds_sink:
1596
+ _adopt_declined_swap(new_sink)
1376
1597
  if closer is not None:
1377
- closer.join(timeout)
1598
+ closer.join(None if deadline is None else max(0.0, deadline - monotonic()))
1378
1599
  def _adopt_declined_swap(new_sink: Sink) -> None:
1379
1600
  """Takes ownership of a sink a worker refused mid-swap (SPEC-035 FR-003).
1380
1601
 
@@ -1397,7 +1618,8 @@ def _adopt_declined_swap(new_sink: Sink) -> None:
1397
1618
  :func:`_note_orphan_emit` carries and for its reason. Without it, an orphan log arming
1398
1619
  ``_state._orphan_sink`` while this thread is inside ``swap_sink``'s first drain, followed by a
1399
1620
  ``shutdown()`` that closes it, lets this re-arm a closed sink for a second ``close()`` at
1400
- exit — reproduced, and ``Sink.close`` promises no idempotency.
1621
+ exit — reproduced. ``sinks/base.py`` asks an implementation to make its release idempotent,
1622
+ but the library does not rely on it, for the reason :func:`_swap_sink` states.
1401
1623
 
1402
1624
  ``incomplete_swaps`` is deliberately not moved. It counts an unconfirmed *drain*, and this
1403
1625
  swap had no drain to confirm — the worker declined before reassigning anything — so counting
@@ -1622,4 +1844,5 @@ def _worker_health() -> Health:
1622
1844
 
1623
1845
 
1624
1846
  _fork.register_child_handler(_mark_inherited)
1847
+ _fork.register_child_handler(_clear_closing_after_fork)
1625
1848
  _fork.register_child_handler(_rebuild_worker_after_fork)
@@ -238,6 +238,7 @@ class Worker:
238
238
  flush_interval: float = 1.0,
239
239
  max_queue: int = 10_000,
240
240
  max_retries: int = 3,
241
+ sink_released: bool = False,
241
242
  ) -> None:
242
243
  """Starts the drain thread and offers the sink this worker's stop signal.
243
244
 
@@ -245,12 +246,28 @@ class Worker:
245
246
  shutdown once-only flag, since ``shutdown`` may be called concurrently by ``atexit``
246
247
  and user code.
247
248
 
249
+ ``sink_released`` says the close is **already discharged** by whoever released this sink,
250
+ never that the sink is unusable (SPEC-044 FR-001). Only ``_lifecycle._get_worker`` passes
251
+ it, and only for a worker built while a ``shutdown()`` was mid-flight over the very sink
252
+ that shutdown's orphan branch had just closed — without it the exit close performs a
253
+ second ``close()``. ``sinks/base.py`` requires an implementation to make its release
254
+ idempotent, but the library must not *rely* on that: it cannot enforce what a
255
+ third-party sink does, and a half-released transport is the failure SPEC-032 exists to
256
+ prevent. So the library performs one close and does not test whether a sink survived
257
+ two. This
258
+ worker still emits to that sink: one that guards its post-close state refuses and the
259
+ batch lands in ``failed_batches``, which is the documented signal on this path. The flag
260
+ describes **one** sink, so :meth:`swap_sink` clears it when it adopts another — otherwise
261
+ the claim would transfer to every sink this worker later held, and the next one would be
262
+ closed by nobody.
263
+
248
264
  Args:
249
265
  sink: The destination every batch is emitted to.
250
266
  batch_size: How many submissions accumulate before an emit is triggered.
251
267
  flush_interval: Seconds before a partial batch is emitted anyway.
252
268
  max_queue: Ceiling on buffered submissions, past which the newest is dropped.
253
269
  max_retries: Retries after a failing emit, floored at zero by :meth:`_emit`.
270
+ sink_released: Whether ``sink``'s close has already been performed elsewhere.
254
271
 
255
272
  Returns:
256
273
  None.
@@ -273,7 +290,7 @@ class Worker:
273
290
  self._drain_finished = threading.Event()
274
291
  self._drain_settled = threading.Event()
275
292
  self._shutdown_done = False
276
- self._sink_closed = False
293
+ self._sink_closed = sink_released
277
294
  self._lock = threading.Lock()
278
295
  self._offer_stop_signal()
279
296
  self._thread = threading.Thread(
@@ -742,6 +759,7 @@ class Worker:
742
759
  if old is new_sink:
743
760
  return True
744
761
  self.sink = new_sink
762
+ self._sink_closed = False
745
763
  self._offer_stop_signal()
746
764
  remaining = None if deadline is None else max(0.0, deadline - time.monotonic())
747
765
  if not (drained and self.flush(remaining)):
@@ -788,6 +806,10 @@ class Worker:
788
806
  ``emit`` — which is why ``sinks/base.py`` requires ``close()`` to tolerate exactly that
789
807
  (SPEC-028 FR-001), and why the sinks holding transport state take their lock in both.
790
808
 
809
+ **It records nothing in the closed-sink latch** (SPEC-044 FR-004): its caller is reached
810
+ through ``_lifecycle._swap_sink``, which latches this same sink before the swap begins,
811
+ and latching again here would only overwrite a record with itself.
812
+
791
813
  ``Sink.close`` takes no timeout, so the close is run on its own thread and joined for
792
814
  what is left of the swap's budget. **An expired join decides only who waits** — it moves
793
815
  no counter and writes no line, which is what dissolves SPEC-028's objection that an
@@ -997,6 +1019,11 @@ class Worker:
997
1019
  ``is_alive()`` is the safety condition rather than a heuristic: it reads ``False`` only
998
1020
  after ``_run`` has returned, so the sink is provably out of use *by the worker*.
999
1021
 
1022
+ **It records nothing in the closed-sink latch, and does not need to** (SPEC-044 FR-004):
1023
+ ``_lifecycle._orphan_sink`` still names this sink where anything named it, and
1024
+ ``worker_owns`` answers ``True``, so ``_close_orphan_sink`` declines rather than re-arming
1025
+ it. The latch exists for a sink this worker has *stopped* holding.
1026
+
1000
1027
  The close runs to completion, inline, and is deliberately **not** bounded — which leaves
1001
1028
  one honest gap. SPEC-028 made ``close()`` take the sink's emit lock, so an application
1002
1029
  thread on the orphan path can hold that lock inside a driver call with no timeout of its