log-foundry 0.10.2.dev68__tar.gz → 0.10.2.dev70__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/PKG-INFO +19 -1
  2. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/README.md +18 -0
  3. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/pyproject.toml +1 -1
  4. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/__init__.py +2 -1
  5. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/context.py +21 -0
  6. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/decorator.py +212 -11
  7. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/model.py +13 -0
  8. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/results.py +7 -1
  9. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/_socket.py +4 -0
  10. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/base.py +48 -1
  11. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/callback.py +5 -0
  12. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/clickhouse.py +4 -0
  13. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/datadog.py +4 -0
  14. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/elasticsearch.py +8 -0
  15. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/eventhubs.py +4 -0
  16. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/file.py +8 -0
  17. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/filtering.py +27 -1
  18. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/firehose.py +4 -0
  19. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/honeycomb.py +4 -0
  20. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/http.py +4 -0
  21. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/kafka.py +35 -0
  22. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/kinesis.py +4 -0
  23. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/logging_sink.py +51 -0
  24. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/logstash.py +5 -0
  25. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/loki.py +4 -0
  26. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/memory.py +4 -0
  27. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/mongodb.py +4 -0
  28. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/multi.py +46 -1
  29. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/nats.py +47 -0
  30. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/newrelic.py +4 -0
  31. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/null.py +4 -0
  32. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/postgres.py +5 -0
  33. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/pubsub.py +73 -0
  34. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/rabbitmq.py +6 -0
  35. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/redis.py +14 -2
  36. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/sentry.py +39 -0
  37. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/sns.py +4 -0
  38. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/splunk.py +4 -0
  39. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/sqlite.py +5 -0
  40. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/sqs.py +4 -0
  41. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/stdout.py +8 -0
  42. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/syslog.py +5 -0
  43. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/transform.py +27 -1
  44. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/LICENSE +0 -0
  45. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/_diag.py +0 -0
  46. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/_fork.py +0 -0
  47. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/_lifecycle.py +0 -0
  48. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/api.py +0 -0
  49. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/config.py +0 -0
  50. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/console.py +0 -0
  51. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/ids.py +0 -0
  52. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/py.typed +0 -0
  53. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sanitize.py +0 -0
  54. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/__init__.py +0 -0
  55. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/_batch.py +0 -0
  56. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/_chunk.py +0 -0
  57. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/_retry.py +0 -0
  58. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/sinks/_time.py +0 -0
  59. {log_foundry-0.10.2.dev68 → log_foundry-0.10.2.dev70}/src/log_foundry/worker.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: log-foundry
3
- Version: 0.10.2.dev68
3
+ Version: 0.10.2.dev70
4
4
  Summary: Generate logs for your console and JSON events for downstream consumption.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -233,6 +233,24 @@ is FIFO, everything submitted before the call is necessarily ahead of that marke
233
233
  exactly why the guarantee is "events submitted before this call", and why concurrent
234
234
  submissions from other threads may or may not be included.
235
235
 
236
+ **`flush()` also empties the sink's own client buffer**, when the sink has one. A sink that
237
+ buffers in a driver rather than writing through — `KafkaSink` hands to librdkafka, `PubSubSink`
238
+ appends an unresolved future, `SentrySink` hands to the SDK's background transport — used to be
239
+ unreachable this way, so `flush()` could return truthy with events sitting in a client. If that
240
+ client cannot be emptied the result is falsy with `reason="sink-flush"`, which is distinct from
241
+ `"abandoned"`: the events are past this library and inside a driver. A **custom sink** that
242
+ buffers should implement `flush()` — it is optional, probed by name, and a sink without one is
243
+ unaffected.
244
+
245
+ **`flush()` also sweeps the spans that are still open**, so an in-span event does not have to wait
246
+ for its span to close to be delivered. The span stays open and usable afterwards: its events go
247
+ now and its `span.end` arrives later, in its own batch. Two consequences worth knowing. Boundary
248
+ events swept this way carry the baggage as of the **flush** rather than as of the close, since
249
+ that completion has to happen before they leave. And the sweep reaches only the **calling
250
+ context's** spans — `contextvars` offers no way to enumerate another thread's or task's context,
251
+ so a `flush()` in a handler that fanned out to tasks does not reach what those tasks have
252
+ buffered; their events arrive when their own spans close.
253
+
236
254
  ## Usage
237
255
 
238
256
  ### `configure(...)`
@@ -197,6 +197,24 @@ is FIFO, everything submitted before the call is necessarily ahead of that marke
197
197
  exactly why the guarantee is "events submitted before this call", and why concurrent
198
198
  submissions from other threads may or may not be included.
199
199
 
200
+ **`flush()` also empties the sink's own client buffer**, when the sink has one. A sink that
201
+ buffers in a driver rather than writing through — `KafkaSink` hands to librdkafka, `PubSubSink`
202
+ appends an unresolved future, `SentrySink` hands to the SDK's background transport — used to be
203
+ unreachable this way, so `flush()` could return truthy with events sitting in a client. If that
204
+ client cannot be emptied the result is falsy with `reason="sink-flush"`, which is distinct from
205
+ `"abandoned"`: the events are past this library and inside a driver. A **custom sink** that
206
+ buffers should implement `flush()` — it is optional, probed by name, and a sink without one is
207
+ unaffected.
208
+
209
+ **`flush()` also sweeps the spans that are still open**, so an in-span event does not have to wait
210
+ for its span to close to be delivered. The span stays open and usable afterwards: its events go
211
+ now and its `span.end` arrives later, in its own batch. Two consequences worth knowing. Boundary
212
+ events swept this way carry the baggage as of the **flush** rather than as of the close, since
213
+ that completion has to happen before they leave. And the sweep reaches only the **calling
214
+ context's** spans — `contextvars` offers no way to enumerate another thread's or task's context,
215
+ so a `flush()` in a handler that fanned out to tasks does not reach what those tasks have
216
+ buffered; their events arrive when their own spans close.
217
+
200
218
  ## Usage
201
219
 
202
220
  ### `configure(...)`
@@ -20,7 +20,7 @@ dependencies = [
20
20
  ]
21
21
 
22
22
  # Optional features. Install with: pip install log-foundry[aws]
23
- version = "0.10.2.dev68"
23
+ version = "0.10.2.dev70"
24
24
 
25
25
  [project.optional-dependencies]
26
26
  aws = ["boto3>=1.43.61"] # SQSSink, SNSSink, KinesisSink, FirehoseSink
@@ -44,7 +44,8 @@ def flush(timeout: float | None = 5.0) -> FlushResult:
44
44
  A :class:`FlushResult`. Truthy when the events submitted before this call reached the
45
45
  sink — so a truthy result means they were delivered, not merely that a drain took place.
46
46
  Falsy carries a ``reason`` naming which outcome occurred: ``"timed-out"``, ``"retired"``,
47
- ``"thread-died"``, ``"queue-full"`` or ``"abandoned"``. It is falsy rather than ``False``:
47
+ ``"thread-died"``, ``"queue-full"``, ``"abandoned"`` or ``"sink-flush"`` — the last meaning
48
+ the queue drained but the sink could not empty its own client buffer (SPEC-036 FR-002). It is falsy rather than ``False``:
48
49
  ``if flush():`` is unchanged, but ``flush() is True`` can no longer hold, which is why the
49
50
  type had to change before ``1.0.0`` rather than after (SPEC-034 FR-007). Events submitted
50
51
  concurrently by another thread may or may not be included, since the caller cannot have
@@ -55,6 +55,27 @@ def current_span() -> Span | None:
55
55
  return stack[-1] if stack else None
56
56
 
57
57
 
58
+ def _live_span_stack() -> tuple[Span, ...]:
59
+ """Returns every span open on this context, outermost first (SPEC-036 FR-001).
60
+
61
+ The library's non-copying read, named as :func:`_live_baggage` is and distinguished from the
62
+ public accessors for the same reason (SPEC-034 FR-003): a caller gets a copy, the library
63
+ reads the live object. There is nothing to copy here in any case — the stack is a tuple, so
64
+ handing it out cannot let anyone mutate it. :func:`current_span` returns only the innermost,
65
+ which is what an event needs and not what a sweep does.
66
+
67
+ Args:
68
+ None.
69
+
70
+ Returns:
71
+ Every span open on this context, outermost first. Empty when none is.
72
+
73
+ Raises:
74
+ None.
75
+ """
76
+ return _span_stack.get()
77
+
78
+
58
79
  def push_span(span: Span) -> contextvars.Token[tuple[Span, ...]]:
59
80
  """Pushes a span onto the stack.
60
81
 
@@ -22,6 +22,7 @@ from log_foundry.ids import (
22
22
  )
23
23
  from log_foundry.model import Span, backfill_baggage, end_event, start_event
24
24
  from log_foundry.results import ContinueResult, FlushResult
25
+ from log_foundry.sinks.base import flush_sink
25
26
  from log_foundry.worker import DEFAULT_SHUTDOWN_TIMEOUT, DEFAULT_SWAP_TIMEOUT, Health, Worker
26
27
 
27
28
  if TYPE_CHECKING:
@@ -60,6 +61,13 @@ shutdown's event has every backoff collapsed to zero.
60
61
  """
61
62
  _orphan_retired = False
62
63
 
64
+ _sweep_lock = threading.Lock()
65
+ """Serializes the span sweep, so two threads cannot deliver one span's buffer twice.
66
+
67
+ The detach is a load and a store, and ``contextvars`` copies the same ``Span`` object into every
68
+ task and into any ``copy_context()`` thread, so the two-reader case is ordinary rather than exotic
69
+ (SPEC-036 FR-001 AC-10). A flush is not a hot path; a sink lock it is not competing with.
70
+ """
63
71
  _loss_lock = threading.Lock()
64
72
  """Guards the two loss counters, and deliberately not ``_worker_lock`` (SPEC-036 FR-003 AC-5).
65
73
 
@@ -197,6 +205,14 @@ def continue_trace(
197
205
  _diag.rejected("parent_span_id given with no trace_id to join", parent_span_id)
198
206
  announced = True
199
207
 
208
+ if adopted is not None and _current_span_was_swept():
209
+ _diag.rejected(
210
+ "the current span has already been flushed; trace context refused",
211
+ traceparent if traceparent is not None else str(trace_id),
212
+ )
213
+ adopted = None
214
+ announced = True
215
+
200
216
  if adopted is not None:
201
217
  context.set_adopted_context(*adopted)
202
218
  _reparent_current_span(*adopted)
@@ -214,6 +230,32 @@ def continue_trace(
214
230
  return ContinueResult(ok=False, reason="rejected" if announced else "nothing-supplied")
215
231
 
216
232
 
233
+ def _current_span_was_swept() -> bool:
234
+ """Reports whether an in-span ``flush()`` has already shipped this span's events.
235
+
236
+ Read from ``context.current_span()`` — what :func:`_reparent_current_span` itself reads —
237
+ **and only when that span is a root**, which is the other half of that function's own guard.
238
+ A swept *child* is not a reason to refuse: the re-parent would have returned early on it and
239
+ rewritten nothing, so a refusal there prevents no corruption. It would still be wrong to
240
+ refuse — the two guards must agree, or the refusal fires where the thing it guards does not
241
+ run — though the adoption it spares reaches less than it appears to: SPEC-024 clears the
242
+ adopted context at the **root** span's close, so one made inside a child does not survive to
243
+ the next root span either. ``continue_trace``'s documented placement on the entry
244
+ point's first line is untouched either way: nothing has been swept that early.
245
+
246
+ Args:
247
+ None.
248
+
249
+ Returns:
250
+ Whether the current span is a root that has been swept.
251
+
252
+ Raises:
253
+ None.
254
+ """
255
+ span = context.current_span()
256
+ return span is not None and span.parent_span_id is None and span.swept
257
+
258
+
217
259
  def _reparent_current_span(trace_id: str, parent_span_id: str | None) -> None:
218
260
  """Moves an already-open root span into the adopted trace, events included.
219
261
 
@@ -712,33 +754,183 @@ def _adopt_declined_swap(new_sink: Sink) -> None:
712
754
  _orphan_sink = new_sink
713
755
 
714
756
 
757
+ def _sweep_open_spans() -> None:
758
+ """Hands the worker every event buffered on an open span in this context (SPEC-036 FR-001).
759
+
760
+ An in-span event lives on ``span.events`` until the span *closes*, and ``Worker.flush``
761
+ drains the *queue* — so a ``flush()`` called inside a ``@trace``d function, which is where
762
+ the README's serverless recipe put it, had by construction nothing to drain. Measured: zero
763
+ of two events delivered, every counter clean, and ``FlushResult`` reporting ``reason=None``.
764
+
765
+ The span stays **open**: its events go now and its ``span.end`` arrives later in its own
766
+ batch. Closing and reopening was rejected — it would emit a ``span.end`` the function never
767
+ reached, with a fabricated ``duration_ms`` and ``status``.
768
+
769
+ Two things must happen before the events leave, and both are why this is not a one-liner.
770
+ The boundary events are backfilled **first**, because SPEC-015 completes them at close by
771
+ iterating ``span.events`` and a swept buffer would ship ``span.start`` with ``fields={}`` —
772
+ the very defect that spec exists to fix, recreated by any in-span flush. They therefore carry
773
+ the baggage as of the flush rather than as of the close, which is a real semantic change and
774
+ the alternative is mutating an event the worker already owns (SPEC-028). And the buffer is
775
+ **detached by swap**, never cleared: ``clear()`` empties the same list object the worker was
776
+ handed.
777
+
778
+ The worker is created when there is something to submit, and **resolved before the buffer is
779
+ detached**. That ordering is the whole of the difference between a lost batch and a delivered
780
+ one: ``_get_worker`` can raise — it ends in ``Thread.start()`` — and a detach that has already
781
+ happened leaves the events in a discarded local while the span reads empty and ``flush()``
782
+ reports success. Measured with the failure injected: 3 of 4 events destroyed, every counter
783
+ zero, on a span that was still open and would have delivered them at its close.
784
+ ``Worker.submit`` raises nothing, so once it is reached the batch is safe. Creating the worker
785
+ at all narrows SPEC-013's refusal rather than contradicting it — that exists so an *empty*
786
+ flush does not stand up a thread, and a sweep that found buffered events is not an empty
787
+ flush. A cold-start Lambda flushing before it returns is exactly this case: the worker is
788
+ built when the first span *closes*, so inside the first traced call there is none.
789
+
790
+ Concurrent sweeps are serialized on ``_sweep_lock``. The detach is a load and a store with a
791
+ real gap between them, and two threads sharing one ``Span`` — which ``contextvars`` makes
792
+ ordinary — can both read the same buffer and deliver it twice: measured, all 8 events
793
+ duplicated with the window held open, and 9 of 25 runs with only a GIL yield between them.
794
+ Rarely preempted on today's build is not a guarantee, and the floor is ``>=3.12`` where a
795
+ free-threading build removes even that. A flush is not a hot path, so a single lock is the
796
+ right cost.
797
+
798
+ **The detach stays one statement, and the two orderings above are not in tension.** A draft
799
+ hoisted the load to the top of the loop so a test could park on it — which put
800
+ ``_get_worker()``, and therefore ``Thread.start()``, *inside* the load-to-store gap: measured,
801
+ a sweep racing a close then delivered the whole batch twice, two ``span.end`` events among
802
+ them, in 67 of 100 unforced trials against 0 before.
803
+
804
+ One statement makes that gap **narrow, not closed**, and the difference matters. It compiles
805
+ to ``LOAD_ATTR … STORE_ATTR`` with no ``CALL`` between, so CPython's eval breaker never runs
806
+ there and today's GIL cannot switch inside it — 0 of 500 unforced trials. Forced with an
807
+ opcode-level preemption it reproduces 10 of 10, and a free-threaded build removes the
808
+ accident entirely while ``requires-python`` has no upper bound. So :func:`_flush` takes this
809
+ same lock rather than relying on the width of a window: that is the *detach-vs-detach* race,
810
+ and a process-global lock is the right instrument for it. The **append** race
811
+ (``api._log`` versus a detach) is a different window needing a per-span lock, and
812
+ ``architecture.md`` §13 declines it on cost.
813
+
814
+ It reaches only the calling context's spans. ``contextvars`` offers no way to enumerate
815
+ another thread's or task's context, so a ``flush()`` in a handler that fanned out does not
816
+ reach what those tasks buffered.
817
+
818
+ Args:
819
+ None.
820
+
821
+ Returns:
822
+ None.
823
+
824
+ Raises:
825
+ Exception: Whatever building the worker raises. :func:`_flush_worker` guards it, because a
826
+ flush is the call most likely to be made in a ``finally``.
827
+ """
828
+ with _sweep_lock:
829
+ for span in context._live_span_stack():
830
+ if not span.events:
831
+ span.swept = True
832
+ continue
833
+ worker = _get_worker()
834
+ backfill_baggage(span, context._live_baggage())
835
+ span.swept = True
836
+ buffered, span.events = span.events, []
837
+ worker.submit(buffered)
838
+
839
+
840
+ def _flush_live_sink() -> bool:
841
+ """Drains whatever the delivering sink holds in its own client (SPEC-036 FR-002).
842
+
843
+ Called **after** the queue drain, because the queue's events have to reach the client buffer
844
+ before it is emptied. A sink with no ``flush`` of its own is unaffected, which is what keeps
845
+ every pre-SPEC-036 sink satisfying the protocol.
846
+
847
+ Which sink is asked follows the ownership rule the rest of this module uses (SPEC-033): a
848
+ live worker's sink if there is one, otherwise the sink an orphan emit actually **reached**.
849
+ Not "a sink has been resolved" — ``configure()`` runs ``_ensure_sink()`` unconditionally, so a
850
+ bare ``configure(service=...)`` has already built a ``StdoutSink`` that nothing was ever
851
+ written to, and materialising a flush against it is the cost SPEC-031 FR-006 declined for the
852
+ close path for the same reason. So a ``flush()`` in a process that has never logged touches
853
+ no sink, which is what FR-001 AC-6 needs to stay true.
854
+
855
+ Args:
856
+ None.
857
+
858
+ Returns:
859
+ Whether the sink's own flush succeeded. ``True`` also when there was no sink to ask, or
860
+ when it holds nothing of its own.
861
+
862
+ Raises:
863
+ None. A failure is reported as a ``FlushResult`` reason by the caller, never raised: a
864
+ flush is the call most likely to be made in a ``finally``.
865
+ """
866
+ worker = _worker
867
+ sink = worker.sink if worker is not None and not worker.retired else _orphan_sink
868
+ if sink is None:
869
+ return True
870
+ try:
871
+ flush_sink(sink)
872
+ except Exception as exc:
873
+ _diag.absorbed("flushing the sink's own buffer", exc, "its client still holds events")
874
+ return False
875
+ return True
876
+
877
+
715
878
  def _flush_worker(timeout: float | None = 5.0) -> FlushResult:
716
879
  """Drains the process worker without retiring it, backing ``flush()`` (SPEC-013 FR-003).
717
880
 
718
- This deliberately does not call :func:`_get_worker`: a process that never logged has
719
- nothing to drain, and building a worker — with the thread and ``atexit`` registration
720
- that brings — in order to flush nothing would be pure cost.
881
+ ~~This deliberately does not call :func:`_get_worker`~~ — narrowed by SPEC-036 FR-001. The
882
+ refusal still holds for an *empty* flush: a process that never logged has nothing to drain,
883
+ and building a worker — with the thread and ``atexit`` registration that brings — in order to
884
+ flush nothing would be pure cost. What changed is that :func:`_sweep_open_spans`, which runs
885
+ first, does build one when it finds buffered events on an open span, because submitting them
886
+ into a worker that does not exist delivers nothing and still reports success.
721
887
 
722
888
  Args:
723
889
  timeout: Seconds to wait for the drain, or ``None`` to wait indefinitely.
724
890
 
725
891
  Returns:
726
892
  A :class:`FlushResult`, truthy when everything outstanding was delivered and when no
727
- worker exists — a process that never logged has nothing to drain, so it has lost
728
- nothing.
893
+ worker exists — a process that never logged has nothing to drain, so it has lost nothing.
894
+ A sweep that could not hand its buffers over reports ``"abandoned"``, the existing token
895
+ for "this call did not deliver them" (SPEC-036 FR-001): the events are still on their open
896
+ spans and their close may yet carry them, but the caller asked *now*, and on the
897
+ cold-start path this exists for there may be no close — reporting success there is the
898
+ exact shape the spec was written to remove. The drain still runs, so whatever was
899
+ submitted before the failure is not held back by it.
900
+
901
+ The sink's own buffer is drained **whichever way the earlier steps went**, and the failure
902
+ reasons are decided afterwards. A draft returned early on a failed sweep or a dead drain
903
+ thread, which skipped it — and by then ``worker.flush`` had already pushed the queue *into*
904
+ that buffer, so the events most worth saving before a freeze were the ones left there. The
905
+ reason reported is the most upstream failure, because that is the one to fix.
729
906
 
730
907
  Raises:
731
908
  None. A flush is the call most likely to be made in a ``finally``, so the library must
732
909
  never be the reason a caller's function fails; a failure is reported by the return
733
910
  value instead (FR-003).
734
911
  """
735
- worker = _worker
736
- if worker is None:
737
- return FlushResult(ok=True)
912
+ swept = True
738
913
  try:
739
- return worker.flush(timeout)
740
- except Exception:
914
+ _sweep_open_spans()
915
+ except Exception as exc:
916
+ _diag.absorbed("sweeping open spans for a flush", exc, "buffered events were not swept")
917
+ swept = False
918
+ worker = _worker
919
+ drained: FlushResult = FlushResult(ok=True)
920
+ thread_died = False
921
+ if worker is not None:
922
+ try:
923
+ drained = worker.flush(timeout)
924
+ except Exception:
925
+ thread_died = True
926
+ sink_drained = _flush_live_sink()
927
+ if not swept:
928
+ return FlushResult(ok=False, reason="abandoned")
929
+ if thread_died:
741
930
  return FlushResult(ok=False, reason="thread-died")
931
+ if not sink_drained:
932
+ return FlushResult(ok=False, reason="sink-flush")
933
+ return drained
742
934
 
743
935
 
744
936
  def _note_orphan_loss() -> None:
@@ -919,6 +1111,14 @@ def _flush(span: Span) -> None:
919
1111
  The late append is now landing in a buffer nothing will emit, which is *also* loss — that
920
1112
  half is ``api._log``'s, keyed on :attr:`Span.closed`.
921
1113
 
1114
+ It takes ``_sweep_lock`` for the detach, because :func:`_sweep_open_spans` performs the same
1115
+ detach on the same attribute and a span can be swept and closed concurrently — measured, the
1116
+ whole batch delivered twice with two ``span.end`` events among them. The hold covers one
1117
+ statement and not the submit; ``Worker.submit`` is a ``put_nowait`` that never blocks, and the
1118
+ cost of the lock on the traced path measured within noise (+0.4% single-threaded, +0.9% across
1119
+ eight threads, 20,000 spans each). The **append** window this does not close is a different
1120
+ one, needs a per-span lock, and is declined in ``architecture.md`` §13.
1121
+
922
1122
  Args:
923
1123
  span: The finished span whose buffered events are submitted.
924
1124
 
@@ -928,7 +1128,8 @@ def _flush(span: Span) -> None:
928
1128
  Raises:
929
1129
  Exception: Whatever creating the worker or submitting raises; :func:`_end` is the guard.
930
1130
  """
931
- events, span.events = span.events, []
1131
+ with _sweep_lock:
1132
+ events, span.events = span.events, []
932
1133
  _get_worker().submit(events)
933
1134
 
934
1135
 
@@ -36,6 +36,18 @@ class Span:
36
36
  span, so a fire-and-forget ``create_task`` can outlive its parent and append to a buffer
37
37
  nothing will emit again. It is read by ``api._log`` at append time, which is the only place
38
38
  that can notice: nothing in the library looks at a span after ``_close_span`` returns.
39
+
40
+ ``swept`` marks a span whose buffered events an in-span ``flush()`` has already handed to the
41
+ worker (SPEC-036 FR-001). Nothing in the delivery path needs it — the sweep is correct without
42
+ it — but ``continue_trace`` does: ``_reparent_current_span`` adopts a context by rewriting the
43
+ events still *buffered* on the open root span, and swept events have left that buffer, so an
44
+ adoption after a sweep would leave one span carrying two trace ids. That is the SPEC-024
45
+ category, wrong data rather than lost data, so the flag lets the adoption refuse instead.
46
+
47
+ The guarantee is **single-threaded**. The flag is not a synchronization primitive:
48
+ ``continue_trace`` reads it and then re-parents across an ordinary function call, so a sweep
49
+ arriving in that gap still splits the span. It is set before the detach so a concurrent reader
50
+ errs toward refusing, and the residual is recorded in ``architecture.md`` §13.
39
51
  """
40
52
 
41
53
  trace_id: str
@@ -46,6 +58,7 @@ class Span:
46
58
  defaults: dict[str, object] = field(default_factory=dict)
47
59
  events: list[dict[str, object]] = field(default_factory=list)
48
60
  closed: bool = False
61
+ swept: bool = False
49
62
 
50
63
 
51
64
  def _iso_now() -> str:
@@ -50,7 +50,13 @@ class FlushResult(_Result):
50
50
  """What :func:`log_foundry.flush` returns.
51
51
 
52
52
  ``reason`` is ``None`` on success. The tokens it can carry today are ``"timed-out"``,
53
- ``"retired"``, ``"thread-died"``, ``"queue-full"`` and ``"abandoned"``.
53
+ ``"retired"``, ``"thread-died"``, ``"queue-full"``, ``"abandoned"`` and ``"sink-flush"``.
54
+
55
+ ``"sink-flush"`` is the one SPEC-036 added (FR-002 AC-8) and it is worth distinguishing: the
56
+ queue drained cleanly and the **sink's own client buffer** did not, so the events are past
57
+ this library and inside a driver. ``"abandoned"`` is the neighbouring case where this call
58
+ could not hand them over at all. New tokens may appear in any release, which is what this
59
+ type exists for — branch on ``bool()``.
54
60
  """
55
61
 
56
62
 
@@ -118,6 +118,10 @@ class SocketTransport:
118
118
  failed: Messages abandoned past the reconnect-retry bound.
119
119
  dropped_oversized: UDP datagrams discarded before any send for exceeding
120
120
  ``max_datagram_bytes``.
121
+
122
+
123
+ It keeps **no** client buffer (SPEC-036 FR-002): ``send_all`` puts the bytes on the
124
+ socket before it returns, and no client object outlives it holding data.
121
125
  """
122
126
 
123
127
  def __init__(
@@ -6,7 +6,7 @@ from abc import abstractmethod
6
6
  from dataclasses import dataclass
7
7
  from typing import Protocol, runtime_checkable
8
8
 
9
- __all__ = ["Sink", "SinkDeliveryError", "SinkLosses", "read_losses"]
9
+ __all__ = ["Sink", "SinkDeliveryError", "SinkLosses", "flush_sink", "read_losses"]
10
10
 
11
11
 
12
12
  class SinkDeliveryError(Exception):
@@ -142,6 +142,24 @@ class Sink(Protocol):
142
142
  the call an operator makes when a destination is already hanging, and sharing one lock
143
143
  would make that poll wait for an in-flight emit and its retry backoff. Where a sink holds
144
144
  both, the order is always transport then counter, never the reverse.
145
+ A sixth is optional in the same way: ``flush() -> None``, which drains whatever the sink is
146
+ holding in its **client** without closing it (SPEC-036 FR-002). A sink that buffers in a
147
+ driver rather than writing through — ``KafkaSink`` hands to librdkafka, ``GooglePubSubSink``
148
+ appends an unresolved future — is unreachable through ``log_foundry.flush()`` without it:
149
+ measured against a stand-in with that shape, ``flush() -> True``, on the wire 0, in the
150
+ client buffer 3, ``health()`` all zeros. It is called **after** the queue drain, so the
151
+ queue's events have reached the client buffer before it is emptied.
152
+
153
+ It is **not a close**, and the difference is the whole point: the sink keeps its transport and
154
+ goes on accepting events afterwards. Like :meth:`emit` it must tolerate being called
155
+ concurrently with an emit (SPEC-028), and like :meth:`emit` it must **raise** when it could
156
+ not deliver what it was holding — that is the only channel by which ``log_foundry.flush()``
157
+ can report ``reason="sink-flush"`` instead of success. :func:`flush_sink` is the probe, and it
158
+ deliberately does **not** behave like :func:`read_losses`: that one swallows a raising
159
+ accessor because a broken reporter must not take ``health()`` down, while this one propagates,
160
+ because a swallowed flush failure is exactly the "sink the worker believes" this file exists
161
+ to prevent.
162
+
145
163
  """
146
164
 
147
165
  @abstractmethod
@@ -260,3 +278,32 @@ def read_losses(sink: object) -> SinkLosses | None:
260
278
  except Exception:
261
279
  return None
262
280
  return losses if isinstance(losses, SinkLosses) else None
281
+
282
+
283
+ def flush_sink(sink: object) -> bool:
284
+ """Calls a sink's optional ``flush()``, letting any failure propagate (SPEC-036 FR-002).
285
+
286
+ The sibling of :func:`read_losses`, written here for the same reason — the probe and its
287
+ guarantees belong in one place — and with the **opposite** failure rule, which is the part
288
+ worth reading twice. ``read_losses`` swallows a raising accessor because a broken reporter
289
+ must not take ``health()`` down with it. This one must not swallow anything: a sink's flush
290
+ failure reaches the caller only through ``log_foundry.flush()``'s result, so absorbing it here
291
+ would produce the exact "sink the worker believes" this module exists to prevent — a
292
+ ``flush()`` reporting success over a client buffer that never went out.
293
+
294
+ Args:
295
+ sink: The sink to probe, of any type.
296
+
297
+ Returns:
298
+ Whether the sink had a ``flush`` to call. ``False`` means it holds nothing of its own, and
299
+ the queue drain was the whole of the flush.
300
+
301
+ Raises:
302
+ Exception: Whatever the sink's ``flush`` raises, deliberately unguarded. The caller turns
303
+ it into a ``FlushResult`` reason; see ``decorator._flush_live_sink``.
304
+ """
305
+ accessor = getattr(sink, "flush", None)
306
+ if not callable(accessor):
307
+ return False
308
+ accessor()
309
+ return True
@@ -22,6 +22,11 @@ class CallbackSink:
22
22
  (SPEC-032 FR-003). Both decisions belong to the callable: this class holds nothing, and what
23
23
  a hook releases is not knowable from here — a callable needing either guarantee must provide
24
24
  it, exactly as a hand-written ``Sink`` implementation would.
25
+
26
+
27
+ It keeps **no** client buffer (SPEC-036 FR-002): it hands each event to a *function*, which
28
+ has returned by the time ``emit`` does. Unlike the three wrapper sinks it wraps no sink, so
29
+ there is nothing to forward a flush to.
25
30
  """
26
31
 
27
32
  def __init__(
@@ -47,6 +47,10 @@ class ClickHouseSink:
47
47
  default auto-generated session, so it is squarely in that case and the lock is required
48
48
  rather than merely prudent. One client per thread would be the alternative, and that is the
49
49
  connection-pool design FR-002 puts out of scope.
50
+
51
+
52
+ It keeps **no** client buffer (SPEC-036 FR-002): the driver call returns only once the
53
+ destination has the batch, so nothing is queued locally between emits.
50
54
  """
51
55
 
52
56
  def __init__(
@@ -28,6 +28,10 @@ class DatadogSink(HTTPSink):
28
28
  sink in the family whose per-event limit is stricter than its request limit, so without
29
29
  it a 2 MB event passes the 5 MB request budget and is rejected by a limit the budget
30
30
  cannot see. All three are the vendor's own figures, from the Logs API's send-logs limits.
31
+
32
+
33
+ It keeps **no** client buffer (SPEC-036 FR-002): each ``emit`` is a request that has
34
+ completed by the time it returns, and no client object outlives it holding data.
31
35
  """
32
36
 
33
37
  MAX_BATCH_COUNT = 1000
@@ -38,6 +38,10 @@ class ElasticsearchSink(HTTPSink):
38
38
  bulk guidance is to find a working size by experiment rather than to send the largest
39
39
  request the server will accept, and a 100 MB bulk is a poor default for a log shipper.
40
40
  Raise it with ``max_batch_bytes=``.
41
+
42
+
43
+ It keeps **no** client buffer (SPEC-036 FR-002): each ``emit`` is a request that has
44
+ completed by the time it returns, and no client object outlives it holding data.
41
45
  """
42
46
 
43
47
  MAX_BATCH_COUNT = 1000
@@ -201,4 +205,8 @@ class OpenSearchSink(ElasticsearchSink):
201
205
  """OpenSearch reuses the Elasticsearch ``_bulk`` protocol verbatim (FR-003).
202
206
 
203
207
  Endpoint and auth differ only by configuration, so this is a straight reuse.
208
+
209
+
210
+ It keeps **no** client buffer (SPEC-036 FR-002): each ``emit`` is a request that has
211
+ completed by the time it returns, and no client object outlives it holding data.
204
212
  """
@@ -23,6 +23,10 @@ class AzureEventHubsSink:
23
23
  1 MB per-batch limit, which the SDK signals by raising ``ValueError`` from ``add``. The
24
24
  worst-case delay (SPEC-027 FR-005) is ``max_retries`` interruptible waits per batch, 0.7 s at
25
25
  the defaults.
26
+
27
+
28
+ It keeps **no** client buffer (SPEC-036 FR-002): the driver call returns only once the
29
+ destination has the batch, so nothing is queued locally between emits.
26
30
  """
27
31
 
28
32
  def __init__(
@@ -71,6 +71,10 @@ class FileSink:
71
71
  concurrently (SPEC-028 FR-002) — this module claimed a single worker thread until that spec
72
72
  measured the orphan path emitting on application threads at the same time. Cross-*process*
73
73
  coordination remains out of scope: two processes appending to one path are on their own.
74
+
75
+
76
+ It keeps **no** client buffer (SPEC-036 FR-002): ``emit`` flushes the stream before it
77
+ returns, so nothing of this sink's is left pending between calls.
74
78
  """
75
79
 
76
80
  def __init__(self, path: str, *, encoding: str = "utf-8") -> None:
@@ -211,6 +215,10 @@ class RotatingFileSink:
211
215
  damage: a second thread mid-``emit`` could write to the handle rotation had just closed, or
212
216
  to the pre-rotation file it had already renamed away. Both are serialized on a lock
213
217
  (SPEC-028 FR-002).
218
+
219
+
220
+ It keeps **no** client buffer (SPEC-036 FR-002): ``emit`` flushes the stream before it
221
+ returns, so nothing of this sink's is left pending between calls.
214
222
  """
215
223
 
216
224
  def __init__(