log-foundry 0.10.2.dev110__tar.gz → 0.10.2.dev111__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/PKG-INFO +107 -45
  2. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/README.md +106 -44
  3. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/pyproject.toml +1 -1
  4. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/LICENSE +0 -0
  5. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/__init__.py +0 -0
  6. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/_diag.py +0 -0
  7. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/_fork.py +0 -0
  8. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/_lifecycle.py +0 -0
  9. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/api.py +0 -0
  10. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/config.py +0 -0
  11. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/console.py +0 -0
  12. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/context.py +0 -0
  13. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/decorator.py +0 -0
  14. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/ids.py +0 -0
  15. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/model.py +0 -0
  16. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/py.typed +0 -0
  17. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/results.py +0 -0
  18. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sanitize.py +0 -0
  19. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/__init__.py +0 -0
  20. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/_batch.py +0 -0
  21. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/_chunk.py +0 -0
  22. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/_retry.py +0 -0
  23. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/_socket.py +0 -0
  24. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/_time.py +0 -0
  25. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/base.py +0 -0
  26. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/callback.py +0 -0
  27. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/clickhouse.py +0 -0
  28. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/datadog.py +0 -0
  29. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/elasticsearch.py +0 -0
  30. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/eventhubs.py +0 -0
  31. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/file.py +0 -0
  32. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/filtering.py +0 -0
  33. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/firehose.py +0 -0
  34. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/honeycomb.py +0 -0
  35. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/http.py +0 -0
  36. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/kafka.py +0 -0
  37. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/kinesis.py +0 -0
  38. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/logging_sink.py +0 -0
  39. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/logstash.py +0 -0
  40. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/loki.py +0 -0
  41. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/memory.py +0 -0
  42. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/mongodb.py +0 -0
  43. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/multi.py +0 -0
  44. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/nats.py +0 -0
  45. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/newrelic.py +0 -0
  46. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/null.py +0 -0
  47. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/postgres.py +0 -0
  48. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/pubsub.py +0 -0
  49. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/rabbitmq.py +0 -0
  50. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/redis.py +0 -0
  51. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/sentry.py +0 -0
  52. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/sns.py +0 -0
  53. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/splunk.py +0 -0
  54. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/sqlite.py +0 -0
  55. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/sqs.py +0 -0
  56. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/stdout.py +0 -0
  57. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/syslog.py +0 -0
  58. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/transform.py +0 -0
  59. {log_foundry-0.10.2.dev110 → log_foundry-0.10.2.dev111}/src/log_foundry/worker.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: log-foundry
3
- Version: 0.10.2.dev110
3
+ Version: 0.10.2.dev111
4
4
  Summary: Generate logs for your console and JSON events for downstream consumption.
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -65,8 +65,11 @@ calls form a tree you can query later.
65
65
  - **Structured, never free-form** — every event is the same named-field JSON shape.
66
66
  - **Safe by default** — never captures your arguments or return values (no accidental PII/secret leakage), and the decorator **never swallows exceptions**.
67
67
  - **Correct under threads and asyncio** — context propagates via `contextvars`.
68
- - **Non-blocking delivery** — finished spans are handed to a background worker; your code never
69
- blocks on sink I/O, and a graceful drain at exit means buffered events aren't lost.
68
+ - **Non-blocking delivery from a traced call** — a finished span's events are handed to a
69
+ background worker, so code inside `@trace` never blocks on sink I/O, and a graceful drain at
70
+ exit means buffered events aren't lost. Two things you can do *deliberately* are synchronous:
71
+ a level call with **no open span**, which emits on your own thread, and `flush()`, which by
72
+ definition waits for the drain it asked for.
70
73
 
71
74
  ---
72
75
 
@@ -88,13 +91,14 @@ pip install 'log-foundry[aws]' # + boto3 for the SQS/SNS/Kinesis/Firehose sink
88
91
  >
89
92
  > - **`health()` and `sink.losses()` return frozen dataclasses**, not `NamedTuple`s. Attribute
90
93
  > access (`h.dropped`, `losses.failed`) is unchanged and is the whole contract; `len(h)`,
91
- > `h[0]` and `queued, dropped, failed = health()` now raise `TypeError`. `Health` has gained a
92
- > field in six consecutive specs and gains more — with no positions, that stops being breaking.
94
+ > `h[0]` and the four-way `queued, dropped, failed_batches, stopped_reason = health()` now
95
+ > raise `TypeError`. `Health` gains fields, and will gain more — with no positions to
96
+ > preserve, that stops being a breaking change.
93
97
  > - **`flush()` returns a `FlushResult` and `continue_trace()` a `ContinueResult`**, each truthy
94
98
  > or falsy with a `reason` naming *why*. `if lf.flush():` is unchanged; **`lf.flush() is True`
95
99
  > is not** — the result is an object. A one-bit return could not grow a reason later without
96
100
  > silently changing what `if flush():` means, which is why it moved now.
97
- > - **`SQSSink`'s injected client is keyword-only** (`SQSSink(url, client=…)`), **`SentrySink`
101
+ > - **`SQSSink`'s injected client is keyword-only** (`SQSSink(queue_url, client=…)`), **`SentrySink`
98
102
  > injects through `client=`** rather than the old `sdk` keyword, with no alias, and the sink attribute the
99
103
  > library assigns for interruptible backoff is **`log_foundry_stop_signal`**, not
100
104
  > `stop_signal` — a prefixed name cannot silently overwrite one your own sink already uses.
@@ -177,9 +181,9 @@ with the child span pointing at its parent via `parent_span_id`:
177
181
 
178
182
  ```json
179
183
  {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.start", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "754adb40e10c445f9ec9e23a2f3dcbf2", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}}
180
- {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "3af73c51540848afbeaba9fdf7a9dce8", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}, "duration_ms": 0.018, "status": "ok"}
184
+ {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "3af73c51540848afbeaba9fdf7a9dce8", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}, "duration_ms": 0.0233328901231289, "status": "ok"}
181
185
  {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.start", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "e789f7e5268b46d8b779c9cbcdde8656", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}}
182
- {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "8f3dbfcfcf4a45f688c738eefef882b0", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}, "duration_ms": 0.326, "status": "ok"}
186
+ {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "8f3dbfcfcf4a45f688c738eefef882b0", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}, "duration_ms": 0.342666869983077, "status": "ok"}
183
187
  ```
184
188
 
185
189
  > **Note on ordering:** the child span (`tax.compute`) finishes first, so its events flush
@@ -191,7 +195,11 @@ with the child span pointing at its parent via `parent_span_id`:
191
195
 
192
196
  A traced call travels through a small pipeline. The first four steps run on your own thread
193
197
  and are deliberately fast; the last two run on a background thread so your code never waits on
194
- the destination.
198
+ the destination. That is the traced path. A level call with no **open** span takes the
199
+ synchronous route instead, described under
200
+ [Flushing and shutdown](#flushing-and-shutdown) — and "open" is the operative word: a task
201
+ that outlives its span still sees that span, but it has closed, so the call takes the
202
+ synchronous route too.
195
203
 
196
204
  1. **You call the code** — a `@trace` function, or one of the `debug`/`info`/… emitters.
197
205
  2. **A span opens** — a record of this one call. It inherits the current trace and parent (see
@@ -353,9 +361,9 @@ def process(): ...
353
361
  - `defaults` — per-decorator fields merged into every event this span emits.
354
362
 
355
363
  The **outermost** decorated call starts a new trace; every nested decorated call becomes a
356
- child span within it. On an exception, the decorator records `status="error"` plus the
357
- exception type and formatted stack, then **re-raises the original exception unchanged** — it
358
- never swallows errors.
364
+ child span within it. On an exception, the decorator records `status="error"` plus an `error`
365
+ sub-document carrying the exception's type, module, message and formatted stack, then
366
+ **re-raises the original exception unchanged** — it never swallows errors.
359
367
 
360
368
  **Async is supported.** Apply `@trace` to an `async def` and it traces the coroutine's actual
361
369
  run — the span opens when the coroutine starts and closes when the awaits complete, not when the
@@ -420,7 +428,7 @@ def process_payment(user_id: int) -> str:
420
428
  It reaches its own name too (`fields={"fields": ...}`), so every reserved word has exactly one
421
429
  route through. A key given both ways takes the keyword's value, since `**kwargs` is what you
422
430
  wrote at the call site and `fields=` is usually a mapping built elsewhere.
423
- - **Orphan logs** — a level call made with no active span is not dropped: it emits a standalone
431
+ - **Orphan logs** — a level call made with no **open** span is not dropped: it emits a standalone
424
432
  one-event span with a fresh `trace_id`, flushed straight to the sink.
425
433
 
426
434
  ### Continuing a trace across processes
@@ -462,9 +470,13 @@ def handler(event, context): # the entry point, deliberately not decorated
462
470
  lf.flush() # the span has closed, so its events are drained
463
471
  ```
464
472
 
465
- `flush()` goes **outside** the traced function, not in its `finally`. An in-span event lives on the
466
- span until the span *closes*, and `flush()` drains the queue — so a `flush()` inside the span has
467
- nothing to drain yet.
473
+ `flush()` works either side of the traced function. Putting it outside, as above, is still the
474
+ clearer shape — the span has closed, so there is nothing to reason about. But a `flush()`
475
+ *inside* an open span is no longer a mistake: it sweeps every open span in the calling context,
476
+ hands their buffered events to the worker and drains them, leaving the spans open. It does not
477
+ deliver a `span.end` that has not happened yet. Before `1.0.0` a flush inside a span swept
478
+ nothing, and this recipe with the `flush()` moved into the traced function delivered **nothing**
479
+ with every counter clean.
468
480
 
469
481
  | Call | Does |
470
482
  |---|---|
@@ -555,8 +567,16 @@ nothing.
555
567
 
556
568
  Wire one up by passing an instance to `configure(sink=...)`; if you never do, the first decorated
557
569
  call falls back to `StdoutSink()`. The **protocol** is a top-level export, alongside
558
- `SinkDeliveryError`, `SinkLosses` and `read_losses`; the **concrete sinks** are not, so import each
559
- from its own module, e.g. `from log_foundry.sinks.sqs import SQSSink`.
570
+ `SinkDeliveryError`, `SinkLosses`, `read_losses` and `flush_sink` — the last two being the probes
571
+ a wrapper sink needs to ask a child for its losses and to push its client-side buffer. The
572
+ **concrete sinks** are not exported, so import each from its own module, e.g.
573
+ `from log_foundry.sinks.sqs import SQSSink`.
574
+
575
+ **Construct `SinkLosses` with keywords**, as the example below does. The public dataclasses are
576
+ keyword-only from `1.0.0`, so a positional `SinkLosses(0, 3)` raises `TypeError` — and it raises
577
+ *inside* your `losses()`, where `read_losses` deliberately swallows it so a broken reporter cannot
578
+ take `health()` down. The symptom is not an error: your sink's loss reporting silently becomes
579
+ `None`.
560
580
 
561
581
  A few conventions hold across every sink below:
562
582
 
@@ -606,8 +626,8 @@ A few conventions hold across every sink below:
606
626
 
607
627
  | Sink | Import from | Configure |
608
628
  |---|---|---|
609
- | `StdoutSink` | `log_foundry.sinks.stdout` | `StdoutSink(stream=sys.stdout)` — one JSON line per event; the zero-config default |
610
- | `StderrSink` | `log_foundry.sinks.stdout` | `StderrSink(stream=sys.stderr)` — same, on stderr (twelve-factor) |
629
+ | `StdoutSink` | `log_foundry.sinks.stdout` | `StdoutSink(stream=None)` — one JSON line per event; `None` resolves to `sys.stdout` **at construction** and is not re-read per write. The zero-config default |
630
+ | `StderrSink` | `log_foundry.sinks.stdout` | `StderrSink(stream=None)` — same, resolving to `sys.stderr` (twelve-factor) |
611
631
  | `NullSink` | `log_foundry.sinks.null` | `NullSink()` — discard everything; `.dropped` counts events |
612
632
  | `MemorySink` | `log_foundry.sinks.memory` | `MemorySink(maxlen=None)` — collect into `.events` (a bounded ring when `maxlen` is set) |
613
633
 
@@ -963,7 +983,7 @@ class MySink:
963
983
 
964
984
  def losses(self) -> SinkLosses:
965
985
  with self._counter_lock: # both fields from one instant
966
- return SinkLosses(dropped=self._dropped, failed=self._failed)
986
+ return SinkLosses(dropped=self._dropped, failed=self._failed) # keywords required
967
987
 
968
988
  def close(self) -> None:
969
989
  with self._lock: # never release under an active writer
@@ -1023,7 +1043,12 @@ Every wrapper shipped here — `MultiSink`, `FilteringSink`, `TransformSink`, `S
1023
1043
 
1024
1044
  Delivery is off the hot path. When a span ends, its events are handed to a per-process
1025
1045
  background worker via a fast, non-blocking submit — your function returns without waiting on
1026
- the sink. The worker batches events (by count and time), emits them on its own thread, retries
1046
+ the sink. A level call with no **open** span has no span to end, so it emits synchronously on
1047
+ the calling thread — including a call from a task that outlived the span it inherited, whose
1048
+ span is present but closed. That path needs no worker, and in a process that only ever logs
1049
+ that way none is ever built; it is a span closing, or a `flush()` sweeping one, that builds it.
1050
+ The rest of this paragraph is about the buffered path.
1051
+ The worker batches events (by count and time), emits them on its own thread, retries
1027
1052
  a failing sink with backoff, and applies backpressure so a slow or down sink can never block or
1028
1053
  back-pressure the app: when its bounded queue is full it drops the newest submissions and counts
1029
1054
  them rather than stalling.
@@ -1051,7 +1076,7 @@ They tell you different things, and they want different responses:
1051
1076
 
1052
1077
  | Field | Means | What to do |
1053
1078
  |---|---|---|
1054
- | `dropped` | The queue filled — the destination is not keeping up. Delivery continues. | Tune `batch_size`/`flush_interval`, or scale the sink. |
1079
+ | `dropped` | The queue filled — the destination is not keeping up. Delivery continues. | Make the destination keep up: scale the sink, or reduce what you log. The worker's batch size, flush interval and queue depth are **not** reachable from the public API, so tuning them is not an option this version offers you. |
1055
1080
  | `failed_batches` | A sink stayed broken through the whole retry budget. Delivery continues. | Fix the destination. |
1056
1081
  | `stopped_reason` | The background thread **died** on that exception type. Nothing further will be delivered, ever. | Restart the process; investigate the named exception. |
1057
1082
  | `sink.dropped` | The sink discarded events **before** attempting delivery — an oversized record, or one the client refused outright. | Read the stderr line: it names the cause. An oversized record means shrink what you log; a refused local produce/publish (Kafka, Pub/Sub) points at the client — a saturated buffer, a bad topic, a credential. |
@@ -1059,7 +1084,7 @@ They tell you different things, and they want different responses:
1059
1084
  | `retired` + `submitted_after_shutdown` | `shutdown()` was called and the process **kept logging**. Those events are queued where nothing will drain them — total loss, for as long as the process runs. | Use `flush()`, not `shutdown()`, in a process that logs again. This is the serverless mistake below. |
1060
1085
  | `incomplete_swaps` | A late `configure(sink=...)` could not confirm the previous sink was drained. The swap took effect; that sink was left open and some queued events may have gone to the new one. | Investigate the previous sink — it was hung or failing. Configure the sink before the first log where you can. |
1061
1086
  | `inherited_sink` | This process is delivering to a sink it **inherited across a `fork`** and may not release, so it will not be closed here. Not a loss and not an alert term. | Nothing, usually. It explains a handle still open after `shutdown()`, and tells you a deployment shares one sink across a fork at all. `True` for a shared `StdoutSink` too, whose `close()` only flushes — so a `True` is not by itself evidence that anything is held. If you want the child to own its transport, build the sink in the worker process (see Forking). |
1062
- | `orphan_lost` | An event logged **with no active span** never reached the sink. That call emits on your own thread with no worker behind it, so no other field here can carry it — it is not a batch, there was no retry, and there may be no worker at all. Covers a sink that failed to *construct* as well as one that raised. | Fix the destination, or the data. The stderr line names the exception type. If a process logs this way at all, this is the field to alert on: nothing else describes that path. |
1087
+ | `orphan_lost` | An event logged **with no open span** never reached the sink. That call emits on your own thread with no worker behind it, so no other field here can carry it — it is not a batch, there was no retry, and there may be no worker at all. Covers a sink that failed to *construct* as well as one that raised. | Fix the destination, or the data. The stderr line names the exception type. If a process logs this way at all, this is the field to alert on: nothing else describes that path. |
1063
1088
  | `in_span_lost` | An event logged **inside a span** could not be built — a value that could not be turned into an event. Always the data, never the destination: the in-span path cannot fail at delivery, which is `failed_batches`. | Fix the call site. Passing a non-string message (an exception object, say) is the common cause. |
1064
1089
  | `closing_sinks` | Swapped-out sinks inside `close()` **right now** — a live gauge, not a counter. It and `queued` are the two fields here that fall as well as rise; the other integer counters only climb. Non-zero on a single read is normal during a swap. | Nothing, unless it stays non-zero. That means a destination is stuck in `close()` and will not release its resources. |
1065
1090
 
@@ -1067,8 +1092,9 @@ They tell you different things, and they want different responses:
1067
1092
  reported. They aggregate different failure populations — one can mean the destination *or* the
1068
1093
  data, the other can only mean the data — so a single number would hide which fix applies.
1069
1094
 
1070
- `h.sink` is a `SinkLosses(dropped, failed)` or `None` — `None` when no worker exists yet, or when
1071
- the configured sink reports nothing (`losses()` is optional). Note the two `dropped` fields count
1095
+ `h.sink` is a `SinkLosses`, carrying `dropped` and `failed`, or `None` — `None` when no worker
1096
+ exists yet, or when the configured sink reports nothing (`losses()` is optional, and a sink whose
1097
+ `losses()` raises reports `None` too). Note the two `dropped` fields count
1072
1098
  different things: the worker's is backpressure at *its* queue, the sink's is an event that never
1073
1099
  reached the wire. They are separate because the remedies do not overlap — and `sink.dropped` is
1074
1100
  itself two causes, which is why the diagnostic line matters. Most sinks drop only what can never
@@ -1109,11 +1135,12 @@ guards its own post-close state — rather than queued where nothing will drain
1109
1135
  the same claim. A stateless sink such as the default `StdoutSink` still accepts it.
1110
1136
 
1111
1137
  Read a snapshot **by attribute** (`h.dropped`), as above — that is the whole contract. `Health`
1112
- and `SinkLosses` are frozen dataclasses, so `len(h)`, `h[0]` and `queued, dropped, failed =
1113
- health()` all raise `TypeError`. They were `NamedTuple`s before `1.0.0` and the tuple shape is
1114
- deliberately gone: `Health` has gained a field in six consecutive specs and gains more, and every
1115
- one of those had to argue that the positions before it were undisturbed. There are no positions to
1116
- disturb now, and adding a field is not a breaking change.
1138
+ and `SinkLosses` are frozen dataclasses, so `len(h)`, `h[0]` and the four-way
1139
+ `queued, dropped, failed_batches, stopped_reason = health()` all raise `TypeError`. They were
1140
+ `NamedTuple`s before `1.0.0`, and the tuple shape is deliberately gone: `Health` gains fields
1141
+ as the library learns to report more, and while it was a tuple every one of those additions
1142
+ had to argue that the positions before it were undisturbed. There are no positions to disturb
1143
+ now, and adding a field is not a breaking change.
1117
1144
 
1118
1145
  `dropped` counts submissions discarded because the queue filled; `failed_batches` counts batches
1119
1146
  abandoned after the retry budget was spent. Overflow also warns on stderr — on the first drop and
@@ -1173,7 +1200,7 @@ was outstanding** — the drain it forces reached the sink, and so did anything
1173
1200
  emitted while it waited its turn. A truthy result is evidence of delivery, not merely that a drain
1174
1201
  took place.
1175
1202
 
1176
- Falsy carries a `reason` saying which of five things happened, because they need different fixes:
1203
+ Falsy carries a `reason` saying which of these happened, because they need different fixes:
1177
1204
 
1178
1205
  | `reason` | Means |
1179
1206
  |---|---|
@@ -1182,6 +1209,7 @@ Falsy carries a `reason` saying which of five things happened, because they need
1182
1209
  | `"thread-died"` | The drain thread is gone; see `health().stopped_reason`. |
1183
1210
  | `"queue-full"` | Backpressure — the queue could not even accept the marker. |
1184
1211
  | `"abandoned"` | A batch was given up on after its retry budget. The destination is broken. |
1212
+ | `"sink-flush"` | Everything queued reached the sink, but the sink's own `flush()` raised — a client-side buffer that did not go out. Distinct from `"abandoned"`: the library delivered, the sink did not. |
1185
1213
 
1186
1214
  ```python
1187
1215
  result = lf.flush(5.0)
@@ -1240,7 +1268,8 @@ def _handler(event, context):
1240
1268
 
1241
1269
  def handler(event, context):
1242
1270
  # NOT decorated, so the span closes when `_handler` returns and its events reach the queue
1243
- # before `flush()` runs. A `flush()` *inside* the traced function has nothing to drain yet.
1271
+ # before `flush()` runs. Since 1.0.0 a `flush()` *inside* the traced function also works —
1272
+ # it sweeps the open span — but keeping it out here means there is nothing to reason about.
1244
1273
  try:
1245
1274
  return _handler(event, context)
1246
1275
  finally:
@@ -1282,9 +1311,9 @@ Every event is the same shape (arch §6). Boundary events add a few fields:
1282
1311
  | `function` | ✓ | span name |
1283
1312
  | `service` / `version` / `env` | ✓ | from `configure(...)` |
1284
1313
  | `fields` | ✓ | merged user fields (config `defaults` → span `defaults` → …) |
1285
- | `duration_ms` | span.end | wall time from a monotonic delta |
1314
+ | `duration_ms` | span.end | wall time from a monotonic delta, unrounded — full float precision |
1286
1315
  | `status` | span.end | `"ok"` or `"error"` |
1287
- | `error` | on failure | `{"type": ..., "stack": ...}` |
1316
+ | `error` | on failure | `{"type": ..., "module": ..., "message": ..., "stack": ...}` |
1288
1317
  | `truncated` | when a ceiling fired | `true`; absent otherwise, never `false` |
1289
1318
 
1290
1319
  IDs are [W3C Trace Context](https://www.w3.org/TR/trace-context/)-compatible by design, so the
@@ -1329,18 +1358,30 @@ poetry run pytest # test (runs in parallel by default; see addopts)
1329
1358
  poetry run pytest -n 0 # ...serially, when debugging a failure
1330
1359
  poetry run ruff check . # lint (line-length 100)
1331
1360
  poetry run mypy # typecheck (strict, over src/)
1361
+ sh scripts/spec-lint.sh # lint the design specs
1362
+ sh scripts/docs-lint.sh # the always-loaded docs tier — NOTHING IN CI RUNS THIS
1363
+ poetry run python scripts/docstring-lint.py # the docstring rule over src/ — ALSO NOT IN CI
1332
1364
  ```
1333
1365
 
1366
+ All six run before a push. The last two are the ones to remember, because they are deliberately
1367
+ not CI jobs: they hold the documentation and the docstrings to rules a reviewer would otherwise
1368
+ have to carry, and keeping them local means the failure lands on whoever caused it rather than on
1369
+ a shared branch. Each has a `-test.sh` beside it that proves its own checks still fire; run that
1370
+ one too if you change the linter.
1371
+
1334
1372
  The library uses a src layout (`src/log_foundry/`) with a single concept per module: `config`,
1335
- `ids`, `model`, `context`, `decorator`, `api`, `console`, `worker`, and the `sinks/` package (the
1336
- `base` protocol, `stdout`, and one module per sink family — see [Sinks](#sinks)).
1373
+ `ids`, `model`, `context`, `decorator`, `api`, `console`, `worker`, `sanitize` and `results`,
1374
+ the internal `_lifecycle`, `_fork` and `_diag`, and the `sinks/` package (the `base` protocol,
1375
+ `stdout`, and one module per sink family — see [Sinks](#sinks)). Anything underscore-prefixed
1376
+ is internal and moves without notice.
1337
1377
  Deeper design docs live in [`docs/`](docs/) — start with [`docs/architecture.md`](docs/architecture.md).
1338
1378
 
1339
1379
  ### Continuous integration
1340
1380
 
1341
- Every pull request runs the checks below. A check whose verdict cannot change on the tree in
1342
- front of it is path-filtered rather than run to the same answer twice, so the *When* column is
1343
- part of the contract:
1381
+ The checks below guard this repository. Most run on every pull request; a check whose verdict
1382
+ cannot change on the tree in front of it is path-filtered rather than run to the same answer
1383
+ twice, and two run only after the merge — so the *When* column is part of the contract and is
1384
+ worth reading before assuming a PR was audited:
1344
1385
 
1345
1386
  | Check | Does | When | Fails the build |
1346
1387
  |---|---|---|---|
@@ -1349,6 +1390,14 @@ part of the contract:
1349
1390
  | [`dependency-review.yml`](.github/workflows/dependency-review.yml) | fails a PR that *introduces* a dependency with a known advisory (`moderate`+) | every PR | yes |
1350
1391
  | [`zizmor.yml`](.github/workflows/zizmor.yml) | static analysis of the workflow files themselves | workflow, action, dependabot or zizmor config touched; also weekly | no — reports to code scanning |
1351
1392
  | CodeQL | `python` + `actions`, `extended` query suite; also weekly | every PR | no — reports to code scanning |
1393
+ | [`integration.yml`](.github/workflows/integration.yml) | the extras-backed sinks against nine real services in containers | sinks, `tests/integration/`, `pyproject.toml`, `poetry.lock` or that workflow touched; also weekly | it goes red, and like every check here it is advisory — `main` requires no status check at all, and this one is furthest from earning that: nine service containers are a flakiness budget |
1394
+ | [`pip-audit.yml`](.github/workflows/pip-audit.yml) | advisories across every extra, `--strict` | **not on PRs at all** — on the merge to `main` when the lockfile, extras, the ignore list or that workflow moved, and weekly | yes, on `main` |
1395
+ | [`scorecard.yml`](.github/workflows/scorecard.yml) | OpenSSF Scorecard over the repository's own supply chain | **not on PRs** — on the merge to `main` when `.github/`, `scripts/`, `SECURITY.md`, `LICENSE` or the lockfile moved, and weekly | no — reports to code scanning |
1396
+
1397
+ **`scripts/docs-lint.sh` is deliberately not in that table**, because nothing in CI runs it. It
1398
+ holds the always-loaded documentation tier to its budgets and is a local pre-push gate, so its
1399
+ failure lands on whoever caused it rather than on a shared branch. Contributors run it, along
1400
+ with `scripts/spec-lint.sh`, before pushing.
1352
1401
 
1353
1402
  On a push to `main` the full `ci.yml` matrix still runs, but as the first job of
1354
1403
  [`release.yml`](.github/workflows/release.yml), which `uses:` this same workflow before it is
@@ -1391,9 +1440,12 @@ everything it instruments:
1391
1440
  [`SECURITY.md`](SECURITY.md#software-bill-of-materials) has the detail.
1392
1441
 
1393
1442
  Scanning runs continuously rather than at release time: CodeQL over the source and the workflows,
1394
- zizmor over the workflows, `dependency-review` on every pull request, a weekly `pip-audit` across
1395
- all eleven extras, and OpenSSF Scorecard. Findings go to code scanning; `dependency-review` and
1396
- `pip-audit` are the two that fail a build.
1443
+ zizmor over the workflows, `dependency-review` on every pull request, `pip-audit` across all
1444
+ eleven extras, and OpenSSF Scorecard. Findings go to code scanning; `dependency-review` and
1445
+ `pip-audit` are the two that fail a build — but note **where**: only `dependency-review` runs
1446
+ on a pull request. `pip-audit` runs weekly and on the merge to `main`, so a floor change is
1447
+ audited after it lands, not before. The weekly re-examination is the point of it: an advisory
1448
+ is published on the advisory database's clock, not on this repository's.
1397
1449
 
1398
1450
  ## Releasing
1399
1451
 
@@ -1416,10 +1468,20 @@ pip ignores pre-releases unless you pass `--pre`.
1416
1468
  Cutting a release is one tag:
1417
1469
 
1418
1470
  ```bash
1419
- git tag -a v0.9.0 -m "log-foundry 0.9.0"
1420
- git push origin v0.9.0
1471
+ git tag -a v1.2.3 -m "log-foundry 1.2.3"
1472
+ git push origin v1.2.3
1421
1473
  ```
1422
1474
 
1475
+ **One precondition, because it fails quietly.** If `docs/release-notes/<tag>.md` exists in the
1476
+ **tagged commit's tree**, the release job uses it as the release body; otherwise it generates
1477
+ one from the commit range. Both are valid, but the lookup is by exact tag name and the checkout
1478
+ takes the tag, so notes written for one version do nothing for another and notes added after
1479
+ the tagged commit are not seen. A release body cannot be amended once published — this
1480
+ repository has immutable releases — so check the file is there and named for the tag before
1481
+ pushing it. [`docs/release-notes/`](docs/release-notes/) holds the ones written so far, and
1482
+ [`docs/spec-delivery/RELEASES.md`](docs/spec-delivery/RELEASES.md) records which specs each
1483
+ released version carried.
1484
+
1423
1485
  Uploads authenticate with PyPI [Trusted Publishing](https://docs.pypi.org/trusted-publishers/)
1424
1486
  (OIDC) through the `pypi` GitHub Environment — there is no API token stored in the repository.
1425
1487
  A tagged build refuses to publish if the derived version doesn't match the tag, and the tagged
@@ -14,8 +14,11 @@ calls form a tree you can query later.
14
14
  - **Structured, never free-form** — every event is the same named-field JSON shape.
15
15
  - **Safe by default** — never captures your arguments or return values (no accidental PII/secret leakage), and the decorator **never swallows exceptions**.
16
16
  - **Correct under threads and asyncio** — context propagates via `contextvars`.
17
- - **Non-blocking delivery** — finished spans are handed to a background worker; your code never
18
- blocks on sink I/O, and a graceful drain at exit means buffered events aren't lost.
17
+ - **Non-blocking delivery from a traced call** — a finished span's events are handed to a
18
+ background worker, so code inside `@trace` never blocks on sink I/O, and a graceful drain at
19
+ exit means buffered events aren't lost. Two things you can do *deliberately* are synchronous:
20
+ a level call with **no open span**, which emits on your own thread, and `flush()`, which by
21
+ definition waits for the drain it asked for.
19
22
 
20
23
  ---
21
24
 
@@ -37,13 +40,14 @@ pip install 'log-foundry[aws]' # + boto3 for the SQS/SNS/Kinesis/Firehose sink
37
40
  >
38
41
  > - **`health()` and `sink.losses()` return frozen dataclasses**, not `NamedTuple`s. Attribute
39
42
  > access (`h.dropped`, `losses.failed`) is unchanged and is the whole contract; `len(h)`,
40
- > `h[0]` and `queued, dropped, failed = health()` now raise `TypeError`. `Health` has gained a
41
- > field in six consecutive specs and gains more — with no positions, that stops being breaking.
43
+ > `h[0]` and the four-way `queued, dropped, failed_batches, stopped_reason = health()` now
44
+ > raise `TypeError`. `Health` gains fields, and will gain more — with no positions to
45
+ > preserve, that stops being a breaking change.
42
46
  > - **`flush()` returns a `FlushResult` and `continue_trace()` a `ContinueResult`**, each truthy
43
47
  > or falsy with a `reason` naming *why*. `if lf.flush():` is unchanged; **`lf.flush() is True`
44
48
  > is not** — the result is an object. A one-bit return could not grow a reason later without
45
49
  > silently changing what `if flush():` means, which is why it moved now.
46
- > - **`SQSSink`'s injected client is keyword-only** (`SQSSink(url, client=…)`), **`SentrySink`
50
+ > - **`SQSSink`'s injected client is keyword-only** (`SQSSink(queue_url, client=…)`), **`SentrySink`
47
51
  > injects through `client=`** rather than the old `sdk` keyword, with no alias, and the sink attribute the
48
52
  > library assigns for interruptible backoff is **`log_foundry_stop_signal`**, not
49
53
  > `stop_signal` — a prefixed name cannot silently overwrite one your own sink already uses.
@@ -126,9 +130,9 @@ with the child span pointing at its parent via `parent_span_id`:
126
130
 
127
131
  ```json
128
132
  {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.start", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "754adb40e10c445f9ec9e23a2f3dcbf2", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}}
129
- {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "3af73c51540848afbeaba9fdf7a9dce8", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}, "duration_ms": 0.018, "status": "ok"}
133
+ {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "3af73c51540848afbeaba9fdf7a9dce8", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}, "duration_ms": 0.0233328901231289, "status": "ok"}
130
134
  {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.start", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "e789f7e5268b46d8b779c9cbcdde8656", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}}
131
- {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "8f3dbfcfcf4a45f688c738eefef882b0", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}, "duration_ms": 0.326, "status": "ok"}
135
+ {"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "8f3dbfcfcf4a45f688c738eefef882b0", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}, "duration_ms": 0.342666869983077, "status": "ok"}
132
136
  ```
133
137
 
134
138
  > **Note on ordering:** the child span (`tax.compute`) finishes first, so its events flush
@@ -140,7 +144,11 @@ with the child span pointing at its parent via `parent_span_id`:
140
144
 
141
145
  A traced call travels through a small pipeline. The first four steps run on your own thread
142
146
  and are deliberately fast; the last two run on a background thread so your code never waits on
143
- the destination.
147
+ the destination. That is the traced path. A level call with no **open** span takes the
148
+ synchronous route instead, described under
149
+ [Flushing and shutdown](#flushing-and-shutdown) — and "open" is the operative word: a task
150
+ that outlives its span still sees that span, but it has closed, so the call takes the
151
+ synchronous route too.
144
152
 
145
153
  1. **You call the code** — a `@trace` function, or one of the `debug`/`info`/… emitters.
146
154
  2. **A span opens** — a record of this one call. It inherits the current trace and parent (see
@@ -302,9 +310,9 @@ def process(): ...
302
310
  - `defaults` — per-decorator fields merged into every event this span emits.
303
311
 
304
312
  The **outermost** decorated call starts a new trace; every nested decorated call becomes a
305
- child span within it. On an exception, the decorator records `status="error"` plus the
306
- exception type and formatted stack, then **re-raises the original exception unchanged** — it
307
- never swallows errors.
313
+ child span within it. On an exception, the decorator records `status="error"` plus an `error`
314
+ sub-document carrying the exception's type, module, message and formatted stack, then
315
+ **re-raises the original exception unchanged** — it never swallows errors.
308
316
 
309
317
  **Async is supported.** Apply `@trace` to an `async def` and it traces the coroutine's actual
310
318
  run — the span opens when the coroutine starts and closes when the awaits complete, not when the
@@ -369,7 +377,7 @@ def process_payment(user_id: int) -> str:
369
377
  It reaches its own name too (`fields={"fields": ...}`), so every reserved word has exactly one
370
378
  route through. A key given both ways takes the keyword's value, since `**kwargs` is what you
371
379
  wrote at the call site and `fields=` is usually a mapping built elsewhere.
372
- - **Orphan logs** — a level call made with no active span is not dropped: it emits a standalone
380
+ - **Orphan logs** — a level call made with no **open** span is not dropped: it emits a standalone
373
381
  one-event span with a fresh `trace_id`, flushed straight to the sink.
374
382
 
375
383
  ### Continuing a trace across processes
@@ -411,9 +419,13 @@ def handler(event, context): # the entry point, deliberately not decorated
411
419
  lf.flush() # the span has closed, so its events are drained
412
420
  ```
413
421
 
414
- `flush()` goes **outside** the traced function, not in its `finally`. An in-span event lives on the
415
- span until the span *closes*, and `flush()` drains the queue — so a `flush()` inside the span has
416
- nothing to drain yet.
422
+ `flush()` works either side of the traced function. Putting it outside, as above, is still the
423
+ clearer shape — the span has closed, so there is nothing to reason about. But a `flush()`
424
+ *inside* an open span is no longer a mistake: it sweeps every open span in the calling context,
425
+ hands their buffered events to the worker and drains them, leaving the spans open. It does not
426
+ deliver a `span.end` that has not happened yet. Before `1.0.0` a flush inside a span swept
427
+ nothing, and this recipe with the `flush()` moved into the traced function delivered **nothing**
428
+ with every counter clean.
417
429
 
418
430
  | Call | Does |
419
431
  |---|---|
@@ -504,8 +516,16 @@ nothing.
504
516
 
505
517
  Wire one up by passing an instance to `configure(sink=...)`; if you never do, the first decorated
506
518
  call falls back to `StdoutSink()`. The **protocol** is a top-level export, alongside
507
- `SinkDeliveryError`, `SinkLosses` and `read_losses`; the **concrete sinks** are not, so import each
508
- from its own module, e.g. `from log_foundry.sinks.sqs import SQSSink`.
519
+ `SinkDeliveryError`, `SinkLosses`, `read_losses` and `flush_sink` — the last two being the probes
520
+ a wrapper sink needs to ask a child for its losses and to push its client-side buffer. The
521
+ **concrete sinks** are not exported, so import each from its own module, e.g.
522
+ `from log_foundry.sinks.sqs import SQSSink`.
523
+
524
+ **Construct `SinkLosses` with keywords**, as the example below does. The public dataclasses are
525
+ keyword-only from `1.0.0`, so a positional `SinkLosses(0, 3)` raises `TypeError` — and it raises
526
+ *inside* your `losses()`, where `read_losses` deliberately swallows it so a broken reporter cannot
527
+ take `health()` down. The symptom is not an error: your sink's loss reporting silently becomes
528
+ `None`.
509
529
 
510
530
  A few conventions hold across every sink below:
511
531
 
@@ -555,8 +575,8 @@ A few conventions hold across every sink below:
555
575
 
556
576
  | Sink | Import from | Configure |
557
577
  |---|---|---|
558
- | `StdoutSink` | `log_foundry.sinks.stdout` | `StdoutSink(stream=sys.stdout)` — one JSON line per event; the zero-config default |
559
- | `StderrSink` | `log_foundry.sinks.stdout` | `StderrSink(stream=sys.stderr)` — same, on stderr (twelve-factor) |
578
+ | `StdoutSink` | `log_foundry.sinks.stdout` | `StdoutSink(stream=None)` — one JSON line per event; `None` resolves to `sys.stdout` **at construction** and is not re-read per write. The zero-config default |
579
+ | `StderrSink` | `log_foundry.sinks.stdout` | `StderrSink(stream=None)` — same, resolving to `sys.stderr` (twelve-factor) |
560
580
  | `NullSink` | `log_foundry.sinks.null` | `NullSink()` — discard everything; `.dropped` counts events |
561
581
  | `MemorySink` | `log_foundry.sinks.memory` | `MemorySink(maxlen=None)` — collect into `.events` (a bounded ring when `maxlen` is set) |
562
582
 
@@ -912,7 +932,7 @@ class MySink:
912
932
 
913
933
  def losses(self) -> SinkLosses:
914
934
  with self._counter_lock: # both fields from one instant
915
- return SinkLosses(dropped=self._dropped, failed=self._failed)
935
+ return SinkLosses(dropped=self._dropped, failed=self._failed) # keywords required
916
936
 
917
937
  def close(self) -> None:
918
938
  with self._lock: # never release under an active writer
@@ -972,7 +992,12 @@ Every wrapper shipped here — `MultiSink`, `FilteringSink`, `TransformSink`, `S
972
992
 
973
993
  Delivery is off the hot path. When a span ends, its events are handed to a per-process
974
994
  background worker via a fast, non-blocking submit — your function returns without waiting on
975
- the sink. The worker batches events (by count and time), emits them on its own thread, retries
995
+ the sink. A level call with no **open** span has no span to end, so it emits synchronously on
996
+ the calling thread — including a call from a task that outlived the span it inherited, whose
997
+ span is present but closed. That path needs no worker, and in a process that only ever logs
998
+ that way none is ever built; it is a span closing, or a `flush()` sweeping one, that builds it.
999
+ The rest of this paragraph is about the buffered path.
1000
+ The worker batches events (by count and time), emits them on its own thread, retries
976
1001
  a failing sink with backoff, and applies backpressure so a slow or down sink can never block or
977
1002
  back-pressure the app: when its bounded queue is full it drops the newest submissions and counts
978
1003
  them rather than stalling.
@@ -1000,7 +1025,7 @@ They tell you different things, and they want different responses:
1000
1025
 
1001
1026
  | Field | Means | What to do |
1002
1027
  |---|---|---|
1003
- | `dropped` | The queue filled — the destination is not keeping up. Delivery continues. | Tune `batch_size`/`flush_interval`, or scale the sink. |
1028
+ | `dropped` | The queue filled — the destination is not keeping up. Delivery continues. | Make the destination keep up: scale the sink, or reduce what you log. The worker's batch size, flush interval and queue depth are **not** reachable from the public API, so tuning them is not an option this version offers you. |
1004
1029
  | `failed_batches` | A sink stayed broken through the whole retry budget. Delivery continues. | Fix the destination. |
1005
1030
  | `stopped_reason` | The background thread **died** on that exception type. Nothing further will be delivered, ever. | Restart the process; investigate the named exception. |
1006
1031
  | `sink.dropped` | The sink discarded events **before** attempting delivery — an oversized record, or one the client refused outright. | Read the stderr line: it names the cause. An oversized record means shrink what you log; a refused local produce/publish (Kafka, Pub/Sub) points at the client — a saturated buffer, a bad topic, a credential. |
@@ -1008,7 +1033,7 @@ They tell you different things, and they want different responses:
1008
1033
  | `retired` + `submitted_after_shutdown` | `shutdown()` was called and the process **kept logging**. Those events are queued where nothing will drain them — total loss, for as long as the process runs. | Use `flush()`, not `shutdown()`, in a process that logs again. This is the serverless mistake below. |
1009
1034
  | `incomplete_swaps` | A late `configure(sink=...)` could not confirm the previous sink was drained. The swap took effect; that sink was left open and some queued events may have gone to the new one. | Investigate the previous sink — it was hung or failing. Configure the sink before the first log where you can. |
1010
1035
  | `inherited_sink` | This process is delivering to a sink it **inherited across a `fork`** and may not release, so it will not be closed here. Not a loss and not an alert term. | Nothing, usually. It explains a handle still open after `shutdown()`, and tells you a deployment shares one sink across a fork at all. `True` for a shared `StdoutSink` too, whose `close()` only flushes — so a `True` is not by itself evidence that anything is held. If you want the child to own its transport, build the sink in the worker process (see Forking). |
1011
- | `orphan_lost` | An event logged **with no active span** never reached the sink. That call emits on your own thread with no worker behind it, so no other field here can carry it — it is not a batch, there was no retry, and there may be no worker at all. Covers a sink that failed to *construct* as well as one that raised. | Fix the destination, or the data. The stderr line names the exception type. If a process logs this way at all, this is the field to alert on: nothing else describes that path. |
1036
+ | `orphan_lost` | An event logged **with no open span** never reached the sink. That call emits on your own thread with no worker behind it, so no other field here can carry it — it is not a batch, there was no retry, and there may be no worker at all. Covers a sink that failed to *construct* as well as one that raised. | Fix the destination, or the data. The stderr line names the exception type. If a process logs this way at all, this is the field to alert on: nothing else describes that path. |
1012
1037
  | `in_span_lost` | An event logged **inside a span** could not be built — a value that could not be turned into an event. Always the data, never the destination: the in-span path cannot fail at delivery, which is `failed_batches`. | Fix the call site. Passing a non-string message (an exception object, say) is the common cause. |
1013
1038
  | `closing_sinks` | Swapped-out sinks inside `close()` **right now** — a live gauge, not a counter. It and `queued` are the two fields here that fall as well as rise; the other integer counters only climb. Non-zero on a single read is normal during a swap. | Nothing, unless it stays non-zero. That means a destination is stuck in `close()` and will not release its resources. |
1014
1039
 
@@ -1016,8 +1041,9 @@ They tell you different things, and they want different responses:
1016
1041
  reported. They aggregate different failure populations — one can mean the destination *or* the
1017
1042
  data, the other can only mean the data — so a single number would hide which fix applies.
1018
1043
 
1019
- `h.sink` is a `SinkLosses(dropped, failed)` or `None` — `None` when no worker exists yet, or when
1020
- the configured sink reports nothing (`losses()` is optional). Note the two `dropped` fields count
1044
+ `h.sink` is a `SinkLosses`, carrying `dropped` and `failed`, or `None` — `None` when no worker
1045
+ exists yet, or when the configured sink reports nothing (`losses()` is optional, and a sink whose
1046
+ `losses()` raises reports `None` too). Note the two `dropped` fields count
1021
1047
  different things: the worker's is backpressure at *its* queue, the sink's is an event that never
1022
1048
  reached the wire. They are separate because the remedies do not overlap — and `sink.dropped` is
1023
1049
  itself two causes, which is why the diagnostic line matters. Most sinks drop only what can never
@@ -1058,11 +1084,12 @@ guards its own post-close state — rather than queued where nothing will drain
1058
1084
  the same claim. A stateless sink such as the default `StdoutSink` still accepts it.
1059
1085
 
1060
1086
  Read a snapshot **by attribute** (`h.dropped`), as above — that is the whole contract. `Health`
1061
- and `SinkLosses` are frozen dataclasses, so `len(h)`, `h[0]` and `queued, dropped, failed =
1062
- health()` all raise `TypeError`. They were `NamedTuple`s before `1.0.0` and the tuple shape is
1063
- deliberately gone: `Health` has gained a field in six consecutive specs and gains more, and every
1064
- one of those had to argue that the positions before it were undisturbed. There are no positions to
1065
- disturb now, and adding a field is not a breaking change.
1087
+ and `SinkLosses` are frozen dataclasses, so `len(h)`, `h[0]` and the four-way
1088
+ `queued, dropped, failed_batches, stopped_reason = health()` all raise `TypeError`. They were
1089
+ `NamedTuple`s before `1.0.0`, and the tuple shape is deliberately gone: `Health` gains fields
1090
+ as the library learns to report more, and while it was a tuple every one of those additions
1091
+ had to argue that the positions before it were undisturbed. There are no positions to disturb
1092
+ now, and adding a field is not a breaking change.
1066
1093
 
1067
1094
  `dropped` counts submissions discarded because the queue filled; `failed_batches` counts batches
1068
1095
  abandoned after the retry budget was spent. Overflow also warns on stderr — on the first drop and
@@ -1122,7 +1149,7 @@ was outstanding** — the drain it forces reached the sink, and so did anything
1122
1149
  emitted while it waited its turn. A truthy result is evidence of delivery, not merely that a drain
1123
1150
  took place.
1124
1151
 
1125
- Falsy carries a `reason` saying which of five things happened, because they need different fixes:
1152
+ Falsy carries a `reason` saying which of these happened, because they need different fixes:
1126
1153
 
1127
1154
  | `reason` | Means |
1128
1155
  |---|---|
@@ -1131,6 +1158,7 @@ Falsy carries a `reason` saying which of five things happened, because they need
1131
1158
  | `"thread-died"` | The drain thread is gone; see `health().stopped_reason`. |
1132
1159
  | `"queue-full"` | Backpressure — the queue could not even accept the marker. |
1133
1160
  | `"abandoned"` | A batch was given up on after its retry budget. The destination is broken. |
1161
+ | `"sink-flush"` | Everything queued reached the sink, but the sink's own `flush()` raised — a client-side buffer that did not go out. Distinct from `"abandoned"`: the library delivered, the sink did not. |
1134
1162
 
1135
1163
  ```python
1136
1164
  result = lf.flush(5.0)
@@ -1189,7 +1217,8 @@ def _handler(event, context):
1189
1217
 
1190
1218
  def handler(event, context):
1191
1219
  # NOT decorated, so the span closes when `_handler` returns and its events reach the queue
1192
- # before `flush()` runs. A `flush()` *inside* the traced function has nothing to drain yet.
1220
+ # before `flush()` runs. Since 1.0.0 a `flush()` *inside* the traced function also works —
1221
+ # it sweeps the open span — but keeping it out here means there is nothing to reason about.
1193
1222
  try:
1194
1223
  return _handler(event, context)
1195
1224
  finally:
@@ -1231,9 +1260,9 @@ Every event is the same shape (arch §6). Boundary events add a few fields:
1231
1260
  | `function` | ✓ | span name |
1232
1261
  | `service` / `version` / `env` | ✓ | from `configure(...)` |
1233
1262
  | `fields` | ✓ | merged user fields (config `defaults` → span `defaults` → …) |
1234
- | `duration_ms` | span.end | wall time from a monotonic delta |
1263
+ | `duration_ms` | span.end | wall time from a monotonic delta, unrounded — full float precision |
1235
1264
  | `status` | span.end | `"ok"` or `"error"` |
1236
- | `error` | on failure | `{"type": ..., "stack": ...}` |
1265
+ | `error` | on failure | `{"type": ..., "module": ..., "message": ..., "stack": ...}` |
1237
1266
  | `truncated` | when a ceiling fired | `true`; absent otherwise, never `false` |
1238
1267
 
1239
1268
  IDs are [W3C Trace Context](https://www.w3.org/TR/trace-context/)-compatible by design, so the
@@ -1278,18 +1307,30 @@ poetry run pytest # test (runs in parallel by default; see addopts)
1278
1307
  poetry run pytest -n 0 # ...serially, when debugging a failure
1279
1308
  poetry run ruff check . # lint (line-length 100)
1280
1309
  poetry run mypy # typecheck (strict, over src/)
1310
+ sh scripts/spec-lint.sh # lint the design specs
1311
+ sh scripts/docs-lint.sh # the always-loaded docs tier — NOTHING IN CI RUNS THIS
1312
+ poetry run python scripts/docstring-lint.py # the docstring rule over src/ — ALSO NOT IN CI
1281
1313
  ```
1282
1314
 
1315
+ All six run before a push. The last two are the ones to remember, because they are deliberately
1316
+ not CI jobs: they hold the documentation and the docstrings to rules a reviewer would otherwise
1317
+ have to carry, and keeping them local means the failure lands on whoever caused it rather than on
1318
+ a shared branch. Each has a `-test.sh` beside it that proves its own checks still fire; run that
1319
+ one too if you change the linter.
1320
+
1283
1321
  The library uses a src layout (`src/log_foundry/`) with a single concept per module: `config`,
1284
- `ids`, `model`, `context`, `decorator`, `api`, `console`, `worker`, and the `sinks/` package (the
1285
- `base` protocol, `stdout`, and one module per sink family — see [Sinks](#sinks)).
1322
+ `ids`, `model`, `context`, `decorator`, `api`, `console`, `worker`, `sanitize` and `results`,
1323
+ the internal `_lifecycle`, `_fork` and `_diag`, and the `sinks/` package (the `base` protocol,
1324
+ `stdout`, and one module per sink family — see [Sinks](#sinks)). Anything underscore-prefixed
1325
+ is internal and moves without notice.
1286
1326
  Deeper design docs live in [`docs/`](docs/) — start with [`docs/architecture.md`](docs/architecture.md).
1287
1327
 
1288
1328
  ### Continuous integration
1289
1329
 
1290
- Every pull request runs the checks below. A check whose verdict cannot change on the tree in
1291
- front of it is path-filtered rather than run to the same answer twice, so the *When* column is
1292
- part of the contract:
1330
+ The checks below guard this repository. Most run on every pull request; a check whose verdict
1331
+ cannot change on the tree in front of it is path-filtered rather than run to the same answer
1332
+ twice, and two run only after the merge — so the *When* column is part of the contract and is
1333
+ worth reading before assuming a PR was audited:
1293
1334
 
1294
1335
  | Check | Does | When | Fails the build |
1295
1336
  |---|---|---|---|
@@ -1298,6 +1339,14 @@ part of the contract:
1298
1339
  | [`dependency-review.yml`](.github/workflows/dependency-review.yml) | fails a PR that *introduces* a dependency with a known advisory (`moderate`+) | every PR | yes |
1299
1340
  | [`zizmor.yml`](.github/workflows/zizmor.yml) | static analysis of the workflow files themselves | workflow, action, dependabot or zizmor config touched; also weekly | no — reports to code scanning |
1300
1341
  | CodeQL | `python` + `actions`, `extended` query suite; also weekly | every PR | no — reports to code scanning |
1342
+ | [`integration.yml`](.github/workflows/integration.yml) | the extras-backed sinks against nine real services in containers | sinks, `tests/integration/`, `pyproject.toml`, `poetry.lock` or that workflow touched; also weekly | it goes red, and like every check here it is advisory — `main` requires no status check at all, and this one is furthest from earning that: nine service containers are a flakiness budget |
1343
+ | [`pip-audit.yml`](.github/workflows/pip-audit.yml) | advisories across every extra, `--strict` | **not on PRs at all** — on the merge to `main` when the lockfile, extras, the ignore list or that workflow moved, and weekly | yes, on `main` |
1344
+ | [`scorecard.yml`](.github/workflows/scorecard.yml) | OpenSSF Scorecard over the repository's own supply chain | **not on PRs** — on the merge to `main` when `.github/`, `scripts/`, `SECURITY.md`, `LICENSE` or the lockfile moved, and weekly | no — reports to code scanning |
1345
+
1346
+ **`scripts/docs-lint.sh` is deliberately not in that table**, because nothing in CI runs it. It
1347
+ holds the always-loaded documentation tier to its budgets and is a local pre-push gate, so its
1348
+ failure lands on whoever caused it rather than on a shared branch. Contributors run it, along
1349
+ with `scripts/spec-lint.sh`, before pushing.
1301
1350
 
1302
1351
  On a push to `main` the full `ci.yml` matrix still runs, but as the first job of
1303
1352
  [`release.yml`](.github/workflows/release.yml), which `uses:` this same workflow before it is
@@ -1340,9 +1389,12 @@ everything it instruments:
1340
1389
  [`SECURITY.md`](SECURITY.md#software-bill-of-materials) has the detail.
1341
1390
 
1342
1391
  Scanning runs continuously rather than at release time: CodeQL over the source and the workflows,
1343
- zizmor over the workflows, `dependency-review` on every pull request, a weekly `pip-audit` across
1344
- all eleven extras, and OpenSSF Scorecard. Findings go to code scanning; `dependency-review` and
1345
- `pip-audit` are the two that fail a build.
1392
+ zizmor over the workflows, `dependency-review` on every pull request, `pip-audit` across all
1393
+ eleven extras, and OpenSSF Scorecard. Findings go to code scanning; `dependency-review` and
1394
+ `pip-audit` are the two that fail a build — but note **where**: only `dependency-review` runs
1395
+ on a pull request. `pip-audit` runs weekly and on the merge to `main`, so a floor change is
1396
+ audited after it lands, not before. The weekly re-examination is the point of it: an advisory
1397
+ is published on the advisory database's clock, not on this repository's.
1346
1398
 
1347
1399
  ## Releasing
1348
1400
 
@@ -1365,10 +1417,20 @@ pip ignores pre-releases unless you pass `--pre`.
1365
1417
  Cutting a release is one tag:
1366
1418
 
1367
1419
  ```bash
1368
- git tag -a v0.9.0 -m "log-foundry 0.9.0"
1369
- git push origin v0.9.0
1420
+ git tag -a v1.2.3 -m "log-foundry 1.2.3"
1421
+ git push origin v1.2.3
1370
1422
  ```
1371
1423
 
1424
+ **One precondition, because it fails quietly.** If `docs/release-notes/<tag>.md` exists in the
1425
+ **tagged commit's tree**, the release job uses it as the release body; otherwise it generates
1426
+ one from the commit range. Both are valid, but the lookup is by exact tag name and the checkout
1427
+ takes the tag, so notes written for one version do nothing for another and notes added after
1428
+ the tagged commit are not seen. A release body cannot be amended once published — this
1429
+ repository has immutable releases — so check the file is there and named for the tag before
1430
+ pushing it. [`docs/release-notes/`](docs/release-notes/) holds the ones written so far, and
1431
+ [`docs/spec-delivery/RELEASES.md`](docs/spec-delivery/RELEASES.md) records which specs each
1432
+ released version carried.
1433
+
1372
1434
  Uploads authenticate with PyPI [Trusted Publishing](https://docs.pypi.org/trusted-publishers/)
1373
1435
  (OIDC) through the `pypi` GitHub Environment — there is no API token stored in the repository.
1374
1436
  A tagged build refuses to publish if the derived version doesn't match the tag, and the tagged
@@ -74,7 +74,7 @@ keywords = [
74
74
  # vulnerability-reporting channel. The repository is still named `log-forge` — the ORIGINAL name,
75
75
  # which PyPI rejected for the distribution — so these URLs deliberately do not match the package
76
76
  # name. See the note on `name` above before "correcting" them.
77
- version = "0.10.2.dev110"
77
+ version = "0.10.2.dev111"
78
78
 
79
79
  [project.urls]
80
80
  Homepage = "https://github.com/agriffi10/log-forge"