log-foundry 0.10.2.dev109__tar.gz → 0.10.2.dev111__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/PKG-INFO +107 -45
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/README.md +106 -44
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/pyproject.toml +1 -1
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/LICENSE +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/__init__.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/_diag.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/_fork.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/_lifecycle.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/api.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/config.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/console.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/context.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/decorator.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/ids.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/model.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/py.typed +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/results.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sanitize.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/__init__.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/_batch.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/_chunk.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/_retry.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/_socket.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/_time.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/base.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/callback.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/clickhouse.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/datadog.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/elasticsearch.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/eventhubs.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/file.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/filtering.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/firehose.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/honeycomb.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/http.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/kafka.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/kinesis.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/logging_sink.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/logstash.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/loki.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/memory.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/mongodb.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/multi.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/nats.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/newrelic.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/null.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/postgres.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/pubsub.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/rabbitmq.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/redis.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/sentry.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/sns.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/splunk.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/sqlite.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/sqs.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/stdout.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/syslog.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/transform.py +0 -0
- {log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/worker.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: log-foundry
|
|
3
|
-
Version: 0.10.2.
|
|
3
|
+
Version: 0.10.2.dev111
|
|
4
4
|
Summary: Generate logs for your console and JSON events for downstream consumption.
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
License-File: LICENSE
|
|
@@ -65,8 +65,11 @@ calls form a tree you can query later.
|
|
|
65
65
|
- **Structured, never free-form** — every event is the same named-field JSON shape.
|
|
66
66
|
- **Safe by default** — never captures your arguments or return values (no accidental PII/secret leakage), and the decorator **never swallows exceptions**.
|
|
67
67
|
- **Correct under threads and asyncio** — context propagates via `contextvars`.
|
|
68
|
-
- **Non-blocking delivery** — finished
|
|
69
|
-
blocks on sink I/O, and a graceful drain at
|
|
68
|
+
- **Non-blocking delivery from a traced call** — a finished span's events are handed to a
|
|
69
|
+
background worker, so code inside `@trace` never blocks on sink I/O, and a graceful drain at
|
|
70
|
+
exit means buffered events aren't lost. Two things you can do *deliberately* are synchronous:
|
|
71
|
+
a level call with **no open span**, which emits on your own thread, and `flush()`, which by
|
|
72
|
+
definition waits for the drain it asked for.
|
|
70
73
|
|
|
71
74
|
---
|
|
72
75
|
|
|
@@ -88,13 +91,14 @@ pip install 'log-foundry[aws]' # + boto3 for the SQS/SNS/Kinesis/Firehose sink
|
|
|
88
91
|
>
|
|
89
92
|
> - **`health()` and `sink.losses()` return frozen dataclasses**, not `NamedTuple`s. Attribute
|
|
90
93
|
> access (`h.dropped`, `losses.failed`) is unchanged and is the whole contract; `len(h)`,
|
|
91
|
-
> `h[0]` and `queued, dropped,
|
|
92
|
-
>
|
|
94
|
+
> `h[0]` and the four-way `queued, dropped, failed_batches, stopped_reason = health()` now
|
|
95
|
+
> raise `TypeError`. `Health` gains fields, and will gain more — with no positions to
|
|
96
|
+
> preserve, that stops being a breaking change.
|
|
93
97
|
> - **`flush()` returns a `FlushResult` and `continue_trace()` a `ContinueResult`**, each truthy
|
|
94
98
|
> or falsy with a `reason` naming *why*. `if lf.flush():` is unchanged; **`lf.flush() is True`
|
|
95
99
|
> is not** — the result is an object. A one-bit return could not grow a reason later without
|
|
96
100
|
> silently changing what `if flush():` means, which is why it moved now.
|
|
97
|
-
> - **`SQSSink`'s injected client is keyword-only** (`SQSSink(
|
|
101
|
+
> - **`SQSSink`'s injected client is keyword-only** (`SQSSink(queue_url, client=…)`), **`SentrySink`
|
|
98
102
|
> injects through `client=`** rather than the old `sdk` keyword, with no alias, and the sink attribute the
|
|
99
103
|
> library assigns for interruptible backoff is **`log_foundry_stop_signal`**, not
|
|
100
104
|
> `stop_signal` — a prefixed name cannot silently overwrite one your own sink already uses.
|
|
@@ -177,9 +181,9 @@ with the child span pointing at its parent via `parent_span_id`:
|
|
|
177
181
|
|
|
178
182
|
```json
|
|
179
183
|
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.start", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "754adb40e10c445f9ec9e23a2f3dcbf2", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}}
|
|
180
|
-
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "3af73c51540848afbeaba9fdf7a9dce8", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}, "duration_ms": 0.
|
|
184
|
+
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "3af73c51540848afbeaba9fdf7a9dce8", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}, "duration_ms": 0.0233328901231289, "status": "ok"}
|
|
181
185
|
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.start", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "e789f7e5268b46d8b779c9cbcdde8656", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}}
|
|
182
|
-
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "8f3dbfcfcf4a45f688c738eefef882b0", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}, "duration_ms": 0.
|
|
186
|
+
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "8f3dbfcfcf4a45f688c738eefef882b0", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}, "duration_ms": 0.342666869983077, "status": "ok"}
|
|
183
187
|
```
|
|
184
188
|
|
|
185
189
|
> **Note on ordering:** the child span (`tax.compute`) finishes first, so its events flush
|
|
@@ -191,7 +195,11 @@ with the child span pointing at its parent via `parent_span_id`:
|
|
|
191
195
|
|
|
192
196
|
A traced call travels through a small pipeline. The first four steps run on your own thread
|
|
193
197
|
and are deliberately fast; the last two run on a background thread so your code never waits on
|
|
194
|
-
the destination.
|
|
198
|
+
the destination. That is the traced path. A level call with no **open** span takes the
|
|
199
|
+
synchronous route instead, described under
|
|
200
|
+
[Flushing and shutdown](#flushing-and-shutdown) — and "open" is the operative word: a task
|
|
201
|
+
that outlives its span still sees that span, but it has closed, so the call takes the
|
|
202
|
+
synchronous route too.
|
|
195
203
|
|
|
196
204
|
1. **You call the code** — a `@trace` function, or one of the `debug`/`info`/… emitters.
|
|
197
205
|
2. **A span opens** — a record of this one call. It inherits the current trace and parent (see
|
|
@@ -353,9 +361,9 @@ def process(): ...
|
|
|
353
361
|
- `defaults` — per-decorator fields merged into every event this span emits.
|
|
354
362
|
|
|
355
363
|
The **outermost** decorated call starts a new trace; every nested decorated call becomes a
|
|
356
|
-
child span within it. On an exception, the decorator records `status="error"` plus
|
|
357
|
-
exception type and formatted stack, then
|
|
358
|
-
never swallows errors.
|
|
364
|
+
child span within it. On an exception, the decorator records `status="error"` plus an `error`
|
|
365
|
+
sub-document carrying the exception's type, module, message and formatted stack, then
|
|
366
|
+
**re-raises the original exception unchanged** — it never swallows errors.
|
|
359
367
|
|
|
360
368
|
**Async is supported.** Apply `@trace` to an `async def` and it traces the coroutine's actual
|
|
361
369
|
run — the span opens when the coroutine starts and closes when the awaits complete, not when the
|
|
@@ -420,7 +428,7 @@ def process_payment(user_id: int) -> str:
|
|
|
420
428
|
It reaches its own name too (`fields={"fields": ...}`), so every reserved word has exactly one
|
|
421
429
|
route through. A key given both ways takes the keyword's value, since `**kwargs` is what you
|
|
422
430
|
wrote at the call site and `fields=` is usually a mapping built elsewhere.
|
|
423
|
-
- **Orphan logs** — a level call made with no
|
|
431
|
+
- **Orphan logs** — a level call made with no **open** span is not dropped: it emits a standalone
|
|
424
432
|
one-event span with a fresh `trace_id`, flushed straight to the sink.
|
|
425
433
|
|
|
426
434
|
### Continuing a trace across processes
|
|
@@ -462,9 +470,13 @@ def handler(event, context): # the entry point, deliberately not decorated
|
|
|
462
470
|
lf.flush() # the span has closed, so its events are drained
|
|
463
471
|
```
|
|
464
472
|
|
|
465
|
-
`flush()`
|
|
466
|
-
|
|
467
|
-
|
|
473
|
+
`flush()` works either side of the traced function. Putting it outside, as above, is still the
|
|
474
|
+
clearer shape — the span has closed, so there is nothing to reason about. But a `flush()`
|
|
475
|
+
*inside* an open span is no longer a mistake: it sweeps every open span in the calling context,
|
|
476
|
+
hands their buffered events to the worker and drains them, leaving the spans open. It does not
|
|
477
|
+
deliver a `span.end` that has not happened yet. Before `1.0.0` a flush inside a span swept
|
|
478
|
+
nothing, and this recipe with the `flush()` moved into the traced function delivered **nothing**
|
|
479
|
+
with every counter clean.
|
|
468
480
|
|
|
469
481
|
| Call | Does |
|
|
470
482
|
|---|---|
|
|
@@ -555,8 +567,16 @@ nothing.
|
|
|
555
567
|
|
|
556
568
|
Wire one up by passing an instance to `configure(sink=...)`; if you never do, the first decorated
|
|
557
569
|
call falls back to `StdoutSink()`. The **protocol** is a top-level export, alongside
|
|
558
|
-
`SinkDeliveryError`, `SinkLosses` and `
|
|
559
|
-
|
|
570
|
+
`SinkDeliveryError`, `SinkLosses`, `read_losses` and `flush_sink` — the last two being the probes
|
|
571
|
+
a wrapper sink needs to ask a child for its losses and to push its client-side buffer. The
|
|
572
|
+
**concrete sinks** are not exported, so import each from its own module, e.g.
|
|
573
|
+
`from log_foundry.sinks.sqs import SQSSink`.
|
|
574
|
+
|
|
575
|
+
**Construct `SinkLosses` with keywords**, as the example below does. The public dataclasses are
|
|
576
|
+
keyword-only from `1.0.0`, so a positional `SinkLosses(0, 3)` raises `TypeError` — and it raises
|
|
577
|
+
*inside* your `losses()`, where `read_losses` deliberately swallows it so a broken reporter cannot
|
|
578
|
+
take `health()` down. The symptom is not an error: your sink's loss reporting silently becomes
|
|
579
|
+
`None`.
|
|
560
580
|
|
|
561
581
|
A few conventions hold across every sink below:
|
|
562
582
|
|
|
@@ -606,8 +626,8 @@ A few conventions hold across every sink below:
|
|
|
606
626
|
|
|
607
627
|
| Sink | Import from | Configure |
|
|
608
628
|
|---|---|---|
|
|
609
|
-
| `StdoutSink` | `log_foundry.sinks.stdout` | `StdoutSink(stream=
|
|
610
|
-
| `StderrSink` | `log_foundry.sinks.stdout` | `StderrSink(stream=
|
|
629
|
+
| `StdoutSink` | `log_foundry.sinks.stdout` | `StdoutSink(stream=None)` — one JSON line per event; `None` resolves to `sys.stdout` **at construction** and is not re-read per write. The zero-config default |
|
|
630
|
+
| `StderrSink` | `log_foundry.sinks.stdout` | `StderrSink(stream=None)` — same, resolving to `sys.stderr` (twelve-factor) |
|
|
611
631
|
| `NullSink` | `log_foundry.sinks.null` | `NullSink()` — discard everything; `.dropped` counts events |
|
|
612
632
|
| `MemorySink` | `log_foundry.sinks.memory` | `MemorySink(maxlen=None)` — collect into `.events` (a bounded ring when `maxlen` is set) |
|
|
613
633
|
|
|
@@ -963,7 +983,7 @@ class MySink:
|
|
|
963
983
|
|
|
964
984
|
def losses(self) -> SinkLosses:
|
|
965
985
|
with self._counter_lock: # both fields from one instant
|
|
966
|
-
return SinkLosses(dropped=self._dropped, failed=self._failed)
|
|
986
|
+
return SinkLosses(dropped=self._dropped, failed=self._failed) # keywords required
|
|
967
987
|
|
|
968
988
|
def close(self) -> None:
|
|
969
989
|
with self._lock: # never release under an active writer
|
|
@@ -1023,7 +1043,12 @@ Every wrapper shipped here — `MultiSink`, `FilteringSink`, `TransformSink`, `S
|
|
|
1023
1043
|
|
|
1024
1044
|
Delivery is off the hot path. When a span ends, its events are handed to a per-process
|
|
1025
1045
|
background worker via a fast, non-blocking submit — your function returns without waiting on
|
|
1026
|
-
the sink.
|
|
1046
|
+
the sink. A level call with no **open** span has no span to end, so it emits synchronously on
|
|
1047
|
+
the calling thread — including a call from a task that outlived the span it inherited, whose
|
|
1048
|
+
span is present but closed. That path needs no worker, and in a process that only ever logs
|
|
1049
|
+
that way none is ever built; it is a span closing, or a `flush()` sweeping one, that builds it.
|
|
1050
|
+
The rest of this paragraph is about the buffered path.
|
|
1051
|
+
The worker batches events (by count and time), emits them on its own thread, retries
|
|
1027
1052
|
a failing sink with backoff, and applies backpressure so a slow or down sink can never block or
|
|
1028
1053
|
back-pressure the app: when its bounded queue is full it drops the newest submissions and counts
|
|
1029
1054
|
them rather than stalling.
|
|
@@ -1051,7 +1076,7 @@ They tell you different things, and they want different responses:
|
|
|
1051
1076
|
|
|
1052
1077
|
| Field | Means | What to do |
|
|
1053
1078
|
|---|---|---|
|
|
1054
|
-
| `dropped` | The queue filled — the destination is not keeping up. Delivery continues. |
|
|
1079
|
+
| `dropped` | The queue filled — the destination is not keeping up. Delivery continues. | Make the destination keep up: scale the sink, or reduce what you log. The worker's batch size, flush interval and queue depth are **not** reachable from the public API, so tuning them is not an option this version offers you. |
|
|
1055
1080
|
| `failed_batches` | A sink stayed broken through the whole retry budget. Delivery continues. | Fix the destination. |
|
|
1056
1081
|
| `stopped_reason` | The background thread **died** on that exception type. Nothing further will be delivered, ever. | Restart the process; investigate the named exception. |
|
|
1057
1082
|
| `sink.dropped` | The sink discarded events **before** attempting delivery — an oversized record, or one the client refused outright. | Read the stderr line: it names the cause. An oversized record means shrink what you log; a refused local produce/publish (Kafka, Pub/Sub) points at the client — a saturated buffer, a bad topic, a credential. |
|
|
@@ -1059,7 +1084,7 @@ They tell you different things, and they want different responses:
|
|
|
1059
1084
|
| `retired` + `submitted_after_shutdown` | `shutdown()` was called and the process **kept logging**. Those events are queued where nothing will drain them — total loss, for as long as the process runs. | Use `flush()`, not `shutdown()`, in a process that logs again. This is the serverless mistake below. |
|
|
1060
1085
|
| `incomplete_swaps` | A late `configure(sink=...)` could not confirm the previous sink was drained. The swap took effect; that sink was left open and some queued events may have gone to the new one. | Investigate the previous sink — it was hung or failing. Configure the sink before the first log where you can. |
|
|
1061
1086
|
| `inherited_sink` | This process is delivering to a sink it **inherited across a `fork`** and may not release, so it will not be closed here. Not a loss and not an alert term. | Nothing, usually. It explains a handle still open after `shutdown()`, and tells you a deployment shares one sink across a fork at all. `True` for a shared `StdoutSink` too, whose `close()` only flushes — so a `True` is not by itself evidence that anything is held. If you want the child to own its transport, build the sink in the worker process (see Forking). |
|
|
1062
|
-
| `orphan_lost` | An event logged **with no
|
|
1087
|
+
| `orphan_lost` | An event logged **with no open span** never reached the sink. That call emits on your own thread with no worker behind it, so no other field here can carry it — it is not a batch, there was no retry, and there may be no worker at all. Covers a sink that failed to *construct* as well as one that raised. | Fix the destination, or the data. The stderr line names the exception type. If a process logs this way at all, this is the field to alert on: nothing else describes that path. |
|
|
1063
1088
|
| `in_span_lost` | An event logged **inside a span** could not be built — a value that could not be turned into an event. Always the data, never the destination: the in-span path cannot fail at delivery, which is `failed_batches`. | Fix the call site. Passing a non-string message (an exception object, say) is the common cause. |
|
|
1064
1089
|
| `closing_sinks` | Swapped-out sinks inside `close()` **right now** — a live gauge, not a counter. It and `queued` are the two fields here that fall as well as rise; the other integer counters only climb. Non-zero on a single read is normal during a swap. | Nothing, unless it stays non-zero. That means a destination is stuck in `close()` and will not release its resources. |
|
|
1065
1090
|
|
|
@@ -1067,8 +1092,9 @@ They tell you different things, and they want different responses:
|
|
|
1067
1092
|
reported. They aggregate different failure populations — one can mean the destination *or* the
|
|
1068
1093
|
data, the other can only mean the data — so a single number would hide which fix applies.
|
|
1069
1094
|
|
|
1070
|
-
`h.sink` is a `SinkLosses
|
|
1071
|
-
the configured sink reports nothing (`losses()` is optional
|
|
1095
|
+
`h.sink` is a `SinkLosses`, carrying `dropped` and `failed`, or `None` — `None` when no worker
|
|
1096
|
+
exists yet, or when the configured sink reports nothing (`losses()` is optional, and a sink whose
|
|
1097
|
+
`losses()` raises reports `None` too). Note the two `dropped` fields count
|
|
1072
1098
|
different things: the worker's is backpressure at *its* queue, the sink's is an event that never
|
|
1073
1099
|
reached the wire. They are separate because the remedies do not overlap — and `sink.dropped` is
|
|
1074
1100
|
itself two causes, which is why the diagnostic line matters. Most sinks drop only what can never
|
|
@@ -1109,11 +1135,12 @@ guards its own post-close state — rather than queued where nothing will drain
|
|
|
1109
1135
|
the same claim. A stateless sink such as the default `StdoutSink` still accepts it.
|
|
1110
1136
|
|
|
1111
1137
|
Read a snapshot **by attribute** (`h.dropped`), as above — that is the whole contract. `Health`
|
|
1112
|
-
and `SinkLosses` are frozen dataclasses, so `len(h)`, `h[0]` and
|
|
1113
|
-
health()` all raise `TypeError`. They were
|
|
1114
|
-
|
|
1115
|
-
|
|
1116
|
-
|
|
1138
|
+
and `SinkLosses` are frozen dataclasses, so `len(h)`, `h[0]` and the four-way
|
|
1139
|
+
`queued, dropped, failed_batches, stopped_reason = health()` all raise `TypeError`. They were
|
|
1140
|
+
`NamedTuple`s before `1.0.0`, and the tuple shape is deliberately gone: `Health` gains fields
|
|
1141
|
+
as the library learns to report more, and while it was a tuple every one of those additions
|
|
1142
|
+
had to argue that the positions before it were undisturbed. There are no positions to disturb
|
|
1143
|
+
now, and adding a field is not a breaking change.
|
|
1117
1144
|
|
|
1118
1145
|
`dropped` counts submissions discarded because the queue filled; `failed_batches` counts batches
|
|
1119
1146
|
abandoned after the retry budget was spent. Overflow also warns on stderr — on the first drop and
|
|
@@ -1173,7 +1200,7 @@ was outstanding** — the drain it forces reached the sink, and so did anything
|
|
|
1173
1200
|
emitted while it waited its turn. A truthy result is evidence of delivery, not merely that a drain
|
|
1174
1201
|
took place.
|
|
1175
1202
|
|
|
1176
|
-
Falsy carries a `reason` saying which of
|
|
1203
|
+
Falsy carries a `reason` saying which of these happened, because they need different fixes:
|
|
1177
1204
|
|
|
1178
1205
|
| `reason` | Means |
|
|
1179
1206
|
|---|---|
|
|
@@ -1182,6 +1209,7 @@ Falsy carries a `reason` saying which of five things happened, because they need
|
|
|
1182
1209
|
| `"thread-died"` | The drain thread is gone; see `health().stopped_reason`. |
|
|
1183
1210
|
| `"queue-full"` | Backpressure — the queue could not even accept the marker. |
|
|
1184
1211
|
| `"abandoned"` | A batch was given up on after its retry budget. The destination is broken. |
|
|
1212
|
+
| `"sink-flush"` | Everything queued reached the sink, but the sink's own `flush()` raised — a client-side buffer that did not go out. Distinct from `"abandoned"`: the library delivered, the sink did not. |
|
|
1185
1213
|
|
|
1186
1214
|
```python
|
|
1187
1215
|
result = lf.flush(5.0)
|
|
@@ -1240,7 +1268,8 @@ def _handler(event, context):
|
|
|
1240
1268
|
|
|
1241
1269
|
def handler(event, context):
|
|
1242
1270
|
# NOT decorated, so the span closes when `_handler` returns and its events reach the queue
|
|
1243
|
-
# before `flush()` runs.
|
|
1271
|
+
# before `flush()` runs. Since 1.0.0 a `flush()` *inside* the traced function also works —
|
|
1272
|
+
# it sweeps the open span — but keeping it out here means there is nothing to reason about.
|
|
1244
1273
|
try:
|
|
1245
1274
|
return _handler(event, context)
|
|
1246
1275
|
finally:
|
|
@@ -1282,9 +1311,9 @@ Every event is the same shape (arch §6). Boundary events add a few fields:
|
|
|
1282
1311
|
| `function` | ✓ | span name |
|
|
1283
1312
|
| `service` / `version` / `env` | ✓ | from `configure(...)` |
|
|
1284
1313
|
| `fields` | ✓ | merged user fields (config `defaults` → span `defaults` → …) |
|
|
1285
|
-
| `duration_ms` | span.end | wall time from a monotonic delta |
|
|
1314
|
+
| `duration_ms` | span.end | wall time from a monotonic delta, unrounded — full float precision |
|
|
1286
1315
|
| `status` | span.end | `"ok"` or `"error"` |
|
|
1287
|
-
| `error` | on failure | `{"type": ..., "stack": ...}` |
|
|
1316
|
+
| `error` | on failure | `{"type": ..., "module": ..., "message": ..., "stack": ...}` |
|
|
1288
1317
|
| `truncated` | when a ceiling fired | `true`; absent otherwise, never `false` |
|
|
1289
1318
|
|
|
1290
1319
|
IDs are [W3C Trace Context](https://www.w3.org/TR/trace-context/)-compatible by design, so the
|
|
@@ -1329,18 +1358,30 @@ poetry run pytest # test (runs in parallel by default; see addopts)
|
|
|
1329
1358
|
poetry run pytest -n 0 # ...serially, when debugging a failure
|
|
1330
1359
|
poetry run ruff check . # lint (line-length 100)
|
|
1331
1360
|
poetry run mypy # typecheck (strict, over src/)
|
|
1361
|
+
sh scripts/spec-lint.sh # lint the design specs
|
|
1362
|
+
sh scripts/docs-lint.sh # the always-loaded docs tier — NOTHING IN CI RUNS THIS
|
|
1363
|
+
poetry run python scripts/docstring-lint.py # the docstring rule over src/ — ALSO NOT IN CI
|
|
1332
1364
|
```
|
|
1333
1365
|
|
|
1366
|
+
All six run before a push. The last two are the ones to remember, because they are deliberately
|
|
1367
|
+
not CI jobs: they hold the documentation and the docstrings to rules a reviewer would otherwise
|
|
1368
|
+
have to carry, and keeping them local means the failure lands on whoever caused it rather than on
|
|
1369
|
+
a shared branch. Each has a `-test.sh` beside it that proves its own checks still fire; run that
|
|
1370
|
+
one too if you change the linter.
|
|
1371
|
+
|
|
1334
1372
|
The library uses a src layout (`src/log_foundry/`) with a single concept per module: `config`,
|
|
1335
|
-
`ids`, `model`, `context`, `decorator`, `api`, `console`, `worker`, and
|
|
1336
|
-
`
|
|
1373
|
+
`ids`, `model`, `context`, `decorator`, `api`, `console`, `worker`, `sanitize` and `results`,
|
|
1374
|
+
the internal `_lifecycle`, `_fork` and `_diag`, and the `sinks/` package (the `base` protocol,
|
|
1375
|
+
`stdout`, and one module per sink family — see [Sinks](#sinks)). Anything underscore-prefixed
|
|
1376
|
+
is internal and moves without notice.
|
|
1337
1377
|
Deeper design docs live in [`docs/`](docs/) — start with [`docs/architecture.md`](docs/architecture.md).
|
|
1338
1378
|
|
|
1339
1379
|
### Continuous integration
|
|
1340
1380
|
|
|
1341
|
-
|
|
1342
|
-
front of it is path-filtered rather than run to the same answer
|
|
1343
|
-
part of the contract
|
|
1381
|
+
The checks below guard this repository. Most run on every pull request; a check whose verdict
|
|
1382
|
+
cannot change on the tree in front of it is path-filtered rather than run to the same answer
|
|
1383
|
+
twice, and two run only after the merge — so the *When* column is part of the contract and is
|
|
1384
|
+
worth reading before assuming a PR was audited:
|
|
1344
1385
|
|
|
1345
1386
|
| Check | Does | When | Fails the build |
|
|
1346
1387
|
|---|---|---|---|
|
|
@@ -1349,6 +1390,14 @@ part of the contract:
|
|
|
1349
1390
|
| [`dependency-review.yml`](.github/workflows/dependency-review.yml) | fails a PR that *introduces* a dependency with a known advisory (`moderate`+) | every PR | yes |
|
|
1350
1391
|
| [`zizmor.yml`](.github/workflows/zizmor.yml) | static analysis of the workflow files themselves | workflow, action, dependabot or zizmor config touched; also weekly | no — reports to code scanning |
|
|
1351
1392
|
| CodeQL | `python` + `actions`, `extended` query suite; also weekly | every PR | no — reports to code scanning |
|
|
1393
|
+
| [`integration.yml`](.github/workflows/integration.yml) | the extras-backed sinks against nine real services in containers | sinks, `tests/integration/`, `pyproject.toml`, `poetry.lock` or that workflow touched; also weekly | it goes red, and like every check here it is advisory — `main` requires no status check at all, and this one is furthest from earning that: nine service containers are a flakiness budget |
|
|
1394
|
+
| [`pip-audit.yml`](.github/workflows/pip-audit.yml) | advisories across every extra, `--strict` | **not on PRs at all** — on the merge to `main` when the lockfile, extras, the ignore list or that workflow moved, and weekly | yes, on `main` |
|
|
1395
|
+
| [`scorecard.yml`](.github/workflows/scorecard.yml) | OpenSSF Scorecard over the repository's own supply chain | **not on PRs** — on the merge to `main` when `.github/`, `scripts/`, `SECURITY.md`, `LICENSE` or the lockfile moved, and weekly | no — reports to code scanning |
|
|
1396
|
+
|
|
1397
|
+
**`scripts/docs-lint.sh` is deliberately not in that table**, because nothing in CI runs it. It
|
|
1398
|
+
holds the always-loaded documentation tier to its budgets and is a local pre-push gate, so its
|
|
1399
|
+
failure lands on whoever caused it rather than on a shared branch. Contributors run it, along
|
|
1400
|
+
with `scripts/spec-lint.sh`, before pushing.
|
|
1352
1401
|
|
|
1353
1402
|
On a push to `main` the full `ci.yml` matrix still runs, but as the first job of
|
|
1354
1403
|
[`release.yml`](.github/workflows/release.yml), which `uses:` this same workflow before it is
|
|
@@ -1391,9 +1440,12 @@ everything it instruments:
|
|
|
1391
1440
|
[`SECURITY.md`](SECURITY.md#software-bill-of-materials) has the detail.
|
|
1392
1441
|
|
|
1393
1442
|
Scanning runs continuously rather than at release time: CodeQL over the source and the workflows,
|
|
1394
|
-
zizmor over the workflows, `dependency-review` on every pull request,
|
|
1395
|
-
|
|
1396
|
-
`pip-audit` are the two that fail a build
|
|
1443
|
+
zizmor over the workflows, `dependency-review` on every pull request, `pip-audit` across all
|
|
1444
|
+
eleven extras, and OpenSSF Scorecard. Findings go to code scanning; `dependency-review` and
|
|
1445
|
+
`pip-audit` are the two that fail a build — but note **where**: only `dependency-review` runs
|
|
1446
|
+
on a pull request. `pip-audit` runs weekly and on the merge to `main`, so a floor change is
|
|
1447
|
+
audited after it lands, not before. The weekly re-examination is the point of it: an advisory
|
|
1448
|
+
is published on the advisory database's clock, not on this repository's.
|
|
1397
1449
|
|
|
1398
1450
|
## Releasing
|
|
1399
1451
|
|
|
@@ -1416,10 +1468,20 @@ pip ignores pre-releases unless you pass `--pre`.
|
|
|
1416
1468
|
Cutting a release is one tag:
|
|
1417
1469
|
|
|
1418
1470
|
```bash
|
|
1419
|
-
git tag -a
|
|
1420
|
-
git push origin
|
|
1471
|
+
git tag -a v1.2.3 -m "log-foundry 1.2.3"
|
|
1472
|
+
git push origin v1.2.3
|
|
1421
1473
|
```
|
|
1422
1474
|
|
|
1475
|
+
**One precondition, because it fails quietly.** If `docs/release-notes/<tag>.md` exists in the
|
|
1476
|
+
**tagged commit's tree**, the release job uses it as the release body; otherwise it generates
|
|
1477
|
+
one from the commit range. Both are valid, but the lookup is by exact tag name and the checkout
|
|
1478
|
+
takes the tag, so notes written for one version do nothing for another and notes added after
|
|
1479
|
+
the tagged commit are not seen. A release body cannot be amended once published — this
|
|
1480
|
+
repository has immutable releases — so check the file is there and named for the tag before
|
|
1481
|
+
pushing it. [`docs/release-notes/`](docs/release-notes/) holds the ones written so far, and
|
|
1482
|
+
[`docs/spec-delivery/RELEASES.md`](docs/spec-delivery/RELEASES.md) records which specs each
|
|
1483
|
+
released version carried.
|
|
1484
|
+
|
|
1423
1485
|
Uploads authenticate with PyPI [Trusted Publishing](https://docs.pypi.org/trusted-publishers/)
|
|
1424
1486
|
(OIDC) through the `pypi` GitHub Environment — there is no API token stored in the repository.
|
|
1425
1487
|
A tagged build refuses to publish if the derived version doesn't match the tag, and the tagged
|
|
@@ -14,8 +14,11 @@ calls form a tree you can query later.
|
|
|
14
14
|
- **Structured, never free-form** — every event is the same named-field JSON shape.
|
|
15
15
|
- **Safe by default** — never captures your arguments or return values (no accidental PII/secret leakage), and the decorator **never swallows exceptions**.
|
|
16
16
|
- **Correct under threads and asyncio** — context propagates via `contextvars`.
|
|
17
|
-
- **Non-blocking delivery** — finished
|
|
18
|
-
blocks on sink I/O, and a graceful drain at
|
|
17
|
+
- **Non-blocking delivery from a traced call** — a finished span's events are handed to a
|
|
18
|
+
background worker, so code inside `@trace` never blocks on sink I/O, and a graceful drain at
|
|
19
|
+
exit means buffered events aren't lost. Two things you can do *deliberately* are synchronous:
|
|
20
|
+
a level call with **no open span**, which emits on your own thread, and `flush()`, which by
|
|
21
|
+
definition waits for the drain it asked for.
|
|
19
22
|
|
|
20
23
|
---
|
|
21
24
|
|
|
@@ -37,13 +40,14 @@ pip install 'log-foundry[aws]' # + boto3 for the SQS/SNS/Kinesis/Firehose sink
|
|
|
37
40
|
>
|
|
38
41
|
> - **`health()` and `sink.losses()` return frozen dataclasses**, not `NamedTuple`s. Attribute
|
|
39
42
|
> access (`h.dropped`, `losses.failed`) is unchanged and is the whole contract; `len(h)`,
|
|
40
|
-
> `h[0]` and `queued, dropped,
|
|
41
|
-
>
|
|
43
|
+
> `h[0]` and the four-way `queued, dropped, failed_batches, stopped_reason = health()` now
|
|
44
|
+
> raise `TypeError`. `Health` gains fields, and will gain more — with no positions to
|
|
45
|
+
> preserve, that stops being a breaking change.
|
|
42
46
|
> - **`flush()` returns a `FlushResult` and `continue_trace()` a `ContinueResult`**, each truthy
|
|
43
47
|
> or falsy with a `reason` naming *why*. `if lf.flush():` is unchanged; **`lf.flush() is True`
|
|
44
48
|
> is not** — the result is an object. A one-bit return could not grow a reason later without
|
|
45
49
|
> silently changing what `if flush():` means, which is why it moved now.
|
|
46
|
-
> - **`SQSSink`'s injected client is keyword-only** (`SQSSink(
|
|
50
|
+
> - **`SQSSink`'s injected client is keyword-only** (`SQSSink(queue_url, client=…)`), **`SentrySink`
|
|
47
51
|
> injects through `client=`** rather than the old `sdk` keyword, with no alias, and the sink attribute the
|
|
48
52
|
> library assigns for interruptible backoff is **`log_foundry_stop_signal`**, not
|
|
49
53
|
> `stop_signal` — a prefixed name cannot silently overwrite one your own sink already uses.
|
|
@@ -126,9 +130,9 @@ with the child span pointing at its parent via `parent_span_id`:
|
|
|
126
130
|
|
|
127
131
|
```json
|
|
128
132
|
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.start", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "754adb40e10c445f9ec9e23a2f3dcbf2", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}}
|
|
129
|
-
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "3af73c51540848afbeaba9fdf7a9dce8", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}, "duration_ms": 0.
|
|
133
|
+
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "6aeb63c0eba85bf4", "parent_span_id": "b02197e75f40eb81", "log_id": "3af73c51540848afbeaba9fdf7a9dce8", "function": "tax.compute", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {"component": "tax"}, "duration_ms": 0.0233328901231289, "status": "ok"}
|
|
130
134
|
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.start", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "e789f7e5268b46d8b779c9cbcdde8656", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}}
|
|
131
|
-
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "8f3dbfcfcf4a45f688c738eefef882b0", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}, "duration_ms": 0.
|
|
135
|
+
{"timestamp": "2026-07-10T00:57:10.411Z", "level": "INFO", "message": "span.end", "trace_id": "8ab2add1480f8f6a52fe97cd23ae6f36", "span_id": "b02197e75f40eb81", "parent_span_id": null, "log_id": "8f3dbfcfcf4a45f688c738eefef882b0", "function": "charge", "service": "billing-api", "version": "1.4.2", "env": "prod", "fields": {}, "duration_ms": 0.342666869983077, "status": "ok"}
|
|
132
136
|
```
|
|
133
137
|
|
|
134
138
|
> **Note on ordering:** the child span (`tax.compute`) finishes first, so its events flush
|
|
@@ -140,7 +144,11 @@ with the child span pointing at its parent via `parent_span_id`:
|
|
|
140
144
|
|
|
141
145
|
A traced call travels through a small pipeline. The first four steps run on your own thread
|
|
142
146
|
and are deliberately fast; the last two run on a background thread so your code never waits on
|
|
143
|
-
the destination.
|
|
147
|
+
the destination. That is the traced path. A level call with no **open** span takes the
|
|
148
|
+
synchronous route instead, described under
|
|
149
|
+
[Flushing and shutdown](#flushing-and-shutdown) — and "open" is the operative word: a task
|
|
150
|
+
that outlives its span still sees that span, but it has closed, so the call takes the
|
|
151
|
+
synchronous route too.
|
|
144
152
|
|
|
145
153
|
1. **You call the code** — a `@trace` function, or one of the `debug`/`info`/… emitters.
|
|
146
154
|
2. **A span opens** — a record of this one call. It inherits the current trace and parent (see
|
|
@@ -302,9 +310,9 @@ def process(): ...
|
|
|
302
310
|
- `defaults` — per-decorator fields merged into every event this span emits.
|
|
303
311
|
|
|
304
312
|
The **outermost** decorated call starts a new trace; every nested decorated call becomes a
|
|
305
|
-
child span within it. On an exception, the decorator records `status="error"` plus
|
|
306
|
-
exception type and formatted stack, then
|
|
307
|
-
never swallows errors.
|
|
313
|
+
child span within it. On an exception, the decorator records `status="error"` plus an `error`
|
|
314
|
+
sub-document carrying the exception's type, module, message and formatted stack, then
|
|
315
|
+
**re-raises the original exception unchanged** — it never swallows errors.
|
|
308
316
|
|
|
309
317
|
**Async is supported.** Apply `@trace` to an `async def` and it traces the coroutine's actual
|
|
310
318
|
run — the span opens when the coroutine starts and closes when the awaits complete, not when the
|
|
@@ -369,7 +377,7 @@ def process_payment(user_id: int) -> str:
|
|
|
369
377
|
It reaches its own name too (`fields={"fields": ...}`), so every reserved word has exactly one
|
|
370
378
|
route through. A key given both ways takes the keyword's value, since `**kwargs` is what you
|
|
371
379
|
wrote at the call site and `fields=` is usually a mapping built elsewhere.
|
|
372
|
-
- **Orphan logs** — a level call made with no
|
|
380
|
+
- **Orphan logs** — a level call made with no **open** span is not dropped: it emits a standalone
|
|
373
381
|
one-event span with a fresh `trace_id`, flushed straight to the sink.
|
|
374
382
|
|
|
375
383
|
### Continuing a trace across processes
|
|
@@ -411,9 +419,13 @@ def handler(event, context): # the entry point, deliberately not decorated
|
|
|
411
419
|
lf.flush() # the span has closed, so its events are drained
|
|
412
420
|
```
|
|
413
421
|
|
|
414
|
-
`flush()`
|
|
415
|
-
|
|
416
|
-
|
|
422
|
+
`flush()` works either side of the traced function. Putting it outside, as above, is still the
|
|
423
|
+
clearer shape — the span has closed, so there is nothing to reason about. But a `flush()`
|
|
424
|
+
*inside* an open span is no longer a mistake: it sweeps every open span in the calling context,
|
|
425
|
+
hands their buffered events to the worker and drains them, leaving the spans open. It does not
|
|
426
|
+
deliver a `span.end` that has not happened yet. Before `1.0.0` a flush inside a span swept
|
|
427
|
+
nothing, and this recipe with the `flush()` moved into the traced function delivered **nothing**
|
|
428
|
+
with every counter clean.
|
|
417
429
|
|
|
418
430
|
| Call | Does |
|
|
419
431
|
|---|---|
|
|
@@ -504,8 +516,16 @@ nothing.
|
|
|
504
516
|
|
|
505
517
|
Wire one up by passing an instance to `configure(sink=...)`; if you never do, the first decorated
|
|
506
518
|
call falls back to `StdoutSink()`. The **protocol** is a top-level export, alongside
|
|
507
|
-
`SinkDeliveryError`, `SinkLosses` and `
|
|
508
|
-
|
|
519
|
+
`SinkDeliveryError`, `SinkLosses`, `read_losses` and `flush_sink` — the last two being the probes
|
|
520
|
+
a wrapper sink needs to ask a child for its losses and to push its client-side buffer. The
|
|
521
|
+
**concrete sinks** are not exported, so import each from its own module, e.g.
|
|
522
|
+
`from log_foundry.sinks.sqs import SQSSink`.
|
|
523
|
+
|
|
524
|
+
**Construct `SinkLosses` with keywords**, as the example below does. The public dataclasses are
|
|
525
|
+
keyword-only from `1.0.0`, so a positional `SinkLosses(0, 3)` raises `TypeError` — and it raises
|
|
526
|
+
*inside* your `losses()`, where `read_losses` deliberately swallows it so a broken reporter cannot
|
|
527
|
+
take `health()` down. The symptom is not an error: your sink's loss reporting silently becomes
|
|
528
|
+
`None`.
|
|
509
529
|
|
|
510
530
|
A few conventions hold across every sink below:
|
|
511
531
|
|
|
@@ -555,8 +575,8 @@ A few conventions hold across every sink below:
|
|
|
555
575
|
|
|
556
576
|
| Sink | Import from | Configure |
|
|
557
577
|
|---|---|---|
|
|
558
|
-
| `StdoutSink` | `log_foundry.sinks.stdout` | `StdoutSink(stream=
|
|
559
|
-
| `StderrSink` | `log_foundry.sinks.stdout` | `StderrSink(stream=
|
|
578
|
+
| `StdoutSink` | `log_foundry.sinks.stdout` | `StdoutSink(stream=None)` — one JSON line per event; `None` resolves to `sys.stdout` **at construction** and is not re-read per write. The zero-config default |
|
|
579
|
+
| `StderrSink` | `log_foundry.sinks.stdout` | `StderrSink(stream=None)` — same, resolving to `sys.stderr` (twelve-factor) |
|
|
560
580
|
| `NullSink` | `log_foundry.sinks.null` | `NullSink()` — discard everything; `.dropped` counts events |
|
|
561
581
|
| `MemorySink` | `log_foundry.sinks.memory` | `MemorySink(maxlen=None)` — collect into `.events` (a bounded ring when `maxlen` is set) |
|
|
562
582
|
|
|
@@ -912,7 +932,7 @@ class MySink:
|
|
|
912
932
|
|
|
913
933
|
def losses(self) -> SinkLosses:
|
|
914
934
|
with self._counter_lock: # both fields from one instant
|
|
915
|
-
return SinkLosses(dropped=self._dropped, failed=self._failed)
|
|
935
|
+
return SinkLosses(dropped=self._dropped, failed=self._failed) # keywords required
|
|
916
936
|
|
|
917
937
|
def close(self) -> None:
|
|
918
938
|
with self._lock: # never release under an active writer
|
|
@@ -972,7 +992,12 @@ Every wrapper shipped here — `MultiSink`, `FilteringSink`, `TransformSink`, `S
|
|
|
972
992
|
|
|
973
993
|
Delivery is off the hot path. When a span ends, its events are handed to a per-process
|
|
974
994
|
background worker via a fast, non-blocking submit — your function returns without waiting on
|
|
975
|
-
the sink.
|
|
995
|
+
the sink. A level call with no **open** span has no span to end, so it emits synchronously on
|
|
996
|
+
the calling thread — including a call from a task that outlived the span it inherited, whose
|
|
997
|
+
span is present but closed. That path needs no worker, and in a process that only ever logs
|
|
998
|
+
that way none is ever built; it is a span closing, or a `flush()` sweeping one, that builds it.
|
|
999
|
+
The rest of this paragraph is about the buffered path.
|
|
1000
|
+
The worker batches events (by count and time), emits them on its own thread, retries
|
|
976
1001
|
a failing sink with backoff, and applies backpressure so a slow or down sink can never block or
|
|
977
1002
|
back-pressure the app: when its bounded queue is full it drops the newest submissions and counts
|
|
978
1003
|
them rather than stalling.
|
|
@@ -1000,7 +1025,7 @@ They tell you different things, and they want different responses:
|
|
|
1000
1025
|
|
|
1001
1026
|
| Field | Means | What to do |
|
|
1002
1027
|
|---|---|---|
|
|
1003
|
-
| `dropped` | The queue filled — the destination is not keeping up. Delivery continues. |
|
|
1028
|
+
| `dropped` | The queue filled — the destination is not keeping up. Delivery continues. | Make the destination keep up: scale the sink, or reduce what you log. The worker's batch size, flush interval and queue depth are **not** reachable from the public API, so tuning them is not an option this version offers you. |
|
|
1004
1029
|
| `failed_batches` | A sink stayed broken through the whole retry budget. Delivery continues. | Fix the destination. |
|
|
1005
1030
|
| `stopped_reason` | The background thread **died** on that exception type. Nothing further will be delivered, ever. | Restart the process; investigate the named exception. |
|
|
1006
1031
|
| `sink.dropped` | The sink discarded events **before** attempting delivery — an oversized record, or one the client refused outright. | Read the stderr line: it names the cause. An oversized record means shrink what you log; a refused local produce/publish (Kafka, Pub/Sub) points at the client — a saturated buffer, a bad topic, a credential. |
|
|
@@ -1008,7 +1033,7 @@ They tell you different things, and they want different responses:
|
|
|
1008
1033
|
| `retired` + `submitted_after_shutdown` | `shutdown()` was called and the process **kept logging**. Those events are queued where nothing will drain them — total loss, for as long as the process runs. | Use `flush()`, not `shutdown()`, in a process that logs again. This is the serverless mistake below. |
|
|
1009
1034
|
| `incomplete_swaps` | A late `configure(sink=...)` could not confirm the previous sink was drained. The swap took effect; that sink was left open and some queued events may have gone to the new one. | Investigate the previous sink — it was hung or failing. Configure the sink before the first log where you can. |
|
|
1010
1035
|
| `inherited_sink` | This process is delivering to a sink it **inherited across a `fork`** and may not release, so it will not be closed here. Not a loss and not an alert term. | Nothing, usually. It explains a handle still open after `shutdown()`, and tells you a deployment shares one sink across a fork at all. `True` for a shared `StdoutSink` too, whose `close()` only flushes — so a `True` is not by itself evidence that anything is held. If you want the child to own its transport, build the sink in the worker process (see Forking). |
|
|
1011
|
-
| `orphan_lost` | An event logged **with no
|
|
1036
|
+
| `orphan_lost` | An event logged **with no open span** never reached the sink. That call emits on your own thread with no worker behind it, so no other field here can carry it — it is not a batch, there was no retry, and there may be no worker at all. Covers a sink that failed to *construct* as well as one that raised. | Fix the destination, or the data. The stderr line names the exception type. If a process logs this way at all, this is the field to alert on: nothing else describes that path. |
|
|
1012
1037
|
| `in_span_lost` | An event logged **inside a span** could not be built — a value that could not be turned into an event. Always the data, never the destination: the in-span path cannot fail at delivery, which is `failed_batches`. | Fix the call site. Passing a non-string message (an exception object, say) is the common cause. |
|
|
1013
1038
|
| `closing_sinks` | Swapped-out sinks inside `close()` **right now** — a live gauge, not a counter. It and `queued` are the two fields here that fall as well as rise; the other integer counters only climb. Non-zero on a single read is normal during a swap. | Nothing, unless it stays non-zero. That means a destination is stuck in `close()` and will not release its resources. |
|
|
1014
1039
|
|
|
@@ -1016,8 +1041,9 @@ They tell you different things, and they want different responses:
|
|
|
1016
1041
|
reported. They aggregate different failure populations — one can mean the destination *or* the
|
|
1017
1042
|
data, the other can only mean the data — so a single number would hide which fix applies.
|
|
1018
1043
|
|
|
1019
|
-
`h.sink` is a `SinkLosses
|
|
1020
|
-
the configured sink reports nothing (`losses()` is optional
|
|
1044
|
+
`h.sink` is a `SinkLosses`, carrying `dropped` and `failed`, or `None` — `None` when no worker
|
|
1045
|
+
exists yet, or when the configured sink reports nothing (`losses()` is optional, and a sink whose
|
|
1046
|
+
`losses()` raises reports `None` too). Note the two `dropped` fields count
|
|
1021
1047
|
different things: the worker's is backpressure at *its* queue, the sink's is an event that never
|
|
1022
1048
|
reached the wire. They are separate because the remedies do not overlap — and `sink.dropped` is
|
|
1023
1049
|
itself two causes, which is why the diagnostic line matters. Most sinks drop only what can never
|
|
@@ -1058,11 +1084,12 @@ guards its own post-close state — rather than queued where nothing will drain
|
|
|
1058
1084
|
the same claim. A stateless sink such as the default `StdoutSink` still accepts it.
|
|
1059
1085
|
|
|
1060
1086
|
Read a snapshot **by attribute** (`h.dropped`), as above — that is the whole contract. `Health`
|
|
1061
|
-
and `SinkLosses` are frozen dataclasses, so `len(h)`, `h[0]` and
|
|
1062
|
-
health()` all raise `TypeError`. They were
|
|
1063
|
-
|
|
1064
|
-
|
|
1065
|
-
|
|
1087
|
+
and `SinkLosses` are frozen dataclasses, so `len(h)`, `h[0]` and the four-way
|
|
1088
|
+
`queued, dropped, failed_batches, stopped_reason = health()` all raise `TypeError`. They were
|
|
1089
|
+
`NamedTuple`s before `1.0.0`, and the tuple shape is deliberately gone: `Health` gains fields
|
|
1090
|
+
as the library learns to report more, and while it was a tuple every one of those additions
|
|
1091
|
+
had to argue that the positions before it were undisturbed. There are no positions to disturb
|
|
1092
|
+
now, and adding a field is not a breaking change.
|
|
1066
1093
|
|
|
1067
1094
|
`dropped` counts submissions discarded because the queue filled; `failed_batches` counts batches
|
|
1068
1095
|
abandoned after the retry budget was spent. Overflow also warns on stderr — on the first drop and
|
|
@@ -1122,7 +1149,7 @@ was outstanding** — the drain it forces reached the sink, and so did anything
|
|
|
1122
1149
|
emitted while it waited its turn. A truthy result is evidence of delivery, not merely that a drain
|
|
1123
1150
|
took place.
|
|
1124
1151
|
|
|
1125
|
-
Falsy carries a `reason` saying which of
|
|
1152
|
+
Falsy carries a `reason` saying which of these happened, because they need different fixes:
|
|
1126
1153
|
|
|
1127
1154
|
| `reason` | Means |
|
|
1128
1155
|
|---|---|
|
|
@@ -1131,6 +1158,7 @@ Falsy carries a `reason` saying which of five things happened, because they need
|
|
|
1131
1158
|
| `"thread-died"` | The drain thread is gone; see `health().stopped_reason`. |
|
|
1132
1159
|
| `"queue-full"` | Backpressure — the queue could not even accept the marker. |
|
|
1133
1160
|
| `"abandoned"` | A batch was given up on after its retry budget. The destination is broken. |
|
|
1161
|
+
| `"sink-flush"` | Everything queued reached the sink, but the sink's own `flush()` raised — a client-side buffer that did not go out. Distinct from `"abandoned"`: the library delivered, the sink did not. |
|
|
1134
1162
|
|
|
1135
1163
|
```python
|
|
1136
1164
|
result = lf.flush(5.0)
|
|
@@ -1189,7 +1217,8 @@ def _handler(event, context):
|
|
|
1189
1217
|
|
|
1190
1218
|
def handler(event, context):
|
|
1191
1219
|
# NOT decorated, so the span closes when `_handler` returns and its events reach the queue
|
|
1192
|
-
# before `flush()` runs.
|
|
1220
|
+
# before `flush()` runs. Since 1.0.0 a `flush()` *inside* the traced function also works —
|
|
1221
|
+
# it sweeps the open span — but keeping it out here means there is nothing to reason about.
|
|
1193
1222
|
try:
|
|
1194
1223
|
return _handler(event, context)
|
|
1195
1224
|
finally:
|
|
@@ -1231,9 +1260,9 @@ Every event is the same shape (arch §6). Boundary events add a few fields:
|
|
|
1231
1260
|
| `function` | ✓ | span name |
|
|
1232
1261
|
| `service` / `version` / `env` | ✓ | from `configure(...)` |
|
|
1233
1262
|
| `fields` | ✓ | merged user fields (config `defaults` → span `defaults` → …) |
|
|
1234
|
-
| `duration_ms` | span.end | wall time from a monotonic delta |
|
|
1263
|
+
| `duration_ms` | span.end | wall time from a monotonic delta, unrounded — full float precision |
|
|
1235
1264
|
| `status` | span.end | `"ok"` or `"error"` |
|
|
1236
|
-
| `error` | on failure | `{"type": ..., "stack": ...}` |
|
|
1265
|
+
| `error` | on failure | `{"type": ..., "module": ..., "message": ..., "stack": ...}` |
|
|
1237
1266
|
| `truncated` | when a ceiling fired | `true`; absent otherwise, never `false` |
|
|
1238
1267
|
|
|
1239
1268
|
IDs are [W3C Trace Context](https://www.w3.org/TR/trace-context/)-compatible by design, so the
|
|
@@ -1278,18 +1307,30 @@ poetry run pytest # test (runs in parallel by default; see addopts)
|
|
|
1278
1307
|
poetry run pytest -n 0 # ...serially, when debugging a failure
|
|
1279
1308
|
poetry run ruff check . # lint (line-length 100)
|
|
1280
1309
|
poetry run mypy # typecheck (strict, over src/)
|
|
1310
|
+
sh scripts/spec-lint.sh # lint the design specs
|
|
1311
|
+
sh scripts/docs-lint.sh # the always-loaded docs tier — NOTHING IN CI RUNS THIS
|
|
1312
|
+
poetry run python scripts/docstring-lint.py # the docstring rule over src/ — ALSO NOT IN CI
|
|
1281
1313
|
```
|
|
1282
1314
|
|
|
1315
|
+
All six run before a push. The last two are the ones to remember, because they are deliberately
|
|
1316
|
+
not CI jobs: they hold the documentation and the docstrings to rules a reviewer would otherwise
|
|
1317
|
+
have to carry, and keeping them local means the failure lands on whoever caused it rather than on
|
|
1318
|
+
a shared branch. Each has a `-test.sh` beside it that proves its own checks still fire; run that
|
|
1319
|
+
one too if you change the linter.
|
|
1320
|
+
|
|
1283
1321
|
The library uses a src layout (`src/log_foundry/`) with a single concept per module: `config`,
|
|
1284
|
-
`ids`, `model`, `context`, `decorator`, `api`, `console`, `worker`, and
|
|
1285
|
-
`
|
|
1322
|
+
`ids`, `model`, `context`, `decorator`, `api`, `console`, `worker`, `sanitize` and `results`,
|
|
1323
|
+
the internal `_lifecycle`, `_fork` and `_diag`, and the `sinks/` package (the `base` protocol,
|
|
1324
|
+
`stdout`, and one module per sink family — see [Sinks](#sinks)). Anything underscore-prefixed
|
|
1325
|
+
is internal and moves without notice.
|
|
1286
1326
|
Deeper design docs live in [`docs/`](docs/) — start with [`docs/architecture.md`](docs/architecture.md).
|
|
1287
1327
|
|
|
1288
1328
|
### Continuous integration
|
|
1289
1329
|
|
|
1290
|
-
|
|
1291
|
-
front of it is path-filtered rather than run to the same answer
|
|
1292
|
-
part of the contract
|
|
1330
|
+
The checks below guard this repository. Most run on every pull request; a check whose verdict
|
|
1331
|
+
cannot change on the tree in front of it is path-filtered rather than run to the same answer
|
|
1332
|
+
twice, and two run only after the merge — so the *When* column is part of the contract and is
|
|
1333
|
+
worth reading before assuming a PR was audited:
|
|
1293
1334
|
|
|
1294
1335
|
| Check | Does | When | Fails the build |
|
|
1295
1336
|
|---|---|---|---|
|
|
@@ -1298,6 +1339,14 @@ part of the contract:
|
|
|
1298
1339
|
| [`dependency-review.yml`](.github/workflows/dependency-review.yml) | fails a PR that *introduces* a dependency with a known advisory (`moderate`+) | every PR | yes |
|
|
1299
1340
|
| [`zizmor.yml`](.github/workflows/zizmor.yml) | static analysis of the workflow files themselves | workflow, action, dependabot or zizmor config touched; also weekly | no — reports to code scanning |
|
|
1300
1341
|
| CodeQL | `python` + `actions`, `extended` query suite; also weekly | every PR | no — reports to code scanning |
|
|
1342
|
+
| [`integration.yml`](.github/workflows/integration.yml) | the extras-backed sinks against nine real services in containers | sinks, `tests/integration/`, `pyproject.toml`, `poetry.lock` or that workflow touched; also weekly | it goes red, and like every check here it is advisory — `main` requires no status check at all, and this one is furthest from earning that: nine service containers are a flakiness budget |
|
|
1343
|
+
| [`pip-audit.yml`](.github/workflows/pip-audit.yml) | advisories across every extra, `--strict` | **not on PRs at all** — on the merge to `main` when the lockfile, extras, the ignore list or that workflow moved, and weekly | yes, on `main` |
|
|
1344
|
+
| [`scorecard.yml`](.github/workflows/scorecard.yml) | OpenSSF Scorecard over the repository's own supply chain | **not on PRs** — on the merge to `main` when `.github/`, `scripts/`, `SECURITY.md`, `LICENSE` or the lockfile moved, and weekly | no — reports to code scanning |
|
|
1345
|
+
|
|
1346
|
+
**`scripts/docs-lint.sh` is deliberately not in that table**, because nothing in CI runs it. It
|
|
1347
|
+
holds the always-loaded documentation tier to its budgets and is a local pre-push gate, so its
|
|
1348
|
+
failure lands on whoever caused it rather than on a shared branch. Contributors run it, along
|
|
1349
|
+
with `scripts/spec-lint.sh`, before pushing.
|
|
1301
1350
|
|
|
1302
1351
|
On a push to `main` the full `ci.yml` matrix still runs, but as the first job of
|
|
1303
1352
|
[`release.yml`](.github/workflows/release.yml), which `uses:` this same workflow before it is
|
|
@@ -1340,9 +1389,12 @@ everything it instruments:
|
|
|
1340
1389
|
[`SECURITY.md`](SECURITY.md#software-bill-of-materials) has the detail.
|
|
1341
1390
|
|
|
1342
1391
|
Scanning runs continuously rather than at release time: CodeQL over the source and the workflows,
|
|
1343
|
-
zizmor over the workflows, `dependency-review` on every pull request,
|
|
1344
|
-
|
|
1345
|
-
`pip-audit` are the two that fail a build
|
|
1392
|
+
zizmor over the workflows, `dependency-review` on every pull request, `pip-audit` across all
|
|
1393
|
+
eleven extras, and OpenSSF Scorecard. Findings go to code scanning; `dependency-review` and
|
|
1394
|
+
`pip-audit` are the two that fail a build — but note **where**: only `dependency-review` runs
|
|
1395
|
+
on a pull request. `pip-audit` runs weekly and on the merge to `main`, so a floor change is
|
|
1396
|
+
audited after it lands, not before. The weekly re-examination is the point of it: an advisory
|
|
1397
|
+
is published on the advisory database's clock, not on this repository's.
|
|
1346
1398
|
|
|
1347
1399
|
## Releasing
|
|
1348
1400
|
|
|
@@ -1365,10 +1417,20 @@ pip ignores pre-releases unless you pass `--pre`.
|
|
|
1365
1417
|
Cutting a release is one tag:
|
|
1366
1418
|
|
|
1367
1419
|
```bash
|
|
1368
|
-
git tag -a
|
|
1369
|
-
git push origin
|
|
1420
|
+
git tag -a v1.2.3 -m "log-foundry 1.2.3"
|
|
1421
|
+
git push origin v1.2.3
|
|
1370
1422
|
```
|
|
1371
1423
|
|
|
1424
|
+
**One precondition, because it fails quietly.** If `docs/release-notes/<tag>.md` exists in the
|
|
1425
|
+
**tagged commit's tree**, the release job uses it as the release body; otherwise it generates
|
|
1426
|
+
one from the commit range. Both are valid, but the lookup is by exact tag name and the checkout
|
|
1427
|
+
takes the tag, so notes written for one version do nothing for another and notes added after
|
|
1428
|
+
the tagged commit are not seen. A release body cannot be amended once published — this
|
|
1429
|
+
repository has immutable releases — so check the file is there and named for the tag before
|
|
1430
|
+
pushing it. [`docs/release-notes/`](docs/release-notes/) holds the ones written so far, and
|
|
1431
|
+
[`docs/spec-delivery/RELEASES.md`](docs/spec-delivery/RELEASES.md) records which specs each
|
|
1432
|
+
released version carried.
|
|
1433
|
+
|
|
1372
1434
|
Uploads authenticate with PyPI [Trusted Publishing](https://docs.pypi.org/trusted-publishers/)
|
|
1373
1435
|
(OIDC) through the `pypi` GitHub Environment — there is no API token stored in the repository.
|
|
1374
1436
|
A tagged build refuses to publish if the derived version doesn't match the tag, and the tagged
|
|
@@ -74,7 +74,7 @@ keywords = [
|
|
|
74
74
|
# vulnerability-reporting channel. The repository is still named `log-forge` — the ORIGINAL name,
|
|
75
75
|
# which PyPI rejected for the distribution — so these URLs deliberately do not match the package
|
|
76
76
|
# name. See the note on `name` above before "correcting" them.
|
|
77
|
-
version = "0.10.2.
|
|
77
|
+
version = "0.10.2.dev111"
|
|
78
78
|
|
|
79
79
|
[project.urls]
|
|
80
80
|
Homepage = "https://github.com/agriffi10/log-forge"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/elasticsearch.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{log_foundry-0.10.2.dev109 → log_foundry-0.10.2.dev111}/src/log_foundry/sinks/logging_sink.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|